diff --git a/docs/devel_doc/openapi.json b/docs/devel_doc/openapi.json index 8737dbf3f..ed380a84b 100644 --- a/docs/devel_doc/openapi.json +++ b/docs/devel_doc/openapi.json @@ -21349,7 +21349,7 @@ } ], "title": "Provider", - "description": "Default provider config: {name, model, credentials_secret}." + "description": "Default provider config: {name, model}. The stack chooses the credential; credentials_secret is rejected." }, "sandbox_image": { "anyOf": [ diff --git a/examples/workflows/README.md b/examples/workflows/README.md new file mode 100644 index 000000000..e43ebdd67 --- /dev/null +++ b/examples/workflows/README.md @@ -0,0 +1,475 @@ +# Workflows on K8s: operator setup and caller guide (post-stack#51) + +This folder is a self-contained, end-to-end worked example of the design in +[stack#51, revision 3](https://github.com/jameswnl/lightspeed-stack/issues/51#issuecomment-5911977694): +lightspeed-stack as the sole policy/secret boundary in front of cloud-agents. + +> **Status: target state, being built (feature-level TDD).** This folder +> defines where issue #51 and cloud-agents#269 end up. Phase 0a is in +> review (PR #57): caller `credentials_secret` -> 400, custom images, +> `advisory` and `spawn: none/local` need admin -> 403, size and count +> caps. The `lightspeed-stack.yaml` blocks marked `PLAN(stack#51)` do not +> load yet, and the catalog/policy behavior in the workflows does not exist +> yet. `verify.py` pins the contract (see "Verification"); each phase should +> make more of it real until the whole folder runs against a deployment +> (Phases 0b–6). + +> **Auth prerequisite.** Role-based policy and the admin gates need an auth +> module that resolves roles: `jwk-token` with `role_rules` (used here) or +> `rh-identity` with `access_rules`. With `k8s`, `noop`, `noop-with-token` +> or `api-key` the resolvers are no-ops: callers only have the `*` role, +> `access_rules` are ignored and every caller counts as admin, so the +> admin-only denials below do not apply. + +## Contents + +| File | Author | Purpose | +|---|---|---| +| `README.md` (this file) | — | Detailed steps: operator setup + authoring + triggering | +| `lightspeed-stack.yaml` | Operator | Provider catalog, secret registry, MCP catalog, policy, spawner | +| `k8s-secrets.yaml` | Operator | Secret values — the only place they exist | +| `triage-github-issue.yaml` | Caller | 2-step triage flow (role `agent-user`) | +| `kb-answer-with-approval.yaml` | Caller | Research → human approval → answer (role `agent-support`) | +| `admin-inline-mcp.yaml` | Caller (admin) | Inline MCP server with a registry secret ref (role `agent-admin`) | +| `post-269-overrides.yaml` | Operator | What changes after cloud-agents#269 (K8s-bound inference credentials, leases) | +| `k8s-deployment.yaml` | Operator | Hardened Deployment, least-privilege RBAC, NetworkPolicy | +| `external-secrets.yaml` | Operator | Production source of the Secrets (External Secrets Operator + Vault) | +| `cases.yaml` | Contract | 63 request/config cases with the expected status, per stage | +| `reference_gate.py` | Contract | Reference model of the gate; replaced by the real endpoint as phases land | +| `verify.py` | Contract | Runs everything above plus deployment cross-checks | + +Assumed deployment: namespace `lightspeed`, stack Deployment + ConfigMap, +Postgres, OpenShell gateway for ephemeral sandboxes. + +## Goal coverage (issue #51 acceptance criteria) + +Each criterion maps to something in this folder that `verify.py` checks, so +"done" for a phase means more of these run against the real stack. + +| #51 criterion | Where it shows up | Checked by | +|---|---|---| +| `ProviderSelection` / `SecretRef` contract | `provider: {name, model, credential_ref?}` in the requests below | `cases.yaml` (credential_ref match / mismatch) | +| Provider-profile and secret registry | `workflow_engine.providers` / `secrets` / `default_*` | load checks, both stages | +| Reject caller-chosen env var / physical ids | `credentials_secret` on run provider and `definition.provider`, unknown provider keys | `cases.yaml` § credentials (400) | +| AuthZ: provider, model, credential, MCP, skills, tools, spawn, images, service accounts, namespaces, limits | `workflow_engine.policy.rules` | `cases.yaml` § policy / spawn / limits (403) | +| In-process handoff contract | Part 3, "After cloud-agents#269" | `post-269-overrides.yaml` run through the same cases | +| Process-boundary handoff contract | same section (lease redeemed over mTLS; Temporal workers get `MCP_ALLOWED_SECRETS` from config) | design only; Phase 6 | +| No secret values in payloads, state, logs | requests carry logical names only; `k8s-secrets.yaml` has placeholders only | `verify.py` secrets checks; Phase 4 canary suite later | +| Inline MCP URLs / literal headers | `admin-inline-mcp.yaml`, `inline_mcp_hosts` | `cases.yaml` § inline MCP | +| OpenShell provider injection for ephemeral | `spawner` block, README Step 6 | config schema check | +| `none` / `local` restrictions | admin-only on workflows | `cases.yaml` § spawn | +| Rotation and revocation | Step 8 (pre-#269 rules, Reloader) and Part 3 (post-#269) | documented; lease tests in Phase 6 | +| Cleanup, redaction, cross-principal tests | not examples; Phase 4 / 6 test suites | cross-principal: `agent-support` vs triage, `agent-user` vs KB | +| Deployment docs: K8s secrets, external managers, OpenShell | `k8s-secrets.yaml` (dev), `external-secrets.yaml`, `k8s-deployment.yaml` | `verify.py` deployment cross-checks | +| Limits (`MAX_*`) | byte, step, MCP and secret-header caps | `cases.yaml` § limits (413 / 422) | + +## The two documents (read this first) + +Everything in this folder is one of two documents. They are authored by +different people, live in different places, and are joined by the stack at +submit time: + +| | `lightspeed-stack.yaml` | Workflow definition YAML | +|---|---|---| +| Author | Operator (platform team) | Caller (API user / pipeline) | +| Lives in | ConfigMap mounted into stack pods | `POST /v1/workflows/run` request body | +| Contains | Catalogs, secret bindings, policy, spawner, MCP URLs | Steps, prompts, logical names, schemas | +| Secrets | Backend *bindings* only (env var / K8s Secret names), never values | Nothing secret-related — logical names only | + +Core principle: workflow payloads may contain logical references to approved +configuration, but never secret material. The stack resolves every logical +name against its catalogs, checks policy for the caller's roles, and only +then persists and starts the run — otherwise 400 (unknown to the deployment) +or 403 (exists, but not for you) before anything is saved. + +--- + +## Part 1 — Operator setup (once per deployment, K8s) + +### Step 1. Put API keys and secrets in K8s Secrets + +Keys live only in K8s Secrets. In production, do not apply values by hand: +`external-secrets.yaml` syncs every Secret below from Vault through External +Secrets Operator (`k8s-secrets.yaml` is the bootstrap/dev equivalent with +`CHANGEME` placeholders; `verify.py` checks both define the same Secrets and +keys). The Secrets: + +- `lightspeed-inference-creds`: `ANTHROPIC_API_KEY`, `OPENAI_TEAM_B_KEY` + (inference keys; pre-cloud-agents#269 bindings must be `backend: env` + because cloud-agents reads `os.environ` by key). +- `mcp-github-readonly-token`: `token` (the full `Authorization` header + value, including the `Bearer ` prefix; open question whether cloud-agents + adds it, see cloud-agents#269). +- `mcp-kb-search-token`: `api-key` (KB service key). +- `mcp-incidents-token`: `token` (only reachable through admin inline MCP). +- `openshell-gateway-tls`: mTLS client identity for stack → gateway. +- `lightspeed-postgres`: `password` for the workflow-state database. +- `lightspeed-postgres-tls`: CA for `ssl_mode: verify-full`. + +```bash +# Fill in every CHANGEME value first, then: +kubectl apply -n lightspeed -f k8s-secrets.yaml # dev / bootstrap +kubectl apply -n lightspeed -f external-secrets.yaml # production +``` + +`k8s-deployment.yaml` wires them into the stack Deployment (non-root, +read-only rootfs, no capabilities, digest-pinned image, default-deny egress +NetworkPolicy, RBAC limited to `get` on named Secrets). Excerpt — inference +keys and DB password as env, TLS as mounted files: + +```yaml +# lightspeed-stack Deployment (excerpt) +env: + - name: ANTHROPIC_API_KEY + valueFrom: {secretKeyRef: {name: lightspeed-inference-creds, key: ANTHROPIC_API_KEY}} + - name: OPENAI_TEAM_B_KEY + valueFrom: {secretKeyRef: {name: lightspeed-inference-creds, key: OPENAI_TEAM_B_KEY}} + - name: POSTGRES_PASSWORD + valueFrom: {secretKeyRef: {name: lightspeed-postgres, key: password}} +volumeMounts: + - {name: gw-tls, mountPath: /etc/openshell-tls, readOnly: true} +volumes: + - name: gw-tls + secret: {secretName: openshell-gateway-tls} +``` + +The stack's ServiceAccount also needs `get` on the MCP secrets (least +privilege: `resourceNames` listing exactly the K8s-bound Secrets; `verify.py` +checks the list matches the registry, and post-#269 adds the inference Secret). + +### Step 2. Define the provider catalog, secret registry, and defaults + +In `lightspeed-stack.yaml` under `workflow_engine` (`PLAN(stack#51)`): + +- `providers`: one entry per logical provider. Each has exactly one + `credential`, so choosing a credential means choosing an entry. + `executor_type` must be in cloud-agents `APPROVED_INFERENCE_PROVIDERS`; + `allowed_models: null` means any model (`[]` fails load). +- `secrets`: logical ref → backend binding (`env` / `k8s` / `file`). +- `default_provider` / `default_model`: the only defaults on cloud-agents + paths. `inference.default_*` become Llama Stack-only. + +This example ships two entries: `claude-prod` (anthropic, +`inference/anthropic-prod`, models pinned) and +`openai-team-b` (openai, `inference/openai-team-b`, any model). + +Config load fails fast instead of misbehaving at runtime: unknown +`executor_type`, dangling ref, `allowed_models: []`, or bad defaults. + +### Step 3. Register MCP servers with credentials as references + +Each `mcp_servers[*]` entry carries the operator-owned URL plus two +`PLAN(stack#51)` fields: + +- `workflow_enabled: true` — required before workflow runs can use it. +- `secret_headers: {Header: {name: }}` — header name to a + registry ref, never a value. + +This example: `github-readonly` (`Authorization` → `mcp/github-readonly`) +and `kb-search` (`X-API-Key` → `mcp/kb-search`). + +Refused at load for `workflow_enabled` entries: request-bound +`authorization_headers` (`client`/`oauth`/`kubernetes`) and non-empty +propagated `headers` (e.g. `x-rh-identity`) — a detached run has no incoming +request, and silently dropping an identity header would run the server without +the caller's identity. The stack generates `MCP_ALLOWED_SECRETS` at startup +from the registry and refuses to start if the process env disagrees. + +### Step 4. Provide skills via the sandbox image, govern with policy + +Skills are directories baked into the sandbox image — the plan does not +change skill packaging: + +```dockerfile +FROM quay.io/example/lightspeed-agentic-sandbox@sha256:a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4 +COPY ./skills/triage /skills/triage +COPY ./skills/kb-search /skills/kb-search +``` + +Enforcement differs by spawn mode: ephemeral gets a per-spawn Landlock +read-only grant on `/skills/`; none/local get a +`SkillsCapability(include=[...])` allow-list. Default is least privilege: +omitted `allowed_skills` means no skills visible, not all. Which roles may +request which skill names is policy (next step). `advisory: true` (blanket +filesystem read in the sandbox) needs an explicit grant. + +### Step 5. Write policy rules per role + +`workflow_engine.policy.rules` (`PLAN(stack#51)`): each rule grants to a set +of roles; a principal's grant is the union of matching rules; nothing is +granted by default. Coarse RBAC (`WORKFLOW_START` etc. in `authorization`) +still gates "may submit at all"; policy gates "may use this". + +This example's three roles: + +- `agent-user` (triage flow): `claude-prod`, `github-readonly` + + `mcp/github-readonly`, skill `triage`, tool `github.read`, spawn + `ephemeral`, the pinned sandbox image, service account `workflow-runner`, + namespace `sandbox-workloads`, limits (900s / 200k tokens / 2 retries). +- `agent-support` (KB flow): `openai-team-b`, `kb-search` + + `mcp/kb-search`, skill `kb-search`, tool `kb.search`, spawn `ephemeral`, + same image, service account `kb-runner`, same namespace, limits + (1200s / 200k tokens / 1 retry). +- `agent-admin`: inherits both, plus both providers, `local`/`none` spawn, + `inline_mcp` (+ allowed hosts, https-only, SSRF-checked), `advisory`. + +### Step 6. Configure the spawner (OpenShell gateway on K8s) + +```yaml +spawner: + type: openshell + openshell_gateway_url: "openshell-gateway.lightspeed.svc:443" + openshell_workspace: lcore-prod + sandbox_image: quay.io/example/lightspeed-agentic-sandbox@sha256:a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4 + openshell_tls_ca: /etc/openshell-tls/ca.crt # from openshell-gateway-tls + openshell_tls_cert: /etc/openshell-tls/tls.crt + openshell_tls_key: /etc/openshell-tls/tls.key + max_pods: 50 +``` + +The gateway owns the K8s-vs-Podman compute decision; the stack proxies +sandbox lifecycle through it. Ephemeral sandboxes receive credentials via +OpenShell provider injection — placeholder in the sandbox env, value only in +the supervisor proxy. + +### Step 7. Deploy and verify the boundary + +```bash +kubectl -n lightspeed create configmap lightspeed-config \ + --from-file=lightspeed-stack.yaml=./lightspeed-stack.yaml +kubectl apply -f k8s-deployment.yaml +kubectl -n lightspeed rollout status deploy/lightspeed-stack +kubectl -n lightspeed logs deploy/lightspeed-stack | grep -i "config\|provider\|MCP_ALLOWED" +``` + +Expected: clean start, generated `MCP_ALLOWED_SECRETS` (names only). Then +prove the denials from Part 2: caller-sent `credentials_secret` → 400, +unknown provider/MCP name → 400, valid-but-ungranted item → 403 with reasons, +`spawn: none` as `agent-user` on workflows → 403. + +### Step 8. Rotation and revocation on K8s (pre-cloud-agents#269 rules) + +- Update the value in Vault; External Secrets syncs the K8s Secret within + `refreshInterval` and Reloader (annotation on the Deployment) **rolls the + stack pods** so they pick up the new env values. By hand: `kubectl edit + secret` + `kubectl rollout restart`. In-flight sandbox steps finish on the old value; new submissions + use the new one after restart. +- Removing a ref from `secrets`/`policy` affects new submissions only — + running steps are unaffected. +- After cloud-agents#269 (lease handoff), revocation fails closed at the + next step (`credential_revoked`) and rotation applies at the next + `acquire_value` with no restart. + +--- + +## Part 2 — Caller: author and trigger a workflow (per run) + +### Step 9. Author the `WorkflowDefinition` + +Schema: `apiVersion` / `kind: AgentWorkflow` / `metadata` / `spec`, optional +top-level `provider` default and `skills` image. Per step: + +- **Input**: `spec.input_prompt`, per-step `prompt` templates with + `{{ steps.X.output.Y }}` references, per-step `context` dicts. +- **Output**: per-step `output_key` (state key) and optional `output_schema` + (dict/JSON-schema constraining that step's output). No workflow-level + output schema — results are read from state/transcripts. +- **Needs**: `mcp_servers` (catalog **names**), `allowed_skills`, `tools` / + `permissions` (service account, `allowed_tools`, `max_tokens`), + `target_namespaces`, `timeout_seconds`, `max_retries`, `spawn`. +- **Provider**: run-level `ProviderSelection {name, model, credential_ref?}` + travels in the API request, not the definition. Optional + `definition.provider` and per-step `inference_provider` must resolve to the + **same catalog entry** as the run provider (same-entry rule; model may + differ). Omit `credentials_secret` everywhere — it is rejected (400). +- **Never include**: secret values, raw `{secret_name, key}` headers, env var + names, K8s Secret names, arbitrary images. + +Merge rules the policy engine applies per step (same precedence as +cloud-agents' normalizer): `spawn` = step → workflow → `ephemeral`; +`mcp_servers` / `allowed_skills` / `permissions` / `service_account` = step → +workflow where `None` inherits and `[]` means explicitly empty; +`sandbox_image` = step `spawn_config` → workflow `spawn_config` → run value +→ spawner default; provider = step → `definition.provider` → run provider. + +The workflow definitions in this folder (plus `admin-inline-mcp.yaml`, Part 3): + +- `triage-github-issue.yaml`: triage → report; inherits run provider + `claude-prod`; workflow-default MCP/skills with the `report` step opting + out via `[]`. +- `kb-answer-with-approval.yaml`: research → human-approval `review` → + respond; `definition.provider` = run entry `openai-team-b`; `research` + overrides to model `gpt-4o-mini` on the same entry (allowed). + +### Step 10. Trigger it: `POST /v1/workflows/run` + +The definition travels inline in `definition`; the run-level provider is a +logical `ProviderSelection`. No secret material anywhere in the request. +The 8-step gate (authenticate → size → parse → catalog → policy → executor +validate → persist/start → audit) runs before anything is saved → `202`. + +Triage flow (as `agent-user`): + +```bash +DEFN=$(python3 -c "import json,yaml;print(json.dumps(yaml.safe_load(open('triage-github-issue.yaml'))))") +curl -s -X POST https://lightspeed.example.com/v1/workflows/run \ + -H "Authorization: Bearer $USER_TOKEN" -H 'Content-Type: application/json' \ + -d "$(jq -n --argjson defn "$DEFN" '{definition: $defn, + provider: {name: "claude-prod", model: "claude-sonnet-4-5"}}')" +# -> 202 {"workflow_id": "", "status": "running"} +``` + +KB flow (as `agent-support`): + +```bash +DEFN=$(python3 -c "import json,yaml;print(json.dumps(yaml.safe_load(open('kb-answer-with-approval.yaml'))))") +curl -s -X POST https://lightspeed.example.com/v1/workflows/run \ + -H "Authorization: Bearer $SUPPORT_TOKEN" -H 'Content-Type: application/json' \ + -d "$(jq -n --argjson defn "$DEFN" '{definition: $defn, + provider: {name: "openai-team-b", model: "gpt-4o"}}')" +# -> 202 {"workflow_id": "", "status": "running"} +``` + +### Step 11. Follow the run; approve the gate + +```bash +# Status / transcripts (both flows) +curl -s https://lightspeed.example.com/v1/workflows/ \ + -H "Authorization: Bearer $TOKEN" +curl -s https://lightspeed.example.com/v1/workflows//transcripts \ + -H "Authorization: Bearer $TOKEN" + +# KB flow pauses at the `review` step; a role with workflow_approve continues it: +curl -s -X POST https://lightspeed.example.com/v1/workflows//approve \ + -H "Authorization: Bearer $SUPPORT_TOKEN" -H 'Content-Type: application/json' \ + -d '{"step_name": "review", "decision": "approved", "approver": "support-lead"}' + +# Cancel a running workflow (role needs workflow_cancel): +curl -s -X POST https://lightspeed.example.com/v1/workflows//cancel \ + -H "Authorization: Bearer $TOKEN" +``` + +Denials to try: add `"credentials_secret"` to the request → 400; unknown +provider/MCP name → 400; `openai-team-b` as `agent-user` → 403; a step naming +a different catalog entry than the run provider → 400 (same-entry rule); +`spawn: none` as non-admin on workflows → 403. + +--- + +## Part 3 — Other surfaces + +> Chat on cloud-agents (`/query/direct`) is out of scope here; it was removed and is +> tracked in [#59](https://github.com/jameswnl/lightspeed-stack/issues/59). + +### Admin inline MCP + +`admin-inline-mcp.yaml` is the exception path: `https` only, no userinfo, +host in `inline_mcp_hosts`, no literal headers, and `secret_headers` as +registry refs (`{name: mcp/incidents}`) that the stack rewrites to +cloud-agents' form. The ref must also be granted via `mcp_secrets`. Prefer a +catalog entry; follow-up: named workflow templates so trusted pipelines do +not need `inline_mcp`. + +### Audit events + +Every decision leaves one structured, names-only event on the dedicated +`audit` logger (route it to its own sink). Examples of what operators should +see for the flows above: + +```json +{"event": "workflow_authorized", "principal": "alice", "roles": ["agent-user"], + "workflow_id": "wf-8f2c", "provider": "claude-prod", "model": "claude-sonnet-4-5", + "credential_ref": "inference/anthropic-prod", "mcp_servers": ["github-readonly"], + "spawn": {"triage": "ephemeral", "report": "ephemeral"}} +{"event": "workflow_denied", "principal": "alice", "status": 403, + "reasons": ["provider 'openai-team-b' not granted"]} +{"event": "secret_accessed", "workflow_id": "wf-8f2c", "step": "triage", + "purpose": "inference:anthropic", "ref": "inference/anthropic-prod", + "backend": "k8s", "outcome": "granted"} +{"event": "lease_released", "workflow_id": "wf-8f2c", "step": "triage", + "purpose": "inference:anthropic", "reason": "completed"} +``` + +No secret value appears in any event, log line, span attribute, metric label, +state record, transcript or API response (Phase 4 canary suite). + +### After cloud-agents#269 (`post-269-overrides.yaml`) + +`verify.py` runs every case against both stages, so you can see exactly what +changes: + +| | Before cloud-agents#269 | After | +|---|---|---| +| Inference credential binding | `backend: env`; stack must be restarted to rotate | `backend: k8s`; read by the stack through the lease provider | +| Handoff | names (`env` key, K8s Secret name) | short-lived lease per step, redeemed then released; cross-process via mTLS | +| Revocation | new submissions only | next step fails closed (`credential_revoked`) | +| Rotation | restart | next lease, no restart | +| `secret_headers` MCP on `none`/`local` | 403 | allowed | +| Same-entry rule | enforced | can relax (each step has its own lease) | +| `spawn: none` / `local` | admin-only (process-wide env) | admin-only lifted once cloud-agents isolates them | +| Stack RBAC | `get` on MCP Secrets | `get` on MCP Secrets + the inference Secret | + +## Gaps in the plan found while building these examples + +Folded into the plan in revision 3 (phase in brackets): + +1. **Inline MCP refs vs the typed parse.** cloud-agents' `MCPServerConfig` + accepts `secret_headers` only as `{secret_name, key}` (`extra=forbid`), so a + registry ref `{name: ...}` fails pipeline step 3 with 422 [Phase 3]. The stack must + rewrite inline registry refs to the executor form for the shape check + (modelled in `reference_gate._shape_copy`). +2. **`MCP_ALLOWED_SECRETS` must cover inline-granted secrets.** The plan + generates it from `workflow_enabled` catalog servers only; a Secret reachable + only through `rules[].mcp_secrets` (admin inline MCP) would be blocked by the + runtime guardrail [Phase 3]. The generator here unions both. +3. **`ADMIN` is never in `authorized_actions`.** Admin gates use the access + resolver with roles stored on `request.state.user_roles` (PR #57), and no-op + auth modules (`k8s`, `noop`, `api-key`) make every caller admin [0a done; + load check in Phase 2, decision D8]. +4. **`Authorization` header values.** The K8s Secret holds the full header + value (including `Bearer `); confirm cloud-agents does not add a prefix [Phase 3 round-trip test]. + +--- + +## Verification + +`verify.py` is the executable contract. Run it from the repo root with the +project venv: + +```bash +uv run python examples/workflows/verify.py +``` + +What it checks (~164 assertions): + +- Every file parses; all three definitions pass the real cloud-agents + `WorkflowDefinition` shape check. +- `lightspeed-stack.yaml` passes the plan's load-time checks in both stages + (unknown `executor_type`, `allowed_models: []`, unbound refs, bad defaults, + request-bound/propagated MCP headers, undefined + policy names), and each of those failures is also pinned as a negative case. +- `cases.yaml`: accepted requests plus every denial (400 / 403 / 413 / 422) + for credentials, catalog, policy, spawn, images, advisory, inline MCP, + and limits, in `pre269` and `post269`. +- The non-`PLAN` subset validates against the current `Configuration` schema. +- The real `JwtRolesResolver` / `GenericAccessResolver` give the roles and + admin semantics the policy assumes. +- Deployment cross-checks: env injection and `MCP_ALLOWED_SECRETS` match the + registry, RBAC `resourceNames` match the K8s-bound Secrets, pod hardening, + digest-pinned images, and the dev and External Secrets sources agree. + +`reference_gate.py` is a model of the design, not the implementation. As each +phase lands, run the same `cases.yaml` against the real endpoint and delete the +matching part of the reference. + +## Production checklist + +- Images pinned by digest (stack, sandbox, policy `sandbox_images`). +- Values only in the secret manager; K8s Secrets synced, never committed. +- Role-resolving auth (`jwk-token` or `rh-identity`), not `k8s`/`noop`. +- `ssl_mode: verify-full` to Postgres; mTLS to the OpenShell gateway. +- Least-privilege RBAC by `resourceNames`; default-deny egress NetworkPolicy. +- Audit logger routed to a separate sink; alerts on `workflow_denied` spikes. +- Rotation runbook: pre-#269 restart via Reloader; post-#269 next lease. +- Deprecation window for the legacy `{name, model}` provider form + (`Deprecation` / `Sunset` headers) communicated to callers before Phase 1. diff --git a/examples/workflows/admin-inline-mcp.yaml b/examples/workflows/admin-inline-mcp.yaml new file mode 100644 index 000000000..75c57babd --- /dev/null +++ b/examples/workflows/admin-inline-mcp.yaml @@ -0,0 +1,36 @@ +# ============================================================================= +# admin-inline-mcp.yaml — trusted-admin workflow using an inline MCP server. +# +# Inline MCP is the exception path (role agent-admin, rules[].inline_mcp): +# - https only, no userinfo in the URL, host must be in inline_mcp_hosts +# - secret_headers carry REGISTRY refs ({name: ...}); the stack validates the +# ref and rewrites it to cloud-agents' {secret_name, key} form. A raw +# {secret_name, key} from a caller is rejected (400). +# - the (principal, server, ref) triple must be granted via mcp_secrets +# Prefer named catalog entries; use this only for one-off trusted pipelines. +# ============================================================================= +apiVersion: cloudagents/v1 +kind: AgentWorkflow +metadata: + name: admin-inline-mcp + description: Query an operator-approved ad-hoc MCP endpoint. + +spec: + input_prompt: "List the open incidents." + spawn: ephemeral + service_account: workflow-runner + timeout_seconds: 600 + steps: + - name: query + type: agent + agent: triage-agent + prompt: Use the incident tools to list open incidents. + output_key: incidents + mcp_servers: + - name: incidents + url: https://mcp.internal.example.com/incidents + secret_headers: + Authorization: {name: mcp/incidents} + timeout_seconds: 300 + permissions: + max_tokens: 20000 diff --git a/examples/workflows/cases.yaml b/examples/workflows/cases.yaml new file mode 100644 index 000000000..9d75f4bac --- /dev/null +++ b/examples/workflows/cases.yaml @@ -0,0 +1,318 @@ +# ============================================================================= +# cases.yaml — the behavioural contract for stack#51 (feature-level TDD). +# +# Each case is a request (or a config) and the status the gate must return. +# verify.py runs them against reference_gate.py today; as phases land they +# should run against the real endpoint and the reference code is deleted. +# +# workflow: base definition file; `set` patches it (dotted path, list items +# addressed by index or by their `name`); `body_set` patches the +# request body; `generate` builds oversized inputs +# as: roles of the caller (default per workflow below) +# expect: HTTP status, or {pre269: N, post269: N} when the stages differ +# reason: substring that must appear in the returned reasons +# spawner: false = stack has no spawner_configuration +# Statuses: 400 unknown to the deployment / forbidden field, 403 exists but not +# for this caller, 413 too large, 422 shape or count, 202 accepted. +# ============================================================================= +defaults: + triage-github-issue: + as: [agent-user] + provider: {name: claude-prod, model: claude-sonnet-4-5} + kb-answer-with-approval: + as: [agent-support] + provider: {name: openai-team-b, model: gpt-4o} + admin-inline-mcp: + as: [agent-admin] + provider: {name: claude-prod, model: claude-sonnet-4-5} + +workflow_cases: + # ---- accepted ------------------------------------------------------------- + - {name: triage as agent-user, workflow: triage-github-issue, expect: 202} + - {name: kb answer as agent-support, workflow: kb-answer-with-approval, expect: 202} + - {name: admin inline MCP with registry ref, workflow: admin-inline-mcp, expect: 202} + - name: credential_ref equal to the entry's credential is accepted + workflow: triage-github-issue + body_set: {provider.credential_ref: {name: inference/anthropic-prod}} + expect: 202 + - name: omitted provider uses workflow_engine defaults + workflow: triage-github-issue + body_set: {provider: null} + expect: 202 + - name: step may pick another model on the same entry + workflow: kb-answer-with-approval + set: {spec.steps.respond.inference_provider: {name: openai-team-b, model: gpt-4o-mini}} + expect: 202 + - name: admin may run spawn none (no secret_headers servers) + workflow: triage-github-issue + as: [agent-admin] + set: {spec.spawn: none, spec.mcp_servers: [], spec.allowed_skills: []} + expect: 202 + - name: admin may use the admin-only azure entry with its pinned model + workflow: triage-github-issue + as: [agent-admin] + body_set: {provider: {name: azure-eu, model: gpt-4o}} + set: {spec.mcp_servers: [], spec.allowed_skills: []} + expect: 202 + + # ---- caller cannot choose credentials (G1, G2) ---------------------------- + - name: run provider credentials_secret (env var name) is rejected + workflow: triage-github-issue + body_set: {provider.credentials_secret: DATABASE_URL} + expect: 400 + reason: credentials_secret + - name: run provider credentials_secret is rejected even when null + workflow: triage-github-issue + body_set: {provider.credentials_secret: null} + expect: 400 + - name: credentials_secret is rejected for admins too + workflow: admin-inline-mcp + body_set: {provider.credentials_secret: ANTHROPIC_API_KEY} + expect: 400 + - name: definition.provider credentials_secret is rejected + workflow: kb-answer-with-approval + set: {provider.credentials_secret: DATABASE_URL} + expect: 400 + - name: credential_ref for another entry's credential is rejected + workflow: triage-github-issue + body_set: {provider.credential_ref: {name: inference/openai-team-b}} + expect: 400 + reason: credential_ref + - name: unknown keys in the provider are rejected + workflow: triage-github-issue + body_set: {provider.base_url: https://evil.example.com} + expect: 400 + + # ---- catalog: unknown to the deployment (400) ------------------------------ + - name: unknown provider + workflow: triage-github-issue + body_set: {provider.name: gpt-everything} + expect: 400 + reason: unknown provider + - name: model outside allowed_models + workflow: triage-github-issue + body_set: {provider.model: claude-opus-4} + expect: 400 + reason: not allowed + - name: step provider on another catalog entry (same-entry rule) + workflow: triage-github-issue + set: {spec.steps.triage.inference_provider: {name: openai-team-b, model: gpt-4o}} + as: [agent-admin] + expect: 400 + reason: same catalog entry + - name: definition.provider on another catalog entry (same-entry rule) + workflow: triage-github-issue + set: {provider: {name: openai-team-b, model: gpt-4o}} + as: [agent-admin] + expect: 400 + reason: same catalog entry + - name: unknown MCP server name + workflow: triage-github-issue + set: {spec.mcp_servers: [github-readwrite]} + expect: 400 + reason: unknown MCP + - name: inline MCP with a raw secret_name/key header (no registry ref) + workflow: admin-inline-mcp + set: + spec.steps.query.mcp_servers.0.secret_headers: + Authorization: {secret_name: lightspeed-postgres, key: password} + expect: 400 + reason: registry refs + - name: inline MCP with an unknown registry ref + workflow: admin-inline-mcp + set: {spec.steps.query.mcp_servers.0.secret_headers.Authorization.name: mcp/nope} + expect: 400 + + # ---- policy: exists, not for this caller (403) ----------------------------- + - name: agent-user cannot use the openai-team-b entry + workflow: triage-github-issue + body_set: {provider: {name: openai-team-b, model: gpt-4o}} + expect: 403 + reason: not granted + - name: agent-support cannot use the admin-only azure entry + workflow: kb-answer-with-approval + body_set: {provider: {name: azure-eu, model: gpt-4o}} + set: + provider: {name: azure-eu, model: gpt-4o} + spec.steps.research.inference_provider: {name: azure-eu, model: gpt-4o} + expect: 403 + - name: agent-support cannot run the triage flow (provider, MCP, skill, tool) + workflow: triage-github-issue + as: [agent-support] + expect: 403 + - name: caller without workflow_start + workflow: triage-github-issue + as: [] + expect: 403 + reason: workflow_start + - name: skill not granted + workflow: triage-github-issue + set: {spec.steps.triage.allowed_skills: [kb-search]} + expect: 403 + reason: skill + - name: tool not granted + workflow: triage-github-issue + set: {spec.steps.triage.tools: [github.write]} + expect: 403 + reason: tool + - name: service account not granted + workflow: triage-github-issue + set: {spec.service_account: kb-runner} + expect: 403 + reason: service account + - name: namespace not granted + workflow: triage-github-issue + set: {spec.steps.triage.target_namespaces: [kube-system]} + expect: 403 + reason: namespace + - name: step timeout over the role limit + workflow: triage-github-issue + set: {spec.steps.triage.timeout_seconds: 3600} + expect: 403 + reason: timeout + - name: step max_tokens over the role limit + workflow: triage-github-issue + set: {spec.steps.triage.permissions.max_tokens: 900000} + expect: 403 + reason: max_tokens + - name: step max_retries over the role limit + workflow: triage-github-issue + set: {spec.steps.triage.max_retries: 9} + expect: 403 + reason: max_retries + - name: workflow timeout over the role limit + workflow: triage-github-issue + set: {spec.timeout_seconds: 7200} + expect: 403 + + # ---- spawn modes, images, advisory ----------------------------------------- + - name: spawn none as agent-user + workflow: triage-github-issue + set: {spec.spawn: none, spec.mcp_servers: [], spec.allowed_skills: []} + expect: 403 + reason: spawn + - name: spawn local on one step as agent-user + workflow: triage-github-issue + set: {spec.steps.report.spawn: local} + expect: 403 + - name: ephemeral steps without spawner_configuration + workflow: triage-github-issue + spawner: false + expect: 403 + reason: spawner_configuration + - name: custom image on the request + workflow: triage-github-issue + body_set: {sandbox_image: evil.example.com/sandbox:latest} + expect: 403 + reason: image + - name: custom image on a step spawn_config + workflow: triage-github-issue + set: {spec.steps.triage.spawn_config.sandbox_image: evil.example.com/sandbox:latest} + expect: 403 + - name: custom skills image + workflow: triage-github-issue + set: {skills: {image: evil.example.com/skills:latest}} + expect: 403 + - name: advisory mode as agent-user + workflow: triage-github-issue + set: {advisory: true} + expect: 403 + reason: advisory + - name: advisory mode as admin + workflow: triage-github-issue + as: [agent-admin] + set: {advisory: true} + expect: 202 + - name: secret_headers MCP on spawn none (needs cloud-agents#269) + workflow: triage-github-issue + as: [agent-admin] + set: {spec.spawn: none} + expect: {pre269: 403, post269: 202} + reason: secret_headers + + # ---- inline MCP ------------------------------------------------------------- + - name: inline MCP as agent-user + workflow: admin-inline-mcp + as: [agent-user] + expect: 403 + reason: inline MCP + - name: inline MCP host not on the allow-list + workflow: admin-inline-mcp + set: {spec.steps.query.mcp_servers.0.url: "https://evil.example.com/mcp"} + expect: 403 + reason: host + - name: inline MCP over plain http + workflow: admin-inline-mcp + set: {spec.steps.query.mcp_servers.0.url: "http://mcp.internal.example.com/incidents"} + expect: 403 + reason: https + - name: inline MCP with credentials in the URL + workflow: admin-inline-mcp + set: {spec.steps.query.mcp_servers.0.url: "https://user:pw@mcp.internal.example.com/incidents"} + expect: 403 + reason: userinfo + - name: inline MCP with a literal header value + workflow: admin-inline-mcp + set: {spec.steps.query.mcp_servers.0.headers: {Authorization: "Bearer ghp_x"}} + expect: 403 + reason: literal header + + # ---- limits (G9) ------------------------------------------------------------- + - name: definition over the byte cap + workflow: triage-github-issue + generate: {pad_bytes: 300000} + expect: 413 + - name: too many steps + workflow: triage-github-issue + generate: {steps: 65} + expect: 422 + - name: too many MCP servers on the workflow default + workflow: triage-github-issue + generate: {mcp_servers: 17} + expect: 422 + - name: too many secret_headers on an inline server + workflow: admin-inline-mcp + generate: {secret_headers: 33} + expect: 422 + - name: missing spec.steps + workflow: triage-github-issue + set: {spec.steps: []} + expect: 422 + - name: legacy step "provider" alias is a shape error + workflow: triage-github-issue + set: {spec.steps.triage.provider: {name: claude-prod}} + expect: 422 + +# Config-load failures. `set` patches the parsed lightspeed-stack.yaml. +config_cases: + - name: allowed_models [] is rejected + set: {workflow_engine.providers.claude-prod.allowed_models: []} + error: allowed_models + - name: executor_type outside APPROVED_INFERENCE_PROVIDERS + set: {workflow_engine.providers.claude-prod.executor_type: ollama} + error: not approved + - name: credential without a secrets binding + set: {workflow_engine.providers.claude-prod.credential.name: inference/missing} + error: no secrets binding + - name: pre-269 inference credential must be env-backed + set: {workflow_engine.secrets.inference/anthropic-prod: {name: inference/anthropic-prod, backend: k8s, secret_name: s, key: k}} + stage: pre269 + error: must be env + - name: default_provider must be in the catalog + set: {workflow_engine.default_provider: gone} + error: default_provider + - name: default_model must pass the default provider's allowed_models + set: {workflow_engine.default_model: claude-opus-4} + error: default_model + - name: workflow_enabled MCP with request-bound auth + set: {mcp_servers.github-readonly.authorization_headers: {Authorization: client}} + error: workflow_enabled + - name: workflow_enabled MCP with propagated headers + set: {mcp_servers.kb-search.headers: [x-rh-identity]} + error: propagated + - name: MCP secret ref without a binding + set: {mcp_servers.kb-search.secret_headers.X-API-Key.name: mcp/missing} + error: no secrets binding + - name: policy rule naming an undefined provider + set: {workflow_engine.policy.rules.0.providers: [ghost]} + error: unknown provider diff --git a/examples/workflows/external-secrets.yaml b/examples/workflows/external-secrets.yaml new file mode 100644 index 000000000..7a79d43ea --- /dev/null +++ b/examples/workflows/external-secrets.yaml @@ -0,0 +1,83 @@ +# ============================================================================= +# external-secrets.yaml — production source of the Secrets in k8s-secrets.yaml. +# +# Values live in the enterprise secret manager (Vault here) and are synced into +# K8s Secrets by External Secrets Operator, so no value is committed to git or +# applied by hand. The stack consumes the resulting K8s Secrets exactly as in +# k8s-secrets.yaml (env injection pre-#269, API reads post-#269), so it needs +# no manager-specific code until a native backend lands. +# k8s-secrets.yaml remains the bootstrap/dev path. +# ============================================================================= +apiVersion: external-secrets.io/v1 +kind: ExternalSecret +metadata: {name: lightspeed-inference-creds, namespace: lightspeed} +spec: + refreshInterval: 1h + secretStoreRef: {kind: ClusterSecretStore, name: vault-lightspeed} + target: {name: lightspeed-inference-creds, creationPolicy: Owner} + data: + - {secretKey: ANTHROPIC_API_KEY, remoteRef: {key: lightspeed/inference/anthropic-prod, property: api_key}} + - {secretKey: OPENAI_TEAM_B_KEY, remoteRef: {key: lightspeed/inference/openai-team-b, property: api_key}} + - {secretKey: AZURE_OPENAI_API_KEY, remoteRef: {key: lightspeed/inference/azure-eu, property: api_key}} +--- +apiVersion: external-secrets.io/v1 +kind: ExternalSecret +metadata: {name: mcp-github-readonly-token, namespace: lightspeed} +spec: + refreshInterval: 1h + secretStoreRef: {kind: ClusterSecretStore, name: vault-lightspeed} + target: {name: mcp-github-readonly-token, creationPolicy: Owner} + data: + - {secretKey: token, remoteRef: {key: lightspeed/mcp/github-readonly, property: authorization}} +--- +apiVersion: external-secrets.io/v1 +kind: ExternalSecret +metadata: {name: mcp-kb-search-token, namespace: lightspeed} +spec: + refreshInterval: 1h + secretStoreRef: {kind: ClusterSecretStore, name: vault-lightspeed} + target: {name: mcp-kb-search-token, creationPolicy: Owner} + data: + - {secretKey: api-key, remoteRef: {key: lightspeed/mcp/kb-search, property: api_key}} +--- +apiVersion: external-secrets.io/v1 +kind: ExternalSecret +metadata: {name: mcp-incidents-token, namespace: lightspeed} +spec: + refreshInterval: 1h + secretStoreRef: {kind: ClusterSecretStore, name: vault-lightspeed} + target: {name: mcp-incidents-token, creationPolicy: Owner} + data: + - {secretKey: token, remoteRef: {key: lightspeed/mcp/incidents, property: authorization}} +--- +apiVersion: external-secrets.io/v1 +kind: ExternalSecret +metadata: {name: lightspeed-postgres, namespace: lightspeed} +spec: + refreshInterval: 1h + secretStoreRef: {kind: ClusterSecretStore, name: vault-lightspeed} + target: {name: lightspeed-postgres, creationPolicy: Owner} + data: + - {secretKey: password, remoteRef: {key: lightspeed/postgres, property: password}} +--- +apiVersion: external-secrets.io/v1 +kind: ExternalSecret +metadata: {name: lightspeed-postgres-tls, namespace: lightspeed} +spec: + refreshInterval: 24h + secretStoreRef: {kind: ClusterSecretStore, name: vault-lightspeed} + target: {name: lightspeed-postgres-tls, creationPolicy: Owner} + data: + - {secretKey: ca.crt, remoteRef: {key: lightspeed/postgres-tls, property: ca}} +--- +apiVersion: external-secrets.io/v1 +kind: ExternalSecret +metadata: {name: openshell-gateway-tls, namespace: lightspeed} +spec: + refreshInterval: 24h + secretStoreRef: {kind: ClusterSecretStore, name: vault-lightspeed} + target: {name: openshell-gateway-tls, creationPolicy: Owner, template: {type: kubernetes.io/tls}} + data: + - {secretKey: ca.crt, remoteRef: {key: lightspeed/openshell-gateway-tls, property: ca}} + - {secretKey: tls.crt, remoteRef: {key: lightspeed/openshell-gateway-tls, property: cert}} + - {secretKey: tls.key, remoteRef: {key: lightspeed/openshell-gateway-tls, property: key}} diff --git a/examples/workflows/k8s-deployment.yaml b/examples/workflows/k8s-deployment.yaml new file mode 100644 index 000000000..4be315573 --- /dev/null +++ b/examples/workflows/k8s-deployment.yaml @@ -0,0 +1,179 @@ +# ============================================================================= +# k8s-deployment.yaml — stack Deployment and its least-privilege surroundings +# (namespace: lightspeed). Pairs with lightspeed-stack.yaml (mounted from the +# `lightspeed-config` ConfigMap) and the Secrets from external-secrets.yaml. +# +# verify.py cross-checks this file against the stack config: +# - every `backend: env` secret is injected as that env var, from a Secret +# - MCP_ALLOWED_SECRETS equals the value the stack would generate +# - the Role grants `get` on exactly the K8s Secrets the registry binds +# - pod hardening (non-root, read-only rootfs, no privilege escalation) +# ============================================================================= +apiVersion: v1 +kind: ServiceAccount +metadata: + name: lightspeed-stack + namespace: lightspeed +automountServiceAccountToken: true # k8s auth + Secret reads +--- +# Pre-cloud-agents#269: only the MCP secrets are read via the API. Inference +# keys arrive as env (backend: env), so they need no RBAC. +apiVersion: rbac.authorization.k8s.io/v1 +kind: Role +metadata: + name: lightspeed-stack-secret-reader + namespace: lightspeed +rules: + - apiGroups: [""] + resources: [secrets] + verbs: [get] + resourceNames: + - mcp-github-readonly-token + - mcp-kb-search-token + - mcp-incidents-token +--- +# Post-cloud-agents#269 variant (bind this one instead of the Role above): the +# lease provider also reads the inference Secret through the API. +apiVersion: rbac.authorization.k8s.io/v1 +kind: Role +metadata: + name: lightspeed-stack-secret-reader-post-269 + namespace: lightspeed + annotations: + example.lightspeed/stage: post-269 +rules: + - apiGroups: [""] + resources: [secrets] + verbs: [get] + resourceNames: + - lightspeed-inference-creds + - mcp-github-readonly-token + - mcp-kb-search-token + - mcp-incidents-token +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: RoleBinding +metadata: + name: lightspeed-stack-secret-reader + namespace: lightspeed +subjects: + - {kind: ServiceAccount, name: lightspeed-stack, namespace: lightspeed} +roleRef: + apiGroup: rbac.authorization.k8s.io + kind: Role + name: lightspeed-stack-secret-reader +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: lightspeed-stack + namespace: lightspeed + annotations: + # Pre-#269 rotation needs a restart; Reloader rolls the pods when the + # synced Secrets change (External Secrets updates them on refreshInterval). + reloader.stakater.com/auto: "true" +spec: + replicas: 2 + strategy: {type: RollingUpdate, rollingUpdate: {maxUnavailable: 0, maxSurge: 1}} + selector: + matchLabels: {app: lightspeed-stack} + template: + metadata: + labels: {app: lightspeed-stack} + spec: + serviceAccountName: lightspeed-stack + securityContext: + runAsNonRoot: true + seccompProfile: {type: RuntimeDefault} + topologySpreadConstraints: + - maxSkew: 1 + topologyKey: kubernetes.io/hostname + whenUnsatisfiable: ScheduleAnyway + labelSelector: {matchLabels: {app: lightspeed-stack}} + containers: + - name: lightspeed-stack + image: quay.io/example/lightspeed-stack@sha256:b1c2d3e4b1c2d3e4b1c2d3e4b1c2d3e4b1c2d3e4b1c2d3e4b1c2d3e4b1c2d3e4 + ports: [{containerPort: 8080, name: http}] + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: {drop: [ALL]} + env: + # backend: env inference bindings (pre-cloud-agents#269 only). + - name: ANTHROPIC_API_KEY + valueFrom: {secretKeyRef: {name: lightspeed-inference-creds, key: ANTHROPIC_API_KEY}} + - name: OPENAI_TEAM_B_KEY + valueFrom: {secretKeyRef: {name: lightspeed-inference-creds, key: OPENAI_TEAM_B_KEY}} + - name: AZURE_OPENAI_API_KEY + valueFrom: {secretKeyRef: {name: lightspeed-inference-creds, key: AZURE_OPENAI_API_KEY}} + - name: POSTGRES_PASSWORD + valueFrom: {secretKeyRef: {name: lightspeed-postgres, key: password}} + # Runtime guardrail only; the stack generates this value at startup + # and refuses to start if the env disagrees (so keep them equal). + - name: MCP_ALLOWED_SECRETS + value: "mcp-github-readonly-token,mcp-incidents-token,mcp-kb-search-token" + readinessProbe: + httpGet: {path: /readiness, port: http} + periodSeconds: 10 + livenessProbe: + httpGet: {path: /liveness, port: http} + initialDelaySeconds: 15 + periodSeconds: 20 + resources: + requests: {cpu: 500m, memory: 1Gi} + limits: {cpu: "2", memory: 2Gi} + volumeMounts: + - {name: config, mountPath: /app-root/lightspeed-stack.yaml, subPath: lightspeed-stack.yaml, readOnly: true} + - {name: gw-tls, mountPath: /etc/openshell-tls, readOnly: true} + - {name: pg-tls, mountPath: /etc/postgres-tls, readOnly: true} + - {name: data, mountPath: /data} + - {name: tmp, mountPath: /tmp} + volumes: + - name: config + configMap: {name: lightspeed-config} + - name: gw-tls + secret: {secretName: openshell-gateway-tls} + - name: pg-tls + secret: {secretName: lightspeed-postgres-tls} + - name: data + persistentVolumeClaim: {claimName: lightspeed-data} + - name: tmp + emptyDir: {} +--- +apiVersion: v1 +kind: Service +metadata: + name: lightspeed-stack + namespace: lightspeed +spec: + selector: {app: lightspeed-stack} + ports: [{name: http, port: 8080, targetPort: http}] +--- +# Default-deny egress with explicit allows, backing the "stack owns egress" +# goal: nothing the operator did not list is reachable from the stack pods. +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: lightspeed-stack + namespace: lightspeed +spec: + podSelector: {matchLabels: {app: lightspeed-stack}} + policyTypes: [Ingress, Egress] + ingress: + - from: + - namespaceSelector: {matchLabels: {kubernetes.io/metadata.name: openshift-ingress}} + ports: [{port: 8080}] + egress: + - to: [{namespaceSelector: {}, podSelector: {matchLabels: {k8s-app: kube-dns}}}] + ports: [{port: 53, protocol: UDP}, {port: 53, protocol: TCP}] + - to: [{podSelector: {matchLabels: {app: postgres}}}] + ports: [{port: 5432}] + - to: [{podSelector: {matchLabels: {app: llama-stack}}}] + ports: [{port: 8321}] + - to: [{podSelector: {matchLabels: {app: openshell-gateway}}}] + ports: [{port: 443}] + # kube-apiserver, LLM providers and MCP hosts all use 443/6443. NetworkPolicy + # has no FQDN match, so pin the destinations with an egress firewall or CIDR + # list limited to the endpoints in lightspeed-stack.yaml (mcp_servers[*].url, + # providers[*].base_url). + - ports: [{port: 443}, {port: 6443}] diff --git a/examples/workflows/k8s-secrets.yaml b/examples/workflows/k8s-secrets.yaml new file mode 100644 index 000000000..138c58c33 --- /dev/null +++ b/examples/workflows/k8s-secrets.yaml @@ -0,0 +1,84 @@ +# ============================================================================= +# BOOTSTRAP / DEV ONLY. Production syncs these Secrets from the secret +# manager with external-secrets.yaml; never commit real values. +# K8s Secret manifests for the workflows example (namespace: lightspeed). +# Replace every CHANGEME value before applying. These are the ONLY place +# where secret values exist: lightspeed-stack.yaml references them by name, +# and workflow definitions / API requests never mention them at all. +# +# kubectl apply -n lightspeed -f k8s-secrets.yaml +# ============================================================================= +apiVersion: v1 +kind: Secret +metadata: + name: lightspeed-inference-creds + namespace: lightspeed +type: Opaque +stringData: + # Bound to inference/anthropic-prod and inference/openai-team-b (backend: env). + # Mounted into the stack Deployment as ANTHROPIC_API_KEY / OPENAI_TEAM_B_KEY. + ANTHROPIC_API_KEY: CHANGEME-anthropic-key + OPENAI_TEAM_B_KEY: CHANGEME-openai-key + AZURE_OPENAI_API_KEY: CHANGEME-azure-key +--- +apiVersion: v1 +kind: Secret +metadata: + name: mcp-github-readonly-token + namespace: lightspeed +type: Opaque +stringData: + # Bound to mcp/github-readonly (backend: k8s, key: token). + # Mounted into ephemeral sandboxes as the Authorization header for github-readonly. + token: "Bearer CHANGEME-github-pat" # full header value, see README +--- +apiVersion: v1 +kind: Secret +metadata: + name: mcp-kb-search-token + namespace: lightspeed +type: Opaque +stringData: + # Bound to mcp/kb-search (backend: k8s, key: api-key). + api-key: CHANGEME-kb-api-key +--- +apiVersion: v1 +kind: Secret +metadata: + name: mcp-incidents-token + namespace: lightspeed +type: Opaque +stringData: + # Bound to mcp/incidents (backend: k8s, key: token); admin inline MCP only. + token: "Bearer CHANGEME-incidents-token" +--- +apiVersion: v1 +kind: Secret +metadata: + name: openshell-gateway-tls + namespace: lightspeed +type: kubernetes.io/tls +data: + # mTLS client identity for stack -> OpenShell gateway. Fill with base64. + ca.crt: CHANGEME-base64-gateway-ca + tls.crt: CHANGEME-base64-client-cert + tls.key: CHANGEME-base64-client-key +--- +apiVersion: v1 +kind: Secret +metadata: + name: lightspeed-postgres + namespace: lightspeed +type: Opaque +stringData: + # Injected as POSTGRES_PASSWORD for database.postgres.password (${env.POSTGRES_PASSWORD}). + password: CHANGEME-postgres-password +--- +apiVersion: v1 +kind: Secret +metadata: + name: lightspeed-postgres-tls + namespace: lightspeed +type: Opaque +stringData: + ca.crt: CHANGEME-postgres-ca-pem diff --git a/examples/workflows/kb-answer-with-approval.yaml b/examples/workflows/kb-answer-with-approval.yaml new file mode 100644 index 000000000..f6462caab --- /dev/null +++ b/examples/workflows/kb-answer-with-approval.yaml @@ -0,0 +1,110 @@ +# ============================================================================= +# kb-answer-with-approval.yaml — caller-authored workflow definition. +# +# Flow: research (search the KB via MCP) -> review (human approval gate) -> +# respond (draft the customer answer from the approved findings). +# +# Logical names used here must exist in lightspeed-stack.yaml and be granted +# to the caller's roles by workflow_engine.policy: +# provider openai-team-b (roles: agent-support) +# MCP server kb-search (roles: agent-support) +# skill kb-search (roles: agent-support) +# spawn ephemeral, image = spawner default, service account kb-runner +# +# Same-entry rule (PLAN(stack#51)): definition.provider and every step +# inference_provider must resolve to the SAME catalog entry as the run-level +# provider. The model may differ per step, as `research` shows below. +# +# Trigger: +# POST /v1/workflows/run (see README.md), then approve the gate with +# POST /v1/workflows/{id}/approve {"step_name": "review", ...} +# ============================================================================= +apiVersion: cloudagents/v1 +kind: AgentWorkflow +metadata: + name: kb-answer-with-approval + description: Research the knowledge base, get human approval, answer the customer. + +# Workflow-level provider default. Must be the same catalog entry as the +# run-level provider (openai-team-b); counts as an override and is checked. +provider: + name: openai-team-b + model: gpt-4o + # NOTE: no credentials_secret — caller-sent credential references are + # rejected (400). The stack rebuilds the provider dict from the catalog. + +spec: + input_prompt: "Customer asks: how do I rotate the database credentials for my deployment?" + + spawn: ephemeral + service_account: kb-runner + timeout_seconds: 1200 + mcp_servers: + - kb-search + + steps: + - name: research + type: agent + agent: kb-research-agent + prompt: >- + Search the knowledge base for documents answering the customer + question in the workflow input. Return the 5 most relevant passages + with document IDs and a 3-bullet extractive summary. Respond with + JSON matching the step output schema. + output_key: findings + output_schema: + type: object + required: [passages, summary_bullets] + properties: + passages: + type: array + items: + type: object + required: [doc_id, passage] + properties: + doc_id: {type: string} + passage: {type: string} + summary_bullets: + type: array + items: {type: string} + # Same catalog entry as the run provider, cheaper model for retrieval. + inference_provider: + name: openai-team-b + model: gpt-4o-mini + allowed_skills: + - kb-search + tools: + - kb.search + target_namespaces: + - sandbox-workloads + timeout_seconds: 600 + max_retries: 1 + permissions: + max_tokens: 100000 + + - name: review + type: human-approval + message: >- + Review the KB findings for the customer question before an answer + is drafted. Approve to continue, reject to stop the workflow. + output_key: decision + + - name: respond + type: agent + agent: kb-research-agent + prompt: >- + Draft a customer-facing answer (under 250 words) from these approved + findings: {{ steps.research.output.findings }} + Cite document IDs. No tool calls needed. + output_key: answer + output_schema: + type: object + required: [answer_markdown] + properties: + answer_markdown: {type: string} + # Inherits openai-team-b / gpt-4o from definition.provider; needs no + # MCP access to draft from already-approved findings. + mcp_servers: [] + timeout_seconds: 300 + permissions: + max_tokens: 20000 diff --git a/examples/workflows/lightspeed-stack.yaml b/examples/workflows/lightspeed-stack.yaml new file mode 100644 index 000000000..30c1f6a5a --- /dev/null +++ b/examples/workflows/lightspeed-stack.yaml @@ -0,0 +1,214 @@ +# ============================================================================= +# lightspeed-stack.yaml — operator configuration for K8s (namespace: lightspeed) +# +# Covers the catalog items needed by the two example workflows in this +# directory: +# - triage-github-issue.yaml (roles: agent-user) +# - kb-answer-with-approval.yaml (roles: agent-support) +# +# Blocks marked PLAN(stack#51) are proposed by +# https://github.com/jameswnl/lightspeed-stack/issues/51#issuecomment-5911977694 +# and do not load on the current schema yet. Everything else is valid today. +# Secret VALUES live only in k8s-secrets.yaml / the Deployment env — this file +# holds backend bindings (env var names, K8s Secret names), never values. +# ============================================================================= +name: Lightspeed Core Stack (workflows on K8s) + +service: + host: 0.0.0.0 + port: 8080 + auth_enabled: true + workers: 2 + color_log: false + access_log: true + +llama_stack: + use_as_library_client: false + url: http://llama-stack.lightspeed.svc:8321 + +user_data_collection: + feedback_enabled: true + feedback_storage: /data/feedback + transcripts_enabled: true + transcripts_storage: /data/transcripts + +# Llama Stack-side defaults. Once PLAN(stack#51) lands, these apply to +# Llama Stack-backed endpoints (/query, /responses) only; workflow runs +# use workflow_engine.default_* instead. +inference: + default_provider: openai + default_model: gpt-4o-mini + +database: + postgres: + host: postgres.lightspeed.svc + port: 5432 + db: lightspeed + user: lightspeed + password: "${env.POSTGRES_PASSWORD}" + namespace: public + ssl_mode: verify-full + ca_cert_path: /etc/postgres-tls/ca.crt # from Secret lightspeed-postgres-tls + +# Role-based policy needs real roles. The k8s, noop and api-key modules use +# no-op resolvers (no roles beyond "*", access_rules ignored, every caller +# counts as admin), so this example uses jwk-token with role_rules. +authentication: + module: jwk-token + skip_for_health_probes: true + skip_for_metrics: true + jwk_config: + url: https://keycloak.example.com/realms/lightspeed/protocol/openid-connect/certs + jwt_configuration: + user_id_claim: sub + username_claim: preferred_username + role_rules: + - jsonpath: "$.realm_access.roles[*]" + operator: contains + value: lightspeed-agent-user + roles: [agent-user] + - jsonpath: "$.realm_access.roles[*]" + operator: contains + value: lightspeed-agent-support + roles: [agent-support] + - jsonpath: "$.realm_access.roles[*]" + operator: contains + value: lightspeed-agent-admin + roles: [agent-admin] + +# Coarse RBAC: "may submit / view / approve / cancel at all". +# An access rule with the `admin` action cannot list other actions (config +# load fails); `admin` implies every other action. +# Fine-grained "may use THIS provider/MCP/skill/spawn/image" lives in +# workflow_engine.policy (PLAN(stack#51)) below. +authorization: + access_rules: + - role: agent-user + actions: [workflow_start, workflow_view, workflow_cancel] + - role: agent-support + actions: [workflow_start, workflow_view, workflow_approve, workflow_cancel] + - role: agent-admin + actions: [admin] + +# --------------------------------------------------------------------------- +# MCP catalog. The stack owns URL/transport/TLS/headers/secrets/egress here; +# callers only send the logical `name`. +# --------------------------------------------------------------------------- +mcp_servers: + - name: github-readonly + provider_id: model-context-protocol + url: https://mcp.internal.example.com/github + workflow_enabled: true # PLAN(stack#51): opt in to workflow use + secret_headers: # PLAN(stack#51): header -> registry ref + Authorization: {name: mcp/github-readonly} + - name: kb-search + provider_id: model-context-protocol + url: https://mcp.internal.example.com/kb + workflow_enabled: true # PLAN(stack#51) + secret_headers: # PLAN(stack#51) + X-API-Key: {name: mcp/kb-search} + +# --------------------------------------------------------------------------- +# Workflow engine + PLAN(stack#51) catalogs and policy. +# --------------------------------------------------------------------------- +workflow_engine: + enabled: true + max_concurrent_workflows: 10 + transcript_retention_days: 30 + + # PLAN(stack#51): provider catalog. Non-empty = governed mode, unknown names deny. + providers: + - name: claude-prod # logical name callers send + executor_type: anthropic # must be in APPROVED_INFERENCE_PROVIDERS + credential: {name: inference/anthropic-prod} # exactly one per entry + allowed_models: [claude-sonnet-4-5, claude-haiku-4-5] # null = any; [] = load failure + - name: openai-team-b + executor_type: openai + credential: {name: inference/openai-team-b} + # allowed_models omitted (null) = any model + - name: azure-eu # admin-only; shows operator-set base_url + executor_type: azure + credential: {name: inference/azure-eu} + allowed_models: [gpt-4o] + base_url: https://lightspeed-eu.openai.azure.com # never caller-settable + + # PLAN(stack#51): defaults for cloud-agents workflow runs. + default_provider: claude-prod + default_model: claude-sonnet-4-5 + + # PLAN(stack#51): logical ref -> backend binding. Operator-only. + secrets: + - name: inference/anthropic-prod + backend: env # must be env pre-cloud-agents#269 + env: ANTHROPIC_API_KEY # injected from Secret lightspeed-inference-creds + - name: inference/openai-team-b + backend: env + env: OPENAI_TEAM_B_KEY + - name: inference/azure-eu + backend: env + env: AZURE_OPENAI_API_KEY + - name: mcp/github-readonly + backend: k8s + secret_name: mcp-github-readonly-token + key: token + - name: mcp/kb-search + backend: k8s + secret_name: mcp-kb-search-token + key: api-key + - name: mcp/incidents # only reachable via admin inline MCP + backend: k8s + secret_name: mcp-incidents-token + key: token + + # PLAN(stack#51): per-role grants. Union of matching rules; default deny. + policy: + rules: + # Triage flow (triage-github-issue.yaml) + - roles: [agent-user] + providers: [claude-prod] + mcp_servers: [github-readonly] + mcp_secrets: [mcp/github-readonly] + spawn: [ephemeral] + sandbox_images: [quay.io/example/lightspeed-agentic-sandbox@sha256:a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4] + skills: [triage] + tools: [github.read] + service_accounts: [workflow-runner] + namespaces: [sandbox-workloads] + limits: {max_timeout_seconds: 900, max_tokens: 200000, max_retries: 2} + # KB answer flow (kb-answer-with-approval.yaml) + - roles: [agent-support] + providers: [openai-team-b] + mcp_servers: [kb-search] + mcp_secrets: [mcp/kb-search] + spawn: [ephemeral] + sandbox_images: [quay.io/example/lightspeed-agentic-sandbox@sha256:a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4] + skills: [kb-search] + tools: [kb.search] + service_accounts: [kb-runner] + namespaces: [sandbox-workloads] + limits: {max_timeout_seconds: 1200, max_tokens: 200000, max_retries: 1} + - roles: [agent-admin] + inherit: [agent-user, agent-support] + providers: [claude-prod, openai-team-b, azure-eu] + spawn: [ephemeral, local, none] + inline_mcp: true + mcp_secrets: [mcp/incidents] + inline_mcp_hosts: [mcp.internal.example.com] + advisory: true + +# --------------------------------------------------------------------------- +# Ephemeral spawner: OpenShell gateway. Sandboxes get credentials via OpenShell +# provider injection (placeholder in sandbox env, value in supervisor proxy). +# The sandbox image also carries /skills/triage and /skills/kb-search +# (see README.md). Requires PLAN(stack#51) policy + secrets above for full +# governance; spawner fields themselves are current schema. +# --------------------------------------------------------------------------- +spawner: + type: openshell + openshell_gateway_url: "openshell-gateway.lightspeed.svc:443" + openshell_workspace: lcore-prod + sandbox_image: quay.io/example/lightspeed-agentic-sandbox@sha256:a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4a1b2c3d4 + openshell_tls_ca: /etc/openshell-tls/ca.crt + openshell_tls_cert: /etc/openshell-tls/tls.crt + openshell_tls_key: /etc/openshell-tls/tls.key + max_pods: 50 diff --git a/examples/workflows/post-269-overrides.yaml b/examples/workflows/post-269-overrides.yaml new file mode 100644 index 000000000..0f1e3841f --- /dev/null +++ b/examples/workflows/post-269-overrides.yaml @@ -0,0 +1,43 @@ +# ============================================================================= +# post-269-overrides.yaml — what changes once cloud-agents#269 (credential +# leases) and stack#51 Phase 6 land. Deep-merged over lightspeed-stack.yaml +# (maps merge, lists are replaced), so `secrets` is restated in full. +# +# Differences from the pre-#269 stage: +# - inference credentials are bound to K8s Secrets, not process env: the +# Deployment no longer injects ANTHROPIC_API_KEY & co., and the stack +# hands cloud-agents a short-lived lease per step (redeemed, then released) +# - rotation applies at the next lease with no restart; revoking a ref fails +# the next step closed (`credential_revoked`) +# - secret_headers MCP servers become usable with spawn none/local +# (cloud-agents resolves refs at the last step) +# - the stack ServiceAccount needs `get` on the inference Secret as well +# (see k8s-deployment.yaml, the "post-269" Role) +# Everything else (catalog, policy, request shapes) is identical by design. +# ============================================================================= +workflow_engine: + secrets: + - name: inference/anthropic-prod + backend: k8s + secret_name: lightspeed-inference-creds + key: ANTHROPIC_API_KEY + - name: inference/openai-team-b + backend: k8s + secret_name: lightspeed-inference-creds + key: OPENAI_TEAM_B_KEY + - name: inference/azure-eu + backend: k8s + secret_name: lightspeed-inference-creds + key: AZURE_OPENAI_API_KEY + - name: mcp/github-readonly + backend: k8s + secret_name: mcp-github-readonly-token + key: token + - name: mcp/kb-search + backend: k8s + secret_name: mcp-kb-search-token + key: api-key + - name: mcp/incidents + backend: k8s + secret_name: mcp-incidents-token + key: token diff --git a/examples/workflows/reference_gate.py b/examples/workflows/reference_gate.py new file mode 100644 index 000000000..fcaa00df7 --- /dev/null +++ b/examples/workflows/reference_gate.py @@ -0,0 +1,352 @@ +"""Reference model of the #51 submission gate and load-time checks. + +This is an executable spec, not the implementation: it encodes the design in +https://github.com/jameswnl/lightspeed-stack/issues/51#issuecomment-5911977694 +so ``cases.yaml`` can pin the expected behaviour before the stack implements +it. Later phases should run the same cases against the real endpoint +(``run_cases.py`` style) and then delete the matching part of this file. + +Stages: ``pre269`` (names only, env-bound inference credentials) and +``post269`` (credential leases, any backend). +""" + +import json +import re +from typing import Any, Optional +from urllib.parse import urlsplit + +from cloud_agents.workflow.core.definition import WorkflowDefinition +from cloud_agents.workflow.core.execution import APPROVED_INFERENCE_PROVIDERS +from pydantic import ValidationError + +from workflow.limits import ( + MAX_DEFINITION_BYTES, + MAX_MCP_SERVERS_PER_STEP, + MAX_SECRET_HEADERS_PER_SERVER, + MAX_WORKFLOW_STEPS, +) + +REQUEST_BOUND = {"client", "oauth", "kubernetes"} +NAME_RE = re.compile(r"^[a-z0-9][a-z0-9/_.-]{0,127}$") + + +class Denied(Exception): + """A gate rejection with its status code and reasons.""" + + def __init__(self, status: int, *reasons: str) -> None: + super().__init__(status, reasons) + self.status = status + self.reasons = list(reasons) + + +# ----------------------------------------------------------------- load time +def generated_mcp_allowed_secrets(stack: dict) -> str: + """Value the stack writes to MCP_ALLOWED_SECRETS at startup.""" + secrets = {s["name"]: s for s in stack["workflow_engine"]["secrets"]} + names = set() + refs = { + ref["name"] + for server in stack["mcp_servers"] + if server.get("workflow_enabled") + for ref in (server.get("secret_headers") or {}).values() + } + # Inline MCP refs granted by policy are mounted too, so the runtime + # guardrail must allow them as well. + for rule in stack["workflow_engine"].get("policy", {}).get("rules", []): + refs |= set(rule.get("mcp_secrets", [])) + for ref in refs: + binding = secrets.get(ref, {}) + if binding.get("backend") == "k8s": + names.add(binding["secret_name"]) + return ",".join(sorted(names)) + + +def load_errors(stack: dict, stage: str = "pre269") -> list[str]: + """Return the config-load failures the plan requires.""" + we = stack["workflow_engine"] + providers = we.get("providers") or [] + secrets = {s["name"]: s for s in we.get("secrets", [])} + mcp = {m["name"]: m for m in stack.get("mcp_servers", [])} + errors: list[str] = [] + seen = set() + for p in providers: + if p["name"] in seen: + errors.append(f"duplicate provider {p['name']}") + seen.add(p["name"]) + if p["executor_type"] not in APPROVED_INFERENCE_PROVIDERS: + errors.append(f"{p['name']}: executor_type not approved") + ref = (p.get("credential") or {}).get("name") + if ref is not None and ref not in secrets: + errors.append(f"{p['name']}: credential {ref} has no secrets binding") + binding = secrets.get(ref or "", {}) + if ref and stage == "pre269" and binding.get("backend") != "env": + errors.append(f"{p['name']}: pre-269 inference credential must be env") + if p.get("allowed_models") == []: + errors.append(f"{p['name']}: allowed_models [] is rejected") + by_name = {p["name"]: p for p in providers} + dp = we.get("default_provider") + if providers and dp not in by_name: + errors.append("default_provider not in catalog") + elif dp: + allowed = by_name[dp].get("allowed_models") + if allowed is not None and we.get("default_model") not in allowed: + errors.append("default_model not allowed by default_provider") + for name, server in mcp.items(): + for ref in (server.get("secret_headers") or {}).values(): + if ref["name"] not in secrets: + errors.append(f"mcp {name}: ref {ref['name']} has no secrets binding") + if server.get("workflow_enabled"): + bad = REQUEST_BOUND & set((server.get("authorization_headers") or {}).values()) + if bad: + errors.append(f"mcp {name}: workflow_enabled with {sorted(bad)}") + if server.get("headers"): + errors.append(f"mcp {name}: workflow_enabled with propagated headers") + for ref in secrets: + if not NAME_RE.match(ref): + errors.append(f"secret name {ref} invalid") + for rule in (we.get("policy") or {}).get("rules", []): + for key, known in (("providers", by_name), ("mcp_servers", mcp), ("mcp_secrets", secrets)): + for item in rule.get(key, []): + if item not in known: + errors.append(f"policy {rule['roles']}: unknown {key[:-1]} {item}") + return errors + + +# -------------------------------------------------------------------- grants +def grants(stack: dict, roles: set[str]) -> dict[str, Any]: + """Union of every policy rule matching the roles (inherit resolved).""" + rules = (stack["workflow_engine"].get("policy") or {}).get("rules", []) + by_role = {r: rule for rule in rules for r in rule["roles"]} + out: dict[str, Any] = {} + + def add(rule: dict, depth: int = 0) -> None: + if depth > 8: + raise ValueError("policy inherit cycle") + for parent in rule.get("inherit", []): + add(by_role[parent], depth + 1) + for key, value in rule.items(): + if key in ("roles", "inherit"): + continue + if isinstance(value, list): + out[key] = sorted(set(out.get(key, [])) | set(value)) + elif isinstance(value, dict): + merged = dict(out.get(key, {})) + for k, v in value.items(): + merged[k] = max(merged.get(k, v), v) + out[key] = merged + else: + out[key] = out.get(key, False) or value + + for role in roles: + if role in by_role: + add(by_role[role]) + return out + + +def actions(stack: dict, roles: set[str]) -> set[str]: + """Coarse RBAC actions for the roles (admin implies all).""" + acts: set[str] = set() + for rule in stack.get("authorization", {}).get("access_rules", []): + if rule["role"] in roles: + acts |= set(rule["actions"]) + return acts | ({"*admin*"} if "admin" in acts else set()) + + +def _can(stack: dict, roles: set[str], action: str) -> bool: + acts = actions(stack, roles) + return "*admin*" in acts or action in acts + + +# ---------------------------------------------------------------- submission +def _steps(definition: dict) -> list[dict]: + return definition["spec"]["steps"] + + +def _agent_steps(definition: dict) -> list[dict]: + return [s for s in _steps(definition) if s.get("type", "agent") == "agent"] + + +def _resolve(stack: dict, name: str, model: Optional[str]) -> tuple[dict, str]: + we = stack["workflow_engine"] + catalog = {p["name"]: p for p in we["providers"]} + if name not in catalog: + raise Denied(400, f"unknown provider '{name}'") + entry = catalog[name] + model = model or we.get("default_model") + allowed = entry.get("allowed_models") + if allowed is not None and model not in allowed: + raise Denied(400, f"model '{model}' not allowed for '{name}'") + return entry, model + + +def _count_errors(definition: dict) -> list[str]: + errors = [] + spec = definition["spec"] + if len(spec["steps"]) > MAX_WORKFLOW_STEPS: + errors.append("too many steps") + for scope in [spec, *spec["steps"]]: + servers = scope.get("mcp_servers") or [] + if len(servers) > MAX_MCP_SERVERS_PER_STEP: + errors.append("too many mcp_servers") + for srv in servers: + if isinstance(srv, dict): + if len(srv.get("secret_headers") or {}) > MAX_SECRET_HEADERS_PER_SERVER: + errors.append("too many secret_headers") + return errors + + +def _shape_copy(definition: dict) -> dict: + """Copy for the cloud-agents shape check with inline registry refs rewritten. + + cloud-agents' MCPServerConfig only accepts ``{secret_name, key}``, so a + registry ref ``{name: ...}`` would fail the typed parse. The stack + validates registry refs itself (step 4), then rewrites them to the + executor form; this models that rewrite for the step-3 shape check. + """ + shaped = json.loads(json.dumps(definition)) + spec = shaped.get("spec", {}) + for scope in [spec, *spec.get("steps", [])]: + for srv in scope.get("mcp_servers") or []: + if not isinstance(srv, dict): + continue + for header, ref in list((srv.get("secret_headers") or {}).items()): + if isinstance(ref, dict) and set(ref) == {"name"}: + srv["secret_headers"][header] = {"secret_name": ref["name"], "key": "value"} + return shaped + + +def submit( + stack: dict, + body: dict, + roles: set[str], + stage: str = "pre269", + spawner: bool = True, +) -> list[str]: + """Run POST /v1/workflows/run steps 1-5; return [] when it would be 202. + + Raises Denied(status, reasons) otherwise. + """ + we = stack["workflow_engine"] + definition = body["definition"] + if not _can(stack, roles, "workflow_start"): # step 1 + raise Denied(403, "workflow_start not granted") + if len(json.dumps(definition).encode()) > MAX_DEFINITION_BYTES: # step 2 + raise Denied(413, "definition too large") + try: # step 3 + WorkflowDefinition.model_validate(_shape_copy(definition)) + except ValidationError as exc: + raise Denied(422, f"shape: {exc.error_count()} errors") from exc + if errs := _count_errors(definition): + raise Denied(422, *errs) + run = body.get("provider") or {} + def_provider = definition.get("provider") or {} + if "credentials_secret" in run or "credentials_secret" in def_provider: + raise Denied(400, "credentials_secret cannot be set by callers") + extra = set(run) - {"name", "model", "credential_ref"} + if extra: + raise Denied(400, f"unknown provider keys {sorted(extra)}") + + # step 4: catalog (400) + name = run.get("name") or we["default_provider"] + entry, model = _resolve(stack, name, run.get("model")) + cref = (run.get("credential_ref") or {}).get("name") + if cref is not None and cref != entry["credential"]["name"]: + raise Denied(400, "credential_ref does not match the provider's credential") + overrides = [def_provider] + [ + s.get("inference_provider") or {} for s in _agent_steps(definition) + ] + for ov in overrides: + if ov.get("name"): + ov_entry, _ = _resolve(stack, ov["name"], ov.get("model")) + if ov_entry["name"] != entry["name"]: + raise Denied(400, "override must use the same catalog entry as the run provider") + mcp = {m["name"]: m for m in stack["mcp_servers"]} + refs = {s["name"] for s in we["secrets"]} + inline: list[dict] = [] + for scope in [definition["spec"], *_agent_steps(definition)]: + for srv in scope.get("mcp_servers") or []: + if isinstance(srv, str): + if srv not in mcp or not mcp[srv].get("workflow_enabled"): + raise Denied(400, f"unknown MCP server '{srv}'") + else: + inline.append(srv) + for ref in (srv.get("secret_headers") or {}).values(): + if not isinstance(ref, dict) or ref.get("name") not in refs: + raise Denied(400, "inline secret_headers must be registry refs") + + # step 5: policy (403) + g = grants(stack, roles) + why: list[str] = [] + if entry["name"] not in g.get("providers", []): + why.append(f"provider '{entry['name']}' not granted") + spec = definition["spec"] + default_image = stack["spawner"]["sandbox_image"] + images = {body.get("sandbox_image"), (spec.get("spawn_config") or {}).get("sandbox_image"), + (definition.get("skills") or {}).get("image")} + for step in _agent_steps(definition): + images.add((step.get("spawn_config") or {}).get("sandbox_image")) + for img in images - {None}: + if img not in g.get("sandbox_images", []): + why.append(f"image '{img}' not granted") + if default_image not in g.get("sandbox_images", []): + why.append("spawner default image not granted") + if definition.get("advisory") and not g.get("advisory"): + why.append("advisory not granted") + for step in _agent_steps(definition): + s = step["name"] + spawn = step.get("spawn") or spec.get("spawn") or "ephemeral" + if spawn not in g.get("spawn", []): + why.append(f"{s}: spawn '{spawn}' not granted") + if spawn == "ephemeral" and not spawner: + why.append(f"{s}: ephemeral needs spawner_configuration") + servers = step["mcp_servers"] if step.get("mcp_servers") is not None else spec.get("mcp_servers") or [] + for srv in servers: + if isinstance(srv, str): + if srv not in g.get("mcp_servers", []): + why.append(f"{s}: MCP '{srv}' not granted") + secret_headers = mcp[srv].get("secret_headers") or {} + for ref in secret_headers.values(): + if ref["name"] not in g.get("mcp_secrets", []): + why.append(f"{s}: MCP secret '{ref['name']}' not granted for '{srv}'") + if secret_headers and spawn != "ephemeral" and stage == "pre269": + why.append(f"{s}: secret_headers need ephemeral before cloud-agents#269") + else: + if not g.get("inline_mcp"): + why.append(f"{s}: inline MCP not granted") + parts = urlsplit(srv.get("url", "")) + if parts.scheme != "https" or parts.username or parts.password: + why.append(f"{s}: inline MCP must be https without userinfo") + if srv.get("headers"): + why.append(f"{s}: inline MCP literal header values are not allowed") + if parts.hostname not in g.get("inline_mcp_hosts", []): + why.append(f"{s}: inline MCP host '{parts.hostname}' not allowed") + for ref in (srv.get("secret_headers") or {}).values(): + if ref["name"] not in g.get("mcp_secrets", []): + why.append(f"{s}: inline MCP secret '{ref['name']}' not granted") + skills = step["allowed_skills"] if step.get("allowed_skills") is not None else spec.get("allowed_skills") or [] + for sk in skills: + if sk not in g.get("skills", []): + why.append(f"{s}: skill '{sk}' not granted") + perms = step.get("permissions") or spec.get("permissions") or {} + for tool in [*step.get("tools", []), *(perms.get("allowed_tools") or [])]: + if tool not in g.get("tools", []): + why.append(f"{s}: tool '{tool}' not granted") + sa = step.get("service_account") or perms.get("service_account") or spec.get("service_account") + if sa and sa not in g.get("service_accounts", []): + why.append(f"{s}: service account '{sa}' not granted") + for ns in step.get("target_namespaces") or []: + if ns not in g.get("namespaces", []): + why.append(f"{s}: namespace '{ns}' not granted") + lim = g.get("limits", {}) + if (step.get("timeout_seconds") or 0) > lim.get("max_timeout_seconds", 0): + why.append(f"{s}: timeout over limit") + if (perms.get("max_tokens") or 0) > lim.get("max_tokens", 0): + why.append(f"{s}: max_tokens over limit") + if step.get("max_retries", 0) > lim.get("max_retries", 0): + why.append(f"{s}: max_retries over limit") + if (spec.get("timeout_seconds") or 0) > g.get("limits", {}).get("max_timeout_seconds", 0): + why.append("workflow timeout over limit") + if why: + raise Denied(403, *why) + return [] + diff --git a/examples/workflows/triage-github-issue.yaml b/examples/workflows/triage-github-issue.yaml new file mode 100644 index 000000000..0bb0afea5 --- /dev/null +++ b/examples/workflows/triage-github-issue.yaml @@ -0,0 +1,84 @@ +# ============================================================================= +# triage-github-issue.yaml — caller-authored workflow definition. +# +# Flow: triage (classify the issue via GitHub MCP) -> report (write the +# triage report from the classification, no tools needed). +# +# Logical names used here must exist in lightspeed-stack.yaml and be granted +# to the caller's roles by workflow_engine.policy: +# provider claude-prod / model claude-sonnet-4-5 (roles: agent-user) +# MCP server github-readonly (roles: agent-user) +# skill triage (roles: agent-user) +# spawn ephemeral, image = spawner default, service account workflow-runner +# +# Trigger (run-level provider + input travel in the API request, not here): +# POST /v1/workflows/run (see README.md for the full curl) +# ============================================================================= +apiVersion: cloudagents/v1 +kind: AgentWorkflow +metadata: + name: triage-github-issue + description: Classify a GitHub issue and write a triage report. + +spec: + input_prompt: "Triage GitHub issue https://github.com/example/repo/issues/1234" + + # Workflow-level defaults: every step inherits these unless it overrides. + spawn: ephemeral + service_account: workflow-runner + timeout_seconds: 900 + mcp_servers: + - github-readonly + allowed_skills: + - triage + + steps: + - name: triage + type: agent + agent: triage-agent + prompt: >- + Read the GitHub issue in the workflow input using the github tools. + Classify it with severity (critical/high/medium/low), area + (backend/frontend/docs/infra), and a one-paragraph summary. + Respond with JSON matching the step output schema. + output_key: classification + output_schema: + type: object + required: [severity, area, summary] + properties: + severity: {type: string, enum: [critical, high, medium, low]} + area: {type: string, enum: [backend, frontend, docs, infra]} + summary: {type: string} + tools: + - github.read + target_namespaces: + - sandbox-workloads + timeout_seconds: 600 + max_retries: 2 + permissions: + max_tokens: 50000 + # inference_provider omitted: inherits the run-level provider + # (claude-prod). Inherited steps are not re-checked; the run-level + # check covers them. + + - name: report + type: agent + agent: triage-agent + prompt: >- + Write a markdown triage report from this classification: + {{ steps.triage.output.classification }} + Keep it under 300 words. No tool calls needed. + output_key: report + output_schema: + type: object + required: [report_markdown] + properties: + report_markdown: {type: string} + # Explicitly empty: [] means "none", while omitting the field would + # inherit the workflow defaults above. The report step needs neither + # MCP access nor skills. + mcp_servers: [] + allowed_skills: [] + timeout_seconds: 300 + permissions: + max_tokens: 20000 diff --git a/examples/workflows/verify.py b/examples/workflows/verify.py new file mode 100644 index 000000000..bb2ba4cb8 --- /dev/null +++ b/examples/workflows/verify.py @@ -0,0 +1,330 @@ +"""Verify the examples/workflows contract (stack#51 + cloud-agents#269). + +Run from anywhere with the project venv: + + uv run python examples/workflows/verify.py + +1. Every file parses; the three workflow definitions pass the real + cloud-agents ``WorkflowDefinition`` shape check. +2. ``lightspeed-stack.yaml`` passes the plan's load-time checks, before and + after cloud-agents#269 (``post-269-overrides.yaml`` merged on top). +3. ``cases.yaml``: every request/config case returns the expected status in + each stage, via ``reference_gate.py`` (later: the real endpoint). +4. The non-PLAN subset validates against the CURRENT ``Configuration``. +5. Auth: real role and access resolvers behave as the policy assumes. +6. Deployment: Deployment, RBAC and Secrets agree with the stack config. +""" + +import asyncio +import base64 +import copy +import json +import os +import sys +import tempfile +from pathlib import Path + +import yaml + +WF_DIR = Path(__file__).resolve().parent +sys.path.insert(0, str(WF_DIR.parent.parent / "src")) +sys.path.insert(0, str(WF_DIR)) + +import reference_gate as gate # noqa: E402 + +failures: list[str] = [] + + +def check(cond: bool, msg: str) -> None: + """Record and print one assertion.""" + print(("PASS " if cond else "FAIL ") + msg) + if not cond: + failures.append(msg) + + +def load(name: str): + """Load one YAML document.""" + with open(WF_DIR / name, encoding="utf-8") as fh: + return yaml.safe_load(fh) + + +def load_all(name: str) -> list[dict]: + """Load a multi-document YAML file.""" + with open(WF_DIR / name, encoding="utf-8") as fh: + return [d for d in yaml.safe_load_all(fh) if d] + + +def deep_merge(base, over): + """Maps merge, everything else (lists, scalars) is replaced.""" + if isinstance(base, dict) and isinstance(over, dict): + out = dict(base) + for key, value in over.items(): + out[key] = deep_merge(base[key], value) if key in base else value + return out + return copy.deepcopy(over) + + +def _key(container, token): + """Resolve a path token: dict key, list index, or list item by name.""" + if isinstance(container, list): + if token.isdigit(): + return int(token) + for i, item in enumerate(container): + if isinstance(item, dict) and item.get("name") == token: + return i + raise KeyError(token) + return token + + +def patch(doc, patches): + """Return a copy of ``doc`` with dotted-path ``patches`` applied.""" + doc = copy.deepcopy(doc) + for path, value in (patches or {}).items(): + tokens = path.split(".") + cur = doc + for tok in tokens[:-1]: + k = _key(cur, tok) + if isinstance(cur, dict) and k not in cur: + cur[k] = {} + cur = cur[k] + cur[_key(cur, tokens[-1])] = value + return doc + + +def generate(definition, spec): + """Build oversized inputs for the limit cases.""" + steps = definition["spec"]["steps"] + if "pad_bytes" in spec: + definition["metadata"]["padding"] = "x" * spec["pad_bytes"] + while len(steps) < spec.get("steps", 0): + clone = copy.deepcopy(steps[0]) + clone["name"] = f"gen{len(steps)}" + steps.append(clone) + if "mcp_servers" in spec: + definition["spec"]["mcp_servers"] = [f"srv{i}" for i in range(spec["mcp_servers"])] + if "secret_headers" in spec: + server = steps[0]["mcp_servers"][0] + server["secret_headers"] = { + f"H{i}": {"name": "mcp/incidents"} for i in range(spec["secret_headers"]) + } + return definition + + +# ---------------------------------------------------------------- 1. parsing +stack = load("lightspeed-stack.yaml") +overrides = load("post-269-overrides.yaml") +stacks = {"pre269": stack, "post269": deep_merge(stack, overrides)} +cases = load("cases.yaml") +definitions = { + name: load(f"{name}.yaml") + for name in ("triage-github-issue", "kb-answer-with-approval", "admin-inline-mcp") +} +for doc in ("k8s-secrets.yaml", "external-secrets.yaml", "k8s-deployment.yaml"): + load_all(doc) +check(True, "all YAML files parse") + +from cloud_agents.workflow.core.definition import WorkflowDefinition # noqa: E402 + +for name, raw in definitions.items(): + WorkflowDefinition.model_validate(gate._shape_copy(raw)) # pylint: disable=W0212 + check(True, f"{name}: WorkflowDefinition shape valid") + +# ------------------------------------------------------------ 2. load checks +for stage, cfg in stacks.items(): + errors = gate.load_errors(cfg, stage) + check(errors == [], f"{stage}: config passes load-time checks {errors}") + + +# ------------------------------------------------------------------ 3. cases +def run(fn, *args, **kwargs): + """Return (status, reasons) the reference gate gives.""" + try: + fn(*args, **kwargs) + return 202, [] + except gate.Denied as exc: + return exc.status, exc.reasons + + +def assert_case(label, stage, got, reasons, case): + """Check status and (optionally) a reason substring.""" + exp = case["expect"] + want = exp[stage] if isinstance(exp, dict) else exp + ok = got == want + if ok and case.get("reason") and want != 202: + ok = any(case["reason"].lower() in r.lower() for r in reasons) + detail = "" if ok else f" (got {got} {reasons[:2]})" + check(ok, f"[{stage}] {label} -> {want}{detail}") + + +for case in cases["workflow_cases"]: + base = definitions[case["workflow"]] + dflt = cases["defaults"][case["workflow"]] + for stage, cfg in stacks.items(): + definition = generate(patch(base, case.get("set")), case.get("generate", {})) + body = patch( + {"definition": definition, "provider": copy.deepcopy(dflt["provider"])}, + case.get("body_set"), + ) + status, reasons = run( + gate.submit, + cfg, + body, + set(case.get("as", dflt["as"])), + stage, + case.get("spawner", True), + ) + assert_case(case["name"], stage, status, reasons, case) + +for case in cases["config_cases"]: + for stage, cfg in stacks.items(): + if case.get("stage", stage) != stage: + continue + errors = gate.load_errors(patch(cfg, case["set"]), stage) + ok = any(case["error"].lower() in e.lower() for e in errors) + extra = "" if ok else f" (got {errors})" + check(ok, f"[{stage}] config: {case['name']} -> load fails{extra}") + +# ------------------------------------------------------ 4. current schema +stripped = copy.deepcopy(stack) +for key in ("providers", "secrets", "default_provider", "default_model", "policy"): + stripped["workflow_engine"].pop(key, None) +for server in stripped["mcp_servers"]: + server.pop("secret_headers", None) + server.pop("workflow_enabled", None) +os.environ.setdefault("POSTGRES_PASSWORD", "test-only") +# FilePath fields point at mounted Secrets in-cluster; stand in dummy files. +for section, keys in ( + (stripped["spawner"], ("openshell_tls_ca", "openshell_tls_cert", "openshell_tls_key")), + (stripped["database"]["postgres"], ("ca_cert_path",)), +): + for key in keys: + fd, path = tempfile.mkstemp() + os.close(fd) + section[key] = path + +from models.config import AccessRule, Action, Configuration, JwtRoleRule # noqa: E402 + +Configuration.model_validate(stripped) +check(True, "non-PLAN subset validates against CURRENT Configuration schema") + +# --------------------------------------------------------------- 5. auth +from authorization.resolvers import GenericAccessResolver, JwtRolesResolver # noqa: E402 + +jwt_cfg = stack["authentication"]["jwk_config"]["jwt_configuration"] +roles_resolver = JwtRolesResolver([JwtRoleRule(**r) for r in jwt_cfg["role_rules"]]) +access = GenericAccessResolver( + [AccessRule(**r) for r in stack["authorization"]["access_rules"]] +) + + +def b64(data): + """Base64url-encode a JSON object (unsigned test token part).""" + return base64.urlsafe_b64encode(json.dumps(data).encode()).rstrip(b"=").decode() + + +def roles_for(token_roles): + """Resolve roles for an unsigned token carrying the given realm roles.""" + token = ".".join([b64({"alg": "none"}), b64({"realm_access": {"roles": token_roles}}), "sig"]) + return asyncio.run(roles_resolver.resolve_roles(("uid", "user", False, token))) + + +user = roles_for(["lightspeed-agent-user"]) +support = roles_for(["lightspeed-agent-support"]) +admin = roles_for(["lightspeed-agent-admin"]) +check( + "agent-user" in user and "agent-support" in support and "agent-admin" in admin, + "token roles map to agent-user/support/admin", +) +check(access.check_access(Action.WORKFLOW_START, user), "agent-user may start") +check(not access.check_access(Action.WORKFLOW_APPROVE, user), "agent-user may not approve") +check(access.check_access(Action.WORKFLOW_APPROVE, support), "agent-support may approve") +check( + not access.check_access(Action.ADMIN, support) and not access.check_access(Action.ADMIN, user), + "only agent-admin is admin", +) +check(access.check_access(Action.ADMIN, admin), "agent-admin is admin") + +# ---------------------------------------------------------- 6. deployment +by_kind: dict[str, list[dict]] = {} +for d in load_all("k8s-deployment.yaml"): + by_kind.setdefault(d["kind"], []).append(d) +pod = by_kind["Deployment"][0]["spec"]["template"]["spec"] +ctr = pod["containers"][0] +env = {e["name"]: e for e in ctr["env"]} +pre = stacks["pre269"]["workflow_engine"]["secrets"] +post = stacks["post269"]["workflow_engine"]["secrets"] + +env_refs = {} +for binding in (b for b in pre if b["backend"] == "env"): + ref = env.get(binding["env"], {}).get("valueFrom", {}).get("secretKeyRef") + check(ref is not None, f"Deployment injects {binding['env']} from a Secret ({binding['name']})") + if ref: + env_refs[binding["env"]] = (ref["name"], ref["key"]) +generated = gate.generated_mcp_allowed_secrets(stacks["pre269"]) +check(env["MCP_ALLOWED_SECRETS"]["value"] == generated, "MCP_ALLOWED_SECRETS equals the generated value") +check( + generated == gate.generated_mcp_allowed_secrets(stacks["post269"]), + "MCP allow-list is the same in both stages", +) + +roles_by_name = {r["metadata"]["name"]: r for r in by_kind["Role"]} + + +def granted(role_name): + """resourceNames the Role grants get on.""" + return set(roles_by_name[role_name]["rules"][0]["resourceNames"]) + + +def k8s_names(secrets): + """K8s Secret names bound with backend: k8s.""" + return {s["secret_name"] for s in secrets if s["backend"] == "k8s"} + + +check( + granted("lightspeed-stack-secret-reader") == k8s_names(pre), + "pre-269 Role grants get on exactly the k8s-bound Secrets", +) +check( + granted("lightspeed-stack-secret-reader-post-269") == k8s_names(post), + "post-269 Role grants get on exactly the k8s-bound Secrets", +) +for role in roles_by_name.values(): + check(role["rules"][0]["verbs"] == ["get"], f"{role['metadata']['name']}: get only, by resourceName") + +check(pod["securityContext"]["runAsNonRoot"] is True, "pod runs as non-root") +sc = ctr["securityContext"] +check( + sc["readOnlyRootFilesystem"] and not sc["allowPrivilegeEscalation"] and sc["capabilities"]["drop"] == ["ALL"], + "container hardened", +) +mounts = {m["name"]: m["mountPath"] for m in ctr["volumeMounts"]} +spawner = stack["spawner"] +check(spawner["openshell_tls_ca"].startswith(mounts["gw-tls"]), "gateway TLS paths match the mount") +check(stack["database"]["postgres"]["ca_cert_path"].startswith(mounts["pg-tls"]), "Postgres CA path matches the mount") +check(stack["database"]["postgres"]["ssl_mode"] == "verify-full", "Postgres uses verify-full TLS") +check(all("@sha256:" in i for i in (ctr["image"], spawner["sandbox_image"])), "stack and sandbox images pinned by digest") +images = {i for r in stack["workflow_engine"]["policy"]["rules"] for i in r.get("sandbox_images", [])} +check(all("@sha256:" in i for i in images), "policy sandbox_images pinned by digest") +check(spawner["sandbox_image"] in images, "spawner default image is granted by policy") + +# Secrets: dev manifest and External Secrets define the same names and keys, +# and cover every Secret/key the stack references in either stage. +dev_docs = load_all("k8s-secrets.yaml") +dev = {d["metadata"]["name"]: set((d.get("stringData") or d.get("data")).keys()) for d in dev_docs} +ext = { + d["spec"]["target"]["name"]: {x["secretKey"] for x in d["spec"]["data"]} + for d in load_all("external-secrets.yaml") +} +check(dev == ext, f"k8s-secrets.yaml and external-secrets.yaml match {sorted(set(dev) ^ set(ext))}") +needed = {(s["secret_name"], s["key"]) for s in post if s["backend"] == "k8s"} +needed |= set(env_refs.values()) +check(all(k in ext.get(n, set()) for n, k in needed), "every referenced Secret/key exists in the secret source") +for d in dev_docs: + values = (d.get("stringData") or d.get("data")).values() + check(all("CHANGEME" in str(v) for v in values), f"{d['metadata']['name']}: placeholders only") + +print() +if failures: + print(f"{len(failures)} FAILED") + sys.exit(1) +print("ALL CHECKS PASSED") diff --git a/src/app/endpoints/workflows.py b/src/app/endpoints/workflows.py index e8dd27c4d..41efec158 100644 --- a/src/app/endpoints/workflows.py +++ b/src/app/endpoints/workflows.py @@ -15,7 +15,7 @@ from authentication import get_auth_dependency from authentication.interface import AuthTuple -from authorization.middleware import authorize +from authorization.middleware import authorize, is_admin from configuration import configuration from log import get_logger from models.api.requests.agents import ApproveWorkflowRequest, RunWorkflowRequest @@ -23,6 +23,10 @@ from utils.endpoints import check_configuration_loaded from workflow.provider_credentials import credentials_secret_for from workflow.spawner_factory import build_spawner +from workflow.submission_guard import ( + enforce_submission_hardening, + reject_oversized_definition, +) logger = get_logger(__name__) router = APIRouter(tags=["workflows"]) @@ -128,17 +132,31 @@ async def start_workflow_handler( Returns: Workflow ID and initial status. """ - _ = request user_id, username, _, _ = auth check_configuration_loaded(configuration) executor = _get_executor() + spawner_config = configuration.spawner_configuration + default_sandbox_image = ( + spawner_config.sandbox_image # pylint: disable=no-member + if spawner_config + else "lightspeed-agentic-sandbox:latest" + ) + + # Pipeline order (issue #51): byte cap (413), shape (422), then the + # stack's own rules -- forbidden caller fields (400), count caps (422) + # and privileged options (403) -- all before anything is persisted. + reject_oversized_definition(body.definition) inference = configuration.inference - provider = body.provider or { - "name": inference.default_provider or "", - "model": inference.default_model or "", - } + caller_provider = dict( + body.provider + or { + "name": inference.default_provider or "", + "model": inference.default_model or "", + } + ) + provider = dict(caller_provider) if "credentials_secret" not in provider: cred_secret = credentials_secret_for(provider.get("name") or "") if cred_secret: @@ -149,13 +167,17 @@ async def start_workflow_handler( # runs, so secret-bearing, malformed, or unapproved-provider input # must be rejected here -- otherwise it returns 202 and lands in # workflow state first. + # `provider` carries the stack's injected credentials reference for the + # shape check; `caller_provider` keeps only the caller's keys so the + # credential-rejection check sees what the caller actually sent. _validate_workflow_submission(body.definition, provider) - - spawner_config = configuration.spawner_configuration - default_sandbox_image = ( - spawner_config.sandbox_image # pylint: disable=no-member - if spawner_config - else "lightspeed-agentic-sandbox:latest" + enforce_submission_hardening( + body.definition, + caller_provider, + body.sandbox_image, + is_admin=is_admin(request), + default_sandbox_image=default_sandbox_image, + spawner_configured=spawner_config is not None, ) workflow_input = { diff --git a/src/authorization/middleware.py b/src/authorization/middleware.py index 93cb5e7e0..c432fa685 100644 --- a/src/authorization/middleware.py +++ b/src/authorization/middleware.py @@ -157,6 +157,30 @@ async def _perform_authorization_check( break if req is not None: req.state.authorized_actions = authorized_actions + req.state.user_roles = user_roles + + +def is_admin(request: Request) -> bool: + """Return whether the request's caller holds the ADMIN action. + + `authorized_actions` never contains ADMIN (an ADMIN grant expands to every + other action), so this asks the access resolver using the roles stored on + `request.state` by the authorization check. Only valid inside an + `@authorize`-decorated handler. + + Parameters: + request: Request that has passed through an `@authorize` endpoint. + + Returns: + True when any of the caller's roles grants ADMIN; False otherwise, + including when no roles were resolved (fail closed). + """ + user_roles = getattr(request.state, "user_roles", None) + if not user_roles: + logger.debug("is_admin called without resolved user_roles") + return False + _, access_resolver = get_authorization_resolvers() + return access_resolver.check_access(Action.ADMIN, user_roles) def authorize(action: Action) -> Callable: diff --git a/src/models/api/requests/agents.py b/src/models/api/requests/agents.py index cfda6a242..c38f22f73 100644 --- a/src/models/api/requests/agents.py +++ b/src/models/api/requests/agents.py @@ -37,7 +37,7 @@ class RunWorkflowRequest(BaseModel): provider: Optional[dict[str, Any]] = Field( None, - description="Default provider config: {name, model, credentials_secret}.", + description="Default provider config: {name, model}. The stack chooses the credential; credentials_secret is rejected.", ) sandbox_image: Optional[str] = Field( diff --git a/src/workflow/limits.py b/src/workflow/limits.py new file mode 100644 index 000000000..b95ff43bd --- /dev/null +++ b/src/workflow/limits.py @@ -0,0 +1,11 @@ +"""Size and count limits for workflow submissions.""" + +from typing import Final + +# Workflow definition limits. Sized against the multi-step transcripts from +# issues #52/#53 (largest: a handful of steps, one or two MCP servers each, +# a few KiB of definition); these leave two orders of magnitude of headroom. +MAX_DEFINITION_BYTES: Final[int] = 262_144 +MAX_WORKFLOW_STEPS: Final[int] = 64 +MAX_MCP_SERVERS_PER_STEP: Final[int] = 16 +MAX_SECRET_HEADERS_PER_SERVER: Final[int] = 32 diff --git a/src/workflow/submission_guard.py b/src/workflow/submission_guard.py new file mode 100644 index 000000000..ca232beac --- /dev/null +++ b/src/workflow/submission_guard.py @@ -0,0 +1,176 @@ +"""Submission-time hardening for POST /v1/workflows/run (issue #51, Phase 0a). + +Closes the holes where a caller could pick the credential env var, the +sandbox image, advisory mode or a non-sandboxed spawn mode, and bounds the +size of a definition. Pure checks: nothing here mutates its inputs. +""" + +import json +from typing import Any, Optional + +from fastapi import HTTPException, status + +from log import get_logger +from workflow.limits import ( + MAX_DEFINITION_BYTES, + MAX_MCP_SERVERS_PER_STEP, + MAX_SECRET_HEADERS_PER_SERVER, + MAX_WORKFLOW_STEPS, +) + +logger = get_logger(__name__) + +_PRIVILEGED_SPAWN_MODES = frozenset({"none", "local"}) +_DEFAULT_SPAWN = "ephemeral" + + +def reject_oversized_definition(definition: dict[str, Any]) -> None: + """Reject a definition whose JSON encoding exceeds the byte cap. + + Parameters: + definition: Raw workflow definition from the request body. + + Raises: + HTTPException: 413 when the definition is over MAX_DEFINITION_BYTES. + """ + size = len(json.dumps(definition).encode("utf-8")) + if size > MAX_DEFINITION_BYTES: + raise HTTPException( + status_code=status.HTTP_413_CONTENT_TOO_LARGE, + detail=f"Workflow definition exceeds {MAX_DEFINITION_BYTES} bytes.", + ) + + +def _as_dict(value: Any) -> dict[str, Any]: + """Return value when it is a dict, else an empty dict.""" + return value if isinstance(value, dict) else {} + + +def _image_of(spawn_config: Any) -> Optional[str]: + """Return the sandbox_image of a raw spawn_config, if any.""" + image = _as_dict(spawn_config).get("sandbox_image") + return image if isinstance(image, str) else None + + +def _count_errors(steps: list[Any], spec: dict[str, Any]) -> list[str]: + """Return count-cap violations for steps, MCP lists and secret headers.""" + errors: list[str] = [] + if len(steps) > MAX_WORKFLOW_STEPS: + errors.append(f"spec.steps has {len(steps)} steps (max {MAX_WORKFLOW_STEPS})") + scopes = [("spec", spec)] + [ + (f"step '{_as_dict(s).get('name', i)}'", _as_dict(s)) + for i, s in enumerate(steps) + ] + for label, scope in scopes: + servers = scope.get("mcp_servers") or [] + if not isinstance(servers, list): + continue + if len(servers) > MAX_MCP_SERVERS_PER_STEP: + errors.append( + f"{label}: {len(servers)} mcp_servers (max {MAX_MCP_SERVERS_PER_STEP})" + ) + for server in servers: + headers = _as_dict(_as_dict(server).get("secret_headers")) + if len(headers) > MAX_SECRET_HEADERS_PER_SERVER: + errors.append( + f"{label}: mcp server has {len(headers)} secret_headers " + f"(max {MAX_SECRET_HEADERS_PER_SERVER})" + ) + return errors + + +def _effective_spawn_modes(steps: list[Any], spec: dict[str, Any]) -> set[str]: + """Return the effective spawn mode of each agent step (step, spec, default). + + Approval steps never spawn anything, so they are skipped. + """ + workflow_spawn = spec.get("spawn") + return { + _as_dict(step).get("spawn") or workflow_spawn or _DEFAULT_SPAWN + for step in steps + if _as_dict(step).get("type", "agent") == "agent" + } + + +def enforce_submission_hardening( # pylint: disable=too-many-arguments + definition: dict[str, Any], + provider: dict[str, Any], + sandbox_image: Optional[str], + *, + is_admin: bool, + default_sandbox_image: str, + spawner_configured: bool, +) -> None: + """Apply the Phase 0a rules to a submission. + + Parameters: + definition: Raw workflow definition from the request body. + provider: Run-level provider as sent by the caller (before the + stack injects its own credentials_secret). + sandbox_image: Request-level sandbox image override, if any. + is_admin: Whether the caller holds the ADMIN action. + default_sandbox_image: The spawner's configured sandbox image. + spawner_configured: Whether a spawner_configuration exists. + + Shape errors (422 from cloud-agents validation) take precedence over + these policy errors because the handler validates first. Denials raise + HTTPException directly and are logged; structured audit events land in + Phase 4 of the #51 plan. + + Raises: + HTTPException: 400 when the caller names a credential, 422 when a + count cap is exceeded, 403 for privileged options used without + ADMIN or for ephemeral steps without a spawner. + """ + if "credentials_secret" in provider or "credentials_secret" in _as_dict( + definition.get("provider") + ): + raise HTTPException( + status_code=status.HTTP_400_BAD_REQUEST, + detail="credentials_secret cannot be set by callers; the stack " + "chooses the credential for the provider.", + ) + + spec = _as_dict(definition.get("spec")) + raw_steps = spec.get("steps") + steps: list[Any] = raw_steps if isinstance(raw_steps, list) else [] + + count_errors = _count_errors(steps, spec) + if count_errors: + raise HTTPException( + status_code=status.HTTP_422_UNPROCESSABLE_CONTENT, + detail={"validation_errors": count_errors}, + ) + + spawn_modes = _effective_spawn_modes(steps, spec) + if "ephemeral" in spawn_modes and not spawner_configured: + raise HTTPException( + status_code=status.HTTP_403_FORBIDDEN, + detail="Ephemeral steps require spawner_configuration.", + ) + + if is_admin: + return + + denied: list[str] = [] + privileged = sorted(spawn_modes & _PRIVILEGED_SPAWN_MODES) + if privileged: + denied.append(f"spawn mode(s) {privileged} require admin") + if definition.get("advisory"): + denied.append("advisory mode requires admin") + + images = { + sandbox_image, + _image_of(spec.get("spawn_config")), + _as_dict(definition.get("skills")).get("image"), + *(_image_of(_as_dict(s).get("spawn_config")) for s in steps), + } + if any(i and i != default_sandbox_image for i in images): + denied.append("a custom sandbox image requires admin") + + if denied: + logger.info("Workflow submission denied: %s", denied) + raise HTTPException( + status_code=status.HTTP_403_FORBIDDEN, + detail={"denied": denied}, + ) diff --git a/tests/unit/authorization/test_middleware.py b/tests/unit/authorization/test_middleware.py index a98c3526b..2a883426a 100644 --- a/tests/unit/authorization/test_middleware.py +++ b/tests/unit/authorization/test_middleware.py @@ -13,6 +13,7 @@ _perform_authorization_check, authorize, get_authorization_resolvers, + is_admin, ) from authorization.resolvers import ( AccessResolver, @@ -325,6 +326,7 @@ async def test_request_state_handling( if request_location != "none": assert mock_request.state.authorized_actions == {Action.QUERY} + assert mock_request.state.user_roles == {"employee", "*"} @pytest.mark.asyncio async def test_everyone_role_added( @@ -405,3 +407,35 @@ async def mock_endpoint(**_: Any) -> str: await mock_endpoint(auth=dummy_auth_tuple) assert exc_info.value.status_code == status.HTTP_403_FORBIDDEN + + +class TestIsAdmin: + """is_admin asks the access resolver, since get_actions never returns ADMIN.""" + + @staticmethod + def _request(mocker: MockerFixture, roles: Any) -> Any: + """Build a request whose state carries the resolved roles.""" + request = mocker.MagicMock(spec=Request) + request.state = mocker.MagicMock() + request.state.user_roles = roles + return request + + def test_admin_role_is_admin(self, mocker: MockerFixture) -> None: + """A role granted ADMIN is admin, though ADMIN is not in get_actions.""" + resolver = GenericAccessResolver( + [ + AccessRule(role="root", actions=[Action.ADMIN]), + AccessRule(role="user", actions=[Action.QUERY]), + ] + ) + mocker.patch( + "authorization.middleware.get_authorization_resolvers", + return_value=(mocker.MagicMock(), resolver), + ) + assert Action.ADMIN not in resolver.get_actions({"root"}) + assert is_admin(self._request(mocker, {"root", "*"})) is True + assert is_admin(self._request(mocker, {"user", "*"})) is False + + def test_missing_roles_is_not_admin(self, mocker: MockerFixture) -> None: + """No resolved roles on the request fails closed.""" + assert is_admin(self._request(mocker, None)) is False diff --git a/tests/unit/cloud_agents/test_submission_guard.py b/tests/unit/cloud_agents/test_submission_guard.py new file mode 100644 index 000000000..47d83863b --- /dev/null +++ b/tests/unit/cloud_agents/test_submission_guard.py @@ -0,0 +1,252 @@ +"""Unit tests for the workflow submission hardening guard (issue #51, Phase 0a).""" + +# pylint: disable=too-few-public-methods + +from __future__ import annotations + +import copy +from typing import Any, Optional + +import pytest +from fastapi import HTTPException + +from workflow.limits import ( + MAX_DEFINITION_BYTES, + MAX_MCP_SERVERS_PER_STEP, + MAX_SECRET_HEADERS_PER_SERVER, + MAX_WORKFLOW_STEPS, +) +from workflow.submission_guard import ( + enforce_submission_hardening, + reject_oversized_definition, +) + +DEFAULT_IMAGE = "spawner-default:v1" + + +def _step(name: str = "agent", **extra: Any) -> dict[str, Any]: + """Build a minimal agent step.""" + return { + "name": name, + "type": "agent", + "prompt": "Do it", + "output_key": "result", + **extra, + } + + +def _definition(steps: Optional[list[dict[str, Any]]] = None, **top: Any) -> dict: + """Build a minimal definition; ``top`` keys land at top level.""" + spec_extra = top.pop("spec", {}) + return { + "apiVersion": "v1", + "kind": "AgentWorkflow", + "metadata": {"name": "t"}, + "spec": {"steps": steps or [_step(spawn="ephemeral")], **spec_extra}, + **top, + } + + +def _check( + definition: dict[str, Any], + *, + provider: Optional[dict[str, Any]] = None, + sandbox_image: Optional[str] = None, + is_admin: bool = False, + spawner_configured: bool = True, +) -> None: + """Call the guard with test defaults.""" + enforce_submission_hardening( + definition, + provider or {"name": "openai", "model": "gpt-4o"}, + sandbox_image, + is_admin=is_admin, + default_sandbox_image=DEFAULT_IMAGE, + spawner_configured=spawner_configured, + ) + + +def _status(definition: dict[str, Any], **kwargs: Any) -> Optional[int]: + """Return the rejection status code, or None when accepted.""" + try: + _check(definition, **kwargs) + except HTTPException as exc: + return exc.status_code + return None + + +class TestCredentialRoutes: + """G1/G2: callers cannot choose the credential env var.""" + + @pytest.mark.parametrize("value", ["DATABASE_URL", None, ""]) + def test_run_provider_credentials_secret_rejected(self, value: Any) -> None: + """Run-level credentials_secret is 400 whatever its value, even for admins.""" + provider = {"name": "openai", "model": "m", "credentials_secret": value} + assert _status(_definition(), provider=provider) == 400 + assert _status(_definition(), provider=provider, is_admin=True) == 400 + + def test_definition_provider_credentials_secret_rejected(self) -> None: + """definition.provider.credentials_secret is 400.""" + definition = _definition( + provider={"name": "bedrock", "model": "m", "credentials_secret": "X_KEY"} + ) + assert _status(definition) == 400 + assert _status(definition, is_admin=True) == 400 + + def test_definition_provider_without_credentials_accepted(self) -> None: + """A definition provider with no credentials_secret passes.""" + definition = _definition(provider={"name": "openai", "model": "m"}) + assert _status(definition) is None + + +class TestSandboxImage: + """G10: callers cannot choose the sandbox image.""" + + def test_default_image_accepted(self) -> None: + """Naming the spawner default image is fine.""" + assert _status(_definition(), sandbox_image=DEFAULT_IMAGE) is None + + @pytest.mark.parametrize( + "mutate", + [ + lambda d: d["spec"].update(spawn_config={"sandbox_image": "evil:1"}), + lambda d: d["spec"]["steps"][0].update( + spawn_config={"sandbox_image": "evil:1"} + ), + lambda d: d.update(skills={"image": "evil:1"}), + ], + ids=["workflow-spawn-config", "step-spawn-config", "skills-image"], + ) + def test_custom_image_in_definition_forbidden(self, mutate: Any) -> None: + """Non-default images in the definition are 403 for non-admins.""" + definition = _definition() + mutate(definition) + assert _status(definition) == 403 + assert _status(definition, is_admin=True) is None + + def test_request_sandbox_image_forbidden(self) -> None: + """A non-default request-level sandbox_image is 403 without ADMIN.""" + assert _status(_definition(), sandbox_image="evil:1") == 403 + assert _status(_definition(), sandbox_image="evil:1", is_admin=True) is None + + +class TestAdvisory: + """G11: advisory mode needs ADMIN.""" + + def test_advisory_forbidden_without_admin(self) -> None: + """advisory: true is 403 for non-admins, allowed for admins.""" + definition = _definition(advisory=True) + assert _status(definition) == 403 + assert _status(definition, is_admin=True) is None + + +class TestSpawnMode: + """Workflow spawn none/local needs ADMIN; ephemeral needs a spawner.""" + + @pytest.mark.parametrize("mode", ["none", "local"]) + def test_step_spawn_forbidden_without_admin(self, mode: str) -> None: + """Step-level none/local is 403 for non-admins.""" + definition = _definition([_step(spawn=mode)]) + assert _status(definition) == 403 + assert _status(definition, is_admin=True) is None + + def test_workflow_spawn_forbidden_without_admin(self) -> None: + """Workflow-level none applies to steps that do not override it.""" + definition = _definition([_step()], spec={"spawn": "none"}) + assert _status(definition) == 403 + + def test_step_override_of_workflow_spawn_uses_effective_mode(self) -> None: + """A step overriding workflow spawn:none with ephemeral passes.""" + definition = _definition([_step(spawn="ephemeral")], spec={"spawn": "none"}) + assert _status(definition) is None + + def test_default_spawn_is_ephemeral_and_needs_spawner(self) -> None: + """No spawn anywhere defaults to ephemeral, 403 without a spawner.""" + definition = _definition([_step()]) + assert _status(definition, spawner_configured=False) == 403 + assert _status(definition, spawner_configured=True) is None + + def test_ephemeral_without_spawner_forbidden_even_for_admin(self) -> None: + """Admins cannot spawn ephemeral steps without a spawner.""" + definition = _definition([_step(spawn="ephemeral")]) + assert _status(definition, is_admin=True, spawner_configured=False) == 403 + + def test_none_without_spawner_fine_for_admin(self) -> None: + """spawn none never needs a spawner.""" + definition = _definition([_step(spawn="none")]) + assert _status(definition, is_admin=True, spawner_configured=False) is None + + +class TestLimits: + """G9: byte and count caps.""" + + def test_oversized_definition_413(self) -> None: + """A definition over MAX_DEFINITION_BYTES is 413.""" + big = _definition([_step(prompt="x" * (MAX_DEFINITION_BYTES + 1))]) + with pytest.raises(HTTPException) as exc_info: + reject_oversized_definition(big) + assert exc_info.value.status_code == 413 + + def test_small_definition_passes_size_check(self) -> None: + """A normal definition passes the size check.""" + reject_oversized_definition(_definition()) + + def test_too_many_steps_422(self) -> None: + """More than MAX_WORKFLOW_STEPS steps is 422.""" + steps = [ + _step(f"s{i}", spawn="ephemeral") for i in range(MAX_WORKFLOW_STEPS + 1) + ] + assert _status(_definition(steps)) == 422 + + def test_max_steps_accepted(self) -> None: + """Exactly MAX_WORKFLOW_STEPS steps passes.""" + steps = [_step(f"s{i}", spawn="ephemeral") for i in range(MAX_WORKFLOW_STEPS)] + assert _status(_definition(steps)) is None + + def test_too_many_mcp_servers_on_step_422(self) -> None: + """More than MAX_MCP_SERVERS_PER_STEP servers on a step is 422.""" + names = [f"srv{i}" for i in range(MAX_MCP_SERVERS_PER_STEP + 1)] + definition = _definition([_step(spawn="ephemeral", mcp_servers=names)]) + assert _status(definition) == 422 + + def test_too_many_workflow_level_mcp_servers_422(self) -> None: + """Workflow-level defaults are counted too.""" + names = [f"srv{i}" for i in range(MAX_MCP_SERVERS_PER_STEP + 1)] + definition = _definition(spec={"mcp_servers": names}) + assert _status(definition) == 422 + + def test_too_many_secret_headers_422(self) -> None: + """An inline MCP entry with too many secret_headers is 422.""" + headers = { + f"h{i}": {"secret_name": "s", "key": "k"} + for i in range(MAX_SECRET_HEADERS_PER_SERVER + 1) + } + server = { + "name": "x", + "url": "https://m.example.com", + "secret_headers": headers, + } + definition = _definition([_step(spawn="ephemeral", mcp_servers=[server])]) + assert _status(definition) == 422 + + +def test_guard_does_not_mutate_inputs() -> None: + """The guard is a pure check.""" + definition = _definition(advisory=False) + before = copy.deepcopy(definition) + _check(definition) + assert definition == before + + +def test_approval_step_does_not_need_spawner() -> None: + """A human-approval step has no spawn mode, so it never needs a spawner.""" + definition = _definition( + [_step(spawn="none"), {"name": "gate", "type": "human-approval"}] + ) + assert _status(definition, is_admin=True, spawner_configured=False) is None + + +def test_non_admin_none_spawn_forbidden_without_spawner() -> None: + """Admin is the only bypass for spawn none; a missing spawner is not.""" + definition = _definition([_step(spawn="none")]) + assert _status(definition, spawner_configured=False) == 403 diff --git a/tests/unit/cloud_agents/test_workflows_endpoint.py b/tests/unit/cloud_agents/test_workflows_endpoint.py index e65cebdf0..befe10540 100644 --- a/tests/unit/cloud_agents/test_workflows_endpoint.py +++ b/tests/unit/cloud_agents/test_workflows_endpoint.py @@ -1,6 +1,6 @@ """Unit tests for the /v1/workflows/* endpoints.""" -# pylint: disable=protected-access,too-few-public-methods,unused-argument,import-outside-toplevel +# pylint: disable=protected-access,too-few-public-methods,unused-argument,import-outside-toplevel,too-many-public-methods from __future__ import annotations @@ -21,6 +21,13 @@ from models.api.requests.agents import ApproveWorkflowRequest, RunWorkflowRequest +def _request(mocker: MockerFixture, admin: bool = True) -> Any: + """Build a mock request whose caller is (by default) an admin.""" + request = mocker.MagicMock() + mocker.patch("app.endpoints.workflows.is_admin", return_value=admin) + return request + + @pytest.fixture(autouse=True) def reset_executor() -> Generator[None, None, None]: """Reset the module-level executor singleton.""" @@ -161,7 +168,7 @@ async def test_starts_workflow( } ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) result = await start_workflow_handler.__wrapped__(request, body, auth) @@ -205,7 +212,7 @@ async def test_one_step_workflow_definition_forwarded( provider={"name": "openai", "model": "gpt-4o-mini"}, ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) result = await start_workflow_handler.__wrapped__(request, body, auth) @@ -230,7 +237,7 @@ async def test_provider_falls_back_to_configured_defaults( definition=_valid_definition(), ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) await start_workflow_handler.__wrapped__(request, body, auth) @@ -256,7 +263,7 @@ async def test_explicit_provider_overrides_configured_defaults( provider={"name": "anthropic", "model": "claude-sonnet-5"}, ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) await start_workflow_handler.__wrapped__(request, body, auth) @@ -306,7 +313,7 @@ async def test_secret_bearing_mcp_definition_rejected_with_422( provider={"name": "openai", "model": "gpt-4o-mini"}, ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) from fastapi import HTTPException @@ -337,7 +344,7 @@ async def test_secret_shaped_run_credentials_rejected_with_422( }, ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) from fastapi import HTTPException @@ -368,7 +375,7 @@ async def test_malformed_definition_rejected_with_422( provider={"name": "openai", "model": "gpt-4o-mini"}, ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) from fastapi import HTTPException @@ -394,7 +401,7 @@ async def test_unapproved_run_provider_rejected_with_422( provider={"name": "bogus", "model": "x"}, ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) from fastapi import HTTPException @@ -434,7 +441,7 @@ async def test_unknown_step_type_rejected_with_422( provider={"name": "openai", "model": "gpt-4o-mini"}, ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) from fastapi import HTTPException @@ -466,7 +473,7 @@ async def test_credentials_secret_added_for_known_provider( provider={"name": "anthropic", "model": "claude-sonnet-5"}, ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) await start_workflow_handler.__wrapped__(request, body, auth) @@ -490,7 +497,7 @@ async def test_credentials_secret_omitted_for_unknown_provider( provider={"name": "bedrock", "model": "some-model"}, ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) await start_workflow_handler.__wrapped__(request, body, auth) @@ -498,32 +505,152 @@ async def test_credentials_secret_omitted_for_unknown_provider( assert "credentials_secret" not in workflow_input["provider"] @pytest.mark.asyncio - async def test_caller_supplied_credentials_secret_not_overridden( + async def test_caller_supplied_credentials_secret_rejected( self, mocker: MockerFixture, mock_config: Any, mock_executor: Any, ) -> None: - """A caller-supplied credentials_secret is preserved as-is.""" + """A caller-supplied credentials_secret is 400 (G1), even for admins.""" + from fastapi import HTTPException + mocker.patch("app.endpoints.workflows.check_configuration_loaded") mock_config.spawner_configuration = None - mock_executor.start.return_value = "wf-abc123" body = RunWorkflowRequest( definition=_valid_definition(), provider={ "name": "openai", "model": "gpt-4o", - "credentials_secret": "MY_CUSTOM_KEY", + "credentials_secret": "DATABASE_URL", }, ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() - await start_workflow_handler.__wrapped__(request, body, auth) + with pytest.raises(HTTPException) as exc_info: + await start_workflow_handler.__wrapped__(_request(mocker), body, auth) - workflow_input = mock_executor.start.call_args[0][0] - assert workflow_input["provider"]["credentials_secret"] == "MY_CUSTOM_KEY" + assert exc_info.value.status_code == 400 + mock_executor.start.assert_not_called() + + @pytest.mark.asyncio + async def test_definition_provider_credentials_secret_rejected( + self, + mocker: MockerFixture, + mock_config: Any, + mock_executor: Any, + ) -> None: + """definition.provider.credentials_secret is 400 (G2).""" + from fastapi import HTTPException + + mocker.patch("app.endpoints.workflows.check_configuration_loaded") + mock_config.spawner_configuration = None + definition = _valid_definition() + definition["provider"] = { + "name": "bedrock", + "model": "m", + "credentials_secret": "DATABASE_URL", + } + body = RunWorkflowRequest(definition=definition) + auth = ("user-1", "testuser", False, "token") + + with pytest.raises(HTTPException) as exc_info: + await start_workflow_handler.__wrapped__(_request(mocker), body, auth) + + assert exc_info.value.status_code == 400 + mock_executor.start.assert_not_called() + + @pytest.mark.asyncio + @pytest.mark.parametrize( + "mutate", + [ + lambda d, b: b.update(sandbox_image="evil:1"), + lambda d, b: d.update(advisory=True), + lambda d, b: d["spec"]["steps"][0].update(spawn="local"), + lambda d, b: d.update(skills={"image": "evil:1"}), + ], + ids=["sandbox-image", "advisory", "local-spawn", "skills-image"], + ) + async def test_privileged_options_need_admin( + self, + mocker: MockerFixture, + mock_config: Any, + mock_executor: Any, + mutate: Any, + ) -> None: + """Non-admins get 403 (nothing started); admins are accepted.""" + from fastapi import HTTPException + + mocker.patch("app.endpoints.workflows.check_configuration_loaded") + spawner_config = mocker.MagicMock() + spawner_config.sandbox_image = "default-sandbox:v1" + mock_config.spawner_configuration = spawner_config + mock_executor.start.return_value = "wf-1" + definition = _valid_definition() + definition["spec"]["steps"][0]["spawn"] = "ephemeral" + extras: dict[str, Any] = {} + mutate(definition, extras) + body = RunWorkflowRequest(definition=definition, **extras) + auth = ("user-1", "testuser", False, "token") + + with pytest.raises(HTTPException) as exc_info: + await start_workflow_handler.__wrapped__( + _request(mocker, admin=False), body, auth + ) + assert exc_info.value.status_code == 403 + mock_executor.start.assert_not_called() + + await start_workflow_handler.__wrapped__(_request(mocker), body, auth) + mock_executor.start.assert_called_once() + + @pytest.mark.asyncio + async def test_non_admin_ephemeral_workflow_accepted( + self, + mocker: MockerFixture, + mock_config: Any, + mock_executor: Any, + ) -> None: + """A normal user can run an ephemeral workflow on the default image.""" + mocker.patch("app.endpoints.workflows.check_configuration_loaded") + spawner_config = mocker.MagicMock() + spawner_config.sandbox_image = "default-sandbox:v1" + mock_config.spawner_configuration = spawner_config + mock_executor.start.return_value = "wf-1" + definition = _valid_definition() + definition["spec"]["steps"][0]["spawn"] = "ephemeral" + body = RunWorkflowRequest(definition=definition) + auth = ("user-1", "testuser", False, "token") + + result = await start_workflow_handler.__wrapped__( + _request(mocker, admin=False), body, auth + ) + + assert result["workflow_id"] == "wf-1" + + @pytest.mark.asyncio + async def test_oversized_definition_rejected_with_413( + self, + mocker: MockerFixture, + mock_config: Any, + mock_executor: Any, + ) -> None: + """A definition over the byte cap is 413 before any validation.""" + from fastapi import HTTPException + + from workflow.limits import MAX_DEFINITION_BYTES + + mocker.patch("app.endpoints.workflows.check_configuration_loaded") + mock_config.spawner_configuration = None + definition = _valid_definition() + definition["spec"]["steps"][0]["prompt"] = "x" * (MAX_DEFINITION_BYTES + 1) + body = RunWorkflowRequest(definition=definition) + auth = ("user-1", "testuser", False, "token") + + with pytest.raises(HTTPException) as exc_info: + await start_workflow_handler.__wrapped__(_request(mocker), body, auth) + + assert exc_info.value.status_code == 413 + mock_executor.start.assert_not_called() @pytest.mark.asyncio async def test_sandbox_image_falls_back_to_spawner_config( @@ -548,7 +675,7 @@ async def test_sandbox_image_falls_back_to_spawner_config( definition=_valid_definition(), ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) await start_workflow_handler.__wrapped__(request, body, auth) @@ -562,7 +689,7 @@ async def test_sandbox_image_request_override_wins( mock_config: Any, mock_executor: Any, ) -> None: - """A request-level sandbox_image wins over spawner_configuration.""" + """An admin's request-level sandbox_image wins over spawner_configuration.""" mocker.patch("app.endpoints.workflows.check_configuration_loaded") spawner_config = mocker.MagicMock() spawner_config.sandbox_image = "configured-sandbox:v3" @@ -574,7 +701,7 @@ async def test_sandbox_image_request_override_wins( sandbox_image="custom-sandbox:v9", ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) await start_workflow_handler.__wrapped__(request, body, auth) @@ -597,7 +724,7 @@ async def test_no_spawner_config_falls_back_to_default_image( definition=_valid_definition(), ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) await start_workflow_handler.__wrapped__(request, body, auth) @@ -635,7 +762,7 @@ async def test_forwards_session_id( session_id="ses-abc123", ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) await start_workflow_handler.__wrapped__(request, body, auth) @@ -673,7 +800,7 @@ async def test_session_id_omitted_forwards_none( }, ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) await start_workflow_handler.__wrapped__(request, body, auth) @@ -715,7 +842,7 @@ async def test_forwards_user_id( }, ) auth = ("user-42", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) await start_workflow_handler.__wrapped__(request, body, auth) @@ -747,7 +874,7 @@ async def test_returns_status( ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) result = await get_workflow_handler.__wrapped__(request, "wf-1", auth) @@ -767,7 +894,7 @@ async def test_not_found_raises_404( mock_executor.get_status.side_effect = KeyError("not found") auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) from fastapi import HTTPException @@ -795,7 +922,7 @@ async def test_approves_step( approver="admin", ) auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) result = await approve_workflow_handler.__wrapped__(request, "wf-1", body, auth) @@ -817,7 +944,7 @@ async def test_cancels_workflow( mocker.patch("app.endpoints.workflows.check_configuration_loaded") auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) result = await cancel_workflow_handler.__wrapped__(request, "wf-1", auth) @@ -842,7 +969,7 @@ async def test_returns_transcripts( } auth = ("user-1", "testuser", False, "token") - request = mocker.MagicMock() + request = _request(mocker) result = await get_transcripts_handler.__wrapped__(request, "wf-1", auth)