diff --git a/docs/architecture/agent-factory.drawio b/docs/architecture/agent-factory.drawio index 85262e432..f7d072ce4 100644 --- a/docs/architecture/agent-factory.drawio +++ b/docs/architecture/agent-factory.drawio @@ -42,8 +42,8 @@ - - + + diff --git a/docs/architecture/ai-platform.drawio b/docs/architecture/ai-platform.drawio index 6fb67bfa9..034be2bee 100644 --- a/docs/architecture/ai-platform.drawio +++ b/docs/architecture/ai-platform.drawio @@ -67,7 +67,7 @@ - + diff --git a/docs/superpowers/plans/2026-09-27-agent-dark-factory-plan.md b/docs/superpowers/plans/2026-09-27-agent-dark-factory-plan.md index bee91eed6..296fdfaa4 100644 --- a/docs/superpowers/plans/2026-09-27-agent-dark-factory-plan.md +++ b/docs/superpowers/plans/2026-09-27-agent-dark-factory-plan.md @@ -11794,10 +11794,16 @@ git commit -m "docs(adr): ADR-0045 merge policy gate" *External review R01:* it also exits 1 when such a workflow grants any `write` permission at workflow level, or at job level for a job not in the script's own allowlist (`sarif-upload: security-events`, `render-diff-comment: pull-requests`, `build-and-push: packages, - security-events`). An allowlisted job must contain no `actions/checkout` of the PR head and no - `run:` step. Fixtures: a workflow-level write fails; an unlisted job with write fails. The - allowlist is a gate path (R17). `build-and-push` fails the last clause today: split the push out - of the PR path, or record it as an exception. The `ci.yaml` job split is its own `fix(ci)` PR. + security-events`, `notify-main-broken: issues` — a recorded exception: push-gated by its `if:`, + deliberately checkout-free, its `run:` steps open the tracking issue only after a broken push + to `main`). An allowlisted job must contain no `actions/checkout` of the PR head, and no + `run:` step that executes on a `pull_request` event — a job whose `if:` pins + `github.event_name == 'push'` satisfies this by construction and is listed as push-gated in + the script. Fixtures: a workflow-level write fails; an unlisted job with write fails; a + push-gated allowlisted job with a `run:` step passes; the same job without the push gate + fails. The allowlist is a gate path (R17). `build-and-push` fails the last clause today: + split the push out of the PR path, or record it as an exception. The `ci.yaml` job split is + its own `fix(ci)` PR. - Tasks `ci:policy-gates`, `ci:workflow-secrets`. The canonical gate list is `no_changed_files.paths` of the rule `agent change approved by a @@ -11878,6 +11884,8 @@ echo PASS # # T8: a pull_request workflow may reference GITHUB_TOKEN and no other secret; agent branches # live in this repo, so their PRs run with its secrets. +# External review R01: such a workflow also grants no write permission, at workflow level +# or in a job outside the script's allowlist. set -uo pipefail HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" SUBJECT="$HERE/../check-workflow-secrets.sh" @@ -11894,6 +11902,11 @@ on: {push: {branches: [main]}} jobs: {a: {runs-on: x, steps: [{run: "echo ${{ secrets.DEPLOY_KEY }}"}]}} EOF WORKFLOWS_DIR="$d" bash "$SUBJECT" >/dev/null 2>&1 || fail "GITHUB_TOKEN, and secrets on push-only workflows, pass" +cat >"$d/push-gated-run.yml" <<'EOF' +on: {pull_request: {}} +jobs: {notify-main-broken: {if: "github.event_name == 'push'", runs-on: x, permissions: {issues: write}, steps: [{run: "echo ok"}]}} +EOF +WORKFLOWS_DIR="$d" bash "$SUBJECT" >/dev/null 2>&1 || fail "a push-gated allowlisted job with a run: step passes" cat >"$d/bad.yml" <<'EOF' on: pull_request_target: @@ -11901,6 +11914,26 @@ jobs: {a: {runs-on: x, steps: [{run: "echo ${{ secrets.SLACK_WEBHOOK }}"}]}} EOF out="$(WORKFLOWS_DIR="$d" bash "$SUBJECT" 2>&1)" && fail "a pull_request_target workflow with a secret fails" grep -q 'bad.yml.*SLACK_WEBHOOK' <<<"$out" || fail "the failure names the file and the secret" +cat >"$d/wf-write.yml" <<'EOF' +on: {pull_request: {}} +permissions: {contents: write} +jobs: {a: {runs-on: x, steps: [{run: "echo ok"}]}} +EOF +out="$(WORKFLOWS_DIR="$d" bash "$SUBJECT" 2>&1)" && fail "a pull_request workflow with a workflow-level write permission fails" +grep -q 'wf-write.yml.*contents' <<<"$out" || fail "the failure names the file and the permission" +cat >"$d/job-write.yml" <<'EOF' +on: {pull_request: {}} +jobs: + upload: {runs-on: x, permissions: {security-events: write}, steps: [{run: "echo ok"}]} +EOF +out="$(WORKFLOWS_DIR="$d" bash "$SUBJECT" 2>&1)" && fail "a pull_request workflow with an unlisted job holding write fails" +grep -q 'job-write.yml.*upload' <<<"$out" || fail "the failure names the file and the job" +cat >"$d/ungated-run.yml" <<'EOF' +on: {pull_request: {}} +jobs: {notify-main-broken: {runs-on: x, permissions: {issues: write}, steps: [{run: "echo ok"}]}} +EOF +out="$(WORKFLOWS_DIR="$d" bash "$SUBJECT" 2>&1)" && fail "an allowlisted job with a run: step and no push gate fails" +grep -q 'ungated-run.yml.*notify-main-broken' <<<"$out" || fail "the failure names the file and the job" [ "$fails" -eq 0 ] || exit 1 echo PASS ``` @@ -11973,21 +12006,76 @@ PY # T8 (SP3 §8): agent branches live in this repository, so their PRs run pull_request workflows # with its secrets. Only GITHUB_TOKEN may appear in such a workflow; a new secret-bearing # workflow must fence agent heads first. +# External review R01: such a workflow also grants no write permission — write scopes live +# only in allowlisted jobs that run no PR code. The allowlist below is a gate path (R17). set -euo pipefail DIR="${WORKFLOWS_DIR:-$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)/.github/workflows}" DIR="$DIR" python3 - <<'PY' import glob, os, re, sys, yaml +# job -> the write scopes it may hold. build-and-push's split out of the PR path is its +# own fix(ci) PR; until it lands the lint flags the job's run steps, and the entry stays +# so the split cannot quietly widen it again. +ALLOWLIST = { + "sarif-upload": {"security-events"}, + "render-diff-comment": {"pull-requests"}, + "build-and-push": {"packages", "security-events"}, + # notify-main-broken: push-gated if:, no checkout by design, run: only opens the tracking issue + "notify-main-broken": {"issues"}, +} + +def write_scopes(perms): + # `permissions: write` (a bare string) is write on every scope + if perms == "write": + return None + if isinstance(perms, dict): + return {k for k, v in perms.items() if v == "write"} + return set() + bad = [] for f in sorted(glob.glob(os.path.join(os.environ["DIR"], "*.y*ml"))): text = open(f).read() doc = yaml.safe_load(text) or {} on = doc.get("on", doc.get(True, {})) # PyYAML reads the key `on` as True events = on if isinstance(on, (dict, list)) else [on] - if not any(e in ("pull_request", "pull_request_target") for e in events): + triggers = {e for e in events if e in ("pull_request", "pull_request_target")} + if not triggers: continue - for name in sorted(set(re.findall(r"secrets\.([A-Za-z0-9_]+)", text)) - {"GITHUB_TOKEN"}): - bad.append(f"{os.path.basename(f)}: secrets.{name} in a pull_request workflow") + name = os.path.basename(f) + for secret in sorted(set(re.findall(r"secrets\.([A-Za-z0-9_]+)", text)) - {"GITHUB_TOKEN"}): + bad.append(f"{name}: secrets.{secret} in a pull_request workflow") + scopes = write_scopes(doc.get("permissions")) + if scopes is None: + bad.append(f"{name}: workflow-level permissions: write (every scope)") + else: + for scope in sorted(scopes): + bad.append(f"{name}: workflow-level {scope}: write") + for job, spec in (doc.get("jobs") or {}).items(): + if not isinstance(spec, dict): + continue + held = write_scopes(spec.get("permissions")) + if held is None: + bad.append(f"{name}: job '{job}' holds write on every scope") + continue + for scope in sorted(held - ALLOWLIST.get(job, set())): + bad.append(f"{name}: job '{job}' holds {scope}: write and is not in the allowlist") + if job in ALLOWLIST: + # An allowlisted job runs no PR code. pull_request's default checkout ref is + # the PR merge commit; pull_request_target's is the base, so only an explicit + # pull_request ref counts there. + steps = [s for s in (spec.get("steps") or []) if isinstance(s, dict)] + # a push-gated if: means the job's run: steps never execute on a pull_request event + job_if = str(spec.get("if") or "") + push_gated = "github.event_name == 'push'" in job_if or "!= 'pull_request'" in job_if + if any("run" in s for s in steps) and not push_gated: + bad.append(f"{name}: allowlisted job '{job}' has a run: step") + for s in steps: + if str(s.get("uses") or "").split("@")[0] != "actions/checkout": + continue + ref = str((s.get("with") or {}).get("ref") or "") + if "github.event.pull_request" in ref or "github.head_ref" in ref \ + or (not ref and "pull_request" in triggers): + bad.append(f"{name}: allowlisted job '{job}' checks out the PR head") for b in bad: print("FAIL:", b, file=sys.stderr) sys.exit(1 if bad else 0) diff --git a/docs/superpowers/specs/2026-09-23-agent-dark-factory-design.md b/docs/superpowers/specs/2026-09-23-agent-dark-factory-design.md index 71e6eb9a5..c5f61af0e 100644 --- a/docs/superpowers/specs/2026-09-23-agent-dark-factory-design.md +++ b/docs/superpowers/specs/2026-09-23-agent-dark-factory-design.md @@ -509,4 +509,4 @@ deciding risk · agents deploying anything · auto-reverting human merges · a m | 3 | `Smana/agent-platform` | `Task` CRD and controller (intake, run-request API, scheduler, triage, templates, run meter, CI/policy watch, auto-merge arming, revert watcher, metrics), strict config, envtest suite, signed chart | Unit and envtest suites green | | 4 | this repo | ADR-0048; HelmRelease, config, factory App key via `agents-secrets`, CNPs, Kyverno `AgentRun` creator rule, Kueue queues under `clusters/aws-0-agent-platform/`; RunLore `notify.templated` block; VMRule + dashboard inside the umbrella | `validate-manifests.sh`, `validate-vmrules.sh` exit 0 | | 5 | live trial | `link-rot` schedule, one review-class issue, the kill-switch drill, all `public`; budgets in shadow for the first week (OD-10) | SC-1…SC-8, SC-10…SC-14 → `/verify-spec` | -| 6 | after SP4's Bedrock backend (OD-12/13) | A RunLore replay: `internal` tasks have no backend before it | SC-9 | +| 6 | after the Anthropic API backend lands (agentgateway migration phase I, ADR-0054; OD-12/13) | A RunLore replay: `internal` tasks have no backend before it | SC-9 | diff --git a/docs/superpowers/specs/2026-09-23-agent-runtime-identity-design.md b/docs/superpowers/specs/2026-09-23-agent-runtime-identity-design.md index 7843fa22a..e035ef279 100644 --- a/docs/superpowers/specs/2026-09-23-agent-runtime-identity-design.md +++ b/docs/superpowers/specs/2026-09-23-agent-runtime-identity-design.md @@ -486,7 +486,7 @@ secrets; `id-token: write` only on push and schedule workflows. | T10 | Unauthorised claims | After SP3, only the factory SA creates `AgentRun`s (SP3's Kyverno rule, admins included), and the factory derives the principal. A repo opts in three times: the branch ruleset, trust policies and App install | Before SP3, the owner creates runs directly. Break-glass is suspending the rule through Flux, which is visible in Git | | T11 | Harness supply chain | Profiles pinned by digest; Trivy; no image field in the claim | Lands with the next reviewed bump | | T12 | MCP data exposure | Read-only, no `secrets`, per-role tools | Logs and ConfigMaps may hold secrets. Cluster-wide `get pods` also exposes pod specs (env `value`, args) and, under `FallbackToLogsOnError`, `status...terminated.message` (log tail) — reachable by every `internal` run, not only reviewer/tester/triager (review M3) | -| T13 | CI tampering | No `workflows` permission; PR CI holds no secrets | `contents: read` in every job that runs PR code. Write scopes live only in jobs that run no PR code (`sarif-upload`, `render-diff-comment`), enforced by the T8 lint (SP3 plan Task 6.3). *External review R01: until that `ci.yaml` fix lands, `security-events: write` is workflow-level, not only on the SARIF upload* | +| T13 | CI tampering | No `workflows` permission; PR CI holds no secrets | `contents: read` in every job that runs PR code. Write scopes live only in jobs that run no PR code (`sarif-upload`, `render-diff-comment`), enforced by the T8 lint (SP3 plan Task 6.3) | | T14 | **Pre-existing:** the `openbao-platform` ClusterSecretStore has no namespace `conditions`, so any namespace can read any `platform/` path | SP1 never uses it (S9). `agents-no-secret-import` blocks ESO objects in `agents` | Any *other* namespace with ExternalSecret rights can read `platform/agents/*`. Fixing the cluster store is out of scope (O1) | **Addressed by SP2 plan P38 (2026-09-27):** agent credentials move to a dedicated `agents` mount that the `external-secrets` policy does not cover. | T15 | `internal` data reaching a SaaS model | `dataClass` is required at creation. The audience binds the class. Z.ai routes only on `public`. Cluster-read MCP tools only on `internal` | A human creating a run can misclassify internal content as `public`. Once SP3 ships it sets the class from the task source | | T16 | Reserved-audience minting from an excluded namespace | Kyverno's global config excludes `kube-system` and `security` (its own namespace) from admission, so a pod there (ESO, cert-manager) can still mint the reserved audiences with a live `TokenRequest` call; `agent-audience-token-request` covers every other namespace | **Accepted:** needs a compromised platform controller in `kube-system` or `security` | diff --git a/website/content/docs/platform/ai-platform/agents/_index.md b/website/content/docs/platform/ai-platform/agents/_index.md index 12688171c..563b0e40c 100644 --- a/website/content/docs/platform/ai-platform/agents/_index.md +++ b/website/content/docs/platform/ai-platform/agents/_index.md @@ -62,7 +62,7 @@ The diagram below shows the **target** architecture: the whole programme once bu each box as deployed on `gcp-0` (noting where its live gate is pending), built but not yet deployed, or planned. -![The Agent Factory's target architecture. Triggers: a GitHub repository (the factory/ready and factory/stop labels; a PR review asking for changes, built but not deployed), RunLore findings (planned), the task agent:run CLI, a developer in a browser (approving is planned), and roomctl (planned). The factory turns a labelled issue into a task: intake and narration from a fixed template, then the Task controller, which starts one implementer per task, opens a room and runs the run meter, with a kill switch beside it; all deployed on gcp-0, live gate pending. The reviewer pair and revise flow are built but not deployed; teams with a tester, Kueue admission and the merge gate (policy-bot and a merger App, auto-merge and rollback in shadow) are planned. Rooms: a web UI behind oauth2-proxy and ZITADEL SSO, the room-broker and its append-only CNPG log are deployed, with the steering, room tools and verdicts still awaiting their live gate; approval cards and fork are planned. The runtime turns an AgentRun claim, through Crossplane, into a default-deny CiliumNetworkPolicy, projected tokens and a gVisor Sandbox pod holding the room-bridge sidecar, the OpenHands harness and an Envoy identity-proxy, on a GKE Sandbox pool on gcp-0 (deployed) or a Karpenter AL2023 pool on aws-0 (built). The proxy sends every call with a per-run JWT to Agent Router (Envoy AI Gateway 1.1.0, deployed and being replaced by agentgateway, selected on 2026-10-01, with a PoC instance on gcp-0), which routes to Z.ai GLM-5.3 (deployed), Claude on Bedrock for aws-0 and on Vertex AI for gcp-0 (planned), the MCP servers and octo-sts, which mints a token for the agents' GitHub App, confined by rulesets to agent/** branches and no tags. Agent Router also carries the agents' room_* tools to the broker; token budgets and tiers are planned. The room-bridge streams events to the broker over TLS with a room token, and the broker posts verdicts on the PR. Spans go through the agent-traces-collector to VictoriaTraces, step logs to VictoriaLogs, and access logs, gen_ai metrics and AgentRun state to VictoriaMetrics, all shown on the agent-run and agent-fleet Grafana dashboards. The same manifests deploy to gcp-0, the live cluster, and aws-0, destroyed and rebuilt on demand](/images/diagrams/agent-factory.svg) +![The Agent Factory's target architecture. Triggers: a GitHub repository (the factory/ready and factory/stop labels; a PR review asking for changes, built but not deployed), RunLore findings (planned), the task agent:run CLI, a developer in a browser (approving is planned), and roomctl (planned). The factory turns a labelled issue into a task: intake and narration from a fixed template, then the Task controller, which starts one implementer per task, opens a room and runs the run meter, with a kill switch beside it; all deployed on gcp-0, live gate pending. The reviewer pair and revise flow are built but not deployed; teams with a tester, Kueue admission and the merge gate (policy-bot and a merger App, auto-merge and rollback in shadow) are planned. Rooms: a web UI behind oauth2-proxy and ZITADEL SSO, the room-broker and its append-only CNPG log are deployed, with the steering, room tools and verdicts still awaiting their live gate; approval cards and fork are planned. The runtime turns an AgentRun claim, through Crossplane, into a default-deny CiliumNetworkPolicy, projected tokens and a gVisor Sandbox pod holding the room-bridge sidecar, the OpenHands harness and an Envoy identity-proxy, on a GKE Sandbox pool on gcp-0 (deployed) or a Karpenter AL2023 pool on aws-0 (built). The proxy sends every call with a per-run JWT to Agent Router (Envoy AI Gateway 1.1.0, deployed and being replaced by agentgateway, selected on 2026-10-01, with a PoC instance on gcp-0), which routes to Z.ai GLM-5.3 (deployed), the Anthropic API direct for internal data with a gateway-held key (planned), Bedrock and Vertex AI optional per cloud (planned), the MCP servers and octo-sts, which mints a token for the agents' GitHub App, confined by rulesets to agent/** branches and no tags. Agent Router also carries the agents' room_* tools to the broker; token budgets and tiers are planned. The room-bridge streams events to the broker over TLS with a room token, and the broker posts verdicts on the PR. Spans go through the agent-traces-collector to VictoriaTraces, step logs to VictoriaLogs, and access logs, gen_ai metrics and AgentRun state to VictoriaMetrics, all shown on the agent-run and agent-fleet Grafana dashboards. The same manifests deploy to gcp-0, the live cluster, and aws-0, destroyed and rebuilt on demand](/images/diagrams/agent-factory.svg) *Source: [`docs/architecture/agent-factory.drawio`](https://github.com/Smana/cloud-native-ref/blob/main/docs/architecture/agent-factory.drawio).* diff --git a/website/static/images/diagrams/agent-factory.svg b/website/static/images/diagrams/agent-factory.svg index 4a1a8fb35..cc00cbd16 100644 --- a/website/static/images/diagrams/agent-factory.svg +++ b/website/static/images/diagrams/agent-factory.svg @@ -1,3 +1,3 @@ -
Agent Factory: target architecture
Agent Factory: target architecture
The whole programme once built: a labelled issue becomes sandboxed agent runs, each under its own identity, collaborating in a room. The agent is not trusted: every control sits outside the sandbox.
The stroke of each box says how far it has got (legend, bottom left). The same manifests deploy to gcp-0 and aws-0.
The whole programme once built: a labelled issue becomes sandboxed agent runs, each under its own identity, collaborating in a room. The agent is not trusted: every control sits outside the sandbox....
People and triggers
People and triggers
GitHub repository
issue label factory/ready · factory/stop on a task
PR review "Request changes" (built, not deployed)
GitHub repository...
RunLore
findings become
factory tasks
RunLore...
CLI
task agent:run
(a hand-launched run)
CLI...
A developer, in a browser
watch a room, post, steer
(approve: planned)
A developer, in a browser...
roomctl
CLI: read, chat, queue, fork
roomctl...
Agent factory (SP3)
Agent factory (SP3)
Intake · narration
labelled issue → Task (fixed template)
started / PR / end comments
live gate pending
Intake · narration...
Reviewer pair · revise
Request changes → a new run
/factory retry
Reviewer pair · revise...
Teams · Kueue admission
implementer, reviewer, tester
Kueue caps and drains sandboxes
Teams · Kueue admission...
Task controller · live gate pending
solo implementer · a room per task
run meter: revokes a run at its
token cap (reads VictoriaMetrics)
Task controller · live gate pending...
Kill switch · live gate pending
agent-factory-stop ConfigMap: all
factory/stop label: one task
control-issue label (planned)
Kill switch · live gate pending...
Merge gate
policy-bot + merger App
auto-merge low-risk classes
auto-merge and rollback in shadow
Merge gate...
Rooms (SP2)
Rooms (SP2)
Web UI · oauth2-proxy
ZITADEL SSO, agents-admin/-member
watch · post or queue · hand to a role
steer, interrupt: live gate pending
Web UI · oauth2-proxy...
room-broker
append-only log API · 2 replicas
Postgres LISTEN/NOTIFY fan-out
room_* MCP tools · verdicts:
live gate pending
room-broker...
Room log
CNPG Postgres, append-only
by grant and trigger
Room log...
Approval cards
a human approves an
agent action (in progress)
Approval cards...
Fork
branch a room from any event
(with roomctl)
Fork...
Runtime and identity (SP1)
Runtime and identity (SP1)
AgentRun claim
Crossplane composes: SA, task
ConfigMap, CNP, agent-sandbox
Sandbox
AgentRun claim...
CiliumNetworkPolicy
default-deny; egress only to
named FQDNs and the router
CiliumNetworkPolicy...
Projected tokens per run
gateway · token exchange,
valid until the deadline
room-broker: 600 s, held by
the room-bridge only
Projected tokens per run...
Sandbox pod
Sandbox pod
room-bridge (native sidecar)
polls the harness, streams events to the broker
room-bridge (native sidecar)...
harness
OpenHands agent-server
+ agent-run
no gateway token
GitHub token in memory
(≤ 1 h)
harness...
identity-proxy
Envoy · attaches the
run's projected token
to every call
identity-proxy...
gVisor · RuntimeClass gvisor
user-space kernel · restricted pod security
no service-account token in the harness
gVisor · RuntimeClass gvisor...
gcp-0 pool
GKE Sandbox,
agents-gvisor
gcp-0 pool...
aws-0 pool
Karpenter AL2023,
agents-gvisor
aws-0 pool...
Agent gateway (SP1, SP4)
Agent gateway (SP1, SP4)
Agent Router (being replaced)
Envoy AI Gateway 1.1.0
per-run JWT, audience = role
+ data class
listeners public · internal · sts
model and MCP routes
Agent Router (being replaced)...
Token budgets · tiers
per run and per fleet (B1–B2)
model tiers by data class
Token budgets · tiers...
agentgateway
the agents' gateway: models, MCP
and the sts listener; selected
2026-10-01 · PoC instance on gcp-0
agentgateway...
Z.ai GLM-5.3
public data · the agents' own key
Z.ai GLM-5.3...
Claude on Bedrock EU
internal data · aws-0, Pod Identity
Claude on Bedrock EU...
Claude on Vertex AI
internal data · gcp-0, Workload Identity
Claude on Vertex AI...
MCP servers
Flux · VictoriaMetrics · VictoriaLogs
MCP servers...
octo-sts
scoped, short-lived installation tokens
octo-sts...
GitHub App (agents)
rulesets: push only agent/** branches,
no tags; cannot merge
GitHub App (agents)...
Observability (O-1)
Observability (O-1)
agent-traces-collector
OpenTelemetry · run id from the pod
metadata allowlist · router, factory spans
agent-traces-collector...
VictoriaTraces
one trace per run
VictoriaTraces...
VictoriaLogs
step log · access log
VictoriaLogs...
VictoriaMetrics
gen_ai metrics · KSM AgentRun state
VMRules alerts
VictoriaMetrics...
Grafana
agent-run and agent-fleet dashboards
Grafana...
Legend
Legend
deployed on gcp-0 (live gate pending where noted)
deployed on gcp-0 (live gate pending where noted)
built and reviewed, not yet deployed
built and reviewed, not yet deployed
planned
planned
platform component
platform component
identity / security
identity / security
storage
storage
outside the cluster
outside the cluster
request / control flow
request / control flow
event / telemetry
event / telemetry
planned flow
planned flow
Where it runs
Where it runs
gcp-0 · GKE
the live cluster: tracks integration/agent-factory
GCP is the primary cloud (programme ADR-0052,
on the programme branches, not yet on main)
gcp-0 · GKE...
aws-0 · EKS
supported, same manifests; destroyed
2026-09-29, rebuilt on demand
aws-0 · EKS...
label
label
narration
narration
Request changes
Request changes
creates runs
creates runs
task agent:run
task agent:run
a room
per task
a room...
merges
merges
verdict comment
verdict comment
per-run JWT
per-run JWT
:8443 TLS · room token
:8443 TLS · room token
room_* MCP tools (:8090)
room_* MCP tools (:8090)
git push, PR
git push, PR
spans
spans
step log
step log
access log, gen_ai metrics
access log, gen_ai metrics
Text is not SVG - cannot display
+
Agent Factory: target architecture
Agent Factory: target architecture
The whole programme once built: a labelled issue becomes sandboxed agent runs, each under its own identity, collaborating in a room. The agent is not trusted: every control sits outside the sandbox.
The stroke of each box says how far it has got (legend, bottom left). The same manifests deploy to gcp-0 and aws-0.
The whole programme once built: a labelled issue becomes sandboxed agent runs, each under its own identity, collaborating in a room. The agent is not trusted: every control sits outside the sandbox....
People and triggers
People and triggers
GitHub repository
issue label factory/ready · factory/stop on a task
PR review "Request changes" (built, not deployed)
GitHub repository...
RunLore
findings become
factory tasks
RunLore...
CLI
task agent:run
(a hand-launched run)
CLI...
A developer, in a browser
watch a room, post, steer
(approve: planned)
A developer, in a browser...
roomctl
CLI: read, chat, queue, fork
roomctl...
Agent factory (SP3)
Agent factory (SP3)
Intake · narration
labelled issue → Task (fixed template)
started / PR / end comments
live gate pending
Intake · narration...
Reviewer pair · revise
Request changes → a new run
/factory retry
Reviewer pair · revise...
Teams · Kueue admission
implementer, reviewer, tester
Kueue caps and drains sandboxes
Teams · Kueue admission...
Task controller · live gate pending
solo implementer · a room per task
run meter: revokes a run at its
token cap (reads VictoriaMetrics)
Task controller · live gate pending...
Kill switch · live gate pending
agent-factory-stop ConfigMap: all
factory/stop label: one task
control-issue label (planned)
Kill switch · live gate pending...
Merge gate
policy-bot + merger App
auto-merge low-risk classes
auto-merge and rollback in shadow
Merge gate...
Rooms (SP2)
Rooms (SP2)
Web UI · oauth2-proxy
ZITADEL SSO, agents-admin/-member
watch · post or queue · hand to a role
steer, interrupt: live gate pending
Web UI · oauth2-proxy...
room-broker
append-only log API · 2 replicas
Postgres LISTEN/NOTIFY fan-out
room_* MCP tools · verdicts:
live gate pending
room-broker...
Room log
CNPG Postgres, append-only
by grant and trigger
Room log...
Approval cards
a human approves an
agent action (in progress)
Approval cards...
Fork
branch a room from any event
(with roomctl)
Fork...
Runtime and identity (SP1)
Runtime and identity (SP1)
AgentRun claim
Crossplane composes: SA, task
ConfigMap, CNP, agent-sandbox
Sandbox
AgentRun claim...
CiliumNetworkPolicy
default-deny; egress only to
named FQDNs and the router
CiliumNetworkPolicy...
Projected tokens per run
gateway · token exchange,
valid until the deadline
room-broker: 600 s, held by
the room-bridge only
Projected tokens per run...
Sandbox pod
Sandbox pod
room-bridge (native sidecar)
polls the harness, streams events to the broker
room-bridge (native sidecar)...
harness
OpenHands agent-server
+ agent-run
no gateway token
GitHub token in memory
(≤ 1 h)
harness...
identity-proxy
Envoy · attaches the
run's projected token
to every call
identity-proxy...
gVisor · RuntimeClass gvisor
user-space kernel · restricted pod security
no service-account token in the harness
gVisor · RuntimeClass gvisor...
gcp-0 pool
GKE Sandbox,
agents-gvisor
gcp-0 pool...
aws-0 pool
Karpenter AL2023,
agents-gvisor
aws-0 pool...
Agent gateway (SP1, SP4)
Agent gateway (SP1, SP4)
Agent Router (being replaced)
Envoy AI Gateway 1.1.0
per-run JWT, audience = role
+ data class
listeners public · internal · sts
model and MCP routes
Agent Router (being replaced)...
Token budgets · tiers
per run and per fleet (B1–B2)
model tiers by data class
Token budgets · tiers...
agentgateway
the agents' gateway: models, MCP
and the sts listener; selected
2026-10-01 · PoC instance on gcp-0
agentgateway...
Z.ai GLM-5.3
public data · the agents' own key
Z.ai GLM-5.3...
Anthropic API (direct)
internal data · gateway-held key
Anthropic API (direct)...
Bedrock · Vertex AI
internal data · optional, per cloud
Bedrock · Vertex AI...
MCP servers
Flux · VictoriaMetrics · VictoriaLogs
MCP servers...
octo-sts
scoped, short-lived installation tokens
octo-sts...
GitHub App (agents)
rulesets: push only agent/** branches,
no tags; cannot merge
GitHub App (agents)...
Observability (O-1)
Observability (O-1)
agent-traces-collector
OpenTelemetry · run id from the pod
metadata allowlist · router, factory spans
agent-traces-collector...
VictoriaTraces
one trace per run
VictoriaTraces...
VictoriaLogs
step log · access log
VictoriaLogs...
VictoriaMetrics
gen_ai metrics · KSM AgentRun state
VMRules alerts
VictoriaMetrics...
Grafana
agent-run and agent-fleet dashboards
Grafana...
Legend
Legend
deployed on gcp-0 (live gate pending where noted)
deployed on gcp-0 (live gate pending where noted)
built and reviewed, not yet deployed
built and reviewed, not yet deployed
planned
planned
platform component
platform component
identity / security
identity / security
storage
storage
outside the cluster
outside the cluster
request / control flow
request / control flow
event / telemetry
event / telemetry
planned flow
planned flow
Where it runs
Where it runs
gcp-0 · GKE
the live cluster: tracks integration/agent-factory
GCP is the primary cloud (programme ADR-0052,
on the programme branches, not yet on main)
gcp-0 · GKE...
aws-0 · EKS
supported, same manifests; destroyed
2026-09-29, rebuilt on demand
aws-0 · EKS...
label
label
narration
narration
Request changes
Request changes
creates runs
creates runs
task agent:run
task agent:run
a room
per task
a room...
merges
merges
verdict comment
verdict comment
per-run JWT
per-run JWT
:8443 TLS · room token
:8443 TLS · room token
room_* MCP tools (:8090)
room_* MCP tools (:8090)
git push, PR
git push, PR
spans
spans
step log
step log
access log, gen_ai metrics
access log, gen_ai metrics
Text is not SVG - cannot display
diff --git a/website/static/images/diagrams/ai-platform-2.svg b/website/static/images/diagrams/ai-platform-2.svg index e79a7b5a5..5a350f7b2 100644 --- a/website/static/images/diagrams/ai-platform-2.svg +++ b/website/static/images/diagrams/ai-platform-2.svg @@ -1,3 +1,3 @@ -
Two gateways, one per kind of caller
Two gateways, one per kind of caller
ai-gateway knows a client by its API key; the agent gateway knows a run by its own token.
ai-gateway knows a client by its API key; the agent gateway knows a run by its own token.
ai-gateway · humans and coding clients
ai-gateway · humans and coding clients
Developers
OpenCode · Continue · OpenWebUI
Developers...
ai-gateway · Envoy AI Gateway
ai-gateway · Envoy AI Gateway
API key check
key from Secrets Manager
API key check...
Semantic Router
picks a model for MoM
Semantic Router...
vLLM models
InferenceService claims
vLLM models...
OpenAI API
tailnet
OpenAI API...
by model
by model
Agent gateway · agent runs
Agent gateway · agent runs
Agent run · gVisor sandbox
Agent run · gVisor sandbox
Harness
OpenHands · no token
Harness...
identity-proxy
Envoy · adds run token
identity-proxy...
Agent gateway · Agent Router today, agentgateway next
Agent gateway · Agent Router today, agentgateway next
Verify run token
JWT on every call
Verify run token...
Meter and route
tokens per run · alias
Meter and route...
Z.ai
GLM
Z.ai...
Claude
Bedrock · Vertex AI
Claude...
MCP tools
Flux · Victoria · room
MCP tools...
octo-sts
GitHub token
octo-sts...
every call
every call
Observability
Victoria stack · Grafana
Observability...
Text is not SVG - cannot display
+
Two gateways, one per kind of caller
Two gateways, one per kind of caller
ai-gateway knows a client by its API key; the agent gateway knows a run by its own token.
ai-gateway knows a client by its API key; the agent gateway knows a run by its own token.
ai-gateway · humans and coding clients
ai-gateway · humans and coding clients
Developers
OpenCode · Continue · OpenWebUI
Developers...
ai-gateway · Envoy AI Gateway
ai-gateway · Envoy AI Gateway
API key check
key from Secrets Manager
API key check...
Semantic Router
picks a model for MoM
Semantic Router...
vLLM models
InferenceService claims
vLLM models...
OpenAI API
tailnet
OpenAI API...
by model
by model
Agent gateway · agent runs
Agent gateway · agent runs
Agent run · gVisor sandbox
Agent run · gVisor sandbox
Harness
OpenHands · no token
Harness...
identity-proxy
Envoy · adds run token
identity-proxy...
Agent gateway · Agent Router today, agentgateway next
Agent gateway · Agent Router today, agentgateway next
Verify run token
JWT on every call
Verify run token...
Meter and route
tokens per run · alias
Meter and route...
Z.ai
GLM
Z.ai...
Claude
Anthropic API (direct)
Claude...
MCP tools
Flux · Victoria · room
MCP tools...
octo-sts
GitHub token
octo-sts...
every call
every call
Observability
Victoria stack · Grafana
Observability...
Text is not SVG - cannot display