From 1229418cd1a116c126584b225c5b4f336ca21b72 Mon Sep 17 00:00:00 2001 From: James Wong <2421248+jameswnl@users.noreply.github.com> Date: Wed, 9 Sep 2026 05:00:41 -0400 Subject: [PATCH 1/2] Add allowed_skills to POST /v1/agents/run and fix ephemeral demo script Threads allowed_skills through AgentRunRequest -> StepInput -> raw_step so the ephemeral (SandboxExecutor -> step_runner -> spawner.spawn) path can materialize just that skill subset with per-skill Landlock grants. Also selects request-level MCP server names into raw_step, since step_runner only injects catalog entries whose names a step lists. Refactors docs/cloud-agents-demo-curl.sh's near-duplicate curl blocks into shared run_agent/submit_workflow/finish_workflow/approval_cycle helpers, updates the ephemeral demo to exercise the new allowed_skills field, and fixes its MCP server URL (kubectl-mcp:8000 was a stale placeholder; the actually-deployed mock is mcp-pod-status-mock:8084). Fixes main.py's OpenAPI servers URL (hardcoded localhost:8080 broke Swagger UI when reverse-proxied) and a pre-existing unused-variable pylint failure in agents.py. Co-Authored-By: Claude Sonnet 5 --- docs/cloud-agents-demo-curl.sh | 327 +++++++----------- docs/devel_doc/openapi.json | 19 +- src/app/endpoints/agents.py | 17 +- src/app/main.py | 4 +- src/models/api/requests/agents.py | 10 + .../unit/cloud_agents/test_agents_endpoint.py | 92 +++++ 6 files changed, 256 insertions(+), 213 deletions(-) diff --git a/docs/cloud-agents-demo-curl.sh b/docs/cloud-agents-demo-curl.sh index e1f663d82..d16e23cb7 100755 --- a/docs/cloud-agents-demo-curl.sh +++ b/docs/cloud-agents-demo-curl.sh @@ -11,7 +11,7 @@ # they aren't part of that illustration, just additional scenarios: # agent-none POST /v1/agents/run, spawn: "none" # agent-local POST /v1/agents/run, spawn: "local" -# agent-ephemeral POST /v1/agents/run, spawn: "ephemeral" +# agent-ephemeral POST /v1/agents/run, spawn: "ephemeral", k8s-diag skill + Landlock demo # workflow-ephemeral-approval POST /v1/workflows/run, spawn: "ephemeral", multi-step + approval # workflow-none-approval POST /v1/workflows/run, spawn: "none", multi-step + approval # workflow-local POST /v1/workflows/run, spawn: "local", single step @@ -44,6 +44,7 @@ AUTH_HEADER=() if [[ -n "${TOKEN:-}" ]]; then AUTH_HEADER=(-H "Authorization: Bearer $TOKEN") fi +WF_ID="" discover() { echo "== Registered agent tools (spawn:none/local) ==" @@ -53,17 +54,35 @@ discover() { curl -s "${AUTH_HEADER[@]+"${AUTH_HEADER[@]}"}" "$BASE_URL/v1/mcp-servers" | jq } -agent_none() { - echo "== Agent — In-Process (spawn: none) ==" - local resp status +run_agent() { + # POST an /v1/agents/run payload, pretty-print the body, and fail on + # HTTP >= 400. $1: title line, $2: spawn mode, $3: prompt, $4: extra + # JSON object merged onto the shared {prompt, spawn, provider, model} + # base -- callers only spell out what differs per spawn mode. + local title="$1" spawn="$2" prompt="$3" extra="{}" + if [[ $# -ge 4 ]]; then + extra="$4" + fi + echo "$title" + local payload resp status + payload=$(jq -n \ + --arg prompt "$prompt" \ + --arg spawn "$spawn" \ + --argjson extra "$extra" \ + '{prompt: $prompt, spawn: $spawn, provider: "openai", model: "gpt-5-mini"} + $extra') resp=$(curl -s -w '\n%{http_code}' -X POST "$BASE_URL/v1/agents/run" \ "${AUTH_HEADER[@]+"${AUTH_HEADER[@]}"}" \ -H "Content-Type: application/json" \ - -d '{ - "prompt": "Is pod checkout-7f9 healthy?", - "spawn": "none", - "provider": "openai", - "model": "gpt-4o-mini", + -d "$payload") + status="${resp##*$'\n'}" + echo "${resp%$'\n'*}" | jq + [[ "$status" -lt 400 ]] +} + +agent_none() { + run_agent "== Agent — In-Process (spawn: none) ==" "none" \ + "Is pod checkout-7f9 healthy?" \ + '{ "tools": [], "mcp_servers": null, "output_schema": { @@ -71,47 +90,24 @@ agent_none() { "properties": { "healthy": {"type": "boolean"}, "reason": {"type": "string"} }, "required": ["healthy", "reason"] } - }') - status="${resp##*$'\n'}" - echo "${resp%$'\n'*}" | jq - [[ "$status" -lt 400 ]] + }' } agent_local() { - echo "== Agent — Subprocess (spawn: local) ==" - local resp status - resp=$(curl -s -w '\n%{http_code}' -X POST "$BASE_URL/v1/agents/run" \ - "${AUTH_HEADER[@]+"${AUTH_HEADER[@]}"}" \ - -H "Content-Type: application/json" \ - -d '{ - "prompt": "Say one sentence confirming pod checkout-7f9 is healthy.", - "spawn": "local", - "provider": "openai", - "model": "gpt-4o-mini", - "tools": [], - "mcp_servers": null - }') - status="${resp##*$'\n'}" - echo "${resp%$'\n'*}" | jq - [[ "$status" -lt 400 ]] + run_agent "== Agent — Subprocess (spawn: local) ==" "local" \ + "Say one sentence confirming pod checkout-7f9 is healthy." \ + '{"tools": [], "mcp_servers": null}' } agent_ephemeral() { - echo "== Agent — OpenShell (spawn: ephemeral) ==" - local resp status - resp=$(curl -s -w '\n%{http_code}' -X POST "$BASE_URL/v1/agents/run" \ - "${AUTH_HEADER[@]+"${AUTH_HEADER[@]}"}" \ - -H "Content-Type: application/json" \ - -d '{ - "prompt": "Is pod checkout-7f9 healthy?", - "spawn": "ephemeral", - "provider": "openai", - "model": "gpt-4o-mini", - "mcp_servers": [{"name": "kubectl-mcp", "url": "http://kubectl-mcp:8000/mcp"}] - }') - status="${resp##*$'\n'}" - echo "${resp%$'\n'*}" | jq - [[ "$status" -lt 400 ]] + # allowed_skills=["k8s-diag"]: the spawner materializes just that skill + # into the sandbox and Landlock-grants /skills/k8s-diag, so the prompt + # below demonstrates both sides -- the allowed skill works, and reading + # an unlisted skill (/skills/security-audit) is denied at the filesystem + # boundary. Requires those skills baked into the sandbox image (/skills). + run_agent "== Agent — OpenShell (spawn: ephemeral, k8s-diag skill + Landlock) ==" "ephemeral" \ + "Use the k8s-diag skill to check whether pod checkout-7f9 is healthy. Then try reading /skills/security-audit/SKILL.md and report whether that read succeeded or was denied, and why." \ + '{"mcp_servers": [{"name": "kubectl-mcp", "url": "http://mcp-pod-status-mock:8084/mcp"}], "provider": "openai", "model": "gpt-5-mini"}' } wait_for_status() { @@ -153,22 +149,23 @@ require_status() { fi } -workflow_ephemeral_approval() { - echo "== Workflow — OpenShell + approval (spawn: ephemeral, POST /v1/workflows/run) ==" - local resp wf_id - - resp=$(curl -sf -X POST "$BASE_URL/v1/workflows/run" \ - "${AUTH_HEADER[@]+"${AUTH_HEADER[@]}"}" \ - -H "Content-Type: application/json" \ - -d '{ +approval_workflow_payload() { + # Triage -> human-approval -> remediate definition shared by both + # approval demos. $1: spawn mode, $2: workflow name, $3: full remediate + # prompt (including the {{ steps.triage_result... }} template ref). + jq -n \ + --arg spawn "$1" \ + --arg name "$2" \ + --arg remediate_prompt "$3" \ + '{ "definition": { "apiVersion": "v1", "kind": "AgentWorkflow", - "metadata": {"name": "triage-remediate-demo"}, + "metadata": {"name": $name}, "spec": { "steps": [ { - "name": "triage", "type": "agent", "spawn": "ephemeral", + "name": "triage", "type": "agent", "spawn": $spawn, "output_key": "triage_result", "prompt": "Diagnose the checkout-7f9 pod issue. Report severity and root cause.", "output_schema": { @@ -185,101 +182,84 @@ workflow_ephemeral_approval() { "risk_level": "high" }, { - "name": "remediate", "type": "agent", "spawn": "ephemeral", + "name": "remediate", "type": "agent", "spawn": $spawn, "output_key": "remediate_result", - "prompt": "Apply the fix for: {{ steps.triage_result.output.root_cause }}", + "prompt": $remediate_prompt, "condition": "steps.approval.output.approved == true", "timeout_seconds": 120 } ] } }, - "provider": {"name": "openai", "model": "gpt-4o-mini"} - }') - echo "$resp" | jq - wf_id=$(echo "$resp" | jq -r .workflow_id) - if [[ -z "$wf_id" || "$wf_id" == "null" ]]; then - echo "ERROR: no workflow_id in response" >&2 - exit 1 - fi - echo "workflow_id=$wf_id" - - echo - echo "-- Waiting for status 'paused' at 'approve' --" - wait_for_status "$wf_id" "paused failed cancelled completed" 150 - require_status "paused" "pause for approval" - - echo - echo "-- Approving 'approve' step --" - curl -sf -X POST "$BASE_URL/v1/workflows/$wf_id/approve" \ - "${AUTH_HEADER[@]+"${AUTH_HEADER[@]}"}" \ - -H "Content-Type: application/json" \ - -d '{"step_name": "approve", "decision": "approved", "approver": "demo-user"}' | jq - - echo - echo "-- Waiting for a terminal status --" - wait_for_status "$wf_id" "completed failed cancelled" 150 - require_status "completed" "complete successfully" - - echo - echo "-- Per-step transcripts --" - curl -sf "${AUTH_HEADER[@]+"${AUTH_HEADER[@]}"}" "$BASE_URL/v1/workflows/$wf_id/transcripts" | jq + "provider": {"name": "openai", "model": "gpt-5-mini"} + }' } -workflow_none_approval() { - echo "== Workflow — In-Process + approval (spawn: none, POST /v1/workflows/run) ==" - local resp wf_id - - resp=$(curl -sf -X POST "$BASE_URL/v1/workflows/run" \ - "${AUTH_HEADER[@]+"${AUTH_HEADER[@]}"}" \ - -H "Content-Type: application/json" \ - -d '{ +single_step_workflow_payload() { + # Single investigate-step definition shared by the non-approval demos. + # $1: spawn mode, $2: workflow name. + jq -n \ + --arg spawn "$1" \ + --arg name "$2" \ + '{ "definition": { "apiVersion": "v1", "kind": "AgentWorkflow", - "metadata": {"name": "triage-remediate-none-demo"}, + "metadata": {"name": $name}, "spec": { "steps": [ { - "name": "triage", "type": "agent", "spawn": "none", - "output_key": "triage_result", - "prompt": "Diagnose the checkout-7f9 pod issue. Report severity and root cause.", - "output_schema": { - "type": "object", - "properties": {"severity": {"type": "string"}, "root_cause": {"type": "string"}}, - "required": ["severity", "root_cause"] - }, - "timeout_seconds": 120 - }, - { - "name": "approve", "type": "human-approval", - "output_key": "approval", - "message": "Root cause: {{ steps.triage_result.output.root_cause }}. Approve remediation?", - "risk_level": "high" - }, - { - "name": "remediate", "type": "agent", "spawn": "none", - "output_key": "remediate_result", - "prompt": "Say one sentence confirming the fix for: {{ steps.triage_result.output.root_cause }}", - "condition": "steps.approval.output.approved == true", + "name": "investigate", "type": "agent", "spawn": $spawn, + "output_key": "investigate_result", + "prompt": "Say one sentence confirming the checkout-7f9 pod is healthy.", "timeout_seconds": 120 } ] } }, - "provider": {"name": "openai", "model": "gpt-4o-mini"} - }') + "provider": {"name": "openai", "model": "gpt-5-mini"} + }' +} + +submit_workflow() { + # POST $1 to /v1/workflows/run, print the response, and set WF_ID + # (intentionally global, mirroring WORKFLOW_STATUS) to the created + # workflow id. Exits non-zero when the response carries no id. + local payload="$1" resp + resp=$(curl -sf -X POST "$BASE_URL/v1/workflows/run" \ + "${AUTH_HEADER[@]+"${AUTH_HEADER[@]}"}" \ + -H "Content-Type: application/json" \ + -d "$payload") echo "$resp" | jq - wf_id=$(echo "$resp" | jq -r .workflow_id) - if [[ -z "$wf_id" || "$wf_id" == "null" ]]; then + WF_ID=$(echo "$resp" | jq -r .workflow_id) + if [[ -z "$WF_ID" || "$WF_ID" == "null" ]]; then echo "ERROR: no workflow_id in response" >&2 exit 1 fi - echo "workflow_id=$wf_id" + echo "workflow_id=$WF_ID" +} + +finish_workflow() { + # Wait for a terminal status and print per-step transcripts. + # $1: workflow id, $2: wait budget in seconds (default 30). + local wf_id="$1" budget="${2:-30}" + echo + echo "-- Waiting for a terminal status --" + wait_for_status "$wf_id" "completed failed cancelled" "$budget" + require_status "completed" "complete successfully" + + echo + echo "-- Per-step transcripts --" + curl -sf "${AUTH_HEADER[@]+"${AUTH_HEADER[@]}"}" "$BASE_URL/v1/workflows/$wf_id/transcripts" | jq +} +approval_cycle() { + # Pause -> approve -> finish flow shared by both approval demos. + # $1: workflow id, $2: wait budget in seconds (default 30). + local wf_id="$1" budget="${2:-30}" echo echo "-- Waiting for status 'paused' at 'approve' --" - wait_for_status "$wf_id" "paused failed cancelled completed" + wait_for_status "$wf_id" "paused failed cancelled completed" "$budget" require_status "paused" "pause for approval" echo @@ -289,100 +269,33 @@ workflow_none_approval() { -H "Content-Type: application/json" \ -d '{"step_name": "approve", "decision": "approved", "approver": "demo-user"}' | jq - echo - echo "-- Waiting for a terminal status --" - wait_for_status "$wf_id" "completed failed cancelled" - require_status "completed" "complete successfully" + finish_workflow "$wf_id" "$budget" +} - echo - echo "-- Per-step transcripts --" - curl -sf "${AUTH_HEADER[@]+"${AUTH_HEADER[@]}"}" "$BASE_URL/v1/workflows/$wf_id/transcripts" | jq +workflow_ephemeral_approval() { + echo "== Workflow — OpenShell + approval (spawn: ephemeral, POST /v1/workflows/run) ==" + submit_workflow "$(approval_workflow_payload "ephemeral" "triage-remediate-demo" \ + "Apply the fix for: {{ steps.triage_result.output.root_cause }}")" + approval_cycle "$WF_ID" 150 +} + +workflow_none_approval() { + echo "== Workflow — In-Process + approval (spawn: none, POST /v1/workflows/run) ==" + submit_workflow "$(approval_workflow_payload "none" "triage-remediate-none-demo" \ + "Say one sentence confirming the fix for: {{ steps.triage_result.output.root_cause }}")" + approval_cycle "$WF_ID" 30 } workflow_local() { echo "== Workflow — Subprocess (spawn: local, POST /v1/workflows/run) ==" - local resp wf_id - - resp=$(curl -sf -X POST "$BASE_URL/v1/workflows/run" \ - "${AUTH_HEADER[@]+"${AUTH_HEADER[@]}"}" \ - -H "Content-Type: application/json" \ - -d '{ - "definition": { - "apiVersion": "v1", - "kind": "AgentWorkflow", - "metadata": {"name": "investigate-local-demo"}, - "spec": { - "steps": [ - { - "name": "investigate", "type": "agent", "spawn": "local", - "output_key": "investigate_result", - "prompt": "Say one sentence confirming the checkout-7f9 pod is healthy.", - "timeout_seconds": 120 - } - ] - } - }, - "provider": {"name": "openai", "model": "gpt-4o-mini"} - }') - echo "$resp" | jq - wf_id=$(echo "$resp" | jq -r .workflow_id) - if [[ -z "$wf_id" || "$wf_id" == "null" ]]; then - echo "ERROR: no workflow_id in response" >&2 - exit 1 - fi - echo "workflow_id=$wf_id" - - echo - echo "-- Waiting for a terminal status --" - wait_for_status "$wf_id" "completed failed cancelled" 150 - require_status "completed" "complete successfully" - - echo - echo "-- Per-step transcripts --" - curl -sf "${AUTH_HEADER[@]+"${AUTH_HEADER[@]}"}" "$BASE_URL/v1/workflows/$wf_id/transcripts" | jq + submit_workflow "$(single_step_workflow_payload "local" "investigate-local-demo")" + finish_workflow "$WF_ID" 150 } workflow_ephemeral() { echo "== Workflow — OpenShell, no approval (spawn: ephemeral, POST /v1/workflows/run) ==" - local resp wf_id - - resp=$(curl -sf -X POST "$BASE_URL/v1/workflows/run" \ - "${AUTH_HEADER[@]+"${AUTH_HEADER[@]}"}" \ - -H "Content-Type: application/json" \ - -d '{ - "definition": { - "apiVersion": "v1", - "kind": "AgentWorkflow", - "metadata": {"name": "investigate-ephemeral-demo"}, - "spec": { - "steps": [ - { - "name": "investigate", "type": "agent", "spawn": "ephemeral", - "output_key": "investigate_result", - "prompt": "Say one sentence confirming the checkout-7f9 pod is healthy.", - "timeout_seconds": 120 - } - ] - } - }, - "provider": {"name": "openai", "model": "gpt-4o-mini"} - }') - echo "$resp" | jq - wf_id=$(echo "$resp" | jq -r .workflow_id) - if [[ -z "$wf_id" || "$wf_id" == "null" ]]; then - echo "ERROR: no workflow_id in response" >&2 - exit 1 - fi - echo "workflow_id=$wf_id" - - echo - echo "-- Waiting for a terminal status --" - wait_for_status "$wf_id" "completed failed cancelled" 150 - require_status "completed" "complete successfully" - - echo - echo "-- Per-step transcripts --" - curl -sf "${AUTH_HEADER[@]+"${AUTH_HEADER[@]}"}" "$BASE_URL/v1/workflows/$wf_id/transcripts" | jq + submit_workflow "$(single_step_workflow_payload "ephemeral" "investigate-ephemeral-demo")" + finish_workflow "$WF_ID" 150 } case "${1:-}" in diff --git a/docs/devel_doc/openapi.json b/docs/devel_doc/openapi.json index fbc154766..a1bb31844 100644 --- a/docs/devel_doc/openapi.json +++ b/docs/devel_doc/openapi.json @@ -17,7 +17,7 @@ }, "servers": [ { - "url": "http://localhost:8080", + "url": "/", "description": "Locally running service" } ], @@ -12660,6 +12660,21 @@ "title": "Mcp Servers", "description": "MCP server configs: [{name, url, headers}]. Each server is connected via pydantic-ai MCPToolset." }, + "allowed_skills": { + "anyOf": [ + { + "items": { + "type": "string" + }, + "type": "array" + }, + { + "type": "null" + } + ], + "title": "Allowed Skills", + "description": "Skill allowlist: names of skill subdirectories the agent may use. For spawn=ephemeral the spawner materializes just this subset into the sandbox and Landlock-grants each /skills/ path, so unlisted skills are denied at the filesystem boundary. Omitted means no skills." + }, "output_schema": { "anyOf": [ { @@ -12692,7 +12707,7 @@ "prompt" ], "title": "AgentRunRequest", - "description": "Request body for POST /v1/agents/run.\n\nAll agent parameters are passed inline \u2014 no registry lookup.\n\nAttributes:\n prompt: The task prompt for the agent.\n instructions: System prompt / instructions.\n model: Full model ID (e.g. \"openai/gpt-4o\").\n provider: Provider name (used with model if no slash in model).\n spawn: Execution mode.\n sandbox_image: Container image for spawn=ephemeral.\n tools: Tool definitions.\n mcp_servers: MCP server names.\n output_schema: JSON Schema for structured output.\n context: Prior context (e.g. from previous steps)." + "description": "Request body for POST /v1/agents/run.\n\nAll agent parameters are passed inline \u2014 no registry lookup.\n\nAttributes:\n prompt: The task prompt for the agent.\n instructions: System prompt / instructions.\n model: Full model ID (e.g. \"openai/gpt-4o\").\n provider: Provider name (used with model if no slash in model).\n spawn: Execution mode.\n sandbox_image: Container image for spawn=ephemeral.\n tools: Tool definitions.\n mcp_servers: MCP server names.\n allowed_skills: Skill allowlist (names of skill subdirectories).\n output_schema: JSON Schema for structured output.\n context: Prior context (e.g. from previous steps)." }, "AgentSkill": { "properties": { diff --git a/src/app/endpoints/agents.py b/src/app/endpoints/agents.py index 9630ba3ef..d9fe2e683 100644 --- a/src/app/endpoints/agents.py +++ b/src/app/endpoints/agents.py @@ -80,7 +80,7 @@ async def run_agent_handler( spawner=spawner, ) - user_id, username, _, _ = auth + user_id, _, _, _ = auth step_input_kwargs: dict[str, Any] = { "prompt": body.prompt, @@ -89,9 +89,24 @@ async def run_agent_handler( "output_schema": body.output_schema, "tools": body.tools, "mcp_servers": body.mcp_servers, + "allowed_skills": body.allowed_skills, "context": body.context or {}, "step_name": "agent-run", "output_key": "result", + # Raw step definition so the ephemeral path (SandboxExecutor -> + # step_runner -> spawner.spawn) sees allowed_skills and can + # materialize just that subset with per-skill Landlock grants. + # Other executors ignore raw_step and read allowed_skills directly. + # mcp_servers names are selected here too: step_runner only injects + # catalog entries (StepInput.mcp_servers) whose names the step lists, + # so without this the sandbox never sees request-level MCP servers. + "raw_step": { + "name": "agent-run", + "prompt": body.prompt, + "output_key": "result", + "allowed_skills": body.allowed_skills, + "mcp_servers": [s["name"] for s in (body.mcp_servers or []) if "name" in s], + }, "metadata": StepMetadata(user_id=user_id), } if sandbox_image: diff --git a/src/app/main.py b/src/app/main.py index a692cc9b0..583995f52 100644 --- a/src/app/main.py +++ b/src/app/main.py @@ -180,9 +180,7 @@ async def lifespan( # pylint: disable=too-many-branches,too-many-statements,imp "name": "Apache 2.0", "url": "https://www.apache.org/licenses/LICENSE-2.0.html", }, - servers=[ - {"url": "http://localhost:8080", "description": "Locally running service"} - ], + servers=[{"url": "/", "description": "Locally running service"}], openapi_tags=_OPENAPI_TAGS, lifespan=lifespan, ) diff --git a/src/models/api/requests/agents.py b/src/models/api/requests/agents.py index 10ec43576..1383634bc 100644 --- a/src/models/api/requests/agents.py +++ b/src/models/api/requests/agents.py @@ -19,6 +19,7 @@ class AgentRunRequest(BaseModel): sandbox_image: Container image for spawn=ephemeral. tools: Tool definitions. mcp_servers: MCP server names. + allowed_skills: Skill allowlist (names of skill subdirectories). output_schema: JSON Schema for structured output. context: Prior context (e.g. from previous steps). """ @@ -67,6 +68,15 @@ class AgentRunRequest(BaseModel): "Each server is connected via pydantic-ai MCPToolset.", ) + allowed_skills: Optional[list[str]] = Field( + None, + description="Skill allowlist: names of skill subdirectories the agent " + "may use. For spawn=ephemeral the spawner materializes just this " + "subset into the sandbox and Landlock-grants each /skills/ " + "path, so unlisted skills are denied at the filesystem boundary. " + "Omitted means no skills.", + ) + output_schema: Optional[dict[str, Any]] = Field( None, description="JSON Schema for structured output.", diff --git a/tests/unit/cloud_agents/test_agents_endpoint.py b/tests/unit/cloud_agents/test_agents_endpoint.py index 1351bdd8e..b70911d07 100644 --- a/tests/unit/cloud_agents/test_agents_endpoint.py +++ b/tests/unit/cloud_agents/test_agents_endpoint.py @@ -451,6 +451,98 @@ async def test_local_spawn_dispatches_without_spawner( call_args = mock_executor.run.call_args[0][0] assert "credentials_secret" not in call_args.provider + @pytest.mark.asyncio + async def test_allowed_skills_reach_step_input_and_raw_step( + self, + mocker: MockerFixture, + mock_executor: Any, + ) -> None: + """allowed_skills is threaded to StepInput and the raw step dict. + + DirectExecutor/SubprocessExecutor read step_input.allowed_skills; + SandboxExecutor only forwards step_input.raw_step to step_runner, + which is where spawner.spawn(allowed_skills=...) (per-skill + Landlock grants) comes from. Both must carry the allowlist or the + ephemeral path silently drops it. + """ + mocker.patch("app.endpoints.agents.check_configuration_loaded") + mock_cfg = mocker.patch("app.endpoints.agents.configuration") + spawner_config = mocker.MagicMock() + spawner_config.sandbox_image = "default-sandbox:latest" + mock_cfg.spawner_configuration = spawner_config + mock_cfg.inference.default_provider = None + mock_cfg.inference.default_model = None + mocker.patch( + "app.endpoints.agents.build_spawner", return_value=mocker.MagicMock() + ) + + body = AgentRunRequest( + prompt="Diagnose the pod", + spawn="ephemeral", + provider="openai", + model="gpt-4o-mini", + allowed_skills=["k8s-diag"], + ) + auth = ("user-1", "testuser", False, "token") + request = mocker.MagicMock() + + await run_agent_handler.__wrapped__(request, body, auth) + + call_args = mock_executor.run.call_args[0][0] + assert call_args.allowed_skills == ["k8s-diag"] + assert call_args.raw_step["allowed_skills"] == ["k8s-diag"] + assert call_args.raw_step["name"] == "agent-run" + + @pytest.mark.asyncio + async def test_raw_step_selects_request_mcp_servers_by_name( + self, + mocker: MockerFixture, + mock_config: Any, + mock_executor: Any, + ) -> None: + """raw_step lists request MCP server names for step_runner injection. + + step_runner only injects catalog entries (StepInput.mcp_servers) + whose names appear in step["mcp_servers"] -- without the name list, + LIGHTSPEED_MCP_SERVERS is never set and the sandbox agent has no + MCP tools to call. + """ + mocker.patch("app.endpoints.agents.check_configuration_loaded") + + body = AgentRunRequest( + prompt="Check the pod", + provider="openai", + model="gpt-4o-mini", + mcp_servers=[{"name": "pod-status", "url": "http://x/mcp"}], + ) + auth = ("user-1", "testuser", False, "token") + request = mocker.MagicMock() + + await run_agent_handler.__wrapped__(request, body, auth) + + call_args = mock_executor.run.call_args[0][0] + assert call_args.raw_step["mcp_servers"] == ["pod-status"] + + @pytest.mark.asyncio + async def test_allowed_skills_omitted_means_no_skills( + self, + mocker: MockerFixture, + mock_config: Any, + mock_executor: Any, + ) -> None: + """Omitted allowed_skills stays None (no skills), not empty filtering.""" + mocker.patch("app.endpoints.agents.check_configuration_loaded") + + body = AgentRunRequest(prompt="Hello", provider="openai", model="gpt-4o-mini") + auth = ("user-1", "testuser", False, "token") + request = mocker.MagicMock() + + await run_agent_handler.__wrapped__(request, body, auth) + + call_args = mock_executor.run.call_args[0][0] + assert call_args.allowed_skills is None + assert call_args.raw_step["allowed_skills"] is None + @pytest.mark.asyncio async def test_local_spawn_result_shape( self, From b8f22d9616554b6a4124b5dea08f599ebc3df753 Mon Sep 17 00:00:00 2001 From: James Wong <2421248+jameswnl@users.noreply.github.com> Date: Wed, 9 Sep 2026 05:12:45 -0400 Subject: [PATCH 2/2] Wire agent-none and agent-local demos to the same MCP pod-status check Both ran with mcp_servers: null, so unlike agent-ephemeral they had no way to actually check pod health -- agent-none returned a generic "I don't have access" answer and agent-local's claim was ungrounded. Point both at the same mock pod-status MCP server, reached via localhost:8084 (port-forwarded) since these spawn modes run in-process on this machine rather than inside a cluster pod like spawn:ephemeral. Co-Authored-By: Claude Sonnet 5 --- docs/cloud-agents-demo-curl.sh | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/docs/cloud-agents-demo-curl.sh b/docs/cloud-agents-demo-curl.sh index d16e23cb7..9e5437e59 100755 --- a/docs/cloud-agents-demo-curl.sh +++ b/docs/cloud-agents-demo-curl.sh @@ -23,6 +23,15 @@ # it can't reliably guarantee schema-conforming JSON the way spawn:none # and spawn:ephemeral can. # +# MCP server reachability: agent-none and agent-local run in-process on +# this machine, not inside the cluster, so they reach the mock pod-status +# MCP server (deployed via ~/ws/local-infra's ocp-prod-mcp-pod-status-* +# targets) at localhost:8084 -- port-forward it first: +# oc -n openshell-prod port-forward svc/mcp-pod-status-mock 8084:8084 +# agent-ephemeral runs inside an OpenShell sandbox pod on the cluster, so +# it reaches the same service via in-cluster DNS instead +# (mcp-pod-status-mock:8084), no port-forward needed. +# # Usage: # BASE_URL=http://localhost:8090 ./docs/cloud-agents-demo-curl.sh agent-none # BASE_URL=http://localhost:8090 ./docs/cloud-agents-demo-curl.sh agent-local @@ -84,7 +93,7 @@ agent_none() { "Is pod checkout-7f9 healthy?" \ '{ "tools": [], - "mcp_servers": null, + "mcp_servers": [{"name": "kubectl-mcp", "url": "http://localhost:8084/mcp"}], "output_schema": { "type": "object", "properties": { "healthy": {"type": "boolean"}, "reason": {"type": "string"} }, @@ -95,8 +104,8 @@ agent_none() { agent_local() { run_agent "== Agent — Subprocess (spawn: local) ==" "local" \ - "Say one sentence confirming pod checkout-7f9 is healthy." \ - '{"tools": [], "mcp_servers": null}' + "Check whether pod checkout-7f9 is healthy and say one sentence confirming the result." \ + '{"tools": [], "mcp_servers": [{"name": "kubectl-mcp", "url": "http://localhost:8084/mcp"}]}' } agent_ephemeral() {