Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
61 commits
Select commit Hold shift + click to select a range
0a8232b
Add bounded parallel E2E runner and repair pipeline recovery
guima-why Oct 5, 2026
1030ccd
Fix AGUI lead-in accounting and capture provider failure facts
guima-why Oct 5, 2026
1eac4b8
Use resolved network subnet in recovery question answers
guima-why Oct 5, 2026
357de4c
Expose safe pre-selector permission diagnostics
guima-why Oct 5, 2026
ad90764
Capture precise schema, adjustment preview and quote guard diagnostics
guima-why Oct 5, 2026
053ea5b
Answer native REPL questions while awaiting pipeline completion
guima-why Oct 5, 2026
7d03921
Preserve native stack polling status and REPL candidate handoff trace
guima-why Oct 5, 2026
98c4879
fix(pipeline): retain schema failure coordinates for model repair
guima-why Oct 5, 2026
d8c71ae
fix(e2e): check actual Step 1 materialization calls
guima-why Oct 5, 2026
4da7995
fix(e2e): wait for native multimodal question checkpoints
guima-why Oct 5, 2026
57d524c
fix(e2e): validate advisory question selections before use
guima-why Oct 5, 2026
9ee9cf8
fix(e2e): fail promptly on deployment replanning and retain failure f…
guima-why Oct 5, 2026
4e935a6
fix(e2e): allow scoped local materialization in AGUI selector flow
guima-why Oct 5, 2026
28da655
test(e2e): retain safe instance geometry and constraint evidence diag…
guima-why Oct 6, 2026
7c2dbd0
fix(a2a): tolerate null bytes in public path projection text
guima-why Oct 6, 2026
60bf882
test(e2e): retain golden source and rollback stream failure evidence
guima-why Oct 6, 2026
de4c1d9
fix(e2e): answer planning handoffs before rollback Step 2 fault
guima-why Oct 6, 2026
a98a020
fix(e2e): handle clarification and image question input boundaries
guima-why Oct 6, 2026
74da4c3
fix(e2e): retain native question boundary during image interruption
guima-why Oct 6, 2026
b1e3c75
fix(e2e): submit image paths without controls in active questions
guima-why Oct 6, 2026
fa90c1c
test(e2e): retain native evidence for target, quote and confirmation …
guima-why Oct 6, 2026
480c31e
fix(e2e): report native reconfirmation instead of waiting for completion
guima-why Oct 6, 2026
7e69ac8
fix(e2e): parse product handoff JSON before safety instructions
guima-why Oct 6, 2026
55dc9bf
test(e2e): correlate native confirmation failures with completion guards
guima-why Oct 6, 2026
e8e7c2f
test(e2e): classify localized native confirmation guard failures
guima-why Oct 6, 2026
7f8141e
fix(e2e): refresh fixture ownership before publishing network facts
guima-why Oct 6, 2026
ad589cb
fix(e2e): distinguish unspecified restrictions from missing question …
guima-why Oct 6, 2026
88e2ece
fix(e2e): select only available network fixtures predating the run
guima-why Oct 6, 2026
4992e73
fix: read coherent OAuth refresh state and control projection fixture…
guima-why Oct 6, 2026
0dd1725
fix: bound Windows subnet reservation lock acquisition retries
guima-why Oct 6, 2026
1058cb0
fix(e2e): honor deferred identities in planning preference answers
guima-why Oct 6, 2026
2fefe7a
test(skill): check manager reuse before enabling idle expiry
guima-why Oct 6, 2026
c123568
fix(e2e): use external input for image parameter questions
guima-why Oct 6, 2026
cfce304
fix(e2e): answer current cloud question and retain native failure dia…
guima-why Oct 6, 2026
e04eae3
test(e2e): write AGUI diagnostic fixture with explicit UTF-8
guima-why Oct 6, 2026
16575f9
fix: observe accepted AGUI pipeline continuation and answer current p…
guima-why Oct 6, 2026
d4f682f
test: cancel permission snapshot batch at the actual tool boundary
guima-why Oct 6, 2026
ed28fb0
fix: observe resumed pipeline past stale input status
guima-why Oct 6, 2026
0918c50
test: synchronize cancellation with runtime factory lifecycle
guima-why Oct 6, 2026
aa11ba0
test: inspect actual VPC references on failed ROS deployment
guima-why Oct 6, 2026
b59aa14
fix(pipeline): retry malformed interrupt verdict within existing budget
guima-why Oct 6, 2026
916f6d7
fix(ci): retain bounded CIDR comparison hashes in live reports
guima-why Oct 6, 2026
47ed020
fix(e2e): follow fresh REPL selection and inspect deployed resource t…
guima-why Oct 7, 2026
1afc87a
ci: bound dependency download concurrency and tolerate mirror latency
guima-why Oct 7, 2026
773b3da
ci: preserve the Desktop workflow scope boundary
guima-why Oct 7, 2026
4abb1a0
test: retain safe A2A smoke task diagnostics on failure
guima-why Oct 7, 2026
a8c322c
fix(e2e): follow fresh selection before REPL rollback confirmation
guima-why Oct 7, 2026
15aa8e3
fix(pipeline): preserve explicit candidate counts during planning
guima-why Oct 7, 2026
e06dc05
test: synchronize duplicate permission registration at the grace boun…
guima-why Oct 7, 2026
4d39a19
fix(e2e): handle recovery questions and clarify local permission policy
guima-why Oct 7, 2026
44c3294
fix(e2e): verify native recovery boundaries and deployed resource types
guima-why Oct 7, 2026
9d95a58
fix(e2e): acknowledge initial question answers before handling redraws
guima-why Oct 7, 2026
8a0e822
test(web): synchronize runtime nonblocking checks with actual lifecycle
guima-why Oct 7, 2026
02e3c89
fix(e2e): distinguish recovered preview errors from ROS creation fail…
guima-why Oct 7, 2026
e768510
fix(e2e): retain safe ROS creation failure counters in CI reports
guima-why Oct 7, 2026
fb94608
fix(e2e): retain safe AGUI protocol milestones when a resume fails
guima-why Oct 7, 2026
a940792
fix(e2e): define unambiguous network fixtures for adjustment and cleanup
guima-why Oct 7, 2026
9ea83a5
fix(e2e): retain failed ROS quote resource error categories
guima-why Oct 7, 2026
fa509f3
fix(pipeline): compare CIDR containment by network membership
guima-why Oct 7, 2026
920b784
test(a2a): isolate resume commit ordering from disk latency
guima-why Oct 7, 2026
6d8c298
test(e2e): diagnose referenced VPC and zone on failed deployment
guima-why Oct 7, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions .github/workflows/test.yml
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,11 @@ concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: true

env:
UV_CONCURRENT_DOWNLOADS: "8"
UV_HTTP_TIMEOUT: "60"
UV_HTTP_RETRIES: "5"

jobs:
skill-bridge-compatibility:
name: Skill bridge (Python ${{ matrix.python-version }})
Expand Down
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -189,5 +189,6 @@ src/iac_code/pipeline/engine/architecture_rules.json.backup-*
# Super Powers
.superpowers/
.worktrees/
ci-e2e-report/
.agents/
spec/
2 changes: 1 addition & 1 deletion scripts/a2a/e2e/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -424,7 +424,7 @@ the rest of the tests.
| `image-ask-waiting` | `ask_user_question` waits for user input, then the server restarts | Static `ask-first-answer.png` / `ask-second-answer.png` image fixtures without `taskId` | Pending ask input is recovered, image answers hydrate the recovered task, and the pipeline completes with VSwitch evidence. |
| `image-selection-waiting` | Step 4 waits for candidate selection, then the server restarts | Static `selection.png` image fixture without `taskId` | Waiting step4 task is recovered, the image selection is accepted, and VSwitch evidence exists. |
| `image-normal-handoff` | Pipeline completes and hands off to normal chat; the normal follow-up is static `normal-followup.png`, then the server restarts | Normal-chat recovery question without `taskId` | Image follow-up stays in the same `contextId`, uses a new normal-chat task, and completed handoff state survives restart. |
| `image-interrupt` | Step 3 receives static `rollback-interrupt.png` as an image rollback to `intent_parsing`, then the server restarts | `继续`, plus selection when needed | The image interrupt is recognized, the pipeline completes as a security-group task, and final deployment evidence is not VSwitch. |
| `image-interrupt` | Step 3 receives `rollback-interrupt.png` with an image caption explicitly requesting a restart from `intent_parsing`; kill the server only after that new attempt starts | `继续`, plus selection when needed | The image interrupt is recognized, the pipeline completes as a security-group task, and final deployment evidence is not VSwitch. |
| `step1-running` | `intent_parsing` running | `继续` | Running pipeline task is recovered and completes; VSwitch evidence exists. |
| `step2-running` | `architecture_planning` running | `继续` | Running pipeline task is recovered and completes; VSwitch evidence exists. |
| `step3-running` | `evaluate_candidates` candidate/sub-pipeline running | `继续` | Sub-pipeline state is recovered and completes; VSwitch evidence exists. |
Expand Down
2 changes: 1 addition & 1 deletion scripts/a2a/e2e/README.zh-CN.md
Original file line number Diff line number Diff line change
Expand Up @@ -490,7 +490,7 @@ provider、tool、真实云调用场景默认会被保护住。只有确认要
| `image-ask-waiting` | `ask_user_question` 等待用户输入,随后重启 server | 不带 `taskId` 发送静态 `ask-first-answer.png` / `ask-second-answer.png` 图片 fixture | pending ask 输入能恢复,图片回答能 hydrate 到恢复后的 task,最终完成并产生 VSwitch 证据。 |
| `image-selection-waiting` | step4 等待候选方案选择,随后重启 server | 不带 `taskId` 发送静态 `selection.png` 图片 fixture | 能恢复等待中的 step4 task,图片选择被接受,并产生 VSwitch 证据。 |
| `image-normal-handoff` | pipeline 完成并 handoff 到 normal chat;normal follow-up 是静态 `normal-followup.png`,随后重启 server | 不带 `taskId` 发送 normal-chat 恢复问题 | 图片 follow-up 保持同一个 `contextId`,使用新的 normal-chat task;completed handoff 状态重启后仍可恢复。 |
| `image-interrupt` | step3 收到静态 `rollback-interrupt.png` 图片,表示回滚到 `intent_parsing`,随后重启 server | `继续`,必要时再选择方案 | 图片 interrupt 能被识别;pipeline 以安全组任务完成,最终部署证据不是 VSwitch。 |
| `image-interrupt` | step3 收到 `rollback-interrupt.png`,图片附注明确请求从 `intent_parsing` 重新解析;只有该新尝试开始后才强杀并重启 server | `继续`,必要时再选择方案 | 图片 interrupt 能被识别;pipeline 以安全组任务完成,最终部署证据不是 VSwitch。 |
| `step1-running` | `intent_parsing` 运行中 | `继续` | running pipeline task 能恢复并完成;存在 VSwitch 证据。 |
| `step2-running` | `architecture_planning` 运行中 | `继续` | running pipeline task 能恢复并完成;存在 VSwitch 证据。 |
| `step3-running` | `evaluate_candidates` 的 candidate/sub-pipeline 运行中 | `继续` | sub-pipeline 状态能恢复并完成;存在 VSwitch 证据。 |
Expand Down
184 changes: 184 additions & 0 deletions scripts/a2a/e2e/cleanup_owned_stacks.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,184 @@
#!/usr/bin/env python3
"""Delete only ROS stacks with an accepted creation receipt in this E2E case."""

from __future__ import annotations

import argparse
import json
import re
import time
from pathlib import Path
from typing import Any


class CleanupOperationError(RuntimeError):
"""Expose a fixed cleanup stage without leaking the cloud error message."""

def __init__(self, stage: str, cause: Exception) -> None:
self.stage = stage
self.cause_type = type(cause).__name__
code = getattr(cause, "code", None)
self.sdk_code = code if isinstance(code, str) and re.fullmatch(r"[A-Za-z][A-Za-z0-9_.-]{0,79}", code) else ""
super().__init__(f"{stage}: {self.cause_type}")


def _record_cleanup_failure(run_dir: Path, stage: str, exc: Exception) -> None:
known_codes = {
"EntityNotExist.Stack", "NotFound.Stack", "StackNotFound", "ActionInProgress",
"Forbidden", "Forbidden.RAM", "InvalidAccessKeyId.NotFound", "SecurityTokenExpired",
"Throttling", "Throttling.User", "InvalidParameter", "DeleteFailed", "DependencyViolation",
}
code = getattr(exc, "code", None)
diagnostic = {"stage": stage, "errorType": "SDKError",
"code": code if isinstance(code, str) and code in known_codes else "unknown"}
with (run_dir / "cleanup-cloud.log").open("a", encoding="utf-8") as stream:
stream.write(json.dumps({"cleanupDiagnostic": diagnostic}) + "\n")


def _stack_body(client: Any, models: Any, stack_id: str, region: str) -> dict[str, Any] | None:
try:
return client.get_stack(models.GetStackRequest(stack_id=stack_id, region_id=region)).body.to_map()
except Exception as exc:
code = getattr(exc, "code", None)
missing_codes = {"EntityNotExist.Stack", "NotFound.Stack", "StackNotFound"}
if isinstance(code, str):
if code in missing_codes:
return None
elif any(re.search(r"\b" + re.escape(marker) + r"\b", str(exc), re.I) for marker in missing_codes):
return None
raise


def cleanup_owned_stacks(run_dir: Path, *, timeout: float = 840) -> dict[str, Any]:
from iac_code.services.session_storage import SessionStorage
from scripts.ci.stack_ownership import creation_receipts

manifest = json.loads((run_dir / "owned-stacks.json").read_text(encoding="utf-8"))
if not isinstance(manifest.get("configDir"), str) or not isinstance(manifest.get("cwd"), str):
raise ValueError("invalid E2E ownership manifest")
storage = SessionStorage(projects_dir=Path(manifest["configDir"]) / "projects")
directories: list[Path] = []
for context_path in sorted((run_dir / "a2a-persistence" / "contexts").glob("*.json")):
context = json.loads(context_path.read_text(encoding="utf-8"))
session_id = context.get("session_id")
if (not isinstance(session_id, str) or re.fullmatch(r"[A-Za-z0-9_-]{1,128}", session_id) is None
or context.get("cwd") != manifest["cwd"]):
raise ValueError("A2A context does not prove this case's session ownership")
session = storage.session_dir(manifest["cwd"], session_id)
if not session.resolve().is_relative_to((Path(manifest["configDir"]) / "projects").resolve()):
raise ValueError("Stack ownership evidence escaped isolated case")
directories.extend([session / "pipeline", session / "a2a" / "pipeline"])
resources = creation_receipts(directories)
from alibabacloud_ros20190910 import models as ros_models

from iac_code.services.cloud_credentials import CloudCredentials
from iac_code.tools.cloud.aliyun.ros_client import RosClientFactory

try:
credential = CloudCredentials().get_provider("aliyun")
except Exception as exc:
raise CleanupOperationError("credential_lookup", exc) from exc
if credential is None:
raise RuntimeError("Aliyun credential is unavailable for E2E teardown")
region = str(manifest.get("regionId") or credential.region_id)
try:
client = RosClientFactory.create(credential, region)
except Exception as exc:
raise CleanupOperationError("client_create", exc) from exc
deadline = time.monotonic() + timeout
deleted: list[str] = []
remaining: list[str] = []
failures: list[str] = []
for resource in resources:
stack_id, name, region = resource["stackId"], resource["stackName"], resource["regionId"]
client = RosClientFactory.create(credential, region)
delete_submitted = False
while time.monotonic() < deadline:
stage = "get_stack"
try:
body = _stack_body(client, ros_models, stack_id, region)
if body is None or body.get("Status") == "DELETE_COMPLETE":
deleted.append(stack_id)
break
if body.get("StackName") != name or body.get("ParentStackId") or body.get("ServiceManaged"):
failures.append(stack_id + ": Stack identity differs from accepted creation receipt")
break
status = str(body.get("Status") or "")
if status == "DELETE_FAILED" and delete_submitted:
failures.append(stack_id + ": accepted deletion failed")
break
if not status.endswith("_IN_PROGRESS") and not delete_submitted:
stage = "delete_stack"
try:
client.delete_stack(ros_models.DeleteStackRequest(stack_id=stack_id, region_id=region))
delete_submitted = True
except Exception as exc:
if getattr(exc, "code", "") in {"EntityNotExist.Stack", "NotFound.Stack", "StackNotFound"}:
deleted.append(stack_id)
break
if getattr(exc, "code", "") != "ActionInProgress":
raise
time.sleep(min(5, max(0, deadline - time.monotonic())))
except Exception as exc:
_record_cleanup_failure(run_dir, stage, exc)
failures.append(stack_id + ": " + type(exc).__name__)
break
else:
failures.append(stack_id + ": cleanup timeout")
if stack_id not in deleted:
remaining.append(stack_id)
# Real pipeline Stack events can reveal an unproven leak. Read those exact
# IDs without granting deletion authority to observations alone.
from scripts.a2a.debugger import _extract_pipeline_envelopes

observed_ids: set[str] = set()
for path in sorted(run_dir.glob("*.events.jsonl"))[:60]:
if path.stat().st_size > 20_000_000:
raise RuntimeError("observed Stack audit evidence exceeds bounded size")
for line in path.read_text(encoding="utf-8").splitlines():
try:
row = json.loads(line)
except ValueError:
continue
for envelope in _extract_pipeline_envelopes(row):
data = envelope.get("data")
if envelope.get("eventType") != "stack_current_changed" or not isinstance(data, dict):
continue
stack_id = data.get("stackId")
if isinstance(stack_id, str) and stack_id:
observed_ids.add(stack_id)
if len(observed_ids) > 60:
raise RuntimeError("observed Stack audit exceeds bounded count")
try:
for stack_id in sorted(observed_ids - set(deleted) - set(remaining)):
body = _stack_body(client, ros_models, stack_id, region)
if body is not None and body.get("Status") != "DELETE_COMPLETE":
failures.append("observed Stack has no accepted creation receipt or cleanup incomplete")
remaining.append(stack_id)
except Exception as exc:
raise CleanupOperationError("observed_stack_audit", exc) from exc
result = {
"status": "failed" if failures or remaining else "completed",
"deletedStackIds": deleted,
"remainingStackIds": remaining,
"failures": failures,
"resources": resources,
}
(run_dir / "cleanup-result.json").write_text(
json.dumps(result, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
)
return result


def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--run-dir", type=Path, required=True)
parser.add_argument("--timeout", type=float, default=840)
args = parser.parse_args()
result = cleanup_owned_stacks(args.run_dir, timeout=args.timeout)
print(json.dumps(result, ensure_ascii=False))
return 0 if result["status"] == "completed" else 1


if __name__ == "__main__":
raise SystemExit(main())
Loading
Loading