diff --git a/docs/architecture/rfcs/external-evidence-research-capability-v0.md b/docs/architecture/rfcs/external-evidence-research-capability-v0.md index c0046e5783..e1b5412c8f 100644 --- a/docs/architecture/rfcs/external-evidence-research-capability-v0.md +++ b/docs/architecture/rfcs/external-evidence-research-capability-v0.md @@ -78,11 +78,11 @@ inventory-only row for the same provider id. ## Product surfaces -- CLI: `external-evidence discover|plan|receipt|admit|retire`. -- Managed Turn: the same five effect-runtime methods. -- Frontend/Lark: not changed in this Core slice. A companion slice should render - the same typed plan/admission projection and readback; it must not invent a - second registry or lifecycle. +- CLI: `external-evidence discover|plan|execute|receipt|admit|readback|retire`. +- Managed Turn: the same five typed effect-runtime methods; explicit provider + execution and ledger projection use the capability's CLI owner. +- Frontend/Lark: existing conversation answer/report and Markdown transports + render the shared validated readback. No independent registry or lifecycle. ## Acceptance @@ -102,6 +102,29 @@ inventory-only row for the same provider id. - retirement waits for downstream coverage of every admitted source; - CLI and effect-runtime TypeScript tests pass from the source checkout. +## Delivery checkpoint (2026-10-02) + +The public GitHub method now completes a bounded real journey: anonymous pinned +file reads, exact-plan receipt validation, a separate parent decision, projection +into the existing deepresearch source ledger, actual lineage readback and retirement. +Optional source refs and literal search terms are bound into the request/plan digest; +legacy requests retain their existing identity. The provider is bundled in extensions +under `method:public-github`; capability and ledger owners remain unchanged. + +Passed: real public-provider/source CLI journey; negative cases for private or stale +readiness, malformed/unpinned sources, plan/admission mutation, partial/empty/failed +reads, independent admission and coverage, wrong-question projection, budget failure +and idempotent replay; packaged desktop/mobile conversation readback and reload; +existing Lark Markdown presentation. Source bodies are not persisted. The shared +Markdown readback uses existing answer/report and Lark transports; no frontend +configuration or parallel evidence authority is needed. + +Commands are in the [versioned capability guide](../../../loopx/capabilities/external_research/README.md#public-github-method--公开-github-方法). +Live Lark delivery, authenticated connector execution and broader semantic research +quality remain untested; this checkpoint does not promote those providers or close +S6/S8. Failed or partial results preserve original-source fallback, and neither a +successful read nor a parent admission certifies evidence completeness. + ## Non-goals - a universal browser/search engine; diff --git a/docs/architecture/rfcs/external-evidence-research-capability-v0.zh-CN.md b/docs/architecture/rfcs/external-evidence-research-capability-v0.zh-CN.md index 898efaee2f..d939db48c7 100644 --- a/docs/architecture/rfcs/external-evidence-research-capability-v0.zh-CN.md +++ b/docs/architecture/rfcs/external-evidence-research-capability-v0.zh-CN.md @@ -60,10 +60,11 @@ Connector registry 继续只拥有库存与遥测。`supported` 绝不映射为 ## 产品入口 -- CLI:`external-evidence discover|plan|receipt|admit|retire`; -- Managed Turn:复用同五个 effect-runtime 方法; -- Frontend/Lark:本 Core 切片不修改。后续 companion slice 只渲染同源 plan/admission - 投影与读回,不建立第二个 registry 或生命周期。 +- CLI:`external-evidence discover|plan|execute|receipt|admit|readback|retire`; +- Managed Turn:复用五个 typed effect-runtime 方法;显式 provider 执行与账本投影 + 使用能力的 CLI owner; +- Frontend/Lark:现有会话答复/报告和 Markdown 运输渲染同源校验回读, + 不建立独立 registry 或生命周期。 ## 验收 @@ -80,6 +81,24 @@ Connector registry 继续只拥有库存与遥测。`supported` 绝不映射为 - 全部被采纳来源完成下游覆盖前不得退休; - CLI 与 effect-runtime TypeScript 测试在源码 checkout 中通过。 +## 交付检查点(2026-10-02) + +公开 GitHub method 已完成有界真实链路:匿名读取固定提交文件、精确 plan 回执校验、 +独立父 Agent 决定、投影到现有 deepresearch 来源账本、实际 lineage 回读与退休。 +可选 source refs 和字面检索词进入 request/plan digest;旧请求身份保持兼容。 +provider 以 `method:public-github` 内置在 extensions,能力和账本 owner 不变。 + +通过:真实公开 provider/源码 CLI 链路;私有或过期 readiness、无效/未固定来源、 +plan/admission 篡改、部分/空/失败读取、独立采纳与覆盖、问题不匹配、预算耗尽及 +幂等重放等负向用例;打包桌面/移动会话回读与重载;现有 Lark Markdown 展示。 +不持久化来源正文。同源 Markdown 沿用现有答复/报告和 Lark 运输路径, +无需新增前端配置或并行证据权威。 + +命令参见[版本化能力指南](../../../loopx/capabilities/external_research/README.md#public-github-method--公开-github-方法)。 +真实 Lark 送达、带凭据 connector 执行和更广泛语义研究质量尚未验证; +该检查点不晋升这些 provider,也不关闭 S6/S8。失败或部分结果保留原始来源退路; +读取成功和父 Agent 采纳均不证明证据完整性。 + ## 非目标 - 通用浏览器或搜索引擎; diff --git a/examples/personal-workspace-browser-smoke.mjs b/examples/personal-workspace-browser-smoke.mjs index 3f6291729b..44372c25af 100644 --- a/examples/personal-workspace-browser-smoke.mjs +++ b/examples/personal-workspace-browser-smoke.mjs @@ -72,6 +72,8 @@ scenarioCatalog.push(turnStepsScenario); scenarioCatalog.push(goalWorkMapScenario); scenarioCatalog.push(performanceDiagnosisScenario); scenarioCatalog.push(blockedNoticeSettingsScenario); +import { externalEvidenceReadbackScenario } from "./personal-workspace-browser/external-evidence-readback.mjs"; +scenarioCatalog.push(externalEvidenceReadbackScenario); const requestedScenario = process.env.LOOPX_PERSONAL_WORKSPACE_SCENARIO; const scenarios = requestedScenario ? scenarioCatalog.filter((scenario) => scenario.id === requestedScenario) diff --git a/examples/personal-workspace-browser/external-evidence-readback.mjs b/examples/personal-workspace-browser/external-evidence-readback.mjs new file mode 100644 index 0000000000..ed2ec67cdf --- /dev/null +++ b/examples/personal-workspace-browser/external-evidence-readback.mjs @@ -0,0 +1,57 @@ +import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; +import { resolve } from "node:path"; +import { resolveTestPython } from "../../scripts/test-python.mjs"; +import { outputDir, repoRoot } from "./fixture.mjs"; +import { openWorkspacePage } from "./scenario-context.mjs"; + +export const externalEvidenceReadbackScenario = { + id: "external-evidence-readback", + async run({browser, collectCoverage, url}) { + // Exercise the product's typed plan/admission and actual downstream ledger, + // then hand the resulting shared readback to the existing answer surface. + const result = spawnSync(resolveTestPython(), ["-X", "utf8", "-c", ` +import tempfile +from pathlib import Path +from loopx.control_plane.effect_runtime import effect_runtime_result as effect +from loopx.capabilities.deep_research.runtime import start_research +from loopx.capabilities.external_research.projection import readback, render_readback +provider = {"provider_id":"method:public-github", "provider_kind":"method", "protocol":"external_evidence_research_v0", "declared":True, "installed":True, "enabled":True, "ready":True, "unavailable_reason":None} +ref = "https://github.com/example/public/blob/" + "a"*40 + "/README.md" +plan = effect("external_evidence.plan", {"request":{"objective":"Inspect public fixture", "user_activity":"Choose a source", "decision":"Whether to use the fixture", "evidence_kinds":["literal_match"], "source_refs":[ref]}, "providers":[provider]}) +receipt = {"schema_version":"loopx_external_evidence_receipt_v0", "plan_id":plan["plan_id"], "request_id":plan["request"]["request_id"], "provider_id":provider["provider_id"], "provider_kind":"method", "status":"succeeded", "summary":"Read pinned public fixture", "completed_at":"2026-10-02T00:00:00Z", "sources":[{"source_ref":ref, "source_family":"github_repository_file", "basis":"observed", "finding":"Literal fixture marker observed at line 1", "content_digest":"sha256:"+"b"*64, "accessed_at":"2026-10-02T00:00:00Z", "limitation":"Retrieval only; completeness unverified"}]} +admission = effect("external_evidence.admit", {"plan":plan, "receipt":receipt, "decision":{"disposition":"admit", "reason":"Synthetic parent checked the direct source", "admitted_source_refs":[ref]}}) +with tempfile.TemporaryDirectory(prefix="lxe-ui-") as folder: + project=Path(folder) + start_research(project, question=plan["request"]["objective"], max_sources=8, max_subquestions=4) + print(render_readback(readback(plan, receipt, admission, project=project, execute=True))) +`], {cwd:repoRoot, encoding:"utf8", env:{...process.env, PYTHONPATH:repoRoot}, timeout:45000}); + assert.equal(result.status, 0, result.stderr); + const markdown = result.stdout; + const context = await openWorkspacePage(browser, url, {collectCoverage}); + const {api, page} = context; + try { + api.answerForMessage = () => markdown; + await page.getByRole("navigation", {name:"管家视图"}).getByRole("button", {name:/^(Chat|对话)$/}).click(); + await page.getByLabel("向 LoopX 发送消息").fill("Show the external evidence readback"); + await page.getByRole("button", {name:"发送",exact:true}).click(); + const answer = page.locator(".personal-channel-timeline .personal-message.is-assistant", {hasText:"External evidence"}); + await answer.waitFor(); + assert.match(await answer.innerText(), /retire_ready/u); + assert.match(await answer.innerText(), /Downstream coverage.*1 admitted/u); + assert.match(await answer.innerText(), /Evidence completeness is unverified/u); + await answer.scrollIntoViewIfNeeded(); + await page.screenshot({path:resolve(outputDir,"external-evidence-desktop.png"),fullPage:false,animations:"disabled"}); + await page.setViewportSize({width:390,height:844}); + await answer.scrollIntoViewIfNeeded(); + assert.equal(await answer.evaluate(el => el.scrollWidth <= el.clientWidth + 1), true); + await page.screenshot({path:resolve(outputDir,"external-evidence-mobile.png"),fullPage:false,animations:"disabled"}); + await page.reload({waitUntil:"networkidle"}); + await page.getByRole("navigation", {name:"管家视图"}).getByRole("button", {name:/^(Chat|对话)$/}).click(); + await answer.waitFor(); + assert.match(await answer.innerText(), /retire_ready/u); + assert.equal(context.errors.length, 0); + return {coverageEntries:context.coverageEntries, note:"Shared typed evidence readback survives conversation reload and fits desktop/mobile."}; + } finally { await context.close(); } + }, +}; diff --git a/examples/public-github-evidence-live-smoke.py b/examples/public-github-evidence-live-smoke.py new file mode 100644 index 0000000000..f3587892e5 --- /dev/null +++ b/examples/public-github-evidence-live-smoke.py @@ -0,0 +1,76 @@ +#!/usr/bin/env python3 +"""Opt-in public GitHub exact-plan journey through the shipped source CLI. + +Only anonymous public GETs and a disposable synthetic research ledger are used. +The explicit synthetic parent decision is separate from provider execution. +""" +from __future__ import annotations + +import argparse +import json +import subprocess +import sys +import tempfile +from pathlib import Path + + +def qualify(source: str, root: Path) -> dict: + def run(*args, json_output=True): + result = subprocess.run([sys.executable, "-X", "utf8", "-m", "loopx.entrypoint", + "--runtime-root", str(root / "runtime"), "--registry", str(root / "registry.json"), + *args, "--format", "json" if json_output else "markdown"], capture_output=True, + text=True, encoding="utf-8", timeout=90, check=True) + return json.loads(result.stdout) if json_output else result.stdout + + def save(name, value): + path = root / name + path.write_text(json.dumps(value), encoding="utf-8") + return str(path) + + objective = "Inspect the public pinned README for the LoopX literal" + plan = run("external-evidence", "plan", "--objective", objective, + "--user-activity", "Choose a public source", "--decision", "Whether the source contains LoopX", + "--evidence-kind", "literal_match", "--public-github", "--source", source, "--search-term", "LoopX") + assert plan["status"] == "ready" + plan_path = save("plan.json", plan) + execution = run("external-evidence", "execute", "--plan-json", plan_path, "--execute") + receipt = execution["receipt"] + assert receipt["status"] == "succeeded" and len(receipt["sources"]) == 1 + assert "observed at lines" in receipt["sources"][0]["finding"] + receipt_path = save("receipt.json", execution) + before = run("external-evidence", "readback", "--plan-json", plan_path, "--receipt-json", receipt_path) + assert before["parent_admission"] is None and before["retirement"] is None + admitted = run("external-evidence", "admit", "--plan-json", plan_path, "--receipt-json", receipt_path, + "--decision", "admit", "--reason", "Synthetic parent checked the pinned file and literal-match finding", + "--admit-source", source) + admission_path = save("admission.json", admitted) + project = root / "research" + run("deepresearch", "start", "--project", str(project), "--question", objective) + args = ("external-evidence", "readback", "--plan-json", plan_path, "--receipt-json", receipt_path, + "--admission-json", admission_path, "--project", str(project)) + assert run(*args)["retirement"]["status"] == "retained" + result = run(*args, "--execute") + assert result["downstream_source_refs"] == [source] + assert result["retirement"]["status"] == "retire_ready" + assert run(*args, "--execute") == result + markdown = run(*args, json_output=False) + assert source in markdown and "retire_ready" in markdown and "Evidence completeness is unverified" in markdown + return {"ok": True, "provider": "method:public-github", "sources_observed": 1, + "explicit_parent_admission": True, "actual_ledger_readback": True, + "retained_before_projection": True, "idempotent_projection": True, + "raw_content_persisted": False, "source_ref": source} + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--execute-public-provider", action="store_true") + parser.add_argument("--source", required=True) + args = parser.parse_args() + if not args.execute_public_provider: + parser.error("--execute-public-provider is required for anonymous public GETs") + with tempfile.TemporaryDirectory(prefix="lxe-") as folder: + print(json.dumps(qualify(args.source, Path(folder)), sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/loopx/capabilities/deep_research/runtime.py b/loopx/capabilities/deep_research/runtime.py index bbabbaed81..0abf998ea2 100644 --- a/loopx/capabilities/deep_research/runtime.py +++ b/loopx/capabilities/deep_research/runtime.py @@ -13,6 +13,7 @@ from pathlib import Path from typing import Any +from ...control_plane.content_digest import ENVELOPED_SHA256_PATTERN from ...file_lock import exclusive_file_lock COMMAND = "/loopx-deepresearch" @@ -238,6 +239,8 @@ def add_source( tool: str, title: str | None, claims: list[dict[str, Any]], + external_evidence: dict[str, str] | None = None, + expected_question: str | None = None, ) -> dict[str, Any]: # Load, validate the whole batch, allocate ids, mutate, and save all under # one project-level lock: concurrent deepresearch commands are a normal @@ -245,13 +248,31 @@ def add_source( # when the atomic rename itself succeeds. with exclusive_file_lock(state_path(project), operation="deepresearch_add_source"): state = _require_active_state(project) + if expected_question is not None and state["question"] != expected_question: + raise ValueError("external evidence objective does not match the active research question") + if external_evidence is not None: + fields = {"plan_id", "admission_id", "receipt_digest", "content_digest"} + if set(external_evidence) != fields or any( + not isinstance(value, str) or not ENVELOPED_SHA256_PATTERN.fullmatch(value) + for value in external_evidence.values() + ): + raise ValueError("external evidence lineage requires exact content-addressed identities") url_or_path = url_or_path.strip() tool = tool.strip() or "unspecified" if not url_or_path: raise ValueError("--url-or-path must be non-empty") normalized = _normalize_source_ref(url_or_path) for source in state["sources"]: - if _normalize_source_ref(str(source["url_or_path"])) == normalized: + same_source = ( + str(source["url_or_path"]).strip().rstrip("/") == url_or_path.rstrip("/") + if external_evidence is not None else + _normalize_source_ref(str(source["url_or_path"])) == normalized + ) + if same_source: + if (external_evidence is not None and source.get("external_evidence") == external_evidence + and [claim["text"] for claim in state["claims"] if claim["id"] in source["claims"]] + == [str(claim.get("text", "")).strip() for claim in claims]): + return {"source_id": source["id"], "claim_ids": source["claims"], "state": state} raise ValueError( f"source already recorded as {source['id']} " f"({source['url_or_path']}); reuse its claims instead of re-reading" @@ -325,6 +346,7 @@ def add_source( "title": (title or "").strip() or None, "accessed_at": _now_iso(), "claims": claim_ids, + **({"external_evidence": dict(external_evidence)} if external_evidence is not None else {}), } ) _save_state(project, state) diff --git a/loopx/capabilities/external_research/README.md b/loopx/capabilities/external_research/README.md index ef3874156e..3535d58a41 100644 --- a/loopx/capabilities/external_research/README.md +++ b/loopx/capabilities/external_research/README.md @@ -106,10 +106,73 @@ credentials, and private notes remain provider-private. `external_evidence.discover`, `external_evidence.plan`, `external_evidence.receipt`, `external_evidence.admit`, and `external_evidence.retire`. -- Frontend and Lark are companion slices. They should render the same plan and - admission projection; neither gets an independent provider registry or - evidence state machine. +- CLI `readback` renders the same validated plan, source, parent decision, + actual deepresearch-ledger coverage and retirement as Markdown. Existing + conversation answer/report and Lark Markdown transports consume that output; + no new UI configuration or evidence state machine is introduced. TypeScript 是 discovery 真值边界、请求身份、provider 准入、provenance 校验、父 Agent 采纳、紧凑投影与 退休条件的唯一语义 owner。Python 仅适配 CLI 与 effect-runtime transport。Managed -Turn 复用同一方法;frontend/Lark 后续只渲染同源投影,不新建 registry 或状态机。 +Turn 复用同一方法;CLI `readback` 的同源 Markdown 可由现有会话答复/报告和 Lark Markdown 运输路径展示,不新建 registry 或状态机。 + +## Public GitHub method / 公开 GitHub 方法 + +The bundled `method:public-github` provider performs anonymous, bounded HTTPS +GETs only. It reads UTF-8 files from explicitly selected full commit SHA URLs. +No token, cookie, private repository, branch-head URL, redirect, proxy credential, +raw page persistence or automatic admission is used. Repository public visibility +is checked at planning and again for each execution read. A saved ready row alone +cannot authorize or prove a successful read. + +该内置 method 只做匿名、有界 HTTPS GET,只接受显式选择的完整 commit SHA 文件 URL。 +规划和执行均检查仓库当前公开性;不使用 token、cookie、私有仓库、分支 head、重定向、 +代理凭据、原文持久化或自动采纳。保留的 ready 行本身不证明执行成功。 + +```bash +loopx external-evidence plan --public-github \ + --objective "Inspect public source" --user-activity "Choose a source" \ + --decision "Whether a literal is present" --evidence-kind literal_match \ + --source https://github.com/OWNER/REPO/blob/FULL_COMMIT_SHA/README.md \ + --search-term LoopX --format json > plan.json +loopx external-evidence execute --plan-json plan.json --execute --format json > execution.json +loopx external-evidence readback --plan-json plan.json --receipt-json execution.json +# The parent separately inspects findings and chooses admit/reject. +loopx external-evidence admit --plan-json plan.json --receipt-json execution.json \ + --decision admit --reason "Direct source answers this bounded decision" \ + --admit-source https://github.com/OWNER/REPO/blob/FULL_COMMIT_SHA/README.md \ + --format json > admission.json +loopx deepresearch start --project research --question "Inspect public source" +loopx external-evidence readback --plan-json plan.json --receipt-json execution.json \ + --admission-json admission.json --project research --execute --format json +loopx external-evidence readback --plan-json plan.json --receipt-json execution.json \ + --admission-json admission.json --project research +``` + +`execute --execute` authorizes source reads only. `readback --execute` authorizes +writing already explicitly admitted compact sources to the **existing** research +ledger. It requires the active research question to match the plan objective. +Retries for the same admission are idempotent. Wrong questions, unrelated existing +sources and exhausted source budgets remain blockers; partial projection stays +retained until every admitted source is actually read back with matching lineage. +Omit `--execute` for read-only projection; omit `--public-github` to avoid the +provider readiness probe. There is no persistent provider enablement to uninstall. +Original sources remain available on partial, empty and failed results. Literal +matches prove neither semantic conclusions nor evidence completeness. + +`execute --execute` 只授权读取来源;`readback --execute` 只把已明确采纳的紧凑证据 +写入现有研究账本,且要求研究问题与 plan objective 相同。同一 admission 可幂等重试; +问题不匹配、已有不相关来源或预算耗尽仍为 blocker。部分下游投影保持 retained, +直至全部已采纳来源以匹配 lineage 实际回读。省略 `--execute` 可只读回读;省略 +`--public-github` 不探测该 provider。没有持久开关需要卸载。部分、空或失败证据 +保留原始来源退路;字面匹配不证明语义结论或证据完整性。 + +Real qualification (anonymous network reads, disposable synthetic ledger): +`uv run --extra test python examples/public-github-evidence-live-smoke.py --execute-public-provider --source https://github.com/OWNER/REPO/blob/FULL_COMMIT_SHA/README.md`. +Packaged conversation readback: set `LOOPX_PERSONAL_WORKSPACE_PACKAGED=1` and +`LOOPX_PERSONAL_WORKSPACE_SCENARIO=external-evidence-readback`, then run +`node examples/personal-workspace-browser-smoke.mjs`. +Live authenticated connector and Lark delivery qualification remain separate; +this method grants no connector credentials or outbound-message authority. + +真实验收会进行匿名网络读取并使用一次性合成账本;打包会话验收沿用上述环境变量和命令。 +真实带凭据 connector 与 Lark 送达资格仍是独立边界;本方法不授予 connector 凭据或外发消息权限。 diff --git a/loopx/capabilities/external_research/catalog_entry.py b/loopx/capabilities/external_research/catalog_entry.py index 047db8bc36..97d874b3ea 100644 --- a/loopx/capabilities/external_research/catalog_entry.py +++ b/loopx/capabilities/external_research/catalog_entry.py @@ -21,12 +21,12 @@ ), "user_value": ( "Discover method and connector inventory, select only a currently ready provider, " - "bind a caller-presented provider receipt to its exact plan, and admit or reject compact " + "execute explicitly selected public GitHub sources, bind a provider receipt to its exact plan, and admit or reject compact " "provenance without copying raw provider content." ), "next_real_step": ( - "run `loopx external-evidence discover --connector-registry`, then provide a current " - "provider inventory to plan and admit the returned provider receipt" + "run `loopx external-evidence plan --public-github --help`, then inspect the returned " + "receipt before explicit admission and downstream ledger readback" ), "entry_command": "loopx external-evidence discover --help", "commands": [ @@ -40,6 +40,16 @@ "purpose": "Bind object, user activity, decision, and evidence kinds to one ready provider.", "write_boundary": "read-only", }, + { + "command": "loopx external-evidence execute --plan-json plan.json --execute", + "purpose": "Read explicit pinned public GitHub sources; parent admission remains separate.", + "write_boundary": "anonymous public HTTPS reads; no raw persistence", + }, + { + "command": "loopx external-evidence readback --plan-json plan.json --receipt-json execution.json ...", + "purpose": "Show the same source lineage, parent decision and actual downstream coverage.", + "write_boundary": "read-only by default; --execute projects admitted sources to the existing research ledger", + }, { "command": "loopx external-evidence receipt --plan-json plan.json --receipt-json receipt.json", "purpose": ( diff --git a/loopx/capabilities/external_research/cli.py b/loopx/capabilities/external_research/cli.py index fd3c4da9db..697741e8bf 100644 --- a/loopx/capabilities/external_research/cli.py +++ b/loopx/capabilities/external_research/cli.py @@ -8,6 +8,8 @@ from ...control_plane.effect_runtime import effect_runtime_result from ..connector_registry.core import load_connector_registry +from ...extensions.public_github_research import execute_public_github, inspect_provider +from .projection import readback, render_readback, validate_receipt PrintPayload = Callable[ @@ -32,6 +34,14 @@ def _load_object(path_text: str, *, label: str) -> dict[str, Any]: return value +def _load_receipt(path_text: str) -> dict: + value = _load_object(path_text, label="external evidence receipt") + receipt = value.get("receipt", value) + if not isinstance(receipt, dict): + raise ValueError("external evidence receipt must be an object") + return receipt + + def _provider_inventory(value: Mapping[str, Any]) -> list[dict[str, object]]: providers = value.get("providers") if not isinstance(providers, list): @@ -80,6 +90,8 @@ def _merge_providers( def _render(payload: dict[str, object]) -> str: + if payload.get("schema_version") == "loopx_external_evidence_readback_v0": + return render_readback(payload) lines = ["# LoopX External Evidence", ""] for field in ( "status", @@ -137,11 +149,27 @@ def register_external_evidence_commands( plan.add_argument("--decision", required=True) plan.add_argument("--evidence-kind", action="append", required=True) plan.add_argument("--constraint", action="append", default=[]) - plan.add_argument("--provider-inventory-json", required=True) + plan.add_argument("--provider-inventory-json") + plan.add_argument("--public-github", action="store_true", help="Opt in to a fresh anonymous public GitHub readiness probe.") + plan.add_argument("--source", action="append", default=[]) + plan.add_argument("--search-term", action="append", default=[]) plan.add_argument("--connector-registry", nargs="?", const="") plan.add_argument("--preferred-provider-id") add_subcommand_format(plan) + execute = actions.add_parser("execute", help="Execute the exact public GitHub plan without admitting evidence.") + execute.add_argument("--plan-json", required=True) + execute.add_argument("--execute", action="store_true", help="Authorize bounded anonymous public-source reads.") + add_subcommand_format(execute) + + project = actions.add_parser("readback", help="Show receipt, parent decision and actual research-ledger coverage.") + project.add_argument("--plan-json", required=True) + project.add_argument("--receipt-json", required=True) + project.add_argument("--admission-json") + project.add_argument("--project", help="Existing deepresearch project; no implicit run is created.") + project.add_argument("--execute", action="store_true", help="Write explicitly admitted sources to the existing ledger.") + add_subcommand_format(project) + receipt = actions.add_parser( "receipt", help="Validate and bind a caller-presented provider receipt to its exact plan.", @@ -194,10 +222,14 @@ def handle_external_evidence_command( {"providers": providers}, ) elif args.external_evidence_action == "plan": - inventory = _load_object( - args.provider_inventory_json, - label="external evidence provider inventory", - ) + if not args.provider_inventory_json and not getattr(args, "public_github", False): + raise ValueError("plan requires provider inventory or explicit --public-github") + inventory = _load_object(args.provider_inventory_json, + label="external evidence provider inventory") if args.provider_inventory_json else {"providers": []} + if getattr(args, "public_github", False): + if len(getattr(args, "source", [])) > 8: + raise ValueError("public GitHub plan allows at most eight sources") + inventory["providers"] = _merge_providers(_provider_inventory(inventory), [inspect_provider(getattr(args, "source", []))]) providers = _merge_providers( _connector_inventory(args.connector_registry), _provider_inventory(inventory), @@ -211,11 +243,31 @@ def handle_external_evidence_command( "decision": args.decision, "evidence_kinds": args.evidence_kind, "constraints": args.constraint, + **({"source_refs": getattr(args, "source", [])} if getattr(args, "source", []) else {}), + **({"search_terms": getattr(args, "search_term", [])} if getattr(args, "search_term", []) else {}), }, "providers": providers, "preferred_provider_id": args.preferred_provider_id, }, ) + elif args.external_evidence_action == "execute": + if not args.execute: + raise ValueError("--execute is required for real public-source reads") + plan = _load_object(args.plan_json, label="external evidence plan") + # Canonical identity must pass the typed owner before any HTTP call. + probe_receipt = {"schema_version": "loopx_external_evidence_receipt_v0", + "plan_id": plan.get("plan_id"), "request_id": plan.get("request", {}).get("request_id"), + "provider_id": plan.get("selected_provider", {}).get("provider_id"), + "provider_kind": plan.get("selected_provider", {}).get("provider_kind"), + "status": "failed", "sources": [], "summary": "Validation only", "completed_at": "not-executed"} + validate_receipt(plan, probe_receipt) + payload = execute_public_github(plan) + payload["observation"] = validate_receipt(plan, payload["receipt"]) + elif args.external_evidence_action == "readback": + payload = readback(_load_object(args.plan_json, label="external evidence plan"), + _load_receipt(args.receipt_json), + _load_object(args.admission_json, label="external evidence admission") if args.admission_json else None, + project=Path(args.project).expanduser() if args.project else None, execute=args.execute) elif args.external_evidence_action == "receipt": payload = effect_runtime_result( "external_evidence.receipt", @@ -223,9 +275,7 @@ def handle_external_evidence_command( "plan": _load_object( args.plan_json, label="external evidence plan" ), - "receipt": _load_object( - args.receipt_json, label="external evidence receipt" - ), + "receipt": _load_receipt(args.receipt_json), }, ) elif args.external_evidence_action == "admit": @@ -235,9 +285,7 @@ def handle_external_evidence_command( "plan": _load_object( args.plan_json, label="external evidence plan" ), - "receipt": _load_object( - args.receipt_json, label="external evidence receipt" - ), + "receipt": _load_receipt(args.receipt_json), "decision": { "disposition": args.decision, "reason": args.reason, diff --git a/loopx/capabilities/external_research/projection.py b/loopx/capabilities/external_research/projection.py new file mode 100644 index 0000000000..3d01582468 --- /dev/null +++ b/loopx/capabilities/external_research/projection.py @@ -0,0 +1,111 @@ +"""Shared external evidence readback for CLI and existing conversation surfaces. + +Typed effect-runtime reductions remain the sole admission/retirement authority. +Deep-research owns its existing durable source ledger; no parallel evidence store. +""" +from __future__ import annotations + +import re +from pathlib import Path + +from ...control_plane.effect_runtime import effect_runtime_result +from ..deep_research.runtime import add_source, load_state + + +def validate_receipt(plan: dict, receipt: dict) -> dict: + return effect_runtime_result("external_evidence.receipt", {"plan": plan, "receipt": receipt}) + + +def _validate_admission(plan: dict, receipt: dict, admission: dict) -> dict: + normalized = effect_runtime_result("external_evidence.admit", {"plan": plan, "receipt": receipt, + "decision": {"disposition": admission.get("disposition"), "reason": admission.get("reason"), + "admitted_source_refs": admission.get("admitted_source_refs", [])}}) + if normalized != admission: + raise ValueError("admission does not match its exact plan and receipt") + return normalized + + +def _lineage(admission: dict, source: dict) -> dict: + return {"plan_id": admission["plan_id"], "admission_id": admission["admission_id"], + "receipt_digest": admission["receipt_digest"], "content_digest": source["content_digest"]} + + +def readback(plan: dict, receipt: dict, admission: dict | None = None, + *, project: Path | None = None, execute: bool = False) -> dict: + observation = validate_receipt(plan, receipt) + receipt = observation["receipt"] + if execute and (admission is None or project is None): + raise ValueError("downstream write requires an explicit parent admission and research project") + normalized = _validate_admission(plan, receipt, admission) if admission is not None else None + write_blockers = [] + if execute and normalized["disposition"] == "admit": + for source in normalized["downstream_projection"]["sources"]: + try: + add_source(project, url_or_path=source["source_ref"], tool="external-evidence", + title="Public evidence", claims=[{"text": source["finding"], "stance": "neutral"}], + external_evidence=_lineage(normalized, source), expected_question=plan["request"]["objective"]) + except ValueError as error: + # A bounded partial projection is retained. Retrying is idempotent + # for the same identity; failed sources never count as covered. + write_blockers.append(str(error)) + break + covered = [] + if project is not None and normalized is not None: + state = load_state(project) + if state is not None and state["question"] == plan["request"]["objective"]: + for source in normalized["downstream_projection"]["sources"]: + for row in state["sources"]: + if (row["url_or_path"] == source["source_ref"] and + row.get("external_evidence") == _lineage(normalized, source) and + any(claim["id"] in row["claims"] and claim["text"] == source["finding"] + for claim in state["claims"])): + covered.append(source["source_ref"]) + break + retirement = effect_runtime_result("external_evidence.retire", {"admission": normalized, + "downstream_source_refs": covered}) if normalized is not None else None + return {"schema_version": "loopx_external_evidence_readback_v0", "plan_id": plan["plan_id"], + "request": plan["request"], "provider_id": receipt["provider_id"], "receipt_status": receipt["status"], + "sources": receipt["sources"], "summary": receipt["summary"], "limitations": receipt["limitations"], + "parent_admission": None if normalized is None else {"disposition": normalized["disposition"], + "reason": normalized["reason"], "admission_id": normalized["admission_id"], + "admitted_source_refs": normalized["admitted_source_refs"]}, + "downstream_source_refs": covered, "retirement": retirement, "write_blockers": write_blockers, + "original_source_fallback_allowed": True, "automatic_admission": False, + "evidence_coverage_observed": False} + + +def _text(value: object) -> str: + return re.sub(r"([\\`*_{}\[\]()<>#!|])", r"\\\1", str(value)).replace("\n", " ") + + +def render_readback(payload: dict) -> str: + admission = payload["parent_admission"] + retirement = payload["retirement"] + lines = ["## External evidence / 外部证据", "", _text(payload["request"]["objective"]), "", + "- Provider / 来源:" + _text(payload["provider_id"]), + "- Receipt / 读取结果:" + payload["receipt_status"], + "- Parent decision / 父 Agent 决定:" + ("pending / 待采纳" if admission is None else _text(admission["disposition"])), + f"- Downstream coverage / 下游覆盖:{len(payload['downstream_source_refs'])} admitted sources / 已采纳来源", + "- Retirement / 退休:" + ("pending / 待决定" if retirement is None else retirement["status"]), + "- Original-source fallback remains available / 可继续使用原始来源。", + "- Evidence completeness is unverified / 证据完整性尚未验证。", ""] + if admission is not None: + lines.extend(["Parent reason / 采纳理由:" + _text(admission["reason"]), ""]) + for ref in payload["request"].get("source_refs", []): + lines.extend(["Original source / 原始来源:" + _text(ref), ""]) + admitted = set(admission["admitted_source_refs"]) if admission is not None else set() + covered = set(payload["downstream_source_refs"]) + for index, source in enumerate(payload["sources"], 1): + ref = source["source_ref"] + lines.extend([f"### Source {index} / 来源 {index}", "", _text(ref), "", + _text(source["finding"]), "", + "- Evidence basis / 依据:" + _text(source["basis"]), + "- Accessed / 读取时间:" + _text(source["accessed_at"]), + "- Digest / 摘要:" + _text(source["content_digest"]), + "- Admission / 采纳:" + ("admitted" if ref in admitted else "not admitted"), + "- Downstream / 下游:" + ("observed" if ref in covered else "not observed"), + "- Limitation / 限制:" + _text(source.get("limitation") or "unspecified"), ""]) + for limitation in [*payload["limitations"], *payload["write_blockers"]]: + lines.append("- " + _text(limitation)) + lines.extend(["", "Plan identity / 计划身份:" + _text(payload["plan_id"])]) + return "\n".join(lines) + "\n" diff --git a/loopx/control_plane/capabilities/external_evidence.ts b/loopx/control_plane/capabilities/external_evidence.ts index 4dcf8ed30b..260f0e8246 100644 --- a/loopx/control_plane/capabilities/external_evidence.ts +++ b/loopx/control_plane/capabilities/external_evidence.ts @@ -92,6 +92,19 @@ function normalizeRequest(value: unknown): JsonObject { (normalized.evidence_kinds as string[]).length > 0, "request.evidence_kinds must not be empty", ); + // Optional source selection is part of the exact request identity. Preserve + // legacy request digests when no source-bound provider is requested. + if (request.source_refs !== undefined) { + const refs = boundedStrings(request.source_refs, "request.source_refs", 8, 2048); + requireThat(refs.length > 0 && new Set(refs).size === refs.length, + "request.source_refs must be nonempty and unique"); + requireThat(refs.every((ref) => SOURCE_REF_RE.test(ref) && !ref.startsWith("file://")), + "request.source_refs must be non-file provenance URIs"); + normalized.source_refs = refs; + } + if (request.search_terms !== undefined) { + normalized.search_terms = boundedStrings(request.search_terms, "request.search_terms", 8, 256); + } normalized.request_id = digest(normalized); return normalized; } diff --git a/loopx/control_plane/coordination/authority_archive.ts b/loopx/control_plane/coordination/authority_archive.ts index 46fc2c35d0..e2515c1667 100644 --- a/loopx/control_plane/coordination/authority_archive.ts +++ b/loopx/control_plane/coordination/authority_archive.ts @@ -2,6 +2,7 @@ import {randomUUID} from "node:crypto"; import {link, open, unlink} from "node:fs/promises"; import {dirname} from "node:path"; +import {syncAuthorityDirectory} from "./file_authority_store.ts"; import type {JsonObject} from "../effect_program.ts"; import type {AuthorityStore, AuthorityStoreCommittedTransaction} from "./authority_store.ts"; import {AuthorityStoreProtocolError, canonicalAuthoritySha256, requireAuthorityStoreId} from "./authority_store_codec.ts"; @@ -73,8 +74,7 @@ export async function exportAuthorityArchive(store: AuthorityStore, goalId: stri await handle.sync(); await handle.close(); closed = true; const verified = await verifyAuthorityArchive(temporary); await link(temporary, output); - const directory = await open(dirname(output), "r"); - try { await directory.sync(); } finally { await directory.close(); } + await syncAuthorityDirectory(dirname(output)); return verified; } finally { if (!closed) await handle.close(); diff --git a/loopx/extensions/public_github_research.py b/loopx/extensions/public_github_research.py new file mode 100644 index 0000000000..6903c9d32e --- /dev/null +++ b/loopx/extensions/public_github_research.py @@ -0,0 +1,131 @@ +"""Explicit public GitHub source-inspection method; no credentials or raw persistence. + +This bundled provider owns HTTP access. External evidence's TypeScript contract +owns exact-plan validation, parent admission and retirement. +""" +from __future__ import annotations + +import hashlib +import json +import re +from datetime import datetime, timezone +from urllib.error import HTTPError, URLError +from urllib.parse import quote, unquote, urlsplit +from urllib.request import HTTPRedirectHandler, ProxyHandler, Request, build_opener + +PROVIDER_ID = "method:public-github" +MAX_SOURCE_BYTES = 1_000_000 +SOURCE = re.compile(r"/([A-Za-z0-9_.-]+)/([A-Za-z0-9_.-]+)/blob/([0-9a-f]{40})/(.+)") + + +class _NoRedirect(HTTPRedirectHandler): + def redirect_request(self, req, fp, code, msg, headers, newurl): + return None + + +def _read(url: str) -> bytes: + # No provider credentials, cookies, user-configured proxy credentials or + # caller-selected origins enter the request. Redirects fail closed. + request = Request(url, headers={"User-Agent": "LoopX-public-evidence", + "Accept": "application/vnd.github+json" if url.startswith("https://api.github.com/") else "text/plain"}) + with build_opener(ProxyHandler({}), _NoRedirect()).open(request, timeout=15) as response: + result = response.read(MAX_SOURCE_BYTES + 1) + if len(result) > MAX_SOURCE_BYTES: + raise ValueError("public source exceeds the bounded read limit") + return result + + +def _source(ref: str) -> tuple[str, str, str, str]: + parsed = urlsplit(ref) + match = SOURCE.fullmatch(parsed.path) + if (parsed.scheme != "https" or parsed.netloc != "github.com" or parsed.query + or parsed.fragment or match is None): + raise ValueError("public GitHub sources require https://github.com/OWNER/REPO/blob/FULL_COMMIT_SHA/PATH") + owner, repo, revision, path = match.groups() + decoded = unquote(path) + if (any(part in {"", ".", ".."} for part in decoded.split("/")) or "\\" in decoded + or any(ord(char) < 32 for char in decoded) or "%" in decoded + or owner in {".", ".."} or repo in {".", ".."}): + raise ValueError("public GitHub source path is invalid") + return owner, repo, revision, decoded + + +def _public_repository(owner: str, repo: str) -> None: + value = json.loads(_read(f"https://api.github.com/repos/{owner}/{repo}")) + if not isinstance(value, dict) or value.get("private") is not False: + raise ValueError("provider only reads currently public GitHub repositories") + + +def inspect_provider(source_refs: list[str]) -> dict: + """Current opt-in readiness, not registry inventory or execution proof.""" + sources = [_source(ref) for ref in source_refs] + if not sources: + raise ValueError("public GitHub inspection requires at least one pinned source") + reason = None + try: + for owner, repo in sorted({(source[0], source[1]) for source in sources}): + _public_repository(owner, repo) + except (OSError, ValueError, URLError): + reason = "public_github_readiness_unavailable" + return {"provider_id": PROVIDER_ID, "provider_kind": "method", + "protocol": "external_evidence_research_v0", "declared": True, + "installed": True, "enabled": True, "ready": reason is None, + "unavailable_reason": reason} + + +def execute_public_github(plan: dict) -> dict: + """Read only exact-plan sources, with fresh public-visibility checks. + + Caller validates the canonical plan through the typed owner before entry. + Findings are retrieval and literal-match facts, not autonomous conclusions. + """ + selected = plan["selected_provider"] + if selected["provider_id"] != PROVIDER_ID or selected["provider_kind"] != "method": + raise ValueError("this executor requires the selected public GitHub method") + request = plan["request"] + refs = request.get("source_refs", []) + sources = [_source(ref) for ref in refs] + if not sources or len(sources) > 8: + raise ValueError("public GitHub execution requires one to eight pinned sources") + terms = request.get("search_terms", []) + records, failures = [], [] + now = datetime.now(timezone.utc).isoformat().replace("+00:00", "Z") + for index, (ref, (owner, repo, revision, path)) in enumerate(zip(refs, sources), 1): + try: + _public_repository(owner, repo) + raw = _read(f"https://raw.githubusercontent.com/{owner}/{repo}/{revision}/{quote(path, safe='/')}") + text = raw.decode("utf-8") + if "\x00" in text: + raise ValueError("source is not UTF-8 text") + if not text.strip(): + failures.append(f"Source {index}: empty source") + continue + matches = [] + for term in terms: + lines = [str(index) for index, line in enumerate(text.splitlines(), 1) if term in line] + matches.append(f"Literal term {json.dumps(term)}: " + + ("observed at lines " + ", ".join(lines[:16]) if lines else "not observed") + + (" (additional matches omitted)" if len(lines) > 16 else "")) + finding = f"Read pinned file {path}; {len(raw)} UTF-8 bytes. " + " ".join(matches) + if len(finding) > 4096: + finding = finding[:4040] + " (additional match metadata omitted)" + records.append({"source_ref": ref, "source_family": "github_repository_file", + "basis": "observed", "finding": finding, + "limitation": "Retrieval/literal matches only; no semantic conclusion, execution test or completeness claim.", + "publication_date": None, "accessed_at": now, + "content_digest": "sha256:" + hashlib.sha256(raw).hexdigest()}) + except HTTPError as error: + failures.append(f"Source {index}: HTTP {error.code}") + except (OSError, ValueError, UnicodeError, URLError): + failures.append(f"Source {index}: source read unavailable") + status = "succeeded" if records else "no_evidence" if all("empty source" in item for item in failures) else "failed" + receipt = {"schema_version": "loopx_external_evidence_receipt_v0", + "plan_id": plan["plan_id"], "request_id": request["request_id"], + "provider_id": PROVIDER_ID, "provider_kind": "method", "status": status, + "sources": records, "summary": f"Read {len(records)} of {len(refs)} requested public sources.", + "limitations": ["Original-source fallback remains available; parent admission is required.", *failures], + "completed_at": now} + return {"receipt": receipt, "execution": {"provider_id": PROVIDER_ID, + "source_reads_observed": len(records), "requested_source_count": len(refs), + "raw_content_persisted": False, "credentials_used": False, + "automatic_admission": False, "evidence_coverage_observed": False}} diff --git a/skills/loopx-self-repair/references/repair-patterns.md b/skills/loopx-self-repair/references/repair-patterns.md index af523e9445..a1f4c7c2e6 100644 --- a/skills/loopx-self-repair/references/repair-patterns.md +++ b/skills/loopx-self-repair/references/repair-patterns.md @@ -24,6 +24,7 @@ teaches a reusable control-plane lesson. | `delegation_runtime_discovery_split` | A coordinator knows a child-count or model preference but cannot discover requester-authorized managed routes, so a direct SDK experiment is mistaken for formal LoopX delegation or the task stays unnecessarily serial. | Goal orchestration readback, requester-scoped binding directory, Turn host/profile readiness, operation journal, signed `before_plan`/`before_delegate`/`after_delegate_result` context, and parent artifact validation. | Execution grants and runtime facts existed only behind the dispatch CLI/MCP while planning consumed a separate prompt or model-preference snapshot; adding provider names to one skill would create another scheduler/config owner. | Store only an ignored Goal-local pointer to the existing operator binding file, project bounded public-safe requester routes through the generic capability context, and keep runtime readiness separate from route selection and task adoption. Recheck runtime/model/budget at dispatch, forbid silent substitution, and reconcile original operation receipts after return. Do not require every heartbeat to use every route or copy credentials/host arguments into registry, frontend, Lark or prompts. | | `goal_runtime_shadows_machine_credential` | A configured machine appears credential-less in another Goal or interpreter. | Machine credential status, Goal runtime root, launching interpreter SDK probe, Turn plan and dispatch. | Goal state location or process environment was mistaken for the machine authentication owner. | Resolve the canonical machine credential at planning and dispatch; keep SDK readiness interpreter-scoped and assignments requester/Goal-scoped. Verify two Goal roots, conflicting Goal-local stores, invalid machine-store refusal and explicit host selection. Never copy a credential or another Goal's grants to make readiness green. | | `budget_metric_overfitting` | A budget failure triggers automatic expansion, or mechanical compaction that removes useful semantics or breaks consumers. | Owning limit, matched base/head measurements, consumer/caller contract, original failure and revised validation. | A regression metric became the objective; historical ceilings or green tests replaced semantic judgment. | Follow the [budget decision guide](../../../docs/development/testing-and-quality.md#budget-failure-decisions), compare true redundancy, compatibility cost and justified headroom, and repair the existing contract/tests and review evidence. Preserve hard limits and frozen qualification results. | +| `validation_fixture_owner_drift` | CI repeatedly fails after a rebase because fixtures retain an old catalog count, omit current caller context or dispatch without the owning claim. | The current source contract, exact failing assertion, base/head entrypoint and durable state readback. | Validation encoded a dated snapshot or skipped current admission instead of preparing the operation it intended to challenge. | Reuse the shipped catalog and package metadata, explicitly isolate runtime routes, and acquire the real queue claim before dispatch. Retain the original negative invariant and exit/readback assertions. Do not weaken routing, execution admission or production boundaries to make stale fixtures pass. | | `skill_import_recreation` | Duplicate LoopX skills return after successful cleanup; imported entries display a fallback brand casing. | Compare installed files and metadata with source-host skills; correlate file creation times with structured host import receipts. | A later external-host import recreates command facades in another discovered root, omits display metadata, and bypasses installer reconciliation. | Attribute the writer from import receipts without guessing the human initiator; exclude already-installed LoopX skills from later imports and rerun managed reconciliation. Ensure standalone workflow entry installation writes Codex metadata, previews missing metadata repair, and records the full installed tree. Preserve user metadata and exact-host invocation behavior. | | `skill_discovery_split_ownership` | Duplicate skill names, conflicting PR-review routes, or canonical and legacy aliases appear together. | Enumerate discovered roots, resolve directory symlinks, compare skill hashes, managed markers, install receipts, and generated metadata. | Workflow and command installers wrote independently to overlapping host roots; dedupe was optional, omitted the bare entry name, or retired copies without proving a replacement. | Repair the shared installer reconciliation and every active installation path; preserve user changes and rich workflows, retire managed aliases from the Codex picker, test repeated installs and custom profiles, then verify a fresh host catalog. Do not treat deleting one visible duplicate or changing invocation policy as a durable fix. | | `skill_discovery_scope_eligibility_conflation` | Project delivery rejects a reusable workflow, and adding a project marker unexpectedly removes connection, repair, or review instructions from global installation. | Canonical scope markers, default shell/CLI and packaged install output, project-copy readback, and doctor required workflows. | One marker was treated as both exclusive project eligibility and default discovery; content richness was mistaken for project authority. | Declare reusable workflows global and capability-local workflows project; accept both explicit declarations for project copies while rejecting missing/unknown markers. Preserve global command routes, existing activation gates, and packaged resource parity. Never repair project delivery by hiding bootstrap or repair instructions from unconnected projects. | diff --git a/tests/architecture/test_content_digest_single_owner.py b/tests/architecture/test_content_digest_single_owner.py index af6d47cfc2..ac348a0836 100644 --- a/tests/architecture/test_content_digest_single_owner.py +++ b/tests/architecture/test_content_digest_single_owner.py @@ -161,6 +161,7 @@ "loopx.capabilities.benchmark_toolkit.runtime_continuity", "loopx.capabilities.benchmark_toolkit.study_projection", "loopx.capabilities.content_ops.item_lifecycle", + "loopx.capabilities.deep_research.runtime", "loopx.capabilities.issue_fix.outcome_projection", "loopx.capabilities.issue_fix.reviewer_notification", "loopx.capabilities.machine_configuration.store", diff --git a/tests/capabilities/test_public_github_evidence.py b/tests/capabilities/test_public_github_evidence.py new file mode 100644 index 0000000000..b490902269 --- /dev/null +++ b/tests/capabilities/test_public_github_evidence.py @@ -0,0 +1,210 @@ +from __future__ import annotations + +import argparse +import copy +import hashlib +import json +from urllib.error import HTTPError + +import pytest + +from loopx.capabilities.external_research import cli +from loopx.capabilities.external_research.projection import readback, render_readback +from loopx.capabilities.deep_research.runtime import add_source, load_state, start_research +from loopx.control_plane.effect_runtime import effect_runtime_result +from loopx.extensions import public_github_research as provider + +REF = "https://github.com/example/public/blob/" + "a" * 40 + "/README.md" +SECOND = REF.replace("README.md", "missing.md") + + +def plan(refs=None): + return effect_runtime_result("external_evidence.plan", {"request": { + "objective": "Inspect public fixture", "user_activity": "Choose a source", + "decision": "Whether the pinned source contains the fixture marker", + "evidence_kinds": ["literal_match"], "source_refs": refs or [REF], "search_terms": ["fixture"]}, + "providers": [{"provider_id": provider.PROVIDER_ID, "provider_kind": "method", + "protocol": "external_evidence_research_v0", "declared": True, "installed": True, + "enabled": True, "ready": True, "unavailable_reason": None}]}) + + +def reader(url): + if url.startswith("https://api.github.com/"): + return b'{"private": false}' + if url.endswith("missing.md"): + raise HTTPError(url, 404, "not found", {}, None) + return b"fixture public data\n" + + +def admission(p, receipt, disposition="admit"): + return effect_runtime_result("external_evidence.admit", {"plan": p, "receipt": receipt, + "decision": {"disposition": disposition, "reason": "Fixture parent decision", + "admitted_source_refs": [s["source_ref"] for s in receipt["sources"]] if disposition == "admit" else []}}) + + +@pytest.mark.parametrize("ref", ["file:///private", REF.replace("https://", "http://"), + REF.replace("github.com", "github.com.evil"), REF.replace("/" + "a"*40 + "/", "/main/"), + REF + "?query=fixture", REF + "#L1", REF.replace("README.md", "../private"), + REF.replace("README.md", "%2e%2e/private"), REF.replace("README.md", "%252e%252e/private")]) +def test_provider_rejects_unpinned_or_out_of_scope_sources(ref, monkeypatch): + monkeypatch.setattr(provider, "_read", lambda _: pytest.fail("invalid input made a HTTP call")) + with pytest.raises(ValueError): + provider.inspect_provider([ref]) + + +def test_fresh_visibility_probe_does_not_trust_old_ready_plan(monkeypatch): + monkeypatch.setattr(provider, "_read", lambda _: b'{"private": true}') + assert provider.inspect_provider([REF])["ready"] is False + output = provider.execute_public_github(plan()) + assert output["receipt"]["status"] == "failed" + assert output["receipt"]["sources"] == [] + assert output["execution"]["automatic_admission"] is False + + +@pytest.mark.parametrize("content,status", [(b"", "no_evidence"), (b"\x00binary", "failed")]) +def test_empty_or_binary_source_preserves_fallback(monkeypatch, content, status): + monkeypatch.setattr(provider, "_read", lambda url: b'{"private":false}' if "api.github.com" in url else content) + p = plan() + output = provider.execute_public_github(p) + assert output["receipt"]["status"] == status + projected = readback(p, output["receipt"]) + assert projected["original_source_fallback_allowed"] is True + assert projected["parent_admission"] is None + assert REF in render_readback(projected) + + +def test_partial_reads_are_observed_not_admitted_or_complete(monkeypatch): + monkeypatch.setattr(provider, "_read", reader) + p = plan([REF, SECOND]) + output = provider.execute_public_github(p) + receipt = output["receipt"] + assert receipt["status"] == "succeeded" and len(receipt["sources"]) == 1 + source = receipt["sources"][0] + assert source["basis"] == "observed" + assert source["content_digest"] == "sha256:" + hashlib.sha256(b"fixture public data\n").hexdigest() + assert "lines 1" in source["finding"] + assert "fixture public data" not in json.dumps(output) + before = readback(p, receipt) + assert before["parent_admission"] is None and before["downstream_source_refs"] == [] + assert before["evidence_coverage_observed"] is False + assert any("HTTP 404" in item for item in before["limitations"]) + rejected = readback(p, receipt, admission(p, receipt, "reject")) + assert rejected["retirement"]["reason"] == "parent_rejected" + + +@pytest.mark.parametrize("mutate", ["source_refs", "search_terms", "objective"]) +def test_execute_rejects_mutated_plan_before_provider_calls(tmp_path, monkeypatch, mutate): + p = plan() + p["request"][mutate] = [SECOND] if mutate == "source_refs" else ["different"] if mutate == "search_terms" else "different" + path = tmp_path / "plan.json" + path.write_text(json.dumps(p), encoding="utf-8") + monkeypatch.setattr(cli, "execute_public_github", lambda _: pytest.fail("mutated plan reached provider")) + payloads = [] + args = argparse.Namespace(command="external-evidence", external_evidence_action="execute", + plan_json=str(path), execute=True) + assert cli.handle_external_evidence_command(args, output_format=lambda _: "json", + print_payload=lambda payload, *_: payloads.append(payload)) == 1 + assert payloads[0]["status"] == "invalid_request" + + +def test_execution_requires_explicit_opt_in(tmp_path, monkeypatch): + monkeypatch.setattr(cli, "execute_public_github", lambda _: pytest.fail("default-off execution called provider")) + args = argparse.Namespace(command="external-evidence", external_evidence_action="execute", + plan_json=str(tmp_path / "absent.json"), execute=False) + assert cli.handle_external_evidence_command(args, output_format=lambda _: "json", + print_payload=lambda *_: None) == 1 + + +def test_parent_admission_and_real_ledger_coverage_are_independent(tmp_path, monkeypatch): + monkeypatch.setattr(provider, "_read", reader) + p = plan() + receipt = provider.execute_public_github(p)["receipt"] + accepted = admission(p, receipt) + start_research(tmp_path, question=p["request"]["objective"], max_sources=8, max_subquestions=4) + retained = readback(p, receipt, accepted, project=tmp_path) + assert retained["retirement"]["status"] == "retained" + projected = readback(p, receipt, accepted, project=tmp_path, execute=True) + assert projected["downstream_source_refs"] == [REF] + assert projected["retirement"]["status"] == "retire_ready" + replay = readback(p, receipt, accepted, project=tmp_path, execute=True) + assert replay == projected + assert len(load_state(tmp_path)["sources"]) == 1 + assert REF in render_readback(replay) and "admitted" in render_readback(replay) + corrupt = copy.deepcopy(accepted) + corrupt["downstream_projection"]["sources"][0]["finding"] = "mutated" + with pytest.raises(ValueError, match="exact plan"): + readback(p, receipt, corrupt, project=tmp_path, execute=True) + + +def test_wrong_question_or_unrelated_source_never_proves_coverage(tmp_path, monkeypatch): + monkeypatch.setattr(provider, "_read", reader) + p = plan() + receipt = provider.execute_public_github(p)["receipt"] + accepted = admission(p, receipt) + start_research(tmp_path, question="Unrelated question", max_sources=8, max_subquestions=4) + result = readback(p, receipt, accepted, project=tmp_path, execute=True) + assert result["write_blockers"] and result["retirement"]["status"] == "retained" + assert not load_state(tmp_path)["sources"] + other = tmp_path / "other" + start_research(other, question=p["request"]["objective"], max_sources=8, max_subquestions=4) + add_source(other, url_or_path=REF, tool="manual", title=None, claims=[{"text":"Unrelated observation"}]) + result = readback(p, receipt, accepted, project=other, execute=True) + assert result["write_blockers"] and result["downstream_source_refs"] == [] + + +def test_partial_downstream_projection_retains_until_all_sources_read_back(tmp_path, monkeypatch): + monkeypatch.setattr(provider, "_read", lambda url: reader(url.replace("second.md", "README.md"))) + p = plan([REF, REF.replace("README.md", "second.md")]) + receipt = provider.execute_public_github(p)["receipt"] + accepted = admission(p, receipt) + start_research(tmp_path, question=p["request"]["objective"], max_sources=1, max_subquestions=4) + result = readback(p, receipt, accepted, project=tmp_path, execute=True) + assert result["downstream_source_refs"] == [REF] + assert result["retirement"]["status"] == "retained" + assert len(result["retirement"]["missing_downstream_source_refs"]) == 1 + assert result["write_blockers"] and result["original_source_fallback_allowed"] + + +def test_existing_lark_sink_preserves_shared_readback_facts(tmp_path, monkeypatch): + from loopx.extensions.lark.presentation.message_card import build_lark_markdown_reply_card + monkeypatch.setattr(provider, "_read", reader) + p = plan() + receipt = provider.execute_public_github(p)["receipt"] + accepted = admission(p, receipt) + start_research(tmp_path, question=p["request"]["objective"], max_sources=8, max_subquestions=4) + result = readback(p, receipt, accepted, project=tmp_path, execute=True) + card = build_lark_markdown_reply_card(render_readback(result)) + content = card["elements"][0]["text"]["content"] + assert "retire_ready" in content and REF in content + assert "Evidence completeness is unverified" in content + assert result["plan_id"] in content + assert "truncated" not in content + + +def test_long_readback_uses_existing_lossless_lark_transport(monkeypatch): + from loopx.extensions.lark.outbound import split_lark_outbound_text + from loopx.extensions.lark.presentation.message_card import build_lark_markdown_reply_card + monkeypatch.setattr(provider, "_read", reader) + p = plan() + receipt = provider.execute_public_github(p)["receipt"] + receipt["sources"][0]["finding"] = "Bounded public finding. " * 150 + result = readback(p, receipt, admission(p, receipt)) + markdown = render_readback(result) + parts = split_lark_outbound_text(markdown, limit=3000, preserve_format=True) + cards = [build_lark_markdown_reply_card(part) for part in parts] + assert len(cards) > 1 + body = "\n".join(card["elements"][0]["text"]["content"] for card in cards) + assert result["plan_id"] in body and REF in body + assert "Evidence completeness is unverified" in body and "truncated" not in body + + +def test_pinned_paths_keep_case_sensitive_identity(tmp_path, monkeypatch): + monkeypatch.setattr(provider, "_read", reader) + refs = [REF, REF.replace("README.md", "readme.md")] + p = plan(refs) + receipt = provider.execute_public_github(p)["receipt"] + accepted = admission(p, receipt) + start_research(tmp_path, question=p["request"]["objective"], max_sources=8, max_subquestions=4) + result = readback(p, receipt, accepted, project=tmp_path, execute=True) + assert result["downstream_source_refs"] == refs + assert result["retirement"]["retire_ready"] is True diff --git a/tests/control_plane/test_native_child_replan_guard_cli.py b/tests/control_plane/test_native_child_replan_guard_cli.py index 465e910cad..6105df8a8f 100644 --- a/tests/control_plane/test_native_child_replan_guard_cli.py +++ b/tests/control_plane/test_native_child_replan_guard_cli.py @@ -34,7 +34,9 @@ def _fixture(tmp_path: Path, monkeypatch: pytest.MonkeyPatch, provider: str, tod state.write_text("---\nstatus: active\n---\n\n# Synthetic Goal\n\n## Agent Todo\n" + ( "\n- [ ] [P1] Validate the original source.\n" f" \n" + f"claimed_by={AGENT} action_kind=validate validation_command=pytest " + "continuation_policy=same_agent_non_delivery " + "required_capabilities=shell%2Cfilesystem_read -->\n" if todo_bound else "" )) index = runtime / "goals" / GOAL / "runs" / "index.jsonl" diff --git a/tests/control_plane/test_todo_projection_recovery.py b/tests/control_plane/test_todo_projection_recovery.py index e310471d00..ce288370eb 100644 --- a/tests/control_plane/test_todo_projection_recovery.py +++ b/tests/control_plane/test_todo_projection_recovery.py @@ -132,8 +132,10 @@ def test_long_committed_todo_rebuilds_from_the_fresh_head_without_a_second_creat assert manager["authority_revision"] == before["provider_revision"] assert manager["todos"][0]["todo_id"] == record["todo_id"] assert manager["todos"][0]["title"].startswith("Independent evidence") - # The existing safe formatter appends its marker outside the text budget. - assert manager["todos"][0]["title"] == listed["todo"]["title"][:419].rstrip() + "..." + # The typed context owner budgets 420 characters including the ellipsis. + assert manager["todos"][0]["title"] == title[:417].rstrip() + "..." + assert len(manager["todos"][0]["title"]) <= 420 + assert manager["todos"][0]["content_truncated"] is True code, create_replay = _cli(registry, *create) assert code == 0 and create_replay["status"] == "replayed", create_replay assert _read(runtime) == before # No new Todo, provider revision or business receipt. diff --git a/tests/control_plane_ts/external_evidence_research.test.ts b/tests/control_plane_ts/external_evidence_research.test.ts index f1a0993893..380672b780 100644 --- a/tests/control_plane_ts/external_evidence_research.test.ts +++ b/tests/control_plane_ts/external_evidence_research.test.ts @@ -392,3 +392,19 @@ test("retirement fails closed on mutated admission semantics", () => { /admission_id does not match/, ); }); + + +test("optional source selection preserves legacy identity and binds source/query mutations", () => { + const legacy = plan(); + assert.equal((legacy.request as Record).source_refs, undefined); + const bound = planExternalEvidenceRequest({request: {...request, + source_refs: ["https://example.com/pinned"], search_terms: ["literal"]}, providers:[methodProvider]}); + assert.notEqual(bound.plan_id, legacy.plan_id); + for (const [field, value] of [["source_refs", ["https://example.com/other"]], ["search_terms", ["different"]]]) { + const changed = structuredClone(bound); + (changed.request as Record)[field as string] = value; + assert.throws(() => recordExternalEvidenceReceiptObservation({plan:changed, receipt:receipt(bound)}), /request_id|plan_id/); + } + assert.throws(() => planExternalEvidenceRequest({request:{...request, source_refs:["file:///private"]}, + providers:[methodProvider]}), /non-file/); +}); diff --git a/tests/test_chat_goal_configuration_api.py b/tests/test_chat_goal_configuration_api.py index a14964e1a4..d4b3ba7d33 100644 --- a/tests/test_chat_goal_configuration_api.py +++ b/tests/test_chat_goal_configuration_api.py @@ -509,7 +509,10 @@ def test_goal_configuration_service_rechecks_revision_before_write( monkeypatch: pytest.MonkeyPatch, ) -> None: registry_path = tmp_path / "registry.json" - registry_path.write_text("{}\n", encoding="utf-8") + import json + initial_registry = json.dumps({"common_runtime_root": str(tmp_path / "runtime"), + "goals": [{"id": "goal-example", "repo": str(tmp_path)}]}) + "\n" + registry_path.write_text(initial_registry, encoding="utf-8") calls: list[dict[str, Any]] = [] def configure_goal_stub(**kwargs: Any) -> dict[str, Any]: @@ -534,4 +537,4 @@ def configure_goal_stub(**kwargs: Any) -> dict[str, Any]: ) assert len(calls) == 1 - assert registry_path.read_text(encoding="utf-8") == "{}\n" + assert registry_path.read_text(encoding="utf-8") == initial_registry diff --git a/tests/test_packaged_skill_metadata.py b/tests/test_packaged_skill_metadata.py index fda1c66f73..26750eeec3 100644 --- a/tests/test_packaged_skill_metadata.py +++ b/tests/test_packaged_skill_metadata.py @@ -30,3 +30,14 @@ def test_packaged_scope_markers_ship_with_workflow_sources(): assert f"skills/{skill_id}/.loopx-skill-scope" in data_files[ f"share/loopx/skills/{skill_id}" ] + + +def test_packaged_skill_display_metadata_is_in_distribution(): + import tomllib + from loopx.skill_install_readback import PACKAGED_HOST_SKILL_IDS + + package = tomllib.loads((REPO_ROOT / "pyproject.toml").read_text(encoding="utf-8")) + data_files = package["tool"]["setuptools"]["data-files"] + for skill_id in PACKAGED_HOST_SKILL_IDS: + assert f"skills/{skill_id}/agents/openai.yaml" in data_files[ + f"share/loopx/skills/{skill_id}/agents"]