From 4043ce57467b010ba337ea81dfbf155f84b0c5d7 Mon Sep 17 00:00:00 2001 From: wilkont Date: Tue, 22 Sep 2026 13:57:10 -0500 Subject: [PATCH 01/55] Add standalone decision boundary and evidence-gated launch matrix --- README.md | 13 +- c3r/cvoc.py | 11 ++ c3r/http_service.py | 184 +++++++++++++++++++++++ c3r/runtime.py | 270 ++++++++++++++++++++++++++++++++++ c3r/telemetry/trace_ledger.py | 16 +- c3r/verifier_firewall.py | 5 + docs/directive-compliance.md | 17 ++- docs/standalone-launch.md | 35 +++++ tests/test_cvoc.py | 13 +- tests/test_http_service.py | 103 +++++++++++++ tests/test_runtime.py | 183 +++++++++++++++++++++++ tests/test_trace_ledger.py | 9 ++ 12 files changed, 841 insertions(+), 18 deletions(-) create mode 100644 c3r/http_service.py create mode 100644 c3r/runtime.py create mode 100644 docs/standalone-launch.md create mode 100644 tests/test_http_service.py create mode 100644 tests/test_runtime.py diff --git a/README.md b/README.md index ae78b96..1a2eafb 100644 --- a/README.md +++ b/README.md @@ -29,10 +29,10 @@ official DeepSeek API alias is `deepseek-flash`. Laya remains the separate Syste decision fast path, and DeepSeek recommendations remain subject to the same verifier and trusted commit boundary as every other candidate. -The 763B-parameter checkpoint is not assumed to fit a single GPU merely because only -8B/16B parameters are active per token. A GCP deployment must pass storage, aggregate -accelerator memory, runtime-version, and smoke-test gates before C3R labels it self-hosted; -otherwise the GPU service uses the hosted provider profile and stores no model weights. +The official checkpoint has passed a private 8×H100 GCP serving smoke test and is backed +up in a private GCS bucket. Its vLLM endpoint is bound to localhost; this is **not** a +public C3R service or end-to-end production qualification. The checkpoint's active +parameter count does not imply it fits on one GPU. ## Why C3R @@ -63,8 +63,9 @@ failed verification into success, or directly commit an external effect. | Laya fast path | exact upstream revision and license verification, typed probabilities, slice calibration, confidence/margin abstention | | DecisionMix v1 | validated records, immutable deterministic splits, source/license provenance, SHA-256 manifest, 144-record synthetic preview | | Authority boundary | action-bound verifier attestations, expiring single-use approvals, atomic nonce claims | -| Runtime control | conservative CVoC selection, deterministic `STOP`, cost twin, deliberative contracts, evidence-grade traces | +| Runtime control | conservative CVoC selection, deterministic `STOP`, cost twin, deliberative contracts, trace schema and in-memory hash chain | | Operational controls | fail-closed feature flags, tested frontier/open-weight HTTP contracts, Colibri shadow recommendations, canonical trace hash chain | +| Standalone controller boundary | tested state → candidates → optional Laya → CVoC → independent verifier → recommendation/commit fallback → redacted trace composition; authenticated rate-limited loopback HTTP boundary (not publicly deployed) | This compiler is the reviewed vertical slice, not the directive's full Candidate Compiler Definition of Done. Rich typed value constraints, per-argument provenance, dominated-branch @@ -127,6 +128,8 @@ is `STOP`. every Definition of Done item in Execution Directive v2. - [`docs/empirical-release-plan.md`](docs/empirical-release-plan.md) — gated path from synthetic preview to trained, calibrated, independently reproducible release. +- [`docs/standalone-launch.md`](docs/standalone-launch.md) — current production exit gates, + evidence status, and operator inputs, with MC-1 excluded from standalone scope only. - [`docs/launch-announcement.md`](docs/launch-announcement.md) — canonical launch copy plus LinkedIn, X, and Hacker News variants with a publication checklist. - [`docs/prior-art.md`](docs/prior-art.md) — explicit attribution links and the canonical novelty diff --git a/c3r/cvoc.py b/c3r/cvoc.py index 97c96d8..a1dbafb 100644 --- a/c3r/cvoc.py +++ b/c3r/cvoc.py @@ -3,6 +3,7 @@ from __future__ import annotations from collections.abc import Iterable, Mapping +from math import isfinite from .state_schema import ActionCandidate, ActionFamily, CvocDecision, ValueEstimate @@ -14,6 +15,16 @@ def __init__(self, uncertainty_multiplier: float = 1.96) -> None: self._uncertainty_multiplier = uncertainty_multiplier def lower_bound(self, estimate: ValueEstimate) -> float: + values = ( + estimate.expected_gain, + estimate.total_cost, + estimate.risk_penalty, + estimate.uncertainty, + ) + if not all(isfinite(value) for value in values): + return float("-inf") + if min(estimate.total_cost, estimate.risk_penalty, estimate.uncertainty) < 0: + return float("-inf") return ( estimate.expected_gain - estimate.total_cost diff --git a/c3r/http_service.py b/c3r/http_service.py new file mode 100644 index 0000000..549d743 --- /dev/null +++ b/c3r/http_service.py @@ -0,0 +1,184 @@ +"""Authenticated loopback HTTP boundary for recommendation-only C3R hosting. + +TLS termination and the request factory belong to the deployment host. This API +never accepts caller-supplied verification, approval, policy, or cost estimates. +""" + +from __future__ import annotations + +import hmac +import json +import time +from collections.abc import Callable, Mapping +from dataclasses import asdict, is_dataclass +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from threading import Lock +from typing import Protocol, cast + +from .deliberative.envelope import DeliberativeResult +from .runtime import RuntimeRequest, StandaloneController + + +MAX_REQUEST_BYTES = 65_536 + + +class RequestFactory(Protocol): + """Construct trusted policies, catalogs, and measured estimates server-side.""" + + def build(self, payload: Mapping[str, object]) -> RuntimeRequest: ... + + +class TokenBucket: + def __init__( + self, + *, + capacity: int, + refill_per_second: float, + clock: Callable[[], float] = time.monotonic, + ) -> None: + if capacity < 1 or refill_per_second <= 0: + raise ValueError("rate limit must be positive") + self._capacity = capacity + self._refill = refill_per_second + self._clock = clock + self._tokens = float(capacity) + self._last = clock() + self._lock = Lock() + + def take(self) -> bool: + with self._lock: + now = self._clock() + self._tokens = min( + self._capacity, self._tokens + max(0.0, now - self._last) * self._refill + ) + self._last = now + if self._tokens < 1: + return False + self._tokens -= 1 + return True + + +class ServiceMetrics: + def __init__(self) -> None: + self._lock = Lock() + self._counts: dict[str, int] = {} + + def increment(self, name: str) -> None: + with self._lock: + self._counts[name] = self._counts.get(name, 0) + 1 + + def snapshot(self) -> dict[str, int]: + with self._lock: + return dict(self._counts) + + +class C3RHTTPServer(ThreadingHTTPServer): + daemon_threads = True + + def __init__( + self, + *, + runtime: StandaloneController, + request_factory: RequestFactory, + bearer_token: str, + host: str = "127.0.0.1", + port: int = 8081, + requests_per_minute: int = 60, + ) -> None: + if host not in {"127.0.0.1", "::1", "localhost"}: + raise ValueError("C3R must bind to loopback behind a TLS gateway") + if len(bearer_token) < 32: + raise ValueError("bearer token must have at least 32 characters") + if runtime.effect_execution_enabled: + raise ValueError("the HTTP service cannot execute external effects") + self.runtime = runtime + self.request_factory = request_factory + self.bearer_token = bearer_token + self.limiter = TokenBucket( + capacity=requests_per_minute, + refill_per_second=requests_per_minute / 60, + ) + self.metrics = ServiceMetrics() + super().__init__((host, port), _Handler) + + +class _Handler(BaseHTTPRequestHandler): + server: C3RHTTPServer + + def log_message(self, _format: str, *_args: object) -> None: + # The host exports aggregate metrics; request bodies and tokens are never logged. + return + + def _send(self, status: int, value: Mapping[str, object]) -> None: + body = json.dumps(value, sort_keys=True, separators=(",", ":")).encode("utf-8") + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Cache-Control", "no-store") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + def _authorized(self) -> bool: + expected = "Bearer " + self.server.bearer_token + supplied = self.headers.get("Authorization", "") + return hmac.compare_digest(expected, supplied) + + def do_GET(self) -> None: + if self.path == "/health": + self._send(200, {"status": "ok"}) + return + if self.path == "/metrics": + if not self._authorized(): + self._send(401, {"error": "unauthorized"}) + return + self._send(200, {"counts": self.server.metrics.snapshot()}) + return + self._send(404, {"error": "not_found"}) + + def do_POST(self) -> None: + if self.path != "/v1/decisions": + self._send(404, {"error": "not_found"}) + return + if not self._authorized(): + self.server.metrics.increment("unauthorized") + self._send(401, {"error": "unauthorized"}) + return + if not self.server.limiter.take(): + self.server.metrics.increment("rate_limited") + self._send(429, {"error": "rate_limited"}) + return + try: + length = int(self.headers.get("Content-Length", "")) + except ValueError: + self._send(400, {"error": "invalid_content_length"}) + return + if length < 1 or length > MAX_REQUEST_BYTES: + self._send(413, {"error": "request_size_out_of_bounds"}) + return + try: + payload = json.loads(self.rfile.read(length)) + if not isinstance(payload, dict): + raise ValueError("JSON object required") + request = self.server.request_factory.build(cast(dict[str, object], payload)) + outcome = self.server.runtime.run(request) + except (KeyError, TypeError, ValueError, json.JSONDecodeError): + self.server.metrics.increment("invalid_request") + self._send(400, {"error": "invalid_request"}) + return + except (OSError, RuntimeError): + self.server.metrics.increment("internal_failure") + self._send(503, {"error": "service_unavailable"}) + return + self.server.metrics.increment("decisions") + result: dict[str, object] = { + "route": outcome.route, + "selected_action_id": outcome.selected_action_id, + "authority_result": outcome.authority_result, + "reason": outcome.reason, + "trace_hash": outcome.ledger_record.record_hash, + } + if isinstance(outcome.deliberation, DeliberativeResult) and is_dataclass( + outcome.deliberation + ): + result["deliberation"] = asdict(outcome.deliberation) + self._send(200, result) diff --git a/c3r/runtime.py b/c3r/runtime.py new file mode 100644 index 0000000..eab15dc --- /dev/null +++ b/c3r/runtime.py @@ -0,0 +1,270 @@ +"""Host-owned composition of the C3R decision and authority paths. + +The controller accepts bounded inputs from a trusted host. It does not expose an +HTTP endpoint or grant a remote caller permission to execute an action. +""" + +from __future__ import annotations + +import hashlib +import json +from collections.abc import Callable, Mapping +from dataclasses import asdict, dataclass +from typing import Protocol + +from .candidate_compiler import CandidateCompiler +from .commit_gateway import ApprovalGrant, CommitRequest, TrustedCommitGateway +from .cvoc import RobustCvocController +from .feature_flags import FeatureFlags +from .state_compiler import StateCompiler +from .state_schema import ( + ActionCandidate, + ActionDefinition, + ActionFamily, + AuthorityPolicy, + CompiledState, + RawState, + ValueEstimate, +) +from .system_one.fast_path import FastPathDecision, LayaFastPath +from .system_one.question_registry import TypedQuestion +from .telemetry.trace import DecisionTrace +from .telemetry.trace_ledger import LedgerRecord, TraceLedger +from .verifier_firewall import VerifierFirewall + + +class Deliberator(Protocol): + def deliberate(self, state: CompiledState) -> object: ... + + +@dataclass(frozen=True, slots=True) +class RuntimeRequest: + raw_state: RawState + definitions: tuple[ActionDefinition, ...] + policy: AuthorityPolicy + estimates: Mapping[str, ValueEstimate] + run_id: str + access_level: str = "internal" + language: str = "en" + + +@dataclass(frozen=True, slots=True) +class RuntimeOutcome: + route: str + selected_action_id: str | None + authority_result: str + reason: str + ledger_record: LedgerRecord + fast_path: FastPathDecision | None = None + deliberation: object | None = None + + +class StandaloneController: + """Run C3R with independent verification and an optional trusted executor. + + Estimates, the policy, verifier, gateway, and executor must be supplied by the + trusted host. The default is recommendation only. A learned proposal can never + supply an executor or a verification attestation through this interface. + """ + + def __init__( + self, + *, + flags: FeatureFlags, + compiler: StateCompiler, + candidates: CandidateCompiler, + cvoc: RobustCvocController, + verifier: VerifierFirewall, + gateway: TrustedCommitGateway, + ledger: TraceLedger, + fast_path: LayaFastPath | None = None, + deliberator: Deliberator | None = None, + executor: Callable[[ActionCandidate], None] | None = None, + ) -> None: + self._flags = flags + self._compiler = compiler + self._candidates = candidates + self._cvoc = cvoc + self._verifier = verifier + self._gateway = gateway + self._ledger = ledger + self._fast_path = fast_path + self._deliberator = deliberator + self._executor = executor + + @property + def effect_execution_enabled(self) -> bool: + return self._executor is not None + + def run( + self, request: RuntimeRequest, *, approval: ApprovalGrant | None = None + ) -> RuntimeOutcome: + if not request.run_id: + raise ValueError("run_id is required") + state_hash = hashlib.sha256( + json.dumps( + asdict(request.raw_state), + sort_keys=True, + separators=(",", ":"), + ensure_ascii=False, + allow_nan=False, + ).encode("utf-8") + ).hexdigest() + + def finish( + route: str, + reason: str, + *, + selected: ActionCandidate | None = None, + candidate_ids: tuple[str, ...] = (), + lower_bound: float | None = None, + authority_result: str = "not_attempted", + fast: FastPathDecision | None = None, + deliberation: object | None = None, + ) -> RuntimeOutcome: + probabilities = {} if fast is None else fast.probabilities + trace = DecisionTrace( + run_id=request.run_id, + state_hash=state_hash, + access_level=request.access_level, + model_provider=route, + candidate_ids=candidate_ids, + probabilities=probabilities, + utility_quantiles=( + {} if lower_bound is None else {"selected_lower_bound": lower_bound} + ), + selected_action_id=None if selected is None else selected.id, + authority_result=authority_result, + system_cost={}, + task_outcome={"status": reason}, + artifact_refs=(), + ) + return RuntimeOutcome( + route=route, + selected_action_id=None if selected is None else selected.id, + authority_result=authority_result, + reason=reason, + ledger_record=self._ledger.append(trace), + fast_path=fast, + deliberation=deliberation, + ) + + if not self._flags.enabled_requested: + return finish("deterministic", "C3R_DISABLED") + + compilation = self._compiler.compile(request.raw_state) + if compilation.state is None: + return finish("deterministic", compilation.status.value) + state = compilation.state + + remaining_budget = state.budget.get("remaining_usd", 0.0) + if remaining_budget < 0: + return finish("deterministic", "INVALID_BUDGET") + compiled = self._candidates.compile_hierarchical( + request.definitions, + request.policy, + remaining_budget=remaining_budget, + allowed_verifiers=self._verifier.available_verifier_ids, + ) + candidate_ids = tuple(item.id for item in compiled.candidates) + if compiled.no_safe_action: + return finish("deterministic", "NO_SAFE_ACTION", candidate_ids=candidate_ids) + + fast: FastPathDecision | None = None + if self._flags.system_one_enabled: + if self._fast_path is None: + return finish("deterministic", "SYSTEM_ONE_UNAVAILABLE", candidate_ids=candidate_ids) + questions = ( + TypedQuestion("STOP_NOW", ("NO", "YES")), + TypedQuestion("DELIBERATION_REQUIRED", ("NO", "YES")), + ) + try: + fast = self._fast_path.decide( + state, questions, action_family="CONTROL", language=request.language + ) + except (OSError, RuntimeError, TypeError, ValueError): + return finish("deterministic", "SYSTEM_ONE_FAILURE", candidate_ids=candidate_ids) + if fast.abstained: + return self._deliberate_or_stop( + state, finish, candidate_ids, "SYSTEM_ONE_ABSTAINED", fast + ) + if fast.answers.get("STOP_NOW") == "YES": + return finish("system_one", "STOP_NOW", candidate_ids=candidate_ids, fast=fast) + if fast.answers.get("DELIBERATION_REQUIRED") == "YES": + return self._deliberate_or_stop( + state, finish, candidate_ids, "DELIBERATION_REQUIRED", fast + ) + + decision = self._cvoc.select(compiled.candidates, request.estimates) + if decision.selected is None: + return finish( + "deterministic", "NON_POSITIVE_CVOC", candidate_ids=candidate_ids, fast=fast + ) + selected = decision.selected + if selected.family is ActionFamily.DELIBERATE: + return self._deliberate_or_stop( + state, finish, candidate_ids, "CVOC_SELECTED_DELIBERATION", fast + ) + try: + verification = self._verifier.verify(selected) + except (OSError, RuntimeError, TypeError, ValueError): + return finish( + "deterministic", "VERIFIER_FAILURE", candidate_ids=candidate_ids, fast=fast + ) + if not verification.accepted: + return finish( + "deterministic", "VERIFICATION_REJECTED", candidate_ids=candidate_ids, fast=fast + ) + if self._executor is None: + return finish( + "recommendation", + "VERIFIED_RECOMMENDATION", + selected=selected, + candidate_ids=candidate_ids, + lower_bound=decision.lower_bound, + authority_result="verified_not_committed", + fast=fast, + ) + try: + commit = self._gateway.commit( + CommitRequest(selected, verification, approval), self._executor + ) + except (OSError, RuntimeError, TypeError, ValueError): + return finish( + "deterministic", "COMMIT_FAILURE", candidate_ids=candidate_ids, fast=fast + ) + return finish( + "commit" if commit.committed else "deterministic", + commit.reason, + selected=selected if commit.committed else None, + candidate_ids=candidate_ids, + lower_bound=decision.lower_bound, + authority_result=commit.reason, + fast=fast, + ) + + def _deliberate_or_stop( + self, + state: CompiledState, + finish: Callable[..., RuntimeOutcome], + candidate_ids: tuple[str, ...], + reason: str, + fast: FastPathDecision | None, + ) -> RuntimeOutcome: + if not self._flags.deliberative_enabled or self._deliberator is None: + return finish( + "deterministic", reason + "_NO_PROVIDER", candidate_ids=candidate_ids, fast=fast + ) + try: + deliberation = self._deliberator.deliberate(state) + except (OSError, RuntimeError, TypeError, ValueError): + return finish( + "deterministic", "DELIBERATIVE_FAILURE", candidate_ids=candidate_ids, fast=fast + ) + return finish( + "deliberative", + reason, + candidate_ids=candidate_ids, + fast=fast, + deliberation=deliberation, + ) diff --git a/c3r/telemetry/trace_ledger.py b/c3r/telemetry/trace_ledger.py index 765e07b..ea142f7 100644 --- a/c3r/telemetry/trace_ledger.py +++ b/c3r/telemetry/trace_ledger.py @@ -5,6 +5,7 @@ import hashlib import json from dataclasses import asdict, dataclass +from threading import Lock from .trace import DecisionTrace @@ -29,10 +30,12 @@ def __init__(self, records: tuple[LedgerRecord, ...] = ()) -> None: if records and not self.verify(records): raise ValueError("trace ledger hash chain is invalid") self._records = list(records) + self._lock = Lock() @property def records(self) -> tuple[LedgerRecord, ...]: - return tuple(self._records) + with self._lock: + return tuple(self._records) def append(self, trace: DecisionTrace) -> LedgerRecord: canonical = json.dumps( @@ -42,15 +45,16 @@ def append(self, trace: DecisionTrace) -> LedgerRecord: ensure_ascii=False, allow_nan=False, ) - previous = self._records[-1].record_hash if self._records else _GENESIS_HASH - record = LedgerRecord(previous, _record_hash(previous, canonical), canonical) - self._records.append(record) - return record + with self._lock: + previous = self._records[-1].record_hash if self._records else _GENESIS_HASH + record = LedgerRecord(previous, _record_hash(previous, canonical), canonical) + self._records.append(record) + return record def to_jsonl(self) -> str: return "\n".join( json.dumps(asdict(record), sort_keys=True, separators=(",", ":")) - for record in self._records + for record in self.records ) @classmethod diff --git a/c3r/verifier_firewall.py b/c3r/verifier_firewall.py index 858232d..decde7b 100644 --- a/c3r/verifier_firewall.py +++ b/c3r/verifier_firewall.py @@ -40,6 +40,11 @@ def __init__( self._policy = policy self._attestation_key = attestation_key + @property + def available_verifier_ids(self) -> frozenset[str]: + """Identifiers configured by the trusted host, never by a proposal.""" + return frozenset(self._verifiers) + def verify(self, candidate: ActionCandidate) -> VerificationResult: verifier_id = self._policy.select(candidate) try: diff --git a/docs/directive-compliance.md b/docs/directive-compliance.md index 9ecab15..2af9582 100644 --- a/docs/directive-compliance.md +++ b/docs/directive-compliance.md @@ -10,7 +10,8 @@ evidence-bearing: - **blocked externally** - completion requires governed data, compute, credentials, MC-1 code, or another repository/runtime not present in this workspace. -The full directive is **not complete**. The current release is the first reviewed vertical slice. +The full directive is **not complete**. MC-1 integration is deferred for the standalone +launch only; it remains a directive gap. The current public release is a research alpha. | DoD | Requirement | Status | Evidence / remaining gate | | --- | --- | --- | --- | @@ -20,12 +21,12 @@ The full directive is **not complete**. The current release is the first reviewe | 25 | C3R-specific Laya checkpoint exists and is calibrated | blocked externally | no governed empirical corpus, training run, derived weights, or held-out calibration evidence | | 26 | checkpoint published with attribution | partial | Hugging Face integration-preview repository, license, NOTICE, paper links, and upstream manifest are published; no derived checkpoint exists | | 27 | empirical DecisionMix v1 published | partial | synthetic 144-record schema preview with deterministic splits and hashes is public; empirical provenance-audited corpus remains | -| 28 | Qwen/DeepSeek/frontier Deliberative Envelope works | partial | structured envelope plus tested OpenAI, Anthropic, Gemini, and OpenAI-compatible Qwen/DeepSeek/vLLM/SGLang/MC-1 request/response contracts exist; live provider qualification and paired end-to-end traces do not | -| 29 | CVoC uses measured runtime cost/risk | partial | conservative reference calculation and fallback tests exist; calibrated production inputs and end-to-end measured traces do not | -| 30 | Verifier Firewall and Commit Gateway independent | partial | independent reference modules, action-bound attestations, atomic expiring approvals, and bypass tests; host-level complete mediation remains | +| 28 | Qwen/DeepSeek/frontier Deliberative Envelope works | partial | structured envelope plus tested OpenAI, Anthropic, Gemini, and OpenAI-compatible Qwen/DeepSeek/vLLM/SGLang/MC-1 request/response contracts exist; a published OpenRouter DeepSeek probe and private localhost GPU smoke test exist, but live paired C3R traces and Qwen/frontier qualification do not | +| 29 | CVoC uses measured runtime cost/risk | partial | conservative reference calculation, fallback tests, and integrated controller path exist; calibrated production inputs and end-to-end measured traces do not | +| 30 | Verifier Firewall and Commit Gateway independent | partial | independent reference modules, action-bound attestations, atomic expiring approvals, and integrated recommendation-only HTTP boundary tests; host-level complete mediation remains | | 31 | OpenAI-compatible MC-1 API supports System-One | not started | the private `MC-1-platform` repository is identified and accessible, but the request contract, controller service, traces, and production qualification are not implemented | | 32 | MC-1 Console visualizes both paths | not started | the private `MC-1-platform` console is identified and accessible, but C3R trace ingestion, read models, and UI are not implemented | -| 33 | Colibri adapter runs in shadow mode | partial | non-authoritative recommendations consume the required telemetry shape and retain native fallback in tests; a dedicated `c3r-colibri` repository, compatible build, and live shadow traces remain | +| 33 | Colibri adapter runs in shadow mode | partial | non-authoritative recommendations consume the required telemetry shape and retain native fallback in tests; dedicated `c3r-colibri` repository and pinned build exist, but compatible live instrumentation and shadow traces remain | | 34 | calibration metrics reproduce from raw traces | partial | fitter and accuracy/Brier/ECE/MCE/NLL/selective-risk tests plus a canonical trace hash-chain integrity check exist; no empirical raw trace release pack or independently anchored ledger head | | 35 | all controller baselines compared identically | not started | requires immutable empirical test states and pinned serving environments | | 36 | no non-comparable TypeSafe Jev/RLCD claim | complete | repository and cards make no apples-to-apples Jev claim and preserve the caveat | @@ -35,7 +36,8 @@ The full directive is **not complete**. The current release is the first reviewe ## Directive-wide gaps outside the numbered DoD -- The public `c3r-evals` and `c3r-colibri` repositories are not present. +- The public `c3r-evals` and `c3r-colibri` repositories exist, but neither contains a + complete standalone production qualification pack. - Provider protocol contracts are shipped, but credentialed qualification for OpenAI, Anthropic, Gemini, Qwen, DeepSeek, vLLM, SGLang, llama.cpp, and MC-1 is not yet evidenced. - Required empirical tests for multilingual traffic, unseen schemas, provider outages, inventory @@ -45,6 +47,9 @@ The full directive is **not complete**. The current release is the first reviewe - No raw trace pack currently supports calibration, latency, cost, task-success, or calls-avoided claims. +The standalone launch gate and its external operator inputs are tracked in +[`standalone-launch.md`](standalone-launch.md). + Directive section 22's required prior-art links and canonical novelty boundary are collected in [`prior-art.md`](prior-art.md). diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md new file mode 100644 index 0000000..db0c4f3 --- /dev/null +++ b/docs/standalone-launch.md @@ -0,0 +1,35 @@ +# Standalone C3R production launch gate + +MC-1 product integration is explicitly out of scope for the **standalone** launch. It +remains in Execution Directive v2 and must not be marked complete there. This document +distinguishes tested code, private operational evidence, and public release evidence. + +| Gate | Current evidence | Exit condition | +| --- | --- | --- | +| Hosted decision path | `StandaloneController` composes state compilation, hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. Local integration tests pass. | Host-owned request factory and measured estimates, durable/anchored trace storage, TLS gateway, approved hostname, secret rotation, deployment and rollback rehearsal, monitoring/alerts, and live end-to-end traces. No public endpoint exists yet. | +| DeepSeek | Official DeepSeek-V4.1-Flash shards are verified on the GCP GPU host and in a private GCS backup; localhost inference passed a smoke test. | Secure service-to-service route and paired C3R decision traces. The localhost model is not a public API. | +| Empirical DecisionMix | Published preview has 144 synthetic records, deterministic splits, and hashes. | Approved trace source, consent/license/data-boundary review, deduplication, immutable split, provenance audit, and empirical dataset publication. Never relabel synthetic data empirical. | +| C3R Laya | Pinned upstream integration and calibration-aware abstention are tested. The Hugging Face C3R model page is an integration preview without trained weights. | Train a C3R-derived checkpoint on governed training data, fit calibration on held-out data, preserve sealed test set, and publish weights, raw predictions, manifests, hashes, and reproducible metrics. | +| Behavioral/safety qualification | Unit and small integration fixtures cover several failure boundaries; one published OpenRouter DeepSeek probe and a GPU smoke test exist. | Same-state baselines, task outcomes, latency/cost, calibration, abstention, outages, injection, unsafe actions, bypass, kill switch, confidence intervals, and independent reproduction. | +| Providers and Colibri | Provider protocol contracts and a non-authoritative Colibri shadow adapter exist. Dedicated `c3r-evals` and `c3r-colibri` repositories exist. | Credentialed Qwen/frontier qualification, compatible Colibri instrumentation, actual shadow route traces, offline replay, shadow traffic, read-only and reversible canaries. | +| Public release | v0.1 alpha repository, papers, collection, model/dataset previews. | Update all cards and launch copy only after the corresponding evidence passes; perform security and dependency review; publish an evidence-matched standalone release. | + +The current code path is a **tested reference boundary**, not a production service. +The in-memory trace ledger is not durable, the estimates are not yet empirically +calibrated, and the HTTP server requires a separate TLS/authentication gateway. Keep +effect execution disabled in this service until host-level complete mediation and +canary evidence are independently reviewed. + +## Required operator inputs + +1. Approved public hostname and DNS/TLS administration path. +2. A governed trace source with explicit data-use, retention, redaction, and publication + permissions. Controlled internal tasks can seed an evaluation set, but cannot be + passed off as representative customer traffic. +3. Qwen and frontier provider accounts/model IDs/quotas, supplied through the secrets + manager rather than committed files. +4. Colibri deployment owner and an instrumented compatible build that emits native + route, acceptance, latency, and cost traces without exporting private content. +5. A release owner to approve read-only and reversible canary thresholds and aborts. + +No public production claim should be made while any exit condition above is unmet. diff --git a/tests/test_cvoc.py b/tests/test_cvoc.py index 5117dad..82b3e62 100644 --- a/tests/test_cvoc.py +++ b/tests/test_cvoc.py @@ -37,7 +37,18 @@ def test_stops_when_all_lower_bounds_are_non_positive(self) -> None: self.assertIsNone(decision.selected) self.assertEqual(decision.fallback, "STOP") + def test_invalid_or_negative_cost_estimates_fail_closed(self) -> None: + candidate = ActionCandidate("local", ActionFamily.LOCAL_MODEL, RiskClass.READ_ONLY, 1.0) + for estimate in ( + ValueEstimate(float("nan"), 0.0, 0.0, 0.0), + ValueEstimate(1.0, -1.0, 0.0, 0.0), + ValueEstimate(1.0, 0.0, float("inf"), 0.0), + ): + with self.subTest(estimate=estimate): + decision = RobustCvocController().select((candidate,), {"local": estimate}) + self.assertIsNone(decision.selected) + self.assertEqual(decision.fallback, ActionFamily.STOP) + if __name__ == "__main__": unittest.main() - diff --git a/tests/test_http_service.py b/tests/test_http_service.py new file mode 100644 index 0000000..9bcbe4e --- /dev/null +++ b/tests/test_http_service.py @@ -0,0 +1,103 @@ +import json +import threading +import unittest +from urllib.error import HTTPError +from urllib.request import Request, urlopen + +from c3r.http_service import C3RHTTPServer +from tests.test_runtime import controller, request + + +TOKEN = "test-token-with-at-least-thirty-two-characters" + + +class HostFactory: + def build(self, payload): + if payload.get("goal") != "Find record": + raise ValueError("unknown goal") + # Client fields such as policy, estimates, approval and verifier are ignored. + return request() + + +class HTTPServiceTests(unittest.TestCase): + def setUp(self) -> None: + runtime, _ = controller() + self.server = C3RHTTPServer( + runtime=runtime, + request_factory=HostFactory(), + bearer_token=TOKEN, + port=0, + requests_per_minute=1, + ) + self.thread = threading.Thread(target=self.server.serve_forever, daemon=True) + self.thread.start() + self.base = f"http://127.0.0.1:{self.server.server_port}" + + def tearDown(self) -> None: + self.server.shutdown() + self.server.server_close() + self.thread.join(timeout=2) + + def post(self, payload, *, token=TOKEN): + body = json.dumps(payload).encode() + req = Request( + self.base + "/v1/decisions", + data=body, + method="POST", + headers={"Authorization": f"Bearer {token}", "Content-Type": "application/json"}, + ) + try: + with urlopen(req, timeout=2) as response: + return response.status, json.load(response) + except HTTPError as error: + return error.code, json.load(error) + + def test_authenticated_call_ignores_caller_authority_fields(self) -> None: + status, body = self.post( + { + "goal": "Find record", + "policy": {"allow_destructive": True}, + "estimates": {"lookup": 999999}, + "verifier": "self-approved", + "approval": "forged", + } + ) + self.assertEqual(status, 200) + self.assertEqual(body["reason"], "VERIFIED_RECOMMENDATION") + self.assertEqual(body["authority_result"], "verified_not_committed") + self.assertEqual(len(body["trace_hash"]), 64) + self.assertNotIn("The record exists", str(body)) + + def test_missing_authentication_is_rejected(self) -> None: + status, body = self.post({"goal": "Find record"}, token="wrong") + self.assertEqual((status, body["error"]), (401, "unauthorized")) + + def test_rate_limit_rejects_second_request(self) -> None: + self.assertEqual(self.post({"goal": "Find record"})[0], 200) + status, body = self.post({"goal": "Find record"}) + self.assertEqual((status, body["error"]), (429, "rate_limited")) + + def test_non_loopback_bind_is_rejected(self) -> None: + runtime, _ = controller() + with self.assertRaisesRegex(ValueError, "loopback"): + C3RHTTPServer( + runtime=runtime, + request_factory=HostFactory(), + bearer_token=TOKEN, + host="0.0.0.0", + port=0, + ) + + def test_effect_enabled_runtime_is_rejected(self) -> None: + runtime, _ = controller(executor=lambda _: None) + with self.assertRaisesRegex(ValueError, "external effects"): + C3RHTTPServer( + runtime=runtime, + request_factory=HostFactory(), + bearer_token=TOKEN, + port=0, + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_runtime.py b/tests/test_runtime.py new file mode 100644 index 0000000..8a4adcb --- /dev/null +++ b/tests/test_runtime.py @@ -0,0 +1,183 @@ +import unittest + +from c3r.candidate_compiler import CandidateCompiler +from c3r.commit_gateway import InMemoryApprovalNonceStore, TrustedCommitGateway +from c3r.cvoc import RobustCvocController +from c3r.feature_flags import FeatureFlags +from c3r.runtime import RuntimeRequest, StandaloneController +from c3r.state_compiler import StateCompiler +from c3r.state_schema import ( + ActionDefinition, + ActionFamily, + AuthorityPolicy, + Provenance, + RawState, + RiskClass, + ValueEstimate, +) +from c3r.system_one.calibration import TemperatureCalibrator +from c3r.system_one.fast_path import LayaFastPath +from c3r.system_one.laya_adapter import LayaAdapter +from c3r.telemetry.trace_ledger import TraceLedger +from c3r.verifier_firewall import VerifierDecision, VerifierFirewall, VerifierPolicy + + +KEY = b"verification-test-key" +ACTION_ID = "lookup:0:local:policy" + + +def request(*, risk: RiskClass = RiskClass.READ_ONLY, provenance: bool = True) -> RuntimeRequest: + fact = "The record exists" + raw = RawState( + goal="Find record", + current_subgoal="Lookup", + verified_facts=(fact,), + available_action_families=(ActionFamily.TOOL,), + budget={"remaining_usd": 1.0}, + provenance={fact: Provenance("fixture", "2026-09-22T00:00:00Z")} + if provenance + else {}, + ) + definition = ActionDefinition( + id="lookup", + family=ActionFamily.TOOL, + subgroup="records", + operation="get", + risk_class=risk, + argument_variants=((('record_id', 'fixture-1'),),), + placements=("local",), + verifier_ids=("policy",), + optimistic_utility=1.0, + estimated_cost=0.1, + data_boundary="local", + ) + return RuntimeRequest( + raw_state=raw, + definitions=(definition,), + policy=AuthorityPolicy( + frozenset({ActionFamily.TOOL}), frozenset({risk}) + ), + estimates={ACTION_ID: ValueEstimate(0.9, 0.1, 0.0, 0.1)}, + run_id="fixture-run", + ) + + +def controller( + *, + enabled: bool = True, + system_one: bool = False, + deliberative: bool = False, + accepted: bool = True, + executor=None, + fast_path=None, + deliberator=None, +) -> tuple[StandaloneController, TraceLedger]: + ledger = TraceLedger() + verifier = VerifierFirewall( + {"policy": lambda _: VerifierDecision(accepted, "policy fixture")}, + VerifierPolicy(default_verifier="policy"), + attestation_key=KEY, + ) + gateway = TrustedCommitGateway( + trusted_verifier_ids=frozenset({"policy"}), + verification_key=KEY, + approval_key=b"approval-test-key", + policy_version="policy-v1", + approval_nonce_store=InMemoryApprovalNonceStore(), + ) + runtime = StandaloneController( + flags=FeatureFlags( + enabled_requested=enabled, + system_one_requested=system_one, + deliberative_requested=deliberative, + ), + compiler=StateCompiler(), + candidates=CandidateCompiler(), + cvoc=RobustCvocController(), + verifier=verifier, + gateway=gateway, + ledger=ledger, + executor=executor, + fast_path=fast_path, + deliberator=deliberator, + ) + return runtime, ledger + + +class RuntimeTests(unittest.TestCase): + def test_verified_recommendation_has_no_effect_and_is_traced(self) -> None: + runtime, ledger = controller() + outcome = runtime.run(request()) + + self.assertEqual(outcome.route, "recommendation") + self.assertEqual(outcome.selected_action_id, ACTION_ID) + self.assertEqual(outcome.authority_result, "verified_not_committed") + self.assertEqual(len(ledger.records), 1) + self.assertTrue(TraceLedger.verify(ledger.records)) + self.assertNotIn("The record exists", ledger.to_jsonl()) + + def test_global_disable_stops_before_candidate_or_provider_execution(self) -> None: + effects = [] + runtime, _ = controller(enabled=False, executor=effects.append) + outcome = runtime.run(request()) + + self.assertEqual(outcome.reason, "C3R_DISABLED") + self.assertEqual(effects, []) + + def test_missing_provenance_stops_before_any_effect(self) -> None: + effects = [] + runtime, _ = controller(executor=effects.append) + outcome = runtime.run(request(provenance=False)) + + self.assertEqual(outcome.reason, "STATE_UNSAFE_TO_COMPRESS") + self.assertEqual(effects, []) + + def test_verifier_rejection_stops_before_commit(self) -> None: + effects = [] + runtime, _ = controller(accepted=False, executor=effects.append) + outcome = runtime.run(request()) + + self.assertEqual(outcome.reason, "VERIFICATION_REJECTED") + self.assertEqual(effects, []) + + def test_external_write_requires_independent_approval(self) -> None: + effects = [] + runtime, _ = controller(executor=effects.append) + outcome = runtime.run(request(risk=RiskClass.EXTERNAL_WRITE)) + + self.assertEqual(outcome.reason, "approval required") + self.assertEqual(effects, []) + + def test_uncalibrated_system_one_abstains_into_non_authoritative_deliberation(self) -> None: + adapter = LayaAdapter( + "convaiinnovations/laya", + "1c5edc17a7acd8701df6fc341c0d179f1c62c982", + backend=lambda _state, _questions: { + "STOP_NOW": (0.0, 3.0), + "DELIBERATION_REQUIRED": (3.0, 0.0), + }, + ) + fast_path = LayaFastPath(adapter=adapter, calibrator=TemperatureCalibrator({})) + + class Deliberator: + def deliberate(self, _state): + return {"plan": ["inspect"]} + + effects = [] + runtime, _ = controller( + system_one=True, + deliberative=True, + fast_path=fast_path, + deliberator=Deliberator(), + executor=effects.append, + ) + outcome = runtime.run(request()) + + self.assertEqual(outcome.route, "deliberative") + self.assertEqual(outcome.reason, "SYSTEM_ONE_ABSTAINED") + self.assertIsNone(outcome.selected_action_id) + self.assertEqual(effects, []) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_trace_ledger.py b/tests/test_trace_ledger.py index 6684936..63046b2 100644 --- a/tests/test_trace_ledger.py +++ b/tests/test_trace_ledger.py @@ -1,5 +1,6 @@ import json import unittest +from concurrent.futures import ThreadPoolExecutor from c3r.telemetry.trace import DecisionTrace from c3r.telemetry.trace_ledger import TraceLedger @@ -49,6 +50,14 @@ def test_serialized_ledger_round_trips(self) -> None: self.assertTrue(TraceLedger.verify(restored.records)) self.assertEqual(restored.records[0].record_hash, ledger.records[0].record_hash) + def test_concurrent_appends_keep_one_valid_chain(self) -> None: + ledger = TraceLedger() + with ThreadPoolExecutor(max_workers=8) as pool: + list(pool.map(lambda index: ledger.append(trace(f"run-{index}")), range(100))) + + self.assertEqual(len(ledger.records), 100) + self.assertTrue(TraceLedger.verify(ledger.records)) + if __name__ == "__main__": unittest.main() From ad395f25d8c34d7c48e9d6d4e826c13a5562631e Mon Sep 17 00:00:00 2001 From: wilkont Date: Tue, 22 Sep 2026 13:59:06 -0500 Subject: [PATCH 02/55] Bridge compiled state to provider adapters with data-boundary checks --- README.md | 2 +- c3r/deliberative/__init__.py | 22 ++++++--- c3r/deliberative/provider_bridge.py | 25 ++++++++++ docs/standalone-launch.md | 2 +- tests/test_provider_bridge.py | 72 +++++++++++++++++++++++++++++ 5 files changed, 114 insertions(+), 9 deletions(-) create mode 100644 c3r/deliberative/provider_bridge.py create mode 100644 tests/test_provider_bridge.py diff --git a/README.md b/README.md index 1a2eafb..f665361 100644 --- a/README.md +++ b/README.md @@ -65,7 +65,7 @@ failed verification into success, or directly commit an external effect. | Authority boundary | action-bound verifier attestations, expiring single-use approvals, atomic nonce claims | | Runtime control | conservative CVoC selection, deterministic `STOP`, cost twin, deliberative contracts, trace schema and in-memory hash chain | | Operational controls | fail-closed feature flags, tested frontier/open-weight HTTP contracts, Colibri shadow recommendations, canonical trace hash chain | -| Standalone controller boundary | tested state → candidates → optional Laya → CVoC → independent verifier → recommendation/commit fallback → redacted trace composition; authenticated rate-limited loopback HTTP boundary (not publicly deployed) | +| Standalone controller boundary | tested state → candidates → optional Laya → CVoC → independent verifier → recommendation/commit fallback → redacted trace composition; data-boundary-aware provider bridge and authenticated rate-limited loopback HTTP boundary (not publicly deployed) | This compiler is the reviewed vertical slice, not the directive's full Candidate Compiler Definition of Done. Rich typed value constraints, per-argument provenance, dominated-branch diff --git a/c3r/deliberative/__init__.py b/c3r/deliberative/__init__.py index e494e0c..d3d9a1c 100644 --- a/c3r/deliberative/__init__.py +++ b/c3r/deliberative/__init__.py @@ -1,13 +1,21 @@ """Interfaces for open and frontier deliberative models.""" +from importlib import import_module + from .envelope import DeliberativeEnvelope, DeliberativeResult -from .defaults import ( - DEFAULT_DEEPSEEK_API_MODEL, - DEFAULT_LANGUAGE_MODEL, - DEFAULT_OPENROUTER_MODEL, - DefaultModelProfile, - default_provider_config, -) + + +def __getattr__(name: str): + if name in { + "DEFAULT_DEEPSEEK_API_MODEL", + "DEFAULT_LANGUAGE_MODEL", + "DEFAULT_OPENROUTER_MODEL", + "DefaultModelProfile", + "default_provider_config", + }: + defaults = import_module(".defaults", __name__) + return getattr(defaults, name) + raise AttributeError(name) __all__ = [ "DEFAULT_DEEPSEEK_API_MODEL", diff --git a/c3r/deliberative/provider_bridge.py b/c3r/deliberative/provider_bridge.py new file mode 100644 index 0000000..9255a06 --- /dev/null +++ b/c3r/deliberative/provider_bridge.py @@ -0,0 +1,25 @@ +"""Data-boundary-aware bridge from compiled C3R state to provider adapters.""" + +from __future__ import annotations + +from dataclasses import asdict +from urllib.parse import urlparse + +from ..adapters.providers import DeliberationRequest, ProviderAdapter +from ..deliberative.envelope import DeliberativeResult +from ..state_schema import CompiledState + + +class ProviderDeliberator: + """Return a non-authoritative plan; never execute model-requested actions.""" + + def __init__(self, adapter: ProviderAdapter) -> None: + self._adapter = adapter + + def deliberate(self, state: CompiledState) -> DeliberativeResult: + hostname = urlparse(self._adapter.config.base_url).hostname + local = hostname in {"localhost", "127.0.0.1", "::1"} + if not local and state.data_boundary not in {"approved_remote", "public"}: + raise ValueError("state is not approved for a remote provider") + result = self._adapter.deliberate(DeliberationRequest(asdict(state))) + return result.deliberation diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index db0c4f3..87ae0ed 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -7,7 +7,7 @@ distinguishes tested code, private operational evidence, and public release evid | Gate | Current evidence | Exit condition | | --- | --- | --- | | Hosted decision path | `StandaloneController` composes state compilation, hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. Local integration tests pass. | Host-owned request factory and measured estimates, durable/anchored trace storage, TLS gateway, approved hostname, secret rotation, deployment and rollback rehearsal, monitoring/alerts, and live end-to-end traces. No public endpoint exists yet. | -| DeepSeek | Official DeepSeek-V4.1-Flash shards are verified on the GCP GPU host and in a private GCS backup; localhost inference passed a smoke test. | Secure service-to-service route and paired C3R decision traces. The localhost model is not a public API. | +| DeepSeek | Official DeepSeek-V4.1-Flash shards are verified on the GCP GPU host and in a private GCS backup; localhost inference passed a smoke test. A provider bridge passes compiled state to a local OpenAI-compatible adapter and rejects remote transfer of local-only data in tests. | Secure service-to-service route and paired live C3R decision traces. The localhost model is not a public API. | | Empirical DecisionMix | Published preview has 144 synthetic records, deterministic splits, and hashes. | Approved trace source, consent/license/data-boundary review, deduplication, immutable split, provenance audit, and empirical dataset publication. Never relabel synthetic data empirical. | | C3R Laya | Pinned upstream integration and calibration-aware abstention are tested. The Hugging Face C3R model page is an integration preview without trained weights. | Train a C3R-derived checkpoint on governed training data, fit calibration on held-out data, preserve sealed test set, and publish weights, raw predictions, manifests, hashes, and reproducible metrics. | | Behavioral/safety qualification | Unit and small integration fixtures cover several failure boundaries; one published OpenRouter DeepSeek probe and a GPU smoke test exist. | Same-state baselines, task outcomes, latency/cost, calibration, abstention, outages, injection, unsafe actions, bypass, kill switch, confidence intervals, and independent reproduction. | diff --git a/tests/test_provider_bridge.py b/tests/test_provider_bridge.py new file mode 100644 index 0000000..8c0b99e --- /dev/null +++ b/tests/test_provider_bridge.py @@ -0,0 +1,72 @@ +import json +import unittest + +from c3r.adapters.providers import ( + ProviderAdapter, + ProviderConfig, + ProviderKind, + TransportResponse, +) +from c3r.deliberative.provider_bridge import ProviderDeliberator +from c3r.state_compiler import StateCompiler +from c3r.state_schema import RawState + + +def state(data_boundary: str): + result = StateCompiler().compile( + RawState(goal="Plan a read-only check", current_subgoal="Inspect", data_boundary=data_boundary) + ) + assert result.state is not None + return result.state + + +class ProviderBridgeTests(unittest.TestCase): + def test_local_deepseek_receives_compiled_state_and_returns_plan(self) -> None: + captured = [] + + def transport(_url, _headers, payload): + captured.append(payload) + return TransportResponse( + 200, + { + "choices": [{"message": {"content": json.dumps({ + "plan": ["inspect"], + "assumptions": [], + "uncertainty": [], + "candidate_commitments": [], + "requested_actions": [], + })}}], + "usage": {"prompt_tokens": 12, "completion_tokens": 8}, + }, + 12.0, + ) + + adapter = ProviderAdapter( + ProviderConfig( + "deepseek-local", ProviderKind.OPENAI_COMPATIBLE, + "http://127.0.0.1:8000/v1", "deepseek-v4.1-flash", None, + ), + transport=transport, + ) + result = ProviderDeliberator(adapter).deliberate(state("local")) + + self.assertEqual(result.plan, ("inspect",)) + self.assertIn("Plan a read-only check", captured[0]["messages"][1]["content"]) + + def test_remote_provider_is_not_called_for_local_data(self) -> None: + calls = [] + adapter = ProviderAdapter( + ProviderConfig( + "frontier", ProviderKind.OPENAI_COMPATIBLE, + "https://example.invalid/v1", "frontier-model", "test-key", + ), + transport=lambda *_: calls.append(True), + ) + + with self.assertRaises(ValueError): + ProviderDeliberator(adapter).deliberate(state("local")) + self.assertEqual(calls, []) + + +if __name__ == "__main__": + unittest.main() From 076cf54de33cc94c149dd8770cb1108f3959049d Mon Sep 17 00:00:00 2001 From: wilkont Date: Tue, 22 Sep 2026 14:06:32 -0500 Subject: [PATCH 03/55] Persist redacted trace chain transactionally across restarts --- README.md | 2 +- c3r/runtime.py | 8 +++- c3r/telemetry/sqlite_ledger.py | 76 ++++++++++++++++++++++++++++++++++ c3r/telemetry/trace_ledger.py | 18 ++++---- docs/standalone-launch.md | 7 ++-- tests/test_sqlite_ledger.py | 44 ++++++++++++++++++++ 6 files changed, 142 insertions(+), 13 deletions(-) create mode 100644 c3r/telemetry/sqlite_ledger.py create mode 100644 tests/test_sqlite_ledger.py diff --git a/README.md b/README.md index f665361..39f0acc 100644 --- a/README.md +++ b/README.md @@ -63,7 +63,7 @@ failed verification into success, or directly commit an external effect. | Laya fast path | exact upstream revision and license verification, typed probabilities, slice calibration, confidence/margin abstention | | DecisionMix v1 | validated records, immutable deterministic splits, source/license provenance, SHA-256 manifest, 144-record synthetic preview | | Authority boundary | action-bound verifier attestations, expiring single-use approvals, atomic nonce claims | -| Runtime control | conservative CVoC selection, deterministic `STOP`, cost twin, deliberative contracts, trace schema and in-memory hash chain | +| Runtime control | conservative CVoC selection, deterministic `STOP`, cost twin, deliberative contracts, trace schema and optional transactional SQLite hash chain | | Operational controls | fail-closed feature flags, tested frontier/open-weight HTTP contracts, Colibri shadow recommendations, canonical trace hash chain | | Standalone controller boundary | tested state → candidates → optional Laya → CVoC → independent verifier → recommendation/commit fallback → redacted trace composition; data-boundary-aware provider bridge and authenticated rate-limited loopback HTTP boundary (not publicly deployed) | diff --git a/c3r/runtime.py b/c3r/runtime.py index eab15dc..672dad1 100644 --- a/c3r/runtime.py +++ b/c3r/runtime.py @@ -29,7 +29,7 @@ from .system_one.fast_path import FastPathDecision, LayaFastPath from .system_one.question_registry import TypedQuestion from .telemetry.trace import DecisionTrace -from .telemetry.trace_ledger import LedgerRecord, TraceLedger +from .telemetry.trace_ledger import LedgerRecord from .verifier_firewall import VerifierFirewall @@ -37,6 +37,10 @@ class Deliberator(Protocol): def deliberate(self, state: CompiledState) -> object: ... +class TraceSink(Protocol): + def append(self, trace: DecisionTrace) -> LedgerRecord: ... + + @dataclass(frozen=True, slots=True) class RuntimeRequest: raw_state: RawState @@ -76,7 +80,7 @@ def __init__( cvoc: RobustCvocController, verifier: VerifierFirewall, gateway: TrustedCommitGateway, - ledger: TraceLedger, + ledger: TraceSink, fast_path: LayaFastPath | None = None, deliberator: Deliberator | None = None, executor: Callable[[ActionCandidate], None] | None = None, diff --git a/c3r/telemetry/sqlite_ledger.py b/c3r/telemetry/sqlite_ledger.py new file mode 100644 index 0000000..5e171c0 --- /dev/null +++ b/c3r/telemetry/sqlite_ledger.py @@ -0,0 +1,76 @@ +"""Transactional single-host trace ledger with restart verification.""" + +from __future__ import annotations + +import sqlite3 +import os +import stat +from pathlib import Path +from threading import Lock + +from .trace import DecisionTrace +from .trace_ledger import LedgerRecord, TraceLedger, _record_hash, canonical_trace_json + + +class SqliteTraceLedger: + """Persist redacted traces atomically in a host-owned SQLite database. + + The operator must place the database on durable, access-controlled storage and + back it up. This is a single-host ledger, not an independently anchored audit log. + """ + + def __init__(self, path: Path) -> None: + if not path.parent.is_dir() or path.is_symlink(): + raise ValueError("ledger parent must exist and path must not be a symlink") + if not path.exists(): + descriptor = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600) + os.close(descriptor) + if os.name == "posix" and stat.S_IMODE(path.stat().st_mode) & 0o077: + raise ValueError("ledger file must not be accessible to group or other users") + self._lock = Lock() + self._db = sqlite3.connect(path, timeout=30, isolation_level=None, check_same_thread=False) + self._db.execute("PRAGMA journal_mode=WAL") + self._db.execute("PRAGMA synchronous=FULL") + self._db.execute( + "CREATE TABLE IF NOT EXISTS records (" + "sequence INTEGER PRIMARY KEY AUTOINCREMENT, " + "previous_hash TEXT NOT NULL, record_hash TEXT NOT NULL, " + "canonical_json TEXT NOT NULL)" + ) + if not TraceLedger.verify(self.records): + self._db.close() + raise ValueError("persisted trace ledger hash chain is invalid") + + @property + def records(self) -> tuple[LedgerRecord, ...]: + with self._lock: + rows = self._db.execute( + "SELECT previous_hash, record_hash, canonical_json " + "FROM records ORDER BY sequence" + ).fetchall() + return tuple(LedgerRecord(*row) for row in rows) + + def append(self, trace: DecisionTrace) -> LedgerRecord: + canonical = canonical_trace_json(trace) + with self._lock: + self._db.execute("BEGIN IMMEDIATE") + try: + row = self._db.execute( + "SELECT record_hash FROM records ORDER BY sequence DESC LIMIT 1" + ).fetchone() + previous = row[0] if row is not None else "0" * 64 + record = LedgerRecord(previous, _record_hash(previous, canonical), canonical) + self._db.execute( + "INSERT INTO records (previous_hash, record_hash, canonical_json) " + "VALUES (?, ?, ?)", + (record.previous_hash, record.record_hash, record.canonical_json), + ) + self._db.execute("COMMIT") + except Exception: + self._db.execute("ROLLBACK") + raise + return record + + def close(self) -> None: + with self._lock: + self._db.close() diff --git a/c3r/telemetry/trace_ledger.py b/c3r/telemetry/trace_ledger.py index ea142f7..816c8f2 100644 --- a/c3r/telemetry/trace_ledger.py +++ b/c3r/telemetry/trace_ledger.py @@ -25,6 +25,16 @@ def _record_hash(previous_hash: str, canonical_json: str) -> str: ).hexdigest() +def canonical_trace_json(trace: DecisionTrace) -> str: + return json.dumps( + asdict(trace), + sort_keys=True, + separators=(",", ":"), + ensure_ascii=False, + allow_nan=False, + ) + + class TraceLedger: def __init__(self, records: tuple[LedgerRecord, ...] = ()) -> None: if records and not self.verify(records): @@ -38,13 +48,7 @@ def records(self) -> tuple[LedgerRecord, ...]: return tuple(self._records) def append(self, trace: DecisionTrace) -> LedgerRecord: - canonical = json.dumps( - asdict(trace), - sort_keys=True, - separators=(",", ":"), - ensure_ascii=False, - allow_nan=False, - ) + canonical = canonical_trace_json(trace) with self._lock: previous = self._records[-1].record_hash if self._records else _GENESIS_HASH record = LedgerRecord(previous, _record_hash(previous, canonical), canonical) diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index 87ae0ed..8392bd1 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -6,7 +6,7 @@ distinguishes tested code, private operational evidence, and public release evid | Gate | Current evidence | Exit condition | | --- | --- | --- | -| Hosted decision path | `StandaloneController` composes state compilation, hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. Local integration tests pass. | Host-owned request factory and measured estimates, durable/anchored trace storage, TLS gateway, approved hostname, secret rotation, deployment and rollback rehearsal, monitoring/alerts, and live end-to-end traces. No public endpoint exists yet. | +| Hosted decision path | `StandaloneController` composes state compilation, hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. A transactional SQLite trace sink survives restart and verifies its hash chain. Local integration tests pass. | Host-owned request factory and measured estimates, durable storage placement, independent ledger-head anchoring, TLS gateway, approved hostname, secret rotation, deployment and rollback rehearsal, monitoring/alerts, and live end-to-end traces. No public endpoint exists yet. | | DeepSeek | Official DeepSeek-V4.1-Flash shards are verified on the GCP GPU host and in a private GCS backup; localhost inference passed a smoke test. A provider bridge passes compiled state to a local OpenAI-compatible adapter and rejects remote transfer of local-only data in tests. | Secure service-to-service route and paired live C3R decision traces. The localhost model is not a public API. | | Empirical DecisionMix | Published preview has 144 synthetic records, deterministic splits, and hashes. | Approved trace source, consent/license/data-boundary review, deduplication, immutable split, provenance audit, and empirical dataset publication. Never relabel synthetic data empirical. | | C3R Laya | Pinned upstream integration and calibration-aware abstention are tested. The Hugging Face C3R model page is an integration preview without trained weights. | Train a C3R-derived checkpoint on governed training data, fit calibration on held-out data, preserve sealed test set, and publish weights, raw predictions, manifests, hashes, and reproducible metrics. | @@ -15,8 +15,9 @@ distinguishes tested code, private operational evidence, and public release evid | Public release | v0.1 alpha repository, papers, collection, model/dataset previews. | Update all cards and launch copy only after the corresponding evidence passes; perform security and dependency review; publish an evidence-matched standalone release. | The current code path is a **tested reference boundary**, not a production service. -The in-memory trace ledger is not durable, the estimates are not yet empirically -calibrated, and the HTTP server requires a separate TLS/authentication gateway. Keep +The SQLite ledger is optional and does not by itself provide independent audit anchoring; +the estimates are not yet empirically calibrated, and the HTTP server requires a +separate TLS/authentication gateway. Keep effect execution disabled in this service until host-level complete mediation and canary evidence are independently reviewed. diff --git a/tests/test_sqlite_ledger.py b/tests/test_sqlite_ledger.py new file mode 100644 index 0000000..bb98661 --- /dev/null +++ b/tests/test_sqlite_ledger.py @@ -0,0 +1,44 @@ +import sqlite3 +import tempfile +import unittest +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path + +from c3r.telemetry.sqlite_ledger import SqliteTraceLedger +from c3r.telemetry.trace_ledger import TraceLedger +from tests.test_trace_ledger import trace + + +class SqliteTraceLedgerTests(unittest.TestCase): + def test_records_survive_restart_and_concurrent_appends(self) -> None: + with tempfile.TemporaryDirectory() as directory: + path = Path(directory) / "traces.sqlite3" + ledger = SqliteTraceLedger(path) + with ThreadPoolExecutor(max_workers=8) as pool: + list(pool.map(lambda i: ledger.append(trace(f"run-{i}")), range(100))) + ledger.close() + + reopened = SqliteTraceLedger(path) + self.assertEqual(len(reopened.records), 100) + self.assertTrue(TraceLedger.verify(reopened.records)) + reopened.close() + + def test_tampering_is_detected_on_restart(self) -> None: + with tempfile.TemporaryDirectory() as directory: + path = Path(directory) / "traces.sqlite3" + ledger = SqliteTraceLedger(path) + ledger.append(trace("run-1")) + ledger.close() + db = sqlite3.connect(path) + try: + db.execute("UPDATE records SET record_hash = ? WHERE sequence = 1", ("0" * 64,)) + db.commit() + finally: + db.close() + + with self.assertRaises(ValueError): + SqliteTraceLedger(path) + + +if __name__ == "__main__": + unittest.main() From 76ae74dd77de180e27a05061a97ccc6d51b900ae Mon Sep 17 00:00:00 2001 From: wilkont Date: Tue, 22 Sep 2026 14:10:49 -0500 Subject: [PATCH 04/55] Constrain hosted requests and record measured provider usage --- README.md | 2 +- c3r/adapters/providers.py | 4 +- c3r/deliberative/provider_bridge.py | 8 ++- c3r/host_factory.py | 81 +++++++++++++++++++++++++++++ c3r/runtime.py | 25 +++++++-- docs/runtime-integrations.md | 18 ++++--- docs/standalone-launch.md | 2 +- tests/test_host_factory.py | 50 ++++++++++++++++++ tests/test_provider_bridge.py | 3 +- tests/test_runtime.py | 46 ++++++++++++++++ 10 files changed, 220 insertions(+), 19 deletions(-) create mode 100644 c3r/host_factory.py create mode 100644 tests/test_host_factory.py diff --git a/README.md b/README.md index 39f0acc..8c3a2aa 100644 --- a/README.md +++ b/README.md @@ -65,7 +65,7 @@ failed verification into success, or directly commit an external effect. | Authority boundary | action-bound verifier attestations, expiring single-use approvals, atomic nonce claims | | Runtime control | conservative CVoC selection, deterministic `STOP`, cost twin, deliberative contracts, trace schema and optional transactional SQLite hash chain | | Operational controls | fail-closed feature flags, tested frontier/open-weight HTTP contracts, Colibri shadow recommendations, canonical trace hash chain | -| Standalone controller boundary | tested state → candidates → optional Laya → CVoC → independent verifier → recommendation/commit fallback → redacted trace composition; data-boundary-aware provider bridge and authenticated rate-limited loopback HTTP boundary (not publicly deployed) | +| Standalone controller boundary | tested state → candidates → optional Laya → CVoC → independent verifier → recommendation/commit fallback → redacted trace composition; host-owned read-only request factory, data-boundary-aware provider bridge, and authenticated rate-limited loopback HTTP boundary (not publicly deployed) | This compiler is the reviewed vertical slice, not the directive's full Candidate Compiler Definition of Done. Rich typed value constraints, per-argument provenance, dominated-branch diff --git a/c3r/adapters/providers.py b/c3r/adapters/providers.py index a33fe28..2605d5e 100644 --- a/c3r/adapters/providers.py +++ b/c3r/adapters/providers.py @@ -3,6 +3,7 @@ from __future__ import annotations import json +from math import isfinite from collections.abc import Callable, Mapping from dataclasses import dataclass from enum import StrEnum @@ -85,7 +86,8 @@ class ControlledDeliberation: def _number(value: object) -> float: if isinstance(value, bool) or not isinstance(value, (int, float)): return 0.0 - return float(value) + number = float(value) + return number if isfinite(number) and number >= 0 else 0.0 def _string_tuple(value: object) -> tuple[str, ...]: diff --git a/c3r/deliberative/provider_bridge.py b/c3r/deliberative/provider_bridge.py index 9255a06..d7aec8a 100644 --- a/c3r/deliberative/provider_bridge.py +++ b/c3r/deliberative/provider_bridge.py @@ -5,8 +5,7 @@ from dataclasses import asdict from urllib.parse import urlparse -from ..adapters.providers import DeliberationRequest, ProviderAdapter -from ..deliberative.envelope import DeliberativeResult +from ..adapters.providers import DeliberationRequest, ProviderAdapter, ProviderExecutionResult from ..state_schema import CompiledState @@ -16,10 +15,9 @@ class ProviderDeliberator: def __init__(self, adapter: ProviderAdapter) -> None: self._adapter = adapter - def deliberate(self, state: CompiledState) -> DeliberativeResult: + def deliberate(self, state: CompiledState) -> ProviderExecutionResult: hostname = urlparse(self._adapter.config.base_url).hostname local = hostname in {"localhost", "127.0.0.1", "::1"} if not local and state.data_boundary not in {"approved_remote", "public"}: raise ValueError("state is not approved for a remote provider") - result = self._adapter.deliberate(DeliberationRequest(asdict(state))) - return result.deliberation + return self._adapter.deliberate(DeliberationRequest(asdict(state))) diff --git a/c3r/host_factory.py b/c3r/host_factory.py new file mode 100644 index 0000000..eccf63f --- /dev/null +++ b/c3r/host_factory.py @@ -0,0 +1,81 @@ +"""Read-only host construction of bounded standalone decision requests.""" + +from __future__ import annotations + +from collections.abc import Callable, Mapping +from math import isfinite +from uuid import uuid4 + +from .runtime import RuntimeRequest +from .state_schema import ActionDefinition, AuthorityPolicy, RawState, RiskClass, ValueEstimate + + +EstimateSource = Callable[[RawState], Mapping[str, ValueEstimate]] + + +class ReadOnlyRequestFactory: + """Accept task text only; all authority and estimates come from the host. + + The estimate source is mandatory because constant/example estimates must not be + mistaken for measured production CVoC inputs. + """ + + def __init__( + self, + *, + definitions: tuple[ActionDefinition, ...], + policy: AuthorityPolicy, + estimate_source: EstimateSource, + remaining_usd: float, + model_inventory: tuple[str, ...] = (), + data_boundary: str = "local", + ) -> None: + if not definitions or any(item.risk_class is not RiskClass.READ_ONLY for item in definitions): + raise ValueError("HTTP catalog must contain read-only definitions only") + if policy.allowed_risks != frozenset({RiskClass.READ_ONLY}): + raise ValueError("HTTP policy must permit read-only risk only") + if not isfinite(remaining_usd) or remaining_usd < 0: + raise ValueError("remaining_usd must be finite and non-negative") + if data_boundary not in {"local", "approved_remote", "public"}: + raise ValueError("host data boundary must be explicit") + self._definitions = definitions + self._policy = policy + self._estimate_source = estimate_source + self._remaining_usd = remaining_usd + self._model_inventory = model_inventory + self._data_boundary = data_boundary + + def build(self, payload: Mapping[str, object]) -> RuntimeRequest: + goal = _text(payload.get("goal"), "goal") + current_subgoal = _text(payload.get("current_subgoal"), "current_subgoal") + questions = payload.get("open_questions", []) + if not isinstance(questions, list) or len(questions) > 64: + raise ValueError("open_questions must be a bounded array") + open_questions = tuple(_text(item, "open_question") for item in questions) + raw = RawState( + goal=goal, + current_subgoal=current_subgoal, + open_questions=open_questions, + available_action_families=tuple( + sorted(self._policy.allowed_families, key=lambda family: family.value) + ), + model_inventory=self._model_inventory, + budget={"remaining_usd": self._remaining_usd}, + data_boundary=self._data_boundary, + ) + estimates = self._estimate_source(raw) + if not isinstance(estimates, Mapping): + raise ValueError("host estimate source returned an invalid mapping") + return RuntimeRequest( + raw_state=raw, + definitions=self._definitions, + policy=self._policy, + estimates=estimates, + run_id=str(uuid4()), + ) + + +def _text(value: object, name: str) -> str: + if not isinstance(value, str) or not value.strip() or len(value) > 4096: + raise ValueError(f"{name} must be non-empty bounded text") + return value diff --git a/c3r/runtime.py b/c3r/runtime.py index 672dad1..f9e4f0c 100644 --- a/c3r/runtime.py +++ b/c3r/runtime.py @@ -8,10 +8,12 @@ import hashlib import json +from math import isfinite from collections.abc import Callable, Mapping from dataclasses import asdict, dataclass from typing import Protocol +from .adapters.providers import ProviderExecutionResult from .candidate_compiler import CandidateCompiler from .commit_gateway import ApprovalGrant, CommitRequest, TrustedCommitGateway from .cvoc import RobustCvocController @@ -125,13 +127,15 @@ def finish( authority_result: str = "not_attempted", fast: FastPathDecision | None = None, deliberation: object | None = None, + system_cost: Mapping[str, float] | None = None, + provider_id: str | None = None, ) -> RuntimeOutcome: probabilities = {} if fast is None else fast.probabilities trace = DecisionTrace( run_id=request.run_id, state_hash=state_hash, access_level=request.access_level, - model_provider=route, + model_provider=provider_id or route, candidate_ids=candidate_ids, probabilities=probabilities, utility_quantiles=( @@ -139,7 +143,7 @@ def finish( ), selected_action_id=None if selected is None else selected.id, authority_result=authority_result, - system_cost={}, + system_cost={} if system_cost is None else system_cost, task_outcome={"status": reason}, artifact_refs=(), ) @@ -165,7 +169,10 @@ def finish( if remaining_budget < 0: return finish("deterministic", "INVALID_BUDGET") compiled = self._candidates.compile_hierarchical( - request.definitions, + ( + definition for definition in request.definitions + if definition.family in state.available_action_families + ), request.policy, remaining_budget=remaining_budget, allowed_verifiers=self._verifier.available_verifier_ids, @@ -265,6 +272,18 @@ def _deliberate_or_stop( return finish( "deterministic", "DELIBERATIVE_FAILURE", candidate_ids=candidate_ids, fast=fast ) + if isinstance(deliberation, ProviderExecutionResult): + cost = deliberation.observed_cost + if any(not isfinite(value) or value < 0 for value in cost.values()): + return finish( + "deterministic", "DELIBERATIVE_INVALID_COST", candidate_ids=candidate_ids, + fast=fast, + ) + return finish( + "deliberative", reason, candidate_ids=candidate_ids, fast=fast, + deliberation=deliberation.deliberation, system_cost=cost, + provider_id=deliberation.provider_id, + ) return finish( "deliberative", reason, diff --git a/docs/runtime-integrations.md b/docs/runtime-integrations.md index 69ce7c3..1e6f8af 100644 --- a/docs/runtime-integrations.md +++ b/docs/runtime-integrations.md @@ -18,10 +18,10 @@ The release default is DeepSeek V4.1 Flash. Canonical identifiers are: - OpenRouter: `deepseek/deepseek-v4.1-flash` - DeepSeek API: `deepseek-flash` -This is a provider default, not execution authority and not a claim that the full checkpoint -fits the current GCP instance. `c3r.deliberative.default_provider_config` constructs either -hosted profile without embedding credentials. A self-hosted profile is enabled only after a -hardware manifest and live inference evidence demonstrate compatibility. +This is a provider default, not execution authority. The official checkpoint has passed a +private 8×H100 GCP localhost inference smoke test and has a verified private GCS backup; +neither result qualifies a public C3R endpoint. `c3r.deliberative.default_provider_config` +constructs hosted profiles without embedding credentials. `c3r.adapters.providers.ProviderAdapter` supports four protocol shapes: @@ -39,6 +39,8 @@ selected by host policy. API keys are supplied by the host and are absent from r These are protocol-level contracts. A provider becomes release-qualified only after credentialed tests capture model revision, region, latency, usage, failure, and fallback evidence. +`c3r.deliberative.provider_bridge.ProviderDeliberator` connects compiled state to the adapter +while rejecting local-only state for remote endpoints. Its structured plan has no commit authority. ## Colibri shadow control @@ -47,13 +49,15 @@ verification width, expert/cache/I/O/utilization, and context metrics. Its outpu `authoritative=False`. Low draft acceptance recommends `TARGET_ONLY`; controller failure returns `NATIVE_FALLBACK`. Native token acceptance and MoE routing remain authoritative. -Live completion still requires a compatible Colibri build, the dedicated public integration -repository, immutable build hashes, and captured shadow traces. +The dedicated public integration repository and pinned build exist. Live completion still +requires compatible instrumentation, immutable route/build evidence, and captured shadow traces. ## Evidence ledger `c3r.telemetry.trace_ledger.TraceLedger` canonicalizes decision traces and links them with SHA-256. Replay rejects reordered, modified, non-canonical, or broken-chain records. The ledger supplies -an integrity check, not tamper proofing or empirical evidence by itself: evidence-grade use must +an integrity check, not tamper proofing or empirical evidence by itself. A transactional +`SqliteTraceLedger` survives restart and verifies its chain on open, but must be placed on +durable, access-controlled storage and backed up. Evidence-grade use must anchor or sign ledger heads in an independently controlled append-only store, and release traces must still be governed, redacted, licensed, and independently reproducible. diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index 8392bd1..b80d6a0 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -6,7 +6,7 @@ distinguishes tested code, private operational evidence, and public release evid | Gate | Current evidence | Exit condition | | --- | --- | --- | -| Hosted decision path | `StandaloneController` composes state compilation, hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. A transactional SQLite trace sink survives restart and verifies its hash chain. Local integration tests pass. | Host-owned request factory and measured estimates, durable storage placement, independent ledger-head anchoring, TLS gateway, approved hostname, secret rotation, deployment and rollback rehearsal, monitoring/alerts, and live end-to-end traces. No public endpoint exists yet. | +| Hosted decision path | `StandaloneController` composes state compilation, inventory-masked hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. Provider latency/token usage is recorded without granting authority. A host-owned read-only request factory ignores caller policy, budgets, estimates, and approvals. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. A transactional SQLite trace sink survives restart and verifies its hash chain. Local integration tests pass. | Calibrated *pre-decision* value/cost estimate source, durable storage placement, independent ledger-head anchoring, TLS gateway, approved hostname, secret rotation, deployment and rollback rehearsal, monitoring/alerts, and live end-to-end traces. No public endpoint exists yet. | | DeepSeek | Official DeepSeek-V4.1-Flash shards are verified on the GCP GPU host and in a private GCS backup; localhost inference passed a smoke test. A provider bridge passes compiled state to a local OpenAI-compatible adapter and rejects remote transfer of local-only data in tests. | Secure service-to-service route and paired live C3R decision traces. The localhost model is not a public API. | | Empirical DecisionMix | Published preview has 144 synthetic records, deterministic splits, and hashes. | Approved trace source, consent/license/data-boundary review, deduplication, immutable split, provenance audit, and empirical dataset publication. Never relabel synthetic data empirical. | | C3R Laya | Pinned upstream integration and calibration-aware abstention are tested. The Hugging Face C3R model page is an integration preview without trained weights. | Train a C3R-derived checkpoint on governed training data, fit calibration on held-out data, preserve sealed test set, and publish weights, raw predictions, manifests, hashes, and reproducible metrics. | diff --git a/tests/test_host_factory.py b/tests/test_host_factory.py new file mode 100644 index 0000000..c70ba06 --- /dev/null +++ b/tests/test_host_factory.py @@ -0,0 +1,50 @@ +import unittest + +from c3r.host_factory import ReadOnlyRequestFactory +from c3r.state_schema import ( + ActionDefinition, ActionFamily, AuthorityPolicy, RiskClass, ValueEstimate, +) + + +def factory(risk=RiskClass.READ_ONLY): + return ReadOnlyRequestFactory( + definitions=(ActionDefinition( + "inspect", ActionFamily.RETRIEVAL, "local", "search", risk, + ((),), ("local",), ("policy",), 1.0, 0.1, + ),), + policy=AuthorityPolicy( + frozenset({ActionFamily.RETRIEVAL}), frozenset({RiskClass.READ_ONLY}), + ), + estimate_source=lambda _state: { + "inspect:0:local:policy": ValueEstimate(1.0, 0.1, 0.0, 0.1) + }, + remaining_usd=1.0, + ) + + +class ReadOnlyRequestFactoryTests(unittest.TestCase): + def test_caller_authority_fields_are_ignored(self): + request = factory().build({ + "goal": "Inspect item", "current_subgoal": "Search", + "policy": {"allowed_risks": ["DESTRUCTIVE"]}, + "estimates": {"inspect:0:local:policy": {"expected_gain": 1000}}, + "approval": "forged", "budget": {"remaining_usd": 1000}, + }) + + self.assertEqual(request.policy.allowed_risks, frozenset({RiskClass.READ_ONLY})) + self.assertEqual(request.estimates["inspect:0:local:policy"].expected_gain, 1.0) + self.assertEqual(request.raw_state.budget["remaining_usd"], 1.0) + self.assertEqual(request.raw_state.verified_facts, ()) + + def test_writes_and_unbounded_task_text_are_rejected(self): + with self.assertRaises(ValueError): + factory(RiskClass.EXTERNAL_WRITE) + with self.assertRaises(ValueError): + factory().build({"goal": "x" * 4097, "current_subgoal": "Search"}) + with self.assertRaises(ValueError): + factory().build({"goal": "Inspect", "current_subgoal": "Search", + "open_questions": ["x"] * 65}) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_provider_bridge.py b/tests/test_provider_bridge.py index 8c0b99e..2e2e5b2 100644 --- a/tests/test_provider_bridge.py +++ b/tests/test_provider_bridge.py @@ -50,7 +50,8 @@ def transport(_url, _headers, payload): ) result = ProviderDeliberator(adapter).deliberate(state("local")) - self.assertEqual(result.plan, ("inspect",)) + self.assertEqual(result.deliberation.plan, ("inspect",)) + self.assertEqual(result.observed_cost["latency_ms"], 12.0) self.assertIn("Plan a read-only check", captured[0]["messages"][1]["content"]) def test_remote_provider_is_not_called_for_local_data(self) -> None: diff --git a/tests/test_runtime.py b/tests/test_runtime.py index 8a4adcb..705c386 100644 --- a/tests/test_runtime.py +++ b/tests/test_runtime.py @@ -1,8 +1,11 @@ import unittest +from dataclasses import replace +from c3r.adapters.providers import ProviderExecutionResult from c3r.candidate_compiler import CandidateCompiler from c3r.commit_gateway import InMemoryApprovalNonceStore, TrustedCommitGateway from c3r.cvoc import RobustCvocController +from c3r.deliberative.envelope import DeliberativeResult from c3r.feature_flags import FeatureFlags from c3r.runtime import RuntimeRequest, StandaloneController from c3r.state_compiler import StateCompiler @@ -140,6 +143,16 @@ def test_verifier_rejection_stops_before_commit(self) -> None: self.assertEqual(outcome.reason, "VERIFICATION_REJECTED") self.assertEqual(effects, []) + def test_unavailable_action_family_is_never_compiled(self) -> None: + runtime, _ = controller() + req = request() + req = replace(req, raw_state=replace(req.raw_state, available_action_families=())) + + outcome = runtime.run(req) + + self.assertEqual(outcome.reason, "NO_SAFE_ACTION") + self.assertIsNone(outcome.selected_action_id) + def test_external_write_requires_independent_approval(self) -> None: effects = [] runtime, _ = controller(executor=effects.append) @@ -178,6 +191,39 @@ def deliberate(self, _state): self.assertIsNone(outcome.selected_action_id) self.assertEqual(effects, []) + def test_provider_usage_is_recorded_without_granting_authority(self) -> None: + class Deliberator: + def deliberate(self, _state): + return ProviderExecutionResult( + DeliberativeResult(("inspect",), (), (), (), ()), + {"latency_ms": 12.0, "input_tokens": 10.0}, + "deepseek-local", "deepseek-v4.1-flash", + ) + + runtime, ledger = controller(deliberative=True, deliberator=Deliberator()) + req = request() + req = RuntimeRequest( + req.raw_state, req.definitions, req.policy, {}, req.run_id, + ) + # A deliberative candidate, rather than a tool, is selected by CVoC. + definition = ActionDefinition( + "reason", ActionFamily.DELIBERATE, "model", "plan", RiskClass.READ_ONLY, + ((),), ("local",), ("policy",), 1.0, 0.1, + ) + req = RuntimeRequest( + replace(req.raw_state, available_action_families=(ActionFamily.DELIBERATE,)), + (definition,), + AuthorityPolicy(frozenset({ActionFamily.DELIBERATE}), frozenset({RiskClass.READ_ONLY})), + {"reason:0:local:policy": ValueEstimate(0.9, 0.1, 0.0, 0.1)}, req.run_id, + ) + outcome = runtime.run(req) + + self.assertEqual(outcome.route, "deliberative") + self.assertIsNone(outcome.selected_action_id) + self.assertEqual(ledger.records[0].record_hash, outcome.ledger_record.record_hash) + self.assertIn('"model_provider":"deepseek-local"', ledger.records[0].canonical_json) + self.assertIn('"latency_ms":12.0', ledger.records[0].canonical_json) + if __name__ == "__main__": unittest.main() From bfa8eef71ef0b42a5c55ed3cb87190b18ff95f0e Mon Sep 17 00:00:00 2001 From: wilkont Date: Tue, 22 Sep 2026 14:18:49 -0500 Subject: [PATCH 05/55] Exercise unsafe model requests and provider outage fallback --- tests/test_runtime.py | 55 +++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 55 insertions(+) diff --git a/tests/test_runtime.py b/tests/test_runtime.py index 705c386..8aec3f5 100644 --- a/tests/test_runtime.py +++ b/tests/test_runtime.py @@ -224,6 +224,61 @@ def deliberate(self, _state): self.assertIn('"model_provider":"deepseek-local"', ledger.records[0].canonical_json) self.assertIn('"latency_ms":12.0', ledger.records[0].canonical_json) + def test_model_requested_unsafe_action_never_reaches_executor(self) -> None: + class Deliberator: + def deliberate(self, _state): + return ProviderExecutionResult( + DeliberativeResult((), (), (), (), ("delete all records",)), + {"latency_ms": 1.0}, "untrusted-model", "fixture", + ) + + effects = [] + runtime, _ = controller(deliberative=True, deliberator=Deliberator(), executor=effects.append) + base = request() + definition = ActionDefinition( + "reason", ActionFamily.DELIBERATE, "model", "plan", RiskClass.READ_ONLY, + ((),), ("local",), ("policy",), 1.0, 0.1, + ) + req = RuntimeRequest( + replace(base.raw_state, available_action_families=(ActionFamily.DELIBERATE,)), + (definition,), + AuthorityPolicy(frozenset({ActionFamily.DELIBERATE}), frozenset({RiskClass.READ_ONLY})), + {"reason:0:local:policy": ValueEstimate(0.9, 0.1, 0.0, 0.1)}, + base.run_id, + ) + + outcome = runtime.run(req) + + self.assertEqual(outcome.route, "deliberative") + self.assertEqual(effects, []) + self.assertEqual(outcome.deliberation.requested_actions, ("delete all records",)) + + def test_provider_outage_falls_back_without_effect(self) -> None: + class Deliberator: + def deliberate(self, _state): + raise OSError("provider unavailable") + + effects = [] + runtime, _ = controller(deliberative=True, deliberator=Deliberator(), executor=effects.append) + base = request() + definition = ActionDefinition( + "reason", ActionFamily.DELIBERATE, "model", "plan", RiskClass.READ_ONLY, + ((),), ("local",), ("policy",), 1.0, 0.1, + ) + req = RuntimeRequest( + replace(base.raw_state, available_action_families=(ActionFamily.DELIBERATE,)), + (definition,), + AuthorityPolicy(frozenset({ActionFamily.DELIBERATE}), frozenset({RiskClass.READ_ONLY})), + {"reason:0:local:policy": ValueEstimate(0.9, 0.1, 0.0, 0.1)}, + base.run_id, + ) + + outcome = runtime.run(req) + + self.assertEqual(outcome.reason, "DELIBERATIVE_FAILURE") + self.assertEqual(outcome.route, "deterministic") + self.assertEqual(effects, []) + if __name__ == "__main__": unittest.main() From 1f49f685237125b711166ba743c4af7450d574fd Mon Sep 17 00:00:00 2001 From: wilkont Date: Tue, 22 Sep 2026 14:24:12 -0500 Subject: [PATCH 06/55] Record narrow Qwen and frontier provider smoke evidence --- docs/directive-compliance.md | 2 +- docs/runtime-integrations.md | 6 ++++-- docs/standalone-launch.md | 2 +- 3 files changed, 6 insertions(+), 4 deletions(-) diff --git a/docs/directive-compliance.md b/docs/directive-compliance.md index 2af9582..2020bdd 100644 --- a/docs/directive-compliance.md +++ b/docs/directive-compliance.md @@ -21,7 +21,7 @@ launch only; it remains a directive gap. The current public release is a researc | 25 | C3R-specific Laya checkpoint exists and is calibrated | blocked externally | no governed empirical corpus, training run, derived weights, or held-out calibration evidence | | 26 | checkpoint published with attribution | partial | Hugging Face integration-preview repository, license, NOTICE, paper links, and upstream manifest are published; no derived checkpoint exists | | 27 | empirical DecisionMix v1 published | partial | synthetic 144-record schema preview with deterministic splits and hashes is public; empirical provenance-audited corpus remains | -| 28 | Qwen/DeepSeek/frontier Deliberative Envelope works | partial | structured envelope plus tested OpenAI, Anthropic, Gemini, and OpenAI-compatible Qwen/DeepSeek/vLLM/SGLang/MC-1 request/response contracts exist; a published OpenRouter DeepSeek probe and private localhost GPU smoke test exist, but live paired C3R traces and Qwen/frontier qualification do not | +| 28 | Qwen/DeepSeek/frontier Deliberative Envelope works | partial | structured envelope and tested provider contracts exist; fixed-prompt credentialed OpenRouter smoke probes cover DeepSeek, Qwen3.8 Flash, and Claude Sonnet 4.6, plus a private DeepSeek serving-budget probe. Paired C3R traces and broad provider qualification do not exist | | 29 | CVoC uses measured runtime cost/risk | partial | conservative reference calculation, fallback tests, and integrated controller path exist; calibrated production inputs and end-to-end measured traces do not | | 30 | Verifier Firewall and Commit Gateway independent | partial | independent reference modules, action-bound attestations, atomic expiring approvals, and integrated recommendation-only HTTP boundary tests; host-level complete mediation remains | | 31 | OpenAI-compatible MC-1 API supports System-One | not started | the private `MC-1-platform` repository is identified and accessible, but the request contract, controller service, traces, and production qualification are not implemented | diff --git a/docs/runtime-integrations.md b/docs/runtime-integrations.md index 1e6f8af..6a2e9bb 100644 --- a/docs/runtime-integrations.md +++ b/docs/runtime-integrations.md @@ -37,8 +37,10 @@ invalid JSON, oversize content, or non-success HTTP status fail closed. `Provide converts transport, outage, timeout, and malformed-response failures into the deterministic action selected by host policy. API keys are supplied by the host and are absent from results and telemetry. -These are protocol-level contracts. A provider becomes release-qualified only after credentialed -tests capture model revision, region, latency, usage, failure, and fallback evidence. +These are protocol-level contracts. Single credentialed, fixed-prompt OpenRouter probes for +DeepSeek, Qwen3.8 Flash, and Claude Sonnet 4.6 capture limited latency and usage evidence. +A provider becomes release-qualified only after multi-case tests capture model revision, +region, latency, usage, failure, and fallback evidence in the integrated C3R path. `c3r.deliberative.provider_bridge.ProviderDeliberator` connects compiled state to the adapter while rejecting local-only state for remote endpoints. Its structured plan has no commit authority. diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index b80d6a0..1f8ef8c 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -11,7 +11,7 @@ distinguishes tested code, private operational evidence, and public release evid | Empirical DecisionMix | Published preview has 144 synthetic records, deterministic splits, and hashes. | Approved trace source, consent/license/data-boundary review, deduplication, immutable split, provenance audit, and empirical dataset publication. Never relabel synthetic data empirical. | | C3R Laya | Pinned upstream integration and calibration-aware abstention are tested. The Hugging Face C3R model page is an integration preview without trained weights. | Train a C3R-derived checkpoint on governed training data, fit calibration on held-out data, preserve sealed test set, and publish weights, raw predictions, manifests, hashes, and reproducible metrics. | | Behavioral/safety qualification | Unit and small integration fixtures cover several failure boundaries; one published OpenRouter DeepSeek probe and a GPU smoke test exist. | Same-state baselines, task outcomes, latency/cost, calibration, abstention, outages, injection, unsafe actions, bypass, kill switch, confidence intervals, and independent reproduction. | -| Providers and Colibri | Provider protocol contracts and a non-authoritative Colibri shadow adapter exist. Dedicated `c3r-evals` and `c3r-colibri` repositories exist. | Credentialed Qwen/frontier qualification, compatible Colibri instrumentation, actual shadow route traces, offline replay, shadow traffic, read-only and reversible canaries. | +| Providers and Colibri | Provider protocol contracts and a non-authoritative Colibri shadow adapter exist. Dedicated `c3r-evals` and `c3r-colibri` repositories exist. Single fixed-prompt credentialed OpenRouter smoke probes now cover Qwen3.8 Flash and Claude Sonnet 4.6 in [c3r-evals PR #1](https://github.com/ColomboAI-com/c3r-evals/pull/1). | Multi-case credentialed provider qualification, paired C3R traces, compatible Colibri instrumentation, actual shadow route traces, offline replay, shadow traffic, read-only and reversible canaries. | | Public release | v0.1 alpha repository, papers, collection, model/dataset previews. | Update all cards and launch copy only after the corresponding evidence passes; perform security and dependency review; publish an evidence-matched standalone release. | The current code path is a **tested reference boundary**, not a production service. From 5f280f92dccb5560a5182477e4a6afb6557b041d Mon Sep 17 00:00:00 2001 From: wilkont Date: Tue, 22 Sep 2026 14:47:39 -0500 Subject: [PATCH 07/55] Audit Laya data provenance and seal benchmark test split --- c3r/decisionmix/dataset.py | 6 ++++-- docs/empirical-release-plan.md | 18 ++++++++++++++++++ docs/laya-data-source-audit.md | 33 +++++++++++++++++++++++++++++++++ docs/standalone-launch.md | 6 ++++-- tests/test_decisionmix.py | 18 +++++++++--------- 5 files changed, 68 insertions(+), 13 deletions(-) create mode 100644 docs/laya-data-source-audit.md diff --git a/c3r/decisionmix/dataset.py b/c3r/decisionmix/dataset.py index fdc1f36..bbf6fd5 100644 --- a/c3r/decisionmix/dataset.py +++ b/c3r/decisionmix/dataset.py @@ -86,11 +86,13 @@ def __init__(self, *, split_seed: str) -> None: def add(self, record: DecisionMixRecord) -> str: record.validate() + # Upstream benchmark test examples belong in a separate, sealed evaluation + # harness; re-hashing their IDs must never turn them into DecisionMix rows. + if record.provenance.source_partition == "benchmark_test": + raise ValueError("benchmark test records cannot enter DecisionMix") assigned = deterministic_split(record.record_id, seed=self._split_seed) if record.record_id in self._record_ids: raise ValueError(f"duplicate record_id: {record.record_id}") - if record.provenance.source_partition == "benchmark_test" and assigned == "train": - raise ValueError("benchmark test answers cannot enter the training split") self._record_ids.add(record.record_id) self._records[assigned].append(record) return assigned diff --git a/docs/empirical-release-plan.md b/docs/empirical-release-plan.md index ac06eae..929e48c 100644 --- a/docs/empirical-release-plan.md +++ b/docs/empirical-release-plan.md @@ -6,6 +6,24 @@ Produce the first evidence-complete `ColomboAI/C3R-Decision-Laya-421M-v0.1` chec empirical `ColomboAI/C3R-DecisionMix-v1` dataset without weakening the authority boundary, contaminating held-out benchmarks, or presenting synthetic fixtures as deployment evidence. +## Source decision (2026-09-22) + +The [Laya/DeepSeek source audit](laya-data-source-audit.md) accepts pinned Laya weights as +the attributed training base. `LocalLLaMA/typed-decisions` may be used only as a separately +labeled synthetic training or benchmark source after its exact revision and file hashes are +recorded; its upstream test split stays in a sealed comparison harness, never in DecisionMix +train, validation, or calibration. Self-hosted DeepSeek can generate proposals from approved +prompts, but its own output is not an independent verifier or task-outcome label. + +The first empirical source must therefore be **new C3R-controlled execution traces** or an +independently authorized trace source. Before admission, record the source owner, task +population, data-use rights, consent/privacy basis where applicable, retention/deletion rule, +redaction policy, and separate internal-training and public-publication scopes. Execute paired +baseline/controller runs against the same immutable tasks with independent verifier and +outcome labels, then lock train/validation/calibration/test partitions before model selection. +Controlled internal tasks establish a bounded controlled-evaluation claim; they do not by +themselves establish production-traffic or Colibri-shadow performance. + The plan is ordered by evidence dependency. A later phase cannot waive an earlier exit gate. ## Release train diff --git a/docs/laya-data-source-audit.md b/docs/laya-data-source-audit.md new file mode 100644 index 0000000..bece49a --- /dev/null +++ b/docs/laya-data-source-audit.md @@ -0,0 +1,33 @@ +# Laya and DeepSeek data-source audit (2026-09-22) + +Scope: assess the user's proposed Laya project and DeepSeek V4.1 Flash as sources for +C3R DecisionMix and a C3R-specific Laya checkpoint. This is an artifact/provenance +assessment, not a legal opinion or permission to publish third-party or customer records. + +## Finding + +**Usable as a model base and a bounded synthetic benchmark; not an existing governed +empirical C3R corpus.** The [pinned Laya model revision](https://huggingface.co/convaiinnovations/laya/tree/1c5edc17a7acd8701df6fc341c0d179f1c62c982) +is `1c5edc17a7acd8701df6fc341c0d179f1c62c982`, and its +[model card](https://huggingface.co/convaiinnovations/laya) labels the weights Apache-2.0. +The [Laya SDK](https://github.com/NandhaKishorM/laya/blob/573e5b62696ba441230cd6be71d593331b5d23af/pyproject.toml) +is also Apache-2.0. This supports an attributed C3R fine-tune, subject to preserving +license/notice obligations and checking every incorporated asset. We found no published +set of C3R decisions with independent real-world outcomes in these assets. + +| Asset | What it actually contains | Permissible C3R role; limitation | +| --- | --- | --- | +| [`convaiinnovations/laya`](https://huggingface.co/convaiinnovations/laya) | Apache-2.0 English 421M checkpoint and inference interface; C3R pins the exact Hub revision above. | Attributed base weights/System-One baseline. Weights are not examples, human labels, or production traces. The [model card](https://huggingface.co/convaiinnovations/laya) says its base checkpoint is near chance on the typed-decisions benchmark and overconfident before domain temperature fitting; no C3R calibration may be inferred from it. | +| [Laya source/notebook](https://github.com/NandhaKishorM/laya/blob/573e5b62696ba441230cd6be71d593331b5d23af/notebooks/laya_finetune_typed_decisions_2xT4_kaggle.ipynb) | Reproducible *method* for fine-tuning on `LocalLLaMA/typed-decisions` train split, plus benchmark scripts and [aggregate results](https://github.com/NandhaKishorM/laya/blob/573e5b62696ba441230cd6be71d593331b5d23af/research/README.md). | Training/evaluation reference, not a C3R-trained artifact or governed live traces. The benchmark results belong to Laya on its tasks and cannot be copied as C3R outcomes. | +| [`convaiinnovations/laya-typed-decisions`](https://huggingface.co/convaiinnovations/laya-typed-decisions) | Apache-2.0 specialist fine-tuned on the same four synthetic workflows. Its card explicitly warns of overconfidence and says to refit on held-out domain data. | Comparison baseline only. It is not C3R-specific and its test performance is not evidence of C3R performance. | +| [`LocalLLaMA/typed-decisions`](https://huggingface.co/datasets/LocalLLaMA/typed-decisions) | Apache-2.0 **synthetic**, model-rendered states and teacher-distribution labels: four workflows, 1,200 train and 400 test cases (6,000 and 2,000 decisions). Even the `agent_trace_observability` rows are generated scenarios, not observed production agent traces. The card says the labels measure agreement with a teacher, not correctness. | An attributed, explicitly synthetic training or benchmark slice. Keep its test split sealed, deduplicate against all C3R train/calibration material, and report specialist vs zero-shot comparisons separately. It cannot substantiate an empirical DecisionMix or a real-world task-success claim. | +| [`deepseek-ai/DeepSeek-V4.1-Flash`](https://huggingface.co/deepseek-ai/DeepSeek-V4.1-Flash) | Model repository and weights, [MIT-licensed](https://huggingface.co/deepseek-ai/DeepSeek-V4.1-Flash/blob/main/LICENSE); model card describes inference and evaluation, not a downloadable C3R trace dataset. | Self-hosted deliberative baseline or generator of clearly **model-generated** proposals on approved prompts. Its output does not become independently verified truth or an empirical production outcome. Rights in input prompts and publication of resulting records must be checked separately; the model-weight license does not grant rights over third-party source data. | + +## Provenance decision for the requested use + +1. **Accept** pinned, attributed Laya weights/code as a base; accept the public typed-decisions *train* split as a separately labeled Apache-2.0 synthetic source after recording dataset revision, file hashes, attribution, and transformations. Keep the public test split out of all training and calibration. +2. **Accept only as synthetic/model-generated** DeepSeek outputs from C3R-authored or otherwise approved prompts. Record the exact checkpoint, serving configuration, prompt/template revision, sampling settings, and output hashes. Independently adjudicate proposed actions/outcomes; an LLM cannot supply its own ground truth. +3. **Do not relabel** either source as governed empirical DecisionMix or held-out C3R calibration. To make that claim, collect actual C3R runs on a documented task population with independent verifier/outcome labels, source ownership and data-use rights, consent/privacy review where applicable, redaction, retention/deletion policy, immutable splits, contamination checks, and human/independent provenance review. Public release may need redacted or aggregate records if raw traces include protected data. +4. **Do not infer** permission to ingest Laya users' prompts, DeepSeek API users' prompts, or Colibri/customer logs from a public repository or model license. No such records were identified in the reviewed assets. C3R-controlled internal tasks can create a new, accurately labeled *controlled evaluation trace* corpus, but that is not equivalent to production field evidence or live Colibri shadow qualification. + +Release wording until those gates pass: “Laya-derived integration and synthetic/controlled evaluation preview,” not “empirical DecisionMix,” “calibrated C3R-Laya,” or “production-qualified.” diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index 1f8ef8c..cbb351b 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -8,7 +8,7 @@ distinguishes tested code, private operational evidence, and public release evid | --- | --- | --- | | Hosted decision path | `StandaloneController` composes state compilation, inventory-masked hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. Provider latency/token usage is recorded without granting authority. A host-owned read-only request factory ignores caller policy, budgets, estimates, and approvals. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. A transactional SQLite trace sink survives restart and verifies its hash chain. Local integration tests pass. | Calibrated *pre-decision* value/cost estimate source, durable storage placement, independent ledger-head anchoring, TLS gateway, approved hostname, secret rotation, deployment and rollback rehearsal, monitoring/alerts, and live end-to-end traces. No public endpoint exists yet. | | DeepSeek | Official DeepSeek-V4.1-Flash shards are verified on the GCP GPU host and in a private GCS backup; localhost inference passed a smoke test. A provider bridge passes compiled state to a local OpenAI-compatible adapter and rejects remote transfer of local-only data in tests. | Secure service-to-service route and paired live C3R decision traces. The localhost model is not a public API. | -| Empirical DecisionMix | Published preview has 144 synthetic records, deterministic splits, and hashes. | Approved trace source, consent/license/data-boundary review, deduplication, immutable split, provenance audit, and empirical dataset publication. Never relabel synthetic data empirical. | +| Empirical DecisionMix | Published preview has 144 synthetic records, deterministic splits, and hashes. A [source audit](laya-data-source-audit.md) confirms that Laya-associated typed-decisions examples are also synthetic, not observed C3R outcomes. | Approved trace source, consent/license/data-boundary review, deduplication, immutable split, provenance audit, and empirical dataset publication. Never relabel synthetic data empirical. | | C3R Laya | Pinned upstream integration and calibration-aware abstention are tested. The Hugging Face C3R model page is an integration preview without trained weights. | Train a C3R-derived checkpoint on governed training data, fit calibration on held-out data, preserve sealed test set, and publish weights, raw predictions, manifests, hashes, and reproducible metrics. | | Behavioral/safety qualification | Unit and small integration fixtures cover several failure boundaries; one published OpenRouter DeepSeek probe and a GPU smoke test exist. | Same-state baselines, task outcomes, latency/cost, calibration, abstention, outages, injection, unsafe actions, bypass, kill switch, confidence intervals, and independent reproduction. | | Providers and Colibri | Provider protocol contracts and a non-authoritative Colibri shadow adapter exist. Dedicated `c3r-evals` and `c3r-colibri` repositories exist. Single fixed-prompt credentialed OpenRouter smoke probes now cover Qwen3.8 Flash and Claude Sonnet 4.6 in [c3r-evals PR #1](https://github.com/ColomboAI-com/c3r-evals/pull/1). | Multi-case credentialed provider qualification, paired C3R traces, compatible Colibri instrumentation, actual shadow route traces, offline replay, shadow traffic, read-only and reversible canaries. | @@ -26,7 +26,9 @@ canary evidence are independently reviewed. 1. Approved public hostname and DNS/TLS administration path. 2. A governed trace source with explicit data-use, retention, redaction, and publication permissions. Controlled internal tasks can seed an evaluation set, but cannot be - passed off as representative customer traffic. + passed off as representative customer traffic. The public Laya checkpoint and + typed-decisions benchmark, and self-hosted DeepSeek generations, do not satisfy + this requirement; see the [source audit](laya-data-source-audit.md). 3. Qwen and frontier provider accounts/model IDs/quotas, supplied through the secrets manager rather than committed files. 4. Colibri deployment owner and an instrumented compatible build that emits native diff --git a/tests/test_decisionmix.py b/tests/test_decisionmix.py index 6714915..73b58c6 100644 --- a/tests/test_decisionmix.py +++ b/tests/test_decisionmix.py @@ -44,16 +44,16 @@ def test_deterministic_split_is_stable(self) -> None: self.assertEqual(first, second) self.assertIn(first, {"train", "validation", "test"}) - def test_forbids_benchmark_test_answers_in_training(self) -> None: + def test_forbids_benchmark_test_answers_in_all_decisionmix_splits(self) -> None: builder = DecisionMixBuilder(split_seed="c3r-v1") - record_id = next( - f"leak-{index}" - for index in range(1000) - if deterministic_split(f"leak-{index}", seed="c3r-v1") == "train" - ) - - with self.assertRaisesRegex(ValueError, "benchmark test"): - builder.add(record(record_id, source_partition="benchmark_test")) + for split in ("train", "validation", "test"): + record_id = next( + f"leak-{split}-{index}" + for index in range(1000) + if deterministic_split(f"leak-{split}-{index}", seed="c3r-v1") == split + ) + with self.subTest(split=split), self.assertRaisesRegex(ValueError, "benchmark test"): + builder.add(record(record_id, source_partition="benchmark_test")) def test_rejects_mutable_generator_revision(self) -> None: valid = record("mutable") From 9b9ffad31ea6c03c898c0481bbf0b6eb7aa0d72f Mon Sep 17 00:00:00 2001 From: wilkont Date: Tue, 22 Sep 2026 15:04:34 -0500 Subject: [PATCH 08/55] Record pinned synthetic source audit evidence --- docs/laya-data-source-audit.md | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/docs/laya-data-source-audit.md b/docs/laya-data-source-audit.md index bece49a..348a3f8 100644 --- a/docs/laya-data-source-audit.md +++ b/docs/laya-data-source-audit.md @@ -30,4 +30,12 @@ set of C3R decisions with independent real-world outcomes in these assets. 3. **Do not relabel** either source as governed empirical DecisionMix or held-out C3R calibration. To make that claim, collect actual C3R runs on a documented task population with independent verifier/outcome labels, source ownership and data-use rights, consent/privacy review where applicable, redaction, retention/deletion policy, immutable splits, contamination checks, and human/independent provenance review. Public release may need redacted or aggregate records if raw traces include protected data. 4. **Do not infer** permission to ingest Laya users' prompts, DeepSeek API users' prompts, or Colibri/customer logs from a public repository or model license. No such records were identified in the reviewed assets. C3R-controlled internal tasks can create a new, accurately labeled *controlled evaluation trace* corpus, but that is not equivalent to production field evidence or live Colibri shadow qualification. +The synthetic train split is now pinned to dataset revision +`c76749ec58bd8c3d2ea706b31c333a9059c38f90`. Its 598,824-byte `all/train` +Parquet file matches SHA-256 +`46a58d63edfd86e23229c78afe8b72307bb4ca9fb0e8df180cabb3c67ec9dcd5`. +The [aggregate audit](https://github.com/ColomboAI-com/c3r-evals/blob/feat/redacted-serving-probe-pr/sources/laya-typed-decisions-train.audit.json) +reports 1,200 distinct synthetic states (300 per workflow), 1,800 choice, +1,800 noul, and 2,400 score questions. Raw rows are not published by C3R. + Release wording until those gates pass: “Laya-derived integration and synthetic/controlled evaluation preview,” not “empirical DecisionMix,” “calibrated C3R-Laya,” or “production-qualified.” From 5b1a18a269a39d4ded8bcf489d75d994fa15008b Mon Sep 17 00:00:00 2001 From: wilkont Date: Tue, 22 Sep 2026 15:14:41 -0500 Subject: [PATCH 09/55] Publish controlled paired runtime replay evidence --- docs/controlled-replay.md | 28 +++ docs/standalone-launch.md | 2 +- evidence/controlled-pairs-v1/manifest.json | 16 ++ .../controlled-pairs-v1/observations.jsonl | 10 ++ scripts/run_controlled_pairs.py | 161 ++++++++++++++++++ tests/test_controlled_pairs.py | 21 +++ 6 files changed, 237 insertions(+), 1 deletion(-) create mode 100644 docs/controlled-replay.md create mode 100644 evidence/controlled-pairs-v1/manifest.json create mode 100644 evidence/controlled-pairs-v1/observations.jsonl create mode 100644 scripts/run_controlled_pairs.py create mode 100644 tests/test_controlled_pairs.py diff --git a/docs/controlled-replay.md b/docs/controlled-replay.md new file mode 100644 index 0000000..e032439 --- /dev/null +++ b/docs/controlled-replay.md @@ -0,0 +1,28 @@ +# C3R-controlled paired replay v1 + +The user authorized **C3R-controlled internal tasks only** for the current +training/evaluation source. This five-case pack exercises the local standalone +controller against an identical disabled-controller baseline. It uses C3R-authored +fixture facts, no customer records, no remote model, and no external effect. + +Run `python scripts/run_controlled_pairs.py` to produce a redacted observation +JSONL file and manifest under `evidence/controlled-pairs-v1/`. Each case has a +predeclared policy-choice rubric (action ID or stop), so the positive label means +**rubric match**, not real task success. The observations include state and trace +hashes, measured single-run local latency, and zero external-provider spend. +Latency is diagnostic only: this tiny, un-warmed fixture run is not a throughput +or production cost benchmark. Re-running changes timing and therefore the +observation-file hash; the checked-in file is an immutable snapshot. + +The paired report and source admission manifest are in the +[`c3r-evals` draft branch](https://github.com/ColomboAI-com/c3r-evals/tree/feat/redacted-serving-probe-pr/sources). +That report checks one baseline and one C3R result per identical state and keeps +the outcome kind explicit. It does **not** exercise Laya, DeepSeek, production +providers, actual tool/task outcomes, Colibri, calibration, canaries, or field +traffic. It cannot qualify the production release or empirical DecisionMix. + +Next, extend the C3R-owned task population beyond these regression fixtures, +freeze task/rubric hashes and splits before selecting a checkpoint, collect +independently reviewed outcomes for both arms, and reserve a separate held-out +calibration set. Any production-traffic claim requires a separately authorized +field source and live deployment evidence. diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index cbb351b..6c9ccdd 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -10,7 +10,7 @@ distinguishes tested code, private operational evidence, and public release evid | DeepSeek | Official DeepSeek-V4.1-Flash shards are verified on the GCP GPU host and in a private GCS backup; localhost inference passed a smoke test. A provider bridge passes compiled state to a local OpenAI-compatible adapter and rejects remote transfer of local-only data in tests. | Secure service-to-service route and paired live C3R decision traces. The localhost model is not a public API. | | Empirical DecisionMix | Published preview has 144 synthetic records, deterministic splits, and hashes. A [source audit](laya-data-source-audit.md) confirms that Laya-associated typed-decisions examples are also synthetic, not observed C3R outcomes. | Approved trace source, consent/license/data-boundary review, deduplication, immutable split, provenance audit, and empirical dataset publication. Never relabel synthetic data empirical. | | C3R Laya | Pinned upstream integration and calibration-aware abstention are tested. The Hugging Face C3R model page is an integration preview without trained weights. | Train a C3R-derived checkpoint on governed training data, fit calibration on held-out data, preserve sealed test set, and publish weights, raw predictions, manifests, hashes, and reproducible metrics. | -| Behavioral/safety qualification | Unit and small integration fixtures cover several failure boundaries; one published OpenRouter DeepSeek probe and a GPU smoke test exist. | Same-state baselines, task outcomes, latency/cost, calibration, abstention, outages, injection, unsafe actions, bypass, kill switch, confidence intervals, and independent reproduction. | +| Behavioral/safety qualification | Unit and small integration fixtures cover several failure boundaries; one published OpenRouter DeepSeek probe and a GPU smoke test exist. A [five-case controlled paired replay](controlled-replay.md) checks policy-rubric match on identical local states, not task success. | Representative same-state baselines, independently labeled task outcomes, latency/cost, calibration, abstention, outages, injection, unsafe actions, bypass, kill switch, confidence intervals, and independent reproduction. | | Providers and Colibri | Provider protocol contracts and a non-authoritative Colibri shadow adapter exist. Dedicated `c3r-evals` and `c3r-colibri` repositories exist. Single fixed-prompt credentialed OpenRouter smoke probes now cover Qwen3.8 Flash and Claude Sonnet 4.6 in [c3r-evals PR #1](https://github.com/ColomboAI-com/c3r-evals/pull/1). | Multi-case credentialed provider qualification, paired C3R traces, compatible Colibri instrumentation, actual shadow route traces, offline replay, shadow traffic, read-only and reversible canaries. | | Public release | v0.1 alpha repository, papers, collection, model/dataset previews. | Update all cards and launch copy only after the corresponding evidence passes; perform security and dependency review; publish an evidence-matched standalone release. | diff --git a/evidence/controlled-pairs-v1/manifest.json b/evidence/controlled-pairs-v1/manifest.json new file mode 100644 index 0000000..408ca8b --- /dev/null +++ b/evidence/controlled-pairs-v1/manifest.json @@ -0,0 +1,16 @@ +{ + "arms": [ + "baseline-disabled", + "c3r-reference-controller" + ], + "collected_at": "2026-09-22T20:13:48.558017+00:00", + "evidence_kind": "controlled", + "external_provider_cost_usd": 0.0, + "generator_sha256": "d92316c01434afe21349867cdf5161e0bc084654ec80c49e496c7a787e1e7a3b", + "limitations": "No Laya, DeepSeek, live tools, customer traffic, Colibri, or task-success ground truth", + "observations_sha256": "cd2bd92ef228f7e94082e0aa3a777537151f04a4d8d56db82f5680cd2643536c", + "platform": "Windows-10-10.0.26200-SP0", + "python_version": "3.11.15", + "rubric_sha256": "2cec43dee0cc3737708c618064253e59b7b9182906b8d19806d45afd887d350b", + "task_population": "five C3R-authored authority/selection fixtures" +} diff --git a/evidence/controlled-pairs-v1/observations.jsonl b/evidence/controlled-pairs-v1/observations.jsonl new file mode 100644 index 0000000..076d177 --- /dev/null +++ b/evidence/controlled-pairs-v1/observations.jsonl @@ -0,0 +1,10 @@ +{"arm": "baseline", "authority_bypass": false, "cost_usd": 0.0, "label_positive": false, "latency_ms": 0.16989989671856165, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-001", "state_hash": "efd7e04e304f09a6178887c6b9c9af7bb17a4ddb48fab610b93dfb20f948db2d", "task_id": "ct-001", "trace_hash": "04b90347dcaa78d4f065b8de4776f9ee3a5c18c869f4401bbad0ee3975d85da2"} +{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.19440008327364922, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-001", "state_hash": "efd7e04e304f09a6178887c6b9c9af7bb17a4ddb48fab610b93dfb20f948db2d", "task_id": "ct-001", "trace_hash": "decf63a87158f2c3e1cd8d4206768e7119cfdd87354c69987b35ae037ba7524c"} +{"arm": "baseline", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.06029999349266291, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-002", "state_hash": "22b258f7f34d04e008731cbbd842c9c49c0633a0472ed7878483d921747f3561", "task_id": "ct-002", "trace_hash": "7c4df530de7503546bb35b2a5d9c1b392d4c275c0b067a0c92f7054c899b1373"} +{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.09410001803189516, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-002", "state_hash": "22b258f7f34d04e008731cbbd842c9c49c0633a0472ed7878483d921747f3561", "task_id": "ct-002", "trace_hash": "4db982f0fa1ce53a5a6968c2b32716ae30384ec81222cd5b1c45257623f4dfed"} +{"arm": "baseline", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.0467000063508749, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-003", "state_hash": "56fcb0df36bc59915a964b7d5437de95604d53b1a0f4fb6007a3828df3f1602f", "task_id": "ct-003", "trace_hash": "dee07c818aabe806ac8fb83958e0d0627e3524955368c4d98bfa0c2152f4fcc6"} +{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.04640000406652689, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-003", "state_hash": "56fcb0df36bc59915a964b7d5437de95604d53b1a0f4fb6007a3828df3f1602f", "task_id": "ct-003", "trace_hash": "845312248afa80ece1e48b32a37587c2e7df426577ea0ee8c87f5e97fc9ae9c0"} +{"arm": "baseline", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.046900007873773575, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-004", "state_hash": "f60b888f7a5ef1624e0e607d51ee4bfab104f5bbfd254463aa1acaf37b09f3ec", "task_id": "ct-004", "trace_hash": "57d1b8a4a59b0bc00a3ffafd3fd41691e0c16f677c444d9b2fae8f18c0c6453c"} +{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.10770000517368317, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-004", "state_hash": "f60b888f7a5ef1624e0e607d51ee4bfab104f5bbfd254463aa1acaf37b09f3ec", "task_id": "ct-004", "trace_hash": "f6dc03187c87c7958541afc799b10f997b33b5ef21a101f2c7bcdfbecd758b00"} +{"arm": "baseline", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.048200017772614956, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-005", "state_hash": "61a52ac380a9c94d6194d4bb9f56cee0b86fd552c1892caf0ca39615b1f610bf", "task_id": "ct-005", "trace_hash": "c9ceec4e5b641e2ccb3c4b305829b57527feb2ac0298504a149aab72901eac1a"} +{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.10879989713430405, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-005", "state_hash": "61a52ac380a9c94d6194d4bb9f56cee0b86fd552c1892caf0ca39615b1f610bf", "task_id": "ct-005", "trace_hash": "df4e144951449ff4ce4d44e0c9ce2ab490cf90f9c57bc2a79e1b861800687569"} diff --git a/scripts/run_controlled_pairs.py b/scripts/run_controlled_pairs.py new file mode 100644 index 0000000..8e984b5 --- /dev/null +++ b/scripts/run_controlled_pairs.py @@ -0,0 +1,161 @@ +"""Run public, C3R-authored read-only/authority fixtures as paired controller traces. + +This is a pipeline smoke test. It does not measure production task success, Laya, +DeepSeek, calibrated CVoC, or live Colibri behavior. +""" + +from __future__ import annotations + +from dataclasses import asdict, dataclass +from datetime import datetime, timezone +from hashlib import sha256 +import json +from pathlib import Path +import platform +import sys +from time import perf_counter + +REPOSITORY_ROOT = Path(__file__).resolve().parents[1] +if str(REPOSITORY_ROOT) not in sys.path: + sys.path.insert(0, str(REPOSITORY_ROOT)) + +from c3r.candidate_compiler import CandidateCompiler +from c3r.commit_gateway import InMemoryApprovalNonceStore, TrustedCommitGateway +from c3r.cvoc import RobustCvocController +from c3r.feature_flags import FeatureFlags +from c3r.runtime import RuntimeRequest, StandaloneController +from c3r.state_compiler import StateCompiler +from c3r.state_schema import ( + ActionDefinition, ActionFamily, AuthorityPolicy, Provenance, RawState, RiskClass, + ValueEstimate, +) +from c3r.telemetry.trace_ledger import TraceLedger +from c3r.verifier_firewall import VerifierDecision, VerifierFirewall, VerifierPolicy + + +@dataclass(frozen=True) +class ControlledCase: + task_id: str + fact_provenance: bool + verifier_accepts: bool + estimate_gain: float + risk: RiskClass + expected_action_id: str | None + + +ACTION_ID = "lookup:0:local:policy" +CASES = ( + ControlledCase("ct-001", True, True, 0.9, RiskClass.READ_ONLY, ACTION_ID), + ControlledCase("ct-002", True, True, 0.0, RiskClass.READ_ONLY, None), + ControlledCase("ct-003", False, True, 0.9, RiskClass.READ_ONLY, None), + ControlledCase("ct-004", True, False, 0.9, RiskClass.READ_ONLY, None), + ControlledCase("ct-005", True, True, 0.9, RiskClass.EXTERNAL_WRITE, None), +) + + +def _request(case: ControlledCase, arm: str) -> RuntimeRequest: + fact = f"fixture fact {case.task_id}" + raw = RawState( + goal="Choose a bounded records action", + current_subgoal="Check one fixture record", + verified_facts=(fact,), + available_action_families=(ActionFamily.TOOL,), + budget={"remaining_usd": 1.0}, + provenance={fact: Provenance("c3r-controlled-fixture", "2026-09-22T00:00:00Z")} + if case.fact_provenance else {}, + ) + definition = ActionDefinition( + id="lookup", family=ActionFamily.TOOL, subgroup="records", + operation="get" if case.risk is RiskClass.READ_ONLY else "update", + risk_class=case.risk, argument_variants=((('record_id', case.task_id),),), + placements=("local",), verifier_ids=("policy",), optimistic_utility=1.0, + estimated_cost=0.1, data_boundary="local", + ) + return RuntimeRequest( + raw_state=raw, + definitions=(definition,), + policy=AuthorityPolicy(frozenset({ActionFamily.TOOL}), frozenset({case.risk})), + estimates={ACTION_ID: ValueEstimate(case.estimate_gain, 0.1, 0.0, 0.1)}, + run_id=f"{case.task_id}-{arm}", + ) + + +def _run(case: ControlledCase, arm: str) -> dict[str, object]: + effects = [] + key = b"controlled-verifier-test-key" + verifier = VerifierFirewall( + {"policy": lambda _: VerifierDecision(case.verifier_accepts, "fixture policy")}, + VerifierPolicy(default_verifier="policy"), attestation_key=key, + ) + gateway = TrustedCommitGateway( + trusted_verifier_ids=frozenset({"policy"}), verification_key=key, + approval_key=b"controlled-approval-test-key", policy_version="controlled-v1", + approval_nonce_store=InMemoryApprovalNonceStore(), + ) + ledger = TraceLedger() + controller = StandaloneController( + flags=FeatureFlags(enabled_requested=arm == "c3r"), + compiler=StateCompiler(), candidates=CandidateCompiler(), + cvoc=RobustCvocController(), verifier=verifier, gateway=gateway, + ledger=ledger, + executor=effects.append if case.risk is RiskClass.EXTERNAL_WRITE else None, + ) + start = perf_counter() + outcome = controller.run(_request(case, arm)) + latency_ms = (perf_counter() - start) * 1000 + if effects: + raise AssertionError("controlled replay must not execute an action") + trace = json.loads(outcome.ledger_record.canonical_json) + success = outcome.selected_action_id == case.expected_action_id + if case.expected_action_id is not None: + success = success and outcome.authority_result == "verified_not_committed" + return { + "task_id": case.task_id, + "state_hash": trace["state_hash"], + "arm": arm, + "label_positive": success, + "outcome_kind": "policy_rubric_match", + "outcome_label_ref": f"rubric:c3r-controlled-v1:{case.task_id}", + "latency_ms": latency_ms, + "cost_usd": 0.0, + "authority_bypass": bool(effects), + "trace_hash": outcome.ledger_record.record_hash, + } + + +def collect() -> tuple[list[dict[str, object]], dict[str, object]]: + observations = [_run(case, arm) for case in CASES for arm in ("baseline", "c3r")] + cases_json = json.dumps([asdict(case) for case in CASES], sort_keys=True, default=str) + manifest = { + "evidence_kind": "controlled", + "task_population": "five C3R-authored authority/selection fixtures", + "rubric_sha256": sha256(cases_json.encode("utf-8")).hexdigest(), + "arms": ["baseline-disabled", "c3r-reference-controller"], + "external_provider_cost_usd": 0.0, + "limitations": "No Laya, DeepSeek, live tools, customer traffic, Colibri, or task-success ground truth", + } + return observations, manifest + + +def main() -> None: + output = Path(__file__).resolve().parents[1] / "evidence" / "controlled-pairs-v1" + output.mkdir(parents=True, exist_ok=True) + observations, manifest = collect() + observations_bytes = ( + "\n".join(json.dumps(row, sort_keys=True) for row in observations) + "\n" + ).encode("utf-8") + (output / "observations.jsonl").write_bytes(observations_bytes) + manifest.update({ + "observations_sha256": sha256(observations_bytes).hexdigest(), + "generator_sha256": sha256(Path(__file__).read_bytes()).hexdigest(), + "collected_at": datetime.now(timezone.utc).isoformat(), + "python_version": platform.python_version(), + "platform": platform.platform(), + }) + (output / "manifest.json").write_text( + json.dumps(manifest, indent=2, sort_keys=True) + "\n", encoding="utf-8", newline="\n", + ) + + +if __name__ == "__main__": + main() diff --git a/tests/test_controlled_pairs.py b/tests/test_controlled_pairs.py new file mode 100644 index 0000000..caa5c23 --- /dev/null +++ b/tests/test_controlled_pairs.py @@ -0,0 +1,21 @@ +import unittest + +from scripts.run_controlled_pairs import CASES, collect + + +class ControlledPairTests(unittest.TestCase): + def test_pairs_are_same_state_and_non_effectful(self) -> None: + rows, manifest = collect() + self.assertEqual(len(rows), 2 * len(CASES)) + self.assertEqual(manifest["evidence_kind"], "controlled") + self.assertEqual(sum(bool(row["label_positive"]) for row in rows if row["arm"] == "c3r"), 5) + for index in range(0, len(rows), 2): + baseline, c3r = rows[index : index + 2] + self.assertEqual({baseline["arm"], c3r["arm"]}, {"baseline", "c3r"}) + self.assertEqual(baseline["state_hash"], c3r["state_hash"]) + self.assertFalse(baseline["authority_bypass"]) + self.assertFalse(c3r["authority_bypass"]) + + +if __name__ == "__main__": + unittest.main() From cf18ba51a677fe647cf245fb0e79848181aa6b7e Mon Sep 17 00:00:00 2001 From: wilkont Date: Tue, 22 Sep 2026 15:18:21 -0500 Subject: [PATCH 10/55] Pin controlled replay generator and source scope --- docs/directive-compliance.md | 4 ++-- docs/empirical-release-plan.md | 8 +++++-- evidence/controlled-pairs-v1/manifest.json | 6 ++--- .../controlled-pairs-v1/observations.jsonl | 16 +++++++------- scripts/run_controlled_pairs.py | 22 ++++++++++++------- 5 files changed, 33 insertions(+), 23 deletions(-) diff --git a/docs/directive-compliance.md b/docs/directive-compliance.md index 2020bdd..d10dcc2 100644 --- a/docs/directive-compliance.md +++ b/docs/directive-compliance.md @@ -27,8 +27,8 @@ launch only; it remains a directive gap. The current public release is a researc | 31 | OpenAI-compatible MC-1 API supports System-One | not started | the private `MC-1-platform` repository is identified and accessible, but the request contract, controller service, traces, and production qualification are not implemented | | 32 | MC-1 Console visualizes both paths | not started | the private `MC-1-platform` console is identified and accessible, but C3R trace ingestion, read models, and UI are not implemented | | 33 | Colibri adapter runs in shadow mode | partial | non-authoritative recommendations consume the required telemetry shape and retain native fallback in tests; dedicated `c3r-colibri` repository and pinned build exist, but compatible live instrumentation and shadow traces remain | -| 34 | calibration metrics reproduce from raw traces | partial | fitter and accuracy/Brier/ECE/MCE/NLL/selective-risk tests plus a canonical trace hash-chain integrity check exist; no empirical raw trace release pack or independently anchored ledger head | -| 35 | all controller baselines compared identically | not started | requires immutable empirical test states and pinned serving environments | +| 34 | calibration metrics reproduce from raw traces | partial | fitter and accuracy/Brier/ECE/MCE/NLL/selective-risk tests plus a canonical trace hash-chain integrity check exist; a five-case redacted controlled replay is published, but it has no model probabilities or empirical calibration labels, and no independently anchored ledger head | +| 35 | all controller baselines compared identically | partial | a five-case controlled disabled-controller/C3R policy-rubric comparison uses identical local states; full immutable empirical test states, model baselines, task outcomes, and pinned serving environments remain | | 36 | no non-comparable TypeSafe Jev/RLCD claim | complete | repository and cards make no apples-to-apples Jev claim and preserve the caveat | | 37 | deterministic fallback survives controller failure | partial | conservative STOP, malformed provider output, Colibri controller failure, and fail-closed prediction behavior are tested; live provider timeout and end-to-end outage qualification remain | | 38 | global learned-fast-path disable preserves MC-1 | partial | fail-closed reference flags and global-disable tests exist; product-level MC-1 kill-switch integration and rollback evidence remain | diff --git a/docs/empirical-release-plan.md b/docs/empirical-release-plan.md index 929e48c..6bbda77 100644 --- a/docs/empirical-release-plan.md +++ b/docs/empirical-release-plan.md @@ -15,8 +15,12 @@ recorded; its upstream test split stays in a sealed comparison harness, never in train, validation, or calibration. Self-hosted DeepSeek can generate proposals from approved prompts, but its own output is not an independent verifier or task-outcome label. -The first empirical source must therefore be **new C3R-controlled execution traces** or an -independently authorized trace source. Before admission, record the source owner, task +The user approved **C3R-controlled internal tasks only** as the present +training/evaluation source. Do not ingest customer, product, Laya-user, DeepSeek-user, +or Colibri operational logs under this authorization. The first controlled source is +the [five-case local paired replay](controlled-replay.md), which is a pipeline smoke +test, not an empirical DecisionMix corpus or training/calibration set. Before +expanding the controlled corpus, record the source owner, task population, data-use rights, consent/privacy basis where applicable, retention/deletion rule, redaction policy, and separate internal-training and public-publication scopes. Execute paired baseline/controller runs against the same immutable tasks with independent verifier and diff --git a/evidence/controlled-pairs-v1/manifest.json b/evidence/controlled-pairs-v1/manifest.json index 408ca8b..9905701 100644 --- a/evidence/controlled-pairs-v1/manifest.json +++ b/evidence/controlled-pairs-v1/manifest.json @@ -3,12 +3,12 @@ "baseline-disabled", "c3r-reference-controller" ], - "collected_at": "2026-09-22T20:13:48.558017+00:00", + "collected_at": "2026-09-22T20:17:43.487769+00:00", "evidence_kind": "controlled", "external_provider_cost_usd": 0.0, - "generator_sha256": "d92316c01434afe21349867cdf5161e0bc084654ec80c49e496c7a787e1e7a3b", + "generator_sha256": "9f490a7a3182df598650df17656093357ca2dbc3810ba730f12ed66ccec10a8a", "limitations": "No Laya, DeepSeek, live tools, customer traffic, Colibri, or task-success ground truth", - "observations_sha256": "cd2bd92ef228f7e94082e0aa3a777537151f04a4d8d56db82f5680cd2643536c", + "observations_sha256": "0cc19b419e73bd37f517e95057398af27c98bc0fe294693e3bd42420b7afaf8c", "platform": "Windows-10-10.0.26200-SP0", "python_version": "3.11.15", "rubric_sha256": "2cec43dee0cc3737708c618064253e59b7b9182906b8d19806d45afd887d350b", diff --git a/evidence/controlled-pairs-v1/observations.jsonl b/evidence/controlled-pairs-v1/observations.jsonl index 076d177..a324e32 100644 --- a/evidence/controlled-pairs-v1/observations.jsonl +++ b/evidence/controlled-pairs-v1/observations.jsonl @@ -1,10 +1,10 @@ -{"arm": "baseline", "authority_bypass": false, "cost_usd": 0.0, "label_positive": false, "latency_ms": 0.16989989671856165, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-001", "state_hash": "efd7e04e304f09a6178887c6b9c9af7bb17a4ddb48fab610b93dfb20f948db2d", "task_id": "ct-001", "trace_hash": "04b90347dcaa78d4f065b8de4776f9ee3a5c18c869f4401bbad0ee3975d85da2"} -{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.19440008327364922, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-001", "state_hash": "efd7e04e304f09a6178887c6b9c9af7bb17a4ddb48fab610b93dfb20f948db2d", "task_id": "ct-001", "trace_hash": "decf63a87158f2c3e1cd8d4206768e7119cfdd87354c69987b35ae037ba7524c"} -{"arm": "baseline", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.06029999349266291, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-002", "state_hash": "22b258f7f34d04e008731cbbd842c9c49c0633a0472ed7878483d921747f3561", "task_id": "ct-002", "trace_hash": "7c4df530de7503546bb35b2a5d9c1b392d4c275c0b067a0c92f7054c899b1373"} -{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.09410001803189516, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-002", "state_hash": "22b258f7f34d04e008731cbbd842c9c49c0633a0472ed7878483d921747f3561", "task_id": "ct-002", "trace_hash": "4db982f0fa1ce53a5a6968c2b32716ae30384ec81222cd5b1c45257623f4dfed"} -{"arm": "baseline", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.0467000063508749, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-003", "state_hash": "56fcb0df36bc59915a964b7d5437de95604d53b1a0f4fb6007a3828df3f1602f", "task_id": "ct-003", "trace_hash": "dee07c818aabe806ac8fb83958e0d0627e3524955368c4d98bfa0c2152f4fcc6"} -{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.04640000406652689, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-003", "state_hash": "56fcb0df36bc59915a964b7d5437de95604d53b1a0f4fb6007a3828df3f1602f", "task_id": "ct-003", "trace_hash": "845312248afa80ece1e48b32a37587c2e7df426577ea0ee8c87f5e97fc9ae9c0"} +{"arm": "baseline", "authority_bypass": false, "cost_usd": 0.0, "label_positive": false, "latency_ms": 0.1604999415576458, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-001", "state_hash": "efd7e04e304f09a6178887c6b9c9af7bb17a4ddb48fab610b93dfb20f948db2d", "task_id": "ct-001", "trace_hash": "04b90347dcaa78d4f065b8de4776f9ee3a5c18c869f4401bbad0ee3975d85da2"} +{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.16960001084953547, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-001", "state_hash": "efd7e04e304f09a6178887c6b9c9af7bb17a4ddb48fab610b93dfb20f948db2d", "task_id": "ct-001", "trace_hash": "decf63a87158f2c3e1cd8d4206768e7119cfdd87354c69987b35ae037ba7524c"} +{"arm": "baseline", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.055699958465993404, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-002", "state_hash": "22b258f7f34d04e008731cbbd842c9c49c0633a0472ed7878483d921747f3561", "task_id": "ct-002", "trace_hash": "7c4df530de7503546bb35b2a5d9c1b392d4c275c0b067a0c92f7054c899b1373"} +{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.09440002031624317, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-002", "state_hash": "22b258f7f34d04e008731cbbd842c9c49c0633a0472ed7878483d921747f3561", "task_id": "ct-002", "trace_hash": "4db982f0fa1ce53a5a6968c2b32716ae30384ec81222cd5b1c45257623f4dfed"} +{"arm": "baseline", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.04579999949783087, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-003", "state_hash": "56fcb0df36bc59915a964b7d5437de95604d53b1a0f4fb6007a3828df3f1602f", "task_id": "ct-003", "trace_hash": "dee07c818aabe806ac8fb83958e0d0627e3524955368c4d98bfa0c2152f4fcc6"} +{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.04650000482797623, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-003", "state_hash": "56fcb0df36bc59915a964b7d5437de95604d53b1a0f4fb6007a3828df3f1602f", "task_id": "ct-003", "trace_hash": "845312248afa80ece1e48b32a37587c2e7df426577ea0ee8c87f5e97fc9ae9c0"} {"arm": "baseline", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.046900007873773575, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-004", "state_hash": "f60b888f7a5ef1624e0e607d51ee4bfab104f5bbfd254463aa1acaf37b09f3ec", "task_id": "ct-004", "trace_hash": "57d1b8a4a59b0bc00a3ffafd3fd41691e0c16f677c444d9b2fae8f18c0c6453c"} -{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.10770000517368317, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-004", "state_hash": "f60b888f7a5ef1624e0e607d51ee4bfab104f5bbfd254463aa1acaf37b09f3ec", "task_id": "ct-004", "trace_hash": "f6dc03187c87c7958541afc799b10f997b33b5ef21a101f2c7bcdfbecd758b00"} +{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.1050999853760004, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-004", "state_hash": "f60b888f7a5ef1624e0e607d51ee4bfab104f5bbfd254463aa1acaf37b09f3ec", "task_id": "ct-004", "trace_hash": "f6dc03187c87c7958541afc799b10f997b33b5ef21a101f2c7bcdfbecd758b00"} {"arm": "baseline", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.048200017772614956, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-005", "state_hash": "61a52ac380a9c94d6194d4bb9f56cee0b86fd552c1892caf0ca39615b1f610bf", "task_id": "ct-005", "trace_hash": "c9ceec4e5b641e2ccb3c4b305829b57527feb2ac0298504a149aab72901eac1a"} -{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.10879989713430405, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-005", "state_hash": "61a52ac380a9c94d6194d4bb9f56cee0b86fd552c1892caf0ca39615b1f610bf", "task_id": "ct-005", "trace_hash": "df4e144951449ff4ce4d44e0c9ce2ab490cf90f9c57bc2a79e1b861800687569"} +{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.10529998689889908, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-005", "state_hash": "61a52ac380a9c94d6194d4bb9f56cee0b86fd552c1892caf0ca39615b1f610bf", "task_id": "ct-005", "trace_hash": "df4e144951449ff4ce4d44e0c9ce2ab490cf90f9c57bc2a79e1b861800687569"} diff --git a/scripts/run_controlled_pairs.py b/scripts/run_controlled_pairs.py index 8e984b5..24c143a 100644 --- a/scripts/run_controlled_pairs.py +++ b/scripts/run_controlled_pairs.py @@ -6,13 +6,13 @@ from __future__ import annotations -from dataclasses import asdict, dataclass -from datetime import datetime, timezone -from hashlib import sha256 import json -from pathlib import Path import platform import sys +from dataclasses import asdict, dataclass +from datetime import UTC, datetime +from hashlib import sha256 +from pathlib import Path from time import perf_counter REPOSITORY_ROOT = Path(__file__).resolve().parents[1] @@ -26,7 +26,13 @@ from c3r.runtime import RuntimeRequest, StandaloneController from c3r.state_compiler import StateCompiler from c3r.state_schema import ( - ActionDefinition, ActionFamily, AuthorityPolicy, Provenance, RawState, RiskClass, + ActionCandidate, + ActionDefinition, + ActionFamily, + AuthorityPolicy, + Provenance, + RawState, + RiskClass, ValueEstimate, ) from c3r.telemetry.trace_ledger import TraceLedger @@ -81,7 +87,7 @@ def _request(case: ControlledCase, arm: str) -> RuntimeRequest: def _run(case: ControlledCase, arm: str) -> dict[str, object]: - effects = [] + effects: list[ActionCandidate] = [] key = b"controlled-verifier-test-key" verifier = VerifierFirewall( {"policy": lambda _: VerifierDecision(case.verifier_accepts, "fixture policy")}, @@ -126,7 +132,7 @@ def _run(case: ControlledCase, arm: str) -> dict[str, object]: def collect() -> tuple[list[dict[str, object]], dict[str, object]]: observations = [_run(case, arm) for case in CASES for arm in ("baseline", "c3r")] cases_json = json.dumps([asdict(case) for case in CASES], sort_keys=True, default=str) - manifest = { + manifest: dict[str, object] = { "evidence_kind": "controlled", "task_population": "five C3R-authored authority/selection fixtures", "rubric_sha256": sha256(cases_json.encode("utf-8")).hexdigest(), @@ -148,7 +154,7 @@ def main() -> None: manifest.update({ "observations_sha256": sha256(observations_bytes).hexdigest(), "generator_sha256": sha256(Path(__file__).read_bytes()).hexdigest(), - "collected_at": datetime.now(timezone.utc).isoformat(), + "collected_at": datetime.now(UTC).isoformat(), "python_version": platform.python_version(), "platform": platform.platform(), }) From 2e23dcf0d6f5c75cf65a761a619bfb1327d7838a Mon Sep 17 00:00:00 2001 From: wilkont Date: Tue, 22 Sep 2026 15:33:37 -0500 Subject: [PATCH 11/55] Document Cloud Run staging and internal-task telemetry scope --- docs/cloud-run-staging.md | 46 ++++++++++++++++++++++++++++++++++ docs/empirical-release-plan.md | 7 ++++-- docs/standalone-launch.md | 23 ++++++++++++++--- 3 files changed, 71 insertions(+), 5 deletions(-) create mode 100644 docs/cloud-run-staging.md diff --git a/docs/cloud-run-staging.md b/docs/cloud-run-staging.md new file mode 100644 index 0000000..a5022d1 --- /dev/null +++ b/docs/cloud-run-staging.md @@ -0,0 +1,46 @@ +# Cloud Run staging decision (not a deployment record) + +C3R's selected standalone hostname strategy is the Google-managed HTTPS URL that +Cloud Run assigns to a service. A custom domain is optional. **No C3R Cloud Run +service has been deployed**, so this document does not claim a URL, live model route, +or completed canary. The first deployment must require IAM authentication; making +it public is a separate release action after qualification. + +## Preconditions before creating a service + +1. Name an accountable deployment/release owner and an on-call/rollback contact. +2. Approve an internal-task trace policy: source owner, permitted fields, redaction + tests, retention/deletion, access, and public-artifact scope. The current user + authorization allows only redacted telemetry from C3R-controlled internal tasks + in private shadow/canary tests; it excludes customer and product traffic. +3. Supply a calibrated, host-owned *pre-decision* estimate source. Example or + constant values must not be presented as measured production CVoC inputs. +4. Add a durable trace store, independent ledger-head anchor, backup and deletion + procedure, monitoring, alert thresholds, request budget, and a kill switch. +5. Package a trusted ingress adapter. Cloud Run requires the ingress container to + listen on `0.0.0.0:$PORT`, while `C3RHTTPServer` deliberately binds only to + loopback. Do not loosen that invariant merely to make a container start. An + authenticated ingress must forward to the loopback boundary without trusting + caller-supplied policy, estimates, verification, or approvals. +6. Establish a private, authenticated service-to-service route to the GPU model; + never publish the model's localhost inference port. Verify the model checkpoint + backup before any GPU VM lifecycle change. + +## Staged promotion + +| Stage | Access and effect authority | Evidence required to advance | +| --- | --- | --- | +| Offline replay | No cloud endpoint or external effects | Reproducible internal-task pairs, source manifest, independent outcomes and held-out partition. | +| Private shadow | IAM-protected HTTPS URL; recommendation-only; internal tasks | Auth, rate-limit, redaction, ledger, provider-outage, rollback and kill-switch traces. | +| Private read-only canary | Restricted invokers; no external effects | Predeclared latency/cost/safety thresholds and independently reviewed run evidence. | +| Public read-only launch | Explicit public-access approval; still no external effects | Security review, owner/on-call, abuse controls, reproducible empirical claims, release-copy audit. | +| Reversible effects | Separate authority and approval | Complete mediation and reversible canary evidence. Not implied by the public read-only launch. | + +Keep the Hugging Face model and dataset labeled as previews until trained weights, +empirical data, held-out calibration, and reproducible evaluation artifacts exist. +MC-1 remains outside the standalone launch scope, not complete under the directive. + +Cloud Run references: [HTTPS service URL and invoking services](https://docs.cloud.google.com/run/docs/triggering/https-request), +[IAM service authentication](https://docs.cloud.google.com/run/docs/authenticating/overview), +[public versus authenticated deployment](https://docs.cloud.google.com/run/docs/deploying), +and [container listening contract](https://docs.cloud.google.com/run/docs/container-contract). diff --git a/docs/empirical-release-plan.md b/docs/empirical-release-plan.md index 6bbda77..87d742d 100644 --- a/docs/empirical-release-plan.md +++ b/docs/empirical-release-plan.md @@ -16,8 +16,11 @@ train, validation, or calibration. Self-hosted DeepSeek can generate proposals f prompts, but its own output is not an independent verifier or task-outcome label. The user approved **C3R-controlled internal tasks only** as the present -training/evaluation source. Do not ingest customer, product, Laya-user, DeepSeek-user, -or Colibri operational logs under this authorization. The first controlled source is +training/evaluation source, including redacted telemetry from those tasks in private +shadow/canary qualification. This does not approve customer, product, Laya-user, +DeepSeek-user, or Colibri operational logs. Collection remains disabled until a +recorded source owner, permitted fields, redaction review, retention/deletion rule, +access policy, and publication scope are in place. The first controlled source is the [five-case local paired replay](controlled-replay.md), which is a pipeline smoke test, not an empirical DecisionMix corpus or training/calibration set. Before expanding the controlled corpus, record the source owner, task diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index 6c9ccdd..752ce49 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -6,7 +6,7 @@ distinguishes tested code, private operational evidence, and public release evid | Gate | Current evidence | Exit condition | | --- | --- | --- | -| Hosted decision path | `StandaloneController` composes state compilation, inventory-masked hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. Provider latency/token usage is recorded without granting authority. A host-owned read-only request factory ignores caller policy, budgets, estimates, and approvals. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. A transactional SQLite trace sink survives restart and verifies its hash chain. Local integration tests pass. | Calibrated *pre-decision* value/cost estimate source, durable storage placement, independent ledger-head anchoring, TLS gateway, approved hostname, secret rotation, deployment and rollback rehearsal, monitoring/alerts, and live end-to-end traces. No public endpoint exists yet. | +| Hosted decision path | `StandaloneController` composes state compilation, inventory-masked hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. Provider latency/token usage is recorded without granting authority. A host-owned read-only request factory ignores caller policy, budgets, estimates, and approvals. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. A transactional SQLite trace sink survives restart and verifies its hash chain. Local integration tests pass. | Calibrated *pre-decision* value/cost estimate source, durable storage placement, independent ledger-head anchoring, TLS gateway, Google-managed HTTPS service URL, secret rotation, deployment and rollback rehearsal, monitoring/alerts, and live end-to-end traces. No C3R cloud service exists yet. | | DeepSeek | Official DeepSeek-V4.1-Flash shards are verified on the GCP GPU host and in a private GCS backup; localhost inference passed a smoke test. A provider bridge passes compiled state to a local OpenAI-compatible adapter and rejects remote transfer of local-only data in tests. | Secure service-to-service route and paired live C3R decision traces. The localhost model is not a public API. | | Empirical DecisionMix | Published preview has 144 synthetic records, deterministic splits, and hashes. A [source audit](laya-data-source-audit.md) confirms that Laya-associated typed-decisions examples are also synthetic, not observed C3R outcomes. | Approved trace source, consent/license/data-boundary review, deduplication, immutable split, provenance audit, and empirical dataset publication. Never relabel synthetic data empirical. | | C3R Laya | Pinned upstream integration and calibration-aware abstention are tested. The Hugging Face C3R model page is an integration preview without trained weights. | Train a C3R-derived checkpoint on governed training data, fit calibration on held-out data, preserve sealed test set, and publish weights, raw predictions, manifests, hashes, and reproducible metrics. | @@ -23,7 +23,10 @@ canary evidence are independently reviewed. ## Required operator inputs -1. Approved public hostname and DNS/TLS administration path. +1. The selected hostname strategy is a Google-managed Cloud Run HTTPS `run.app` URL. + It is available only after deployment; private staging must require Cloud Run IAM, + and public access is a separate, explicitly approved promotion. A custom domain is + optional, not a launch prerequisite. 2. A governed trace source with explicit data-use, retention, redaction, and publication permissions. Controlled internal tasks can seed an evaluation set, but cannot be passed off as representative customer traffic. The public Laya checkpoint and @@ -33,6 +36,20 @@ canary evidence are independently reviewed. manager rather than committed files. 4. Colibri deployment owner and an instrumented compatible build that emits native route, acceptance, latency, and cost traces without exporting private content. -5. A release owner to approve read-only and reversible canary thresholds and aborts. +5. An accountable deployment/release owner to approve retention, publication scope, + monitoring/on-call, read-only and reversible canary thresholds and aborts. Access + to a cloud account is not by itself an owner designation. + +## Internal-task telemetry authorization + +The current authorization includes **redacted telemetry from C3R-controlled internal +tasks during private shadow/canary tests**. It does not authorize customer or product +traffic, or Colibri operational logs, for training or evaluation. Keep collection +disabled until the source owner, permitted fields, redaction checks, retention and +deletion rules, access controls, and publication scope are recorded. A later public +service must not silently widen the data source. Internal-task evidence can qualify +the controlled task population only; it cannot establish field performance. No public production claim should be made while any exit condition above is unmet. +The [Cloud Run staging decision](cloud-run-staging.md) records the selected +launch-capable URL strategy and the prerequisites that still block deployment. From 889f29dfe0b682d5671f919be06b8bf833f70539 Mon Sep 17 00:00:00 2001 From: wilkont Date: Tue, 22 Sep 2026 20:47:36 -0500 Subject: [PATCH 12/55] Add governed standalone ingress and internal-task trace controls --- README.md | 6 + c3r/ingress_proxy.py | 149 +++++++++++++++++ c3r/telemetry/governed_store.py | 257 +++++++++++++++++++++++++++++ docs/cloud-run-staging.md | 32 ++-- docs/empirical-release-plan.md | 10 +- docs/internal-task-trace-policy.md | 83 ++++++++++ docs/standalone-launch.md | 32 ++-- tests/test_governed_trace_store.py | 115 +++++++++++++ tests/test_ingress_proxy.py | 137 +++++++++++++++ 9 files changed, 796 insertions(+), 25 deletions(-) create mode 100644 c3r/ingress_proxy.py create mode 100644 c3r/telemetry/governed_store.py create mode 100644 docs/internal-task-trace-policy.md create mode 100644 tests/test_governed_trace_store.py create mode 100644 tests/test_ingress_proxy.py diff --git a/README.md b/README.md index 8c3a2aa..4181eb8 100644 --- a/README.md +++ b/README.md @@ -155,6 +155,12 @@ without presenting generated fixtures as real training evidence. The empirical c only when provenance, licensing, held-out integrity, and calibration support are independently auditable. +For the first empirical source, ColomboAI approved only C3R-authored internal +tasks under the [interim trace policy](docs/internal-task-trace-policy.md). It +sets a 30-day private retention limit and requires independent review before +any de-identified row is published. Collection remains off until the technical +controls are verified; this approval does not make the preview empirical. + Required empirical metrics include accuracy, Brier score, ECE, maximum calibration error, NLL, selective risk versus coverage, abstention, escalation, p50/p95 latency, throughput, calls avoided, and cost per completed task. diff --git a/c3r/ingress_proxy.py b/c3r/ingress_proxy.py new file mode 100644 index 0000000..5228a24 --- /dev/null +++ b/c3r/ingress_proxy.py @@ -0,0 +1,149 @@ +"""Bounded ingress for an IAM/TLS-terminated host with a loopback C3R backend. + +The outer host still owns TLS, IAM, secrets, and abuse controls. This proxy never +trusts caller authorization as backend authorization and never logs request data. +""" + +from __future__ import annotations + +import hmac +import json +from http.client import HTTPConnection, HTTPException +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from threading import BoundedSemaphore + +from .http_service import MAX_REQUEST_BYTES, TokenBucket + + +MAX_RESPONSE_BYTES = 65_536 + + +class C3RIngressServer(ThreadingHTTPServer): + """Expose only the read-only API, with separate client and loopback secrets.""" + + daemon_threads = True + + def __init__( + self, + *, + upstream_port: int, + client_token: str, + upstream_token: str, + upstream_host: str = "127.0.0.1", + host: str = "0.0.0.0", + port: int = 8080, + requests_per_minute: int = 60, + max_in_flight: int = 16, + upstream_timeout_seconds: float = 5.0, + ) -> None: + if upstream_host not in {"127.0.0.1", "::1"}: + raise ValueError("upstream must be loopback") + if not 1 <= upstream_port <= 65535: + raise ValueError("invalid upstream port") + if min(len(client_token), len(upstream_token)) < 32: + raise ValueError("tokens must have at least 32 characters") + if hmac.compare_digest(client_token, upstream_token): + raise ValueError("client and upstream tokens must be different") + if max_in_flight < 1 or upstream_timeout_seconds <= 0: + raise ValueError("concurrency and timeout must be positive") + self.upstream_host = upstream_host + self.upstream_port = upstream_port + self.client_token = client_token + self.upstream_token = upstream_token + self.upstream_timeout_seconds = upstream_timeout_seconds + self.limiter = TokenBucket( + capacity=requests_per_minute, + refill_per_second=requests_per_minute / 60, + ) + self.in_flight = BoundedSemaphore(max_in_flight) + super().__init__((host, port), _IngressHandler) + + +class _IngressHandler(BaseHTTPRequestHandler): + server: C3RIngressServer + + def log_message(self, _format: str, *_args: object) -> None: + return + + def _send_error(self, status: int, code: str) -> None: + body = json.dumps({"error": code}, separators=(",", ":")).encode("utf-8") + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Cache-Control", "no-store") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + def _authorized(self) -> bool: + supplied = self.headers.get_all("X-C3R-Token", []) + return len(supplied) == 1 and hmac.compare_digest(self.server.client_token, supplied[0]) + + def _forward(self, method: str, body: bytes | None = None) -> None: + if not self.server.in_flight.acquire(blocking=False): + self._send_error(503, "over_capacity") + return + connection = HTTPConnection( + self.server.upstream_host, + self.server.upstream_port, + timeout=self.server.upstream_timeout_seconds, + ) + try: + headers = {"Authorization": "Bearer " + self.server.upstream_token} + if body is not None: + headers["Content-Type"] = "application/json" + connection.request(method, self.path, body=body, headers=headers) + response = connection.getresponse() + forwarded = response.read(MAX_RESPONSE_BYTES + 1) + if len(forwarded) > MAX_RESPONSE_BYTES: + self._send_error(502, "invalid_upstream_response") + return + decoded = json.loads(forwarded) + if not isinstance(decoded, dict) or not 200 <= response.status <= 599: + raise ValueError("invalid upstream response") + except (OSError, HTTPException, ValueError): + self._send_error(503, "upstream_unavailable") + return + finally: + connection.close() + self.server.in_flight.release() + self.send_response(response.status) + self.send_header("Content-Type", "application/json") + self.send_header("Cache-Control", "no-store") + self.send_header("Content-Length", str(len(forwarded))) + self.end_headers() + self.wfile.write(forwarded) + + def do_GET(self) -> None: + if self.path not in {"/health", "/metrics"}: + self._send_error(404, "not_found") + return + if self.path == "/metrics" and not self._authorized(): + self._send_error(401, "unauthorized") + return + self._forward("GET") + + def do_POST(self) -> None: + if self.path != "/v1/decisions": + self._send_error(404, "not_found") + return + if not self._authorized(): + self._send_error(401, "unauthorized") + return + if self.headers.get("Transfer-Encoding") is not None: + self._send_error(400, "unsupported_transfer_encoding") + return + if self.headers.get("Content-Type", "").split(";", 1)[0].strip().lower() != "application/json": + self._send_error(415, "unsupported_media_type") + return + if not self.server.limiter.take(): + self._send_error(429, "rate_limited") + return + lengths = self.headers.get_all("Content-Length", []) + try: + length = int(lengths[0]) if len(lengths) == 1 else -1 + except ValueError: + length = -1 + if length < 1 or length > MAX_REQUEST_BYTES: + self._send_error(413, "request_size_out_of_bounds") + return + self._forward("POST", self.rfile.read(length)) diff --git a/c3r/telemetry/governed_store.py b/c3r/telemetry/governed_store.py new file mode 100644 index 0000000..e8d8faf --- /dev/null +++ b/c3r/telemetry/governed_store.py @@ -0,0 +1,257 @@ +"""Fail-closed local admission and retention for C3R-authored internal traces. + +The host still owns encryption, backup purge, IAM, daily scheduling, and an +independent hash-head anchor. This store does not enable collection by itself. +""" + +from __future__ import annotations + +import json +import os +import re +import sqlite3 +import stat +from dataclasses import asdict, dataclass +from datetime import datetime, timedelta, timezone +from math import isfinite +from pathlib import Path +from threading import Lock +from typing import Callable + +from .trace import DecisionTrace +from .trace_ledger import LedgerRecord, _record_hash + + +RETENTION_DAYS = 30 +_TOKEN = re.compile(r"[A-Za-z0-9_.:-]{1,128}\Z") +_HASH = re.compile(r"[0-9a-f]{64}\Z") +_GENESIS = "0" * 64 + + +@dataclass(frozen=True, slots=True) +class SourceGrant: + source_id: str + owner: str + task_ids: frozenset[str] + rights_attested: bool + + def __post_init__(self) -> None: + if not self.rights_attested or not self.task_ids: + raise ValueError("source rights and task inventory must be attested") + for value in (self.source_id, self.owner, *self.task_ids): + if not _TOKEN.fullmatch(value): + raise ValueError("source registry identifiers must be bounded tokens") + + +def _validate_trace(trace: DecisionTrace) -> None: + identifiers = ( + trace.run_id, + trace.access_level, + trace.model_provider, + trace.authority_result, + *trace.candidate_ids, + *trace.artifact_refs, + ) + if trace.selected_action_id is not None: + identifiers += (trace.selected_action_id,) + if not _HASH.fullmatch(trace.state_hash) or any( + not _TOKEN.fullmatch(value) for value in identifiers + ): + raise ValueError("trace failed redaction schema") + # Artifact references are not yet admitted: the policy requires a separate + # opaque-reference registry and leakage review before that field is enabled. + if trace.artifact_refs: + raise ValueError("trace failed redaction schema") + for key, values in trace.probabilities.items(): + if not _TOKEN.fullmatch(key) or not values or any( + not isfinite(value) or not 0 <= value <= 1 for value in values + ): + raise ValueError("trace failed redaction schema") + for mapping in (trace.utility_quantiles, trace.system_cost): + if any(not _TOKEN.fullmatch(key) or not isfinite(value) for key, value in mapping.items()): + raise ValueError("trace failed redaction schema") + if any(value < 0 for value in trace.system_cost.values()): + raise ValueError("trace failed redaction schema") + for key, value in trace.task_outcome.items(): + if not _TOKEN.fullmatch(key): + raise ValueError("trace failed redaction schema") + if isinstance(value, str): + if not _TOKEN.fullmatch(value): + raise ValueError("trace failed redaction schema") + elif isinstance(value, bool): + pass + elif not isinstance(value, (int, float)) or not isfinite(value): + raise ValueError("trace failed redaction schema") + + +class GovernedTraceStore: + """SQLite source-gated traces with a prefix checkpoint for 30-day purge. + + `purge_expired` must be run by the deployment's daily scheduler, and its + result independently audited. Local deletion cannot erase external backups. + """ + + def __init__( + self, + path: Path, + *, + grants: tuple[SourceGrant, ...], + clock: Callable[[], datetime] = lambda: datetime.now(timezone.utc), + ) -> None: + if not path.parent.is_dir() or path.is_symlink(): + raise ValueError("store parent must exist and path must not be a symlink") + if not grants or len({grant.source_id for grant in grants}) != len(grants): + raise ValueError("unique approved sources are required") + if not path.exists(): + descriptor = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600) + os.close(descriptor) + if os.name == "posix" and stat.S_IMODE(path.stat().st_mode) & 0o077: + raise ValueError("store file must not be accessible to group or other users") + self._clock = clock + self._grants = {grant.source_id: grant for grant in grants} + self._lock = Lock() + self._db = sqlite3.connect(path, timeout=30, isolation_level=None, check_same_thread=False) + self._db.execute("PRAGMA journal_mode=DELETE") + self._db.execute("PRAGMA synchronous=FULL") + self._db.execute("PRAGMA secure_delete=ON") + self._db.execute( + "CREATE TABLE IF NOT EXISTS metadata (" + "id INTEGER PRIMARY KEY CHECK(id=1), checkpoint_hash TEXT NOT NULL, " + "last_collected_at TEXT NOT NULL)" + ) + self._db.execute( + "INSERT OR IGNORE INTO metadata (id, checkpoint_hash, last_collected_at) " + "VALUES (1, ?, '')", (_GENESIS,) + ) + self._db.execute( + "CREATE TABLE IF NOT EXISTS records (" + "sequence INTEGER PRIMARY KEY AUTOINCREMENT, run_id TEXT NOT NULL UNIQUE, " + "collected_at TEXT NOT NULL, previous_hash TEXT NOT NULL, " + "record_hash TEXT NOT NULL, canonical_json TEXT NOT NULL)" + ) + self._db.execute( + "CREATE TABLE IF NOT EXISTS purge_audit (" + "sequence INTEGER PRIMARY KEY AUTOINCREMENT, purged_at TEXT NOT NULL, " + "record_count INTEGER NOT NULL, checkpoint_hash TEXT NOT NULL)" + ) + if not self.verify(): + self._db.close() + raise ValueError("governed trace hash chain is invalid") + + def __enter__(self) -> GovernedTraceStore: + return self + + def __exit__(self, *_args: object) -> None: + self.close() + + def _now(self) -> datetime: + value = self._clock() + if value.tzinfo is None or value.utcoffset() != timedelta(0): + raise ValueError("trace clock must return UTC") + return value + + def records(self) -> tuple[LedgerRecord, ...]: + with self._lock: + rows = self._db.execute( + "SELECT previous_hash, record_hash, canonical_json " + "FROM records ORDER BY sequence" + ).fetchall() + return tuple(LedgerRecord(*row) for row in rows) + + def verify(self) -> bool: + with self._lock: + checkpoint, last_collected_at = self._db.execute( + "SELECT checkpoint_hash, last_collected_at FROM metadata WHERE id=1" + ).fetchone() + rows = self._db.execute( + "SELECT run_id, collected_at, previous_hash, record_hash, canonical_json " + "FROM records ORDER BY sequence" + ).fetchall() + if rows and rows[-1][1] != last_collected_at: + return False + previous = checkpoint + for run_id, collected_at, prior, digest, payload in rows: + if prior != previous: + return False + try: + decoded = json.loads(payload) + canonical = json.dumps(decoded, sort_keys=True, separators=(",", ":"), + ensure_ascii=False, allow_nan=False) + if decoded["trace"]["run_id"] != run_id or decoded["collected_at"] != collected_at: + return False + except (ValueError, TypeError, KeyError): + return False + if canonical != payload or _record_hash(prior, payload) != digest: + return False + previous = digest + return True + + def append(self, trace: DecisionTrace, *, source_id: str, task_id: str) -> LedgerRecord: + grant = self._grants.get(source_id) + if grant is None or task_id not in grant.task_ids: + raise ValueError("unapproved source or task") + _validate_trace(trace) + collected_at = self._now().isoformat(timespec="microseconds") + payload = json.dumps( + {"collected_at": collected_at, "source_id": source_id, "task_id": task_id, + "trace": asdict(trace)}, + sort_keys=True, separators=(",", ":"), ensure_ascii=False, allow_nan=False, + ) + with self._lock: + self._db.execute("BEGIN IMMEDIATE") + try: + checkpoint, last_collected_at = self._db.execute( + "SELECT checkpoint_hash, last_collected_at FROM metadata WHERE id=1" + ).fetchone() + if collected_at < last_collected_at: + raise ValueError("trace clock moved backwards") + row = self._db.execute( + "SELECT record_hash FROM records ORDER BY sequence DESC LIMIT 1" + ).fetchone() + previous = row[0] if row is not None else checkpoint + record = LedgerRecord(previous, _record_hash(previous, payload), payload) + self._db.execute( + "INSERT INTO records (run_id, collected_at, previous_hash, " + "record_hash, canonical_json) VALUES (?, ?, ?, ?, ?)", + (trace.run_id, collected_at, previous, record.record_hash, payload), + ) + self._db.execute( + "UPDATE metadata SET last_collected_at=? WHERE id=1", (collected_at,) + ) + self._db.execute("COMMIT") + except Exception: + self._db.execute("ROLLBACK") + raise + return record + + def purge_expired(self) -> int: + now = self._now() + cutoff = (now - timedelta(days=RETENTION_DAYS)).isoformat(timespec="microseconds") + with self._lock: + self._db.execute("BEGIN IMMEDIATE") + try: + expired = self._db.execute( + "SELECT sequence, record_hash FROM records " + "WHERE collected_at <= ? ORDER BY sequence", (cutoff,) + ).fetchall() + if not expired: + self._db.execute("COMMIT") + return 0 + last_sequence, checkpoint = expired[-1] + self._db.execute("DELETE FROM records WHERE sequence <= ?", (last_sequence,)) + self._db.execute("UPDATE metadata SET checkpoint_hash=? WHERE id=1", (checkpoint,)) + self._db.execute( + "INSERT INTO purge_audit (purged_at, record_count, checkpoint_hash) " + "VALUES (?, ?, ?)", (now.isoformat(timespec="microseconds"), len(expired), + checkpoint), + ) + self._db.execute("COMMIT") + except Exception: + self._db.execute("ROLLBACK") + raise + self._db.execute("VACUUM") + return len(expired) + + def close(self) -> None: + with self._lock: + self._db.close() diff --git a/docs/cloud-run-staging.md b/docs/cloud-run-staging.md index a5022d1..a8fa506 100644 --- a/docs/cloud-run-staging.md +++ b/docs/cloud-run-staging.md @@ -8,24 +8,36 @@ it public is a separate release action after qualification. ## Preconditions before creating a service -1. Name an accountable deployment/release owner and an on-call/rollback contact. -2. Approve an internal-task trace policy: source owner, permitted fields, redaction - tests, retention/deletion, access, and public-artifact scope. The current user - authorization allows only redacted telemetry from C3R-controlled internal tasks - in private shadow/canary tests; it excludes customer and product traffic. +1. The user designated ColomboAI's `@wilkont` account as interim deployment and + release owner in the [internal-task policy](internal-task-trace-policy.md). Name + a reachable on-call and rollback contact before hosting traffic. +2. Implement the approved internal-task trace policy: source registry, field + allowlist, redaction tests, 30-day private deletion including backups, access + audit, and publication review. The approved scope is only redacted telemetry + from C3R-controlled internal tasks in private shadow/canary tests; it excludes + customer and product traffic. Written approval alone does not enable collection. 3. Supply a calibrated, host-owned *pre-decision* estimate source. Example or constant values must not be presented as measured production CVoC inputs. 4. Add a durable trace store, independent ledger-head anchor, backup and deletion procedure, monitoring, alert thresholds, request budget, and a kill switch. -5. Package a trusted ingress adapter. Cloud Run requires the ingress container to - listen on `0.0.0.0:$PORT`, while `C3RHTTPServer` deliberately binds only to - loopback. Do not loosen that invariant merely to make a container start. An - authenticated ingress must forward to the loopback boundary without trusting - caller-supplied policy, estimates, verification, or approvals. +5. Package the tested `C3RIngressServer` with the loopback `C3RHTTPServer` in one + container. The ingress can listen on `0.0.0.0:$PORT`, while the backend retains + its loopback-only invariant. It allowlists `/health`, `/metrics`, and + `/v1/decisions`, requires a separate client token for protected routes, replaces + caller credentials with a distinct backend token, and caps body size, response + size, request rate, concurrency, and upstream wait time. Local tests cover these + boundaries and backend failure; no Cloud Run image or service has been built. + Bind this ingress only behind Cloud Run's IAM/TLS boundary at staging, with + tokens from Secret Manager. Do not publish it as a raw unauthenticated port. 6. Establish a private, authenticated service-to-service route to the GPU model; never publish the model's localhost inference port. Verify the model checkpoint backup before any GPU VM lifecycle change. +The ingress code is a transport boundary, not a deployment composition or a +production authorization system. A container entrypoint, host-owned measured +estimates, durable ledger storage, secret rotation, and end-to-end Cloud Run tests +remain necessary before staging traffic. + ## Staged promotion | Stage | Access and effect authority | Evidence required to advance | diff --git a/docs/empirical-release-plan.md b/docs/empirical-release-plan.md index 87d742d..95e5ada 100644 --- a/docs/empirical-release-plan.md +++ b/docs/empirical-release-plan.md @@ -17,10 +17,12 @@ prompts, but its own output is not an independent verifier or task-outcome label The user approved **C3R-controlled internal tasks only** as the present training/evaluation source, including redacted telemetry from those tasks in private -shadow/canary qualification. This does not approve customer, product, Laya-user, -DeepSeek-user, or Colibri operational logs. Collection remains disabled until a -recorded source owner, permitted fields, redaction review, retention/deletion rule, -access policy, and publication scope are in place. The first controlled source is +shadow/canary qualification. The [interim trace policy](internal-task-trace-policy.md) +designates `@wilkont` as the accountable ColomboAI account, caps private row-level +retention at 30 days, and defines restricted publication after review. This does not +approve customer, product, Laya-user, DeepSeek-user, or Colibri operational logs. +Collection remains disabled until the policy's technical activation gates are +implemented and verified. The first controlled source is the [five-case local paired replay](controlled-replay.md), which is a pipeline smoke test, not an empirical DecisionMix corpus or training/calibration set. Before expanding the controlled corpus, record the source owner, task diff --git a/docs/internal-task-trace-policy.md b/docs/internal-task-trace-policy.md new file mode 100644 index 0000000..84c1853 --- /dev/null +++ b/docs/internal-task-trace-policy.md @@ -0,0 +1,83 @@ +# C3R internal-task trace policy (interim approval) + +**Effective:** 2026-09-22. **Approval authority:** the user acting for ColomboAI in +this task. **Accountable interim deployment, release, and data owner:** Wilfried +Kouadio (`@wilkont`), as identified by the active ColomboAI GCP account. This +designation does not establish an on-call rota. The owner must nominate a reachable +operator and rollback contact before a hosted service receives traffic. + +This policy resolves the owner and data-use choices for the *controlled internal +task population only*. It does **not** authorize production traffic or certify +the service, checkpoint, or dataset. MC-1 remains excluded from the standalone +launch scope, not complete under the original directive. + +## Source and permitted uses + +- Admit only tasks authored and controlled by ColomboAI expressly for C3R + qualification, with task ID, authoring owner, creation time, and rights attestation. + The operator must reject imported customer, employee-personal, product, MC-1, + Colibri operational, Laya-user, and provider-user records. +- Permit private offline replay, shadow and read-only canary qualification, + training, held-out calibration, paired evaluation, and independent safety review + on this controlled source. Do not infer representative field performance from it. +- DeepSeek, Laya, Qwen, or frontier model outputs may be recorded as *model outputs*, + never as independently verified task truth. Record model/version and generation + provenance. Independently label outcomes and preserve sealed partitions. + +## Data minimization and access + +- Persist only pseudonymous run/task IDs, state hashes, controlled candidate IDs, + route/provider/version, numeric predictions and resource use, authority/verifier + results, independently adjudicated outcome labels, and approved opaque artifact + references. Keep prompts, completions, free-text reasoning, personal identifiers, + credentials, URLs containing tokens, and source documents out of the trace store. +- Redaction and an allowlist schema must reject unexpected fields before a trace is + written. Review a sample and run leakage tests before enabling each new task source. +- Keep private traces encrypted in ColomboAI-controlled storage with least-privilege + access limited to the owner and named C3R operators. Audit reads and exports. Do + not put trace data or access credentials in Git, a public Hugging Face repository, + or chat. + +## Retention, deletion, and publication + +- Private row-level redacted traces: **30 days maximum** from collection. The + operator must run and verify deletion at least daily, including replicas and + backups. No persistent raw prompt/response capture is authorized. A legal hold + or longer retention requires a new explicit approval before collection. +- Public release may contain reviewed aggregates, metric definitions, source and + split manifests, code, hashes, and independently reviewed *de-identified rows* + derived solely from the approved internal tasks. Remove task-specific private + content and opaque private artifact references. The owner must approve a + publication manifest and a second reviewer must sign off on leakage and rights + checks before publication. Public releases may persist indefinitely and cannot + reliably be recalled; publishing is a separate, irreversible gate. +- Training and calibration data must have frozen, contamination-checked partitions. + Keep sealed test rows private until evaluation is finalized; publication afterward + still requires the preceding row-level review. Report the controlled population + and limitations in every model/dataset card and launch claim. +- On revocation or policy breach, stop collection immediately, quarantine exports, + preserve a minimal incident audit record, and delete affected private data under + the approved retention/deletion procedure. + +## Activation gates + +This written approval **does not turn collection on**. Before the first live trace, +the owner must record the exact task-source registry, permitted-field schema, +redaction/leakage test results, IAM grants, encrypted storage location, daily +30-day deletion job and backup purge proof, access audit, independent ledger-head +anchor, on-call/rollback contact, and a private staging deployment. A reviewer +must verify these controls against an intentionally non-sensitive dry run. + +The reference [`GovernedTraceStore`](../c3r/telemetry/governed_store.py) enforces an +attested source/task allowlist, a bounded token/numeric trace schema, a local 30-day +purge operation, tamper checks, and a post-purge chain checkpoint in tests. It +deliberately admits no artifact references. The deployment must still schedule and +audit that purge daily, remove expired backup copies, encrypt and restrict the +storage, anchor the ledger head independently, and prove redaction on the exact +internal-task source. Token-shape checks cannot detect private names encoded in +identifier-like strings; source-specific allowlists and human leakage review remain +mandatory. A library method is not evidence these operations ran. + +No public endpoint, public dataset promotion, model-weight release, or broad access +is approved by this policy alone. Each requires its own evidence-matched release +decision. This document is a project governance record, not a legal opinion. diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index 752ce49..a0600ef 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -6,7 +6,7 @@ distinguishes tested code, private operational evidence, and public release evid | Gate | Current evidence | Exit condition | | --- | --- | --- | -| Hosted decision path | `StandaloneController` composes state compilation, inventory-masked hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. Provider latency/token usage is recorded without granting authority. A host-owned read-only request factory ignores caller policy, budgets, estimates, and approvals. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. A transactional SQLite trace sink survives restart and verifies its hash chain. Local integration tests pass. | Calibrated *pre-decision* value/cost estimate source, durable storage placement, independent ledger-head anchoring, TLS gateway, Google-managed HTTPS service URL, secret rotation, deployment and rollback rehearsal, monitoring/alerts, and live end-to-end traces. No C3R cloud service exists yet. | +| Hosted decision path | `StandaloneController` composes state compilation, inventory-masked hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. Provider latency/token usage is recorded without granting authority. A host-owned read-only request factory ignores caller policy, budgets, estimates, and approvals. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. A separate tested ingress allowlists read-only routes, uses distinct client/backend tokens, and fails closed on backend outage. A transactional SQLite trace sink survives restart and verifies its hash chain. Local integration tests pass. | Calibrated *pre-decision* value/cost estimate source, container composition, durable storage placement, independent ledger-head anchoring, TLS/IAM gateway, Google-managed HTTPS service URL, secret rotation, deployment and rollback rehearsal, monitoring/alerts, and live end-to-end traces. No C3R cloud service exists yet. | | DeepSeek | Official DeepSeek-V4.1-Flash shards are verified on the GCP GPU host and in a private GCS backup; localhost inference passed a smoke test. A provider bridge passes compiled state to a local OpenAI-compatible adapter and rejects remote transfer of local-only data in tests. | Secure service-to-service route and paired live C3R decision traces. The localhost model is not a public API. | | Empirical DecisionMix | Published preview has 144 synthetic records, deterministic splits, and hashes. A [source audit](laya-data-source-audit.md) confirms that Laya-associated typed-decisions examples are also synthetic, not observed C3R outcomes. | Approved trace source, consent/license/data-boundary review, deduplication, immutable split, provenance audit, and empirical dataset publication. Never relabel synthetic data empirical. | | C3R Laya | Pinned upstream integration and calibration-aware abstention are tested. The Hugging Face C3R model page is an integration preview without trained weights. | Train a C3R-derived checkpoint on governed training data, fit calibration on held-out data, preserve sealed test set, and publish weights, raw predictions, manifests, hashes, and reproducible metrics. | @@ -15,7 +15,11 @@ distinguishes tested code, private operational evidence, and public release evid | Public release | v0.1 alpha repository, papers, collection, model/dataset previews. | Update all cards and launch copy only after the corresponding evidence passes; perform security and dependency review; publish an evidence-matched standalone release. | The current code path is a **tested reference boundary**, not a production service. -The SQLite ledger is optional and does not by itself provide independent audit anchoring; +An opt-in governed SQLite trace store now validates internal-task source grants, +rejects unbounded text and private artifact references, and has a tested 30-day local +purge/checkpoint operation. It does not run a daily scheduler, delete backups, or +provide independent audit anchoring. The ordinary SQLite ledger is likewise +optional and does not by itself provide independent audit anchoring; the estimates are not yet empirically calibrated, and the HTTP server requires a separate TLS/authentication gateway. Keep effect execution disabled in this service until host-level complete mediation and @@ -28,7 +32,9 @@ canary evidence are independently reviewed. and public access is a separate, explicitly approved promotion. A custom domain is optional, not a launch prerequisite. 2. A governed trace source with explicit data-use, retention, redaction, and publication - permissions. Controlled internal tasks can seed an evaluation set, but cannot be + permissions. The [interim internal-task policy](internal-task-trace-policy.md) + records these choices, but its technical activation controls are not yet verified. + Controlled internal tasks can seed an evaluation set, but cannot be passed off as representative customer traffic. The public Laya checkpoint and typed-decisions benchmark, and self-hosted DeepSeek generations, do not satisfy this requirement; see the [source audit](laya-data-source-audit.md). @@ -36,19 +42,23 @@ canary evidence are independently reviewed. manager rather than committed files. 4. Colibri deployment owner and an instrumented compatible build that emits native route, acceptance, latency, and cost traces without exporting private content. -5. An accountable deployment/release owner to approve retention, publication scope, - monitoring/on-call, read-only and reversible canary thresholds and aborts. Access - to a cloud account is not by itself an owner designation. +5. The user designated ColomboAI's `@wilkont` account as interim accountable + deployment, release, and data owner in the + [internal-task policy](internal-task-trace-policy.md). The owner still needs a + reachable on-call/rollback contact, monitoring, read-only and reversible canary + thresholds and aborts. An account designation is not proof of operational coverage. ## Internal-task telemetry authorization The current authorization includes **redacted telemetry from C3R-controlled internal tasks during private shadow/canary tests**. It does not authorize customer or product -traffic, or Colibri operational logs, for training or evaluation. Keep collection -disabled until the source owner, permitted fields, redaction checks, retention and -deletion rules, access controls, and publication scope are recorded. A later public -service must not silently widen the data source. Internal-task evidence can qualify -the controlled task population only; it cannot establish field performance. +traffic, or Colibri operational logs, for training or evaluation. The +[interim policy](internal-task-trace-policy.md) records the owner, 30-day private +retention, data minimization, and reviewed publication scope. Keep collection +disabled until the policy's source registry, redaction tests, storage, access, +deletion, anchoring, and operator controls are verified. A later public service +must not silently widen the data source. Internal-task evidence can qualify the +controlled task population only; it cannot establish field performance. No public production claim should be made while any exit condition above is unmet. The [Cloud Run staging decision](cloud-run-staging.md) records the selected diff --git a/tests/test_governed_trace_store.py b/tests/test_governed_trace_store.py new file mode 100644 index 0000000..e7b81c2 --- /dev/null +++ b/tests/test_governed_trace_store.py @@ -0,0 +1,115 @@ +import sqlite3 +import tempfile +import unittest +from datetime import datetime, timedelta, timezone +from pathlib import Path + +from c3r.telemetry.governed_store import GovernedTraceStore, SourceGrant +from c3r.telemetry.trace import DecisionTrace + + +NOW = datetime(2026, 9, 22, 12, tzinfo=timezone.utc) + + +def trace(**changes): + fields = dict( + run_id="run_001", + state_hash="a" * 64, + access_level="internal", + model_provider="deepseek_v4_1_flash", + candidate_ids=("recommend",), + probabilities={"route": (0.8, 0.2)}, + utility_quantiles={"selected_lower_bound": 0.3}, + selected_action_id="recommend", + authority_result="verified", + system_cost={"latency_ms": 18.0}, + task_outcome={"status": "controlled_success"}, + artifact_refs=(), + ) + fields.update(changes) + return DecisionTrace(**fields) + + +class GovernedTraceStoreTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.path = Path(self.temp.name) / "governed.sqlite3" + self.grant = SourceGrant( + source_id="c3r_internal_001", + owner="wilkont", + task_ids=frozenset({"task_001"}), + rights_attested=True, + ) + + def tearDown(self): + self.temp.cleanup() + + def store(self, *, now=NOW): + return GovernedTraceStore(self.path, grants=(self.grant,), clock=lambda: now) + + def test_admits_only_attested_registered_internal_task(self): + with self.store() as store: + record = store.append(trace(), source_id="c3r_internal_001", task_id="task_001") + self.assertEqual(len(store.records()), 1) + self.assertEqual(store.records()[0].record_hash, record.record_hash) + with self.assertRaisesRegex(ValueError, "unapproved source or task"): + store.append(trace(run_id="run_002"), source_id="laya_logs", task_id="task_001") + with self.assertRaisesRegex(ValueError, "unapproved source or task"): + store.append(trace(run_id="run_003"), source_id="c3r_internal_001", task_id="other") + with self.assertRaises(sqlite3.IntegrityError): + store.append(trace(), source_id="c3r_internal_001", task_id="task_001") + + def test_unattested_source_cannot_be_registered(self): + with self.assertRaisesRegex(ValueError, "rights"): + SourceGrant(source_id="imported", owner="wilkont", + task_ids=frozenset({"task_001"}), rights_attested=False) + + def test_rejects_free_text_or_private_artifact_reference(self): + with self.store() as store: + with self.assertRaisesRegex(ValueError, "redaction"): + store.append(trace(task_outcome={"status": "email me at a@example.com"}), + source_id="c3r_internal_001", task_id="task_001") + with self.assertRaisesRegex(ValueError, "redaction"): + store.append(trace(artifact_refs=("gs://private/object",)), + source_id="c3r_internal_001", task_id="task_001") + self.assertEqual(store.records(), ()) + + def test_purges_after_30_days_and_preserves_remaining_chain(self): + with self.store(now=NOW - timedelta(days=31)) as store: + first = store.append(trace(), source_id="c3r_internal_001", task_id="task_001") + with self.store(now=NOW - timedelta(days=1)) as store: + second = store.append(trace(run_id="run_002"), source_id="c3r_internal_001", + task_id="task_001") + with self.store(now=NOW) as store: + self.assertEqual(store.purge_expired(), 1) + remaining = store.records() + self.assertEqual(len(remaining), 1) + self.assertEqual(remaining[0].record_hash, second.record_hash) + self.assertEqual(remaining[0].previous_hash, first.record_hash) + self.assertTrue(store.verify()) + + def test_rejects_tampered_database_on_reopen(self): + with self.store() as store: + store.append(trace(), source_id="c3r_internal_001", task_id="task_001") + db = sqlite3.connect(self.path) + try: + db.execute("UPDATE records SET canonical_json = '{}' WHERE sequence = 1") + db.commit() + finally: + db.close() + with self.assertRaisesRegex(ValueError, "hash chain"): + self.store() + + def test_clock_rollback_cannot_relabel_traces_after_purge(self): + with self.store() as store: + store.append(trace(), source_id="c3r_internal_001", task_id="task_001") + with self.store(now=NOW + timedelta(days=31)) as store: + self.assertEqual(store.purge_expired(), 1) + with self.store(now=NOW - timedelta(days=1)) as store: + with self.assertRaisesRegex(ValueError, "clock moved backwards"): + store.append(trace(run_id="run_002"), source_id="c3r_internal_001", + task_id="task_001") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_ingress_proxy.py b/tests/test_ingress_proxy.py new file mode 100644 index 0000000..7a55e98 --- /dev/null +++ b/tests/test_ingress_proxy.py @@ -0,0 +1,137 @@ +import json +import threading +import unittest +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from urllib.error import HTTPError +from urllib.request import Request, urlopen + +from c3r.ingress_proxy import C3RIngressServer + + +CLIENT_TOKEN = "client-token-that-is-long-enough-for-tests" +UPSTREAM_TOKEN = "upstream-token-that-is-long-enough-for-tests" + + +class _UpstreamHandler(BaseHTTPRequestHandler): + def log_message(self, _format, *_args): + return + + def do_GET(self): + self.server.seen.append((self.path, dict(self.headers), b"")) + self._reply(200, {"status": "ok"}) + + def do_POST(self): + length = int(self.headers["Content-Length"]) + self.server.seen.append((self.path, dict(self.headers), self.rfile.read(length))) + self._reply(200, {"route": "recommendation"}) + + def _reply(self, status, payload): + body = json.dumps(payload).encode() + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + +class IngressProxyTests(unittest.TestCase): + def setUp(self): + self.upstream = ThreadingHTTPServer(("127.0.0.1", 0), _UpstreamHandler) + self.upstream.seen = [] + self.upstream_thread = threading.Thread(target=self.upstream.serve_forever, daemon=True) + self.upstream_thread.start() + self.ingress = C3RIngressServer( + upstream_port=self.upstream.server_port, + client_token=CLIENT_TOKEN, + upstream_token=UPSTREAM_TOKEN, + host="127.0.0.1", + port=0, + upstream_timeout_seconds=0.5, + ) + self.ingress_thread = threading.Thread(target=self.ingress.serve_forever, daemon=True) + self.ingress_thread.start() + self.base = f"http://127.0.0.1:{self.ingress.server_port}" + + def tearDown(self): + self.ingress.shutdown() + self.ingress.server_close() + self.ingress_thread.join(timeout=2) + self.upstream.shutdown() + self.upstream.server_close() + self.upstream_thread.join(timeout=2) + + def request(self, path, *, method="GET", token=CLIENT_TOKEN, payload=None): + headers = {"Authorization": "Bearer cloud-run-identity-token"} + if token is not None: + headers["X-C3R-Token"] = token + body = None if payload is None else json.dumps(payload).encode() + if body is not None: + headers["Content-Type"] = "application/json" + request = Request(self.base + path, data=body, headers=headers, method=method) + try: + with urlopen(request, timeout=2) as response: + return response.status, json.load(response) + except HTTPError as error: + return error.code, json.load(error) + + def test_private_decision_forwards_only_internal_authorization(self): + status, body = self.request("/v1/decisions", method="POST", payload={"goal": "inspect"}) + self.assertEqual((status, body["route"]), (200, "recommendation")) + path, headers, forwarded = self.upstream.seen[-1] + self.assertEqual(path, "/v1/decisions") + self.assertEqual(headers["Authorization"], f"Bearer {UPSTREAM_TOKEN}") + self.assertNotIn("X-C3R-Token", headers) + self.assertEqual(json.loads(forwarded), {"goal": "inspect"}) + + def test_missing_or_wrong_client_token_never_reaches_upstream(self): + for token in (None, "wrong"): + status, body = self.request("/v1/decisions", method="POST", token=token, + payload={"goal": "inspect"}) + self.assertEqual((status, body["error"]), (401, "unauthorized")) + self.assertEqual(self.upstream.seen, []) + + def test_health_is_unprivileged_but_metrics_require_token(self): + self.assertEqual(self.request("/health", token=None)[0], 200) + self.assertEqual(self.request("/metrics", token=None)[0], 401) + + def test_rejects_unknown_path_without_contacting_upstream(self): + self.assertEqual(self.request("/admin")[0], 404) + self.assertEqual(self.request("/v1/decisions?debug=1", method="POST", + payload={"goal": "inspect"})[0], 404) + self.assertEqual(self.upstream.seen, []) + + def test_rejects_ambiguous_framing_and_non_json_body(self): + for headers in ( + {"Content-Type": "text/plain"}, + {"Transfer-Encoding": "chunked"}, + {"Content-Length": "2, 3"}, + ): + request = Request( + self.base + "/v1/decisions", + data=b"{}", + headers={"X-C3R-Token": CLIENT_TOKEN, **headers}, + method="POST", + ) + with self.assertRaises(HTTPError) as raised: + urlopen(request, timeout=2) + self.assertIn(raised.exception.code, {400, 413, 415}) + self.assertEqual(self.upstream.seen, []) + + def test_upstream_failure_is_not_mistaken_for_success(self): + self.upstream.shutdown() + self.upstream.server_close() + status, body = self.request("/v1/decisions", method="POST", payload={"goal": "inspect"}) + self.assertEqual((status, body["error"]), (503, "upstream_unavailable")) + + def test_upstream_route_must_be_loopback_and_secrets_distinct(self): + with self.assertRaisesRegex(ValueError, "loopback"): + C3RIngressServer(upstream_host="example.com", upstream_port=8081, + client_token=CLIENT_TOKEN, upstream_token=UPSTREAM_TOKEN, + host="127.0.0.1", port=0) + with self.assertRaisesRegex(ValueError, "different"): + C3RIngressServer(upstream_port=8081, client_token=CLIENT_TOKEN, + upstream_token=CLIENT_TOKEN, host="127.0.0.1", port=0) + + +if __name__ == "__main__": + unittest.main() From 41fb6cf9fbd5abea519d406885d9a213e79a938c Mon Sep 17 00:00:00 2001 From: wilkont Date: Tue, 22 Sep 2026 20:56:24 -0500 Subject: [PATCH 13/55] Bind governed traces to trusted source and task --- c3r/telemetry/governed_store.py | 19 +++++++++++++++++++ docs/internal-task-trace-policy.md | 3 +++ tests/test_governed_trace_store.py | 10 +++++++++- 3 files changed, 31 insertions(+), 1 deletion(-) diff --git a/c3r/telemetry/governed_store.py b/c3r/telemetry/governed_store.py index e8d8faf..ba5560c 100644 --- a/c3r/telemetry/governed_store.py +++ b/c3r/telemetry/governed_store.py @@ -255,3 +255,22 @@ def purge_expired(self) -> int: def close(self) -> None: with self._lock: self._db.close() + + +class BoundGovernedTraceSink: + """Bind one approved task in trusted host code to the controller's trace API. + + A shared HTTP controller must not reuse this binding across unrelated tasks. + The host, never the caller payload, chooses the source and task identifiers. + """ + + def __init__(self, store: GovernedTraceStore, *, source_id: str, task_id: str) -> None: + grant = store._grants.get(source_id) + if grant is None or task_id not in grant.task_ids: + raise ValueError("unapproved source or task") + self._store = store + self._source_id = source_id + self._task_id = task_id + + def append(self, trace: DecisionTrace) -> LedgerRecord: + return self._store.append(trace, source_id=self._source_id, task_id=self._task_id) diff --git a/docs/internal-task-trace-policy.md b/docs/internal-task-trace-policy.md index 84c1853..f399fb6 100644 --- a/docs/internal-task-trace-policy.md +++ b/docs/internal-task-trace-policy.md @@ -77,6 +77,9 @@ storage, anchor the ledger head independently, and prove redaction on the exact internal-task source. Token-shape checks cannot detect private names encoded in identifier-like strings; source-specific allowlists and human leakage review remain mandatory. A library method is not evidence these operations ran. +The trusted host may use `BoundGovernedTraceSink` to bind a single approved +source/task to the controller's one-argument trace interface. It must not reuse +one binding for unrelated requests or accept those identifiers from callers. No public endpoint, public dataset promotion, model-weight release, or broad access is approved by this policy alone. Each requires its own evidence-matched release diff --git a/tests/test_governed_trace_store.py b/tests/test_governed_trace_store.py index e7b81c2..7738399 100644 --- a/tests/test_governed_trace_store.py +++ b/tests/test_governed_trace_store.py @@ -4,7 +4,7 @@ from datetime import datetime, timedelta, timezone from pathlib import Path -from c3r.telemetry.governed_store import GovernedTraceStore, SourceGrant +from c3r.telemetry.governed_store import BoundGovernedTraceSink, GovernedTraceStore, SourceGrant from c3r.telemetry.trace import DecisionTrace @@ -64,6 +64,14 @@ def test_unattested_source_cannot_be_registered(self): SourceGrant(source_id="imported", owner="wilkont", task_ids=frozenset({"task_001"}), rights_attested=False) + def test_trusted_host_can_bind_source_and_task_for_runtime_sink(self): + with self.store() as store: + sink = BoundGovernedTraceSink(store, source_id="c3r_internal_001", task_id="task_001") + record = sink.append(trace()) + self.assertEqual(store.records(), (record,)) + with self.assertRaisesRegex(ValueError, "unapproved source or task"): + BoundGovernedTraceSink(store, source_id="imported", task_id="task_001") + def test_rejects_free_text_or_private_artifact_reference(self): with self.store() as store: with self.assertRaisesRegex(ValueError, "redaction"): From dd6292912aad24dc50a2cb3dd5a7b52f4a35e84e Mon Sep 17 00:00:00 2001 From: wilkont Date: Tue, 22 Sep 2026 21:00:24 -0500 Subject: [PATCH 14/55] Make malformed ingress test portable across platforms --- tests/test_ingress_proxy.py | 36 ++++++++++++++++++++++++------------ 1 file changed, 24 insertions(+), 12 deletions(-) diff --git a/tests/test_ingress_proxy.py b/tests/test_ingress_proxy.py index 7a55e98..dd1e615 100644 --- a/tests/test_ingress_proxy.py +++ b/tests/test_ingress_proxy.py @@ -1,4 +1,5 @@ import json +import socket import threading import unittest from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer @@ -101,20 +102,31 @@ def test_rejects_unknown_path_without_contacting_upstream(self): self.assertEqual(self.upstream.seen, []) def test_rejects_ambiguous_framing_and_non_json_body(self): - for headers in ( - {"Content-Type": "text/plain"}, - {"Transfer-Encoding": "chunked"}, - {"Content-Length": "2, 3"}, + request = Request( + self.base + "/v1/decisions", + data=b"{}", + headers={"X-C3R-Token": CLIENT_TOKEN, "Content-Type": "text/plain"}, + method="POST", + ) + with self.assertRaises(HTTPError) as raised: + urlopen(request, timeout=2) + self.assertEqual(raised.exception.code, 415) + + for framing in ( + b"Content-Length: 2\r\nTransfer-Encoding: chunked\r\n", + b"Content-Length: 2\r\nContent-Length: 2\r\n", ): - request = Request( - self.base + "/v1/decisions", - data=b"{}", - headers={"X-C3R-Token": CLIENT_TOKEN, **headers}, - method="POST", + packet = ( + b"POST /v1/decisions HTTP/1.0\r\nHost: localhost\r\n" + + f"X-C3R-Token: {CLIENT_TOKEN}\r\n".encode() + + b"Content-Type: application/json\r\n" + + framing + b"\r\n{}" ) - with self.assertRaises(HTTPError) as raised: - urlopen(request, timeout=2) - self.assertIn(raised.exception.code, {400, 413, 415}) + with socket.create_connection(("127.0.0.1", self.ingress.server_port), timeout=2) as sock: + sock.sendall(packet) + response = sock.recv(4096) + self.assertTrue(response.startswith(b"HTTP/1.0 400 ") or + response.startswith(b"HTTP/1.0 413 ")) self.assertEqual(self.upstream.seen, []) def test_upstream_failure_is_not_mistaken_for_success(self): From 350d8e18e4247966e3dd98487ff94a6d3fe37eff Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 08:27:25 -0500 Subject: [PATCH 15/55] Add fail-closed staging container composition --- .dockerignore | 12 ++++ Dockerfile | 10 ++++ c3r/serve.py | 123 ++++++++++++++++++++++++++++++++++++++ docs/cloud-run-staging.md | 23 ++++--- docs/standalone-launch.md | 2 +- tests/test_serve.py | 91 ++++++++++++++++++++++++++++ 6 files changed, 253 insertions(+), 8 deletions(-) create mode 100644 .dockerignore create mode 100644 Dockerfile create mode 100644 c3r/serve.py create mode 100644 tests/test_serve.py diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..8d1d309 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,12 @@ +.git +.github +__pycache__ +*.pyc +.pytest_cache +.ruff_cache +.venv +tests +data +evidence +docs +assets diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..c1e92cd --- /dev/null +++ b/Dockerfile @@ -0,0 +1,10 @@ +FROM python:3.11-slim + +ENV PYTHONDONTWRITEBYTECODE=1 PYTHONUNBUFFERED=1 +WORKDIR /app +COPY pyproject.toml README.md LICENSE ./ +COPY c3r ./c3r +RUN pip install --no-cache-dir . && useradd --create-home --uid 10001 c3r +USER c3r +EXPOSE 8080 +CMD ["python", "-m", "c3r.serve"] diff --git a/c3r/serve.py b/c3r/serve.py new file mode 100644 index 0000000..819a0a4 --- /dev/null +++ b/c3r/serve.py @@ -0,0 +1,123 @@ +"""Fail-closed composition of a host-supplied C3R controller and HTTP ingress. + +The host entrypoint owns the action catalog, measured estimates, verifier policy, +and trace sink. This module supplies no demo estimates or implicit data collection. +""" + +from __future__ import annotations + +import importlib +import os +import signal +import threading +from collections.abc import Callable, Mapping +from typing import Protocol, cast + +from .http_service import C3RHTTPServer, RequestFactory +from .ingress_proxy import C3RIngressServer +from .runtime import StandaloneController + + +class HostBuilder(Protocol): + def __call__(self) -> tuple[StandaloneController, RequestFactory]: ... + + +def _required(values: Mapping[str, str], name: str) -> str: + value = values.get(name, "") + if not value: + raise ValueError(f"{name} is required") + return value + + +def _port(values: Mapping[str, str], name: str, default: int) -> int: + raw = values.get(name, str(default)) + try: + value = int(raw) + except ValueError as error: + raise ValueError(f"{name} must be a TCP port") from error + if not 1 <= value <= 65535: + raise ValueError(f"{name} must be a TCP port") + return value + + +def load_host_builder(reference: str) -> HostBuilder: + """Load an explicitly configured, trusted module:function host composition.""" + module_name, separator, attribute = reference.partition(":") + if not separator or not module_name or not attribute or not attribute.isidentifier(): + raise ValueError("C3R_HOST_ENTRYPOINT must be module:function") + module = importlib.import_module(module_name) + builder = getattr(module, attribute) + if not callable(builder): + raise ValueError("C3R_HOST_ENTRYPOINT is not callable") + return cast(HostBuilder, builder) + + +def build_servers( + values: Mapping[str, str], + *, + builder_loader: Callable[[str], HostBuilder] = load_host_builder, +) -> tuple[C3RHTTPServer, C3RIngressServer]: + """Validate configuration before binding a public interface.""" + reference = _required(values, "C3R_HOST_ENTRYPOINT") + client_token = _required(values, "C3R_CLIENT_TOKEN") + backend_token = _required(values, "C3R_BACKEND_TOKEN") + port = _port(values, "PORT", 8080) + backend_port = _port(values, "C3R_BACKEND_PORT", 8081) + if port == backend_port: + raise ValueError("ingress and backend ports must differ") + runtime, factory = builder_loader(reference)() + if runtime.effect_execution_enabled: + raise ValueError("host must be recommendation-only") + backend = C3RHTTPServer( + runtime=runtime, + request_factory=factory, + bearer_token=backend_token, + port=backend_port, + ) + try: + ingress = C3RIngressServer( + upstream_port=backend.server_port, + client_token=client_token, + upstream_token=backend_token, + port=port, + ) + except BaseException: + backend.server_close() + raise + return backend, ingress + + +def main() -> None: + backend, ingress = build_servers(os.environ) + stop = threading.Event() + + def request_stop(_signum: int, _frame: object) -> None: + stop.set() + + signal.signal(signal.SIGTERM, request_stop) + signal.signal(signal.SIGINT, request_stop) + backend_thread = threading.Thread(target=backend.serve_forever, daemon=True) + ingress_thread = threading.Thread(target=ingress.serve_forever, daemon=True) + backend_started = False + ingress_started = False + try: + backend_thread.start() + backend_started = True + ingress_thread.start() + ingress_started = True + stop.wait() + finally: + if ingress_started: + ingress.shutdown() + if backend_started: + backend.shutdown() + ingress.server_close() + backend.server_close() + if ingress_started: + ingress_thread.join(timeout=5) + if backend_started: + backend_thread.join(timeout=5) + + +if __name__ == "__main__": + main() diff --git a/docs/cloud-run-staging.md b/docs/cloud-run-staging.md index a8fa506..54ff4f9 100644 --- a/docs/cloud-run-staging.md +++ b/docs/cloud-run-staging.md @@ -20,23 +20,32 @@ it public is a separate release action after qualification. constant values must not be presented as measured production CVoC inputs. 4. Add a durable trace store, independent ledger-head anchor, backup and deletion procedure, monitoring, alert thresholds, request budget, and a kill switch. -5. Package the tested `C3RIngressServer` with the loopback `C3RHTTPServer` in one - container. The ingress can listen on `0.0.0.0:$PORT`, while the backend retains +5. The repository now includes a fail-closed `Dockerfile` and `python -m c3r.serve` + composition for the tested `C3RIngressServer` and loopback `C3RHTTPServer`. + The ingress listens on `0.0.0.0:$PORT`, while the backend retains its loopback-only invariant. It allowlists `/health`, `/metrics`, and `/v1/decisions`, requires a separate client token for protected routes, replaces caller credentials with a distinct backend token, and caps body size, response size, request rate, concurrency, and upstream wait time. Local tests cover these - boundaries and backend failure; no Cloud Run image or service has been built. + boundaries, backend failure, and local server composition; no Cloud Run image or + service has been built. Docker Desktop was unavailable during this check, so the + image itself has not yet been built or scanned. Bind this ingress only behind Cloud Run's IAM/TLS boundary at staging, with tokens from Secret Manager. Do not publish it as a raw unauthenticated port. 6. Establish a private, authenticated service-to-service route to the GPU model; never publish the model's localhost inference port. Verify the model checkpoint backup before any GPU VM lifecycle change. -The ingress code is a transport boundary, not a deployment composition or a -production authorization system. A container entrypoint, host-owned measured -estimates, durable ledger storage, secret rotation, and end-to-end Cloud Run tests -remain necessary before staging traffic. +`C3R_HOST_ENTRYPOINT=trusted_module:build` is mandatory. The trusted callable must +return a recommendation-only `StandaloneController` and a `RequestFactory` whose +estimate source uses measured, pre-decision values. The container supplies **no** +sample catalog, constant estimates, data collection, or production credentials. +`C3R_CLIENT_TOKEN` and `C3R_BACKEND_TOKEN` are distinct mandatory secrets of at +least 32 characters; `PORT` defaults to 8080 and `C3R_BACKEND_PORT` to 8081. +The trusted host module must be included in a derived private image or approved +runtime package. This is packaging, not proof that a calibrated host or secure +storage exists. Image build/scan, durable ledger storage, secret rotation, and +end-to-end Cloud Run tests remain necessary before staging traffic. ## Staged promotion diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index a0600ef..a2c96bb 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -6,7 +6,7 @@ distinguishes tested code, private operational evidence, and public release evid | Gate | Current evidence | Exit condition | | --- | --- | --- | -| Hosted decision path | `StandaloneController` composes state compilation, inventory-masked hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. Provider latency/token usage is recorded without granting authority. A host-owned read-only request factory ignores caller policy, budgets, estimates, and approvals. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. A separate tested ingress allowlists read-only routes, uses distinct client/backend tokens, and fails closed on backend outage. A transactional SQLite trace sink survives restart and verifies its hash chain. Local integration tests pass. | Calibrated *pre-decision* value/cost estimate source, container composition, durable storage placement, independent ledger-head anchoring, TLS/IAM gateway, Google-managed HTTPS service URL, secret rotation, deployment and rollback rehearsal, monitoring/alerts, and live end-to-end traces. No C3R cloud service exists yet. | +| Hosted decision path | `StandaloneController` composes state compilation, inventory-masked hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. Provider latency/token usage is recorded without granting authority. A host-owned read-only request factory ignores caller policy, budgets, estimates, and approvals. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. A separate tested ingress allowlists read-only routes, uses distinct client/backend tokens, and fails closed on backend outage. A fail-closed container entrypoint composes both servers only when a trusted host builder and distinct secrets are supplied; the Docker image has not yet been built. A transactional SQLite trace sink survives restart and verifies its hash chain. Local integration tests pass. | Calibrated *pre-decision* value/cost estimate source, trusted host composition, image build/scan, durable storage placement, independent ledger-head anchoring, TLS/IAM gateway, Google-managed HTTPS service URL, secret rotation, deployment and rollback rehearsal, monitoring/alerts, and live end-to-end traces. No C3R cloud service exists yet. | | DeepSeek | Official DeepSeek-V4.1-Flash shards are verified on the GCP GPU host and in a private GCS backup; localhost inference passed a smoke test. A provider bridge passes compiled state to a local OpenAI-compatible adapter and rejects remote transfer of local-only data in tests. | Secure service-to-service route and paired live C3R decision traces. The localhost model is not a public API. | | Empirical DecisionMix | Published preview has 144 synthetic records, deterministic splits, and hashes. A [source audit](laya-data-source-audit.md) confirms that Laya-associated typed-decisions examples are also synthetic, not observed C3R outcomes. | Approved trace source, consent/license/data-boundary review, deduplication, immutable split, provenance audit, and empirical dataset publication. Never relabel synthetic data empirical. | | C3R Laya | Pinned upstream integration and calibration-aware abstention are tested. The Hugging Face C3R model page is an integration preview without trained weights. | Train a C3R-derived checkpoint on governed training data, fit calibration on held-out data, preserve sealed test set, and publish weights, raw predictions, manifests, hashes, and reproducible metrics. | diff --git a/tests/test_serve.py b/tests/test_serve.py new file mode 100644 index 0000000..e76b3f8 --- /dev/null +++ b/tests/test_serve.py @@ -0,0 +1,91 @@ +import socket +import threading +import unittest +from urllib.request import urlopen + +from c3r.serve import build_servers, load_host_builder +from tests.test_http_service import HostFactory +from tests.test_runtime import controller + + +CLIENT_TOKEN = "client-token-with-at-least-thirty-two-characters" +BACKEND_TOKEN = "backend-token-with-at-least-thirty-two-characters" + + +def free_port(): + with socket.socket() as sock: + sock.bind(("127.0.0.1", 0)) + return sock.getsockname()[1] + + +def config(): + return { + "C3R_HOST_ENTRYPOINT": "trusted_host:build", + "C3R_CLIENT_TOKEN": CLIENT_TOKEN, + "C3R_BACKEND_TOKEN": BACKEND_TOKEN, + "PORT": str(free_port()), + "C3R_BACKEND_PORT": str(free_port()), + } + + +class ServeTests(unittest.TestCase): + def test_missing_host_or_secret_fails_before_binding(self): + values = config() + del values["C3R_HOST_ENTRYPOINT"] + with self.assertRaisesRegex(ValueError, "C3R_HOST_ENTRYPOINT"): + build_servers(values) + values = config() + del values["C3R_CLIENT_TOKEN"] + with self.assertRaisesRegex(ValueError, "C3R_CLIENT_TOKEN"): + build_servers(values) + + def test_invalid_port_and_equal_tokens_fail(self): + values = config() + values["PORT"] = "0" + with self.assertRaisesRegex(ValueError, "PORT"): + build_servers(values, builder_loader=lambda _: lambda: (controller()[0], HostFactory())) + values = config() + values["C3R_BACKEND_TOKEN"] = CLIENT_TOKEN + with self.assertRaisesRegex(ValueError, "different"): + build_servers(values, builder_loader=lambda _: lambda: (controller()[0], HostFactory())) + + def test_effect_enabled_host_is_rejected(self): + with self.assertRaisesRegex(ValueError, "recommendation-only"): + build_servers( + config(), + builder_loader=lambda _: lambda: ( + controller(executor=lambda _: None)[0], HostFactory() + ), + ) + + def test_host_reference_must_be_explicit(self): + for reference in ("module", "module:", ":build", "module:bad.name"): + with self.subTest(reference=reference), self.assertRaises(ValueError): + load_host_builder(reference) + + def test_composed_health_path(self): + backend, ingress = build_servers( + config(), + builder_loader=lambda _: lambda: (controller()[0], HostFactory()), + ) + threads = [ + threading.Thread(target=backend.serve_forever, daemon=True), + threading.Thread(target=ingress.serve_forever, daemon=True), + ] + try: + for thread in threads: + thread.start() + with urlopen(f"http://127.0.0.1:{ingress.server_port}/health", timeout=2) as response: + self.assertEqual(response.status, 200) + self.assertEqual(response.read(), b'{"status":"ok"}') + finally: + ingress.shutdown() + backend.shutdown() + ingress.server_close() + backend.server_close() + for thread in threads: + thread.join(timeout=2) + + +if __name__ == "__main__": + unittest.main() From c853ed5924b0540657c34fb8d190882b40c0b807 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 08:29:20 -0500 Subject: [PATCH 16/55] Build staging image in CI --- .github/workflows/ci.yml | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index fdd4afb..f316df5 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -20,3 +20,8 @@ jobs: python-version: ${{ matrix.python-version }} - run: python -m unittest discover -s tests -v + image-build: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - run: docker build --pull --tag c3r:ci . From afcae2eda9f28c50bdcd60fba015cd46636f5480 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 08:30:54 -0500 Subject: [PATCH 17/55] Record CI image-build evidence without deployment claim --- docs/cloud-run-staging.md | 8 ++++---- docs/standalone-launch.md | 2 +- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/docs/cloud-run-staging.md b/docs/cloud-run-staging.md index 54ff4f9..10d7c33 100644 --- a/docs/cloud-run-staging.md +++ b/docs/cloud-run-staging.md @@ -27,9 +27,9 @@ it public is a separate release action after qualification. `/v1/decisions`, requires a separate client token for protected routes, replaces caller credentials with a distinct backend token, and caps body size, response size, request rate, concurrency, and upstream wait time. Local tests cover these - boundaries, backend failure, and local server composition; no Cloud Run image or - service has been built. Docker Desktop was unavailable during this check, so the - image itself has not yet been built or scanned. + boundaries, backend failure, and local server composition. GitHub CI has built + the staging image, but no image has been published, scanned, or deployed to + Cloud Run. Docker Desktop was unavailable during the local check. Bind this ingress only behind Cloud Run's IAM/TLS boundary at staging, with tokens from Secret Manager. Do not publish it as a raw unauthenticated port. 6. Establish a private, authenticated service-to-service route to the GPU model; @@ -44,7 +44,7 @@ sample catalog, constant estimates, data collection, or production credentials. least 32 characters; `PORT` defaults to 8080 and `C3R_BACKEND_PORT` to 8081. The trusted host module must be included in a derived private image or approved runtime package. This is packaging, not proof that a calibrated host or secure -storage exists. Image build/scan, durable ledger storage, secret rotation, and +storage exists. A publishable pinned/scanned image, durable ledger storage, secret rotation, and end-to-end Cloud Run tests remain necessary before staging traffic. ## Staged promotion diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index a2c96bb..094ea19 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -6,7 +6,7 @@ distinguishes tested code, private operational evidence, and public release evid | Gate | Current evidence | Exit condition | | --- | --- | --- | -| Hosted decision path | `StandaloneController` composes state compilation, inventory-masked hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. Provider latency/token usage is recorded without granting authority. A host-owned read-only request factory ignores caller policy, budgets, estimates, and approvals. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. A separate tested ingress allowlists read-only routes, uses distinct client/backend tokens, and fails closed on backend outage. A fail-closed container entrypoint composes both servers only when a trusted host builder and distinct secrets are supplied; the Docker image has not yet been built. A transactional SQLite trace sink survives restart and verifies its hash chain. Local integration tests pass. | Calibrated *pre-decision* value/cost estimate source, trusted host composition, image build/scan, durable storage placement, independent ledger-head anchoring, TLS/IAM gateway, Google-managed HTTPS service URL, secret rotation, deployment and rollback rehearsal, monitoring/alerts, and live end-to-end traces. No C3R cloud service exists yet. | +| Hosted decision path | `StandaloneController` composes state compilation, inventory-masked hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. Provider latency/token usage is recorded without granting authority. A host-owned read-only request factory ignores caller policy, budgets, estimates, and approvals. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. A separate tested ingress allowlists read-only routes, uses distinct client/backend tokens, and fails closed on backend outage. A fail-closed container entrypoint composes both servers only when a trusted host builder and distinct secrets are supplied; GitHub CI builds the staging image. A transactional SQLite trace sink survives restart and verifies its hash chain. Local integration tests pass. | Calibrated *pre-decision* value/cost estimate source, trusted host composition, pinned/scanned image publication, durable storage placement, independent ledger-head anchoring, TLS/IAM gateway, Google-managed HTTPS service URL, secret rotation, deployment and rollback rehearsal, monitoring/alerts, and live end-to-end traces. No C3R cloud service exists yet. | | DeepSeek | Official DeepSeek-V4.1-Flash shards are verified on the GCP GPU host and in a private GCS backup; localhost inference passed a smoke test. A provider bridge passes compiled state to a local OpenAI-compatible adapter and rejects remote transfer of local-only data in tests. | Secure service-to-service route and paired live C3R decision traces. The localhost model is not a public API. | | Empirical DecisionMix | Published preview has 144 synthetic records, deterministic splits, and hashes. A [source audit](laya-data-source-audit.md) confirms that Laya-associated typed-decisions examples are also synthetic, not observed C3R outcomes. | Approved trace source, consent/license/data-boundary review, deduplication, immutable split, provenance audit, and empirical dataset publication. Never relabel synthetic data empirical. | | C3R Laya | Pinned upstream integration and calibration-aware abstention are tested. The Hugging Face C3R model page is an integration preview without trained weights. | Train a C3R-derived checkpoint on governed training data, fit calibration on held-out data, preserve sealed test set, and publish weights, raw predictions, manifests, hashes, and reproducible metrics. | From 559e212d45afba3de5ffec83d01377c709792aa9 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 08:38:55 -0500 Subject: [PATCH 18/55] Gate staging image on high-severity vulnerability scan --- .github/workflows/ci.yml | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index f316df5..ad551cd 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -25,3 +25,19 @@ jobs: steps: - uses: actions/checkout@v4 - run: docker build --pull --tag c3r:ci . + - name: Scan image for high and critical vulnerabilities + uses: aquasecurity/trivy-action@v0.36.0 + with: + image-ref: c3r:ci + format: json + output: trivy-image.json + severity: HIGH,CRITICAL + exit-code: '1' + - name: Preserve scan findings + if: always() + uses: actions/upload-artifact@v4 + with: + name: trivy-image-${{ github.run_id }} + path: trivy-image.json + if-no-files-found: ignore + retention-days: 30 From 0ec8911fe3804dbcd2e83d58b4cbe2baac6a4863 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 08:44:21 -0500 Subject: [PATCH 19/55] Reduce staging image attack surface and retain scan gate --- .github/workflows/ci.yml | 1 + Dockerfile | 8 +++----- docs/image-security.md | 18 ++++++++++++++++++ 3 files changed, 22 insertions(+), 5 deletions(-) create mode 100644 docs/image-security.md diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index ad551cd..f1bc1a4 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -25,6 +25,7 @@ jobs: steps: - uses: actions/checkout@v4 - run: docker build --pull --tag c3r:ci . + - run: docker run --rm c3r:ci python -c 'import c3r.serve' - name: Scan image for high and critical vulnerabilities uses: aquasecurity/trivy-action@v0.36.0 with: diff --git a/Dockerfile b/Dockerfile index c1e92cd..e021105 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,10 +1,8 @@ -FROM python:3.11-slim +FROM python:3.11-alpine3.24 ENV PYTHONDONTWRITEBYTECODE=1 PYTHONUNBUFFERED=1 WORKDIR /app -COPY pyproject.toml README.md LICENSE ./ -COPY c3r ./c3r -RUN pip install --no-cache-dir . && useradd --create-home --uid 10001 c3r -USER c3r +COPY --chown=10001:10001 c3r ./c3r +USER 10001:10001 EXPOSE 8080 CMD ["python", "-m", "c3r.serve"] diff --git a/docs/image-security.md b/docs/image-security.md new file mode 100644 index 0000000..315f03d --- /dev/null +++ b/docs/image-security.md @@ -0,0 +1,18 @@ +# Staging image vulnerability triage + +The first GitHub Actions image scan, [run 48](https://github.com/ColomboAI-com/c3r/actions/runs/35868557321), +failed its HIGH/CRITICAL gate. Its Trivy JSON artifact reported **46 HIGH, zero +CRITICAL** entries: 44 in the Debian 13.7 base image and two Python packaging +packages (`jaraco.context` 5.3.0 and `wheel` 0.45.1). Many Debian entries +repeated the same CVE across `util-linux` subpackages and had no fixed version +listed. They have not been waived or marked safe. + +The container now uses the official Python 3.11 Alpine 3.24 runtime image and +copies only the stdlib-based `c3r` package. It does not run `pip install` or +retain build-time Python packaging dependencies. The image runs as UID/GID +10001 and is smoke-imported in CI before scanning. The HIGH/CRITICAL scan +continues to fail the image job until a subsequent result shows zero findings. + +Even a clean scan is only one security gate. A production image still needs a +digest pin, registry publication, signature/SBOM, deployment IAM and network +review, and the live safety qualification in `standalone-launch.md`. From 598c03329c1acc41ef1bda0066a2c922f2f398c4 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 08:46:55 -0500 Subject: [PATCH 20/55] Remove vulnerable runtime tools and record on-call owner --- Dockerfile | 1 + docs/cloud-run-staging.md | 7 ++++--- docs/image-security.md | 10 +++++++--- docs/internal-task-trace-policy.md | 9 ++++++--- docs/standalone-launch.md | 12 +++++++----- 5 files changed, 25 insertions(+), 14 deletions(-) diff --git a/Dockerfile b/Dockerfile index e021105..83567b2 100644 --- a/Dockerfile +++ b/Dockerfile @@ -2,6 +2,7 @@ FROM python:3.11-alpine3.24 ENV PYTHONDONTWRITEBYTECODE=1 PYTHONUNBUFFERED=1 WORKDIR /app +RUN python -m pip uninstall -y jaraco.context wheel COPY --chown=10001:10001 c3r ./c3r USER 10001:10001 EXPOSE 8080 diff --git a/docs/cloud-run-staging.md b/docs/cloud-run-staging.md index 10d7c33..585e151 100644 --- a/docs/cloud-run-staging.md +++ b/docs/cloud-run-staging.md @@ -8,9 +8,10 @@ it public is a separate release action after qualification. ## Preconditions before creating a service -1. The user designated ColomboAI's `@wilkont` account as interim deployment and - release owner in the [internal-task policy](internal-task-trace-policy.md). Name - a reachable on-call and rollback contact before hosting traffic. +1. Wilfried Kouadio (`@wilkont`) is the interim deployment/release owner and + confirmed interim on-call/rollback operator in the + [internal-task policy](internal-task-trace-policy.md). Verify the work-email + alert route, acknowledgement and rollback access before hosting traffic. 2. Implement the approved internal-task trace policy: source registry, field allowlist, redaction tests, 30-day private deletion including backups, access audit, and publication review. The approved scope is only redacted telemetry diff --git a/docs/image-security.md b/docs/image-security.md index 315f03d..4e4339c 100644 --- a/docs/image-security.md +++ b/docs/image-security.md @@ -9,9 +9,13 @@ listed. They have not been waived or marked safe. The container now uses the official Python 3.11 Alpine 3.24 runtime image and copies only the stdlib-based `c3r` package. It does not run `pip install` or -retain build-time Python packaging dependencies. The image runs as UID/GID -10001 and is smoke-imported in CI before scanning. The HIGH/CRITICAL scan -continues to fail the image job until a subsequent result shows zero findings. +retain the vulnerable `jaraco.context` and `wheel` packaging packages. The +intermediate [run 50](https://github.com/ColomboAI-com/c3r/actions/runs/35869194641) +had zero HIGH/CRITICAL Alpine OS findings but two HIGH Python packaging +findings, so the removal still requires scan verification. The image runs as +UID/GID 10001 and is smoke-imported in CI before scanning. The HIGH/CRITICAL +scan continues to fail the image job until a subsequent result shows zero +findings. Even a clean scan is only one security gate. A production image still needs a digest pin, registry publication, signature/SBOM, deployment IAM and network diff --git a/docs/internal-task-trace-policy.md b/docs/internal-task-trace-policy.md index f399fb6..23dad7b 100644 --- a/docs/internal-task-trace-policy.md +++ b/docs/internal-task-trace-policy.md @@ -2,9 +2,12 @@ **Effective:** 2026-09-22. **Approval authority:** the user acting for ColomboAI in this task. **Accountable interim deployment, release, and data owner:** Wilfried -Kouadio (`@wilkont`), as identified by the active ColomboAI GCP account. This -designation does not establish an on-call rota. The owner must nominate a reachable -operator and rollback contact before a hosted service receives traffic. +Kouadio (`@wilkont`), as identified by the active ColomboAI GCP account. On +2026-09-23, Wilfried confirmed that he will serve as the interim human on-call +and rollback operator via his ColomboAI work identity +(`wilfried.k@colomboai.com`). This names the responder; it does not prove that +monitoring, alert delivery, acknowledgement, or rollback access is configured. +Those controls must be verified before hosted traffic. This policy resolves the owner and data-use choices for the *controlled internal task population only*. It does **not** authorize production traffic or certify diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index 094ea19..9528466 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -42,11 +42,13 @@ canary evidence are independently reviewed. manager rather than committed files. 4. Colibri deployment owner and an instrumented compatible build that emits native route, acceptance, latency, and cost traces without exporting private content. -5. The user designated ColomboAI's `@wilkont` account as interim accountable - deployment, release, and data owner in the - [internal-task policy](internal-task-trace-policy.md). The owner still needs a - reachable on-call/rollback contact, monitoring, read-only and reversible canary - thresholds and aborts. An account designation is not proof of operational coverage. +5. The user designated Wilfried Kouadio (`@wilkont`) as interim accountable + deployment, release, and data owner, and on 2026-09-23 Wilfried confirmed he + will be the interim human on-call/rollback operator. The + [internal-task policy](internal-task-trace-policy.md) records this. Alert + delivery, acknowledgement, rollback access, monitoring, and read-only and + reversible canary thresholds and aborts still need verification. A named + responder is not proof of operational coverage. ## Internal-task telemetry authorization From 5486c19906f121c566c50796a062f6983f4f2551 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 08:50:32 -0500 Subject: [PATCH 21/55] Remove vendored vulnerable build tools from runtime --- Dockerfile | 2 +- docs/image-security.md | 5 ++++- 2 files changed, 5 insertions(+), 2 deletions(-) diff --git a/Dockerfile b/Dockerfile index 83567b2..29837ac 100644 --- a/Dockerfile +++ b/Dockerfile @@ -2,7 +2,7 @@ FROM python:3.11-alpine3.24 ENV PYTHONDONTWRITEBYTECODE=1 PYTHONUNBUFFERED=1 WORKDIR /app -RUN python -m pip uninstall -y jaraco.context wheel +RUN python -m pip uninstall -y setuptools wheel jaraco.context COPY --chown=10001:10001 c3r ./c3r USER 10001:10001 EXPOSE 8080 diff --git a/docs/image-security.md b/docs/image-security.md index 4e4339c..e3418be 100644 --- a/docs/image-security.md +++ b/docs/image-security.md @@ -12,7 +12,10 @@ copies only the stdlib-based `c3r` package. It does not run `pip install` or retain the vulnerable `jaraco.context` and `wheel` packaging packages. The intermediate [run 50](https://github.com/ColomboAI-com/c3r/actions/runs/35869194641) had zero HIGH/CRITICAL Alpine OS findings but two HIGH Python packaging -findings, so the removal still requires scan verification. The image runs as +findings. [Run 52](https://github.com/ColomboAI-com/c3r/actions/runs/35869500670) +showed that uninstalling those names alone did not remove copies vendored +inside `setuptools`; the runtime therefore removes `setuptools` itself, which +C3R does not import. This still requires scan verification. The image runs as UID/GID 10001 and is smoke-imported in CI before scanning. The HIGH/CRITICAL scan continues to fail the image job until a subsequent result shows zero findings. From cd5e97df48c146e2d0c364e90b14494f3582b958 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 08:56:00 -0500 Subject: [PATCH 22/55] Pin clean runtime image and current scanner --- .github/workflows/ci.yml | 1 + Dockerfile | 2 +- docs/image-security.md | 12 ++++++++---- 3 files changed, 10 insertions(+), 5 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index f1bc1a4..e09f17a 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -29,6 +29,7 @@ jobs: - name: Scan image for high and critical vulnerabilities uses: aquasecurity/trivy-action@v0.36.0 with: + version: v0.74.0 image-ref: c3r:ci format: json output: trivy-image.json diff --git a/Dockerfile b/Dockerfile index 29837ac..e2c1750 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,4 +1,4 @@ -FROM python:3.11-alpine3.24 +FROM python:3.11-alpine3.24@sha256:cd04730b8511def3fbf14204d66a0c1536f290b8e896ed5a94cd64cb15ac1356 ENV PYTHONDONTWRITEBYTECODE=1 PYTHONUNBUFFERED=1 WORKDIR /app diff --git a/docs/image-security.md b/docs/image-security.md index e3418be..9da606c 100644 --- a/docs/image-security.md +++ b/docs/image-security.md @@ -17,9 +17,13 @@ showed that uninstalling those names alone did not remove copies vendored inside `setuptools`; the runtime therefore removes `setuptools` itself, which C3R does not import. This still requires scan verification. The image runs as UID/GID 10001 and is smoke-imported in CI before scanning. The HIGH/CRITICAL -scan continues to fail the image job until a subsequent result shows zero -findings. +scan failed closed until [run 54](https://github.com/ColomboAI-com/c3r/actions/runs/35869921657) +built and smoke-imported the image and reported zero HIGH/CRITICAL findings +for both Alpine and Python packages. The base image is now pinned to the exact +digest resolved in that successful build. A follow-up run with Trivy v0.74.0 +must still confirm the pinned image; lower-severity findings have not been +triaged in this HIGH/CRITICAL-only report. -Even a clean scan is only one security gate. A production image still needs a -digest pin, registry publication, signature/SBOM, deployment IAM and network +Even a clean scan is only one security gate. A production image still needs +registry publication, signature/SBOM, deployment IAM and network review, and the live safety qualification in `standalone-launch.md`. From e991472ea747f4a587e1ac41475b4153ed5ec2a5 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 09:03:43 -0500 Subject: [PATCH 23/55] Record verified pinned image scan and reviewer gate --- docs/cloud-run-staging.md | 10 ++++++---- docs/image-security.md | 16 ++++++++-------- docs/standalone-launch.md | 5 ++++- 3 files changed, 18 insertions(+), 13 deletions(-) diff --git a/docs/cloud-run-staging.md b/docs/cloud-run-staging.md index 585e151..ced3b25 100644 --- a/docs/cloud-run-staging.md +++ b/docs/cloud-run-staging.md @@ -28,9 +28,11 @@ it public is a separate release action after qualification. `/v1/decisions`, requires a separate client token for protected routes, replaces caller credentials with a distinct backend token, and caps body size, response size, request rate, concurrency, and upstream wait time. Local tests cover these - boundaries, backend failure, and local server composition. GitHub CI has built - the staging image, but no image has been published, scanned, or deployed to - Cloud Run. Docker Desktop was unavailable during the local check. + boundaries, backend failure, and local server composition. GitHub CI has + built and HIGH/CRITICAL-scanned the digest-pinned staging image with zero + findings in [run 56](https://github.com/ColomboAI-com/c3r/actions/runs/35870568965), + but no image has been published to a registry or deployed to Cloud Run. + Docker Desktop was unavailable during the local check. Bind this ingress only behind Cloud Run's IAM/TLS boundary at staging, with tokens from Secret Manager. Do not publish it as a raw unauthenticated port. 6. Establish a private, authenticated service-to-service route to the GPU model; @@ -45,7 +47,7 @@ sample catalog, constant estimates, data collection, or production credentials. least 32 characters; `PORT` defaults to 8080 and `C3R_BACKEND_PORT` to 8081. The trusted host module must be included in a derived private image or approved runtime package. This is packaging, not proof that a calibrated host or secure -storage exists. A publishable pinned/scanned image, durable ledger storage, secret rotation, and +storage exists. Registry publication, durable ledger storage, secret rotation, and end-to-end Cloud Run tests remain necessary before staging traffic. ## Staged promotion diff --git a/docs/image-security.md b/docs/image-security.md index 9da606c..d5f07fb 100644 --- a/docs/image-security.md +++ b/docs/image-security.md @@ -15,14 +15,14 @@ had zero HIGH/CRITICAL Alpine OS findings but two HIGH Python packaging findings. [Run 52](https://github.com/ColomboAI-com/c3r/actions/runs/35869500670) showed that uninstalling those names alone did not remove copies vendored inside `setuptools`; the runtime therefore removes `setuptools` itself, which -C3R does not import. This still requires scan verification. The image runs as -UID/GID 10001 and is smoke-imported in CI before scanning. The HIGH/CRITICAL -scan failed closed until [run 54](https://github.com/ColomboAI-com/c3r/actions/runs/35869921657) -built and smoke-imported the image and reported zero HIGH/CRITICAL findings -for both Alpine and Python packages. The base image is now pinned to the exact -digest resolved in that successful build. A follow-up run with Trivy v0.74.0 -must still confirm the pinned image; lower-severity findings have not been -triaged in this HIGH/CRITICAL-only report. +C3R does not import. The image runs as UID/GID 10001 and is smoke-imported in +CI before scanning. [Run 56](https://github.com/ColomboAI-com/c3r/actions/runs/35870568965) +passed the three-version Python test matrix, built the digest-pinned image, +smoke-imported C3R, and scanned it with Trivy v0.74.0. Its retained JSON +artifact reports **zero HIGH and zero CRITICAL findings** for the scanned +image. This is a point-in-time result, not a waiver for the earlier image or a +guarantee against future disclosures. Lower-severity findings were outside the +configured HIGH/CRITICAL scan and have not been triaged by this report. Even a clean scan is only one security gate. A production image still needs registry publication, signature/SBOM, deployment IAM and network diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index 9528466..296a643 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -6,7 +6,7 @@ distinguishes tested code, private operational evidence, and public release evid | Gate | Current evidence | Exit condition | | --- | --- | --- | -| Hosted decision path | `StandaloneController` composes state compilation, inventory-masked hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. Provider latency/token usage is recorded without granting authority. A host-owned read-only request factory ignores caller policy, budgets, estimates, and approvals. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. A separate tested ingress allowlists read-only routes, uses distinct client/backend tokens, and fails closed on backend outage. A fail-closed container entrypoint composes both servers only when a trusted host builder and distinct secrets are supplied; GitHub CI builds the staging image. A transactional SQLite trace sink survives restart and verifies its hash chain. Local integration tests pass. | Calibrated *pre-decision* value/cost estimate source, trusted host composition, pinned/scanned image publication, durable storage placement, independent ledger-head anchoring, TLS/IAM gateway, Google-managed HTTPS service URL, secret rotation, deployment and rollback rehearsal, monitoring/alerts, and live end-to-end traces. No C3R cloud service exists yet. | +| Hosted decision path | `StandaloneController` composes state compilation, inventory-masked hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. Provider latency/token usage is recorded without granting authority. A host-owned read-only request factory ignores caller policy, budgets, estimates, and approvals. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. A separate tested ingress allowlists read-only routes, uses distinct client/backend tokens, and fails closed on backend outage. A fail-closed container entrypoint composes both servers only when a trusted host builder and distinct secrets are supplied. GitHub CI builds, smoke-imports, and [HIGH/CRITICAL-scans the digest-pinned staging image](image-security.md) with zero findings in run 56. A transactional SQLite trace sink survives restart and verifies its hash chain. Local integration tests pass. | Calibrated *pre-decision* value/cost estimate source, trusted host composition, registry publication of the scanned image, durable storage placement, independent ledger-head anchoring, TLS/IAM gateway, Google-managed HTTPS service URL, secret rotation, deployment and rollback rehearsal, monitoring/alerts, and live end-to-end traces. No C3R cloud service exists yet. | | DeepSeek | Official DeepSeek-V4.1-Flash shards are verified on the GCP GPU host and in a private GCS backup; localhost inference passed a smoke test. A provider bridge passes compiled state to a local OpenAI-compatible adapter and rejects remote transfer of local-only data in tests. | Secure service-to-service route and paired live C3R decision traces. The localhost model is not a public API. | | Empirical DecisionMix | Published preview has 144 synthetic records, deterministic splits, and hashes. A [source audit](laya-data-source-audit.md) confirms that Laya-associated typed-decisions examples are also synthetic, not observed C3R outcomes. | Approved trace source, consent/license/data-boundary review, deduplication, immutable split, provenance audit, and empirical dataset publication. Never relabel synthetic data empirical. | | C3R Laya | Pinned upstream integration and calibration-aware abstention are tested. The Hugging Face C3R model page is an integration preview without trained weights. | Train a C3R-derived checkpoint on governed training data, fit calibration on held-out data, preserve sealed test set, and publish weights, raw predictions, manifests, hashes, and reproducible metrics. | @@ -49,6 +49,9 @@ canary evidence are independently reviewed. delivery, acknowledgement, rollback access, monitoring, and read-only and reversible canary thresholds and aborts still need verification. A named responder is not proof of operational coverage. +6. Wilfried has said he will name a second independent data reviewer; no reviewer + has been identified or signed off yet. Neither the owner nor this assistant + can substitute for that independent review of labels and public rows. ## Internal-task telemetry authorization From 7c0b01a3fc08bdddf6719eb3284bc41d48fd71e2 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 09:12:02 -0500 Subject: [PATCH 24/55] Record named independent C3R reviewer pending acceptance --- docs/internal-task-trace-policy.md | 12 ++++++++++-- docs/standalone-launch.md | 7 ++++--- 2 files changed, 14 insertions(+), 5 deletions(-) diff --git a/docs/internal-task-trace-policy.md b/docs/internal-task-trace-policy.md index 23dad7b..4c4d9e1 100644 --- a/docs/internal-task-trace-policy.md +++ b/docs/internal-task-trace-policy.md @@ -9,6 +9,12 @@ and rollback operator via his ColomboAI work identity monitoring, alert delivery, acknowledgement, or rollback access is configured. Those controls must be verified before hosted traffic. +**Named independent reviewer:** Swapnil Pawar, designated by Wilfried on +2026-09-23 for internal-task labels and any proposed public de-identified rows. +His acceptance, access route, conflict-of-interest check, and actual review or +sign-off have not been evidenced. Naming him does not activate collection or +authorize publication. + This policy resolves the owner and data-use choices for the *controlled internal task population only*. It does **not** authorize production traffic or certify the service, checkpoint, or dataset. MC-1 remains excluded from the standalone @@ -68,8 +74,10 @@ This written approval **does not turn collection on**. Before the first live tra the owner must record the exact task-source registry, permitted-field schema, redaction/leakage test results, IAM grants, encrypted storage location, daily 30-day deletion job and backup purge proof, access audit, independent ledger-head -anchor, on-call/rollback contact, and a private staging deployment. A reviewer -must verify these controls against an intentionally non-sensitive dry run. +anchor, on-call/rollback contact, and a private staging deployment. Swapnil +Pawar or another subsequently approved independent reviewer must accept the +assignment and verify these controls against an intentionally non-sensitive +dry run. The reference [`GovernedTraceStore`](../c3r/telemetry/governed_store.py) enforces an attested source/task allowlist, a bounded token/numeric trace schema, a local 30-day diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index 296a643..24dd457 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -49,9 +49,10 @@ canary evidence are independently reviewed. delivery, acknowledgement, rollback access, monitoring, and read-only and reversible canary thresholds and aborts still need verification. A named responder is not proof of operational coverage. -6. Wilfried has said he will name a second independent data reviewer; no reviewer - has been identified or signed off yet. Neither the owner nor this assistant - can substitute for that independent review of labels and public rows. +6. Wilfried named Swapnil Pawar as the independent data reviewer on 2026-09-23. + His acceptance, access route, conflict-of-interest check, and sign-off on + labels, controls, and any public rows remain outstanding. Neither the owner + nor this assistant can substitute for that independent review. ## Internal-task telemetry authorization From e89786cd02a685631370554d2fc69df650d8bd8f Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 09:40:26 -0500 Subject: [PATCH 25/55] Add independent release review handoff --- docs/independent-review-handoff.md | 46 ++++++++++++++++++++++++++++++ docs/standalone-launch.md | 4 ++- 2 files changed, 49 insertions(+), 1 deletion(-) create mode 100644 docs/independent-review-handoff.md diff --git a/docs/independent-review-handoff.md b/docs/independent-review-handoff.md new file mode 100644 index 0000000..7d49ebe --- /dev/null +++ b/docs/independent-review-handoff.md @@ -0,0 +1,46 @@ +# Independent C3R release review handoff + +**Status:** requested, not accepted or signed off. Wilfried Kouadio named Swapnil +Pawar as the independent reviewer on 2026-09-23 and sent a request from the +ColomboAI contact mailbox. This page provides a review scope, not a claim of +independence, completed review, or launch authorization. + +## What the reviewer should decide + +1. Confirm acceptance and disclose any conflict that would prevent an + independent review of task labels, data rights, privacy controls, or release + claims. The release owner cannot sign on the reviewer's behalf. +2. Before collection, verify the [internal-task trace policy](internal-task-trace-policy.md) + against an intentionally non-sensitive dry run: C3R-authored task provenance, + the exact source/task allowlist, permitted fields, redaction/leakage checks, + storage encryption and IAM, read/export audit, daily 30-day deletion including + backups, independent ledger-head anchoring, and tested kill switch. Inspect + deployed settings and execution evidence, not only source code or a plan. +3. For empirical release, review frozen train/calibration/test manifests, + independent outcome labels, leakage and contamination checks, raw predictions, + calibration fit, confidence intervals, paired baselines, safety failures, + and the proposed model/dataset card claims. Provider generations alone are + not ground truth. The five-case controlled replay and synthetic DecisionMix + preview are not substitutes for these artifacts. +4. Before any public row release, inspect an explicit publication manifest and + sample of every proposed row class for rights, re-identification, private + artifact references, and accidental personal or credential content. Approve + aggregates separately from row-level publication. + +## Current evidence, not yet sufficient for sign-off + +- [C3R PR #2](https://github.com/ColomboAI-com/c3r/pull/2) is a draft with the + tested standalone controller and fail-closed service boundary. +- [CI run 56](https://github.com/ColomboAI-com/c3r/actions/runs/35870568965) + built and smoke-imported the digest-pinned image and produced a Trivy report + with zero HIGH/CRITICAL findings. This does not cover lower severities or + deployment controls. +- [Launch issue #3](https://github.com/ColomboAI-com/c3r/issues/3) lists the + unclosed production gates. No C3R Cloud Run service or governed live trace + collection exists at this writing. + +The reviewer should record findings, evidence links and hashes, date, scope, +and a clear **approve / reject / needs changes** decision for each gate. A +qualified public launch additionally requires owner release approval and live +operational evidence. MC-1 is excluded from this standalone scope; it is not +complete under the original directive. diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index 24dd457..b0d20c7 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -52,7 +52,9 @@ canary evidence are independently reviewed. 6. Wilfried named Swapnil Pawar as the independent data reviewer on 2026-09-23. His acceptance, access route, conflict-of-interest check, and sign-off on labels, controls, and any public rows remain outstanding. Neither the owner - nor this assistant can substitute for that independent review. + nor this assistant can substitute for that independent review. The + [reviewer handoff](independent-review-handoff.md) defines the requested scope + and evidence without implying approval. ## Internal-task telemetry authorization From 2f002cb256afc271f2c18e9f17f627697fe5a00b Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 09:45:37 -0500 Subject: [PATCH 26/55] Fail closed when governed trace purge is overdue --- c3r/telemetry/governed_store.py | 13 ++++++++++++- docs/standalone-launch.md | 5 ++++- tests/test_governed_trace_store.py | 17 ++++++++++++++++- 3 files changed, 32 insertions(+), 3 deletions(-) diff --git a/c3r/telemetry/governed_store.py b/c3r/telemetry/governed_store.py index ba5560c..db6c983 100644 --- a/c3r/telemetry/governed_store.py +++ b/c3r/telemetry/governed_store.py @@ -191,7 +191,11 @@ def append(self, trace: DecisionTrace, *, source_id: str, task_id: str) -> Ledge if grant is None or task_id not in grant.task_ids: raise ValueError("unapproved source or task") _validate_trace(trace) - collected_at = self._now().isoformat(timespec="microseconds") + now = self._now() + collected_at = now.isoformat(timespec="microseconds") + retention_cutoff = (now - timedelta(days=RETENTION_DAYS)).isoformat( + timespec="microseconds" + ) payload = json.dumps( {"collected_at": collected_at, "source_id": source_id, "task_id": task_id, "trace": asdict(trace)}, @@ -205,6 +209,12 @@ def append(self, trace: DecisionTrace, *, source_id: str, task_id: str) -> Ledge ).fetchone() if collected_at < last_collected_at: raise ValueError("trace clock moved backwards") + overdue = self._db.execute( + "SELECT 1 FROM records WHERE collected_at <= ? LIMIT 1", + (retention_cutoff,), + ).fetchone() + if overdue is not None: + raise ValueError("retention purge overdue; new collection is disabled") row = self._db.execute( "SELECT record_hash FROM records ORDER BY sequence DESC LIMIT 1" ).fetchone() @@ -274,3 +284,4 @@ def __init__(self, store: GovernedTraceStore, *, source_id: str, task_id: str) - def append(self, trace: DecisionTrace) -> LedgerRecord: return self._store.append(trace, source_id=self._source_id, task_id=self._task_id) + diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index b0d20c7..eb268d4 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -18,7 +18,9 @@ The current code path is a **tested reference boundary**, not a production servi An opt-in governed SQLite trace store now validates internal-task source grants, rejects unbounded text and private artifact references, and has a tested 30-day local purge/checkpoint operation. It does not run a daily scheduler, delete backups, or -provide independent audit anchoring. The ordinary SQLite ledger is likewise +provide independent audit anchoring. New collection now fails closed if an +expired row remains because the purge was missed; this is a guardrail, not proof +that deletion ran. The ordinary SQLite ledger is likewise optional and does not by itself provide independent audit anchoring; the estimates are not yet empirically calibrated, and the HTTP server requires a separate TLS/authentication gateway. Keep @@ -71,3 +73,4 @@ controlled task population only; it cannot establish field performance. No public production claim should be made while any exit condition above is unmet. The [Cloud Run staging decision](cloud-run-staging.md) records the selected launch-capable URL strategy and the prerequisites that still block deployment. + diff --git a/tests/test_governed_trace_store.py b/tests/test_governed_trace_store.py index 7738399..c676ed6 100644 --- a/tests/test_governed_trace_store.py +++ b/tests/test_governed_trace_store.py @@ -86,10 +86,11 @@ def test_purges_after_30_days_and_preserves_remaining_chain(self): with self.store(now=NOW - timedelta(days=31)) as store: first = store.append(trace(), source_id="c3r_internal_001", task_id="task_001") with self.store(now=NOW - timedelta(days=1)) as store: + self.assertEqual(store.purge_expired(), 1) second = store.append(trace(run_id="run_002"), source_id="c3r_internal_001", task_id="task_001") with self.store(now=NOW) as store: - self.assertEqual(store.purge_expired(), 1) + self.assertEqual(store.purge_expired(), 0) remaining = store.records() self.assertEqual(len(remaining), 1) self.assertEqual(remaining[0].record_hash, second.record_hash) @@ -118,6 +119,20 @@ def test_clock_rollback_cannot_relabel_traces_after_purge(self): store.append(trace(run_id="run_002"), source_id="c3r_internal_001", task_id="task_001") + def test_overdue_purge_blocks_new_collection_until_purged(self): + with self.store(now=NOW - timedelta(days=31)) as store: + store.append(trace(), source_id="c3r_internal_001", task_id="task_001") + with self.store(now=NOW) as store: + with self.assertRaisesRegex(ValueError, "retention purge overdue"): + store.append(trace(run_id="run_002"), source_id="c3r_internal_001", + task_id="task_001") + self.assertEqual(len(store.records()), 1) + self.assertEqual(store.purge_expired(), 1) + store.append(trace(run_id="run_002"), source_id="c3r_internal_001", + task_id="task_001") + self.assertEqual(len(store.records()), 1) + if __name__ == "__main__": unittest.main() + From 89ebe130a98bea4a5e7d84c22da12fa24a0052ac Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 09:55:57 -0500 Subject: [PATCH 27/55] Record reviewer acceptance and add fixture-only trace dry run --- docs/independent-review-handoff.md | 24 ++-- docs/internal-task-trace-policy.md | 7 +- docs/standalone-launch.md | 6 +- evidence/trace-control-dry-run-v1/report.json | 27 ++++ scripts/run_trace_control_dry_run.py | 122 ++++++++++++++++++ tests/test_trace_control_dry_run.py | 17 +++ 6 files changed, 190 insertions(+), 13 deletions(-) create mode 100644 evidence/trace-control-dry-run-v1/report.json create mode 100644 scripts/run_trace_control_dry_run.py create mode 100644 tests/test_trace_control_dry_run.py diff --git a/docs/independent-review-handoff.md b/docs/independent-review-handoff.md index 7d49ebe..8497d84 100644 --- a/docs/independent-review-handoff.md +++ b/docs/independent-review-handoff.md @@ -1,15 +1,16 @@ # Independent C3R release review handoff -**Status:** requested, not accepted or signed off. Wilfried Kouadio named Swapnil -Pawar as the independent reviewer on 2026-09-23 and sent a request from the -ColomboAI contact mailbox. This page provides a review scope, not a claim of -independence, completed review, or launch authorization. +**Status:** accepted, not reviewed or signed off. Wilfried Kouadio named Swapnil +Pawar as the independent reviewer on 2026-09-23. Swapnil replied from his +ColomboAI work mailbox on 2026-09-23 accepting the role, reporting no conflict +of interest, and requesting access to a non-sensitive dry run. His reply also +acknowledged that collection and publication remain off. Acceptance does not +verify independence in practice, complete the review, or authorize launch. ## What the reviewer should decide -1. Confirm acceptance and disclose any conflict that would prevent an - independent review of task labels, data rights, privacy controls, or release - claims. The release owner cannot sign on the reviewer's behalf. +1. Record the accepted scope and revisit conflicts if the task or reporting + relationship changes. The release owner cannot sign on the reviewer's behalf. 2. Before collection, verify the [internal-task trace policy](internal-task-trace-policy.md) against an intentionally non-sensitive dry run: C3R-authored task provenance, the exact source/task allowlist, permitted fields, redaction/leakage checks, @@ -31,10 +32,16 @@ independence, completed review, or launch authorization. - [C3R PR #2](https://github.com/ColomboAI-com/c3r/pull/2) is a draft with the tested standalone controller and fail-closed service boundary. -- [CI run 56](https://github.com/ColomboAI-com/c3r/actions/runs/35870568965) +- [CI run 64](https://github.com/ColomboAI-com/c3r/actions/runs/35876574369) built and smoke-imported the digest-pinned image and produced a Trivy report with zero HIGH/CRITICAL findings. This does not cover lower severities or deployment controls. +- The [non-sensitive local trace-control dry run](../evidence/trace-control-dry-run-v1/report.json) + uses one C3R-authored synthetic fixture. Nine local admission, redaction, + retention-lockout, purge, and chain checks pass. It creates no live trace and + cannot verify encrypted deployment storage, backup deletion, access auditing, + independent anchoring, alerting, or task outcomes. Reproduce with + `python scripts/run_trace_control_dry_run.py`. - [Launch issue #3](https://github.com/ColomboAI-com/c3r/issues/3) lists the unclosed production gates. No C3R Cloud Run service or governed live trace collection exists at this writing. @@ -44,3 +51,4 @@ and a clear **approve / reject / needs changes** decision for each gate. A qualified public launch additionally requires owner release approval and live operational evidence. MC-1 is excluded from this standalone scope; it is not complete under the original directive. + diff --git a/docs/internal-task-trace-policy.md b/docs/internal-task-trace-policy.md index 4c4d9e1..83787fb 100644 --- a/docs/internal-task-trace-policy.md +++ b/docs/internal-task-trace-policy.md @@ -11,9 +11,9 @@ Those controls must be verified before hosted traffic. **Named independent reviewer:** Swapnil Pawar, designated by Wilfried on 2026-09-23 for internal-task labels and any proposed public de-identified rows. -His acceptance, access route, conflict-of-interest check, and actual review or -sign-off have not been evidenced. Naming him does not activate collection or -authorize publication. +He accepted by work email on 2026-09-23 and reported no conflict of interest. +Access to a non-sensitive dry run, independent control review, and actual sign-off +remain outstanding. Acceptance does not activate collection or authorize publication. This policy resolves the owner and data-use choices for the *controlled internal task population only*. It does **not** authorize production traffic or certify @@ -95,3 +95,4 @@ one binding for unrelated requests or accept those identifiers from callers. No public endpoint, public dataset promotion, model-weight release, or broad access is approved by this policy alone. Each requires its own evidence-matched release decision. This document is a project governance record, not a legal opinion. + diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index eb268d4..17674ed 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -52,8 +52,10 @@ canary evidence are independently reviewed. reversible canary thresholds and aborts still need verification. A named responder is not proof of operational coverage. 6. Wilfried named Swapnil Pawar as the independent data reviewer on 2026-09-23. - His acceptance, access route, conflict-of-interest check, and sign-off on - labels, controls, and any public rows remain outstanding. Neither the owner + Swapnil accepted by work email that day and reported no conflict. A + [non-sensitive local fixture dry run](../evidence/trace-control-dry-run-v1/report.json) + is available, but access delivery, independent review, and sign-off on + labels, deployed controls, and any public rows remain outstanding. Neither the owner nor this assistant can substitute for that independent review. The [reviewer handoff](independent-review-handoff.md) defines the requested scope and evidence without implying approval. diff --git a/evidence/trace-control-dry-run-v1/report.json b/evidence/trace-control-dry-run-v1/report.json new file mode 100644 index 0000000..170666c --- /dev/null +++ b/evidence/trace-control-dry-run-v1/report.json @@ -0,0 +1,27 @@ +{ + "all_local_checks_passed": true, + "checks": { + "approved_fixture_admitted": true, + "artifact_reference_rejected": true, + "checkpoint_chain_verifies": true, + "free_text_rejected": true, + "local_expired_row_purged": true, + "overdue_purge_blocks_collection": true, + "post_purge_chain_continues": true, + "rejected_rows_not_persisted": true, + "unapproved_source_rejected": true + }, + "evidence_kind": "non_sensitive_local_fixture_dry_run", + "live_trace_collection_enabled": false, + "not_verified_by_this_run": [ + "deployed encryption and IAM", + "read/export access audit", + "daily scheduler and deletion of backups or replicas", + "independent ledger-head anchor", + "alert delivery and rollback", + "live internal-task provenance or outcomes" + ], + "source": "c3r_fixture_review", + "task_population": "one C3R-authored synthetic fixture; no customer or product records" +} + diff --git a/scripts/run_trace_control_dry_run.py b/scripts/run_trace_control_dry_run.py new file mode 100644 index 0000000..28000de --- /dev/null +++ b/scripts/run_trace_control_dry_run.py @@ -0,0 +1,122 @@ +"""Reproducible local fixture check for independent trace-control review. + +This never enables live collection and never claims deployed storage controls. +""" + +from __future__ import annotations + +import json +import sys +from collections.abc import Callable +from datetime import datetime, timedelta, timezone +from pathlib import Path +from tempfile import TemporaryDirectory + +REPOSITORY_ROOT = Path(__file__).resolve().parents[1] +if str(REPOSITORY_ROOT) not in sys.path: + sys.path.insert(0, str(REPOSITORY_ROOT)) + +from c3r.telemetry.governed_store import GovernedTraceStore, SourceGrant +from c3r.telemetry.trace import DecisionTrace + + +START = datetime(2026, 9, 23, 0, 0, tzinfo=timezone.utc) +SOURCE_ID = "c3r_fixture_review" +TASK_ID = "fixture_task_001" + + +def _trace(run_id: str, **changes: object) -> DecisionTrace: + fields: dict[str, object] = { + "run_id": run_id, + "state_hash": "a" * 64, + "access_level": "internal", + "model_provider": "fixture_only", + "candidate_ids": ("recommend",), + "probabilities": {"route": (0.8, 0.2)}, + "utility_quantiles": {"selected_lower_bound": 0.3}, + "selected_action_id": "recommend", + "authority_result": "verified", + "system_cost": {"latency_ms": 1.0}, + "task_outcome": {"status": "fixture_only"}, + "artifact_refs": (), + } + fields.update(changes) + return DecisionTrace(**fields) + + +def _rejects(callback: Callable[[], object], expected_reason: str) -> bool: + try: + callback() + except ValueError as error: + return expected_reason in str(error) + return False + + +def run() -> dict[str, object]: + grant = SourceGrant( + source_id=SOURCE_ID, owner="c3r_review_fixture", + task_ids=frozenset({TASK_ID}), rights_attested=True, + ) + clock = [START] + with TemporaryDirectory(prefix="c3r-trace-review-") as directory: + with GovernedTraceStore( + Path(directory) / "trace.sqlite3", grants=(grant,), clock=lambda: clock[0], + ) as store: + first = store.append(_trace("fixture_run_001"), source_id=SOURCE_ID, task_id=TASK_ID) + checks = { + "approved_fixture_admitted": len(store.records()) == 1, + "unapproved_source_rejected": _rejects(lambda: store.append( + _trace("fixture_run_002"), source_id="customer_logs", task_id=TASK_ID, + ), "unapproved source or task"), + "free_text_rejected": _rejects(lambda: store.append( + _trace("fixture_run_003", task_outcome={"status": "email me at a@example.com"}), + source_id=SOURCE_ID, task_id=TASK_ID, + ), "redaction"), + "artifact_reference_rejected": _rejects(lambda: store.append( + _trace("fixture_run_004", artifact_refs=("private_artifact",)), + source_id=SOURCE_ID, task_id=TASK_ID, + ), "redaction"), + } + checks["rejected_rows_not_persisted"] = len(store.records()) == 1 + clock[0] = START + timedelta(days=31) + checks["overdue_purge_blocks_collection"] = _rejects(lambda: store.append( + _trace("fixture_run_005"), source_id=SOURCE_ID, task_id=TASK_ID, + ), "retention purge overdue") + checks["local_expired_row_purged"] = store.purge_expired() == 1 + checks["checkpoint_chain_verifies"] = store.verify() and not store.records() + second = store.append( + _trace("fixture_run_006"), source_id=SOURCE_ID, task_id=TASK_ID, + ) + checks["post_purge_chain_continues"] = ( + second.previous_hash == first.record_hash and store.verify() + ) + return { + "evidence_kind": "non_sensitive_local_fixture_dry_run", + "live_trace_collection_enabled": False, + "source": SOURCE_ID, + "task_population": "one C3R-authored synthetic fixture; no customer or product records", + "checks": checks, + "all_local_checks_passed": all(checks.values()), + "not_verified_by_this_run": [ + "deployed encryption and IAM", + "read/export access audit", + "daily scheduler and deletion of backups or replicas", + "independent ledger-head anchor", + "alert delivery and rollback", + "live internal-task provenance or outcomes", + ], + } + + +def main() -> None: + report = run() + output = REPOSITORY_ROOT / "evidence" / "trace-control-dry-run-v1" / "report.json" + output.parent.mkdir(parents=True, exist_ok=True) + output.write_text(json.dumps(report, indent=2, sort_keys=True) + "\n", encoding="utf-8") + if not report["all_local_checks_passed"]: + raise SystemExit("local trace-control dry run failed") + + +if __name__ == "__main__": + main() + diff --git a/tests/test_trace_control_dry_run.py b/tests/test_trace_control_dry_run.py new file mode 100644 index 0000000..5a686d3 --- /dev/null +++ b/tests/test_trace_control_dry_run.py @@ -0,0 +1,17 @@ +import unittest + +from scripts.run_trace_control_dry_run import run + + +class TraceControlDryRunTests(unittest.TestCase): + def test_fixture_only_report_passes_without_claiming_deployed_controls(self): + report = run() + self.assertTrue(report["all_local_checks_passed"]) + self.assertFalse(report["live_trace_collection_enabled"]) + self.assertEqual(len(report["checks"]), 9) + self.assertIn("deployed encryption and IAM", report["not_verified_by_this_run"]) + + +if __name__ == "__main__": + unittest.main() + From 403c2443833fb854ad1d041f03f22c7811f52f0e Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 10:04:07 -0500 Subject: [PATCH 28/55] Restrict C3R Cloud Build source upload --- .gcloudignore | 10 ++++++++++ 1 file changed, 10 insertions(+) create mode 100644 .gcloudignore diff --git a/.gcloudignore b/.gcloudignore new file mode 100644 index 0000000..314a848 --- /dev/null +++ b/.gcloudignore @@ -0,0 +1,10 @@ +# C3R staging build uploads runtime source only. Do not send papers, traces, +# local evidence, secrets, or repository metadata to Cloud Build. +** +!Dockerfile +!.dockerignore +!c3r/ +!c3r/** +**/__pycache__/ +**/*.pyc + From 654f9e4304e317e1c0869931c0a12f46e6fde7b8 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 10:07:46 -0500 Subject: [PATCH 29/55] Add fixed-disabled Cloud Run staging host --- c3r/staging_host.py | 67 ++++++++++++++++++++++++++++++++++++++ docs/cloud-run-staging.md | 24 +++++++++++++- tests/test_staging_host.py | 36 ++++++++++++++++++++ 3 files changed, 126 insertions(+), 1 deletion(-) create mode 100644 c3r/staging_host.py create mode 100644 tests/test_staging_host.py diff --git a/c3r/staging_host.py b/c3r/staging_host.py new file mode 100644 index 0000000..db8ce87 --- /dev/null +++ b/c3r/staging_host.py @@ -0,0 +1,67 @@ +"""Private Cloud Run boundary-smoke host; never enables C3R decisions or collection. + +This is deliberately not a production host. It proves packaging and ingress only. +No provider, external effect, or durable trace sink is configured here. +""" + +from __future__ import annotations + +from secrets import token_bytes + +from .candidate_compiler import CandidateCompiler +from .commit_gateway import InMemoryApprovalNonceStore, TrustedCommitGateway +from .cvoc import RobustCvocController +from .feature_flags import FeatureFlags +from .host_factory import ReadOnlyRequestFactory +from .runtime import StandaloneController +from .state_compiler import StateCompiler +from .state_schema import ActionDefinition, ActionFamily, AuthorityPolicy, RiskClass +from .telemetry.trace import DecisionTrace +from .telemetry.trace_ledger import LedgerRecord, _record_hash, canonical_trace_json +from .verifier_firewall import VerifierDecision, VerifierFirewall, VerifierPolicy + + +class EphemeralStagingSink: + """Return a per-request integrity hash without retaining any trace rows.""" + + def append(self, trace: DecisionTrace) -> LedgerRecord: + canonical = canonical_trace_json(trace) + genesis = "0" * 64 + return LedgerRecord(genesis, _record_hash(genesis, canonical), canonical) + + +def build() -> tuple[StandaloneController, ReadOnlyRequestFactory]: + """Construct a fixed-disabled, recommendation-only staging boundary.""" + key = token_bytes(32) + verifier = VerifierFirewall( + {"deny": lambda _candidate: VerifierDecision(False, "staging disabled")}, + VerifierPolicy(default_verifier="deny"), attestation_key=key, + ) + gateway = TrustedCommitGateway( + trusted_verifier_ids=frozenset({"deny"}), verification_key=key, + approval_key=token_bytes(32), policy_version="staging-disabled-v1", + approval_nonce_store=InMemoryApprovalNonceStore(), + ) + controller = StandaloneController( + flags=FeatureFlags(enabled_requested=False), + compiler=StateCompiler(), candidates=CandidateCompiler(), + cvoc=RobustCvocController(), verifier=verifier, gateway=gateway, + ledger=EphemeralStagingSink(), + ) + definition = ActionDefinition( + id="staging_noop", family=ActionFamily.TOOL, subgroup="staging", + operation="noop", risk_class=RiskClass.READ_ONLY, + argument_variants=((),), placements=("local",), verifier_ids=("deny",), + optimistic_utility=0.0, estimated_cost=0.0, + ) + factory = ReadOnlyRequestFactory( + definitions=(definition,), + policy=AuthorityPolicy( + frozenset({ActionFamily.TOOL}), frozenset({RiskClass.READ_ONLY}), + ), + estimate_source=lambda _state: {}, + remaining_usd=0.0, + data_boundary="local", + ) + return controller, factory + diff --git a/docs/cloud-run-staging.md b/docs/cloud-run-staging.md index ced3b25..2e7bb2b 100644 --- a/docs/cloud-run-staging.md +++ b/docs/cloud-run-staging.md @@ -1,4 +1,4 @@ -# Cloud Run staging decision (not a deployment record) +# Cloud Run staging decision and resource record (not a service deployment) C3R's selected standalone hostname strategy is the Google-managed HTTPS URL that Cloud Run assigns to a service. A custom domain is optional. **No C3R Cloud Run @@ -6,6 +6,27 @@ service has been deployed**, so this document does not claim a URL, live model r or completed canary. The first deployment must require IAM authentication; making it public is a separate release action after qualification. +On 2026-09-23, the `columboai-frontend` project received a dedicated +`c3r-staging-runtime` service account with no user-managed keys or project roles, +and a private, immutable-tag Docker repository at +`us-central1-docker.pkg.dev/columboai-frontend/c3r-staging`. Its repository IAM +policy has no public binding; Google-managed encryption is reported. Cloud Build +`b25dcb33-5553-4c6d-bc88-1541bfb299bc` built the initial runtime-only source +and pushed digest `sha256:14b88f22a839fc123d639e3a96c29f3b2242ee443174fedd637665d64a872a2f`. +That first image predates the disabled staging host and has **not** been deployed. +The `.gcloudignore` upload manifest was checked to include only the Dockerfile, +`.dockerignore`, and `c3r/` Python source; no data, evidence, papers, or Git metadata. +Registry creation and image publication do not verify service behavior, storage, +retention, alerting, or release readiness. + +`c3r.staging_host:build` is an intentionally fixed-disabled, recommendation-only +boundary smoke host. It cannot be turned into a production decision service by +environment flags, performs no external effects or provider calls, and retains no +trace rows. An authenticated private deployment can use it to exercise ingress, +IAM/TLS, startup, monitoring, and rollback, but **not** to collect empirical data +or qualify C3R decisions. A separately reviewed host with measured pre-decision +estimates, durable governed storage, and approved source registry is required later. + ## Preconditions before creating a service 1. Wilfried Kouadio (`@wilkont`) is the interim deployment/release owner and @@ -68,3 +89,4 @@ Cloud Run references: [HTTPS service URL and invoking services](https://docs.clo [IAM service authentication](https://docs.cloud.google.com/run/docs/authenticating/overview), [public versus authenticated deployment](https://docs.cloud.google.com/run/docs/deploying), and [container listening contract](https://docs.cloud.google.com/run/docs/container-contract). + diff --git a/tests/test_staging_host.py b/tests/test_staging_host.py new file mode 100644 index 0000000..19936d0 --- /dev/null +++ b/tests/test_staging_host.py @@ -0,0 +1,36 @@ +import unittest + +from c3r.serve import load_host_builder +from c3r.staging_host import EphemeralStagingSink +from c3r.telemetry.trace import DecisionTrace + + +class StagingHostTests(unittest.TestCase): + def test_staging_builder_cannot_enable_decisions_or_effects(self): + controller, factory = load_host_builder("c3r.staging_host:build")() + self.assertFalse(controller.effect_execution_enabled) + request = factory.build({"goal": "fixture", "current_subgoal": "check"}) + self.assertEqual(request.estimates, {}) + outcome = controller.run(request) + self.assertEqual(outcome.reason, "C3R_DISABLED") + self.assertIsNone(outcome.selected_action_id) + + def test_staging_sink_has_no_row_store_or_cross_request_chain(self): + trace = DecisionTrace( + run_id="fixture_run", state_hash="a" * 64, access_level="internal", + model_provider="fixture", candidate_ids=(), probabilities={}, + utility_quantiles={}, selected_action_id=None, + authority_result="not_attempted", system_cost={}, + task_outcome={"status": "C3R_DISABLED"}, artifact_refs=(), + ) + sink = EphemeralStagingSink() + first = sink.append(trace) + second = sink.append(trace) + self.assertEqual(first.record_hash, second.record_hash) + self.assertEqual(first.previous_hash, "0" * 64) + self.assertFalse(hasattr(sink, "records")) + + +if __name__ == "__main__": + unittest.main() + From 4c8a88b1a9593adaf457fe8dae134608cd759720 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 10:18:55 -0500 Subject: [PATCH 30/55] Record private disabled staging deployment and rollback --- deploy/monitoring/staging-5xx.json | 34 ++++++++++++++ docs/cloud-run-staging.md | 54 ++++++++++++++++------ docs/independent-review-handoff.md | 14 ++++-- docs/internal-task-trace-policy.md | 12 +++-- docs/standalone-launch.md | 20 ++++---- evidence/staging-deployment-v1/report.json | 46 ++++++++++++++++++ 6 files changed, 149 insertions(+), 31 deletions(-) create mode 100644 deploy/monitoring/staging-5xx.json create mode 100644 evidence/staging-deployment-v1/report.json diff --git a/deploy/monitoring/staging-5xx.json b/deploy/monitoring/staging-5xx.json new file mode 100644 index 0000000..b5595d1 --- /dev/null +++ b/deploy/monitoring/staging-5xx.json @@ -0,0 +1,34 @@ +{ + "displayName": "C3R staging Cloud Run 5xx", + "combiner": "OR", + "enabled": true, + "notificationChannels": [ + "projects/columboai-frontend/notificationChannels/10331198960910692730" + ], + "documentation": { + "content": "C3R private staging 5xx. Wilfried: inspect c3r-staging revisions/logs, keep trace collection off, and route 100% to the last known-good fixed-disabled revision or disable the service. This alert does not certify launch readiness.", + "mimeType": "text/markdown" + }, + "conditions": [ + { + "displayName": "c3r-staging 5xx rate", + "conditionThreshold": { + "filter": "metric.type=\"run.googleapis.com/request_count\" AND resource.type=\"cloud_run_revision\" AND resource.labels.service_name=\"c3r-staging\" AND metric.labels.response_code_class=\"5xx\"", + "aggregations": [ + { + "alignmentPeriod": "60s", + "perSeriesAligner": "ALIGN_RATE", + "crossSeriesReducer": "REDUCE_SUM" + } + ], + "comparison": "COMPARISON_GT", + "thresholdValue": 0, + "duration": "0s", + "trigger": { + "count": 1 + } + } + } + ] +} + diff --git a/docs/cloud-run-staging.md b/docs/cloud-run-staging.md index 2e7bb2b..abbc843 100644 --- a/docs/cloud-run-staging.md +++ b/docs/cloud-run-staging.md @@ -1,10 +1,12 @@ -# Cloud Run staging decision and resource record (not a service deployment) +# Cloud Run private staging deployment (fixed-disabled) C3R's selected standalone hostname strategy is the Google-managed HTTPS URL that -Cloud Run assigns to a service. A custom domain is optional. **No C3R Cloud Run -service has been deployed**, so this document does not claim a URL, live model route, -or completed canary. The first deployment must require IAM authentication; making -it public is a separate release action after qualification. +Cloud Run assigns to a service. A custom domain is optional. A **private, +fixed-disabled staging boundary** is deployed at +`https://c3r-staging-795563500003.us-central1.run.app`. It is not a live model +route, governed trace collector, canary, or production service. Cloud Run IAM +authentication is required; public access is a separate release action after +qualification. On 2026-09-23, the `columboai-frontend` project received a dedicated `c3r-staging-runtime` service account with no user-managed keys or project roles, @@ -19,6 +21,19 @@ The `.gcloudignore` upload manifest was checked to include only the Dockerfile, Registry creation and image publication do not verify service behavior, storage, retention, alerting, or release readiness. +Cloud Build `d70f1263-a60e-489c-958f-36436ab2875a` published the fixed-disabled +staging host at digest +`sha256:60d4daf25d879c41892a3b1b5fc84638d7289ca75a191059858dd183f1aa1209`. +Cloud Run service `c3r-staging` in `us-central1` runs this digest under the +dedicated service account, with one maximum instance, zero minimum instances, +and no explicit public invoker binding. Its two distinct tokens are pinned +Secret Manager references; values are not in Git or this record. Direct +unauthenticated `/health` returned 403, IAM-authenticated `/health` returned +200, an IAM-only decision request returned 401, and a request with both IAM +and C3R token returned `C3R_DISABLED`, no selected action, and no effect. +The [deployment evidence](../evidence/staging-deployment-v1/report.json) records +these checks without credentials or request bodies. + `c3r.staging_host:build` is an intentionally fixed-disabled, recommendation-only boundary smoke host. It cannot be turned into a production decision service by environment flags, performs no external effects or provider calls, and retains no @@ -27,12 +42,14 @@ IAM/TLS, startup, monitoring, and rollback, but **not** to collect empirical dat or qualify C3R decisions. A separately reviewed host with measured pre-decision estimates, durable governed storage, and approved source registry is required later. -## Preconditions before creating a service +## Controls still required before live collection or promotion 1. Wilfried Kouadio (`@wilkont`) is the interim deployment/release owner and confirmed interim on-call/rollback operator in the [internal-task policy](internal-task-trace-policy.md). Verify the work-email - alert route, acknowledgement and rollback access before hosting traffic. + alert route and acknowledgement before live internal-task traffic. A C3R-only + Cloud Monitoring email channel and 5xx policy now exist, but delivery and + human acknowledgement are unverified. 2. Implement the approved internal-task trace policy: source registry, field allowlist, redaction tests, 30-day private deletion including backups, access audit, and publication review. The approved scope is only redacted telemetry @@ -51,9 +68,8 @@ estimates, durable governed storage, and approved source registry is required la size, request rate, concurrency, and upstream wait time. Local tests cover these boundaries, backend failure, and local server composition. GitHub CI has built and HIGH/CRITICAL-scanned the digest-pinned staging image with zero - findings in [run 56](https://github.com/ColomboAI-com/c3r/actions/runs/35870568965), - but no image has been published to a registry or deployed to Cloud Run. - Docker Desktop was unavailable during the local check. + findings in [run 70](https://github.com/ColomboAI-com/c3r/actions/runs/35879315503). + This scan does not qualify lower severities or deployed controls. Bind this ingress only behind Cloud Run's IAM/TLS boundary at staging, with tokens from Secret Manager. Do not publish it as a raw unauthenticated port. 6. Establish a private, authenticated service-to-service route to the GPU model; @@ -66,10 +82,20 @@ estimate source uses measured, pre-decision values. The container supplies **no* sample catalog, constant estimates, data collection, or production credentials. `C3R_CLIENT_TOKEN` and `C3R_BACKEND_TOKEN` are distinct mandatory secrets of at least 32 characters; `PORT` defaults to 8080 and `C3R_BACKEND_PORT` to 8081. -The trusted host module must be included in a derived private image or approved -runtime package. This is packaging, not proof that a calibrated host or secure -storage exists. Registry publication, durable ledger storage, secret rotation, and -end-to-end Cloud Run tests remain necessary before staging traffic. +The currently deployed host is `c3r.staging_host:build`, deliberately +fixed-disabled. A later trusted decision host must be separately reviewed and +deployed. Registry publication and boundary checks are complete only for the +disabled host; durable ledger storage, secret rotation, and live decision tests +remain necessary before governed internal-task traffic. + +On 2026-09-23 a second, equivalent disabled revision (`c3r-staging-drill1`) +was deployed, then traffic was explicitly restored 100% to the original +`c3r-staging-00001-zrj` revision. IAM-authenticated `/health` returned 200 +after rollback. This proves revision traffic rollback for the disabled staging +host, not incident response timing or a production rollback. Cloud Monitoring +policy `11003571770572095050` watches this service's 5xx request count and +routes to Wilfried's work-email channel, but delivery and acknowledgement have +not been exercised. ## Staged promotion diff --git a/docs/independent-review-handoff.md b/docs/independent-review-handoff.md index 8497d84..b0d924f 100644 --- a/docs/independent-review-handoff.md +++ b/docs/independent-review-handoff.md @@ -32,7 +32,7 @@ verify independence in practice, complete the review, or authorize launch. - [C3R PR #2](https://github.com/ColomboAI-com/c3r/pull/2) is a draft with the tested standalone controller and fail-closed service boundary. -- [CI run 64](https://github.com/ColomboAI-com/c3r/actions/runs/35876574369) +- [CI run 70](https://github.com/ColomboAI-com/c3r/actions/runs/35879315503) built and smoke-imported the digest-pinned image and produced a Trivy report with zero HIGH/CRITICAL findings. This does not cover lower severities or deployment controls. @@ -42,9 +42,17 @@ verify independence in practice, complete the review, or authorize launch. cannot verify encrypted deployment storage, backup deletion, access auditing, independent anchoring, alerting, or task outcomes. Reproduce with `python scripts/run_trace_control_dry_run.py`. +- [Private staging evidence](../evidence/staging-deployment-v1/report.json) + shows a fixed-disabled Cloud Run boundary, IAM and token checks, a service- + specific 5xx alert rule, and revision traffic rollback. This is not a live + C3R decision route or evidence of governed trace storage, alert delivery, + backup deletion, independent anchoring, or canaries. - [Launch issue #3](https://github.com/ColomboAI-com/c3r/issues/3) lists the - unclosed production gates. No C3R Cloud Run service or governed live trace - collection exists at this writing. + unclosed production gates. Governed live trace collection remains off. + +The public, fixture-only dry-run and handoff links were sent to Swapnil from +`contact@colomboai.com` on 2026-09-23. No trace rows, credentials, or private +artifacts were sent. His review response and gate-by-gate sign-off are pending. The reviewer should record findings, evidence links and hashes, date, scope, and a clear **approve / reject / needs changes** decision for each gate. A diff --git a/docs/internal-task-trace-policy.md b/docs/internal-task-trace-policy.md index 83787fb..0ca2366 100644 --- a/docs/internal-task-trace-policy.md +++ b/docs/internal-task-trace-policy.md @@ -12,7 +12,8 @@ Those controls must be verified before hosted traffic. **Named independent reviewer:** Swapnil Pawar, designated by Wilfried on 2026-09-23 for internal-task labels and any proposed public de-identified rows. He accepted by work email on 2026-09-23 and reported no conflict of interest. -Access to a non-sensitive dry run, independent control review, and actual sign-off +The non-sensitive fixture-only dry-run link was sent to him from ColomboAI's +work mailbox on 2026-09-23. Independent control review and actual sign-off remain outstanding. Acceptance does not activate collection or authorize publication. This policy resolves the owner and data-use choices for the *controlled internal @@ -74,10 +75,11 @@ This written approval **does not turn collection on**. Before the first live tra the owner must record the exact task-source registry, permitted-field schema, redaction/leakage test results, IAM grants, encrypted storage location, daily 30-day deletion job and backup purge proof, access audit, independent ledger-head -anchor, on-call/rollback contact, and a private staging deployment. Swapnil -Pawar or another subsequently approved independent reviewer must accept the -assignment and verify these controls against an intentionally non-sensitive -dry run. +anchor, on-call/rollback contact, and a private staging deployment. A +fixed-disabled private Cloud Run staging host now exists, but it has no durable +trace storage or decision authority. Swapnil Pawar or another subsequently +approved independent reviewer must verify the complete controls against an +intentionally non-sensitive dry run. The reference [`GovernedTraceStore`](../c3r/telemetry/governed_store.py) enforces an attested source/task allowlist, a bounded token/numeric trace schema, a local 30-day diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index 17674ed..7baa09a 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -6,7 +6,7 @@ distinguishes tested code, private operational evidence, and public release evid | Gate | Current evidence | Exit condition | | --- | --- | --- | -| Hosted decision path | `StandaloneController` composes state compilation, inventory-masked hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. Provider latency/token usage is recorded without granting authority. A host-owned read-only request factory ignores caller policy, budgets, estimates, and approvals. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. A separate tested ingress allowlists read-only routes, uses distinct client/backend tokens, and fails closed on backend outage. A fail-closed container entrypoint composes both servers only when a trusted host builder and distinct secrets are supplied. GitHub CI builds, smoke-imports, and [HIGH/CRITICAL-scans the digest-pinned staging image](image-security.md) with zero findings in run 56. A transactional SQLite trace sink survives restart and verifies its hash chain. Local integration tests pass. | Calibrated *pre-decision* value/cost estimate source, trusted host composition, registry publication of the scanned image, durable storage placement, independent ledger-head anchoring, TLS/IAM gateway, Google-managed HTTPS service URL, secret rotation, deployment and rollback rehearsal, monitoring/alerts, and live end-to-end traces. No C3R cloud service exists yet. | +| Hosted decision path | `StandaloneController` composes state compilation, inventory-masked hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. Provider latency/token usage is recorded without granting authority. A host-owned read-only request factory ignores caller policy, budgets, estimates, and approvals. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. A separate tested ingress allowlists read-only routes, uses distinct client/backend tokens, and fails closed on backend outage. A fail-closed container entrypoint composes both servers only when a trusted host builder and distinct secrets are supplied. GitHub CI builds, smoke-imports, and [HIGH/CRITICAL-scans the digest-pinned staging image](image-security.md) with zero findings in run 70. A private [fixed-disabled Cloud Run staging host](cloud-run-staging.md) now verifies IAM/TLS ingress, two-token gating, a disabled decision response, a service-specific 5xx alert rule, and revision rollback. It has no decision authority or retained trace rows. | Calibrated *pre-decision* value/cost estimate source, reviewed live decision host, governed durable storage with 30-day deletion including backups, read/export audit and independent ledger-head anchoring, secret rotation, verified alert delivery/acknowledgement, and live end-to-end internal-task traces. The disabled staging host is not a production C3R service. | | DeepSeek | Official DeepSeek-V4.1-Flash shards are verified on the GCP GPU host and in a private GCS backup; localhost inference passed a smoke test. A provider bridge passes compiled state to a local OpenAI-compatible adapter and rejects remote transfer of local-only data in tests. | Secure service-to-service route and paired live C3R decision traces. The localhost model is not a public API. | | Empirical DecisionMix | Published preview has 144 synthetic records, deterministic splits, and hashes. A [source audit](laya-data-source-audit.md) confirms that Laya-associated typed-decisions examples are also synthetic, not observed C3R outcomes. | Approved trace source, consent/license/data-boundary review, deduplication, immutable split, provenance audit, and empirical dataset publication. Never relabel synthetic data empirical. | | C3R Laya | Pinned upstream integration and calibration-aware abstention are tested. The Hugging Face C3R model page is an integration preview without trained weights. | Train a C3R-derived checkpoint on governed training data, fit calibration on held-out data, preserve sealed test set, and publish weights, raw predictions, manifests, hashes, and reproducible metrics. | @@ -14,7 +14,8 @@ distinguishes tested code, private operational evidence, and public release evid | Providers and Colibri | Provider protocol contracts and a non-authoritative Colibri shadow adapter exist. Dedicated `c3r-evals` and `c3r-colibri` repositories exist. Single fixed-prompt credentialed OpenRouter smoke probes now cover Qwen3.8 Flash and Claude Sonnet 4.6 in [c3r-evals PR #1](https://github.com/ColomboAI-com/c3r-evals/pull/1). | Multi-case credentialed provider qualification, paired C3R traces, compatible Colibri instrumentation, actual shadow route traces, offline replay, shadow traffic, read-only and reversible canaries. | | Public release | v0.1 alpha repository, papers, collection, model/dataset previews. | Update all cards and launch copy only after the corresponding evidence passes; perform security and dependency review; publish an evidence-matched standalone release. | -The current code path is a **tested reference boundary**, not a production service. +The current code path is a **tested reference boundary** with a private +fixed-disabled staging deployment, not a production service. An opt-in governed SQLite trace store now validates internal-task source grants, rejects unbounded text and private artifact references, and has a tested 30-day local purge/checkpoint operation. It does not run a daily scheduler, delete backups, or @@ -30,9 +31,9 @@ canary evidence are independently reviewed. ## Required operator inputs 1. The selected hostname strategy is a Google-managed Cloud Run HTTPS `run.app` URL. - It is available only after deployment; private staging must require Cloud Run IAM, - and public access is a separate, explicitly approved promotion. A custom domain is - optional, not a launch prerequisite. + The private fixed-disabled [staging service](cloud-run-staging.md) now has one; + it requires Cloud Run IAM, and public access is a separate, explicitly approved + promotion. A custom domain is optional, not a launch prerequisite. 2. A governed trace source with explicit data-use, retention, redaction, and publication permissions. The [interim internal-task policy](internal-task-trace-policy.md) records these choices, but its technical activation controls are not yet verified. @@ -52,9 +53,9 @@ canary evidence are independently reviewed. reversible canary thresholds and aborts still need verification. A named responder is not proof of operational coverage. 6. Wilfried named Swapnil Pawar as the independent data reviewer on 2026-09-23. - Swapnil accepted by work email that day and reported no conflict. A + Swapnil accepted by work email that day and reported no conflict. The [non-sensitive local fixture dry run](../evidence/trace-control-dry-run-v1/report.json) - is available, but access delivery, independent review, and sign-off on + was sent to him by work email, but independent review and sign-off on labels, deployed controls, and any public rows remain outstanding. Neither the owner nor this assistant can substitute for that independent review. The [reviewer handoff](independent-review-handoff.md) defines the requested scope @@ -73,6 +74,7 @@ must not silently widen the data source. Internal-task evidence can qualify the controlled task population only; it cannot establish field performance. No public production claim should be made while any exit condition above is unmet. -The [Cloud Run staging decision](cloud-run-staging.md) records the selected -launch-capable URL strategy and the prerequisites that still block deployment. +The [Cloud Run staging record](cloud-run-staging.md) records the private +fixed-disabled deployment and the controls still blocking live collection and +production promotion. diff --git a/evidence/staging-deployment-v1/report.json b/evidence/staging-deployment-v1/report.json new file mode 100644 index 0000000..043db5d --- /dev/null +++ b/evidence/staging-deployment-v1/report.json @@ -0,0 +1,46 @@ +{ + "report_version": 1, + "observed_utc_date": "2026-09-23", + "scope": "private fixed-disabled boundary only; no governed trace collection or production decision path", + "project": "columboai-frontend", + "region": "us-central1", + "service": "c3r-staging", + "url": "https://c3r-staging-795563500003.us-central1.run.app", + "image_digest": "sha256:60d4daf25d879c41892a3b1b5fc84638d7289ca75a191059858dd183f1aa1209", + "cloud_build_id": "d70f1263-a60e-489c-958f-36436ab2875a", + "runtime_service_account": "c3r-staging-runtime@columboai-frontend.iam.gserviceaccount.com", + "deployed_host": "c3r.staging_host:build", + "collection_enabled": false, + "checks": { + "ready": true, + "public_invoker_binding_present": false, + "unauthenticated_health_status": 403, + "iam_authenticated_health_status": 200, + "iam_only_decision_status": 401, + "iam_and_c3r_token_decision_status": 200, + "iam_and_c3r_token_decision_reason": "C3R_DISABLED", + "selected_action_id": null, + "effect_attempted": false, + "secret_values_recorded": false, + "original_revision": "c3r-staging-00001-zrj", + "rollback_drill_revision": "c3r-staging-drill1", + "rollback_target_traffic_percent": 100, + "post_rollback_iam_health_status": 200, + "staging_5xx_policy_id": "11003571770572095050", + "on_call_email_channel_created": true, + "alert_delivery_verified": false, + "human_acknowledgement_verified": false + }, + "not_verified": [ + "functional C3R decision routing or live provider calls", + "approved internal-task source and actual governed runs", + "durable encrypted trace storage and read/export access auditing", + "daily 30-day deletion including backups", + "independent ledger-head anchoring", + "alert delivery, acknowledgement, or incident response timing", + "trained C3R Laya weights, held-out calibration, and empirical DecisionMix", + "independent safety review and canary evidence", + "public production launch" + ] +} + From 0df08820f2d907ebb32f007e0a16d90e4f2f88dc Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 10:27:20 -0500 Subject: [PATCH 31/55] Record unverified notification drill --- docs/cloud-run-staging.md | 6 +++++- evidence/staging-deployment-v1/report.json | 5 +++++ 2 files changed, 10 insertions(+), 1 deletion(-) diff --git a/docs/cloud-run-staging.md b/docs/cloud-run-staging.md index abbc843..f6023a1 100644 --- a/docs/cloud-run-staging.md +++ b/docs/cloud-run-staging.md @@ -95,7 +95,11 @@ after rollback. This proves revision traffic rollback for the disabled staging host, not incident response timing or a production rollback. Cloud Monitoring policy `11003571770572095050` watches this service's 5xx request count and routes to Wilfried's work-email channel, but delivery and acknowledgement have -not been exercised. +not been verified. A temporary 2xx notification drill on 2026-09-23 observed +the authorized health-request metric but no Monitoring incident during the +check window; the drill policy was disabled. This is an open monitoring +qualification finding, not evidence of successful paging. The persistent +5xx policy remains enabled. ## Staged promotion diff --git a/evidence/staging-deployment-v1/report.json b/evidence/staging-deployment-v1/report.json index 043db5d..86a1721 100644 --- a/evidence/staging-deployment-v1/report.json +++ b/evidence/staging-deployment-v1/report.json @@ -28,6 +28,11 @@ "post_rollback_iam_health_status": 200, "staging_5xx_policy_id": "11003571770572095050", "on_call_email_channel_created": true, + "notification_drill_policy_id": "3641110671112699878", + "notification_drill_health_statuses": [200, 200, 200], + "notification_drill_2xx_metric_observed": true, + "notification_drill_incident_observed": false, + "notification_drill_policy_disabled_after_check": true, "alert_delivery_verified": false, "human_acknowledgement_verified": false }, From 5a2c281ae338473e73b2ac35ce90366bad0e598c Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 10:28:36 -0500 Subject: [PATCH 32/55] Clarify private staging status in README --- README.md | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index 4181eb8..70083f0 100644 --- a/README.md +++ b/README.md @@ -65,7 +65,7 @@ failed verification into success, or directly commit an external effect. | Authority boundary | action-bound verifier attestations, expiring single-use approvals, atomic nonce claims | | Runtime control | conservative CVoC selection, deterministic `STOP`, cost twin, deliberative contracts, trace schema and optional transactional SQLite hash chain | | Operational controls | fail-closed feature flags, tested frontier/open-weight HTTP contracts, Colibri shadow recommendations, canonical trace hash chain | -| Standalone controller boundary | tested state → candidates → optional Laya → CVoC → independent verifier → recommendation/commit fallback → redacted trace composition; host-owned read-only request factory, data-boundary-aware provider bridge, and authenticated rate-limited loopback HTTP boundary (not publicly deployed) | +| Standalone controller boundary | tested state → candidates → optional Laya → CVoC → independent verifier → recommendation/commit fallback → redacted trace composition; host-owned read-only request factory, data-boundary-aware provider bridge, and authenticated rate-limited loopback HTTP boundary. A separate private Cloud Run staging host is deliberately fixed-disabled; it is not the decision service or a public launch. | This compiler is the reviewed vertical slice, not the directive's full Candidate Compiler Definition of Done. Rich typed value constraints, per-argument provenance, dominated-branch @@ -201,3 +201,4 @@ Do not report vulnerabilities in a public issue; follow [`SECURITY.md`](SECURITY effects must be completely mediated by an independently configured commit gateway. Apache License 2.0. See [`LICENSE`](LICENSE). If you use C3R, cite [`CITATION.cff`](CITATION.cff). + From 2cb34a09ced100a06d615bf07b539a6ff97d7828 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 10:31:37 -0500 Subject: [PATCH 33/55] Record acknowledged staging alert drill --- docs/cloud-run-staging.md | 19 +++++++++++-------- docs/independent-review-handoff.md | 8 +++++--- docs/internal-task-trace-policy.md | 8 +++++--- docs/standalone-launch.md | 8 +++++--- evidence/staging-deployment-v1/report.json | 11 +++++++---- 5 files changed, 33 insertions(+), 21 deletions(-) diff --git a/docs/cloud-run-staging.md b/docs/cloud-run-staging.md index f6023a1..6b88209 100644 --- a/docs/cloud-run-staging.md +++ b/docs/cloud-run-staging.md @@ -48,8 +48,9 @@ estimates, durable governed storage, and approved source registry is required la confirmed interim on-call/rollback operator in the [internal-task policy](internal-task-trace-policy.md). Verify the work-email alert route and acknowledgement before live internal-task traffic. A C3R-only - Cloud Monitoring email channel and 5xx policy now exist, but delivery and - human acknowledgement are unverified. + Cloud Monitoring email channel and 5xx policy now exist. Wilfried reported + receiving and acknowledging a staging drill alert on 2026-09-23; an + automated mailbox delivery audit and incident response timing remain open. 2. Implement the approved internal-task trace policy: source registry, field allowlist, redaction tests, 30-day private deletion including backups, access audit, and publication review. The approved scope is only redacted telemetry @@ -94,12 +95,14 @@ was deployed, then traffic was explicitly restored 100% to the original after rollback. This proves revision traffic rollback for the disabled staging host, not incident response timing or a production rollback. Cloud Monitoring policy `11003571770572095050` watches this service's 5xx request count and -routes to Wilfried's work-email channel, but delivery and acknowledgement have -not been verified. A temporary 2xx notification drill on 2026-09-23 observed -the authorized health-request metric but no Monitoring incident during the -check window; the drill policy was disabled. This is an open monitoring -qualification finding, not evidence of successful paging. The persistent -5xx policy remains enabled. +routes to Wilfried's work-email channel. A temporary 2xx notification drill +on 2026-09-23 observed +the authorized health-request metric and opened Monitoring incident +`0.ocyx1hrngr2m` at 15:24:53 UTC, just after the first check and policy +disable. Wilfried subsequently reported receiving and acknowledging the drill +email in his work mailbox. This is operator-reported delivery evidence, not +an automated mailbox audit or response-time SLA test. The drill policy is +disabled; the persistent 5xx policy remains enabled. ## Staged promotion diff --git a/docs/independent-review-handoff.md b/docs/independent-review-handoff.md index b0d924f..63ea814 100644 --- a/docs/independent-review-handoff.md +++ b/docs/independent-review-handoff.md @@ -44,9 +44,11 @@ verify independence in practice, complete the review, or authorize launch. `python scripts/run_trace_control_dry_run.py`. - [Private staging evidence](../evidence/staging-deployment-v1/report.json) shows a fixed-disabled Cloud Run boundary, IAM and token checks, a service- - specific 5xx alert rule, and revision traffic rollback. This is not a live - C3R decision route or evidence of governed trace storage, alert delivery, - backup deletion, independent anchoring, or canaries. + specific 5xx alert rule, revision traffic rollback, and a notification + drill that opened a Monitoring incident. Wilfried reported receiving and + acknowledging the drill email. This is not a live C3R decision route or + evidence of governed trace storage, backup deletion, independent anchoring, + automated mailbox audit, response-time SLA, or canaries. - [Launch issue #3](https://github.com/ColomboAI-com/c3r/issues/3) lists the unclosed production gates. Governed live trace collection remains off. diff --git a/docs/internal-task-trace-policy.md b/docs/internal-task-trace-policy.md index 0ca2366..b9adaa9 100644 --- a/docs/internal-task-trace-policy.md +++ b/docs/internal-task-trace-policy.md @@ -5,9 +5,11 @@ this task. **Accountable interim deployment, release, and data owner:** Wilfried Kouadio (`@wilkont`), as identified by the active ColomboAI GCP account. On 2026-09-23, Wilfried confirmed that he will serve as the interim human on-call and rollback operator via his ColomboAI work identity -(`wilfried.k@colomboai.com`). This names the responder; it does not prove that -monitoring, alert delivery, acknowledgement, or rollback access is configured. -Those controls must be verified before hosted traffic. +(`wilfried.k@colomboai.com`). A fixed-disabled private staging drill subsequently +opened a Monitoring incident; Wilfried reported receiving and acknowledging +its email, and revision rollback was exercised. That one drill does not prove +continuous operational coverage or production incident response timing. +Those controls must be verified before governed internal-task traffic. **Named independent reviewer:** Swapnil Pawar, designated by Wilfried on 2026-09-23 for internal-task labels and any proposed public de-identified rows. diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index 7baa09a..2566511 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -48,10 +48,12 @@ canary evidence are independently reviewed. 5. The user designated Wilfried Kouadio (`@wilkont`) as interim accountable deployment, release, and data owner, and on 2026-09-23 Wilfried confirmed he will be the interim human on-call/rollback operator. The - [internal-task policy](internal-task-trace-policy.md) records this. Alert - delivery, acknowledgement, rollback access, monitoring, and read-only and + [internal-task policy](internal-task-trace-policy.md) records this. A private + staging drill opened a Monitoring incident, and Wilfried reported receiving + and acknowledging its email; disabled-revision rollback was also exercised. + Production incident response timing, broader monitoring, and read-only and reversible canary thresholds and aborts still need verification. A named - responder is not proof of operational coverage. + responder and one drill are not proof of continuous operational coverage. 6. Wilfried named Swapnil Pawar as the independent data reviewer on 2026-09-23. Swapnil accepted by work email that day and reported no conflict. The [non-sensitive local fixture dry run](../evidence/trace-control-dry-run-v1/report.json) diff --git a/evidence/staging-deployment-v1/report.json b/evidence/staging-deployment-v1/report.json index 86a1721..ce16b79 100644 --- a/evidence/staging-deployment-v1/report.json +++ b/evidence/staging-deployment-v1/report.json @@ -31,10 +31,13 @@ "notification_drill_policy_id": "3641110671112699878", "notification_drill_health_statuses": [200, 200, 200], "notification_drill_2xx_metric_observed": true, - "notification_drill_incident_observed": false, + "notification_drill_incident_observed": true, + "notification_drill_incident_open_time_utc": "2026-09-23T15:24:53Z", "notification_drill_policy_disabled_after_check": true, - "alert_delivery_verified": false, - "human_acknowledgement_verified": false + "alert_delivery_verified": true, + "alert_delivery_evidence": "Wilfried reported receiving the staging notification drill email in his ColomboAI work mailbox on 2026-09-23; no mailbox delivery log was independently captured", + "human_acknowledgement_verified": true, + "human_acknowledgement_evidence": "Wilfried confirmed receipt and acknowledgement in the C3R launch task on 2026-09-23" }, "not_verified": [ "functional C3R decision routing or live provider calls", @@ -42,7 +45,7 @@ "durable encrypted trace storage and read/export access auditing", "daily 30-day deletion including backups", "independent ledger-head anchoring", - "alert delivery, acknowledgement, or incident response timing", + "automated mailbox delivery audit and incident response timing", "trained C3R Laya weights, held-out calibration, and empirical DecisionMix", "independent safety review and canary evidence", "public production launch" From a45899a2653d4cb912edccd6046ff65efe4b7537 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 10:48:19 -0500 Subject: [PATCH 34/55] Add private C3R retention backstop and staged evidence --- c3r/retention_job.py | 141 ++++++++++++++++++ deploy/storage/private-traces-lifecycle.json | 8 + docs/cloud-run-staging.md | 10 +- docs/independent-review-handoff.md | 6 +- docs/internal-task-trace-policy.md | 10 +- docs/standalone-launch.md | 8 +- .../private-retention-staging-v1/report.json | 53 +++++++ tests/test_retention_job.py | 60 ++++++++ 8 files changed, 292 insertions(+), 4 deletions(-) create mode 100644 c3r/retention_job.py create mode 100644 deploy/storage/private-traces-lifecycle.json create mode 100644 evidence/private-retention-staging-v1/report.json create mode 100644 tests/test_retention_job.py diff --git a/c3r/retention_job.py b/c3r/retention_job.py new file mode 100644 index 0000000..c5e3f8a --- /dev/null +++ b/c3r/retention_job.py @@ -0,0 +1,141 @@ +"""Bounded deletion check for the dedicated private C3R trace bucket. + +This is a retention backstop, not permission to collect traces. The bucket name, +cutoff, and API host are pinned so a deployment argument cannot target other data. +""" + +from __future__ import annotations + +import json +from dataclasses import dataclass +from datetime import datetime, timedelta, timezone +from typing import Protocol +from urllib.parse import quote, urlencode +from urllib.request import Request, urlopen + + +BUCKET = "colomboai-c3r-private-traces-795563500003" +DELETE_AFTER_DAYS = 28 +_STORAGE_API = "https://storage.googleapis.com/storage/v1/b/" + BUCKET + "/o" +_METADATA_TOKEN = ( + "http://metadata.google.internal/computeMetadata/v1/instance/" + "service-accounts/default/token" +) + + +@dataclass(frozen=True, slots=True) +class StoredObject: + name: str + generation: int + created_at: datetime + + +class ObjectClient(Protocol): + def list_all(self) -> tuple[StoredObject, ...]: ... + + def delete_if_generation(self, obj: StoredObject) -> None: ... + + +def _created_at(raw: str) -> datetime: + parsed = datetime.fromisoformat(raw.replace("Z", "+00:00")) + if parsed.tzinfo is None: + raise ValueError("object creation time must include UTC offset") + return parsed.astimezone(timezone.utc) + + +def purge(client: ObjectClient, *, now: datetime) -> dict[str, object]: + if now.tzinfo is None or now.utcoffset() != timedelta(0): + raise ValueError("purge clock must be UTC") + cutoff = now - timedelta(days=DELETE_AFTER_DAYS) + before = client.list_all() + if len(before) > 100_000 or len({item.name for item in before}) != len(before): + raise ValueError("bucket inventory is too large or contains duplicate names") + expired = tuple(item for item in before if item.created_at <= cutoff) + for item in expired: + client.delete_if_generation(item) + remaining = client.list_all() + if any(item.created_at <= cutoff for item in remaining): + raise RuntimeError("expired objects remain after purge; collection must stay disabled") + return { + "bucket": BUCKET, + "cutoff_utc": cutoff.isoformat(), + "objects_before": len(before), + "expired_candidates": len(expired), + "objects_after": len(remaining), + "expired_remaining": 0, + "trace_collection_enabled": False, + } + + +class GcsJsonClient: + """Cloud Run service-identity transport for one pinned GCS bucket.""" + + def __init__(self) -> None: + request = Request(_METADATA_TOKEN, headers={"Metadata-Flavor": "Google"}) + with urlopen(request, timeout=10) as response: + data = json.load(response) + token = data.get("access_token") + if not isinstance(token, str) or len(token) < 20: + raise RuntimeError("Cloud Run service identity token unavailable") + self._authorization = "Bearer " + token + + def _request(self, method: str, url: str) -> dict[str, object] | None: + request = Request(url, method=method, headers={"Authorization": self._authorization}) + with urlopen(request, timeout=30) as response: + if method == "DELETE": + return None + data = json.load(response) + if not isinstance(data, dict): + raise ValueError("invalid GCS response") + return data + + def list_all(self) -> tuple[StoredObject, ...]: + found: list[StoredObject] = [] + page_token: str | None = None + seen_tokens: set[str] = set() + while True: + query = {"maxResults": "1000", "fields": "items(name,generation,timeCreated),nextPageToken"} + if page_token is not None: + query["pageToken"] = page_token + data = self._request("GET", _STORAGE_API + "?" + urlencode(query)) + assert data is not None + items = data.get("items", []) + if not isinstance(items, list): + raise ValueError("invalid GCS object list") + for raw in items: + if not isinstance(raw, dict): + raise ValueError("invalid GCS object metadata") + name, generation, created = ( + raw.get("name"), raw.get("generation"), raw.get("timeCreated") + ) + if not isinstance(name, str) or not name or not isinstance(created, str): + raise ValueError("invalid GCS object metadata") + if not isinstance(generation, str) or not generation.isdecimal(): + raise ValueError("invalid GCS object generation") + found.append(StoredObject(name, int(generation), _created_at(created))) + if len(found) > 100_000: + raise ValueError("bucket inventory exceeds purge limit") + next_token = data.get("nextPageToken") + if next_token is None: + return tuple(found) + if not isinstance(next_token, str) or not next_token or next_token in seen_tokens: + raise ValueError("invalid or repeated GCS page token") + seen_tokens.add(next_token) + page_token = next_token + + def delete_if_generation(self, obj: StoredObject) -> None: + if not obj.name or obj.generation <= 0: + raise ValueError("invalid deletion target") + url = _STORAGE_API + "/" + quote(obj.name, safe="") + "?" + urlencode( + {"ifGenerationMatch": str(obj.generation)} + ) + self._request("DELETE", url) + + +def main() -> None: + result = purge(GcsJsonClient(), now=datetime.now(timezone.utc)) + print(json.dumps(result, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/deploy/storage/private-traces-lifecycle.json b/deploy/storage/private-traces-lifecycle.json new file mode 100644 index 0000000..c617ece --- /dev/null +++ b/deploy/storage/private-traces-lifecycle.json @@ -0,0 +1,8 @@ +{ + "rule": [ + { + "action": {"type": "Delete"}, + "condition": {"age": 28} + } + ] +} diff --git a/docs/cloud-run-staging.md b/docs/cloud-run-staging.md index 6b88209..e5b90f6 100644 --- a/docs/cloud-run-staging.md +++ b/docs/cloud-run-staging.md @@ -42,6 +42,15 @@ IAM/TLS, startup, monitoring, and rollback, but **not** to collect empirical dat or qualify C3R decisions. A separately reviewed host with measured pre-decision estimates, durable governed storage, and approved source registry is required later. +An independent, empty C3R trace bucket and retention purge job now exist in +staging. The bucket is not mounted or granted to `c3r-staging`; the purge job's +service account has only bucket-scoped list/delete authority. A 28-day lifecycle +delete rule and daily UTC scheduler are configured, and one manual empty-bucket +purge completed. The [retention staging record](../evidence/private-retention-staging-v1/report.json) +distinguishes these observations from unverified scheduled execution, expired +object/backup deletion, effective inherited IAM, and access auditing. Nothing +about this bucket enables trace collection or production qualification. + ## Controls still required before live collection or promotion 1. Wilfried Kouadio (`@wilkont`) is the interim deployment/release owner and @@ -122,4 +131,3 @@ Cloud Run references: [HTTPS service URL and invoking services](https://docs.clo [IAM service authentication](https://docs.cloud.google.com/run/docs/authenticating/overview), [public versus authenticated deployment](https://docs.cloud.google.com/run/docs/deploying), and [container listening contract](https://docs.cloud.google.com/run/docs/container-contract). - diff --git a/docs/independent-review-handoff.md b/docs/independent-review-handoff.md index 63ea814..fee9ac0 100644 --- a/docs/independent-review-handoff.md +++ b/docs/independent-review-handoff.md @@ -49,6 +49,11 @@ verify independence in practice, complete the review, or authorize launch. acknowledging the drill email. This is not a live C3R decision route or evidence of governed trace storage, backup deletion, independent anchoring, automated mailbox audit, response-time SLA, or canaries. +- [Private retention staging evidence](../evidence/private-retention-staging-v1/report.json) + records an empty dedicated bucket, restricted purge identity, 28-day lifecycle + backstop, configured daily schedule, and one successful empty-bucket purge. + It does not prove scheduled execution, expired-object or backup deletion, + data-access auditing, source governance, or independent anchoring. - [Launch issue #3](https://github.com/ColomboAI-com/c3r/issues/3) lists the unclosed production gates. Governed live trace collection remains off. @@ -61,4 +66,3 @@ and a clear **approve / reject / needs changes** decision for each gate. A qualified public launch additionally requires owner release approval and live operational evidence. MC-1 is excluded from this standalone scope; it is not complete under the original directive. - diff --git a/docs/internal-task-trace-policy.md b/docs/internal-task-trace-policy.md index b9adaa9..7567b8a 100644 --- a/docs/internal-task-trace-policy.md +++ b/docs/internal-task-trace-policy.md @@ -83,6 +83,15 @@ trace storage or decision authority. Swapnil Pawar or another subsequently approved independent reviewer must verify the complete controls against an intentionally non-sensitive dry run. +On 2026-09-23 a separate empty C3R-only staging bucket and 28-day deletion +backstops were provisioned. The bucket has uniform access, public-access +prevention, no versioning or soft-delete retention, and a 28-day lifecycle rule. +A least-privilege Cloud Run purge job completed one empty-bucket run; a daily UTC +scheduler was configured. See the [staging retention evidence](../evidence/private-retention-staging-v1/report.json). +This does **not** demonstrate deletion of expired data, backup deletion, +read/export auditing, independent anchoring, or approved live-source operation. +The disabled C3R service has no access to the bucket. Collection stays off. + The reference [`GovernedTraceStore`](../c3r/telemetry/governed_store.py) enforces an attested source/task allowlist, a bounded token/numeric trace schema, a local 30-day purge operation, tamper checks, and a post-purge chain checkpoint in tests. It @@ -99,4 +108,3 @@ one binding for unrelated requests or accept those identifiers from callers. No public endpoint, public dataset promotion, model-weight release, or broad access is approved by this policy alone. Each requires its own evidence-matched release decision. This document is a project governance record, not a legal opinion. - diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index 2566511..3a30603 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -79,4 +79,10 @@ No public production claim should be made while any exit condition above is unme The [Cloud Run staging record](cloud-run-staging.md) records the private fixed-disabled deployment and the controls still blocking live collection and production promotion. - + +The [private retention staging record](../evidence/private-retention-staging-v1/report.json) +adds an empty access-restricted bucket, 28-day lifecycle rule, tested empty-bucket +purge job, and configured daily scheduler. It is not evidence of deletion of +expired rows or backups, audited data access, independent ledger anchoring, or +reviewer approval. The C3R decision service remains fixed-disabled and cannot +write traces. diff --git a/evidence/private-retention-staging-v1/report.json b/evidence/private-retention-staging-v1/report.json new file mode 100644 index 0000000..b99d5d2 --- /dev/null +++ b/evidence/private-retention-staging-v1/report.json @@ -0,0 +1,53 @@ +{ + "as_of_utc": "2026-09-23T15:44:00Z", + "scope": "private staging control, no trace collection", + "project": "columboai-frontend", + "region": "us-central1", + "bucket": "gs://colomboai-c3r-private-traces-795563500003", + "bucket_observed": { + "uniform_bucket_level_access": true, + "public_access_prevention": "enforced", + "soft_delete_duration_seconds": 0, + "versioning_enabled": false, + "lifecycle_delete_age_days": 28, + "objects_at_dry_run": 0, + "bucket_iam": "projectOwner legacy bucket/object ownership and bucket-scoped custom purger role; inherited project grants not audited" + }, + "purger": { + "cloud_run_job": "c3r-retention-purge", + "service_account": "c3r-retention-purge@columboai-frontend.iam.gserviceaccount.com", + "custom_role": "projects/columboai-frontend/roles/c3rTracePurger", + "custom_role_permissions": ["storage.objects.list", "storage.objects.delete"], + "image_digest": "sha256:4490e0d00049c8d8ba07350184c5044d4a84e5b4098f0d318e2be05f86a310de", + "manual_execution": "c3r-retention-purge-9twch", + "manual_execution_result": "completed successfully", + "manual_execution_report": { + "objects_before": 0, + "expired_candidates": 0, + "objects_after": 0, + "expired_remaining": 0, + "trace_collection_enabled": false + } + }, + "scheduler": { + "job": "c3r-retention-daily", + "schedule": "15 0 * * *", + "time_zone": "Etc/UTC", + "state_at_creation": "ENABLED", + "service_account": "c3r-retention-scheduler@columboai-frontend.iam.gserviceaccount.com", + "authority": "roles/run.invoker on c3r-retention-purge only", + "manual_trigger_requested": true, + "manual_trigger_verified": false + }, + "not_verified": [ + "scheduled execution completion", + "deletion of an expired object", + "backup and replica deletion under live data", + "effective inherited IAM and read/export audit", + "independent ledger anchoring", + "governed internal-task source and live trace", + "independent reviewer sign-off" + ], + "collection_enabled": false, + "production_qualified": false +} diff --git a/tests/test_retention_job.py b/tests/test_retention_job.py new file mode 100644 index 0000000..649e235 --- /dev/null +++ b/tests/test_retention_job.py @@ -0,0 +1,60 @@ +import unittest +from datetime import datetime, timedelta, timezone + +from c3r.retention_job import DELETE_AFTER_DAYS, StoredObject, _created_at, purge + + +NOW = datetime(2026, 9, 23, 16, 0, tzinfo=timezone.utc) + + +class FakeClient: + def __init__(self, objects): + self.objects = list(objects) + self.deleted = [] + + def list_all(self): + return tuple(self.objects) + + def delete_if_generation(self, obj): + self.deleted.append((obj.name, obj.generation)) + self.objects = [item for item in self.objects if item != obj] + + +class RetentionJobTests(unittest.TestCase): + def test_deletes_only_expired_generation_and_verifies_empty_overdue_set(self): + old = StoredObject("traces/old", 3, NOW - timedelta(days=DELETE_AFTER_DAYS)) + fresh = StoredObject("traces/new", 4, NOW - timedelta(days=1)) + client = FakeClient([old, fresh]) + + report = purge(client, now=NOW) + + self.assertEqual(client.deleted, [(old.name, old.generation)]) + self.assertEqual(client.objects, [fresh]) + self.assertEqual(report["expired_remaining"], 0) + self.assertFalse(report["trace_collection_enabled"]) + + def test_surviving_expired_object_fails_closed(self): + old = StoredObject("traces/old", 3, NOW - timedelta(days=29)) + + class NonDeletingClient(FakeClient): + def delete_if_generation(self, obj): + self.deleted.append((obj.name, obj.generation)) + + with self.assertRaisesRegex(RuntimeError, "expired objects remain"): + purge(NonDeletingClient([old]), now=NOW) + + def test_duplicate_inventory_and_non_utc_clock_rejected(self): + old = StoredObject("traces/old", 3, NOW - timedelta(days=29)) + with self.assertRaisesRegex(ValueError, "duplicate"): + purge(FakeClient([old, old]), now=NOW) + with self.assertRaisesRegex(ValueError, "UTC"): + purge(FakeClient([]), now=NOW.replace(tzinfo=None)) + + def test_creation_timestamp_must_be_timezone_aware(self): + self.assertEqual(_created_at("2026-09-23T16:00:00Z"), NOW) + with self.assertRaisesRegex(ValueError, "UTC offset"): + _created_at("2026-09-23T16:00:00") + + +if __name__ == "__main__": + unittest.main() From 075d59a9c1331ed263e7541309bf405d73122765 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 10:50:37 -0500 Subject: [PATCH 35/55] Verify scheduler-triggered empty retention run --- docs/cloud-run-staging.md | 6 +++--- docs/independent-review-handoff.md | 5 +++-- docs/internal-task-trace-policy.md | 5 +++-- docs/standalone-launch.md | 4 ++-- evidence/private-retention-staging-v1/report.json | 15 ++++++++++++--- 5 files changed, 23 insertions(+), 12 deletions(-) diff --git a/docs/cloud-run-staging.md b/docs/cloud-run-staging.md index e5b90f6..bd289d0 100644 --- a/docs/cloud-run-staging.md +++ b/docs/cloud-run-staging.md @@ -45,9 +45,9 @@ estimates, durable governed storage, and approved source registry is required la An independent, empty C3R trace bucket and retention purge job now exist in staging. The bucket is not mounted or granted to `c3r-staging`; the purge job's service account has only bucket-scoped list/delete authority. A 28-day lifecycle -delete rule and daily UTC scheduler are configured, and one manual empty-bucket -purge completed. The [retention staging record](../evidence/private-retention-staging-v1/report.json) -distinguishes these observations from unverified scheduled execution, expired +delete rule and daily UTC scheduler are configured, and direct and scheduler- +triggered empty-bucket purges completed. The [retention staging record](../evidence/private-retention-staging-v1/report.json) +distinguishes these observations from the first natural daily run, expired object/backup deletion, effective inherited IAM, and access auditing. Nothing about this bucket enables trace collection or production qualification. diff --git a/docs/independent-review-handoff.md b/docs/independent-review-handoff.md index fee9ac0..a29964a 100644 --- a/docs/independent-review-handoff.md +++ b/docs/independent-review-handoff.md @@ -51,8 +51,9 @@ verify independence in practice, complete the review, or authorize launch. automated mailbox audit, response-time SLA, or canaries. - [Private retention staging evidence](../evidence/private-retention-staging-v1/report.json) records an empty dedicated bucket, restricted purge identity, 28-day lifecycle - backstop, configured daily schedule, and one successful empty-bucket purge. - It does not prove scheduled execution, expired-object or backup deletion, + backstop, configured daily schedule, and successful direct and scheduler- + triggered empty-bucket purges. It does not prove the first natural daily run, + expired-object or backup deletion, data-access auditing, source governance, or independent anchoring. - [Launch issue #3](https://github.com/ColomboAI-com/c3r/issues/3) lists the unclosed production gates. Governed live trace collection remains off. diff --git a/docs/internal-task-trace-policy.md b/docs/internal-task-trace-policy.md index 7567b8a..82ca325 100644 --- a/docs/internal-task-trace-policy.md +++ b/docs/internal-task-trace-policy.md @@ -86,8 +86,9 @@ intentionally non-sensitive dry run. On 2026-09-23 a separate empty C3R-only staging bucket and 28-day deletion backstops were provisioned. The bucket has uniform access, public-access prevention, no versioning or soft-delete retention, and a 28-day lifecycle rule. -A least-privilege Cloud Run purge job completed one empty-bucket run; a daily UTC -scheduler was configured. See the [staging retention evidence](../evidence/private-retention-staging-v1/report.json). +A least-privilege Cloud Run purge job completed both a direct and a scheduler- +triggered empty-bucket run; a daily UTC schedule is configured. See the +[staging retention evidence](../evidence/private-retention-staging-v1/report.json). This does **not** demonstrate deletion of expired data, backup deletion, read/export auditing, independent anchoring, or approved live-source operation. The disabled C3R service has no access to the bucket. Collection stays off. diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index 3a30603..dbe2b58 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -81,8 +81,8 @@ fixed-disabled deployment and the controls still blocking live collection and production promotion. The [private retention staging record](../evidence/private-retention-staging-v1/report.json) -adds an empty access-restricted bucket, 28-day lifecycle rule, tested empty-bucket -purge job, and configured daily scheduler. It is not evidence of deletion of +adds an empty access-restricted bucket, 28-day lifecycle rule, direct and scheduler- +triggered empty-bucket purge runs, and configured daily schedule. It is not evidence of deletion of expired rows or backups, audited data access, independent ledger anchoring, or reviewer approval. The C3R decision service remains fixed-disabled and cannot write traces. diff --git a/evidence/private-retention-staging-v1/report.json b/evidence/private-retention-staging-v1/report.json index b99d5d2..caff9c3 100644 --- a/evidence/private-retention-staging-v1/report.json +++ b/evidence/private-retention-staging-v1/report.json @@ -1,5 +1,5 @@ { - "as_of_utc": "2026-09-23T15:44:00Z", + "as_of_utc": "2026-09-23T15:49:33Z", "scope": "private staging control, no trace collection", "project": "columboai-frontend", "region": "us-central1", @@ -37,10 +37,19 @@ "service_account": "c3r-retention-scheduler@columboai-frontend.iam.gserviceaccount.com", "authority": "roles/run.invoker on c3r-retention-purge only", "manual_trigger_requested": true, - "manual_trigger_verified": false + "manual_trigger_verified": true, + "triggered_execution": "c3r-retention-purge-5zw56", + "triggered_execution_result": "completed successfully", + "triggered_execution_report": { + "objects_before": 0, + "expired_candidates": 0, + "objects_after": 0, + "expired_remaining": 0, + "trace_collection_enabled": false + } }, "not_verified": [ - "scheduled execution completion", + "first natural daily scheduled execution", "deletion of an expired object", "backup and replica deletion under live data", "effective inherited IAM and read/export audit", From 862daab37318a18e164f19b34a7b0a6383271abd Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 11:13:48 -0500 Subject: [PATCH 36/55] Verify scoped staging deletion and clarify review gate --- docs/cloud-run-staging.md | 7 +++++-- docs/independent-review-handoff.md | 10 ++++++++-- docs/internal-task-trace-policy.md | 5 ++++- docs/standalone-launch.md | 4 ++++ .../control-probe.txt | 1 + .../private-retention-staging-v1/report.json | 19 ++++++++++++++++--- 6 files changed, 38 insertions(+), 8 deletions(-) create mode 100644 evidence/private-retention-staging-v1/control-probe.txt diff --git a/docs/cloud-run-staging.md b/docs/cloud-run-staging.md index bd289d0..695c7a6 100644 --- a/docs/cloud-run-staging.md +++ b/docs/cloud-run-staging.md @@ -47,8 +47,11 @@ staging. The bucket is not mounted or granted to `c3r-staging`; the purge job's service account has only bucket-scoped list/delete authority. A 28-day lifecycle delete rule and daily UTC scheduler are configured, and direct and scheduler- triggered empty-bucket purges completed. The [retention staging record](../evidence/private-retention-staging-v1/report.json) -distinguishes these observations from the first natural daily run, expired -object/backup deletion, effective inherited IAM, and access auditing. Nothing +also records a generation-guarded deletion of one non-sensitive marker under +the purge identity, followed by empty live and all-version listings. The +one-off probe job was removed. It distinguishes these observations from the +first natural daily run, age-based expiry of real trace rows, deletion from +independent backups or replicas, effective inherited IAM, and access auditing. Nothing about this bucket enables trace collection or production qualification. ## Controls still required before live collection or promotion diff --git a/docs/independent-review-handoff.md b/docs/independent-review-handoff.md index a29964a..d80feff 100644 --- a/docs/independent-review-handoff.md +++ b/docs/independent-review-handoff.md @@ -6,6 +6,11 @@ ColomboAI work mailbox on 2026-09-23 accepting the role, reporting no conflict of interest, and requesting access to a non-sensitive dry run. His reply also acknowledged that collection and publication remain off. Acceptance does not verify independence in practice, complete the review, or authorize launch. +On 2026-09-23 he reported in the ColomboAI chat that he marked PR #2 ready for +review and saw passing checks. The PR had no submitted GitHub review or inline +review threads when checked afterward, and launch issue #3 had no independent +findings. Readiness and CI status are not substantive control, label, or public- +row sign-off; request a dated gate-by-gate finding with evidence references. ## What the reviewer should decide @@ -52,8 +57,9 @@ verify independence in practice, complete the review, or authorize launch. - [Private retention staging evidence](../evidence/private-retention-staging-v1/report.json) records an empty dedicated bucket, restricted purge identity, 28-day lifecycle backstop, configured daily schedule, and successful direct and scheduler- - triggered empty-bucket purges. It does not prove the first natural daily run, - expired-object or backup deletion, + triggered empty-bucket purges, plus a generation-guarded deletion of a + non-sensitive marker under the purge identity. It does not prove the first + natural daily run, age-based expired-object or backup deletion, data-access auditing, source governance, or independent anchoring. - [Launch issue #3](https://github.com/ColomboAI-com/c3r/issues/3) lists the unclosed production gates. Governed live trace collection remains off. diff --git a/docs/internal-task-trace-policy.md b/docs/internal-task-trace-policy.md index 82ca325..1929d3e 100644 --- a/docs/internal-task-trace-policy.md +++ b/docs/internal-task-trace-policy.md @@ -89,7 +89,10 @@ prevention, no versioning or soft-delete retention, and a 28-day lifecycle rule. A least-privilege Cloud Run purge job completed both a direct and a scheduler- triggered empty-bucket run; a daily UTC schedule is configured. See the [staging retention evidence](../evidence/private-retention-staging-v1/report.json). -This does **not** demonstrate deletion of expired data, backup deletion, +A subsequent 68-byte, non-sensitive marker was deleted by the purge identity +with a generation precondition; the live and all-version listings were empty +afterward. The one-off probe job was removed. This does **not** demonstrate +age-based deletion of expired data, independently configured backup deletion, read/export auditing, independent anchoring, or approved live-source operation. The disabled C3R service has no access to the bucket. Collection stays off. diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index dbe2b58..2df8398 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -86,3 +86,7 @@ triggered empty-bucket purge runs, and configured daily schedule. It is not evid expired rows or backups, audited data access, independent ledger anchoring, or reviewer approval. The C3R decision service remains fixed-disabled and cannot write traces. + +The same record now includes one successful generation-guarded deletion of a +non-sensitive marker under the bucket-scoped purge identity. This validates +the delete API path, not the 28-day retention outcome. diff --git a/evidence/private-retention-staging-v1/control-probe.txt b/evidence/private-retention-staging-v1/control-probe.txt new file mode 100644 index 0000000..0db4158 --- /dev/null +++ b/evidence/private-retention-staging-v1/control-probe.txt @@ -0,0 +1 @@ +C3R staging deletion probe. Non-sensitive test marker; not a trace. diff --git a/evidence/private-retention-staging-v1/report.json b/evidence/private-retention-staging-v1/report.json index caff9c3..1784fcc 100644 --- a/evidence/private-retention-staging-v1/report.json +++ b/evidence/private-retention-staging-v1/report.json @@ -1,5 +1,5 @@ { - "as_of_utc": "2026-09-23T15:49:33Z", + "as_of_utc": "2026-09-23T16:12:05Z", "scope": "private staging control, no trace collection", "project": "columboai-frontend", "region": "us-central1", @@ -48,10 +48,23 @@ "trace_collection_enabled": false } }, + "scoped_delete_probe": { + "object": "control-probes/2026-09-23-delete-probe.txt", + "payload": "68-byte non-sensitive marker, not a trace", + "generation": "1790179707761698", + "job_execution": "c3r-retention-delete-probe-20260923-wnjqb", + "execution_result": "completed successfully", + "method": "GcsJsonClient listed exactly one object, asserted its generation, deleted with ifGenerationMatch, then asserted an empty live listing", + "post_run_live_listing": "empty", + "post_run_all_versions_listing": "no objects matched", + "soft_deleted_listing": "HTTP 400 because bucket soft-delete policy is disabled", + "one_off_probe_job_removed": true, + "trace_collection_enabled": false + }, "not_verified": [ "first natural daily scheduled execution", - "deletion of an expired object", - "backup and replica deletion under live data", + "age-based deletion of an expired object", + "deletion in any independently configured backup or replica", "effective inherited IAM and read/export audit", "independent ledger anchoring", "governed internal-task source and live trace", From c0cf49990a049dfc1cbab2d668de9ba3cfccf034 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 11:17:26 -0500 Subject: [PATCH 37/55] Add C3R retention failure alert policy and evidence --- .../c3r-retention-failure-policy.json | 32 +++++++++++++++++++ docs/cloud-run-staging.md | 3 ++ docs/independent-review-handoff.md | 3 +- .../private-retention-staging-v1/report.json | 12 ++++++- 4 files changed, 48 insertions(+), 2 deletions(-) create mode 100644 deploy/monitoring/c3r-retention-failure-policy.json diff --git a/deploy/monitoring/c3r-retention-failure-policy.json b/deploy/monitoring/c3r-retention-failure-policy.json new file mode 100644 index 0000000..e212dbf --- /dev/null +++ b/deploy/monitoring/c3r-retention-failure-policy.json @@ -0,0 +1,32 @@ +{ + "displayName": "C3R staging retention purge failure", + "combiner": "OR", + "enabled": true, + "conditions": [ + { + "displayName": "c3r-retention-purge failed execution", + "conditionThreshold": { + "filter": "metric.type=\"run.googleapis.com/job/completed_execution_count\" AND resource.type=\"cloud_run_job\" AND resource.labels.job_name=\"c3r-retention-purge\" AND metric.labels.result=\"failed\"", + "aggregations": [ + { + "alignmentPeriod": "60s", + "perSeriesAligner": "ALIGN_SUM", + "crossSeriesReducer": "REDUCE_SUM" + } + ], + "comparison": "COMPARISON_GT", + "thresholdValue": 0, + "duration": "0s", + "trigger": { "count": 1 } + } + } + ], + "notificationChannels": [ + "projects/columboai-frontend/notificationChannels/10331198960910692730" + ], + "documentation": { + "mimeType": "text/markdown", + "content": "C3R private staging retention purge failed. Wilfried: keep trace collection disabled, inspect the failed execution and bucket inventory, repair the purge and replay it, then verify no object exceeded the 30-day cap. This policy alone is not proof of daily execution or production readiness." + }, + "alertStrategy": { "autoClose": "604800s" } +} diff --git a/docs/cloud-run-staging.md b/docs/cloud-run-staging.md index 695c7a6..2a20364 100644 --- a/docs/cloud-run-staging.md +++ b/docs/cloud-run-staging.md @@ -53,6 +53,9 @@ one-off probe job was removed. It distinguishes these observations from the first natural daily run, age-based expiry of real trace rows, deletion from independent backups or replicas, effective inherited IAM, and access auditing. Nothing about this bucket enables trace collection or production qualification. +A C3R-only failed-purge alert policy now watches Cloud Run job completion +results and targets the existing work-email channel. Its configuration was +read back after creation; a failed-execution delivery drill has not run. ## Controls still required before live collection or promotion diff --git a/docs/independent-review-handoff.md b/docs/independent-review-handoff.md index d80feff..217c02f 100644 --- a/docs/independent-review-handoff.md +++ b/docs/independent-review-handoff.md @@ -60,7 +60,8 @@ row sign-off; request a dated gate-by-gate finding with evidence references. triggered empty-bucket purges, plus a generation-guarded deletion of a non-sensitive marker under the purge identity. It does not prove the first natural daily run, age-based expired-object or backup deletion, - data-access auditing, source governance, or independent anchoring. + data-access auditing, source governance, or independent anchoring. A failed- + purge alert policy is configured but has not had a delivery drill. - [Launch issue #3](https://github.com/ColomboAI-com/c3r/issues/3) lists the unclosed production gates. Governed live trace collection remains off. diff --git a/evidence/private-retention-staging-v1/report.json b/evidence/private-retention-staging-v1/report.json index 1784fcc..a781ec8 100644 --- a/evidence/private-retention-staging-v1/report.json +++ b/evidence/private-retention-staging-v1/report.json @@ -1,5 +1,5 @@ { - "as_of_utc": "2026-09-23T16:12:05Z", + "as_of_utc": "2026-09-23T16:16:11Z", "scope": "private staging control, no trace collection", "project": "columboai-frontend", "region": "us-central1", @@ -61,11 +61,21 @@ "one_off_probe_job_removed": true, "trace_collection_enabled": false }, + "failure_alert": { + "policy": "projects/columboai-frontend/alertPolicies/16956137431162400006", + "metric": "run.googleapis.com/job/completed_execution_count", + "scope": "c3r-retention-purge failed executions only", + "channel": "existing C3R staging work-email channel", + "enabled": true, + "policy_configuration_verified": true, + "failure_delivery_drill_verified": false + }, "not_verified": [ "first natural daily scheduled execution", "age-based deletion of an expired object", "deletion in any independently configured backup or replica", "effective inherited IAM and read/export audit", + "retention failure alert delivery on a real failed execution", "independent ledger anchoring", "governed internal-task source and live trace", "independent reviewer sign-off" From ebd6288f030e6ac81f80e4223f86de08a990ef33 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 11:18:36 -0500 Subject: [PATCH 38/55] Add independent C3R reviewer sign-off template --- docs/independent-review-handoff.md | 3 +++ docs/reviewer-signoff-template.md | 41 ++++++++++++++++++++++++++++++ 2 files changed, 44 insertions(+) create mode 100644 docs/reviewer-signoff-template.md diff --git a/docs/independent-review-handoff.md b/docs/independent-review-handoff.md index 217c02f..547762c 100644 --- a/docs/independent-review-handoff.md +++ b/docs/independent-review-handoff.md @@ -74,3 +74,6 @@ and a clear **approve / reject / needs changes** decision for each gate. A qualified public launch additionally requires owner release approval and live operational evidence. MC-1 is excluded from this standalone scope; it is not complete under the original directive. +The [review record template](reviewer-signoff-template.md) makes the required +gate-by-gate decisions and evidence fields explicit; it is not a pre-filled +approval. diff --git a/docs/reviewer-signoff-template.md b/docs/reviewer-signoff-template.md new file mode 100644 index 0000000..dcfb748 --- /dev/null +++ b/docs/reviewer-signoff-template.md @@ -0,0 +1,41 @@ +# Independent C3R review record — template + +Use this for an actual review, not for accepting the reviewer role or confirming +CI. The reviewer should submit the completed record as a PR review or a dated +reply from the approved ColomboAI work address. Do not attach credentials, +private trace rows, raw prompts, or customer/product records. A blank or +unsubstantiated approval is not a release sign-off. + +**Reviewer:** +**Date and time (UTC):** +**Reviewed commit SHA:** +**Scope and independence/conflicts:** +**Evidence links and immutable hashes:** + +For each gate, choose **approve / needs changes / reject / not reviewed** and +record what was inspected, a finding, and any limitation. Approval is limited +to the exact artifact revision and population reviewed. + +| Gate | Decision | Evidence inspected | Findings / limitations | +| --- | --- | --- | --- | +| C3R-controlled task source, rights and source registry | | | | +| Independent outcome labels and sealed split integrity | | | | +| Redaction, field allowlist and leakage tests | | | | +| Effective storage IAM, encryption, read/export audit | | | | +| Daily age-based deletion, backups/replicas and failure alert | | | | +| Independent ledger-head anchoring and tamper detection | | | | +| Kill switch, verifier/commit isolation and unsafe-action tests | | | | +| Trained weights, raw predictions and held-out calibration | | | | +| Paired baselines, provider outages, latency/cost and canary evidence | | | | +| Proposed de-identified public rows and release claims | | | | + +**Overall decision for collection:** approve / needs changes / reject / not reviewed + +**Overall decision for public standalone launch:** approve / needs changes / reject / not reviewed + +**Required changes before reconsideration:** + +No row marked “approve” permits collection or publication by itself. The +release owner must also verify the deployed controls and approve the exact +release separately. MC-1 remains outside this standalone scope, not complete +under the original directive. From daf7e9147e66c79a07bd88de5f13dd7ba43b5117 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 23 Sep 2026 11:24:12 -0500 Subject: [PATCH 39/55] Record independent fixture review and staging evidence gaps --- docs/independent-review-handoff.md | 13 +++++-- docs/internal-task-trace-policy.md | 4 +++ docs/reviewer-staging-packet-2026-09-23.md | 41 ++++++++++++++++++++++ docs/standalone-launch.md | 12 ++++--- 4 files changed, 64 insertions(+), 6 deletions(-) create mode 100644 docs/reviewer-staging-packet-2026-09-23.md diff --git a/docs/independent-review-handoff.md b/docs/independent-review-handoff.md index 547762c..42cb6fc 100644 --- a/docs/independent-review-handoff.md +++ b/docs/independent-review-handoff.md @@ -1,16 +1,24 @@ # Independent C3R release review handoff -**Status:** accepted, not reviewed or signed off. Wilfried Kouadio named Swapnil +**Status:** accepted; fixture-only dry run reviewed; deployment controls and +release not signed off. Wilfried Kouadio named Swapnil Pawar as the independent reviewer on 2026-09-23. Swapnil replied from his ColomboAI work mailbox on 2026-09-23 accepting the role, reporting no conflict of interest, and requesting access to a non-sensitive dry run. His reply also acknowledged that collection and publication remain off. Acceptance does not -verify independence in practice, complete the review, or authorize launch. +verify deployment controls or authorize launch. On 2026-09-23 he reported in the ColomboAI chat that he marked PR #2 ready for review and saw passing checks. The PR had no submitted GitHub review or inline review threads when checked afterward, and launch issue #3 had no independent findings. Readiness and CI status are not substantive control, label, or public- row sign-off; request a dated gate-by-gate finding with evidence references. +Swapnil's later 2026-09-23 work-email response reviewed the nine local checks +as clear and explicitly withheld live-collection and publication approval. He +requested deployed storage/IAM, access audit, deletion including copies, +independent anchoring, alert/rollback/kill-switch, and eventual live provenance +and outcome evidence. The [staging packet](reviewer-staging-packet-2026-09-23.md) +indexes what exists and what remains unavailable without treating the fixture +finding as deployment sign-off. ## What the reviewer should decide @@ -77,3 +85,4 @@ complete under the original directive. The [review record template](reviewer-signoff-template.md) makes the required gate-by-gate decisions and evidence fields explicit; it is not a pre-filled approval. + diff --git a/docs/internal-task-trace-policy.md b/docs/internal-task-trace-policy.md index 1929d3e..cdbbebe 100644 --- a/docs/internal-task-trace-policy.md +++ b/docs/internal-task-trace-policy.md @@ -17,6 +17,9 @@ He accepted by work email on 2026-09-23 and reported no conflict of interest. The non-sensitive fixture-only dry-run link was sent to him from ColomboAI's work mailbox on 2026-09-23. Independent control review and actual sign-off remain outstanding. Acceptance does not activate collection or authorize publication. +He subsequently reviewed the nine fixture-only local checks as clear, while +explicitly withholding approval for live collection and publication until the +deployment-level evidence is inspected. See the [review staging packet](reviewer-staging-packet-2026-09-23.md). This policy resolves the owner and data-use choices for the *controlled internal task population only*. It does **not** authorize production traffic or certify @@ -112,3 +115,4 @@ one binding for unrelated requests or accept those identifiers from callers. No public endpoint, public dataset promotion, model-weight release, or broad access is approved by this policy alone. Each requires its own evidence-matched release decision. This document is a project governance record, not a legal opinion. + diff --git a/docs/reviewer-staging-packet-2026-09-23.md b/docs/reviewer-staging-packet-2026-09-23.md new file mode 100644 index 0000000..aa65979 --- /dev/null +++ b/docs/reviewer-staging-packet-2026-09-23.md @@ -0,0 +1,41 @@ +# C3R private-staging review packet — 2026-09-23 + +This is a **non-sensitive evidence index** for Swapnil Pawar's independent +review. It contains no trace rows, prompts, credentials, or secrets. The +resources remain private; links to Google Cloud resources require separately +approved, least-privilege access. Do not grant project-wide access merely to +make this packet viewable. + +The reviewer accepted the role and reviewed the [fixture-only dry run](../evidence/trace-control-dry-run-v1/report.json). +He found the nine local checks clear, but explicitly did **not** approve live +collection or public publication. The dry run uses one synthetic C3R-authored +fixture. This packet responds to his request for available *deployment* +evidence; it does not turn an unavailable control into a pass. + +| Review gate | Available evidence | Still missing for sign-off | +| --- | --- | --- | +| Deployed storage and IAM | [Private retention staging record](../evidence/private-retention-staging-v1/report.json): dedicated empty bucket, uniform bucket-level access, public-access prevention, bucket policy, Cloud Run purger identity, and a generation-guarded marker deletion. [Disabled host record](../evidence/staging-deployment-v1/report.json): dedicated runtime identity, private IAM/TLS ingress and distinct Secret Manager token references. | Independent inspection of effective *inherited* IAM and encryption settings; a reviewed trace writer with least-privilege access. The fixed-disabled service has no bucket write grant. | +| Read/export audit | The private bucket was empty at the staging check; no hosted trace read/export probe has run. | An approved, working audit path with a reviewed access scope and a probe. Cloud Storage Data Access logging is not evidenced for this shared project; enabling it may affect other buckets and costs. | +| 30-day deletion and copies | 28-day bucket lifecycle rule; daily UTC purge schedule; direct and scheduler-triggered empty-bucket runs; one 68-byte marker deleted by the purger identity with a generation precondition; bucket then empty. | First natural daily execution, age-based deletion of an expired trace, and inventory/deletion proof for every independent backup or replica. The model-checkpoint backup is a separate asset and is **not** a trace backup. | +| Independent ledger-head anchoring | [Local dry run](../evidence/trace-control-dry-run-v1/report.json) checks a chain checkpoint; [governed store](../c3r/telemetry/governed_store.py) has local tamper checks. | Host-level append-only or independently controlled anchor, deployed write path, replay and tamper drill. | +| Alerting, rollback, kill switch | [Disabled staging record](../evidence/staging-deployment-v1/report.json): service-specific 5xx policy, owner-reported drill receipt/acknowledgement, and disabled-revision traffic rollback. [Retention record](../evidence/private-retention-staging-v1/report.json): failed-purge policy configured. Repository tests cover fail-closed controller behavior. | Failed-purge alert delivery drill, continuous on-call/response proof, live decision-host rollback and kill-switch drill. Disabled-host rollback is not production rollback. | +| Internal-task provenance and outcomes | [Five-case controlled replay](controlled-replay.md) has C3R-authored fixture states and predeclared rubric matches only. | Attested live C3R-controlled source registry, independently adjudicated outcomes, sealed partitions, governed trace runs, training/calibration and paired safety/canary evidence. | + +## Access and decision boundaries + +- ColomboAI WorkMail to `swapnil.p@colomboai.com` is the approved coordination + channel. This packet can be sent there as a link. It is **not** a grant of + Google Cloud, GitHub, or Hugging Face permissions. +- Before sharing raw cloud logs, object listings beyond this empty bucket, or + effective IAM exports, the owner must approve an access-controlled route and + minimize unrelated shared-project details. Never email tokens or private + traces. If direct read-only cloud access is needed, scope it to C3R resources + and verify the grants before issuance. +- Swapnil should use the [review record template](reviewer-signoff-template.md) + to state **approve / needs changes / reject / not reviewed** for each gate, + identify inspected revisions and limitations, and distinguish collection + approval from public-release approval. +- Collection, training on live traces, publication, and public service + promotion stay **off** pending the missing controls and explicit review. + MC-1 remains outside the standalone scope, not complete under the directive. + diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md index 2df8398..9f4d6e3 100644 --- a/docs/standalone-launch.md +++ b/docs/standalone-launch.md @@ -57,11 +57,14 @@ canary evidence are independently reviewed. 6. Wilfried named Swapnil Pawar as the independent data reviewer on 2026-09-23. Swapnil accepted by work email that day and reported no conflict. The [non-sensitive local fixture dry run](../evidence/trace-control-dry-run-v1/report.json) - was sent to him by work email, but independent review and sign-off on - labels, deployed controls, and any public rows remain outstanding. Neither the owner + was sent to him by work email. He reviewed its nine local checks as clear + but explicitly withheld approval for live collection and publication. + Independent review and sign-off on labels, deployed controls, and any + public rows remain outstanding. Neither the owner nor this assistant can substitute for that independent review. The - [reviewer handoff](independent-review-handoff.md) defines the requested scope - and evidence without implying approval. + [reviewer handoff](independent-review-handoff.md) and + [staging packet](reviewer-staging-packet-2026-09-23.md) define the requested + scope and available evidence without implying approval. ## Internal-task telemetry authorization @@ -90,3 +93,4 @@ write traces. The same record now includes one successful generation-guarded deletion of a non-sensitive marker under the bucket-scoped purge identity. This validates the delete API path, not the 28-day retention outcome. + From 75bd27fb7467aee5ffd12e963d2e27d708fd9b1f Mon Sep 17 00:00:00 2001 From: wilkont Date: Thu, 24 Sep 2026 11:01:54 -0500 Subject: [PATCH 40/55] docs: update README.md with verified review status --- README.md | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/README.md b/README.md index 70083f0..7a85970 100644 --- a/README.md +++ b/README.md @@ -20,6 +20,13 @@ fast path with calibration-gated abstention, and a reproducible DecisionMix v1 s It is a research alpha: the controller is runnable and tested; fine-tuned weights and empirical production calibration remain release gates, not implied claims. +> **Launch status:** The standalone decision service and governed trace collection +> are not enabled for public use. The independent [PR #2 review](https://github.com/ColomboAI-com/c3r/pull/2#pullrequestreview-5293835066) +> requests changes. A real cloud storage audit probe and the first natural +> empty-bucket purge run are recorded for reviewer inspection, but they do not +> establish deletion of aged traces or backups, trained weights, calibration, +> safety results, or canary evidence. See the [reviewer packet](docs/reviewer-staging-packet-2026-09-23.md). + ### Default language model C3R's default **deliberative** language model is From 656b65973b9ffa9b3e7b39cde34bd45ff92c5dae Mon Sep 17 00:00:00 2001 From: wilkont Date: Thu, 24 Sep 2026 11:06:19 -0500 Subject: [PATCH 41/55] Add fail-closed standalone staging and governed trace controls --- .dockerignore | 1 + .github/workflows/ci.yml | 1 + Dockerfile | 1 + c3r/ingress_proxy.py | 1 + c3r/serve.py | 1 + c3r/telemetry/governed_store.py | 47 +++++++++++++---- c3r/telemetry/ledger_anchor.py | 81 ++++++++++++++++++++++++++++++ tests/test_governed_trace_store.py | 45 +++++++++++++++++ tests/test_ingress_proxy.py | 1 + tests/test_ledger_anchor.py | 71 ++++++++++++++++++++++++++ tests/test_serve.py | 1 + 11 files changed, 240 insertions(+), 11 deletions(-) create mode 100644 c3r/telemetry/ledger_anchor.py create mode 100644 tests/test_ledger_anchor.py diff --git a/.dockerignore b/.dockerignore index 8d1d309..351e208 100644 --- a/.dockerignore +++ b/.dockerignore @@ -10,3 +10,4 @@ data evidence docs assets + diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index e09f17a..463d22b 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -43,3 +43,4 @@ jobs: path: trivy-image.json if-no-files-found: ignore retention-days: 30 + diff --git a/Dockerfile b/Dockerfile index e2c1750..56396d8 100644 --- a/Dockerfile +++ b/Dockerfile @@ -7,3 +7,4 @@ COPY --chown=10001:10001 c3r ./c3r USER 10001:10001 EXPOSE 8080 CMD ["python", "-m", "c3r.serve"] + diff --git a/c3r/ingress_proxy.py b/c3r/ingress_proxy.py index 5228a24..dff9669 100644 --- a/c3r/ingress_proxy.py +++ b/c3r/ingress_proxy.py @@ -147,3 +147,4 @@ def do_POST(self) -> None: self._send_error(413, "request_size_out_of_bounds") return self._forward("POST", self.rfile.read(length)) + diff --git a/c3r/serve.py b/c3r/serve.py index 819a0a4..8fffeec 100644 --- a/c3r/serve.py +++ b/c3r/serve.py @@ -121,3 +121,4 @@ def request_stop(_signum: int, _frame: object) -> None: if __name__ == "__main__": main() + diff --git a/c3r/telemetry/governed_store.py b/c3r/telemetry/governed_store.py index db6c983..2bd3fc0 100644 --- a/c3r/telemetry/governed_store.py +++ b/c3r/telemetry/governed_store.py @@ -18,6 +18,7 @@ from threading import Lock from typing import Callable +from .ledger_anchor import LedgerHead, capture_head from .trace import DecisionTrace from .trace_ledger import LedgerRecord, _record_hash @@ -158,33 +159,57 @@ def records(self) -> tuple[LedgerRecord, ...]: ).fetchall() return tuple(LedgerRecord(*row) for row in rows) - def verify(self) -> bool: + def _verified_head(self) -> tuple[int, str] | None: with self._lock: checkpoint, last_collected_at = self._db.execute( "SELECT checkpoint_hash, last_collected_at FROM metadata WHERE id=1" ).fetchone() rows = self._db.execute( - "SELECT run_id, collected_at, previous_hash, record_hash, canonical_json " - "FROM records ORDER BY sequence" + "SELECT sequence, run_id, collected_at, previous_hash, " + "record_hash, canonical_json FROM records ORDER BY sequence" ).fetchall() - if rows and rows[-1][1] != last_collected_at: - return False + sequence_row = self._db.execute( + "SELECT seq FROM sqlite_sequence WHERE name='records'" + ).fetchone() + sequence = int(sequence_row[0]) if sequence_row is not None else 0 + if rows and rows[-1][2] != last_collected_at: + return None + if rows and rows[-1][0] != sequence: + return None previous = checkpoint - for run_id, collected_at, prior, digest, payload in rows: + for _, run_id, collected_at, prior, digest, payload in rows: if prior != previous: - return False + return None try: decoded = json.loads(payload) canonical = json.dumps(decoded, sort_keys=True, separators=(",", ":"), ensure_ascii=False, allow_nan=False) if decoded["trace"]["run_id"] != run_id or decoded["collected_at"] != collected_at: - return False + return None except (ValueError, TypeError, KeyError): - return False + return None if canonical != payload or _record_hash(prior, payload) != digest: - return False + return None previous = digest - return True + return sequence, previous + + def verify(self) -> bool: + return self._verified_head() is not None + + def snapshot_head(self, *, policy_version: str) -> LedgerHead: + """Capture a verified head for signing outside the trace-writer identity. + + The caller must send this to a separately controlled signer and durable + destination. Capturing a head locally is not independent anchoring. + """ + state = self._verified_head() + if state is None: + raise ValueError("governed trace hash chain is invalid") + sequence, record_hash = state + return capture_head( + sequence=sequence, record_hash=record_hash, + policy_version=policy_version, clock=self._clock, + ) def append(self, trace: DecisionTrace, *, source_id: str, task_id: str) -> LedgerRecord: grant = self._grants.get(source_id) diff --git a/c3r/telemetry/ledger_anchor.py b/c3r/telemetry/ledger_anchor.py new file mode 100644 index 0000000..2ac6296 --- /dev/null +++ b/c3r/telemetry/ledger_anchor.py @@ -0,0 +1,81 @@ +"""Portable signed ledger-head envelopes; deployment must supply independent signing. + +This module never holds a private key. A production signer and anchor destination +must be controlled separately from the trace writer (for example, Cloud KMS and +an access-separated evidence project). Local success is not deployment proof. +""" + +from __future__ import annotations + +import json +import re +from dataclasses import asdict, dataclass +from datetime import datetime, timezone +from typing import Callable + + +_HASH = re.compile(r"[0-9a-f]{64}\Z") + + +@dataclass(frozen=True, slots=True) +class LedgerHead: + sequence: int + record_hash: str + captured_at: str + policy_version: str + + def __post_init__(self) -> None: + if self.sequence < 0 or not _HASH.fullmatch(self.record_hash): + raise ValueError("invalid ledger head") + if not self.policy_version or len(self.policy_version) > 64: + raise ValueError("invalid policy version") + try: + instant = datetime.fromisoformat(self.captured_at) + except ValueError as error: + raise ValueError("invalid capture timestamp") from error + if instant.tzinfo is None or instant.utcoffset().total_seconds() != 0: + raise ValueError("capture timestamp must be UTC") + + def payload(self) -> bytes: + return json.dumps( + asdict(self), sort_keys=True, separators=(",", ":"), ensure_ascii=False + ).encode("utf-8") + + +@dataclass(frozen=True, slots=True) +class SignedLedgerAnchor: + head: LedgerHead + key_id: str + signature_hex: str + + +def capture_head( + *, sequence: int, record_hash: str, policy_version: str, + clock: Callable[[], datetime] = lambda: datetime.now(timezone.utc), +) -> LedgerHead: + return LedgerHead(sequence, record_hash, clock().isoformat(), policy_version) + + +def sign_head( + head: LedgerHead, *, key_id: str, signer: Callable[[bytes], bytes] +) -> SignedLedgerAnchor: + if not key_id or len(key_id) > 256: + raise ValueError("invalid signer key identifier") + signature = signer(head.payload()) + if not signature: + raise ValueError("empty signature") + return SignedLedgerAnchor(head, key_id, signature.hex()) + + +def verify_anchor( + anchor: SignedLedgerAnchor, *, sequence: int, record_hash: str, + verifier: Callable[[str, bytes, bytes], bool], +) -> bool: + if anchor.head.sequence != sequence or anchor.head.record_hash != record_hash: + return False + try: + signature = bytes.fromhex(anchor.signature_hex) + except ValueError: + return False + return bool(signature) and verifier(anchor.key_id, anchor.head.payload(), signature) + diff --git a/tests/test_governed_trace_store.py b/tests/test_governed_trace_store.py index c676ed6..8b956ba 100644 --- a/tests/test_governed_trace_store.py +++ b/tests/test_governed_trace_store.py @@ -1,3 +1,4 @@ +import hmac import sqlite3 import tempfile import unittest @@ -5,6 +6,7 @@ from pathlib import Path from c3r.telemetry.governed_store import BoundGovernedTraceSink, GovernedTraceStore, SourceGrant +from c3r.telemetry.ledger_anchor import sign_head, verify_anchor from c3r.telemetry.trace import DecisionTrace @@ -132,6 +134,49 @@ def test_overdue_purge_blocks_new_collection_until_purged(self): task_id="task_001") self.assertEqual(len(store.records()), 1) + def test_external_anchor_detects_clean_chain_tail_removal(self): + # The test key stands in for a separate signer; it is not deployment evidence. + key = b"fixture-only-signer" + sign = lambda payload: hmac.digest(key, payload, "sha256") + verify = lambda _key_id, payload, signature: hmac.compare_digest( + sign(payload), signature + ) + with self.store() as store: + store.append(trace(), source_id="c3r_internal_001", task_id="task_001") + store.append(trace(run_id="run_002"), source_id="c3r_internal_001", + task_id="task_001") + anchored = sign_head(store.snapshot_head(policy_version="fixture-v1"), + key_id="fixture-only", signer=sign) + + # A writer with DB access can erase a tail and make local replay valid. + # The independent prior signature must still reject the altered state. + db = sqlite3.connect(self.path) + try: + db.execute("DELETE FROM records WHERE sequence=2") + db.execute("UPDATE sqlite_sequence SET seq=1 WHERE name='records'") + db.execute("UPDATE metadata SET last_collected_at=(" + "SELECT collected_at FROM records WHERE sequence=1) WHERE id=1") + db.commit() + finally: + db.close() + with self.store() as store: + self.assertTrue(store.verify()) + current = store.snapshot_head(policy_version="fixture-v1") + self.assertFalse(verify_anchor(anchored, sequence=current.sequence, + record_hash=current.record_hash, + verifier=verify)) + + def test_head_snapshot_preserves_purged_prefix_checkpoint(self): + with self.store(now=NOW - timedelta(days=31)) as store: + first = store.append(trace(), source_id="c3r_internal_001", task_id="task_001") + before = store.snapshot_head(policy_version="fixture-v1") + with self.store(now=NOW) as store: + self.assertEqual(store.purge_expired(), 1) + after = store.snapshot_head(policy_version="fixture-v1") + self.assertEqual(after.sequence, before.sequence) + self.assertEqual(after.record_hash, first.record_hash) + self.assertEqual(store.records(), ()) + if __name__ == "__main__": unittest.main() diff --git a/tests/test_ingress_proxy.py b/tests/test_ingress_proxy.py index dd1e615..00c614f 100644 --- a/tests/test_ingress_proxy.py +++ b/tests/test_ingress_proxy.py @@ -147,3 +147,4 @@ def test_upstream_route_must_be_loopback_and_secrets_distinct(self): if __name__ == "__main__": unittest.main() + diff --git a/tests/test_ledger_anchor.py b/tests/test_ledger_anchor.py new file mode 100644 index 0000000..064f14c --- /dev/null +++ b/tests/test_ledger_anchor.py @@ -0,0 +1,71 @@ +import hmac +import unittest +from dataclasses import replace +from datetime import datetime, timezone + +from c3r.telemetry.ledger_anchor import capture_head, sign_head, verify_anchor +from c3r.telemetry.trace import DecisionTrace +from c3r.telemetry.trace_ledger import TraceLedger + + +def trace(run_id: str) -> DecisionTrace: + return DecisionTrace( + run_id=run_id, + state_hash="a" * 64, + access_level="internal", + model_provider="fixture", + candidate_ids=("STOP",), + probabilities={}, + utility_quantiles={"STOP": 0.0}, + selected_action_id="STOP", + authority_result="not-requested", + system_cost={"latency_ms": 1.0}, + task_outcome={"success": True}, + artifact_refs=(), + ) + + +class LedgerAnchorTests(unittest.TestCase): + def test_signed_head_matches_chain_and_detects_removal_or_alteration(self) -> None: + # Test-only signer. Deployment must use a key unavailable to the writer. + test_key = b"fixture-only-key" + sign = lambda payload: hmac.digest(test_key, payload, "sha256") + verify = lambda _key_id, payload, signature: hmac.compare_digest( + sign(payload), signature + ) + ledger = TraceLedger() + ledger.append(trace("one")) + ledger.append(trace("two")) + head = capture_head( + sequence=2, + record_hash=ledger.records[-1].record_hash, + policy_version="fixture-v1", + clock=lambda: datetime(2026, 9, 23, tzinfo=timezone.utc), + ) + anchor = sign_head(head, key_id="test-only", signer=sign) + + self.assertTrue(TraceLedger.verify(ledger.records)) + self.assertTrue(verify_anchor(anchor, sequence=2, + record_hash=ledger.records[-1].record_hash, + verifier=verify)) + self.assertFalse(verify_anchor(anchor, sequence=1, + record_hash=ledger.records[0].record_hash, + verifier=verify)) + self.assertFalse(verify_anchor(anchor, sequence=2, + record_hash="f" * 64, verifier=verify)) + self.assertFalse(verify_anchor(replace(anchor, signature_hex="00"), sequence=2, + record_hash=head.record_hash, verifier=verify)) + + def test_rejects_invalid_head_and_empty_signature(self) -> None: + with self.assertRaises(ValueError): + capture_head(sequence=-1, record_hash="0" * 64, policy_version="v1") + with self.assertRaises(ValueError): + capture_head(sequence=0, record_hash="not-a-hash", policy_version="v1") + head = capture_head(sequence=0, record_hash="0" * 64, policy_version="v1") + with self.assertRaises(ValueError): + sign_head(head, key_id="test-only", signer=lambda _payload: b"") + + +if __name__ == "__main__": + unittest.main() + diff --git a/tests/test_serve.py b/tests/test_serve.py index e76b3f8..1be9976 100644 --- a/tests/test_serve.py +++ b/tests/test_serve.py @@ -89,3 +89,4 @@ def test_composed_health_path(self): if __name__ == "__main__": unittest.main() + From 37f2b740c1fac6407df45f4959d0fdf1537da42d Mon Sep 17 00:00:00 2001 From: wilkont Date: Thu, 24 Sep 2026 14:37:00 -0500 Subject: [PATCH 42/55] Bind retention purge target to dedicated Cloud Run project --- c3r/retention_job.py | 74 +++++++++++++++++++++++++++++++++----------- 1 file changed, 56 insertions(+), 18 deletions(-) diff --git a/c3r/retention_job.py b/c3r/retention_job.py index c5e3f8a..2f87f59 100644 --- a/c3r/retention_job.py +++ b/c3r/retention_job.py @@ -1,12 +1,16 @@ """Bounded deletion check for the dedicated private C3R trace bucket. -This is a retention backstop, not permission to collect traces. The bucket name, -cutoff, and API host are pinned so a deployment argument cannot target other data. +This is a retention backstop, not permission to collect traces. The deployment +must supply a narrow dedicated-project bucket target. Bucket-scoped IAM is still +required: name validation is defense in depth, not an authorization boundary. """ from __future__ import annotations import json +import os +import re +from collections.abc import Mapping from dataclasses import dataclass from datetime import datetime, timedelta, timezone from typing import Protocol @@ -14,9 +18,11 @@ from urllib.request import Request, urlopen -BUCKET = "colomboai-c3r-private-traces-795563500003" DELETE_AFTER_DAYS = 28 -_STORAGE_API = "https://storage.googleapis.com/storage/v1/b/" + BUCKET + "/o" +_BUCKET = re.compile(r"colomboai-c3r-staging-traces-([0-9]{12})\Z") +_METADATA_PROJECT_NUMBER = ( + "http://metadata.google.internal/computeMetadata/v1/project/numeric-project-id" +) _METADATA_TOKEN = ( "http://metadata.google.internal/computeMetadata/v1/instance/" "service-accounts/default/token" @@ -31,9 +37,28 @@ class StoredObject: class ObjectClient(Protocol): + bucket: str + def list_all(self) -> tuple[StoredObject, ...]: ... - def delete_if_generation(self, obj: StoredObject) -> None: ... + def delete_generation(self, obj: StoredObject) -> None: ... + + +def _validated_bucket(bucket: str) -> str: + if not _BUCKET.fullmatch(bucket): + raise ValueError("C3R_TRACE_BUCKET must name the dedicated C3R staging trace bucket") + return bucket + + +def bucket_from_environment(values: Mapping[str, str]) -> str: + return _validated_bucket(values.get("C3R_TRACE_BUCKET", "")) + + +def _verify_runtime_project(bucket: str, project_number: str) -> None: + match = _BUCKET.fullmatch(_validated_bucket(bucket)) + assert match is not None + if match.group(1) != project_number: + raise ValueError("C3R_TRACE_BUCKET does not match the Cloud Run project number") def _created_at(raw: str) -> datetime: @@ -44,20 +69,21 @@ def _created_at(raw: str) -> datetime: def purge(client: ObjectClient, *, now: datetime) -> dict[str, object]: + bucket = _validated_bucket(client.bucket) if now.tzinfo is None or now.utcoffset() != timedelta(0): raise ValueError("purge clock must be UTC") cutoff = now - timedelta(days=DELETE_AFTER_DAYS) before = client.list_all() - if len(before) > 100_000 or len({item.name for item in before}) != len(before): - raise ValueError("bucket inventory is too large or contains duplicate names") + if len(before) > 100_000 or len({(item.name, item.generation) for item in before}) != len(before): + raise ValueError("bucket inventory is too large or contains duplicate generations") expired = tuple(item for item in before if item.created_at <= cutoff) for item in expired: - client.delete_if_generation(item) + client.delete_generation(item) remaining = client.list_all() if any(item.created_at <= cutoff for item in remaining): raise RuntimeError("expired objects remain after purge; collection must stay disabled") return { - "bucket": BUCKET, + "bucket": bucket, "cutoff_utc": cutoff.isoformat(), "objects_before": len(before), "expired_candidates": len(expired), @@ -68,9 +94,16 @@ def purge(client: ObjectClient, *, now: datetime) -> dict[str, object]: class GcsJsonClient: - """Cloud Run service-identity transport for one pinned GCS bucket.""" - - def __init__(self) -> None: + """Cloud Run service-identity transport for one validated GCS bucket.""" + + def __init__(self, bucket: str) -> None: + self.bucket = _validated_bucket(bucket) + project_request = Request(_METADATA_PROJECT_NUMBER, + headers={"Metadata-Flavor": "Google"}) + with urlopen(project_request, timeout=10) as response: + project_number = response.read(64).decode("ascii") + _verify_runtime_project(bucket, project_number) + self._storage_api = "https://storage.googleapis.com/storage/v1/b/" + bucket + "/o" request = Request(_METADATA_TOKEN, headers={"Metadata-Flavor": "Google"}) with urlopen(request, timeout=10) as response: data = json.load(response) @@ -94,10 +127,11 @@ def list_all(self) -> tuple[StoredObject, ...]: page_token: str | None = None seen_tokens: set[str] = set() while True: - query = {"maxResults": "1000", "fields": "items(name,generation,timeCreated),nextPageToken"} + query = {"maxResults": "1000", "versions": "true", + "fields": "items(name,generation,timeCreated),nextPageToken"} if page_token is not None: query["pageToken"] = page_token - data = self._request("GET", _STORAGE_API + "?" + urlencode(query)) + data = self._request("GET", self._storage_api + "?" + urlencode(query)) assert data is not None items = data.get("items", []) if not isinstance(items, list): @@ -123,19 +157,23 @@ def list_all(self) -> tuple[StoredObject, ...]: seen_tokens.add(next_token) page_token = next_token - def delete_if_generation(self, obj: StoredObject) -> None: + def delete_generation(self, obj: StoredObject) -> None: if not obj.name or obj.generation <= 0: raise ValueError("invalid deletion target") - url = _STORAGE_API + "/" + quote(obj.name, safe="") + "?" + urlencode( - {"ifGenerationMatch": str(obj.generation)} + # GCS identifies a version by (name, generation); including generation + # permanently targets that exact version, including a noncurrent one. + url = self._storage_api + "/" + quote(obj.name, safe="") + "?" + urlencode( + {"generation": str(obj.generation)} ) self._request("DELETE", url) def main() -> None: - result = purge(GcsJsonClient(), now=datetime.now(timezone.utc)) + bucket = bucket_from_environment(os.environ) + result = purge(GcsJsonClient(bucket), now=datetime.now(timezone.utc)) print(json.dumps(result, sort_keys=True)) if __name__ == "__main__": main() + From a7fb2987637f96a3ac6400b8c966b77d4c1193dc Mon Sep 17 00:00:00 2001 From: wilkont Date: Thu, 24 Sep 2026 14:37:12 -0500 Subject: [PATCH 43/55] Test Cloud Run metadata-bound retention target and version deletion --- tests/test_retention_job.py | 80 +++++++++++++++++++++++++++++++++++-- 1 file changed, 77 insertions(+), 3 deletions(-) diff --git a/tests/test_retention_job.py b/tests/test_retention_job.py index 649e235..5eb817e 100644 --- a/tests/test_retention_job.py +++ b/tests/test_retention_job.py @@ -1,21 +1,28 @@ +import io import unittest from datetime import datetime, timedelta, timezone +from unittest.mock import patch -from c3r.retention_job import DELETE_AFTER_DAYS, StoredObject, _created_at, purge +from c3r.retention_job import ( + DELETE_AFTER_DAYS, GcsJsonClient, StoredObject, _created_at, + _verify_runtime_project, bucket_from_environment, purge, +) NOW = datetime(2026, 9, 23, 16, 0, tzinfo=timezone.utc) +BUCKET = "colomboai-c3r-staging-traces-123456789012" class FakeClient: def __init__(self, objects): + self.bucket = BUCKET self.objects = list(objects) self.deleted = [] def list_all(self): return tuple(self.objects) - def delete_if_generation(self, obj): + def delete_generation(self, obj): self.deleted.append((obj.name, obj.generation)) self.objects = [item for item in self.objects if item != obj] @@ -37,7 +44,7 @@ def test_surviving_expired_object_fails_closed(self): old = StoredObject("traces/old", 3, NOW - timedelta(days=29)) class NonDeletingClient(FakeClient): - def delete_if_generation(self, obj): + def delete_generation(self, obj): self.deleted.append((obj.name, obj.generation)) with self.assertRaisesRegex(RuntimeError, "expired objects remain"): @@ -50,11 +57,78 @@ def test_duplicate_inventory_and_non_utc_clock_rejected(self): with self.assertRaisesRegex(ValueError, "UTC"): purge(FakeClient([]), now=NOW.replace(tzinfo=None)) + def test_distinct_generations_are_purged_without_treating_them_as_duplicates(self): + old = StoredObject("traces/replaced", 3, NOW - timedelta(days=29)) + fresh = StoredObject("traces/replaced", 4, NOW - timedelta(days=1)) + client = FakeClient([old, fresh]) + + report = purge(client, now=NOW) + + self.assertEqual(client.deleted, [(old.name, old.generation)]) + self.assertEqual(client.objects, [fresh]) + self.assertEqual(report["expired_remaining"], 0) + + def test_gcs_transport_lists_versions_and_deletes_exact_generation(self): + client = object.__new__(GcsJsonClient) + client.bucket = BUCKET + client._storage_api = "https://storage.googleapis.com/storage/v1/b/" + BUCKET + "/o" + requests = [] + + def fake_request(method, url): + requests.append((method, url)) + if method == "GET": + return {"items": [ + {"name": "traces/replaced", "generation": "3", "timeCreated": "2026-08-01T00:00:00Z"}, + {"name": "traces/replaced", "generation": "4", "timeCreated": "2026-09-23T00:00:00Z"}, + ]} + return None + + client._request = fake_request + objects = client.list_all() + client.delete_generation(objects[0]) + + self.assertEqual([obj.generation for obj in objects], [3, 4]) + self.assertIn("versions=true", requests[0][1]) + self.assertIn("generation=3", requests[1][1]) + self.assertNotIn("ifGenerationMatch", requests[1][1]) + def test_creation_timestamp_must_be_timezone_aware(self): self.assertEqual(_created_at("2026-09-23T16:00:00Z"), NOW) with self.assertRaisesRegex(ValueError, "UTC offset"): _created_at("2026-09-23T16:00:00") + def test_dedicated_bucket_must_be_explicit_and_narrow(self): + self.assertEqual(bucket_from_environment({"C3R_TRACE_BUCKET": BUCKET}), BUCKET) + for value in ("", "colomboai-c3r-private-traces-123456789012", + "unrelated-bucket", "colomboai-c3r-staging-traces-123456789012/other"): + with self.subTest(value=value), self.assertRaisesRegex(ValueError, "C3R_TRACE_BUCKET"): + bucket_from_environment({"C3R_TRACE_BUCKET": value}) + client = FakeClient([]) + client.bucket = "unrelated-bucket" + with self.assertRaisesRegex(ValueError, "C3R_TRACE_BUCKET"): + purge(client, now=NOW) + + def test_runtime_project_must_match_bucket_suffix(self): + _verify_runtime_project(BUCKET, "123456789012") + with self.assertRaisesRegex(ValueError, "project number"): + _verify_runtime_project(BUCKET, "999999999999") + + def test_gcs_client_rejects_wrong_project_before_credential_request(self): + with patch("c3r.retention_job.urlopen", return_value=io.BytesIO(b"999999999999")) as open_url: + with self.assertRaisesRegex(ValueError, "project number"): + GcsJsonClient(BUCKET) + self.assertEqual(open_url.call_count, 1) + + def test_gcs_client_accepts_metadata_bound_bucket(self): + token = b'{"access_token":"' + b"x" * 24 + b'"}' + with patch("c3r.retention_job.urlopen", side_effect=[ + io.BytesIO(b"123456789012"), io.BytesIO(token), + ]) as open_url: + client = GcsJsonClient(BUCKET) + self.assertEqual(client.bucket, BUCKET) + self.assertEqual(open_url.call_count, 2) + if __name__ == "__main__": unittest.main() + From 7b62ff2aeff2e4571c860f071d6d953ec0f07b20 Mon Sep 17 00:00:00 2001 From: wilkont Date: Thu, 24 Sep 2026 14:47:54 -0500 Subject: [PATCH 44/55] Add dedicated retention image build recipe --- Dockerfile.retention | 9 +++++++++ 1 file changed, 9 insertions(+) create mode 100644 Dockerfile.retention diff --git a/Dockerfile.retention b/Dockerfile.retention new file mode 100644 index 0000000..9146807 --- /dev/null +++ b/Dockerfile.retention @@ -0,0 +1,9 @@ +FROM python:3.11-alpine3.24@sha256:cd04730b8511def3fbf14204d66a0c1536f290b8e896ed5a94cd64cb15ac1356 + +ENV PYTHONDONTWRITEBYTECODE=1 PYTHONUNBUFFERED=1 +WORKDIR /app +RUN python -m pip uninstall -y setuptools wheel jaraco.context +COPY --chown=10001:10001 c3r ./c3r +USER 10001:10001 +CMD ["python", "-m", "c3r.retention_job"] + From 2f4a73b1e119edeb3a4b3dca8882b7cb08c60019 Mon Sep 17 00:00:00 2001 From: wilkont Date: Thu, 24 Sep 2026 14:48:26 -0500 Subject: [PATCH 45/55] Build and scan retention image in CI --- .github/workflows/ci.yml | 26 +++++++++++++++++++++++++- 1 file changed, 25 insertions(+), 1 deletion(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 463d22b..65b57b3 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -43,4 +43,28 @@ jobs: path: trivy-image.json if-no-files-found: ignore retention-days: 30 - + + retention-image-build: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - run: docker build --pull --file Dockerfile.retention --tag c3r-retention:ci . + - run: docker run --rm c3r-retention:ci python -c 'import c3r.retention_job' + - name: Scan retention image for high and critical vulnerabilities + uses: aquasecurity/trivy-action@v0.36.0 + with: + version: v0.74.0 + image-ref: c3r-retention:ci + format: json + output: trivy-retention-image.json + severity: HIGH,CRITICAL + exit-code: '1' + - name: Preserve retention scan findings + if: always() + uses: actions/upload-artifact@v4 + with: + name: trivy-retention-image-${{ github.run_id }} + path: trivy-retention-image.json + if-no-files-found: ignore + retention-days: 30 + From 3befd38a2d5304fa20fd751fce912645e8ea8d49 Mon Sep 17 00:00:00 2001 From: wilkont Date: Tue, 29 Sep 2026 10:47:47 -0500 Subject: [PATCH 46/55] Fail closed on standalone effect execution --- c3r/runtime.py | 47 +++++++++++---------------- c3r/staging_host.py | 8 +---- scripts/run_controlled_pairs.py | 17 ++-------- tests/test_http_service.py | 6 ++-- tests/test_runtime.py | 56 ++++++++++++++------------------- tests/test_serve.py | 5 ++- 6 files changed, 54 insertions(+), 85 deletions(-) diff --git a/c3r/runtime.py b/c3r/runtime.py index f9e4f0c..ecc1fb1 100644 --- a/c3r/runtime.py +++ b/c3r/runtime.py @@ -15,7 +15,6 @@ from .adapters.providers import ProviderExecutionResult from .candidate_compiler import CandidateCompiler -from .commit_gateway import ApprovalGrant, CommitRequest, TrustedCommitGateway from .cvoc import RobustCvocController from .feature_flags import FeatureFlags from .state_compiler import StateCompiler @@ -26,6 +25,7 @@ AuthorityPolicy, CompiledState, RawState, + RiskClass, ValueEstimate, ) from .system_one.fast_path import FastPathDecision, LayaFastPath @@ -66,11 +66,12 @@ class RuntimeOutcome: class StandaloneController: - """Run C3R with independent verification and an optional trusted executor. + """Run C3R with independent verification and recommendation-only outcomes. - Estimates, the policy, verifier, gateway, and executor must be supplied by the - trusted host. The default is recommendation only. A learned proposal can never - supply an executor or a verification attestation through this interface. + Estimates, the policy, and verifier must be supplied by the trusted host. + Supplying an executor is rejected: arbitrary external effects cannot be + atomically committed with the trace ledger. A learned proposal cannot grant + authority. """ def __init__( @@ -81,30 +82,28 @@ def __init__( candidates: CandidateCompiler, cvoc: RobustCvocController, verifier: VerifierFirewall, - gateway: TrustedCommitGateway, ledger: TraceSink, fast_path: LayaFastPath | None = None, deliberator: Deliberator | None = None, executor: Callable[[ActionCandidate], None] | None = None, ) -> None: + if executor is not None: + raise ValueError("external effects are unsupported by StandaloneController") self._flags = flags self._compiler = compiler self._candidates = candidates self._cvoc = cvoc self._verifier = verifier - self._gateway = gateway self._ledger = ledger self._fast_path = fast_path self._deliberator = deliberator - self._executor = executor @property def effect_execution_enabled(self) -> bool: - return self._executor is not None + """Capability flag retained for fail-closed hosting checks.""" + return False - def run( - self, request: RuntimeRequest, *, approval: ApprovalGrant | None = None - ) -> RuntimeOutcome: + def run(self, request: RuntimeRequest) -> RuntimeOutcome: if not request.run_id: raise ValueError("run_id is required") state_hash = hashlib.sha256( @@ -226,31 +225,21 @@ def finish( return finish( "deterministic", "VERIFICATION_REJECTED", candidate_ids=candidate_ids, fast=fast ) - if self._executor is None: + if selected.risk_class is not RiskClass.READ_ONLY: return finish( - "recommendation", - "VERIFIED_RECOMMENDATION", - selected=selected, + "deterministic", + "EFFECT_EXECUTION_UNAVAILABLE", candidate_ids=candidate_ids, lower_bound=decision.lower_bound, - authority_result="verified_not_committed", fast=fast, ) - try: - commit = self._gateway.commit( - CommitRequest(selected, verification, approval), self._executor - ) - except (OSError, RuntimeError, TypeError, ValueError): - return finish( - "deterministic", "COMMIT_FAILURE", candidate_ids=candidate_ids, fast=fast - ) return finish( - "commit" if commit.committed else "deterministic", - commit.reason, - selected=selected if commit.committed else None, + "recommendation", + "VERIFIED_RECOMMENDATION", + selected=selected, candidate_ids=candidate_ids, lower_bound=decision.lower_bound, - authority_result=commit.reason, + authority_result="verified_not_committed", fast=fast, ) diff --git a/c3r/staging_host.py b/c3r/staging_host.py index db8ce87..40da96d 100644 --- a/c3r/staging_host.py +++ b/c3r/staging_host.py @@ -9,7 +9,6 @@ from secrets import token_bytes from .candidate_compiler import CandidateCompiler -from .commit_gateway import InMemoryApprovalNonceStore, TrustedCommitGateway from .cvoc import RobustCvocController from .feature_flags import FeatureFlags from .host_factory import ReadOnlyRequestFactory @@ -37,15 +36,10 @@ def build() -> tuple[StandaloneController, ReadOnlyRequestFactory]: {"deny": lambda _candidate: VerifierDecision(False, "staging disabled")}, VerifierPolicy(default_verifier="deny"), attestation_key=key, ) - gateway = TrustedCommitGateway( - trusted_verifier_ids=frozenset({"deny"}), verification_key=key, - approval_key=token_bytes(32), policy_version="staging-disabled-v1", - approval_nonce_store=InMemoryApprovalNonceStore(), - ) controller = StandaloneController( flags=FeatureFlags(enabled_requested=False), compiler=StateCompiler(), candidates=CandidateCompiler(), - cvoc=RobustCvocController(), verifier=verifier, gateway=gateway, + cvoc=RobustCvocController(), verifier=verifier, ledger=EphemeralStagingSink(), ) definition = ActionDefinition( diff --git a/scripts/run_controlled_pairs.py b/scripts/run_controlled_pairs.py index 24c143a..0ff8068 100644 --- a/scripts/run_controlled_pairs.py +++ b/scripts/run_controlled_pairs.py @@ -20,13 +20,11 @@ sys.path.insert(0, str(REPOSITORY_ROOT)) from c3r.candidate_compiler import CandidateCompiler -from c3r.commit_gateway import InMemoryApprovalNonceStore, TrustedCommitGateway from c3r.cvoc import RobustCvocController from c3r.feature_flags import FeatureFlags from c3r.runtime import RuntimeRequest, StandaloneController from c3r.state_compiler import StateCompiler from c3r.state_schema import ( - ActionCandidate, ActionDefinition, ActionFamily, AuthorityPolicy, @@ -87,30 +85,21 @@ def _request(case: ControlledCase, arm: str) -> RuntimeRequest: def _run(case: ControlledCase, arm: str) -> dict[str, object]: - effects: list[ActionCandidate] = [] key = b"controlled-verifier-test-key" verifier = VerifierFirewall( {"policy": lambda _: VerifierDecision(case.verifier_accepts, "fixture policy")}, VerifierPolicy(default_verifier="policy"), attestation_key=key, ) - gateway = TrustedCommitGateway( - trusted_verifier_ids=frozenset({"policy"}), verification_key=key, - approval_key=b"controlled-approval-test-key", policy_version="controlled-v1", - approval_nonce_store=InMemoryApprovalNonceStore(), - ) ledger = TraceLedger() controller = StandaloneController( flags=FeatureFlags(enabled_requested=arm == "c3r"), compiler=StateCompiler(), candidates=CandidateCompiler(), - cvoc=RobustCvocController(), verifier=verifier, gateway=gateway, + cvoc=RobustCvocController(), verifier=verifier, ledger=ledger, - executor=effects.append if case.risk is RiskClass.EXTERNAL_WRITE else None, ) start = perf_counter() outcome = controller.run(_request(case, arm)) latency_ms = (perf_counter() - start) * 1000 - if effects: - raise AssertionError("controlled replay must not execute an action") trace = json.loads(outcome.ledger_record.canonical_json) success = outcome.selected_action_id == case.expected_action_id if case.expected_action_id is not None: @@ -124,7 +113,7 @@ def _run(case: ControlledCase, arm: str) -> dict[str, object]: "outcome_label_ref": f"rubric:c3r-controlled-v1:{case.task_id}", "latency_ms": latency_ms, "cost_usd": 0.0, - "authority_bypass": bool(effects), + "authority_bypass": False, # effects are structurally unavailable in this controller "trace_hash": outcome.ledger_record.record_hash, } @@ -144,7 +133,7 @@ def collect() -> tuple[list[dict[str, object]], dict[str, object]]: def main() -> None: - output = Path(__file__).resolve().parents[1] / "evidence" / "controlled-pairs-v1" + output = Path(__file__).resolve().parents[1] / "evidence" / "controlled-pairs-v2" output.mkdir(parents=True, exist_ok=True) observations, manifest = collect() observations_bytes = ( diff --git a/tests/test_http_service.py b/tests/test_http_service.py index 9bcbe4e..5a5d89d 100644 --- a/tests/test_http_service.py +++ b/tests/test_http_service.py @@ -89,10 +89,12 @@ def test_non_loopback_bind_is_rejected(self) -> None: ) def test_effect_enabled_runtime_is_rejected(self) -> None: - runtime, _ = controller(executor=lambda _: None) + class EffectCapableRuntime: + effect_execution_enabled = True + with self.assertRaisesRegex(ValueError, "external effects"): C3RHTTPServer( - runtime=runtime, + runtime=EffectCapableRuntime(), request_factory=HostFactory(), bearer_token=TOKEN, port=0, diff --git a/tests/test_runtime.py b/tests/test_runtime.py index 8aec3f5..65729ca 100644 --- a/tests/test_runtime.py +++ b/tests/test_runtime.py @@ -3,7 +3,6 @@ from c3r.adapters.providers import ProviderExecutionResult from c3r.candidate_compiler import CandidateCompiler -from c3r.commit_gateway import InMemoryApprovalNonceStore, TrustedCommitGateway from c3r.cvoc import RobustCvocController from c3r.deliberative.envelope import DeliberativeResult from c3r.feature_flags import FeatureFlags @@ -81,13 +80,6 @@ def controller( VerifierPolicy(default_verifier="policy"), attestation_key=KEY, ) - gateway = TrustedCommitGateway( - trusted_verifier_ids=frozenset({"policy"}), - verification_key=KEY, - approval_key=b"approval-test-key", - policy_version="policy-v1", - approval_nonce_store=InMemoryApprovalNonceStore(), - ) runtime = StandaloneController( flags=FeatureFlags( enabled_requested=enabled, @@ -98,7 +90,6 @@ def controller( candidates=CandidateCompiler(), cvoc=RobustCvocController(), verifier=verifier, - gateway=gateway, ledger=ledger, executor=executor, fast_path=fast_path, @@ -108,6 +99,13 @@ def controller( class RuntimeTests(unittest.TestCase): + def test_executor_configuration_is_rejected_before_any_effect(self) -> None: + effects = [] + with self.assertRaisesRegex(ValueError, "external effects"): + controller(executor=effects.append) + + self.assertEqual(effects, []) + def test_verified_recommendation_has_no_effect_and_is_traced(self) -> None: runtime, ledger = controller() outcome = runtime.run(request()) @@ -120,28 +118,25 @@ def test_verified_recommendation_has_no_effect_and_is_traced(self) -> None: self.assertNotIn("The record exists", ledger.to_jsonl()) def test_global_disable_stops_before_candidate_or_provider_execution(self) -> None: - effects = [] - runtime, _ = controller(enabled=False, executor=effects.append) + runtime, _ = controller(enabled=False) outcome = runtime.run(request()) self.assertEqual(outcome.reason, "C3R_DISABLED") - self.assertEqual(effects, []) + self.assertFalse(runtime.effect_execution_enabled) def test_missing_provenance_stops_before_any_effect(self) -> None: - effects = [] - runtime, _ = controller(executor=effects.append) + runtime, _ = controller() outcome = runtime.run(request(provenance=False)) self.assertEqual(outcome.reason, "STATE_UNSAFE_TO_COMPRESS") - self.assertEqual(effects, []) + self.assertFalse(runtime.effect_execution_enabled) def test_verifier_rejection_stops_before_commit(self) -> None: - effects = [] - runtime, _ = controller(accepted=False, executor=effects.append) + runtime, _ = controller(accepted=False) outcome = runtime.run(request()) self.assertEqual(outcome.reason, "VERIFICATION_REJECTED") - self.assertEqual(effects, []) + self.assertFalse(runtime.effect_execution_enabled) def test_unavailable_action_family_is_never_compiled(self) -> None: runtime, _ = controller() @@ -153,13 +148,14 @@ def test_unavailable_action_family_is_never_compiled(self) -> None: self.assertEqual(outcome.reason, "NO_SAFE_ACTION") self.assertIsNone(outcome.selected_action_id) - def test_external_write_requires_independent_approval(self) -> None: - effects = [] - runtime, _ = controller(executor=effects.append) + def test_external_write_is_unavailable_to_recommendation_only_controller(self) -> None: + runtime, ledger = controller() outcome = runtime.run(request(risk=RiskClass.EXTERNAL_WRITE)) - self.assertEqual(outcome.reason, "approval required") - self.assertEqual(effects, []) + self.assertEqual(outcome.reason, "EFFECT_EXECUTION_UNAVAILABLE") + self.assertEqual(outcome.route, "deterministic") + self.assertIsNone(outcome.selected_action_id) + self.assertTrue(TraceLedger.verify(ledger.records)) def test_uncalibrated_system_one_abstains_into_non_authoritative_deliberation(self) -> None: adapter = LayaAdapter( @@ -176,20 +172,18 @@ class Deliberator: def deliberate(self, _state): return {"plan": ["inspect"]} - effects = [] runtime, _ = controller( system_one=True, deliberative=True, fast_path=fast_path, deliberator=Deliberator(), - executor=effects.append, ) outcome = runtime.run(request()) self.assertEqual(outcome.route, "deliberative") self.assertEqual(outcome.reason, "SYSTEM_ONE_ABSTAINED") self.assertIsNone(outcome.selected_action_id) - self.assertEqual(effects, []) + self.assertFalse(runtime.effect_execution_enabled) def test_provider_usage_is_recorded_without_granting_authority(self) -> None: class Deliberator: @@ -232,8 +226,7 @@ def deliberate(self, _state): {"latency_ms": 1.0}, "untrusted-model", "fixture", ) - effects = [] - runtime, _ = controller(deliberative=True, deliberator=Deliberator(), executor=effects.append) + runtime, _ = controller(deliberative=True, deliberator=Deliberator()) base = request() definition = ActionDefinition( "reason", ActionFamily.DELIBERATE, "model", "plan", RiskClass.READ_ONLY, @@ -250,7 +243,7 @@ def deliberate(self, _state): outcome = runtime.run(req) self.assertEqual(outcome.route, "deliberative") - self.assertEqual(effects, []) + self.assertFalse(runtime.effect_execution_enabled) self.assertEqual(outcome.deliberation.requested_actions, ("delete all records",)) def test_provider_outage_falls_back_without_effect(self) -> None: @@ -258,8 +251,7 @@ class Deliberator: def deliberate(self, _state): raise OSError("provider unavailable") - effects = [] - runtime, _ = controller(deliberative=True, deliberator=Deliberator(), executor=effects.append) + runtime, _ = controller(deliberative=True, deliberator=Deliberator()) base = request() definition = ActionDefinition( "reason", ActionFamily.DELIBERATE, "model", "plan", RiskClass.READ_ONLY, @@ -277,7 +269,7 @@ def deliberate(self, _state): self.assertEqual(outcome.reason, "DELIBERATIVE_FAILURE") self.assertEqual(outcome.route, "deterministic") - self.assertEqual(effects, []) + self.assertFalse(runtime.effect_execution_enabled) if __name__ == "__main__": diff --git a/tests/test_serve.py b/tests/test_serve.py index 1be9976..17b0f60 100644 --- a/tests/test_serve.py +++ b/tests/test_serve.py @@ -50,11 +50,14 @@ def test_invalid_port_and_equal_tokens_fail(self): build_servers(values, builder_loader=lambda _: lambda: (controller()[0], HostFactory())) def test_effect_enabled_host_is_rejected(self): + class EffectCapableRuntime: + effect_execution_enabled = True + with self.assertRaisesRegex(ValueError, "recommendation-only"): build_servers( config(), builder_loader=lambda _: lambda: ( - controller(executor=lambda _: None)[0], HostFactory() + EffectCapableRuntime(), HostFactory() ), ) From 3e859bbd9ee0fce91f9465f2cce5ea2c7c34f4e2 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 30 Sep 2026 14:26:28 -0500 Subject: [PATCH 47/55] Add scoped stateless C3R API with guarded CLM System-One --- .gcloudignore | 2 + .gcloudignore.verify | 16 ++++ .github/workflows/ci.yml | 4 +- NOTICE | 5 + README.md | 73 +++++++++++--- c3r/feature_flags.py | 4 + c3r/http_service.py | 55 ++++++++++- c3r/ingress_proxy.py | 9 +- c3r/runtime.py | 54 +++++++++-- c3r/serve.py | 14 +++ c3r/staging_host.py | 11 +-- c3r/system_one/__init__.py | 9 +- c3r/system_one/clm_adapter.py | 168 +++++++++++++++++++++++++++++++++ c3r/system_one/factory.py | 36 +++++++ c3r/system_one/fast_path.py | 48 +++++++++- c3r/system_one/laya_adapter.py | 12 +++ c3r/telemetry/ephemeral.py | 15 +++ cloudbuild.retention.yaml | 18 ++++ cloudbuild.verify.yaml | 42 +++++++++ docs/release-policy.md | 25 +++++ docs/stateless-core-api.md | 82 ++++++++++++++++ tests/test_clm_adapter.py | 113 ++++++++++++++++++++++ tests/test_ingress_proxy.py | 9 ++ tests/test_runtime.py | 66 ++++++++++++- tests/test_serve.py | 40 ++++++++ tests/test_stateless_api.py | 88 +++++++++++++++++ third_party/CLM_VERSION | 8 ++ 27 files changed, 984 insertions(+), 42 deletions(-) create mode 100644 .gcloudignore.verify create mode 100644 c3r/system_one/clm_adapter.py create mode 100644 c3r/system_one/factory.py create mode 100644 c3r/telemetry/ephemeral.py create mode 100644 cloudbuild.retention.yaml create mode 100644 cloudbuild.verify.yaml create mode 100644 docs/release-policy.md create mode 100644 docs/stateless-core-api.md create mode 100644 tests/test_clm_adapter.py create mode 100644 tests/test_stateless_api.py create mode 100644 third_party/CLM_VERSION diff --git a/.gcloudignore b/.gcloudignore index 314a848..f5b0533 100644 --- a/.gcloudignore +++ b/.gcloudignore @@ -2,6 +2,8 @@ # local evidence, secrets, or repository metadata to Cloud Build. ** !Dockerfile +!Dockerfile.retention +!cloudbuild.retention.yaml !.dockerignore !c3r/ !c3r/** diff --git a/.gcloudignore.verify b/.gcloudignore.verify new file mode 100644 index 0000000..45e37b9 --- /dev/null +++ b/.gcloudignore.verify @@ -0,0 +1,16 @@ +# Separate verification upload: source and synthetic tests only. +# Never upload local evidence, credentials, trace rows, papers, or Git metadata. +** +!Dockerfile +!Dockerfile.retention +!cloudbuild.verify.yaml +!.dockerignore +!c3r/ +!c3r/** +!tests/ +!tests/** +!scripts/ +!scripts/run_controlled_pairs.py +!scripts/run_trace_control_dry_run.py +**/__pycache__/ +**/*.pyc diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 65b57b3..c10d5e7 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -49,6 +49,9 @@ jobs: steps: - uses: actions/checkout@v4 - run: docker build --pull --file Dockerfile.retention --tag c3r-retention:ci . + - name: Verify retention entrypoint + run: | + test "$(docker image inspect c3r-retention:ci --format '{{json .Config.Cmd}}')" = '["python","-m","c3r.retention_job"]' - run: docker run --rm c3r-retention:ci python -c 'import c3r.retention_job' - name: Scan retention image for high and critical vulnerabilities uses: aquasecurity/trivy-action@v0.36.0 @@ -67,4 +70,3 @@ jobs: path: trivy-retention-image.json if-no-files-found: ignore retention-days: 30 - diff --git a/NOTICE b/NOTICE index 159169e..a6c84f2 100644 --- a/NOTICE +++ b/NOTICE @@ -13,3 +13,8 @@ Upstream model: https://huggingface.co/convaiinnovations/laya Upstream source: https://github.com/NandhaKishorM/laya No derived Laya weights are included in this repository at this time. + +The optional CLM service adapter targets Contrastive-LM/CLM at upstream source +commit bb42c6c5bf914fd449bed2f6ca65be80602cb1f7. Upstream CLM source is +Apache-2.0: https://github.com/Contrastive-LM/CLM. No CLM source, encoder +weights, or projection-head weights are redistributed in this repository. diff --git a/README.md b/README.md index 7a85970..2f87f44 100644 --- a/README.md +++ b/README.md @@ -15,25 +15,41 @@ C3R is an open-core control plane that decides **which computation is worth perf It evaluates tools, retrieval, local and frontier models, verification, placement, and stopping as typed candidates under one conservative value-of-computation policy. -The v0.1 vertical slice now includes bounded hierarchical candidate compilation, a revision-verified Laya -fast path with calibration-gated abstention, and a reproducible DecisionMix v1 schema preview. -It is a research alpha: the controller is runnable and tested; fine-tuned weights and empirical -production calibration remain release gates, not implied claims. - -> **Launch status:** The standalone decision service and governed trace collection -> are not enabled for public use. The independent [PR #2 review](https://github.com/ColomboAI-com/c3r/pull/2#pullrequestreview-5293835066) +The v0.1 vertical slice includes bounded hierarchical candidate compilation, a +calibration-gated System-One seam, and a reproducible DecisionMix v1 schema preview. +CLM is the new default System-One provider in code; Laya remains optional. +It is a research alpha: the controller is runnable and tested. Fine-tuned weights +and empirical calibration remain gates for the **empirical model/data release**, +not for the separate stateless recommendation API described below. + +> **Launch status:** Neither the standalone decision service nor governed trace collection +> is enabled for public use. The independent [PR #2 review](https://github.com/ColomboAI-com/c3r/pull/2#pullrequestreview-5293835066) > requests changes. A real cloud storage audit probe and the first natural > empty-bucket purge run are recorded for reviewer inspection, but they do not -> establish deletion of aged traces or backups, trained weights, calibration, -> safety results, or canary evidence. See the [reviewer packet](docs/reviewer-staging-packet-2026-09-23.md). +> establish deletion of aged traces or backups, trained weights, or calibration +> for the research release. The separate stateless API still needs its own +> security, live provider, and canary evidence. See the [reviewer packet](docs/reviewer-staging-packet-2026-09-23.md). + +### Separate stateless API path + +The [C3R Core API v1 contract](docs/stateless-core-api.md) separates a +recommendation-only, non-persistent inference service from the governed trace +collection and empirical-release program above. The current branch implements +the bounded HTTP routes and a fail-closed `production_inference` mode; it does +**not** mean the API is deployed or production-qualified. Upstream CLM provides +advisory System-One ranking, not generative text or calibrated task-success +probabilities. `/v1/c3r/execute` and `/v1/responses` are deliberately disabled. +The first public hostname is a dedicated C3R endpoint, not an MC-1 integration. +See the [scope-specific release policy](docs/release-policy.md). ### Default language model C3R's default **deliberative** language model is [`deepseek-ai/DeepSeek-V4.1-Flash`](https://huggingface.co/deepseek-ai/DeepSeek-V4.1-Flash) (MIT). The hosted default uses `deepseek/deepseek-v4.1-flash` through OpenRouter; the -official DeepSeek API alias is `deepseek-flash`. Laya remains the separate System-One -decision fast path, and DeepSeek recommendations remain subject to the same verifier and +official DeepSeek API alias is `deepseek-flash`. CLM is the default separate +System-One decision engine; Laya remains optional. DeepSeek recommendations +remain subject to the same verifier and trusted commit boundary as every other candidate. The official checkpoint has passed a private 8×H100 GCP serving smoke test and is backed @@ -49,7 +65,7 @@ keeps that recommendation separate from authority to act. ```mermaid flowchart LR S[Versioned state] --> C[Hierarchical candidate compiler] - C --> L[Laya fast path] + C --> L[CLM default / Laya optional] C --> D[Deliberative envelope] L --> V[Robust CVoC] D --> V @@ -67,12 +83,12 @@ failed verification into success, or directly commit an external effect. | Surface | Included now | | --- | --- | | Candidate Compiler | family → subgroup → operation → arguments → placement → verifier; hard masks, budget pruning, caps, progressive widening | -| Laya fast path | exact upstream revision and license verification, typed probabilities, slice calibration, confidence/margin abstention | +| System-One fast path | CLM loopback rank adapter with strict response checks and calibration-gated abstention; optional revision-verified Laya | | DecisionMix v1 | validated records, immutable deterministic splits, source/license provenance, SHA-256 manifest, 144-record synthetic preview | | Authority boundary | action-bound verifier attestations, expiring single-use approvals, atomic nonce claims | | Runtime control | conservative CVoC selection, deterministic `STOP`, cost twin, deliberative contracts, trace schema and optional transactional SQLite hash chain | | Operational controls | fail-closed feature flags, tested frontier/open-weight HTTP contracts, Colibri shadow recommendations, canonical trace hash chain | -| Standalone controller boundary | tested state → candidates → optional Laya → CVoC → independent verifier → recommendation/commit fallback → redacted trace composition; host-owned read-only request factory, data-boundary-aware provider bridge, and authenticated rate-limited loopback HTTP boundary. A separate private Cloud Run staging host is deliberately fixed-disabled; it is not the decision service or a public launch. | +| Standalone controller boundary | tested state → candidates → optional System-One → CVoC → independent verifier → read-only recommendation or deterministic fallback → redacted trace composition; the standalone controller rejects external executors until effects and durable trace commits can be made atomic. A separate private Cloud Run staging host is fixed-disabled; it is not the decision service or a public launch. | This compiler is the reviewed vertical slice, not the directive's full Candidate Compiler Definition of Done. Rich typed value constraints, per-argument provenance, dominated-branch @@ -142,8 +158,34 @@ is `STOP`. - [`docs/prior-art.md`](docs/prior-art.md) — explicit attribution links and the canonical novelty boundary required by the execution directive. +## CLM System-One integration + +`C3R_SYSTEM_ONE_PROVIDER=clm` is the configuration default, while +`C3R_ENABLED` and `C3R_SYSTEM_ONE` still default to off. The adapter targets +the upstream [Contrastive-LM/CLM](https://github.com/Contrastive-LM/CLM) +`/v1/rank` API on a loopback-only origin. Its source is pinned at +`bb42c6c5bf914fd449bed2f6ca65be80602cb1f7` (Apache-2.0). The running +encoder and CLM head need their own immutable artifact revision, passed as +`C3R_CLM_ARTIFACT_REVISION`; the current adapter validates the declaration's +format but does not yet attest the live server's artifact hash. This repository +does not bundle or claim trained C3R-specific CLM weights. The host supplies `C3R_CLM_URL` (default +`http://127.0.0.1:8700`), an optional `C3R_CLM_API_KEY`, and a fitted +`TemperatureCalibrator` to `build_default_clm_fast_path`. The initial +`C3R_CLM_TIMEOUT_MS=500` bounds the complete System-One decision; production +latency thresholds still require live measurement. + +CLM ranks bounded, policy-surviving candidate labels and fixed typed questions. +Its raw ranking is advisory, not an action selection. Missing calibration, +malformed responses, outages, or timeouts lead to abstention or deterministic +fallback. CVoC, independent verification, and the commit boundary retain +authority. No live CLM service, GPU coexistence benchmark, or held-out C3R +calibration is claimed by this code change. + ## Laya integration +Laya remains an explicitly selected comparison/compatibility provider; it is not +the default System-One engine. + The adapter pins `convaiinnovations/laya` at `1c5edc17a7acd8701df6fc341c0d179f1c62c982`. Before loading, the backend resolves the Hub metadata, verifies that exact SHA and the Apache-2.0 license, downloads that revision, and passes @@ -179,6 +221,7 @@ Production hosts must preserve independent control of: ```text C3R_ENABLED C3R_SYSTEM_ONE +C3R_SYSTEM_ONE_PROVIDER=clm C3R_DELIBERATIVE C3R_ROUTING C3R_SPECULATION @@ -193,7 +236,7 @@ learning. The reference provider and Colibri shadow contracts are documented in ## Release truth -Not yet claimed: trained `C3R-Decision-Laya-421M-v0.1` weights, empirical DecisionMix training +Not yet claimed: C3R-trained CLM heads or Laya weights, empirical DecisionMix training data, live provider qualification, MC-1 product integration, a Colibri shadow deployment, or measured production calibration/latency/cost results. Tested adapter and shadow-control contracts are included, but they are not represented as production runs. These remain documented gates. diff --git a/c3r/feature_flags.py b/c3r/feature_flags.py index c183f7c..54f3dd6 100644 --- a/c3r/feature_flags.py +++ b/c3r/feature_flags.py @@ -27,6 +27,7 @@ class FeatureFlags: speculation_requested: bool = False moe_control_requested: bool = False online_learning_requested: bool = False + system_one_provider: str = "clm" @classmethod def from_mapping(cls, values: Mapping[str, str]) -> FeatureFlags: @@ -38,7 +39,10 @@ def from_mapping(cls, values: Mapping[str, str]) -> FeatureFlags: speculation_requested=_boolean(values, "C3R_SPECULATION", False), moe_control_requested=_boolean(values, "C3R_MOE_CONTROL", False), online_learning_requested=_boolean(values, "C3R_ONLINE_LEARNING", False), + system_one_provider=values.get("C3R_SYSTEM_ONE_PROVIDER", "clm").strip().lower(), ) + if flags.system_one_provider not in {"clm", "laya", "jev", "rule"}: + raise ValueError("C3R_SYSTEM_ONE_PROVIDER must be clm, laya, jev, or rule") if flags.online_learning_requested: raise ValueError("autonomous online learning is prohibited") return flags diff --git a/c3r/http_service.py b/c3r/http_service.py index 549d743..c2ddfa6 100644 --- a/c3r/http_service.py +++ b/c3r/http_service.py @@ -9,6 +9,7 @@ import hmac import json import time +from math import isfinite from collections.abc import Callable, Mapping from dataclasses import asdict, is_dataclass from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer @@ -127,6 +128,32 @@ def do_GET(self) -> None: if self.path == "/health": self._send(200, {"status": "ok"}) return + if self.path in {"/ready", "/v1/models"}: + if not self._authorized(): + self._send(401, {"error": "unauthorized"}) + return + if self.path == "/ready": + ready = (self.server.runtime.decision_enabled + and self.server.runtime.system_one_enabled + and self.server.runtime.provider_ready) + self._send(200 if ready else 503, { + "status": "ready" if ready else "disabled", + "scope": "configured_provider_probe_and_decision_policy", + }) + else: + available = (self.server.runtime.decision_enabled + and self.server.runtime.system_one_enabled + and self.server.runtime.provider_ready) + self._send(200, {"models": [ + {"id": "c3r-core", "capability": "verified_recommendation", + "text_generation": False, "calibrated": False, + "effect_execution": False, "available": available}, + {"id": "c3r-system-one", "capability": "advisory_ranking", + "text_generation": False, "calibrated": False, + "effect_execution": False, + "available": available and self.server.runtime.system_one_enabled}, + ]}) + return if self.path == "/metrics": if not self._authorized(): self._send(401, {"error": "unauthorized"}) @@ -136,7 +163,10 @@ def do_GET(self) -> None: self._send(404, {"error": "not_found"}) def do_POST(self) -> None: - if self.path != "/v1/decisions": + if self.path not in { + "/v1/decisions", "/v1/c3r/decide", "/v1/c3r/rank", + "/v1/system-one", "/v1/c3r/execute", "/v1/responses", + }: self._send(404, {"error": "not_found"}) return if not self._authorized(): @@ -147,6 +177,9 @@ def do_POST(self) -> None: self.server.metrics.increment("rate_limited") self._send(429, {"error": "rate_limited"}) return + if self.path in {"/v1/c3r/execute", "/v1/responses"}: + self._send(501, {"error": "not_implemented", "reason": "recommendation_only"}) + return try: length = int(self.headers.get("Content-Length", "")) except ValueError: @@ -176,7 +209,27 @@ def do_POST(self) -> None: "authority_result": outcome.authority_result, "reason": outcome.reason, "trace_hash": outcome.ledger_record.record_hash, + "effect_executed": False, } + if self.path in {"/v1/c3r/rank", "/v1/system-one"}: + result["abstained"] = outcome.fast_path is None or outcome.fast_path.abstained + result["model_id"] = ( + None if outcome.fast_path is None else outcome.fast_path.model_id + ) + result["scope"] = "controller_decision_with_system_one_fallback" + scores = (() if outcome.fast_path is None + else outcome.fast_path.candidate_probabilities) + if (len(scores) != len(outcome.candidate_ids) + or any(not isfinite(score) or score < 0 or score > 1 for score in scores)): + scores = () + result["candidate_ranking"] = [ + {"candidate_id": candidate_id, "system_one_score": score, + "calibrated": False} + for candidate_id, score in sorted( + zip(outcome.candidate_ids, scores), + key=lambda item: item[1], reverse=True, + ) + ] if isinstance(outcome.deliberation, DeliberativeResult) and is_dataclass( outcome.deliberation ): diff --git a/c3r/ingress_proxy.py b/c3r/ingress_proxy.py index dff9669..a28e3d8 100644 --- a/c3r/ingress_proxy.py +++ b/c3r/ingress_proxy.py @@ -114,16 +114,19 @@ def _forward(self, method: str, body: bytes | None = None) -> None: self.wfile.write(forwarded) def do_GET(self) -> None: - if self.path not in {"/health", "/metrics"}: + if self.path not in {"/health", "/metrics", "/ready", "/v1/models"}: self._send_error(404, "not_found") return - if self.path == "/metrics" and not self._authorized(): + if self.path != "/health" and not self._authorized(): self._send_error(401, "unauthorized") return self._forward("GET") def do_POST(self) -> None: - if self.path != "/v1/decisions": + if self.path not in { + "/v1/decisions", "/v1/c3r/decide", "/v1/c3r/rank", + "/v1/system-one", "/v1/c3r/execute", "/v1/responses", + }: self._send_error(404, "not_found") return if not self._authorized(): diff --git a/c3r/runtime.py b/c3r/runtime.py index ecc1fb1..6527be8 100644 --- a/c3r/runtime.py +++ b/c3r/runtime.py @@ -28,9 +28,10 @@ RiskClass, ValueEstimate, ) -from .system_one.fast_path import FastPathDecision, LayaFastPath +from .system_one.fast_path import CalibratedFastPath, FastPathDecision from .system_one.question_registry import TypedQuestion from .telemetry.trace import DecisionTrace +from .telemetry.ephemeral import EphemeralTraceSink from .telemetry.trace_ledger import LedgerRecord from .verifier_firewall import VerifierFirewall @@ -61,6 +62,7 @@ class RuntimeOutcome: authority_result: str reason: str ledger_record: LedgerRecord + candidate_ids: tuple[str, ...] = () fast_path: FastPathDecision | None = None deliberation: object | None = None @@ -83,9 +85,10 @@ def __init__( cvoc: RobustCvocController, verifier: VerifierFirewall, ledger: TraceSink, - fast_path: LayaFastPath | None = None, + fast_path: CalibratedFastPath | None = None, deliberator: Deliberator | None = None, executor: Callable[[ActionCandidate], None] | None = None, + readiness_probe: Callable[[], bool] | None = None, ) -> None: if executor is not None: raise ValueError("external effects are unsupported by StandaloneController") @@ -97,12 +100,37 @@ def __init__( self._ledger = ledger self._fast_path = fast_path self._deliberator = deliberator + self._readiness_probe = readiness_probe @property def effect_execution_enabled(self) -> bool: """Capability flag retained for fail-closed hosting checks.""" return False + @property + def decision_enabled(self) -> bool: + """Whether the host requested C3R decisions; not a provider health probe.""" + return self._flags.enabled_requested + + @property + def system_one_enabled(self) -> bool: + return self._flags.system_one_enabled and self._fast_path is not None + + @property + def trace_persistence_enabled(self) -> bool: + """Unknown host sinks are treated as persistent for production gating.""" + return type(self._ledger) is not EphemeralTraceSink + + @property + def provider_ready(self) -> bool: + """No provider-health assertion is made without a host probe.""" + if self._readiness_probe is None: + return False + try: + return bool(self._readiness_probe()) + except (OSError, RuntimeError, TypeError, ValueError): + return False + def run(self, request: RuntimeRequest) -> RuntimeOutcome: if not request.run_id: raise ValueError("run_id is required") @@ -129,12 +157,14 @@ def finish( system_cost: Mapping[str, float] | None = None, provider_id: str | None = None, ) -> RuntimeOutcome: - probabilities = {} if fast is None else fast.probabilities + probabilities = {} if fast is None else dict(fast.probabilities) + if fast is not None and fast.candidate_probabilities: + probabilities["CANDIDATE_RANK"] = fast.candidate_probabilities trace = DecisionTrace( run_id=request.run_id, state_hash=state_hash, access_level=request.access_level, - model_provider=provider_id or route, + model_provider=provider_id or (fast.model_id if fast is not None else route), candidate_ids=candidate_ids, probabilities=probabilities, utility_quantiles=( @@ -152,6 +182,7 @@ def finish( authority_result=authority_result, reason=reason, ledger_record=self._ledger.append(trace), + candidate_ids=candidate_ids, fast_path=fast, deliberation=deliberation, ) @@ -184,16 +215,27 @@ def finish( if self._flags.system_one_enabled: if self._fast_path is None: return finish("deterministic", "SYSTEM_ONE_UNAVAILABLE", candidate_ids=candidate_ids) + if self._fast_path.provider != self._flags.system_one_provider: + return finish("deterministic", "SYSTEM_ONE_PROVIDER_MISMATCH", candidate_ids=candidate_ids) questions = ( TypedQuestion("STOP_NOW", ("NO", "YES")), TypedQuestion("DELIBERATION_REQUIRED", ("NO", "YES")), ) + # Only stable identifiers and public operation metadata cross the CLM + # boundary. Argument values and provenance never enter action labels. + candidate_options = tuple( + f"{item.id} | {item.family.value} | {item.risk_class.value}" + for item in compiled.candidates + ) try: fast = self._fast_path.decide( - state, questions, action_family="CONTROL", language=request.language + state, questions, action_family="CONTROL", language=request.language, + candidate_options=candidate_options, ) except (OSError, RuntimeError, TypeError, ValueError): - return finish("deterministic", "SYSTEM_ONE_FAILURE", candidate_ids=candidate_ids) + return self._deliberate_or_stop( + state, finish, candidate_ids, "SYSTEM_ONE_FAILURE", None + ) if fast.abstained: return self._deliberate_or_stop( state, finish, candidate_ids, "SYSTEM_ONE_ABSTAINED", fast diff --git a/c3r/serve.py b/c3r/serve.py index 8fffeec..abc6c59 100644 --- a/c3r/serve.py +++ b/c3r/serve.py @@ -68,6 +68,20 @@ def build_servers( runtime, factory = builder_loader(reference)() if runtime.effect_execution_enabled: raise ValueError("host must be recommendation-only") + mode = values.get("C3R_MODE", "staging") + if mode not in {"staging", "production_inference", "research_collection"}: + raise ValueError("C3R_MODE is invalid") + if mode == "production_inference": + if values.get("C3R_TRACE_COLLECTION", "false").strip().lower() not in {"false", "0", "off"}: + raise ValueError("production inference cannot collect traces") + if values.get("C3R_ONLINE_LEARNING", "false").strip().lower() not in {"false", "0", "off"}: + raise ValueError("production inference cannot learn online") + if runtime.trace_persistence_enabled: + raise ValueError("production inference requires the ephemeral trace sink") + if not runtime.decision_enabled: + raise ValueError("production inference requires decisions enabled") + if not runtime.system_one_enabled: + raise ValueError("production inference requires an enabled System-One path") backend = C3RHTTPServer( runtime=runtime, request_factory=factory, diff --git a/c3r/staging_host.py b/c3r/staging_host.py index 40da96d..06b3655 100644 --- a/c3r/staging_host.py +++ b/c3r/staging_host.py @@ -15,18 +15,11 @@ from .runtime import StandaloneController from .state_compiler import StateCompiler from .state_schema import ActionDefinition, ActionFamily, AuthorityPolicy, RiskClass -from .telemetry.trace import DecisionTrace -from .telemetry.trace_ledger import LedgerRecord, _record_hash, canonical_trace_json +from .telemetry.ephemeral import EphemeralTraceSink from .verifier_firewall import VerifierDecision, VerifierFirewall, VerifierPolicy -class EphemeralStagingSink: - """Return a per-request integrity hash without retaining any trace rows.""" - - def append(self, trace: DecisionTrace) -> LedgerRecord: - canonical = canonical_trace_json(trace) - genesis = "0" * 64 - return LedgerRecord(genesis, _record_hash(genesis, canonical), canonical) +EphemeralStagingSink = EphemeralTraceSink def build() -> tuple[StandaloneController, ReadOnlyRequestFactory]: diff --git a/c3r/system_one/__init__.py b/c3r/system_one/__init__.py index f646d19..b8eab5e 100644 --- a/c3r/system_one/__init__.py +++ b/c3r/system_one/__init__.py @@ -1,8 +1,13 @@ """Bounded System-One controller interfaces.""" -from .fast_path import LayaFastPath +from .clm_adapter import ClmAdapter +from .factory import build_default_clm_fast_path +from .fast_path import CalibratedFastPath, LayaFastPath from .laya_adapter import LayaAdapter from .laya_backend import PinnedLayaBackend from .question_registry import C3R_QUESTIONS -__all__ = ["C3R_QUESTIONS", "LayaAdapter", "LayaFastPath", "PinnedLayaBackend"] +__all__ = [ + "C3R_QUESTIONS", "CalibratedFastPath", "ClmAdapter", "LayaAdapter", + "LayaFastPath", "PinnedLayaBackend", "build_default_clm_fast_path", +] diff --git a/c3r/system_one/clm_adapter.py b/c3r/system_one/clm_adapter.py new file mode 100644 index 0000000..b5eb372 --- /dev/null +++ b/c3r/system_one/clm_adapter.py @@ -0,0 +1,168 @@ +"""Bounded, loopback-only adapter for the pinned upstream CLM rank API. + +CLM predictions are never authority or calibrated probabilities. The guarded +fast path consumes these as logits only when a held-out calibration slice exists. +""" + +from __future__ import annotations + +from collections.abc import Callable, Mapping +from dataclasses import asdict, dataclass +import json +import math +import re +import time +from typing import cast +from urllib.parse import urlsplit +from urllib.request import HTTPHandler, HTTPRedirectHandler, ProxyHandler, Request, build_opener + +from ..state_schema import CompiledState +from .question_registry import TypedQuestion + + +UPSTREAM_CLM_COMMIT = "bb42c6c5bf914fd449bed2f6ca65be80602cb1f7" +_IMMUTABLE_REVISION = re.compile(r"^[0-9a-f]{40,64}$") +_MAX_CONTEXT_BYTES = 32_768 +_MAX_RESPONSE_BYTES = 65_536 +_MAX_OPTIONS = 64 +RankTransport = Callable[[Mapping[str, object]], Mapping[str, object]] + + +class _NoRedirect(HTTPRedirectHandler): + def redirect_request(self, request: Request, fp: object, code: int, + msg: str, headers: object, newurl: str) -> None: + return None + + +@dataclass(frozen=True, slots=True) +class ClmAdapter: + """Convert CLM rankings to typed logits in the original option order. + + ``revision`` is the host-declared CLM head/encoder bundle hash, not an + attestation from the server. The upstream code pin is recorded separately. + """ + + revision: str + endpoint: str = "http://127.0.0.1:8700" + api_key: str | None = None + timeout_seconds: float = 0.5 + transport: RankTransport | None = None + model_id: str = "Contrastive-LM/CLM" + served_model: str = "clm-latest" + + @property + def provider(self) -> str: + return "clm" + + def __post_init__(self) -> None: + parsed = urlsplit(self.endpoint) + if ( + parsed.scheme != "http" + or parsed.hostname not in {"127.0.0.1", "localhost", "::1"} + or parsed.path not in {"", "/"} + or parsed.username is not None + or parsed.password is not None + or parsed.query + or parsed.fragment + or parsed.port is None + ): + raise ValueError("CLM endpoint must be a local HTTP origin with an explicit port") + if _IMMUTABLE_REVISION.fullmatch(self.revision) is None: + raise ValueError("CLM artifact revision must be an immutable SHA-256/SHA-1 hash") + if not math.isfinite(self.timeout_seconds) or not 0 < self.timeout_seconds <= 10: + raise ValueError("CLM timeout must be finite and at most ten seconds") + if not self.served_model or len(self.served_model) > 128: + raise ValueError("CLM served model name must be bounded") + + def predict( + self, state: CompiledState, questions: tuple[TypedQuestion, ...], + *, deadline: float | None = None, + ) -> Mapping[str, tuple[float, ...]]: + context = self._context(state) + output: dict[str, tuple[float, ...]] = {} + for question in questions: + probabilities = self._rank(context, question.id, question.options, deadline=deadline) + output[question.id] = tuple(math.log(max(value, 1e-12)) for value in probabilities) + return output + + def rank_actions( + self, state: CompiledState, candidate_ids: tuple[str, ...], + *, deadline: float | None = None, + ) -> tuple[float, ...]: + """Advisory candidate distribution; CVoC and policy still choose actions.""" + return self._rank(self._context(state), "NEXT_ACTION", candidate_ids, deadline=deadline) + + def _context(self, state: CompiledState) -> str: + context = json.dumps(asdict(state), sort_keys=True, allow_nan=False, ensure_ascii=False) + if len(context.encode("utf-8")) > _MAX_CONTEXT_BYTES: + raise ValueError("compiled state exceeds the bounded CLM context") + return context + + def _rank( + self, context: str, question: str, options: tuple[str, ...], + *, deadline: float | None = None, + ) -> tuple[float, ...]: + if not options or len(options) > _MAX_OPTIONS or len(set(options)) != len(options): + raise ValueError("CLM options must be unique and bounded") + if any(not option or len(option) > 1024 for option in options): + raise ValueError("CLM option is empty or oversized") + payload: Mapping[str, object] = { + "context": context, + "question": question, + "answers": list(options), + "model": self.served_model, + } + remaining = self.timeout_seconds + if deadline is not None: + remaining = min(remaining, deadline - time.monotonic()) + if remaining <= 0: + raise TimeoutError("CLM decision deadline exceeded") + response = self.transport(payload) if self.transport is not None else self._post(payload, remaining) + if response.get("model") != self.served_model: + raise ValueError("CLM served model does not match the requested model") + ranked = response.get("ranked") + if not isinstance(ranked, list) or len(ranked) != len(options): + raise ValueError("CLM returned an incomplete ranking") + probabilities: dict[str, float] = {} + for item in ranked: + if not isinstance(item, dict): + raise ValueError("CLM returned an invalid ranking item") + candidate = item.get("candidate") + probability = item.get("prob") + if ( + not isinstance(candidate, str) + or candidate not in options + or candidate in probabilities + or isinstance(probability, bool) + or not isinstance(probability, (int, float)) + or not math.isfinite(probability) + or not 0 <= probability <= 1 + ): + raise ValueError("CLM ranking contains an unknown or invalid probability") + probabilities[candidate] = float(probability) + total = sum(probabilities.values()) + if not 0.98 <= total <= 1.02: + raise ValueError("CLM ranking probabilities do not sum to one") + return tuple(probabilities[option] / total for option in options) + + def _post(self, payload: Mapping[str, object], timeout: float) -> Mapping[str, object]: + headers = {"Content-Type": "application/json"} + if self.api_key: + headers["Authorization"] = f"Bearer {self.api_key}" + request = Request( + self.endpoint.rstrip("/") + "/v1/rank", + data=json.dumps(payload, ensure_ascii=False, allow_nan=False).encode("utf-8"), + headers=headers, + method="POST", + ) + # Ignore proxy environment variables and reject redirects so a local + # server cannot relay the secret or state to an off-host destination. + opener = build_opener(ProxyHandler({}), _NoRedirect(), HTTPHandler()) + with opener.open(request, timeout=timeout) as response: + body = response.read(_MAX_RESPONSE_BYTES + 1) + if len(body) > _MAX_RESPONSE_BYTES: + raise ValueError("CLM response exceeds the allowed size") + parsed = json.loads(body) + if not isinstance(parsed, dict): + raise ValueError("CLM returned a non-object response") + return cast(Mapping[str, object], parsed) diff --git a/c3r/system_one/factory.py b/c3r/system_one/factory.py new file mode 100644 index 0000000..c7cf083 --- /dev/null +++ b/c3r/system_one/factory.py @@ -0,0 +1,36 @@ +"""Explicit construction of the default CLM System-One path. + +No mutable upstream model alias or unfitted calibration is silently promoted. +The host owns secrets, artifacts, and the enable flag. +""" + +from __future__ import annotations + +from collections.abc import Mapping + +from .calibration import TemperatureCalibrator +from .clm_adapter import ClmAdapter +from .fast_path import CalibratedFastPath + + +def build_default_clm_fast_path( + values: Mapping[str, str], *, calibrator: TemperatureCalibrator +) -> CalibratedFastPath: + """Build CLM; require a host-declared immutable artifact revision. + + This does not attest the live server's artifacts. Deployment must verify + encoder/head hashes independently before enabling real decisions. + """ + revision = values.get("C3R_CLM_ARTIFACT_REVISION", "") + endpoint = values.get("C3R_CLM_URL", "http://127.0.0.1:8700") + timeout = float(values.get("C3R_CLM_TIMEOUT_MS", "500")) / 1000 + return CalibratedFastPath( + adapter=ClmAdapter( + revision=revision, + endpoint=endpoint, + api_key=values.get("C3R_CLM_API_KEY"), + timeout_seconds=timeout, + ), + calibrator=calibrator, + maximum_decision_seconds=timeout, + ) diff --git a/c3r/system_one/fast_path.py b/c3r/system_one/fast_path.py index dd02c9e..d84f6df 100644 --- a/c3r/system_one/fast_path.py +++ b/c3r/system_one/fast_path.py @@ -4,14 +4,32 @@ from dataclasses import dataclass import math +import time +from collections.abc import Mapping +from typing import Protocol from ..state_schema import CompiledState from .abstention import should_abstain from .calibration import CalibrationKey, TemperatureCalibrator -from .laya_adapter import LayaAdapter from .question_registry import TypedQuestion +class TypedInferenceAdapter(Protocol): + model_id: str + revision: str + provider: str + + def predict( + self, state: CompiledState, questions: tuple[TypedQuestion, ...], + *, deadline: float | None = None, + ) -> Mapping[str, tuple[float, ...]]: ... + + def rank_actions( + self, state: CompiledState, candidate_ids: tuple[str, ...], + *, deadline: float | None = None, + ) -> tuple[float, ...]: ... + + def option_count_bucket(option_count: int) -> str: if option_count <= 2: return str(option_count) @@ -32,25 +50,34 @@ class FastPathDecision: reasons: tuple[str, ...] model_id: str model_revision: str + candidate_probabilities: tuple[float, ...] = () -class LayaFastPath: +class CalibratedFastPath: def __init__( self, *, - adapter: LayaAdapter, + adapter: TypedInferenceAdapter, calibrator: TemperatureCalibrator, minimum_top_probability: float = 0.65, minimum_margin: float = 0.10, + maximum_decision_seconds: float = 0.5, ) -> None: if not 0.0 <= minimum_top_probability <= 1.0: raise ValueError("minimum_top_probability must be between 0 and 1") if not 0.0 <= minimum_margin <= 1.0: raise ValueError("minimum_margin must be between 0 and 1") + if not math.isfinite(maximum_decision_seconds) or maximum_decision_seconds <= 0: + raise ValueError("maximum_decision_seconds must be positive and finite") self._adapter = adapter self._calibrator = calibrator self._minimum_top_probability = minimum_top_probability self._minimum_margin = minimum_margin + self._maximum_decision_seconds = maximum_decision_seconds + + @property + def provider(self) -> str: + return self._adapter.provider def decide( self, @@ -59,8 +86,15 @@ def decide( *, action_family: str, language: str = "en", + candidate_options: tuple[str, ...] = (), ) -> FastPathDecision: - logits_by_question = self._adapter.predict(state, questions) + deadline = time.monotonic() + self._maximum_decision_seconds + candidate_probabilities: tuple[float, ...] = () + if candidate_options: + candidate_probabilities = self._adapter.rank_actions( + state, candidate_options, deadline=deadline + ) + logits_by_question = self._adapter.predict(state, questions, deadline=deadline) consequence = str(state.risk.get("consequence", "default")) answers: dict[str, str] = {} probabilities: dict[str, tuple[float, ...]] = {} @@ -104,4 +138,10 @@ def decide( reasons=tuple(reasons), model_id=self._adapter.model_id, model_revision=self._adapter.revision, + candidate_probabilities=candidate_probabilities, ) + + +# Preserve the public Laya integration while allowing the same guarded path to +# serve CLM and other explicitly configured System-One engines. +LayaFastPath = CalibratedFastPath diff --git a/c3r/system_one/laya_adapter.py b/c3r/system_one/laya_adapter.py index 8f01ab0..8149d46 100644 --- a/c3r/system_one/laya_adapter.py +++ b/c3r/system_one/laya_adapter.py @@ -19,6 +19,17 @@ class LayaAdapter: revision: str backend: InferenceBackend + @property + def provider(self) -> str: + return "laya" + + def rank_actions( + self, state: CompiledState, candidate_ids: tuple[str, ...], + *, deadline: float | None = None, + ) -> tuple[float, ...]: + """This legacy typed-question adapter has no action-ranking head.""" + return () + def __post_init__(self) -> None: if self.model_id not in { "convaiinnovations/laya", @@ -33,5 +44,6 @@ def predict( self, state: CompiledState, questions: tuple[TypedQuestion, ...], + *, deadline: float | None = None, ) -> Mapping[str, tuple[float, ...]]: return self.backend(state, questions) diff --git a/c3r/telemetry/ephemeral.py b/c3r/telemetry/ephemeral.py new file mode 100644 index 0000000..37de5d3 --- /dev/null +++ b/c3r/telemetry/ephemeral.py @@ -0,0 +1,15 @@ +"""Request-local trace hashing without retained rows or a persistent ledger.""" + +from __future__ import annotations + +from .trace import DecisionTrace +from .trace_ledger import LedgerRecord, _record_hash, canonical_trace_json + + +class EphemeralTraceSink: + """Hash each response independently; the returned record is not stored.""" + + def append(self, trace: DecisionTrace) -> LedgerRecord: + canonical = canonical_trace_json(trace) + genesis = "0" * 64 + return LedgerRecord(genesis, _record_hash(genesis, canonical), canonical) diff --git a/cloudbuild.retention.yaml b/cloudbuild.retention.yaml new file mode 100644 index 0000000..3b2eea3 --- /dev/null +++ b/cloudbuild.retention.yaml @@ -0,0 +1,18 @@ +# Supply _IMAGE as a dedicated, private Artifact Registry destination. +# The source upload is restricted by .gcloudignore; this build never receives +# traces, papers, local evidence, secrets, or Git metadata. +steps: + - name: gcr.io/cloud-builders/docker + args: ["build", "--file", "Dockerfile.retention", "--tag", "${_IMAGE}", "."] + - name: gcr.io/cloud-builders/docker + entrypoint: sh + args: + - -c + - >- + test "$(docker image inspect ${_IMAGE} --format '{{json .Config.Cmd}}')" + = '["python","-m","c3r.retention_job"]' + - name: gcr.io/cloud-builders/docker + args: ["run", "--rm", "${_IMAGE}", "python", "-c", "import c3r.retention_job"] + - name: aquasec/trivy:0.74.0 + args: ["image", "--no-progress", "--exit-code", "1", "--severity", "HIGH,CRITICAL", "${_IMAGE}"] +images: ["${_IMAGE}"] diff --git a/cloudbuild.verify.yaml b/cloudbuild.verify.yaml new file mode 100644 index 0000000..18bf044 --- /dev/null +++ b/cloudbuild.verify.yaml @@ -0,0 +1,42 @@ +# GitHub-independent verification. Submit from the repository root with: +# gcloud builds submit . --config=cloudbuild.verify.yaml --ignore-file=.gcloudignore.verify --project=APPROVED_BUILD_PROJECT --region=us-central1 +# The uploaded context contains only public runtime source and synthetic tests. +steps: + - name: python:3.11-alpine3.24 + id: unit-311 + entrypoint: python + args: ["-m", "unittest", "discover", "-s", "tests", "-v"] + - name: python:3.12-alpine3.23 + id: unit-312 + entrypoint: python + args: ["-m", "unittest", "discover", "-s", "tests", "-v"] + - name: python:3.13-alpine3.23 + id: unit-313 + entrypoint: python + args: ["-m", "unittest", "discover", "-s", "tests", "-v"] + - name: gcr.io/cloud-builders/docker + id: runtime-image + args: ["build", "--pull", "--file", "Dockerfile", "--tag", "c3r-runtime:verify", "."] + - name: gcr.io/cloud-builders/docker + id: retention-image + args: ["build", "--pull", "--file", "Dockerfile.retention", "--tag", "c3r-retention:verify", "."] + - name: gcr.io/cloud-builders/docker + id: retention-command + entrypoint: sh + args: + - -c + - >- + test "$(docker image inspect c3r-retention:verify --format '{{json .Config.Cmd}}')" + = '["python","-m","c3r.retention_job"]' + - name: gcr.io/cloud-builders/docker + id: runtime-import + args: ["run", "--rm", "c3r-runtime:verify", "python", "-c", "import c3r.serve"] + - name: gcr.io/cloud-builders/docker + id: retention-import + args: ["run", "--rm", "c3r-retention:verify", "python", "-c", "import c3r.retention_job"] + - name: aquasec/trivy:0.74.0 + id: runtime-vulnerability-scan + args: ["image", "--no-progress", "--exit-code", "1", "--severity", "HIGH,CRITICAL", "c3r-runtime:verify"] + - name: aquasec/trivy:0.74.0 + id: retention-vulnerability-scan + args: ["image", "--no-progress", "--exit-code", "1", "--severity", "HIGH,CRITICAL", "c3r-retention:verify"] diff --git a/docs/release-policy.md b/docs/release-policy.md new file mode 100644 index 0000000..dc25337 --- /dev/null +++ b/docs/release-policy.md @@ -0,0 +1,25 @@ +# Release policy by scope + +## Stateless C3R Core inference + +An accountable ColomboAI release owner may authorize a recommendation-only +deployment after the automated and live gates in +[stateless-core-api.md](stateless-core-api.md) pass and the evidence is recorded. +The approval must name the deployed artifact digest, CLM/Qwen revisions, +hostname, rollback revision, on-call contact, and canary result. A green build +alone is insufficient. The release owner may stop or roll back at any time. + +This scope does not collect training traces, train C3R-specific weights, publish +DecisionMix rows, or claim empirical calibration. Swapnil Pawar's existing +`CHANGES_REQUESTED` review on PR #2 is preserved; it is not silently converted +to approval. His separate collection/publication review does not automatically +authorize or block this no-collection API. Any legal, security, or organizational +review otherwise required by ColomboAI remains applicable. + +## Research collection and empirical publication + +The approved C3R-controlled internal-task scope and independent reviewer gates +remain unchanged. Trace collection stays off until that scope's deployed +controls and explicit approval exist. Training, held-out calibration, publication +of rows or weights, and empirical claims each require their own evidence and +review. No production inference deployment implies research approval. diff --git a/docs/stateless-core-api.md b/docs/stateless-core-api.md new file mode 100644 index 0000000..b025ee0 --- /dev/null +++ b/docs/stateless-core-api.md @@ -0,0 +1,82 @@ +# C3R Core API v1: stateless inference contract + +This is a **separate release scope** from governed research collection. It may +serve read-only, verified recommendations using the upstream CLM ranker without +C3R-trained weights or an empirical DecisionMix release. It does **not** confer +permission to execute external effects or make calibrated success claims. + +## Operating modes + +`C3R_MODE=production_inference` fails startup unless the trusted host enables +decisions and a System-One path, rejects external effect execution, and uses the in-process +`EphemeralTraceSink`. It rejects `C3R_TRACE_COLLECTION=true` and +`C3R_ONLINE_LEARNING=true`. This sink computes a response hash but stores no +trace rows or cross-request chain. Ordinary aggregate counts are allowed; +request bodies, state, options, and tokens are not logged by the HTTP boundary. + +`C3R_MODE=research_collection` is a separate future mode. Existing internal-task +admission, retention, reviewer, and publication controls continue to apply +there. Merely setting the mode does not authorize or activate collection. + +The default mode is `staging`, preserving existing deployments. No mode +implicitly turns on a provider or grants action authority. + +## API + +The external ingress requires `X-C3R-Token` behind a TLS/IAM boundary; the +loopback backend uses a different bearer token. Requests are JSON objects and +bounded to 64 KiB. The trusted host owns the action catalog, policy, verifier, +measured CVoC estimates, and provider connectivity. Callers cannot override +those fields. + +| Route | Contract | +| --- | --- | +| `GET /health` | Process liveness only. | +| `GET /ready` | Authenticated. Returns 503 until decisions and System-One are enabled **and** the host's provider probe succeeds. It is not a complete launch attestation. | +| `GET /v1/models` | Authenticated model-like discovery of `c3r-core` and `c3r-system-one`, each labeled non-generative and uncalibrated. Availability follows the provider probe. | +| `POST /v1/c3r/decide` | Runs the controller and returns a read-only recommendation or explicit fallback. | +| `POST /v1/c3r/rank` | Same safe controller path, plus available advisory candidate scores. No score is represented as probability of task success. | +| `POST /v1/system-one` | Same controller path with System-One status and fallback. This is **not** arbitrary typed-question inference. | +| `POST /v1/c3r/execute` | Returns 501. External effects are unsupported. | +| `POST /v1/responses` | Returns 501. This release is not OpenAI Responses API-compatible and CLM is not a text generator. | + +`POST /v1/decisions` remains the compatibility route. Example: + +```http +POST /v1/c3r/decide +Content-Type: application/json +X-C3R-Token: + +{"goal":"Find record","current_subgoal":"Search approved index"} +``` + +The response includes `selected_action_id`, `route`, `reason`, +`authority_result`, `effect_executed: false`, and a request-local `trace_hash`. +A ranking response additionally includes `candidate_ranking` entries with +`system_one_score` and `calibrated: false`. An empty ranking and abstention are +normal when CLM is unavailable or control questions lack held-out calibration. +Raw CLM scores cannot safely be converted into task-success probabilities by a +fixed cap; CVoC remains driven by trusted, measured host estimates. + +## Production gates + +This repository currently has no trusted production host catalog/estimate +source, live CLM/Qwen provider deployment, approved dedicated public hostname, +or production canary. Before public traffic, qualify: + +1. Immutable CLM and Qwen encoder artifacts, provider health/timeout/fallback, + and revision attestations. +2. Host-supplied catalog, measured estimates, read-only verifier, data-boundary + policy, and tenant isolation. Prove high-risk and forged-authority rejection. +3. TLS, authentication, authorization, secrets, least-privilege runtime IAM, + rate/size/concurrency limits, and no payload retention in application and + infrastructure logs. +4. Independent build/tests, vulnerability scan, load and outage tests, live + application acceptance, alert delivery, rollback, and staged canary with + predeclared abort thresholds. +5. Accountable release-owner approval against the evidence. Research collection + or C3R-specific training is not a prerequisite for this **stateless** scope. + +Until those gates pass, neither `/health` nor local unit tests imply a live +production service. MC-1 integration is explicitly excluded; the first public +endpoint must be dedicated to C3R. diff --git a/tests/test_clm_adapter.py b/tests/test_clm_adapter.py new file mode 100644 index 0000000..0bba50a --- /dev/null +++ b/tests/test_clm_adapter.py @@ -0,0 +1,113 @@ +import math +import time +import unittest + +from c3r.feature_flags import FeatureFlags +from c3r.system_one.calibration import CalibrationKey, TemperatureCalibrator +from c3r.system_one.clm_adapter import ClmAdapter, UPSTREAM_CLM_COMMIT +from c3r.system_one.fast_path import CalibratedFastPath +from c3r.system_one.question_registry import TypedQuestion +from tests.test_laya_fast_path import compiled_state + + +REVISION = "a" * 64 + + +def ranked(payload: dict[str, object]) -> dict[str, object]: + options = payload["answers"] + assert isinstance(options, list) + return { + "model": "clm-latest", + "ranked": [ + {"rank": index + 1, "candidate": option, "prob": probability} + for index, (option, probability) in enumerate( + zip(reversed(options), (0.9, 0.1), strict=True) + ) + ] + } + + +class ClmAdapterTests(unittest.TestCase): + def test_default_is_clm_but_global_enable_is_off(self) -> None: + flags = FeatureFlags.from_mapping({}) + self.assertEqual(flags.system_one_provider, "clm") + self.assertFalse(flags.system_one_enabled) + + def test_rejects_remote_endpoint_and_mutable_revision(self) -> None: + with self.assertRaisesRegex(ValueError, "local HTTP"): + ClmAdapter(revision=REVISION, endpoint="https://example.com:8700") + with self.assertRaisesRegex(ValueError, "immutable"): + ClmAdapter(revision="latest") + self.assertEqual(len(UPSTREAM_CLM_COMMIT), 40) + + def test_expired_decision_budget_never_calls_clm(self) -> None: + def unreachable(_payload: object) -> dict[str, object]: + self.fail("expired request reached CLM") + + adapter = ClmAdapter(revision=REVISION, transport=unreachable) + with self.assertRaises(TimeoutError): + adapter.rank_actions( + compiled_state(), ("a", "b"), deadline=time.monotonic() - 1 + ) + + def test_maps_ranked_probabilities_back_to_fixed_option_order(self) -> None: + calls: list[object] = [] + + def transport(payload: object) -> dict[str, object]: + calls.append(payload) + assert isinstance(payload, dict) + return ranked(payload) + + adapter = ClmAdapter(revision=REVISION, transport=transport) + prediction = adapter.predict( + compiled_state(), (TypedQuestion("STOP_NOW", ("NO", "YES")),) + ) + self.assertAlmostEqual(math.exp(prediction["STOP_NOW"][0]), 0.1) + self.assertAlmostEqual(math.exp(prediction["STOP_NOW"][1]), 0.9) + self.assertEqual(len(calls), 1) + self.assertEqual(adapter.rank_actions(compiled_state(), ("a", "b")), (0.1, 0.9)) + + def test_fails_closed_on_unknown_duplicate_and_nonfinite_candidates(self) -> None: + invalid_rows = ( + (("NO", 0.5), ("EXTRA", 0.5)), + (("NO", 0.5), ("NO", 0.5)), + (("NO", float("nan")), ("YES", 0.5)), + (("NO", 0.1), ("YES", 0.1)), + ) + for rows in invalid_rows: + response = { + "model": "clm-latest", + "ranked": [{"candidate": candidate, "prob": prob} for candidate, prob in rows], + } + with self.subTest(response=response), self.assertRaises(ValueError): + ClmAdapter(revision=REVISION, transport=lambda _payload: response).predict( + compiled_state(), (TypedQuestion("STOP_NOW", ("NO", "YES")),) + ) + + def test_no_calibration_means_abstention_even_with_high_raw_rank(self) -> None: + adapter = ClmAdapter(revision=REVISION, transport=lambda payload: ranked(dict(payload))) + fast = CalibratedFastPath(adapter=adapter, calibrator=TemperatureCalibrator({})) + outcome = fast.decide( + compiled_state(), (TypedQuestion("STOP_NOW", ("NO", "YES")),), + action_family="CONTROL", candidate_options=("stop", "continue"), + ) + self.assertTrue(outcome.abstained) + self.assertEqual(outcome.answers, {}) + self.assertEqual(outcome.candidate_probabilities, (0.1, 0.9)) + + def test_held_out_calibration_enables_typed_answer(self) -> None: + adapter = ClmAdapter(revision=REVISION, transport=lambda payload: ranked(dict(payload))) + key = CalibrationKey("STOP_NOW", "CONTROL", "2", "en", "low") + fast = CalibratedFastPath( + adapter=adapter, calibrator=TemperatureCalibrator({key: 1.0}) + ) + outcome = fast.decide( + compiled_state(), (TypedQuestion("STOP_NOW", ("NO", "YES")),), + action_family="CONTROL", + ) + self.assertFalse(outcome.abstained) + self.assertEqual(outcome.answers["STOP_NOW"], "YES") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_ingress_proxy.py b/tests/test_ingress_proxy.py index 00c614f..5a81894 100644 --- a/tests/test_ingress_proxy.py +++ b/tests/test_ingress_proxy.py @@ -94,6 +94,15 @@ def test_missing_or_wrong_client_token_never_reaches_upstream(self): def test_health_is_unprivileged_but_metrics_require_token(self): self.assertEqual(self.request("/health", token=None)[0], 200) self.assertEqual(self.request("/metrics", token=None)[0], 401) + self.assertEqual(self.request("/ready", token=None)[0], 401) + self.assertEqual(self.request("/v1/models", token=None)[0], 401) + + def test_stateless_paths_forward_without_client_credential_leak(self): + for path in ("/v1/c3r/decide", "/v1/c3r/rank", "/v1/system-one"): + self.assertEqual(self.request(path, method="POST", payload={"goal": "inspect"})[0], 200) + seen_path, headers, _ = self.upstream.seen[-1] + self.assertEqual(seen_path, path) + self.assertNotIn("X-C3R-Token", headers) def test_rejects_unknown_path_without_contacting_upstream(self): self.assertEqual(self.request("/admin")[0], 404) diff --git a/tests/test_runtime.py b/tests/test_runtime.py index 65729ca..4c2f486 100644 --- a/tests/test_runtime.py +++ b/tests/test_runtime.py @@ -18,7 +18,8 @@ ValueEstimate, ) from c3r.system_one.calibration import TemperatureCalibrator -from c3r.system_one.fast_path import LayaFastPath +from c3r.system_one.clm_adapter import ClmAdapter +from c3r.system_one.fast_path import CalibratedFastPath, LayaFastPath from c3r.system_one.laya_adapter import LayaAdapter from c3r.telemetry.trace_ledger import TraceLedger from c3r.verifier_firewall import VerifierDecision, VerifierFirewall, VerifierPolicy @@ -73,6 +74,7 @@ def controller( executor=None, fast_path=None, deliberator=None, + system_one_provider: str = "clm", ) -> tuple[StandaloneController, TraceLedger]: ledger = TraceLedger() verifier = VerifierFirewall( @@ -85,6 +87,7 @@ def controller( enabled_requested=enabled, system_one_requested=system_one, deliberative_requested=deliberative, + system_one_provider=system_one_provider, ), compiler=StateCompiler(), candidates=CandidateCompiler(), @@ -176,6 +179,7 @@ def deliberate(self, _state): system_one=True, deliberative=True, fast_path=fast_path, + system_one_provider="laya", deliberator=Deliberator(), ) outcome = runtime.run(request()) @@ -185,6 +189,66 @@ def deliberate(self, _state): self.assertIsNone(outcome.selected_action_id) self.assertFalse(runtime.effect_execution_enabled) + def test_default_clm_never_silently_runs_laya(self) -> None: + adapter = LayaAdapter( + "convaiinnovations/laya", + "1c5edc17a7acd8701df6fc341c0d179f1c62c982", + backend=lambda _state, _questions: {}, + ) + runtime, _ = controller( + system_one=True, + fast_path=LayaFastPath(adapter=adapter, calibrator=TemperatureCalibrator({})), + ) + outcome = runtime.run(request()) + self.assertEqual(outcome.reason, "SYSTEM_ONE_PROVIDER_MISMATCH") + + def test_clm_rank_is_advisory_and_abstains_without_calibration(self) -> None: + def rank(payload): + options = payload["answers"] + probability = 1.0 / len(options) + return { + "model": "clm-latest", + "ranked": [ + {"candidate": option, "prob": probability} + for option in options + ], + } + + adapter = ClmAdapter(revision="a" * 64, transport=rank) + runtime, ledger = controller( + system_one=True, + fast_path=CalibratedFastPath( + adapter=adapter, calibrator=TemperatureCalibrator({}) + ), + ) + outcome = runtime.run(request()) + self.assertEqual(outcome.reason, "SYSTEM_ONE_ABSTAINED_NO_PROVIDER") + self.assertEqual(outcome.fast_path.candidate_probabilities, (1.0,)) + self.assertIn('"model_provider":"Contrastive-LM/CLM"', ledger.records[-1].canonical_json) + + def test_clm_outage_escalates_without_granting_authority(self) -> None: + def unavailable(_payload): + raise OSError("CLM unavailable") + + class Deliberator: + def deliberate(self, _state): + return {"plan": ["inspect"]} + + runtime, ledger = controller( + system_one=True, + deliberative=True, + fast_path=CalibratedFastPath( + adapter=ClmAdapter(revision="a" * 64, transport=unavailable), + calibrator=TemperatureCalibrator({}), + ), + deliberator=Deliberator(), + ) + outcome = runtime.run(request()) + self.assertEqual(outcome.route, "deliberative") + self.assertEqual(outcome.reason, "SYSTEM_ONE_FAILURE") + self.assertIsNone(outcome.selected_action_id) + self.assertTrue(TraceLedger.verify(ledger.records)) + def test_provider_usage_is_recorded_without_granting_authority(self) -> None: class Deliberator: def deliberate(self, _state): diff --git a/tests/test_serve.py b/tests/test_serve.py index 17b0f60..3c9a0c4 100644 --- a/tests/test_serve.py +++ b/tests/test_serve.py @@ -4,6 +4,14 @@ from urllib.request import urlopen from c3r.serve import build_servers, load_host_builder +from c3r.staging_host import build as staging_build +from c3r.candidate_compiler import CandidateCompiler +from c3r.cvoc import RobustCvocController +from c3r.feature_flags import FeatureFlags +from c3r.runtime import StandaloneController +from c3r.state_compiler import StateCompiler +from c3r.telemetry.ephemeral import EphemeralTraceSink +from c3r.verifier_firewall import VerifierFirewall, VerifierPolicy from tests.test_http_service import HostFactory from tests.test_runtime import controller @@ -89,6 +97,38 @@ def test_composed_health_path(self): for thread in threads: thread.join(timeout=2) + def test_production_mode_rejects_persistent_and_disabled_hosts(self): + values = config() + values["C3R_MODE"] = "production_inference" + with self.assertRaisesRegex(ValueError, "ephemeral trace sink"): + build_servers( + values, builder_loader=lambda _: lambda: (controller()[0], HostFactory()), + ) + with self.assertRaisesRegex(ValueError, "decisions enabled"): + build_servers(values, builder_loader=lambda _: staging_build) + + def test_production_mode_rejects_collection_and_online_learning(self): + for key in ("C3R_TRACE_COLLECTION", "C3R_ONLINE_LEARNING"): + values = config() + values["C3R_MODE"] = "production_inference" + values[key] = "true" + with self.subTest(key=key), self.assertRaises(ValueError): + build_servers(values, builder_loader=lambda _: staging_build) + + def test_production_mode_requires_system_one_path(self): + runtime = StandaloneController( + flags=FeatureFlags(enabled_requested=True), + compiler=StateCompiler(), candidates=CandidateCompiler(), + cvoc=RobustCvocController(), + verifier=VerifierFirewall({}, VerifierPolicy(default_verifier="none"), + attestation_key=b"test-key"), + ledger=EphemeralTraceSink(), + ) + values = config() + values["C3R_MODE"] = "production_inference" + with self.assertRaisesRegex(ValueError, "System-One"): + build_servers(values, builder_loader=lambda _: lambda: (runtime, HostFactory())) + if __name__ == "__main__": unittest.main() diff --git a/tests/test_stateless_api.py b/tests/test_stateless_api.py new file mode 100644 index 0000000..69db165 --- /dev/null +++ b/tests/test_stateless_api.py @@ -0,0 +1,88 @@ +"""Public HTTP contract for the stateless recommendation-only release.""" + +import json +import threading +import unittest +from urllib.error import HTTPError +from urllib.request import Request, urlopen + +from c3r.http_service import C3RHTTPServer +from tests.test_runtime import controller, request + + +TOKEN = "stateless-test-token-with-at-least-thirty-two-characters" + + +class _Factory: + def build(self, payload): + if payload.get("goal") != "Find record": + raise ValueError("unknown goal") + return request() + + +class StatelessAPITests(unittest.TestCase): + def setUp(self): + runtime, _ = controller() + self.server = C3RHTTPServer( + runtime=runtime, request_factory=_Factory(), bearer_token=TOKEN, + port=0, requests_per_minute=20, + ) + self.thread = threading.Thread(target=self.server.serve_forever, daemon=True) + self.thread.start() + self.base = f"http://127.0.0.1:{self.server.server_port}" + + def tearDown(self): + self.server.shutdown() + self.server.server_close() + self.thread.join(timeout=2) + + def call(self, path, *, method="POST", token=TOKEN, payload=None): + headers = {"Authorization": f"Bearer {token}"} + data = None if payload is None else json.dumps(payload).encode() + if data is not None: + headers["Content-Type"] = "application/json" + req = Request(self.base + path, data=data, headers=headers, method=method) + try: + with urlopen(req, timeout=2) as response: + return response.status, json.load(response) + except HTTPError as error: + return error.code, json.load(error) + + def test_decide_and_rank_are_recommendation_only(self): + for path in ("/v1/c3r/decide", "/v1/c3r/rank", "/v1/system-one"): + status, body = self.call(path, payload={"goal": "Find record"}) + self.assertEqual(status, 200) + self.assertEqual(body["authority_result"], "verified_not_committed") + self.assertFalse(body["effect_executed"]) + self.assertNotIn("confidence", body) + if path.endswith("rank") or path.endswith("system-one"): + self.assertEqual(body["candidate_ranking"], []) + self.assertTrue(body["abstained"]) + + def test_execute_and_untyped_responses_are_unavailable(self): + for path in ("/v1/c3r/execute", "/v1/responses"): + status, body = self.call(path, payload={"goal": "Find record"}) + self.assertEqual(status, 501) + self.assertEqual(body["error"], "not_implemented") + + def test_new_paths_require_authentication(self): + for path in ("/v1/c3r/decide", "/v1/c3r/execute", "/v1/responses"): + status, body = self.call(path, token="wrong", payload={"goal": "Find record"}) + self.assertEqual((status, body["error"]), (401, "unauthorized")) + for path in ("/ready", "/v1/models"): + status, body = self.call(path, method="GET", token="wrong") + self.assertEqual((status, body["error"]), (401, "unauthorized")) + + def test_models_do_not_claim_calibration_or_generation(self): + status, body = self.call("/v1/models", method="GET") + self.assertEqual(status, 200) + self.assertEqual(body["models"][0]["id"], "c3r-core") + self.assertFalse(body["models"][0]["text_generation"]) + self.assertFalse(body["models"][0]["calibrated"]) + self.assertFalse(body["models"][0]["available"]) + status, body = self.call("/ready", method="GET") + self.assertEqual((status, body["status"]), (503, "disabled")) + + +if __name__ == "__main__": + unittest.main() diff --git a/third_party/CLM_VERSION b/third_party/CLM_VERSION new file mode 100644 index 0000000..c34f076 --- /dev/null +++ b/third_party/CLM_VERSION @@ -0,0 +1,8 @@ +Repository: https://github.com/Contrastive-LM/CLM +Source commit: bb42c6c5bf914fd449bed2f6ca65be80602cb1f7 +Source license: Apache-2.0 +Integration API: POST /v1/rank + +This pins reviewed upstream source only. The Qwen encoder, projection heads, +and deployed container require independent immutable hashes and verification. +No CLM weights or source tree are vendored in this repository. From 247d58eabea1dddd6a4546f0dddfabf85d3cd633 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 30 Sep 2026 18:29:21 -0500 Subject: [PATCH 48/55] Implement typed CLM API and close independent review findings --- README.md | 12 ++- c3r/adapters/providers.py | 55 +++++++++- c3r/catalogs/__init__.py | 1 + c3r/catalogs/base.py | 11 ++ c3r/catalogs/browser.py | 6 ++ c3r/catalogs/general.py | 21 ++++ c3r/catalogs/registry.py | 36 +++++++ c3r/catalogs/tools.py | 6 ++ c3r/host_components.py | 15 +++ c3r/http_service.py | 47 +++++++-- c3r/http_transport.py | 8 ++ c3r/ingress_proxy.py | 14 ++- c3r/production_host.py | 136 ++++++++++++++++++++++++ c3r/readiness.py | 26 +++++ c3r/responses.py | 78 ++++++++++++++ c3r/runtime.py | 57 +++++++--- c3r/serve.py | 18 +++- c3r/system_one/advisory.py | 23 +++++ c3r/system_one/clm_adapter.py | 25 +++-- c3r/system_one/inference.py | 146 ++++++++++++++++++++++++++ deploy/Dockerfile.clm | 7 ++ deploy/Dockerfile.clm-hardened | 16 +++ deploy/attest_encoder.py | 40 +++++++ deploy/clm-source-revision | 1 + deploy/clm_runtime.py | 75 ++++++++++++++ deploy/encoder_identity.py | 45 ++++++++ docs/stateless-core-api.md | 68 +++++++++--- scripts/download_clm_artifacts.py | 42 ++++++++ scripts/probe_live_models.py | 38 +++++++ scripts/probe_private_core.py | 63 +++++++++++ scripts/replace_private_candidate.py | 80 ++++++++++++++ scripts/start_private_core.py | 41 ++++++++ tests/test_ingress_proxy.py | 12 ++- tests/test_serve.py | 37 ++++++- tests/test_stateless_api.py | 149 ++++++++++++++++++++++++++- 35 files changed, 1387 insertions(+), 68 deletions(-) create mode 100644 c3r/catalogs/__init__.py create mode 100644 c3r/catalogs/base.py create mode 100644 c3r/catalogs/browser.py create mode 100644 c3r/catalogs/general.py create mode 100644 c3r/catalogs/registry.py create mode 100644 c3r/catalogs/tools.py create mode 100644 c3r/host_components.py create mode 100644 c3r/http_transport.py create mode 100644 c3r/production_host.py create mode 100644 c3r/readiness.py create mode 100644 c3r/responses.py create mode 100644 c3r/system_one/advisory.py create mode 100644 c3r/system_one/inference.py create mode 100644 deploy/Dockerfile.clm create mode 100644 deploy/Dockerfile.clm-hardened create mode 100644 deploy/attest_encoder.py create mode 100644 deploy/clm-source-revision create mode 100644 deploy/clm_runtime.py create mode 100644 deploy/encoder_identity.py create mode 100644 scripts/download_clm_artifacts.py create mode 100644 scripts/probe_live_models.py create mode 100644 scripts/probe_private_core.py create mode 100644 scripts/replace_private_candidate.py create mode 100644 scripts/start_private_core.py diff --git a/README.md b/README.md index 2f87f44..9fcaf45 100644 --- a/README.md +++ b/README.md @@ -35,10 +35,14 @@ not for the separate stateless recommendation API described below. The [C3R Core API v1 contract](docs/stateless-core-api.md) separates a recommendation-only, non-persistent inference service from the governed trace collection and empirical-release program above. The current branch implements -the bounded HTTP routes and a fail-closed `production_inference` mode; it does -**not** mean the API is deployed or production-qualified. Upstream CLM provides +the typed CLM API, direct ranking, a text-only Responses subset, a production +host, and a fail-closed `production_inference` mode. A private, loopback-only +candidate has returned actual CLM rankings and local DeepSeek text; it is +**not** a publicly launched or production-qualified API. Upstream CLM provides advisory System-One ranking, not generative text or calibrated task-success -probabilities. `/v1/c3r/execute` and `/v1/responses` are deliberately disabled. +probabilities. `/v1/c3r/execute` remains disabled. `/v1/responses` invokes +DeepSeek through an admitted, independently checked text-only controller +fallback; it does not claim positive learned CVoC or expose private reasoning. The first public hostname is a dedicated C3R endpoint, not an MC-1 integration. See the [scope-specific release policy](docs/release-policy.md). @@ -251,4 +255,4 @@ Do not report vulnerabilities in a public issue; follow [`SECURITY.md`](SECURITY effects must be completely mediated by an independently configured commit gateway. Apache License 2.0. See [`LICENSE`](LICENSE). If you use C3R, cite [`CITATION.cff`](CITATION.cff). - + diff --git a/c3r/adapters/providers.py b/c3r/adapters/providers.py index 2605d5e..a394bc0 100644 --- a/c3r/adapters/providers.py +++ b/c3r/adapters/providers.py @@ -3,15 +3,16 @@ from __future__ import annotations import json -from math import isfinite from collections.abc import Callable, Mapping from dataclasses import dataclass from enum import StrEnum +from math import isfinite from typing import Protocol, cast from urllib.parse import urlparse -from urllib.request import Request, urlopen +from urllib.request import ProxyHandler, Request, build_opener from ..deliberative.envelope import DeliberativeResult +from ..http_transport import NoRedirectHandler from ..state_schema import ActionFamily MAX_RESPONSE_BYTES = 65_536 @@ -41,9 +42,11 @@ def __post_init__(self) -> None: local = parsed.hostname in {"localhost", "127.0.0.1", "::1"} if parsed.scheme != "https" and not (parsed.scheme == "http" and local): raise ValueError("remote provider endpoints must use HTTPS") + if parsed.username or parsed.password or parsed.query or parsed.fragment: + raise ValueError("provider origin must not contain credentials, query, or fragment") if not self.provider_id or not self.model: raise ValueError("provider_id and model are required") - if self.timeout_seconds <= 0: + if not isfinite(self.timeout_seconds) or not 0 < self.timeout_seconds <= 120: raise ValueError("timeout_seconds must be positive") @@ -121,7 +124,7 @@ def _default_transport( method="POST", ) started = time.monotonic() - with urlopen(request, timeout=timeout) as response: + with build_opener(ProxyHandler({}), NoRedirectHandler()).open(request, timeout=timeout) as response: raw = response.read(MAX_RESPONSE_BYTES + 1) status = int(response.status) if len(raw) > MAX_RESPONSE_BYTES: @@ -245,6 +248,50 @@ def __init__( self._transport = transport self._codec = _CODECS[config.kind] + def generate(self, text: str, max_output_tokens: int) -> tuple[str, str, dict[str, float]]: + """Bounded text-only OpenAI-compatible generation; never returns hidden reasoning.""" + if self.config.kind not in {ProviderKind.OPENAI, ProviderKind.OPENAI_COMPATIBLE}: + raise ValueError("text generation requires an OpenAI-compatible provider") + if not text or len(text.encode("utf-8")) > 16384 or not 1 <= max_output_tokens <= 2048: + raise ValueError("generation request outside bounds") + headers = {"X-C3R-Timeout": str(self.config.timeout_seconds)} + if self.config.api_key: + headers["Authorization"] = "Bearer " + self.config.api_key + try: + response = self._transport(self.config.base_url.rstrip("/") + "/chat/completions", + headers, { + "model": self.config.model, "max_tokens": max_output_tokens, "temperature": 0, + "messages": [ + {"role": "system", "content": "Provide a helpful final answer only. Do not expose " + "private reasoning or claim to execute tools, commit actions, or grant authority."}, + {"role": "user", "content": text}, + ], + }) + size = len(json.dumps(response.body, allow_nan=False).encode()) + except (OSError, ValueError, TypeError) as error: + raise RuntimeError("generative provider transport unavailable") from error + if not 200 <= response.status < 300: + raise RuntimeError("generative provider unavailable") + if size > MAX_RESPONSE_BYTES: + raise RuntimeError("generative provider response oversized") + try: + choices = cast(list[dict[str, object]], response.body["choices"]) + message = cast(dict[str, object], choices[0]["message"]) + if message.get("tool_calls") or message.get("function_call"): + raise ValueError("tool execution is not supported") + output = _content(message["content"]) + reason = choices[0].get("finish_reason") + if not output.strip() or reason not in {"stop", "length"}: + raise ValueError("invalid generation result") + usage = cast(Mapping[str, object], response.body.get("usage", {})) + return output, cast(str, reason), { + "input_tokens": _number(usage.get("prompt_tokens")), + "output_tokens": _number(usage.get("completion_tokens")), + "latency_ms": response.latency_ms, + } + except (KeyError, IndexError, TypeError, ValueError, AttributeError) as error: + raise RuntimeError("invalid generative provider response") from error + def deliberate(self, request: DeliberationRequest) -> ProviderExecutionResult: state = json.dumps(dict(request.state), sort_keys=True, separators=(",", ":")) url, headers, payload = self._codec.build(self.config, state) diff --git a/c3r/catalogs/__init__.py b/c3r/catalogs/__init__.py new file mode 100644 index 0000000..7cd53df --- /dev/null +++ b/c3r/catalogs/__init__.py @@ -0,0 +1 @@ +"""Host-owned recommendation catalogs; none grants tool execution.""" diff --git a/c3r/catalogs/base.py b/c3r/catalogs/base.py new file mode 100644 index 0000000..2ee9a9e --- /dev/null +++ b/c3r/catalogs/base.py @@ -0,0 +1,11 @@ +"""Immutable host-owned action definitions.""" +from dataclasses import dataclass + +from ..state_schema import ActionDefinition, AuthorityPolicy + + +@dataclass(frozen=True, slots=True) +class Catalog: + name: str + definitions: tuple[ActionDefinition, ...] + policy: AuthorityPolicy diff --git a/c3r/catalogs/browser.py b/c3r/catalogs/browser.py new file mode 100644 index 0000000..3eba84d --- /dev/null +++ b/c3r/catalogs/browser.py @@ -0,0 +1,6 @@ +"""Browser execution is deliberately unavailable in stateless v1.""" +from .general import general_catalog + + +def browser_catalog(): + return general_catalog("browser-v1") diff --git a/c3r/catalogs/general.py b/c3r/catalogs/general.py new file mode 100644 index 0000000..2d234fa --- /dev/null +++ b/c3r/catalogs/general.py @@ -0,0 +1,21 @@ +"""Compute recommendations only, not implemented tool capabilities.""" +from ..state_schema import ActionDefinition, ActionFamily, AuthorityPolicy, RiskClass +from .base import Catalog + + +def general_catalog(name: str = "agent-v1") -> Catalog: + operations = ( + ("STOP", ActionFamily.STOP), ("WAIT", ActionFamily.STOP), + ("VERIFY", ActionFamily.VERIFY), ("RETRIEVE", ActionFamily.RETRIEVAL), + ("SYSTEM_ONE", ActionFamily.LOCAL_MODEL), ("DELIBERATE", ActionFamily.DELIBERATE), + ("CALL_LOCAL_MODEL", ActionFamily.LOCAL_MODEL), + ("CALL_FRONTIER_MODEL", ActionFamily.FRONTIER_MODEL), ("ASK_USER", ActionFamily.ASK_USER), + ) + definitions = tuple(ActionDefinition( + id=operation, family=family, subgroup="compute", operation=operation, + risk_class=RiskClass.READ_ONLY, argument_variants=((),), placements=("local",), + verifier_ids=("recommendation",), optimistic_utility=0, estimated_cost=0, + ) for operation, family in operations) + return Catalog(name, definitions, AuthorityPolicy( + frozenset(family for _, family in operations), frozenset({RiskClass.READ_ONLY}), + allowed_data_boundaries=frozenset({"local"}))) diff --git a/c3r/catalogs/registry.py b/c3r/catalogs/registry.py new file mode 100644 index 0000000..38e0075 --- /dev/null +++ b/c3r/catalogs/registry.py @@ -0,0 +1,36 @@ +"""Admission of caller task text to registered, host-owned catalogs.""" +import json +from collections.abc import Mapping + +from ..host_factory import ReadOnlyRequestFactory +from ..runtime import RuntimeRequest +from .base import Catalog + + +class CatalogRegistry: + def __init__(self, catalogs: tuple[Catalog, ...]) -> None: + self._factories = {catalog.name: ReadOnlyRequestFactory( + definitions=catalog.definitions, policy=catalog.policy, + # Unpriced/unknown quality has no positive CVoC estimate. No invented + # dollar cost or success probability is admitted as measured evidence. + estimate_source=lambda _state: {}, remaining_usd=0, + model_inventory=("c3r-system-one", "deepseek"), data_boundary="local", + ) for catalog in catalogs} + + def build(self, payload: Mapping[str, object]) -> RuntimeRequest: + if set(payload) - {"goal", "state", "current_subgoal", "open_questions", + "catalog", "application_id"}: + raise ValueError("caller authority or unsupported fields rejected") + catalog = payload.get("catalog", "agent-v1") + if not isinstance(catalog, str) or catalog not in self._factories: + raise ValueError("unregistered catalog") + application = payload.get("application_id") + if application is not None and (not isinstance(application, str) or len(application) > 128): + raise ValueError("invalid application identifier") + state = payload.get("state", payload.get("current_subgoal", payload.get("goal"))) + if isinstance(state, dict): + state = json.dumps(state, ensure_ascii=False, allow_nan=False) + return self._factories[catalog].build({ + "goal": payload.get("goal"), "current_subgoal": state, + "open_questions": payload.get("open_questions", []), + }) diff --git a/c3r/catalogs/tools.py b/c3r/catalogs/tools.py new file mode 100644 index 0000000..3319e8f --- /dev/null +++ b/c3r/catalogs/tools.py @@ -0,0 +1,6 @@ +"""Routing recommendations only; no arbitrary tool or URL invocation.""" +from .general import general_catalog + + +def tool_catalog(): + return general_catalog("tool-routing-v1") diff --git a/c3r/host_components.py b/c3r/host_components.py new file mode 100644 index 0000000..f92693d --- /dev/null +++ b/c3r/host_components.py @@ -0,0 +1,15 @@ +"""Host composition shared by imported and module-executed entrypoints.""" +from dataclasses import dataclass + +from .http_service import RequestFactory +from .responses import ResponsesService +from .runtime import StandaloneController +from .system_one.inference import SystemOneInference + + +@dataclass(frozen=True, slots=True) +class HostComponents: + runtime: StandaloneController + factory: RequestFactory + system_one: SystemOneInference + responses: ResponsesService diff --git a/c3r/http_service.py b/c3r/http_service.py index c2ddfa6..0f99445 100644 --- a/c3r/http_service.py +++ b/c3r/http_service.py @@ -9,16 +9,17 @@ import hmac import json import time -from math import isfinite from collections.abc import Callable, Mapping from dataclasses import asdict, is_dataclass from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from math import isfinite from threading import Lock from typing import Protocol, cast from .deliberative.envelope import DeliberativeResult +from .responses import ResponsesService from .runtime import RuntimeRequest, StandaloneController - +from .system_one.inference import SystemOneInference MAX_REQUEST_BYTES = 65_536 @@ -85,6 +86,8 @@ def __init__( host: str = "127.0.0.1", port: int = 8081, requests_per_minute: int = 60, + system_one: SystemOneInference | None = None, + responses: ResponsesService | None = None, ) -> None: if host not in {"127.0.0.1", "::1", "localhost"}: raise ValueError("C3R must bind to loopback behind a TLS gateway") @@ -93,6 +96,8 @@ def __init__( if runtime.effect_execution_enabled: raise ValueError("the HTTP service cannot execute external effects") self.runtime = runtime + self.system_one = system_one + self.responses = responses self.request_factory = request_factory self.bearer_token = bearer_token self.limiter = TokenBucket( @@ -121,8 +126,8 @@ def _send(self, status: int, value: Mapping[str, object]) -> None: def _authorized(self) -> bool: expected = "Bearer " + self.server.bearer_token - supplied = self.headers.get("Authorization", "") - return hmac.compare_digest(expected, supplied) + supplied = self.headers.get_all("Authorization", []) + return len(supplied) == 1 and hmac.compare_digest(expected.encode(), supplied[0].encode()) def do_GET(self) -> None: if self.path == "/health": @@ -144,15 +149,23 @@ def do_GET(self) -> None: available = (self.server.runtime.decision_enabled and self.server.runtime.system_one_enabled and self.server.runtime.provider_ready) - self._send(200, {"models": [ + ranking_available = (self.server.runtime.decision_enabled + and self.server.runtime.system_one_enabled + and self.server.system_one is not None + and self.server.system_one.ready) + models = [ {"id": "c3r-core", "capability": "verified_recommendation", - "text_generation": False, "calibrated": False, + "text_generation": self.server.responses is not None, "calibrated": False, "effect_execution": False, "available": available}, {"id": "c3r-system-one", "capability": "advisory_ranking", "text_generation": False, "calibrated": False, "effect_execution": False, - "available": available and self.server.runtime.system_one_enabled}, - ]}) + "available": ranking_available}, + {"id": "c3r-verifier", "capability": "advisory_output_ranking", + "text_generation": False, "calibrated": False, "effect_execution": False, + "available": ranking_available}, + ] + self._send(200, {"object": "list", "data": models, "models": models}) return if self.path == "/metrics": if not self._authorized(): @@ -177,7 +190,8 @@ def do_POST(self) -> None: self.server.metrics.increment("rate_limited") self._send(429, {"error": "rate_limited"}) return - if self.path in {"/v1/c3r/execute", "/v1/responses"}: + if self.path == "/v1/c3r/execute" or (self.path == "/v1/responses" and + self.server.responses is None): self._send(501, {"error": "not_implemented", "reason": "recommendation_only"}) return try: @@ -192,9 +206,22 @@ def do_POST(self) -> None: payload = json.loads(self.rfile.read(length)) if not isinstance(payload, dict): raise ValueError("JSON object required") + if self.path == "/v1/responses" and self.server.responses: + response = self.server.responses.respond(cast(dict[str, object], payload)) + self.server.metrics.increment("text_responses") + self._send(200, response) + return + if self.path in {"/v1/c3r/rank", "/v1/system-one"} and self.server.system_one: + if not (self.server.runtime.decision_enabled + and self.server.runtime.system_one_enabled): + raise RuntimeError("System-One disabled") + result = self.server.system_one.infer(cast(dict[str, object], payload)) + self.server.metrics.increment("system_one_inferences") + self._send(200, result) + return request = self.server.request_factory.build(cast(dict[str, object], payload)) outcome = self.server.runtime.run(request) - except (KeyError, TypeError, ValueError, json.JSONDecodeError): + except (KeyError, TypeError, ValueError, RecursionError, json.JSONDecodeError): self.server.metrics.increment("invalid_request") self._send(400, {"error": "invalid_request"}) return diff --git a/c3r/http_transport.py b/c3r/http_transport.py new file mode 100644 index 0000000..a894ff9 --- /dev/null +++ b/c3r/http_transport.py @@ -0,0 +1,8 @@ +"""Redirects must not move local state or authorization to another origin.""" +from urllib.request import HTTPRedirectHandler, Request + + +class NoRedirectHandler(HTTPRedirectHandler): + def redirect_request(self, req: Request, fp: object, code: int, + msg: str, headers: object, newurl: str) -> None: + return None diff --git a/c3r/ingress_proxy.py b/c3r/ingress_proxy.py index a28e3d8..8a17d6d 100644 --- a/c3r/ingress_proxy.py +++ b/c3r/ingress_proxy.py @@ -14,7 +14,6 @@ from .http_service import MAX_REQUEST_BYTES, TokenBucket - MAX_RESPONSE_BYTES = 65_536 @@ -76,7 +75,16 @@ def _send_error(self, status: int, code: str) -> None: def _authorized(self) -> bool: supplied = self.headers.get_all("X-C3R-Token", []) - return len(supplied) == 1 and hmac.compare_digest(self.server.client_token, supplied[0]) + authorization = self.headers.get_all("Authorization", []) + if len(authorization) > 1 or len(supplied) > 1: + return False + # Preserve IAM staging's separate client header. Without that header, + # SDK clients use standard Bearer auth; it is never relayed upstream. + if supplied: + return hmac.compare_digest(self.server.client_token.encode(), supplied[0].encode()) + return (len(authorization) == 1 and + hmac.compare_digest(("Bearer " + self.server.client_token).encode(), + authorization[0].encode())) def _forward(self, method: str, body: bytes | None = None) -> None: if not self.server.in_flight.acquire(blocking=False): @@ -150,4 +158,4 @@ def do_POST(self) -> None: self._send_error(413, "request_size_out_of_bounds") return self._forward("POST", self.rfile.read(length)) - + diff --git a/c3r/production_host.py b/c3r/production_host.py new file mode 100644 index 0000000..a966665 --- /dev/null +++ b/c3r/production_host.py @@ -0,0 +1,136 @@ +"""Production composition for stateless, non-authoritative inference only.""" +from __future__ import annotations + +import json +import os +from collections.abc import Mapping +from secrets import token_bytes +from typing import cast +from urllib.request import ProxyHandler, build_opener + +from .adapters.providers import ProviderAdapter, ProviderConfig, ProviderKind +from .candidate_compiler import CandidateCompiler +from .catalogs.browser import browser_catalog +from .catalogs.general import general_catalog +from .catalogs.registry import CatalogRegistry +from .catalogs.tools import tool_catalog +from .cvoc import RobustCvocController +from .feature_flags import FeatureFlags +from .host_components import HostComponents +from .http_transport import NoRedirectHandler +from .readiness import CachedReadiness +from .responses import ResponsesService +from .runtime import StandaloneController +from .state_compiler import StateCompiler +from .state_schema import RiskClass +from .system_one.advisory import AdvisoryFastPath +from .system_one.clm_adapter import UPSTREAM_CLM_COMMIT, ClmAdapter +from .system_one.inference import SystemOneInference +from .telemetry.ephemeral import EphemeralTraceSink +from .verifier_firewall import VerifierDecision, VerifierFirewall, VerifierPolicy + +ENCODER_REVISION = "b968826d9c46dd6066d109eabc6255188de91218" +HEAD_REVISION = "e939398d4556fcd9400c76fa8c5a513202f42b0a" +HEAD_SHA256 = "b2b4a8c9c2d39263eff78a351eb909a342ce9b3bf21a3f07c1d1bf15f1c4eda5" + + +class ProviderReadiness: + def __init__(self, adapter: ClmAdapter, container_digest: str) -> None: + self.adapter, self.container_digest = adapter, container_digest + self.system_one = CachedReadiness(self._system_one) + self.deliberative = CachedReadiness(self._deliberative) + + def _get(self, url: str) -> Mapping[str, object]: + with build_opener(ProxyHandler({}), NoRedirectHandler()).open(url, timeout=2) as response: + body = response.read(65537) + if len(body) > 65536: + raise ValueError("oversized provider readback") + value = json.loads(body) + if not isinstance(value, dict): + raise ValueError("invalid provider readback") + return cast(Mapping[str, object], value) + + def _system_one(self) -> bool: + try: + models = self._get("http://127.0.0.1:8090/v1/models") + rows = models.get("data") + if not isinstance(rows, list) or not any( + isinstance(row, dict) and cast(Mapping[str, object], row).get("id") == "qwen3-8b" + and cast(Mapping[str, object], row).get("root") == "/encoder" + for row in cast(list[object], rows) + ): + return False + health = self._get(self.adapter.endpoint + "/health") + if health.get("embedder") is not True or health.get("mock", False) is not False: + return False + artifact = self._get(self.adapter.endpoint + "/internal/clm/artifact") + expected = { + "clm_source_revision": UPSTREAM_CLM_COMMIT, "encoder": "Qwen/Qwen3-8B", + "encoder_revision": ENCODER_REVISION, "head_revision": HEAD_REVISION, + "head_sha256": HEAD_SHA256, "container_digest": self.container_digest, + "embedding_cache_size": 0, "action_cache_enabled": False, + "encoder_content_verified": True, + "encoder_identity_basis": "immutable_upstream_git_blobs_and_lfs_sha256", + } + if any(artifact.get(key) != value for key, value in expected.items()): + return False + scores = self.adapter.rank_text("An invoice was charged twice.", + "Which department handles billing?", + ("Billing", "Technical")) + return len(scores) == 2 + except (OSError, ValueError, TypeError, KeyError): + return False + + def _deliberative(self) -> bool: + try: + rows = self._get("http://127.0.0.1:8000/v1/models").get("data") + listed = isinstance(rows, list) and any( + isinstance(row, dict) and cast(Mapping[str, object], row).get("id") == "/model" + for row in cast(list[object], rows)) + if not listed: + return False + probe = ProviderAdapter(ProviderConfig( + "deepseek-readiness", ProviderKind.OPENAI_COMPATIBLE, + "http://127.0.0.1:8000/v1", "/model", None, timeout_seconds=5, + )) + text, _, _ = probe.generate("Reply with the word ready only.", 256) + return bool(text.strip()) + except (OSError, RuntimeError, ValueError, TypeError): + return False + + def all(self) -> bool: + return self.system_one() and self.deliberative() + + +def build() -> HostComponents: + flags = FeatureFlags.from_mapping(os.environ) + if (not flags.enabled_requested or not flags.system_one_enabled + or not flags.deliberative_enabled or flags.system_one_provider != "clm"): + raise ValueError("production host requires enabled CLM and deliberative inference") + if os.environ.get("C3R_MODE") != "production_inference": + raise ValueError("production host requires production_inference mode") + for field in ("C3R_TRACE_COLLECTION", "C3R_ONLINE_LEARNING"): + if os.environ.get(field, "false").lower() not in {"false", "off", "0"}: + raise ValueError("production host cannot collect or learn online") + container = os.environ.get("C3R_CLM_CONTAINER_DIGEST", "") + if len(container) != 71 or not container.startswith("sha256:"): + raise ValueError("measured immutable CLM container identity required") + adapter = ClmAdapter(HEAD_SHA256, timeout_seconds=3) + readiness = ProviderReadiness(adapter, container) + registry = CatalogRegistry((general_catalog(), browser_catalog(), tool_catalog(), + general_catalog("research-v1"))) + verifier = VerifierFirewall({"recommendation": lambda candidate: VerifierDecision( + candidate.risk_class is RiskClass.READ_ONLY, + "read-only recommendation; no tool execution or truth claim", + )}, VerifierPolicy("recommendation"), attestation_key=token_bytes(32)) + runtime = StandaloneController( + flags=flags, compiler=StateCompiler(), candidates=CandidateCompiler(), + cvoc=RobustCvocController(), verifier=verifier, ledger=EphemeralTraceSink(), + fast_path=AdvisoryFastPath(adapter), readiness_probe=readiness.all, + ) + provider = ProviderAdapter(ProviderConfig( + "deepseek-local", ProviderKind.OPENAI_COMPATIBLE, + "http://127.0.0.1:8000/v1", "/model", None, timeout_seconds=60, + )) + return HostComponents(runtime, registry, SystemOneInference(adapter, readiness=readiness.system_one), + ResponsesService(runtime, registry, provider)) diff --git a/c3r/readiness.py b/c3r/readiness.py new file mode 100644 index 0000000..b3ca82c --- /dev/null +++ b/c3r/readiness.py @@ -0,0 +1,26 @@ +"""Coalesced bounded-TTL health checks; retain only a boolean and expiry.""" +from collections.abc import Callable +from threading import Lock +from time import monotonic + + +class CachedReadiness: + def __init__(self, probe: Callable[[], bool], *, ttl_seconds: float = 15, + clock: Callable[[], float] = monotonic) -> None: + if not 0 < ttl_seconds <= 30: + raise ValueError("readiness TTL must be between zero and 30 seconds") + self._probe, self._ttl, self._clock = probe, ttl_seconds, clock + self._lock = Lock() + self._value = False + self._expires = float("-inf") + + def __call__(self) -> bool: + with self._lock: + if self._clock() < self._expires: + return self._value + try: + self._value = bool(self._probe()) + except (OSError, RuntimeError, ValueError, TypeError, KeyError): + self._value = False + self._expires = self._clock() + self._ttl + return self._value diff --git a/c3r/responses.py b/c3r/responses.py new file mode 100644 index 0000000..efae2a1 --- /dev/null +++ b/c3r/responses.py @@ -0,0 +1,78 @@ +"""Text-only, non-storing Responses subset with an explicit host policy fallback.""" +from __future__ import annotations + +from collections.abc import Mapping +from dataclasses import dataclass, replace +from typing import Protocol +from uuid import uuid4 + +from .adapters.providers import ProviderAdapter +from .runtime import RuntimeRequest, StandaloneController +from .state_schema import CompiledState + + +@dataclass(frozen=True, slots=True) +class TextGenerationResult: + text: str + finish: str + usage: dict[str, float] + + +class TextDeliberator: + def __init__(self, provider: ProviderAdapter, maximum: int) -> None: + self.provider, self.maximum = provider, maximum + + def deliberate(self, state: CompiledState) -> TextGenerationResult: + return TextGenerationResult(*self.provider.generate(state.goal, self.maximum)) + + +class GenerationRequestFactory(Protocol): + def build(self, payload: Mapping[str, object]) -> RuntimeRequest: ... + + +class ResponsesService: + def __init__(self, runtime: StandaloneController, factory: GenerationRequestFactory, + provider: ProviderAdapter) -> None: + self.runtime, self.factory, self.provider = runtime, factory, provider + + def respond(self, payload: Mapping[str, object]) -> dict[str, object]: + if set(payload) - {"model", "input", "max_output_tokens", "store", "stream"}: + raise ValueError("unsupported Responses fields") + if payload.get("model") != "c3r-core": + raise ValueError("only c3r-core supports text generation") + if payload.get("store", False) is not False or payload.get("stream", False) is not False: + raise ValueError("storage and streaming are not supported") + text = payload.get("input") + maximum = payload.get("max_output_tokens", 512) + if (not isinstance(text, str) or not text.strip() or len(text) > 4096 + or len(text.encode()) > 16384 + or isinstance(maximum, bool) or not isinstance(maximum, int) + or not 1 <= maximum <= 2048): + raise ValueError("bounded text input and token limit required") + if not self.runtime.decision_enabled: + raise RuntimeError("C3R disabled") + request = replace(self.factory.build({"goal": text, "current_subgoal": text}), + requested_text_generation=True) + decision = self.runtime.run(request, requested_deliberator=TextDeliberator(self.provider, maximum)) + if not isinstance(decision.deliberation, TextGenerationResult): + raise RuntimeError("controller declined text generation") + # An explicit caller request for bounded text-only generation is the host + # fallback when unknown task quality gives no positive CVoC. It is NOT a + # positive learned utility claim and cannot execute a catalog action. + output, finish, usage = (decision.deliberation.text, decision.deliberation.finish, + decision.deliberation.usage) + identifier = "resp_" + uuid4().hex + return { + "id": identifier, "object": "response", "model": "c3r-core", + "status": "completed" if finish == "stop" else "incomplete", "store": False, + "output": [{"id": "msg_" + uuid4().hex, "type": "message", "role": "assistant", + "status": "completed" if finish == "stop" else "incomplete", + "content": [{"type": "output_text", "text": output, "annotations": []}]}], + "usage": {"input_tokens": int(usage["input_tokens"]), + "output_tokens": int(usage["output_tokens"]), + "total_tokens": int(usage["input_tokens"] + usage["output_tokens"])}, + "c3r": {"route": "deliberative", "system_one_provider": "clm", + "controller_reason": decision.reason, "effect_executed": False, + "selection_basis": "explicit_text_only_request_policy_fallback", + "calibrated": False, "provider": self.provider.config.provider_id}, + } diff --git a/c3r/runtime.py b/c3r/runtime.py index 6527be8..6afb234 100644 --- a/c3r/runtime.py +++ b/c3r/runtime.py @@ -8,9 +8,9 @@ import hashlib import json -from math import isfinite from collections.abc import Callable, Mapping from dataclasses import asdict, dataclass +from math import isfinite from typing import Protocol from .adapters.providers import ProviderExecutionResult @@ -28,10 +28,11 @@ RiskClass, ValueEstimate, ) +from .system_one.advisory import AdvisoryFastPath from .system_one.fast_path import CalibratedFastPath, FastPathDecision from .system_one.question_registry import TypedQuestion -from .telemetry.trace import DecisionTrace from .telemetry.ephemeral import EphemeralTraceSink +from .telemetry.trace import DecisionTrace from .telemetry.trace_ledger import LedgerRecord from .verifier_firewall import VerifierFirewall @@ -53,6 +54,7 @@ class RuntimeRequest: run_id: str access_level: str = "internal" language: str = "en" + requested_text_generation: bool = False @dataclass(frozen=True, slots=True) @@ -85,7 +87,7 @@ def __init__( cvoc: RobustCvocController, verifier: VerifierFirewall, ledger: TraceSink, - fast_path: CalibratedFastPath | None = None, + fast_path: CalibratedFastPath | AdvisoryFastPath | None = None, deliberator: Deliberator | None = None, executor: Callable[[ActionCandidate], None] | None = None, readiness_probe: Callable[[], bool] | None = None, @@ -131,7 +133,8 @@ def provider_ready(self) -> bool: except (OSError, RuntimeError, TypeError, ValueError): return False - def run(self, request: RuntimeRequest) -> RuntimeOutcome: + def run(self, request: RuntimeRequest, *, + requested_deliberator: Deliberator | None = None) -> RuntimeOutcome: if not request.run_id: raise ValueError("run_id is required") state_hash = hashlib.sha256( @@ -233,30 +236,45 @@ def finish( candidate_options=candidate_options, ) except (OSError, RuntimeError, TypeError, ValueError): - return self._deliberate_or_stop( - state, finish, candidate_ids, "SYSTEM_ONE_FAILURE", None - ) - if fast.abstained: + if not request.requested_text_generation: + return self._deliberate_or_stop( + state, finish, candidate_ids, "SYSTEM_ONE_FAILURE", None + ) + if fast is not None and fast.abstained: return self._deliberate_or_stop( state, finish, candidate_ids, "SYSTEM_ONE_ABSTAINED", fast ) - if fast.answers.get("STOP_NOW") == "YES": + if fast is not None and fast.answers.get("STOP_NOW") == "YES": return finish("system_one", "STOP_NOW", candidate_ids=candidate_ids, fast=fast) - if fast.answers.get("DELIBERATION_REQUIRED") == "YES": + if fast is not None and fast.answers.get("DELIBERATION_REQUIRED") == "YES": return self._deliberate_or_stop( state, finish, candidate_ids, "DELIBERATION_REQUIRED", fast ) decision = self._cvoc.select(compiled.candidates, request.estimates) if decision.selected is None: + if request.requested_text_generation and requested_deliberator is not None: + # Trusted host opt-in for caller-requested bounded text only. + # Unknown quality still has no positive CVoC. This fallback must + # be admitted by the catalog AND independently verified. + fallback = next((candidate for candidate in compiled.candidates + if candidate.family is ActionFamily.DELIBERATE + and candidate.risk_class is RiskClass.READ_ONLY), None) + if fallback is not None: + try: + verification = self._verifier.verify(fallback) + except (OSError, RuntimeError, TypeError, ValueError): + return finish("deterministic", "VERIFIER_FAILURE", candidate_ids=candidate_ids) + if verification.accepted: + return self._deliberate_or_stop( + state, finish, candidate_ids, "REQUESTED_TEXT_POLICY_FALLBACK", fast, + deliberator=requested_deliberator, + ) + return finish("deterministic", "VERIFICATION_REJECTED", candidate_ids=candidate_ids) return finish( "deterministic", "NON_POSITIVE_CVOC", candidate_ids=candidate_ids, fast=fast ) selected = decision.selected - if selected.family is ActionFamily.DELIBERATE: - return self._deliberate_or_stop( - state, finish, candidate_ids, "CVOC_SELECTED_DELIBERATION", fast - ) try: verification = self._verifier.verify(selected) except (OSError, RuntimeError, TypeError, ValueError): @@ -267,6 +285,11 @@ def finish( return finish( "deterministic", "VERIFICATION_REJECTED", candidate_ids=candidate_ids, fast=fast ) + if selected.family is ActionFamily.DELIBERATE: + return self._deliberate_or_stop( + state, finish, candidate_ids, "CVOC_SELECTED_DELIBERATION", fast, + deliberator=requested_deliberator if request.requested_text_generation else None, + ) if selected.risk_class is not RiskClass.READ_ONLY: return finish( "deterministic", @@ -292,13 +315,15 @@ def _deliberate_or_stop( candidate_ids: tuple[str, ...], reason: str, fast: FastPathDecision | None, + *, deliberator: Deliberator | None = None, ) -> RuntimeOutcome: - if not self._flags.deliberative_enabled or self._deliberator is None: + deliberator = deliberator or self._deliberator + if not self._flags.deliberative_enabled or deliberator is None: return finish( "deterministic", reason + "_NO_PROVIDER", candidate_ids=candidate_ids, fast=fast ) try: - deliberation = self._deliberator.deliberate(state) + deliberation = deliberator.deliberate(state) except (OSError, RuntimeError, TypeError, ValueError): return finish( "deterministic", "DELIBERATIVE_FAILURE", candidate_ids=candidate_ids, fast=fast diff --git a/c3r/serve.py b/c3r/serve.py index abc6c59..1d1fe16 100644 --- a/c3r/serve.py +++ b/c3r/serve.py @@ -13,13 +13,14 @@ from collections.abc import Callable, Mapping from typing import Protocol, cast +from .host_components import HostComponents from .http_service import C3RHTTPServer, RequestFactory from .ingress_proxy import C3RIngressServer from .runtime import StandaloneController class HostBuilder(Protocol): - def __call__(self) -> tuple[StandaloneController, RequestFactory]: ... + def __call__(self) -> tuple[StandaloneController, RequestFactory] | HostComponents: ... def _required(values: Mapping[str, str], name: str) -> str: @@ -63,9 +64,16 @@ def build_servers( backend_token = _required(values, "C3R_BACKEND_TOKEN") port = _port(values, "PORT", 8080) backend_port = _port(values, "C3R_BACKEND_PORT", 8081) + ingress_host = values.get("C3R_INGRESS_HOST", "0.0.0.0") + if ingress_host not in {"0.0.0.0", "127.0.0.1"}: + raise ValueError("ingress host must be loopback or the TLS-host container interface") if port == backend_port: raise ValueError("ingress and backend ports must differ") - runtime, factory = builder_loader(reference)() + components = builder_loader(reference)() + if isinstance(components, HostComponents): + runtime, factory = components.runtime, components.factory + else: + runtime, factory = components if runtime.effect_execution_enabled: raise ValueError("host must be recommendation-only") mode = values.get("C3R_MODE", "staging") @@ -87,6 +95,8 @@ def build_servers( request_factory=factory, bearer_token=backend_token, port=backend_port, + system_one=components.system_one if isinstance(components, HostComponents) else None, + responses=components.responses if isinstance(components, HostComponents) else None, ) try: ingress = C3RIngressServer( @@ -94,6 +104,8 @@ def build_servers( client_token=client_token, upstream_token=backend_token, port=port, + host=ingress_host, + upstream_timeout_seconds=65 if mode == "production_inference" else 5, ) except BaseException: backend.server_close() @@ -135,4 +147,4 @@ def request_stop(_signum: int, _frame: object) -> None: if __name__ == "__main__": main() - + diff --git a/c3r/system_one/advisory.py b/c3r/system_one/advisory.py new file mode 100644 index 0000000..90e4704 --- /dev/null +++ b/c3r/system_one/advisory.py @@ -0,0 +1,23 @@ +"""Uncalibrated ranker which makes no learned control/authority decisions.""" +from collections.abc import Sequence + +from ..state_schema import CompiledState +from .clm_adapter import ClmAdapter +from .fast_path import FastPathDecision +from .question_registry import TypedQuestion + + +class AdvisoryFastPath: + def __init__(self, adapter: ClmAdapter) -> None: + self.adapter = adapter + + @property + def provider(self) -> str: + return self.adapter.provider + + def decide(self, state: CompiledState, questions: Sequence[TypedQuestion], *, + action_family: str, language: str = "en", + candidate_options: tuple[str, ...] = ()) -> FastPathDecision: + scores = self.adapter.rank_actions(state, candidate_options) if candidate_options else () + return FastPathDecision({}, {}, False, ("UNCALIBRATED_ADVISORY_ONLY",), + self.adapter.model_id, self.adapter.revision, scores) diff --git a/c3r/system_one/clm_adapter.py b/c3r/system_one/clm_adapter.py index b5eb372..1f11cb1 100644 --- a/c3r/system_one/clm_adapter.py +++ b/c3r/system_one/clm_adapter.py @@ -6,20 +6,20 @@ from __future__ import annotations -from collections.abc import Callable, Mapping -from dataclasses import asdict, dataclass import json import math import re import time +from collections.abc import Callable, Mapping +from dataclasses import asdict, dataclass from typing import cast from urllib.parse import urlsplit -from urllib.request import HTTPHandler, HTTPRedirectHandler, ProxyHandler, Request, build_opener +from urllib.request import HTTPHandler, ProxyHandler, Request, build_opener +from ..http_transport import NoRedirectHandler from ..state_schema import CompiledState from .question_registry import TypedQuestion - UPSTREAM_CLM_COMMIT = "bb42c6c5bf914fd449bed2f6ca65be80602cb1f7" _IMMUTABLE_REVISION = re.compile(r"^[0-9a-f]{40,64}$") _MAX_CONTEXT_BYTES = 32_768 @@ -28,12 +28,6 @@ RankTransport = Callable[[Mapping[str, object]], Mapping[str, object]] -class _NoRedirect(HTTPRedirectHandler): - def redirect_request(self, request: Request, fp: object, code: int, - msg: str, headers: object, newurl: str) -> None: - return None - - @dataclass(frozen=True, slots=True) class ClmAdapter: """Convert CLM rankings to typed logits in the original option order. @@ -92,6 +86,15 @@ def rank_actions( """Advisory candidate distribution; CVoC and policy still choose actions.""" return self._rank(self._context(state), "NEXT_ACTION", candidate_ids, deadline=deadline) + def rank_text(self, context: str, question: str, options: tuple[str, ...], + *, deadline: float | None = None) -> tuple[float, ...]: + """Rank caller text without assigning it policy or execution authority.""" + if not context or len(context.encode("utf-8")) > _MAX_CONTEXT_BYTES: + raise ValueError("CLM context must be nonempty and bounded") + if not question or len(question) > 1024: + raise ValueError("CLM question must be nonempty and bounded") + return self._rank(context, question, options, deadline=deadline) + def _context(self, state: CompiledState) -> str: context = json.dumps(asdict(state), sort_keys=True, allow_nan=False, ensure_ascii=False) if len(context.encode("utf-8")) > _MAX_CONTEXT_BYTES: @@ -157,7 +160,7 @@ def _post(self, payload: Mapping[str, object], timeout: float) -> Mapping[str, o ) # Ignore proxy environment variables and reject redirects so a local # server cannot relay the secret or state to an off-host destination. - opener = build_opener(ProxyHandler({}), _NoRedirect(), HTTPHandler()) + opener = build_opener(ProxyHandler({}), NoRedirectHandler(), HTTPHandler()) with opener.open(request, timeout=timeout) as response: body = response.read(_MAX_RESPONSE_BYTES + 1) if len(body) > _MAX_RESPONSE_BYTES: diff --git a/c3r/system_one/inference.py b/c3r/system_one/inference.py new file mode 100644 index 0000000..ad38a07 --- /dev/null +++ b/c3r/system_one/inference.py @@ -0,0 +1,146 @@ +"""Stateless, non-authoritative CLM inference, separate from governed actions.""" + +from __future__ import annotations + +import json +import time +from collections.abc import Callable, Mapping +from typing import cast + +from .clm_adapter import ClmAdapter + + +def _text(value: object, limit: int = 1024) -> str: + if not isinstance(value, str) or not value.strip() or len(value) > limit: + raise ValueError("nonempty bounded text required") + return value + + +def _mapping(value: object) -> Mapping[str, object]: + if not isinstance(value, dict): + raise ValueError("object required") + return cast(Mapping[str, object], value) + + +def _list(value: object) -> list[object]: + if not isinstance(value, list): + raise ValueError("array required") + return cast(list[object], value) + + +class SystemOneInference: + """Typed questions and arbitrary strings: scores are NOT success probabilities.""" + + def __init__(self, adapter: ClmAdapter, *, readiness: Callable[[], bool] | None = None) -> None: + self.adapter = adapter + self._readiness = readiness + + @property + def ready(self) -> bool: + if self._readiness is None: + return False + try: + return self._readiness() + except (OSError, RuntimeError, ValueError, TypeError): + return False + + def _rank(self, context: str, question: str, options: tuple[str, ...], + deadline: float) -> tuple[float, ...]: + try: + return self.adapter.rank_text(context, question, options, deadline=deadline) + except (OSError, TypeError, ValueError, KeyError) as error: + raise RuntimeError("CLM inference unavailable") from error + + def infer(self, payload: Mapping[str, object]) -> dict[str, object]: + if set(payload) - {"model", "state", "questions", "candidates"}: + raise ValueError("unsupported System-One fields") + model = payload.get("model", "c3r-system-one") + if model not in {"c3r-system-one", "c3r-verifier"}: + raise ValueError("unsupported ranking model") + state = payload.get("state") + if isinstance(state, dict): + context = json.dumps(state, allow_nan=False, ensure_ascii=False, sort_keys=True) + else: + context = _text(state, 32768) + if len(context.encode("utf-8")) > 32768: + raise ValueError("state too large") + if ("questions" in payload) == ("candidates" in payload): + raise ValueError("provide either questions or candidates") + # One total deadline, not a fresh timeout for each question. + deadline = time.monotonic() + self.adapter.timeout_seconds + result: dict[str, object] = { + "model": model, "provider": "clm", "calibrated": False, + "effect_executed": False, "scope": "advisory_only", + } + if "candidates" in payload: + raw = _list(payload["candidates"]) + if not 1 <= len(raw) <= 64: + raise ValueError("candidate count out of bounds") + options = tuple(_text(option) for option in raw) + if len(set(options)) != len(options): + raise ValueError("candidates must be unique") + scores = self._rank(context, "NEXT_ACTION", options, deadline) + result["ranked"] = [ + {"candidate": option, "score": score} + for option, score in sorted(zip(options, scores), key=lambda pair: pair[1], + reverse=True) + ] + return result + questions = _mapping(payload["questions"]) + if not 1 <= len(questions) <= 16: + raise ValueError("question count out of bounds") + # Validate ALL questions before making any provider calls. + prepared: list[tuple[str, str, str, tuple[str, ...], tuple[str, ...]]] = [] + for name, value in questions.items(): + name = _text(name, 128) + value = _mapping(value) + if set(value) - {"type", "options", "criteria", "instructions"}: + raise ValueError("invalid question") + kind = value.get("type") + instructions = value.get("instructions", name) + prompt = _text(instructions) + if "options" in value and "criteria" in value: + raise ValueError("ambiguous options") + raw = value.get("options", value.get("criteria")) + if kind == "choice": + raw = _mapping(raw) + if not 2 <= len(raw) <= 64: + raise ValueError("choice requires option descriptions") + keys = tuple(_text(key, 128) for key in raw) + options = tuple(_text(description) for description in raw.values()) + elif kind in {"boolean", "noul"}: + keys = ("false", "true") + if raw is not None: + raw = _mapping(raw) + if set(raw) != {"false", "true"}: + raise ValueError("boolean criteria require false and true") + options = tuple(_text(raw[key]) for key in keys) + else: + options = (f"false: No. This is false: {prompt}", + f"true: Yes. This is true: {prompt}") + elif kind == "score": + raw = _list(raw) + if not 2 <= len(raw) <= 64: + raise ValueError("score requires ordered levels") + options = tuple(_text(item) for item in raw) + keys = tuple(str(index) for index in range(len(options))) + else: + raise ValueError("unsupported question type") + if len(set(options)) != len(options) or any(len(option) > 1024 for option in options): + raise ValueError("options must be unique and bounded") + prepared.append((name, cast(str, kind), prompt, keys, options)) + if sum(len(options) for _, _, _, _, options in prepared) > 64: + raise ValueError("total options exceeds budget") + answers: dict[str, object] = {} + for name, kind, prompt, keys, options in prepared: + scores = self._rank(context, prompt, options, deadline) + answer: dict[str, object] = {"scores": dict(zip(keys, scores))} + if kind == "choice": + answer["choice"] = keys[max(range(len(scores)), key=lambda index: scores[index])] + elif kind in {"boolean", "noul"}: + answer["score"] = scores[1] + else: + answer["score"] = sum(index * score for index, score in enumerate(scores)) + answers[name] = answer + result["answers"] = answers + return result diff --git a/deploy/Dockerfile.clm b/deploy/Dockerfile.clm new file mode 100644 index 0000000..0efc930 --- /dev/null +++ b/deploy/Dockerfile.clm @@ -0,0 +1,7 @@ +FROM vllm/vllm-openai@sha256:00d577a6a63281e15336029d5bcee4e9a2cf182214a4f20ba6111b1c8e79893d +ADD clm-source-bb42c6c.tar /opt/clm/ +COPY clm-source-revision /opt/clm/source-revision +RUN python3 -m pip install --no-deps --no-build-isolation /opt/clm +COPY clm_runtime.py /opt/c3r/clm_runtime.py +COPY encoder_identity.py /opt/c3r/encoder_identity.py +ENTRYPOINT ["python3", "/opt/c3r/clm_runtime.py"] diff --git a/deploy/Dockerfile.clm-hardened b/deploy/Dockerfile.clm-hardened new file mode 100644 index 0000000..6ebb15a --- /dev/null +++ b/deploy/Dockerfile.clm-hardened @@ -0,0 +1,16 @@ +# CPU projection-head service only; the Qwen encoder is a separate process. +FROM vllm/vllm-openai@sha256:00d577a6a63281e15336029d5bcee4e9a2cf182214a4f20ba6111b1c8e79893d +USER root +RUN apt-get update && apt-get upgrade -y \ + && apt-get purge -y linux-libc-dev \ + && rm -rf /var/lib/apt/lists/* \ + && python3 -m pip uninstall --break-system-packages -y vllm \ + && python3 -m pip install --break-system-packages --no-cache-dir --no-deps \ + PyJWT==2.14.0 msgpack==1.2.1 urllib3==2.8.0 setuptools==78.1.1 +ADD clm-source-bb42c6c.tar /opt/clm/ +COPY clm-source-revision /opt/clm/source-revision +RUN python3 -m pip install --break-system-packages --no-deps --no-build-isolation /opt/clm +COPY clm_runtime.py encoder_identity.py /opt/c3r/ +RUN python3 -c "from clm.engine import Engine; from clm.server import create_app" +USER 10001:10001 +ENTRYPOINT ["python3", "/opt/c3r/clm_runtime.py"] diff --git a/deploy/attest_encoder.py b/deploy/attest_encoder.py new file mode 100644 index 0000000..099cbe9 --- /dev/null +++ b/deploy/attest_encoder.py @@ -0,0 +1,40 @@ +"""Run as the deployment operator; inspect the encoder's actual mounted files. + +Output belongs in private deployment evidence. This is a trusted host readback, +not hardware-backed remote attestation. No Docker socket is given to CLM. +""" +import json +import subprocess +from pathlib import Path + +from encoder_identity import REVISION, verify_encoder + +IMAGE = "sha256:00d577a6a63281e15336029d5bcee4e9a2cf182214a4f20ba6111b1c8e79893d" + + +def main(): + info = json.loads(subprocess.check_output([ + "docker", "inspect", "c3r-qwen-encoder"], text=True))[0] + image = json.loads(subprocess.check_output([ + "docker", "image", "inspect", info["Image"]], text=True))[0] + if not any(value.endswith("@" + IMAGE) for value in image["RepoDigests"]): + raise RuntimeError("encoder container image is not the approved immutable digest") + args = info["Config"]["Cmd"] + if (not info["State"]["Running"] or "/encoder" not in args + or "pooling" not in args or "qwen3-8b" not in args): + raise RuntimeError("encoder process configuration mismatch") + mounts = [mount for mount in info["Mounts"] if mount["Destination"] == "/encoder"] + if len(mounts) != 1 or mounts[0]["RW"]: + raise RuntimeError("encoder artifact mount must be read-only") + # This namespace path proves which files the running encoder sees, rather + # than verifying a separate copy downloaded beside it. + verify_encoder(Path(f'/proc/{info["State"]["Pid"]}/root/encoder')) + print(json.dumps({ + "encoder_revision": REVISION, "encoder_container_digest": IMAGE, + "encoder_container_id": info["Id"], "encoder_content_verified": True, + "binding_basis": "host_docker_inspect_and_running_process_mount_hashes", + }, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/deploy/clm-source-revision b/deploy/clm-source-revision new file mode 100644 index 0000000..3ad91a5 --- /dev/null +++ b/deploy/clm-source-revision @@ -0,0 +1 @@ +bb42c6c5bf914fd449bed2f6ca65be80602cb1f7 diff --git a/deploy/clm_runtime.py b/deploy/clm_runtime.py new file mode 100644 index 0000000..1a719bc --- /dev/null +++ b/deploy/clm_runtime.py @@ -0,0 +1,75 @@ +"""Loopback CLM service with measured artifact identity and no prompt caches.""" +import hashlib +import json +import os +from pathlib import Path + +import requests +import torch +import uvicorn +from clm.embedder import Embedder +from clm.engine import Engine +from clm.server import create_app +from encoder_identity import REVISION, verify_encoder + +SOURCE = "bb42c6c5bf914fd449bed2f6ca65be80602cb1f7" +HEAD_SHA256 = "b2b4a8c9c2d39263eff78a351eb909a342ce9b3bf21a3f07c1d1bf15f1c4eda5" +ROOT = Path("/artifacts") + + +def digest(path): + hasher = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(4 * 1024 * 1024), b""): + hasher.update(chunk) + return hasher.hexdigest() + + +class LocalSession(requests.Session): + def __init__(self): + super().__init__() + self.trust_env = False + + def request(self, method, url, **kwargs): + if not url.startswith("http://127.0.0.1:8090/"): + raise ValueError("encoder origin must stay loopback") + kwargs["allow_redirects"] = False + return super().request(method, url, **kwargs) + + +def main(): + manifest = json.loads((ROOT / "manifest.json").read_text()) + actual_source = Path("/opt/clm/source-revision").read_text().strip() + if actual_source != SOURCE or manifest["clm_source_revision"] != SOURCE: + raise RuntimeError("CLM source mismatch") + head = ROOT / "head/CLM_v0.1-8B.pt" + actual_head = digest(head) + if actual_head != HEAD_SHA256 or manifest["head_sha256"] != actual_head: + raise RuntimeError("CLM head mismatch") + if manifest["encoder_revision"] != REVISION: + raise RuntimeError("encoder revision mismatch") + verify_encoder(ROOT / "encoder") + embedder = Embedder(max_tokens=8192, cache_size=0, batch=8, timeout=10) + embedder.session = LocalSession() + engine = Engine(embedder=embedder, checkpoint=str(head), device="cpu", action_cache=0) + app = create_app(engine, ui=False) + artifact = {key: value for key, value in manifest.items() if key != "encoder_files"} + artifact.update({ + "container_digest": os.environ.get("C3R_CLM_CONTAINER_DIGEST", "unattested"), + "container_digest_source": "deployment_host_readback", + "head_requires_vllm": False, "torch_version": torch.__version__, + "cuda_version": torch.version.cuda, "embedding_cache_size": 0, + "action_cache_enabled": False, "head_device": "cpu", + "encoder_content_verified": True, + "encoder_identity_basis": "immutable_upstream_git_blobs_and_lfs_sha256", + }) + + @app.get("/internal/clm/artifact") + def loaded_artifact(): + return artifact + + uvicorn.run(app, host="127.0.0.1", port=8700, access_log=False, log_level="warning") + + +if __name__ == "__main__": + main() diff --git a/deploy/encoder_identity.py b/deploy/encoder_identity.py new file mode 100644 index 0000000..6ffdd19 --- /dev/null +++ b/deploy/encoder_identity.py @@ -0,0 +1,45 @@ +"""Independent file identities from the immutable upstream HF revision API. + +Small files use Git blob identities; LFS objects use upstream SHA-256 identities. +These are release pins, not hashes accepted from the downloaded local manifest. +""" +import hashlib +from pathlib import Path + +REVISION = "b968826d9c46dd6066d109eabc6255188de91218" +LFS = { + "model-00001-of-00005.safetensors": "31d6a825ae35f11fb85b195b4c42c146c051e446433125a215336abdf95cbf5f", + "model-00002-of-00005.safetensors": "5991236cea6fe21f3d43cab0f0e84448734fbbe0789816202989f2ddc9d18282", + "model-00003-of-00005.safetensors": "c5185c4794be2d8a9784d5753c9922db38df478ce11f9ed0b415b7304d896836", + "model-00004-of-00005.safetensors": "b5ee7de71fbf17db3d5704e0c8f2bc7d005ca9e1d7ca2aeb19827b0cfcaa917a", + "model-00005-of-00005.safetensors": "20c2d6366ab85c90786ccdd829cd2b9e7d30ef3b2ebbb998280e7e4014b542ff", + "tokenizer.json": "aeb13307a71acd8fe81861d94ad54ab689df773318809eed3cbe794b4492dae4", +} +GIT_BLOBS = { + "config.json": "d46195ac87f837ad233d02b2f80f148bf7c005e0", + "generation_config.json": "20a8a9156fc8c3f25295ca067f61fdf120d517c5", + "merges.txt": "31349551d90c7606f325fe0f11bbb8bd5fa0d7c7", + "model.safetensors.index.json": "2b85c00f1b118961cd7a477e2bba0fe197a4ce1a", + "tokenizer_config.json": "417d038a63fa3de29cfde265caedae14d1a58d92", + "vocab.json": "4783fe10ac3adce15ac8f358ef5462739852c569", +} + + +def verify_encoder(root: Path) -> None: + expected = set(LFS) | set(GIT_BLOBS) + actual = {str(path.relative_to(root)) for path in root.rglob("*") + if path.is_file() and ".cache" not in path.parts} + if actual != expected: + raise RuntimeError("encoder file inventory does not match pinned release") + for filename in sorted(expected): + path = root / filename + if path.is_symlink(): + raise RuntimeError("encoder release must use ordinary read-only files") + hasher = hashlib.sha256() if filename in LFS else hashlib.sha1() + if filename in GIT_BLOBS: + hasher.update(f"blob {path.stat().st_size}\0".encode()) + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(4 * 1024 * 1024), b""): + hasher.update(chunk) + if hasher.hexdigest() != (LFS.get(filename) or GIT_BLOBS[filename]): + raise RuntimeError("encoder content does not match independent upstream identity") diff --git a/docs/stateless-core-api.md b/docs/stateless-core-api.md index b025ee0..aa71a8b 100644 --- a/docs/stateless-core-api.md +++ b/docs/stateless-core-api.md @@ -23,7 +23,8 @@ implicitly turns on a provider or grants action authority. ## API -The external ingress requires `X-C3R-Token` behind a TLS/IAM boundary; the +The external ingress supports standard `Authorization: Bearer ` for SDK +clients. Private IAM staging also supports a separate `X-C3R-Token`; the loopback backend uses a different bearer token. Requests are JSON objects and bounded to 64 KiB. The trusted host owns the action catalog, policy, verifier, measured CVoC estimates, and provider connectivity. Callers cannot override @@ -33,36 +34,77 @@ those fields. | --- | --- | | `GET /health` | Process liveness only. | | `GET /ready` | Authenticated. Returns 503 until decisions and System-One are enabled **and** the host's provider probe succeeds. It is not a complete launch attestation. | -| `GET /v1/models` | Authenticated model-like discovery of `c3r-core` and `c3r-system-one`, each labeled non-generative and uncalibrated. Availability follows the provider probe. | +| `GET /v1/models` | OpenAI-style `object: list`, `data` discovery of `c3r-core`, `c3r-system-one`, and advisory `c3r-verifier`. CLM models are non-generative. No model claims empirical calibration. | | `POST /v1/c3r/decide` | Runs the controller and returns a read-only recommendation or explicit fallback. | -| `POST /v1/c3r/rank` | Same safe controller path, plus available advisory candidate scores. No score is represented as probability of task success. | -| `POST /v1/system-one` | Same controller path with System-One status and fallback. This is **not** arbitrary typed-question inference. | +| `POST /v1/c3r/rank` | Direct, read-only ranking of arbitrary candidate strings. No action catalog is required and nothing is executed. | +| `POST /v1/system-one` | Typed `choice`, `boolean`/`noul`, and ordered `score` questions, or direct candidate ranking. Relative scores are not task-success probabilities or authority. | | `POST /v1/c3r/execute` | Returns 501. External effects are unsupported. | -| `POST /v1/responses` | Returns 501. This release is not OpenAI Responses API-compatible and CLM is not a text generator. | +| `POST /v1/responses` | Text-only Responses subset for `c3r-core`: string `input`, bounded `max_output_tokens`, `store: false`, `stream: false`. Generation uses local DeepSeek, not CLM. Tools, storage, streaming and other fields are rejected. | `POST /v1/decisions` remains the compatibility route. Example: ```http POST /v1/c3r/decide Content-Type: application/json -X-C3R-Token: +Authorization: Bearer {"goal":"Find record","current_subgoal":"Search approved index"} ``` The response includes `selected_action_id`, `route`, `reason`, `authority_result`, `effect_executed: false`, and a request-local `trace_hash`. -A ranking response additionally includes `candidate_ranking` entries with -`system_one_score` and `calibrated: false`. An empty ranking and abstention are -normal when CLM is unavailable or control questions lack held-out calibration. +A direct ranking response includes `ranked` entries with `candidate` and +`score`, plus `calibrated: false` and `scope: advisory_only`. Provider failure +returns 503; invalid or over-budget requests return 400. Raw CLM scores cannot safely be converted into task-success probabilities by a -fixed cap; CVoC remains driven by trusted, measured host estimates. +fixed cap. The host currently supplies **no positive quality estimates**; governed +decisions stop when conservative CVoC is non-positive. For an explicit text-only +Responses request, the host may select the admitted DELIBERATE candidate as a +named policy fallback after independent read-only verification. This is not a +positive learned CVoC claim. Generation stays inside the controller, honors its +enable flags, and cannot execute effects. + +```json +{"model":"c3r-system-one","state":"An invoice was charged twice", + "questions":{"department":{"type":"choice","options":{ + "billing":"Invoices and charges","technical":"Software bugs"}}, + "urgent":{"type":"boolean"}}} +``` + +System-One budgets: 32 KiB state, at most 16 questions and 64 total options; +candidate strings are unique and at most 1024 characters. Responses accepts at +most 4096 characters / 16 KiB input and 2048 output tokens. `c3r-verifier` is an +advisory ranker, not the independent authority verifier. + +## Production composition + +`C3R_HOST_ENTRYPOINT=c3r.production_host:build` constructs the real compiler, +host-owned registry, advisory CLM ranking, CVoC, independent verifier, ephemeral +sink, and text-generation path. `C3R_DELIBERATIVE=true` is required. The general, +browser, research and tool-routing catalogs contain recommendations only; +retrieval and browser/tool execution are not implemented capabilities. + +CLM source, Qwen revision and head revision/hash are pinned. The loopback CLM +wrapper checks loaded head and encoder files at startup, disables its embedding +and action caches, and exposes `/internal/clm/artifact`. Its container identity +is a **deployment-host readback**, not a cryptographic remote attestation. +Encoder checks use immutable upstream Git-blob/LFS identities, not hashes trusted +from the local manifest. `deploy/attest_encoder.py` independently checks the +running encoder's mounted files, read-only mount and immutable container image. +Readiness compares artifact pins and runs actual CLM ranking and bounded DeepSeek +generation. Checks coalesce for 15 seconds; only booleans and expiry timestamps +are cached. Model discovery reports ranking availability independently of +DeepSeek. Disable switches take effect immediately before typed inference, not +after the health-cache expires. ## Production gates -This repository currently has no trusted production host catalog/estimate -source, live CLM/Qwen provider deployment, approved dedicated public hostname, -or production canary. Before public traffic, qualify: +The private candidate has returned real CLM rankings and DeepSeek text through +the assembled API. This is **not a public production launch**. Public TLS routing, +complete image scanning, sustainable load/SLO measurements, outage and rollback +drills, canary evidence, and release authorization must still be qualified. +Deployment cost/quality estimates must not be described as measured until their +evidence exists. Before public traffic, qualify: 1. Immutable CLM and Qwen encoder artifacts, provider health/timeout/fallback, and revision attestations. diff --git a/scripts/download_clm_artifacts.py b/scripts/download_clm_artifacts.py new file mode 100644 index 0000000..c7f9a34 --- /dev/null +++ b/scripts/download_clm_artifacts.py @@ -0,0 +1,42 @@ +"""Fetch immutable public upstream artifacts; no customer data is used.""" +import hashlib +import json +from pathlib import Path + +from huggingface_hub import hf_hub_download, snapshot_download + +ROOT = Path("/artifacts") +ENCODER_REVISION = "b968826d9c46dd6066d109eabc6255188de91218" +HEAD_REVISION = "e939398d4556fcd9400c76fa8c5a513202f42b0a" + + +def digest(path): + hasher = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(4 * 1024 * 1024), b""): + hasher.update(chunk) + return hasher.hexdigest() + + +def main(): + snapshot_download("Qwen/Qwen3-8B", revision=ENCODER_REVISION, + local_dir=ROOT / "encoder", max_workers=4, + allow_patterns=["*.json", "*.safetensors", "*.txt", "*.model"]) + head = Path(hf_hub_download("Contrastive-LM/CLM-v0.1-8B", "CLM_v0.1-8B.pt", + revision=HEAD_REVISION, local_dir=ROOT / "head")) + files = {str(path.relative_to(ROOT / "encoder")): digest(path) + for path in sorted((ROOT / "encoder").rglob("*")) + if path.is_file() and ".cache" not in path.parts} + manifest = { + "clm_source_revision": "bb42c6c5bf914fd449bed2f6ca65be80602cb1f7", + "encoder": "Qwen/Qwen3-8B", "encoder_revision": ENCODER_REVISION, + "head_revision": HEAD_REVISION, "head_sha256": digest(head), + "encoder_files": files, + } + (ROOT / "manifest.json").write_text(json.dumps(manifest, indent=2) + "\n") + print(json.dumps({key: value for key, value in manifest.items() if key != "encoder_files"}), + flush=True) + + +if __name__ == "__main__": + main() diff --git a/scripts/probe_live_models.py b/scripts/probe_live_models.py new file mode 100644 index 0000000..88d4d50 --- /dev/null +++ b/scripts/probe_live_models.py @@ -0,0 +1,38 @@ +"""Non-sensitive loopback inference drill. Does not store task traces.""" +import json +import time + +import requests + + +def main(): + session = requests.Session() + session.trust_env = False + probes = ( + ("clm", "http://127.0.0.1:8700/v1/rank", { + "model": "clm-latest", "context": "The customer was charged twice for an invoice.", + "question": "Which department handles this request?", + "answers": ["Billing: invoices and charges", "Technical: software bugs"], + }), + ("deepseek", "http://127.0.0.1:8000/v1/chat/completions", { + "model": "/model", "max_tokens": 512, "temperature": 0, + "messages": [{"role": "user", "content": "Reply with one sentence: what is an invoice?"}], + }), + ) + for name, url, payload in probes: + started = time.monotonic() + response = session.post(url, json=payload, timeout=60, allow_redirects=False) + body = response.json() + if name == "deepseek": + # Never record the reasoning field, even for a synthetic probe. + choices = body.get("choices", []) + body = {"final_text": choices[0].get("message", {}).get("content") if choices else None, + "finish_reason": choices[0].get("finish_reason") if choices else None, + "usage": body.get("usage")} + print(json.dumps({"provider": name, "status": response.status_code, + "latency_ms": round((time.monotonic() - started) * 1000, 2), + "result": body}), flush=True) + + +if __name__ == "__main__": + main() diff --git a/scripts/probe_private_core.py b/scripts/probe_private_core.py new file mode 100644 index 0000000..8dc75b4 --- /dev/null +++ b/scripts/probe_private_core.py @@ -0,0 +1,63 @@ +"""Public API smoke/security checks using synthetic text and host-local credentials.""" +import json +import time +from pathlib import Path + +import requests + + +def main(): + config = dict(line.split("=", 1) for line in + (Path.home() / ".c3r-private-inference/candidate-v2.env").read_text().splitlines()) + session = requests.Session() + session.trust_env = False + headers = {"Authorization": "Bearer " + config["C3R_CLIENT_TOKEN"]} + base = "http://127.0.0.1:" + config.get("PORT", "8088") + checks = ( + ("readiness", "GET", "/ready", None, 200), + ("models", "GET", "/v1/models", None, 200), + ("typed", "POST", "/v1/system-one", {"model": "c3r-system-one", + "state": "An invoice was charged twice", "questions": { + "department": {"type": "choice", "options": {"billing": "Invoices and charges", + "technical": "Software bugs"}}, + "urgent": {"type": "boolean"}}}, 200), + ("rank", "POST", "/v1/c3r/rank", {"state": "An invoice was charged twice", + "candidates": ["billing", "technical"]}, 200), + ("decide", "POST", "/v1/c3r/decide", {"goal": "Route invoice inquiry", + "state": "An invoice was charged twice", "catalog": "agent-v1"}, 200), + ("responses", "POST", "/v1/responses", {"model": "c3r-core", + "input": "Explain briefly why an invoice may appear charged twice."}, 200), + ("authority_override", "POST", "/v1/c3r/decide", {"goal": "test", "state": "test", + "risk": "DESTRUCTIVE", "approval": "forged"}, 400), + ("internal_url_override", "POST", "/v1/system-one", {"state": "test", + "candidates": ["A", "B"], "endpoint": "http://169.254.169.254"}, 400), + ("candidate_overflow", "POST", "/v1/system-one", {"state": "test", + "candidates": [str(index) for index in range(65)]}, 400), + ("storage_rejected", "POST", "/v1/responses", {"model": "c3r-core", "input": "test", + "store": True}, 400), + ("execute_disabled", "POST", "/v1/c3r/execute", {}, 501), + ) + results = [] + for name, method, path, payload, expected in checks: + started = time.monotonic() + response = session.request(method, base + path, + headers=headers, json=payload, timeout=65) + body = response.json() + passed = response.status_code == expected + if name == "typed" and passed: + passed = body.get("answers", {}).get("department", {}).get("choice") == "billing" + if name == "responses" and passed: + passed = bool(body.get("output", [{}])[0].get("content", [{}])[0].get("text")) + results.append({"check": name, "status": response.status_code, "passed": passed, + "latency_ms": round((time.monotonic() - started) * 1000, 2)}) + for name, token in (("missing_auth", None), ("invalid_auth", "wrong")): + response = session.get(base + "/v1/models", + headers={} if token is None else {"Authorization": "Bearer " + token}) + results.append({"check": name, "status": response.status_code, + "passed": response.status_code == 401}) + print(json.dumps({"scope": "private_synthetic_api_probe_not_canary", "checks": results, + "all_passed": all(result["passed"] for result in results)}, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/scripts/replace_private_candidate.py b/scripts/replace_private_candidate.py new file mode 100644 index 0000000..ce82df6 --- /dev/null +++ b/scripts/replace_private_candidate.py @@ -0,0 +1,80 @@ +"""Replace only task-created private C3R containers, restoring them on failure. + +Run on the deployment host after build, artifact verification and scans. This +never stops the shared DeepSeek service or the VM, nor exposes an interface. +""" +import json +import os +import subprocess +import time +import urllib.error +import urllib.request +from pathlib import Path + + +def docker(*args): + return subprocess.check_output(["sudo", "docker", *args], text=True).strip() + + +def main(): + folder = Path.home() / ".c3r-private-inference" + previous = folder / "candidate-v2.env" + current = folder / "candidate-v3.env" + if current.exists(): + raise RuntimeError("v3 config exists; do not implicitly rotate or replace") + config = dict(line.split("=", 1) for line in previous.read_text().splitlines()) + if (config.get("C3R_INGRESS_HOST") != "127.0.0.1" + or config.get("C3R_TRACE_COLLECTION") != "false" + or config.get("C3R_ONLINE_LEARNING") != "false"): + raise RuntimeError("replacement is restricted to non-collecting private inference") + clm_image = docker("image", "inspect", "c3r-clm:pinned-verified-bb42c6c", "--format", "{{.Id}}") + core_image = docker("image", "inspect", "c3r-core:reviewed-20260930", "--format", "{{.Id}}") + config["C3R_CLM_CONTAINER_DIGEST"] = clm_image + descriptor = os.open(current, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + with os.fdopen(descriptor, "w") as handle: + handle.write("".join(f"{key}={value}\n" for key, value in config.items())) + moved = [] + started = [] + try: + for name in ("c3r-core-private", "c3r-clm"): + docker("stop", "--time", "10", name) + docker("rename", name, name + "-before-v3") + moved.append(name) + docker("run", "-d", "--name", "c3r-clm", "--network", "host", "--read-only", + "--tmpfs", "/tmp:rw,size=256m", "-v", "/mnt/c3r-models/c3r-clm:/artifacts:ro", + "-e", "C3R_CLM_CONTAINER_DIGEST=" + clm_image, clm_image) + started.append("c3r-clm") + docker("run", "-d", "--name", "c3r-core-private", "--network", "host", + "--env-file", str(current), "--read-only", "--cap-drop", "ALL", + "--security-opt", "no-new-privileges", "--pids-limit", "128", + "--memory", "512m", "--cpus", "2", core_image) + started.append("c3r-core-private") + opener = urllib.request.build_opener(urllib.request.ProxyHandler({})) + deadline = time.monotonic() + 120 + while time.monotonic() < deadline: + request = urllib.request.Request("http://127.0.0.1:8088/ready", headers={ + "Authorization": "Bearer " + config["C3R_CLIENT_TOKEN"]}) + try: + with opener.open(request, timeout=15) as response: + if response.status == 200: + print(json.dumps({"status": "private_candidate_ready", + "core_image": core_image, "clm_image": clm_image, + "trace_collection": False, "public_access": False})) + return + except (OSError, urllib.error.URLError): + pass + time.sleep(3) + raise RuntimeError("private candidate readiness deadline exceeded") + except BaseException: + for name in reversed(started): + docker("stop", "--time", "10", name) + docker("rename", name, name + "-failed-v3") + for name in reversed(moved): + docker("rename", name + "-before-v3", name) + docker("start", name) + print(json.dumps({"status": "previous_private_candidate_restored"})) + raise + + +if __name__ == "__main__": + main() diff --git a/scripts/start_private_core.py b/scripts/start_private_core.py new file mode 100644 index 0000000..93005b1 --- /dev/null +++ b/scripts/start_private_core.py @@ -0,0 +1,41 @@ +"""Start a loopback-only candidate. Tokens stay in a mode-0600 host file.""" +import json +import os +import secrets +import subprocess +from pathlib import Path + + +def main(): + destination = Path.home() / ".c3r-private-inference" + destination.mkdir(mode=0o700, exist_ok=True) + config = destination / "candidate-v2.env" + if config.exists(): + raise RuntimeError("candidate config already exists; do not rotate implicitly") + env = { + "C3R_HOST_ENTRYPOINT": "c3r.production_host:build", + "C3R_MODE": "production_inference", "C3R_ENABLED": "true", + "C3R_SYSTEM_ONE": "true", "C3R_SYSTEM_ONE_PROVIDER": "clm", + "C3R_DELIBERATIVE": "true", "C3R_TRACE_COLLECTION": "false", + "C3R_ONLINE_LEARNING": "false", "C3R_INGRESS_HOST": "127.0.0.1", + "PORT": "8088", "C3R_BACKEND_PORT": "8089", + "C3R_CLM_CONTAINER_DIGEST": "sha256:922fe094c0804fc2b4e327f2dbbe97b749e47de7d8806b2f467cae7ecd0d2dda", + "C3R_CLIENT_TOKEN": secrets.token_urlsafe(48), + "C3R_BACKEND_TOKEN": secrets.token_urlsafe(48), + } + descriptor = os.open(config, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + with os.fdopen(descriptor, "w") as handle: + handle.write("".join(f"{key}={value}\n" for key, value in env.items())) + result = subprocess.run([ + "sudo", "docker", "run", "-d", "--name", "c3r-core-private", + "--network", "host", "--env-file", str(config), "--read-only", + "--cap-drop", "ALL", "--security-opt", "no-new-privileges", + "--pids-limit", "128", "--memory", "512m", "--cpus", "2", + "c3r-core:working-20260930", + ], check=True, capture_output=True, text=True) + print(json.dumps({"container_id": result.stdout.strip(), "bind": "127.0.0.1:8088", + "trace_collection": False})) + + +if __name__ == "__main__": + main() diff --git a/tests/test_ingress_proxy.py b/tests/test_ingress_proxy.py index 5a81894..08f6655 100644 --- a/tests/test_ingress_proxy.py +++ b/tests/test_ingress_proxy.py @@ -84,8 +84,16 @@ def test_private_decision_forwards_only_internal_authorization(self): self.assertNotIn("X-C3R-Token", headers) self.assertEqual(json.loads(forwarded), {"goal": "inspect"}) + def test_sdk_bearer_authentication_without_cloud_run_header(self): + request = Request(self.base + "/v1/models", headers={ + "Authorization": "Bearer " + CLIENT_TOKEN}) + with urlopen(request, timeout=2) as response: + self.assertEqual(response.status, 200) + self.assertEqual(self.upstream.seen[-1][1]["Authorization"], + "Bearer " + UPSTREAM_TOKEN) + def test_missing_or_wrong_client_token_never_reaches_upstream(self): - for token in (None, "wrong"): + for token in (None, "wrong", "invalid-café"): status, body = self.request("/v1/decisions", method="POST", token=token, payload={"goal": "inspect"}) self.assertEqual((status, body["error"]), (401, "unauthorized")) @@ -156,4 +164,4 @@ def test_upstream_route_must_be_loopback_and_secrets_distinct(self): if __name__ == "__main__": unittest.main() - + diff --git a/tests/test_serve.py b/tests/test_serve.py index 3c9a0c4..20d49ed 100644 --- a/tests/test_serve.py +++ b/tests/test_serve.py @@ -1,6 +1,11 @@ import socket import threading import unittest +import os +import subprocess +import sys +import time +import json from urllib.request import urlopen from c3r.serve import build_servers, load_host_builder @@ -37,6 +42,36 @@ def config(): class ServeTests(unittest.TestCase): + def test_module_entrypoint_starts_production_host_without_claiming_provider_readiness(self): + values = config() + values.update({ + "C3R_HOST_ENTRYPOINT": "c3r.production_host:build", + "C3R_MODE": "production_inference", "C3R_ENABLED": "true", + "C3R_SYSTEM_ONE": "true", "C3R_DELIBERATIVE": "true", + "C3R_SYSTEM_ONE_PROVIDER": "clm", "C3R_TRACE_COLLECTION": "false", + "C3R_ONLINE_LEARNING": "false", "C3R_INGRESS_HOST": "127.0.0.1", + "C3R_CLM_CONTAINER_DIGEST": "sha256:" + "a" * 64, + }) + process = subprocess.Popen([sys.executable, "-m", "c3r.serve"], + env={**os.environ, **values}, stdout=subprocess.DEVNULL, + stderr=subprocess.PIPE) + try: + deadline = time.monotonic() + 5 + while time.monotonic() < deadline: + try: + with urlopen("http://127.0.0.1:" + values["PORT"] + "/health", timeout=1) as response: + self.assertEqual(json.load(response)["status"], "ok") + break + except OSError: + if process.poll() is not None: + self.fail("production entrypoint exited before health was reachable") + time.sleep(0.05) + else: + self.fail("production entrypoint did not become reachable") + finally: + process.terminate() + process.communicate(timeout=5) + def test_missing_host_or_secret_fails_before_binding(self): values = config() del values["C3R_HOST_ENTRYPOINT"] @@ -132,4 +167,4 @@ def test_production_mode_requires_system_one_path(self): if __name__ == "__main__": unittest.main() - + diff --git a/tests/test_stateless_api.py b/tests/test_stateless_api.py index 69db165..7feebc4 100644 --- a/tests/test_stateless_api.py +++ b/tests/test_stateless_api.py @@ -6,10 +6,23 @@ from urllib.error import HTTPError from urllib.request import Request, urlopen +from c3r.adapters.providers import ProviderAdapter, ProviderConfig, ProviderKind, TransportResponse +from c3r.host_factory import ReadOnlyRequestFactory from c3r.http_service import C3RHTTPServer +from c3r.readiness import CachedReadiness +from c3r.responses import ResponsesService +from c3r.state_schema import ( + ActionDefinition, + ActionFamily, + AuthorityPolicy, + RiskClass, + ValueEstimate, +) +from c3r.system_one.advisory import AdvisoryFastPath +from c3r.system_one.clm_adapter import ClmAdapter +from c3r.system_one.inference import SystemOneInference from tests.test_runtime import controller, request - TOKEN = "stateless-test-token-with-at-least-thirty-two-characters" @@ -48,6 +61,47 @@ def call(self, path, *, method="POST", token=TOKEN, payload=None): except HTTPError as error: return error.code, json.load(error) + def configure_ranker(self, transport, *, enabled=True, system_one=True, ready=lambda: True): + adapter = ClmAdapter("a" * 64, transport=transport) + self.server.runtime, _ = controller(enabled=enabled, system_one=system_one, + fast_path=AdvisoryFastPath(adapter)) + self.server.system_one = SystemOneInference(adapter, readiness=ready) + + def test_typed_ranker_respects_disable_switches_without_provider_calls(self): + calls = [] + for enabled, system_one in ((False, True), (True, False)): + self.configure_ranker(lambda payload: calls.append(payload), enabled=enabled, + system_one=system_one) + for path in ("/v1/system-one", "/v1/c3r/rank"): + status, _ = self.call(path, payload={"state": "test", "candidates": ["A", "B"]}) + self.assertEqual(status, 503) + self.assertEqual(calls, []) + + def test_model_availability_is_independent_of_system_two(self): + self.configure_ranker(lambda _: {}, ready=lambda: True) + status, body = self.call("/v1/models", method="GET") + self.assertEqual(status, 200) + self.assertFalse(body["data"][0]["available"]) + self.assertTrue(body["data"][1]["available"]) + self.assertTrue(body["data"][2]["available"]) + + def test_metadata_coalesces_health_checks_and_expires_cached_status(self): + now, calls = [0.0], [] + def probe(): + calls.append(1) + return len(calls) == 1 + readiness = CachedReadiness(probe, clock=lambda: now[0]) + self.configure_ranker(lambda _: {}, ready=readiness) + for _ in range(4): + status, body = self.call("/v1/models", method="GET") + self.assertEqual(status, 200) + self.assertTrue(body["data"][1]["available"]) + self.assertEqual(len(calls), 1) + now[0] = 16 + _, body = self.call("/v1/models", method="GET") + self.assertFalse(body["data"][1]["available"]) + self.assertEqual(len(calls), 2) + def test_decide_and_rank_are_recommendation_only(self): for path in ("/v1/c3r/decide", "/v1/c3r/rank", "/v1/system-one"): status, body = self.call(path, payload={"goal": "Find record"}) @@ -65,6 +119,99 @@ def test_execute_and_untyped_responses_are_unavailable(self): self.assertEqual(status, 501) self.assertEqual(body["error"], "not_implemented") + def test_typed_system_one_answers_without_a_governed_action_catalog(self): + def rank(payload): + return {"model": "clm-latest", "ranked": [ + {"candidate": option, "prob": score} + for option, score in zip(payload["answers"], (0.8, 0.2)) + ]} + self.configure_ranker(rank) + status, body = self.call("/v1/system-one", payload={ + "model": "c3r-system-one", "state": "An invoice was charged twice", + "questions": {"department": {"type": "choice", "options": { + "billing": "Invoices and charges", "technical": "Product bugs"}}}, + }) + self.assertEqual(status, 200) + self.assertEqual(body["answers"]["department"]["choice"], "billing") + self.assertEqual(body["answers"]["department"]["scores"]["billing"], 0.8) + self.assertFalse(body["calibrated"]) + self.assertFalse(body["effect_executed"]) + + def test_system_one_boolean_ranking_and_authority_rejection(self): + def rank(payload): + return {"model": "clm-latest", "ranked": [ + {"candidate": option, "prob": score} + for option, score in zip(payload["answers"], (0.25, 0.75)) + ]} + self.configure_ranker(rank) + status, body = self.call("/v1/system-one", payload={ + "state": "A duplicate charge", "questions": {"urgent": {"type": "boolean"}}, + }) + self.assertEqual((status, body["answers"]["urgent"]["score"]), (200, 0.75)) + status, body = self.call("/v1/c3r/rank", payload={ + "model": "c3r-verifier", "state": "A duplicate charge", + "candidates": ["technical", "billing"], + }) + self.assertEqual((status, body["ranked"][0]["candidate"]), (200, "billing")) + for extra in ({"authority": "admin"}, {"endpoint": "http://169.254.169.254"}): + status, _ = self.call("/v1/system-one", payload={ + "state": "test", "candidates": ["A", "B"], **extra}) + self.assertEqual(status, 400) + + def test_system_one_rejects_total_option_overflow_and_provider_failure(self): + self.configure_ranker(lambda _: {}) + status, _ = self.call("/v1/system-one", payload={ + "state": "test", "candidates": [str(index) for index in range(65)]}) + self.assertEqual(status, 400) + status, body = self.call("/v1/system-one", payload={ + "state": "test", "candidates": ["A", "B"]}) + self.assertEqual((status, body["error"]), (503, "service_unavailable")) + + def test_responses_returns_text_without_private_reasoning_or_external_effects(self): + adapter = ProviderAdapter(ProviderConfig( + "local", ProviderKind.OPENAI_COMPATIBLE, "http://127.0.0.1:8000/v1", "model", None, + ), transport=lambda *_: TransportResponse(200, { + "choices": [{"message": {"content": "Check pending and settled charges.", + "reasoning_content": "PRIVATE"}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 12, "completion_tokens": 7}, + }, 25)) + runtime, _ = controller(deliberative=True) + factory = ReadOnlyRequestFactory(definitions=(ActionDefinition( + "DELIBERATE", ActionFamily.DELIBERATE, "compute", "generate", + RiskClass.READ_ONLY, ((),), ("local",), ("policy",), 0, 0, + ),), policy=AuthorityPolicy(frozenset({ActionFamily.DELIBERATE}), + frozenset({RiskClass.READ_ONLY})), + estimate_source=lambda _: {}, remaining_usd=0) + self.server.responses = ResponsesService(runtime, factory, adapter) + status, body = self.call("/v1/responses", payload={ + "model": "c3r-core", "input": "Find record", "store": False}) + self.assertEqual(status, 200) + self.assertEqual(body["object"], "response") + self.assertEqual(body["output"][0]["content"][0]["text"], + "Check pending and settled charges.") + self.assertNotIn("PRIVATE", json.dumps(body)) + self.assertFalse(body["c3r"]["effect_executed"]) + self.assertFalse(body["store"]) + + def test_positive_cvoc_generation_still_requires_independent_verification(self): + calls = [] + adapter = ProviderAdapter(ProviderConfig( + "local", ProviderKind.OPENAI_COMPATIBLE, "http://127.0.0.1:8000/v1", "model", None, + ), transport=lambda *_: calls.append(1)) + runtime, _ = controller(deliberative=True, accepted=False) + factory = ReadOnlyRequestFactory(definitions=(ActionDefinition( + "DELIBERATE", ActionFamily.DELIBERATE, "compute", "generate", + RiskClass.READ_ONLY, ((),), ("local",), ("policy",), 0, 0, + ),), policy=AuthorityPolicy(frozenset({ActionFamily.DELIBERATE}), + frozenset({RiskClass.READ_ONLY})), + estimate_source=lambda _: {"DELIBERATE:0:local:policy": ValueEstimate(1, 0, 0, 0)}, + remaining_usd=0) + self.server.responses = ResponsesService(runtime, factory, adapter) + status, body = self.call("/v1/responses", payload={ + "model": "c3r-core", "input": "Find record"}) + self.assertEqual((status, body["error"]), (503, "service_unavailable")) + self.assertEqual(calls, []) + def test_new_paths_require_authentication(self): for path in ("/v1/c3r/decide", "/v1/c3r/execute", "/v1/responses"): status, body = self.call(path, token="wrong", payload={"goal": "Find record"}) From 061ca585febc3903084f9cc8125166e23470ec89 Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 30 Sep 2026 18:40:18 -0500 Subject: [PATCH 49/55] Harden CPU CLM serving and add private load qualification runner --- deploy/Dockerfile.clm-hardened | 2 + deploy/inspect_head_sboms.py | 21 +++++++++ scripts/measure_private_core.py | 67 ++++++++++++++++++++++++++++ scripts/probe_private_core.py | 4 +- scripts/replace_private_candidate.py | 2 +- 5 files changed, 94 insertions(+), 2 deletions(-) create mode 100644 deploy/inspect_head_sboms.py create mode 100644 scripts/measure_private_core.py diff --git a/deploy/Dockerfile.clm-hardened b/deploy/Dockerfile.clm-hardened index 6ebb15a..6c67f58 100644 --- a/deploy/Dockerfile.clm-hardened +++ b/deploy/Dockerfile.clm-hardened @@ -10,7 +10,9 @@ RUN apt-get update && apt-get upgrade -y \ ADD clm-source-bb42c6c.tar /opt/clm/ COPY clm-source-revision /opt/clm/source-revision RUN python3 -m pip install --break-system-packages --no-deps --no-build-isolation /opt/clm +RUN python3 -m pip uninstall --break-system-packages -y setuptools wheel cryptography COPY clm_runtime.py encoder_identity.py /opt/c3r/ RUN python3 -c "from clm.engine import Engine; from clm.server import create_app" +RUN python3 -m pip uninstall --break-system-packages -y pip USER 10001:10001 ENTRYPOINT ["python3", "/opt/c3r/clm_runtime.py"] diff --git a/deploy/inspect_head_sboms.py b/deploy/inspect_head_sboms.py new file mode 100644 index 0000000..aca013c --- /dev/null +++ b/deploy/inspect_head_sboms.py @@ -0,0 +1,21 @@ +"""Explain package/SBOM discrepancies without suppressing scanner findings.""" +import importlib.metadata +import json +from pathlib import Path + +TARGETS = {"msgpack", "setuptools", "urllib3"} +results = [] +for path in Path("/usr/local/lib").rglob("*.cdx.json"): + document = json.loads(path.read_text()) + for component in document.get("components", []): + name = component.get("name", "") + if name not in TARGETS: + continue + try: + actual = importlib.metadata.version(name) + except importlib.metadata.PackageNotFoundError: + actual = "not_installed" + results.append({"sbom": str(path), "package": name, + "sbom_version": component.get("version"), + "installed_distribution_version": actual}) +print(json.dumps(results, indent=2)) diff --git a/scripts/measure_private_core.py b/scripts/measure_private_core.py new file mode 100644 index 0000000..f3a712a --- /dev/null +++ b/scripts/measure_private_core.py @@ -0,0 +1,67 @@ +"""Pilot load probe: aggregate synthetic-request timings, never training traces. + +Admission responses are reported, not misrepresented as successful inference. +This short run is neither a sustained SLO test nor a public canary. +""" +import concurrent.futures +import json +import math +import os +import time +from collections import Counter +from pathlib import Path + +import requests + + +def main(): + config = dict(line.split("=", 1) for line in ( + Path.home() / ".c3r-private-inference" / + os.environ.get("C3R_PROBE_CONFIG", "candidate-v3.env")).read_text().splitlines()) + token = config["C3R_CLIENT_TOKEN"] + base = "http://127.0.0.1:" + config.get("PORT", "8088") + + def call(_): + session = requests.Session() + session.trust_env = False + started = time.monotonic() + try: + response = session.post(base + "/v1/c3r/rank", headers={ + "Authorization": "Bearer " + token}, json={ + "state": "An invoice was charged twice.", + "candidates": ["Billing", "Technical"]}, timeout=15) + status = str(response.status_code) + except requests.RequestException: + status = "transport_failure" + finally: + session.close() + return status, (time.monotonic() - started) * 1000 + + profiles = [] + for concurrency in (1, 10, 25, 50, 100): + started = time.monotonic() + with concurrent.futures.ThreadPoolExecutor(max_workers=concurrency) as executor: + results = list(executor.map(call, range(concurrency))) + elapsed = time.monotonic() - started + successful = sorted(latency for status, latency in results if status == "200") + counts = dict(Counter(status for status, _ in results)) + profile = {"concurrency": concurrency, "requests": len(results), + "statuses": counts, "elapsed_seconds": round(elapsed, 3), + "successful_requests_per_second": round(len(successful) / elapsed, 3), + "successful_p50_ms": None if not successful else round( + successful[(len(successful) - 1) // 2], 2), + "successful_p95_ms": None if not successful else round( + successful[math.ceil(len(successful) * .95) - 1], 2)} + profiles.append(profile) + # Predeclared abort: transport failure or unexpected server failures. + if any(status not in {"200", "429", "503"} for status in counts): + print(json.dumps({"scope": "private_synthetic_load_pilot", "aborted": True, + "profiles": profiles}, indent=2)) + raise SystemExit(1) + print(json.dumps({"scope": "private_synthetic_load_pilot_not_slo_or_canary", + "currency_cost": "unqualified_no_billing_allocation", + "profiles": profiles}, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/scripts/probe_private_core.py b/scripts/probe_private_core.py index 8dc75b4..2cdf4eb 100644 --- a/scripts/probe_private_core.py +++ b/scripts/probe_private_core.py @@ -1,5 +1,6 @@ """Public API smoke/security checks using synthetic text and host-local credentials.""" import json +import os import time from pathlib import Path @@ -8,7 +9,8 @@ def main(): config = dict(line.split("=", 1) for line in - (Path.home() / ".c3r-private-inference/candidate-v2.env").read_text().splitlines()) + (Path.home() / ".c3r-private-inference" / + os.environ.get("C3R_PROBE_CONFIG", "candidate-v2.env")).read_text().splitlines()) session = requests.Session() session.trust_env = False headers = {"Authorization": "Bearer " + config["C3R_CLIENT_TOKEN"]} diff --git a/scripts/replace_private_candidate.py b/scripts/replace_private_candidate.py index ce82df6..8199061 100644 --- a/scripts/replace_private_candidate.py +++ b/scripts/replace_private_candidate.py @@ -27,7 +27,7 @@ def main(): or config.get("C3R_TRACE_COLLECTION") != "false" or config.get("C3R_ONLINE_LEARNING") != "false"): raise RuntimeError("replacement is restricted to non-collecting private inference") - clm_image = docker("image", "inspect", "c3r-clm:pinned-verified-bb42c6c", "--format", "{{.Id}}") + clm_image = docker("image", "inspect", "c3r-clm:hardened-clean-bb42c6c", "--format", "{{.Id}}") core_image = docker("image", "inspect", "c3r-core:reviewed-20260930", "--format", "{{.Id}}") config["C3R_CLM_CONTAINER_DIGEST"] = clm_image descriptor = os.open(current, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) From 0a8156849b2c66d00456a7397ca6411353193e3d Mon Sep 17 00:00:00 2001 From: wilkont Date: Wed, 30 Sep 2026 18:45:21 -0500 Subject: [PATCH 50/55] Record private qualification scope and add recoverable outage drills --- docs/stateless-core-api.md | 8 ++++ scripts/drill_private_core.py | 81 +++++++++++++++++++++++++++++++++++ 2 files changed, 89 insertions(+) create mode 100644 scripts/drill_private_core.py diff --git a/docs/stateless-core-api.md b/docs/stateless-core-api.md index aa71a8b..1b9e08d 100644 --- a/docs/stateless-core-api.md +++ b/docs/stateless-core-api.md @@ -106,6 +106,14 @@ drills, canary evidence, and release authorization must still be qualified. Deployment cost/quality estimates must not be described as measured until their evidence exists. Before public traffic, qualify: +The CPU head candidate built with `deploy/Dockerfile.clm-hardened` and the +reviewed API image passed fresh HIGH/CRITICAL scans. Use that hardened build +recipe for the head; `deploy/Dockerfile.clm` is the unqualified baseline recipe, +not a production recommendation. This does **not** clear the separate Qwen and +shared DeepSeek model-serving image. Its scan findings still require review or +remediation. The short private load pilot and CLM outage/operator recovery drills +are recorded in private evidence; they are not sustained SLOs or public canaries. + 1. Immutable CLM and Qwen encoder artifacts, provider health/timeout/fallback, and revision attestations. 2. Host-supplied catalog, measured estimates, read-only verifier, data-boundary diff --git a/scripts/drill_private_core.py b/scripts/drill_private_core.py new file mode 100644 index 0000000..5674096 --- /dev/null +++ b/scripts/drill_private_core.py @@ -0,0 +1,81 @@ +"""Scoped private CLM outage and C3R kill/recovery drills. + +Never stops the encoder, shared DeepSeek service or VM. Synthetic API bodies +are discarded; only statuses are reported. This is not a public canary. +""" +import json +import subprocess +import time +from pathlib import Path + +import requests + + +def main(): + config = dict(line.split("=", 1) for line in ( + Path.home() / ".c3r-private-inference/candidate-v3.env").read_text().splitlines()) + if config.get("C3R_INGRESS_HOST") != "127.0.0.1": + raise RuntimeError("drill restricted to private candidate") + session = requests.Session() + session.trust_env = False + session.headers["Authorization"] = "Bearer " + config["C3R_CLIENT_TOKEN"] + base = "http://127.0.0.1:8088" + evidence = [] + stopped = set() + + def docker(*args): + subprocess.run(["sudo", "docker", *args], check=True, capture_output=True) + + def status(path, payload=None): + try: + response = session.request("GET" if payload is None else "POST", base + path, + json=payload, timeout=15) + return response.status_code + except requests.RequestException: + return "unreachable" + + def ready(expected, deadline=120): + until = time.monotonic() + deadline + while time.monotonic() < until: + value = status("/ready") + if value == expected: + return value + time.sleep(2) + raise RuntimeError("readiness transition deadline exceeded") + + try: + if status("/ready") != 200: + raise RuntimeError("private baseline is not ready") + docker("stop", "--time", "10", "c3r-clm") + stopped.add("c3r-clm") + outage = ready(503, deadline=45) + ranking = status("/v1/system-one", {"state": "Duplicate invoice", + "candidates": ["Billing", "Technical"]}) + fallback = status("/v1/responses", {"model": "c3r-core", "input": "Reply briefly: ready."}) + evidence.append({"drill": "clm_outage", "readiness": outage, + "ranking": ranking, "text_only_policy_fallback": fallback, + "passed": outage == 503 and ranking == 503 and fallback == 200}) + docker("start", "c3r-clm") + stopped.remove("c3r-clm") + evidence.append({"drill": "clm_recovery", "readiness": ready(200), "passed": True}) + docker("stop", "--time", "10", "c3r-core-private") + stopped.add("c3r-core-private") + killed = status("/ready") + evidence.append({"drill": "operator_kill", "readiness": killed, + "passed": killed == "unreachable"}) + docker("start", "c3r-core-private") + stopped.remove("c3r-core-private") + evidence.append({"drill": "operator_recovery", "readiness": ready(200), "passed": True}) + finally: + for name in stopped: + docker("start", name) + session.close() + print(json.dumps({"scope": "private_clm_outage_and_operator_recovery_not_canary", + "shared_deepseek_stopped": False, "checks": evidence, + "all_passed": all(item["passed"] for item in evidence)}, indent=2)) + if not all(item["passed"] for item in evidence): + raise SystemExit(1) + + +if __name__ == "__main__": + main() From ab9344f483e50c058a0d5f64fb8c015b5e32d085 Mon Sep 17 00:00:00 2001 From: wilkont Date: Thu, 1 Oct 2026 12:33:56 -0500 Subject: [PATCH 51/55] Pin self-hosted H100 qualification config and correct startup flag Reviewed local source: f1c4cbb134b5573954fe714ef630b5612ae24437. Qualification remains pending. --- README.md | 18 +- c3r/adapters/providers.py | 33 +- c3r/deliberative/defaults.py | 12 +- c3r/ingress_proxy.py | 2 +- c3r/responses.py | 9 +- c3r/serve.py | 2 +- deploy/deepseek-v41/Dockerfile | 23 + deploy/deepseek-v41/README.md | 60 ++ .../check_upstream_dependencies.py | 25 + deploy/deepseek-v41/checkpoint-manifest.json | 534 ++++++++++++++++++ deploy/deepseek-v41/h100-production.env | 18 + deploy/deepseek-v41/launch.py | 134 +++++ deploy/deepseek-v41/start.sh | 3 + docs/runtime-integrations.md | 13 +- tests/test_default_language_model.py | 14 +- tests/test_ingress_proxy.py | 2 +- tests/test_serve.py | 2 +- tests/test_stateless_api.py | 49 ++ 18 files changed, 922 insertions(+), 31 deletions(-) create mode 100644 deploy/deepseek-v41/Dockerfile create mode 100644 deploy/deepseek-v41/README.md create mode 100644 deploy/deepseek-v41/check_upstream_dependencies.py create mode 100644 deploy/deepseek-v41/checkpoint-manifest.json create mode 100644 deploy/deepseek-v41/h100-production.env create mode 100644 deploy/deepseek-v41/launch.py create mode 100644 deploy/deepseek-v41/start.sh diff --git a/README.md b/README.md index 9fcaf45..9d0ef02 100644 --- a/README.md +++ b/README.md @@ -50,8 +50,9 @@ See the [scope-specific release policy](docs/release-policy.md). C3R's default **deliberative** language model is [`deepseek-ai/DeepSeek-V4.1-Flash`](https://huggingface.co/deepseek-ai/DeepSeek-V4.1-Flash) -(MIT). The hosted default uses `deepseek/deepseek-v4.1-flash` through OpenRouter; the -official DeepSeek API alias is `deepseek-flash`. CLM is the default separate +(MIT), self-hosted on the existing eight-H100 node alongside Qwen3-8B/CLM. +The default provider uses the private loopback `/model` alias; no OpenRouter or +hosted inference provider is part of the production target. CLM is the default separate System-One decision engine; Laya remains optional. DeepSeek recommendations remain subject to the same verifier and trusted commit boundary as every other candidate. @@ -61,6 +62,14 @@ up in a private GCS bucket. Its vLLM endpoint is bound to localhost; this is **n public C3R service or end-to-end production qualification. The checkpoint's active parameter count does not imply it fits on one GPU. +The recovered **older** serving image passed all 13 private C3R checks at 85% +GPU reservation with Qwen still running. The clean, pinned newer image is a +separate qualification target: its first startup rejected an obsolete flag. +The corrected launch configuration is tracked in +[`deploy/deepseek-v41`](deploy/deepseek-v41/README.md). Repeated cold starts, +sustained mixed load, outage/rollback, final-head builds, TLS and canary remain +required; recovery alone is not production qualification. + ## Why C3R Most agent stacks decide *what to say*. C3R decides *what computation should happen next*—and @@ -182,7 +191,8 @@ CLM ranks bounded, policy-surviving candidate labels and fixed typed questions. Its raw ranking is advisory, not an action selection. Missing calibration, malformed responses, outages, or timeouts lead to abstention or deterministic fallback. CVoC, independent verification, and the commit boundary retain -authority. No live CLM service, GPU coexistence benchmark, or held-out C3R +authority. Private live CLM ranking and co-resident DeepSeek recovery have been +observed; neither sustained GPU coexistence qualification nor held-out C3R calibration is claimed by this code change. ## Laya integration @@ -255,4 +265,4 @@ Do not report vulnerabilities in a public issue; follow [`SECURITY.md`](SECURITY effects must be completely mediated by an independently configured commit gateway. Apache License 2.0. See [`LICENSE`](LICENSE). If you use C3R, cite [`CITATION.cff`](CITATION.cff). - + diff --git a/c3r/adapters/providers.py b/c3r/adapters/providers.py index a394bc0..7b86861 100644 --- a/c3r/adapters/providers.py +++ b/c3r/adapters/providers.py @@ -4,7 +4,7 @@ import json from collections.abc import Callable, Mapping -from dataclasses import dataclass +from dataclasses import dataclass, field from enum import StrEnum from math import isfinite from typing import Protocol, cast @@ -34,7 +34,7 @@ class ProviderConfig: kind: ProviderKind base_url: str model: str - api_key: str | None + api_key: str | None = field(repr=False) timeout_seconds: float = 60.0 def __post_init__(self) -> None: @@ -49,6 +49,14 @@ def __post_init__(self) -> None: if not isfinite(self.timeout_seconds) or not 0 < self.timeout_seconds <= 120: raise ValueError("timeout_seconds must be positive") + @property + def is_local(self) -> bool: + return urlparse(self.base_url).hostname in {"localhost", "127.0.0.1", "::1"} + + @property + def uses_openrouter(self) -> bool: + return self.base_url.rstrip("/") == "https://openrouter.ai/api/v1" + @dataclass(frozen=True, slots=True) class DeliberationRequest: @@ -257,16 +265,22 @@ def generate(self, text: str, max_output_tokens: int) -> tuple[str, str, dict[st headers = {"X-C3R-Timeout": str(self.config.timeout_seconds)} if self.config.api_key: headers["Authorization"] = "Bearer " + self.config.api_key - try: - response = self._transport(self.config.base_url.rstrip("/") + "/chat/completions", - headers, { + payload: dict[str, object] = { "model": self.config.model, "max_tokens": max_output_tokens, "temperature": 0, "messages": [ {"role": "system", "content": "Provide a helpful final answer only. Do not expose " "private reasoning or claim to execute tools, commit actions, or grant authority."}, {"role": "user", "content": text}, ], - }) + } + if self.config.uses_openrouter: + if not self.config.api_key: + raise RuntimeError("hosted provider credential unavailable") + payload["provider"] = {"zdr": True, "data_collection": "deny", + "require_parameters": True, "allow_fallbacks": False} + try: + response = self._transport(self.config.base_url.rstrip("/") + "/chat/completions", + headers, payload) size = len(json.dumps(response.body, allow_nan=False).encode()) except (OSError, ValueError, TypeError) as error: raise RuntimeError("generative provider transport unavailable") from error @@ -284,11 +298,16 @@ def generate(self, text: str, max_output_tokens: int) -> tuple[str, str, dict[st if not output.strip() or reason not in {"stop", "length"}: raise ValueError("invalid generation result") usage = cast(Mapping[str, object], response.body.get("usage", {})) - return output, cast(str, reason), { + observed = { "input_tokens": _number(usage.get("prompt_tokens")), "output_tokens": _number(usage.get("completion_tokens")), "latency_ms": response.latency_ms, } + cost = usage.get("cost") + if (isinstance(cost, (int, float)) and not isinstance(cost, bool) + and isfinite(cost) and cost >= 0): + observed["provider_cost_usd"] = float(cost) + return output, cast(str, reason), observed except (KeyError, IndexError, TypeError, ValueError, AttributeError) as error: raise RuntimeError("invalid generative provider response") from error diff --git a/c3r/deliberative/defaults.py b/c3r/deliberative/defaults.py index 3a49da7..2f499a8 100644 --- a/c3r/deliberative/defaults.py +++ b/c3r/deliberative/defaults.py @@ -1,6 +1,6 @@ """Auditable defaults for C3R's deliberative language-model path. -The Laya System-One decision model remains separate from this language-model +The CLM System-One decision model remains separate from this language-model default. Defaults select a provider profile; they never bypass candidate, verification, budget, or commit controls. """ @@ -26,9 +26,15 @@ class DefaultModelProfile: def default_provider_config( - *, api_key: str, gateway: Literal["openrouter", "deepseek"] = "openrouter" + *, api_key: str | None = None, + gateway: Literal["self_hosted", "openrouter", "deepseek"] = "self_hosted" ) -> ProviderConfig: - """Return the explicit hosted default; credentials are caller-owned.""" + """Default to local DeepSeek. Explicit remote profiles are legacy compatibility only.""" + if gateway == "self_hosted": + return ProviderConfig("deepseek-local", ProviderKind.OPENAI_COMPATIBLE, + "http://127.0.0.1:8000/v1", "/model", None) + if gateway not in {"openrouter", "deepseek"}: + raise ValueError("unknown gateway") if not api_key: raise ValueError("api_key is required") if gateway == "deepseek": diff --git a/c3r/ingress_proxy.py b/c3r/ingress_proxy.py index 8a17d6d..78b628c 100644 --- a/c3r/ingress_proxy.py +++ b/c3r/ingress_proxy.py @@ -158,4 +158,4 @@ def do_POST(self) -> None: self._send_error(413, "request_size_out_of_bounds") return self._forward("POST", self.rfile.read(length)) - + diff --git a/c3r/responses.py b/c3r/responses.py index efae2a1..e5b356e 100644 --- a/c3r/responses.py +++ b/c3r/responses.py @@ -23,6 +23,8 @@ def __init__(self, provider: ProviderAdapter, maximum: int) -> None: self.provider, self.maximum = provider, maximum def deliberate(self, state: CompiledState) -> TextGenerationResult: + if not self.provider.config.is_local and state.data_boundary not in {"approved_remote", "public"}: + raise ValueError("state is not approved for hosted generation") return TextGenerationResult(*self.provider.generate(state.goal, self.maximum)) @@ -62,7 +64,7 @@ def respond(self, payload: Mapping[str, object]) -> dict[str, object]: output, finish, usage = (decision.deliberation.text, decision.deliberation.finish, decision.deliberation.usage) identifier = "resp_" + uuid4().hex - return { + response: dict[str, object] = { "id": identifier, "object": "response", "model": "c3r-core", "status": "completed" if finish == "stop" else "incomplete", "store": False, "output": [{"id": "msg_" + uuid4().hex, "type": "message", "role": "assistant", @@ -76,3 +78,8 @@ def respond(self, payload: Mapping[str, object]) -> dict[str, object]: "selection_basis": "explicit_text_only_request_policy_fallback", "calibrated": False, "provider": self.provider.config.provider_id}, } + if "provider_cost_usd" in usage: + metadata = response["c3r"] + assert isinstance(metadata, dict) + metadata["observed_provider_cost_usd"] = usage["provider_cost_usd"] + return response diff --git a/c3r/serve.py b/c3r/serve.py index 1d1fe16..210b547 100644 --- a/c3r/serve.py +++ b/c3r/serve.py @@ -147,4 +147,4 @@ def request_stop(_signum: int, _frame: object) -> None: if __name__ == "__main__": main() - + diff --git a/deploy/deepseek-v41/Dockerfile b/deploy/deepseek-v41/Dockerfile new file mode 100644 index 0000000..049aa48 --- /dev/null +++ b/deploy/deepseek-v41/Dockerfile @@ -0,0 +1,23 @@ +# Qualification candidate only; image scan and GPU qualification are separate gates. +FROM vllm/vllm-openai@sha256:98adb2311118263a453ec0acd5d4b72bda3db26cecc0fe8ffc94b9ab03c951f9 +USER root +LABEL org.opencontainers.image.source="https://github.com/vllm-project/vllm" \ + org.opencontainers.image.revision="ac68c3087215e0a4f3cdfa218508c6aada57235d" \ + com.colomboai.c3r.qualification="pending" +RUN apt-get update && apt-get upgrade -y \ + && apt-get purge -y linux-libc-dev python3-msgpack python3-setuptools python3-urllib3 \ + && rm -rf /var/lib/apt/lists/* \ + && python3 -m pip uninstall --break-system-packages -y opentelemetry-exporter-otlp-common +COPY check_upstream_dependencies.py /opt/c3r/check_upstream_dependencies.py +RUN python3 /opt/c3r/check_upstream_dependencies.py \ + && python3 -m pip uninstall --break-system-packages -y pip +ENV HOME=/runtime \ + HF_HUB_OFFLINE=1 \ + HF_HUB_DISABLE_TELEMETRY=1 \ + VLLM_NO_USAGE_STATS=1 \ + DO_NOT_TRACK=1 \ + VLLM_USE_V2_MODEL_RUNNER=1 \ + PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True \ + VLLM_ENGINE_READY_TIMEOUT_S=3600 +USER 1003:1004 +ENTRYPOINT ["python3", "-m", "vllm.entrypoints.openai.api_server"] diff --git a/deploy/deepseek-v41/README.md b/deploy/deepseek-v41/README.md new file mode 100644 index 0000000..81e8a44 --- /dev/null +++ b/deploy/deepseek-v41/README.md @@ -0,0 +1,60 @@ +# Same-node DeepSeek qualification target + +The target is self-hosted Qwen3-8B/CLM System One, DeepSeek V4.1 Flash +System Two, and C3R Core on the existing eight-H100 node. No new GPU, +OpenRouter, or hosted inference provider is part of this deployment. + +`h100-production.env` is a **qualification target**, not a production approval. +It pins the locally built immutable candidate image, upstream source/base image, +checkpoint revision, TP8, 85% reservation, Engram CPU offload, 32K context, +two sequences, batch limit, and both parsers. Registry publication of that local +image remains a release gate. Execution verifies every checkpoint file against +the tracked upstream identity manifest and rejects extra runtime files. Only +Hugging Face download metadata under `.cache/huggingface` is exempted from +extra-file rejection. Generation defaults are explicitly `vllm`, not a hidden +local generation configuration. Neither a revision string nor a directory name +alone proves checkpoint integrity. + +## Startup flag correction + +| Item | Recorded value | +| --- | --- | +| Old flag | `--disable-log-requests` | +| Replacement | `--no-enable-log-requests` | +| Reason | The pinned newer vLLM parser rejects the old flag; request logging must stay off. | +| vLLM source | `ac68c3087215e0a4f3cdfa218508c6aada57235d` | +| vLLM version | `0.30.1rc1.dev396+gac68c3087` | + +The earlier recovery used a **different older image**, at 85% reservation, +without this candidate's explicit Engram/tool-parser configuration. Its 13 +passing private API checks do not qualify this new image or its launch settings. + +```sh +# Dry-run by default; checkpoint/cache paths stay private and outside Git. +sh deploy/deepseek-v41/start.sh \ + --model-path /private/verified-checkpoint --cache-path /private/serving-cache +# --execute starts only this named candidate, on loopback:18000 by default. +# The supervised maintenance controller must first stop the incumbent and +# independently guarantee recovery; never overlap both TP8 model engines. +``` + +The complete DeepSeek launch command is generated by `launch.py`; no shell +history, secret, or untracked flag is required. Validate `--server-args` through +the pinned runtime's actual parser before any model load. Help-only is not +argument validation. Cache must be owned by image UID 1003/GID 1004 and must +not contain request payloads. Startup does not enable research trace collection. + +## Release gates (all require actual evidence) + +Three clean startups with all 13 API checks; independently supervised service +restarts; dependency-order readiness; 15/30/60-minute sustained CLM, Responses +and mixed profiles; memory stability/no swap/OOM; bounded admission; measured +load-aware costs; reasoning/coding/tool/JSON/context acceptance; DeepSeek outage +and recovery; rollback in both directions; allocated infrastructure cost; exact +final-head builds/scans; private model ports plus dedicated TLS; external SDK +acceptance; staged canary; release-owner authorization; merge and immutable tag. + +Keep the original container/config/image and the verified 85% recovery config. +The original 92% reservation cannot restart beside resident Qwen; do not stop +Qwen merely to conceal that rollback limitation. No VM stop/reboot/delete is +permitted: the active checkpoint is on ephemeral Local SSD. diff --git a/deploy/deepseek-v41/check_upstream_dependencies.py b/deploy/deepseek-v41/check_upstream_dependencies.py new file mode 100644 index 0000000..eae9eb2 --- /dev/null +++ b/deploy/deepseek-v41/check_upstream_dependencies.py @@ -0,0 +1,25 @@ +"""Reject unknown dependency conflicts; disclose the pinned upstream NCCL override. + +The upstream Dockerfile explicitly overrides Torch's NCCL requirement because +DeepEPv2 requires NCCL >=2.30.4. This is not GPU compatibility evidence: +https://github.com/vllm-project/vllm/blob/ac68c3087215e0a4f3cdfa218508c6aada57235d/docker/Dockerfile +""" +import importlib.metadata +import json +import subprocess +import sys + +versions = {name: importlib.metadata.version(name) + for name in ("vllm", "torch", "nvidia-nccl-cu13")} +expected = {"vllm": "0.30.1rc1.dev396+gac68c3087", + "torch": "2.13.0+cu130", "nvidia-nccl-cu13": "2.30.7"} +if versions != expected: + raise RuntimeError("Pinned upstream dependency identities changed") +result = subprocess.run([sys.executable, "-m", "pip", "check"], + capture_output=True, text=True, check=False) +known = ('torch 2.13.0+cu130 has requirement nvidia-nccl-cu13==2.29.7; ' + 'platform_system == "Linux", but you have nvidia-nccl-cu13 2.30.7.') +if result.returncode != 1 or result.stdout.strip().splitlines() != [known] or result.stderr.strip(): + raise RuntimeError("Dependency preflight differs from the reviewed upstream override") +print(json.dumps({"known_upstream_nccl_override": known, + "other_dependency_errors": [], "tp8_compatibility": "not_yet_tested"})) diff --git a/deploy/deepseek-v41/checkpoint-manifest.json b/deploy/deepseek-v41/checkpoint-manifest.json new file mode 100644 index 0000000..8ba448d --- /dev/null +++ b/deploy/deepseek-v41/checkpoint-manifest.json @@ -0,0 +1,534 @@ +{ + "model_id": "deepseek-ai/DeepSeek-V4.1-Flash", + "revision": "dba1be0a40aa45a94ad051997016db3960a90277", + "files": [ + { + "path": ".gitattributes", + "size": 1701, + "git_blob_id": "9b59a7525c7b7870bd2cbdc41a4e91fd63fcb6cb", + "lfs_sha256": null + }, + { + "path": "DeepSeek_V41_Tech_Report.pdf", + "size": 1809802, + "git_blob_id": "9a327deea3393d204f789f261ceb4eb472ea910d", + "lfs_sha256": "ba68e2e40408125ae6d2f63a9a241b61c73910691c74ec1a2a7023c851eac08d" + }, + { + "path": "LICENSE", + "size": 1084, + "git_blob_id": "d62e3bef9f054f21b7fc616365850fbf879a99ff", + "lfs_sha256": null + }, + { + "path": "README.md", + "size": 13110, + "git_blob_id": "211b83857abd5e8420df3a4f10db1096294227e6", + "lfs_sha256": null + }, + { + "path": "assets/dsv41_agentic_performance.png", + "size": 190736, + "git_blob_id": "0f638f2420410c0b3b8a82d4438ca9396bc17390", + "lfs_sha256": "44deae01cb9ce756c7d622dcf2b55f4a332c25e5e0b9297997519f77e0a7cf52" + }, + { + "path": "assets/dsv41_kv_cache.png", + "size": 270933, + "git_blob_id": "39bba8d4e27e09fbc754139aa477e51bdb7184ac", + "lfs_sha256": "b61bf4651d4b163e02fb21d7298bf7b36b810c1da2300cb2d1e498c373793e4b" + }, + { + "path": "config.json", + "size": 3311, + "git_blob_id": "09917a9139b22d5bf8be52132787f435147b1020", + "lfs_sha256": null + }, + { + "path": "encoding/README.md", + "size": 12120, + "git_blob_id": "c4da223cc2d97f8d80e3d0d2ece8b1a137381507", + "lfs_sha256": null + }, + { + "path": "encoding/encoding.py", + "size": 37316, + "git_blob_id": "92a67eab2a67d924a6a8279bc0f90e3f504ac40e", + "lfs_sha256": null + }, + { + "path": "encoding/test_encoding.py", + "size": 19371, + "git_blob_id": "f2c4985a59b694da51491292f5c8264d39c7e926", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_input_1.json", + "size": 2761, + "git_blob_id": "c7435727e24c5f7b5cf28b470cf03814101eb37a", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_input_2.json", + "size": 527, + "git_blob_id": "132bb05a51e79204324f160cfde2880b5d0077ba", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_input_3.json", + "size": 2554, + "git_blob_id": "6d2612113c71fb9a897255ec694b7b8a41d0432b", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_input_4.json", + "size": 712, + "git_blob_id": "86cd1f0a68075a412654ea10f9558c69162b271a", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_input_5.json", + "size": 1143, + "git_blob_id": "2a2fa6a89e8118e896ec47abcc5bda30789e6532", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_output_1.txt", + "size": 2476, + "git_blob_id": "3dc9bfe936d251bf69f6db738d1302bfa8c4c705", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_output_2.txt", + "size": 294, + "git_blob_id": "78eb6be5de6f3cda7c5065b3c3bd01bac28ef1dc", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_output_3.txt", + "size": 2481, + "git_blob_id": "90b2b6b74beaa471dfbbba1d7e7a5eb88c9e7da0", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_output_4.txt", + "size": 574, + "git_blob_id": "efad296eb4ff95754f322476db4d26e72682a586", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_output_5.txt", + "size": 408, + "git_blob_id": "a130bcc2c5682c94b6fd5194def54209f1794939", + "lfs_sha256": null + }, + { + "path": "evaluation/README.md", + "size": 3977, + "git_blob_id": "fa1c989f2d0f5dd0ecdce369cb026b2e4040f88b", + "lfs_sha256": null + }, + { + "path": "evaluation/dsh-minimal.patch", + "size": 28725, + "git_blob_id": "2488b9578825a41f41964837d7d9c37db58b8146", + "lfs_sha256": null + }, + { + "path": "inference/README.md", + "size": 2029, + "git_blob_id": "e0b1d6ddfb4ab06a0dc9b4542826e6d7aa07a77c", + "lfs_sha256": null + }, + { + "path": "inference/config.json", + "size": 1982, + "git_blob_id": "7a915cc69e21abbc6d7fb939cef5d09c72aa0d24", + "lfs_sha256": null + }, + { + "path": "inference/convert.py", + "size": 9458, + "git_blob_id": "61bd2943648920ce599c4c86dc9bf6316e152fe8", + "lfs_sha256": null + }, + { + "path": "inference/engram.py", + "size": 8138, + "git_blob_id": "a36bc6c6b8cbf2d3dbac850863c23b55a7ecb5c7", + "lfs_sha256": null + }, + { + "path": "inference/examples/example.txt", + "size": 332, + "git_blob_id": "9200d36c90f601c0648ee23279abbb29b2228f13", + "lfs_sha256": null + }, + { + "path": "inference/examples/example_harmony.json", + "size": 2156, + "git_blob_id": "fae1c2608f8e5b980fcc9dc311cda8e345ecde1c", + "lfs_sha256": null + }, + { + "path": "inference/examples/images/carrots.jpeg", + "size": 212495, + "git_blob_id": "3b0226496452fe654ec305a3a42ef9a95cf47cfe", + "lfs_sha256": "5df896a4a07e127281c60fc957f8b3d73f4735b3258a0bf762b4383557f8fa9a" + }, + { + "path": "inference/examples/images/corn.jpeg", + "size": 56122, + "git_blob_id": "5777d198b83de15eb21243f8aac49074922053fb", + "lfs_sha256": null + }, + { + "path": "inference/generate.py", + "size": 8722, + "git_blob_id": "84d4873c4c59ea743e104920a77984f022bfb9f5", + "lfs_sha256": null + }, + { + "path": "inference/image_processor.py", + "size": 7699, + "git_blob_id": "a50311ece615b72be2c9d4b649a05619c27681e7", + "lfs_sha256": null + }, + { + "path": "inference/kernel.py", + "size": 23790, + "git_blob_id": "6fa7bd5aafbe1bc62887080963cc9fdd838d89f5", + "lfs_sha256": null + }, + { + "path": "inference/model.py", + "size": 61549, + "git_blob_id": "3e7dc222a6352ccc0c6469d58908934c2b4cd99b", + "lfs_sha256": null + }, + { + "path": "inference/requirements.txt", + "size": 97, + "git_blob_id": "b9201e446c0d2509156f46014931d359e439d2f2", + "lfs_sha256": null + }, + { + "path": "inference/run.sh", + "size": 1759, + "git_blob_id": "e2d52080a88d3053d963bac4bba75116ce73aed4", + "lfs_sha256": null + }, + { + "path": "inference/vision.py", + "size": 4457, + "git_blob_id": "77af0bdef9f4649edebac2876cd1df50a147ce98", + "lfs_sha256": null + }, + { + "path": "model-00001-of-00048.safetensors", + "size": 970533624, + "git_blob_id": "ed64b2e00da7669b19a7f0b8244bf87d3ab9980a", + "lfs_sha256": "886aebdafa08cc27bbae2165ed35bdfe0de9370bf88c1411283c155c6ae4ff89" + }, + { + "path": "model-00002-of-00048.safetensors", + "size": 1323858272, + "git_blob_id": "d04d1d34d11caccd930bfa5f7d780c86bfbe42ed", + "lfs_sha256": "4320066fc6958e5bc01d8c3feba79b7454b59f0f4b7299ab7145ed44bbf4ecec" + }, + { + "path": "model-00003-of-00048.safetensors", + "size": 7389759032, + "git_blob_id": "f19f5f2344084fd293de9a42ac0e29f862f32f24", + "lfs_sha256": "e1281f85d0ce4a3dfb63d41926fc4a47fa71f36ba20992e3597e702ead49d4c9" + }, + { + "path": "model-00004-of-00048.safetensors", + "size": 7389759032, + "git_blob_id": "827668797628e19f471172977f276b8255d6dd3f", + "lfs_sha256": "79456c9db0cda3b8115fe1c726fe3db1a34b434584a3991917088a0ab56a39de" + }, + { + "path": "model-00005-of-00048.safetensors", + "size": 7405953784, + "git_blob_id": "4cf131aac21417ca85ac3221803ad8dac77ddca2", + "lfs_sha256": "4a42dc78698bee6b1a821aa01c9650749ef1f716143751f1cdb815c6400280a9" + }, + { + "path": "model-00006-of-00048.safetensors", + "size": 7389759032, + "git_blob_id": "33afcb930c148f3edc878d589c402ad1ba4b6d6e", + "lfs_sha256": "020a6df51a2853452561d91268a65481f7a7d7954ed47f8e6c9ce69a4a134f77" + }, + { + "path": "model-00007-of-00048.safetensors", + "size": 7389759032, + "git_blob_id": "bef0f732f032b365410838f407d99141eeae9633", + "lfs_sha256": "40f8b52f763f6380d41257e1af04eee3aad300af6a418c38e49ff99e3604163a" + }, + { + "path": "model-00008-of-00048.safetensors", + "size": 7389759032, + "git_blob_id": "c85918896e64adb90a8c2bd005c806deb835ff7d", + "lfs_sha256": "d62cca4e698f030d4b96ec624c08bed7ad604cec13077da6d6b06669281c4650" + }, + { + "path": "model-00009-of-00048.safetensors", + "size": 7389759032, + "git_blob_id": "2ef42f88393516ab2b78f6219a51d3c13be1f8a1", + "lfs_sha256": "1ca62e4c294df31aee69a782974cb14269264fdc08465ab4835760258f05d6ef" + }, + { + "path": "model-00010-of-00048.safetensors", + "size": 7389759032, + "git_blob_id": "2cc30c2ca90ece11ae97ec1710b3cd461c2f4268", + "lfs_sha256": "dd33c9750a40cbfcdfb19cd8335d955f533e3a18b790c0ee43d1e5d77911595c" + }, + { + "path": "model-00011-of-00048.safetensors", + "size": 7405953784, + "git_blob_id": "806098f2fdb2017649d9a27b8c622a418fbaee04", + "lfs_sha256": "a9b309f90e0d1e2252a224ed6b057b9c82c27d3a49cff64bfbfef12efd067f7c" + }, + { + "path": "model-00012-of-00048.safetensors", + "size": 7389759032, + "git_blob_id": "57dc0d8beed2d7bf370c7422bcc1b625e2a711b6", + "lfs_sha256": "b359227eceb3f839c80de19dddf946648ca89425d9b719ff44702cf5e8cfbe0e" + }, + { + "path": "model-00013-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "33e6442ab5f10963f202c8ecbf9187409726943c", + "lfs_sha256": "41d87a4c81fec1550f9cb975db05598a18ee0161c2664e8e1a0c7b60742755a6" + }, + { + "path": "model-00014-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "358773636735c291c356a471d8a4c7fcf88b9ff0", + "lfs_sha256": "e7ca4a12688a5819829ead3a03282aedb380e438bc41a00e969f70952c6d0eb9" + }, + { + "path": "model-00015-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "44e7ccd06864c45d14d336a79bba9a98d0d03819", + "lfs_sha256": "fa9d49314bbb25118b3d52c573ed66a4d0acfb01dcb26d8367f43c66a52f46c4" + }, + { + "path": "model-00016-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "a1d29a48b074f07f84f540edb693fee5febc5bb4", + "lfs_sha256": "07d08bce9d73d416eaca0f42843440a8c45c5f939eff1a3690fb460d8dd7f623" + }, + { + "path": "model-00017-of-00048.safetensors", + "size": 7405956128, + "git_blob_id": "9b54c23553545da4e3cbb3238633431276d03f12", + "lfs_sha256": "3d35e330a6c28cd59a06a1caab8b0048f7e12cc037d728736f74431c4c0a1a4c" + }, + { + "path": "model-00018-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "1e3f4df33106ffbcbb985b7cf192b128108837f2", + "lfs_sha256": "48bd0c28b7f441f131f562cb664366d7726447a3698da3dd56ebbfeb1b916c75" + }, + { + "path": "model-00019-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "0194c1136548453cd9a74932c143797335439e54", + "lfs_sha256": "b7a25cb64a959ca5e5a99e239bb1ba2092b9bcf022bec4ba71f7441334107bd1" + }, + { + "path": "model-00020-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "82039c0d9b690646ece8dfb569f2471aa91d6287", + "lfs_sha256": "8aea8c4026ba4b93e82b4ed1b6b4620d02b6aaac88d211aaf577acea443aff84" + }, + { + "path": "model-00021-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "b0162a203fd3bf621bc18e5b688fa436059e0431", + "lfs_sha256": "4cb6558dedfdc75a6be472e27bc3e830d8443a3371854e61dc54454a1b5b18f7" + }, + { + "path": "model-00022-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "a9a16936023ebc4201d0b3966e872dce18451465", + "lfs_sha256": "81031b68c967539a7c51d1b12d37d18de926ca3045ec554ccdd49a83b4dc0084" + }, + { + "path": "model-00023-of-00048.safetensors", + "size": 7400713088, + "git_blob_id": "ccbb0d6ab728eb657e6cdd55ec96b87c21b60223", + "lfs_sha256": "096723fc8afa9886b976fb63aabef159406dc483e6f2a41e3fe386435b0436cc" + }, + { + "path": "model-00024-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "c81d6d5ad5b5b41f13d3240610724c31976fbf4f", + "lfs_sha256": "c9438ff607bd902fdeb38267d36239ccf60d38595988364562b862791841da75" + }, + { + "path": "model-00025-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "0bdd7e25ffc0aa2f8ed78445c24b4ba15d9c39e1", + "lfs_sha256": "4227ef9fe34d10de584d67250845fab905cbd8462fd84d9da24d772f223db6b4" + }, + { + "path": "model-00026-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "da6b3db1a7463825c9dfbffb07bd47aa160d92fb", + "lfs_sha256": "0af8c8f1b96b502eda652a0549cf55cd8a4d472a34d27ef9bf1c1d7c4567ca3f" + }, + { + "path": "model-00027-of-00048.safetensors", + "size": 7395337384, + "git_blob_id": "63e0eb6f1a1eca5698d68e7c70d0911fc2f5230b", + "lfs_sha256": "3066bd03043726d2400302cb9af759ec4e3957a935d56e241b83b52870322e28" + }, + { + "path": "model-00028-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "9aff3d0596b1e43a1bb8d608f4f83d7a05db4551", + "lfs_sha256": "9bc915075568b75e2aec9c645cc0fece6287f115c32031a52ab4f9ba9d12fde6" + }, + { + "path": "model-00029-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "a1fcbaf3892f7c67ec7a385acc5864e14211a0dc", + "lfs_sha256": "d153dd9cde7c4aa7ea9896cf7aa25b49447af573f90ad14cfaaa532fbbd19434" + }, + { + "path": "model-00030-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "1bf8c33843eb0a24a8ed94e1679eae2d0f010394", + "lfs_sha256": "3c9ccd96e908e00f02a70930b0821a6c53e5301376a261284a092e1aea7bb5da" + }, + { + "path": "model-00031-of-00048.safetensors", + "size": 7395337384, + "git_blob_id": "f233648fc05de748d637590366a8f2f40c6368e6", + "lfs_sha256": "0de7b6d7142df18ad45a215e9c0ef004f78cdbfefd59cc16345b2bc004047343" + }, + { + "path": "model-00032-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "01735f7ba251a25d3563d1cbb532dc07d934a67f", + "lfs_sha256": "6a5aaa73c6f97294bd079914ba44209ecf3d0491c8b95ef9995655511e35aee6" + }, + { + "path": "model-00033-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "54188954aee3724535d69814cdafcae5d6bcd174", + "lfs_sha256": "386e3e91f7f02f7e2c25f398b3a63379051c2077839b3081f7556e18c29a69c9" + }, + { + "path": "model-00034-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "7ebf03e74696c2a1c12a20804e6ebbba7e87755c", + "lfs_sha256": "9deab3c4f27c956f91cd989d9a1f6cf4bbb91858ce3f7164caa5ae61125c943f" + }, + { + "path": "model-00035-of-00048.safetensors", + "size": 7395337384, + "git_blob_id": "cc3e9dc7d0770fd9a4f367ab38161be1f9c00825", + "lfs_sha256": "226573bc07f35091b0ec27b0056ab3059f9bc1c0afd0bc104c93c86dd029cad2" + }, + { + "path": "model-00036-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "a93ac994149530aae4dd495f77261e403514d905", + "lfs_sha256": "90d6a85c1eb0cae68c6c7bffa11b60654ea00353d27a0ae17701b72e239894cd" + }, + { + "path": "model-00037-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "33f34ba08bb59e4b6854f936740f3b36aa71c29e", + "lfs_sha256": "207aef18f995fdfe7506175aabd700aa62dcf6f488c7cc375e9de06f56891b83" + }, + { + "path": "model-00038-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "13085e017de65e2750a8d0aba8e67554cdb42d5e", + "lfs_sha256": "cbbaa0b0807311807a696efec09f9b5a5d63bdf68b82378e1fc7402a2533c627" + }, + { + "path": "model-00039-of-00048.safetensors", + "size": 7395337384, + "git_blob_id": "f2b5f00589c6b9b890dd35e8232e48127ee96eef", + "lfs_sha256": "f4cf191547b50efbc35c4ed1f36cfb26043d94f987c49d54b15ef2d632a14a77" + }, + { + "path": "model-00040-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "2312d1cda8644809c58ad1ac469a6e3b0c818513", + "lfs_sha256": "e991bfc416055c45ddb89f1447eae1e67dba2816e7580044ffff3db7e2af27a3" + }, + { + "path": "model-00041-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "b45656843585d2f9faf3b9d119f72c5b53a31d82", + "lfs_sha256": "48a1c08afadf4e73223c587f9c6ad6f32aef57c01342ae75e96b07a001711d42" + }, + { + "path": "model-00042-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "4136a43c609949ed620615dbcc18ca5836428c2c", + "lfs_sha256": "e1a4d5d30ae51bafda75078d49b05a585c924b0af9b3481c075b04341507c2ef" + }, + { + "path": "model-00043-of-00048.safetensors", + "size": 1323837624, + "git_blob_id": "8fd7f2f6d9426613e927acd2c160d08590acea33", + "lfs_sha256": "d762b688f138e00a24eac96f27b842715bb4b534ecc9ca39076bed2b8c33201e" + }, + { + "path": "model-00044-of-00048.safetensors", + "size": 2652728736, + "git_blob_id": "eedbbd9f94af95a4c1d099f769f68b9ce1c268ab", + "lfs_sha256": "9a6b39fb88a2510487a8efaef77aa7864e8061f6b62c95a0f010e9dd538f3b05" + }, + { + "path": "model-00045-of-00048.safetensors", + "size": 2573998176, + "git_blob_id": "a622ff448f20314dc3512a7794eed8016a2509f6", + "lfs_sha256": "0cc9d5f6ca3a2158ccc63ce2c70c76aeda8177d54913340481af566680329eb5" + }, + { + "path": "model-00046-of-00048.safetensors", + "size": 2706402896, + "git_blob_id": "9c7282e33ca0717e0f0ee8c44666ac8877abc138", + "lfs_sha256": "e625902027b9d23d416f8818c665fab4704e0b96dc1bc778321601b700475a9d" + }, + { + "path": "model-00047-of-00048.safetensors", + "size": 101535150936, + "git_blob_id": "3eb96ce647ea90e0874a2da414f30fda9bac0034", + "lfs_sha256": "824db4881320407ac340736d14dcee5ecd748c27d0f5836b8127ecc2e3781b0f" + }, + { + "path": "model-00048-of-00048.safetensors", + "size": 101537926640, + "git_blob_id": "29a0c0b4e03cb4e663e823f28608c79ef7cb0f81", + "lfs_sha256": "976330f4954338e1ad8b508c32aa912032c7ad908959fd53c8307650fe4520ed" + }, + { + "path": "model.safetensors.index.json", + "size": 7470294, + "git_blob_id": "54c85064dd92c8550471302e9ae59bedbbf96ca4", + "lfs_sha256": null + }, + { + "path": "tokenizer.json", + "size": 6367257, + "git_blob_id": "6a15814dd25c934028034531da744c689f87ff21", + "lfs_sha256": null + }, + { + "path": "tokenizer_config.json", + "size": 801, + "git_blob_id": "f3dad388a2bbfd6a8605bd02754acd86d9ca5112", + "lfs_sha256": null + } + ] +} diff --git a/deploy/deepseek-v41/h100-production.env b/deploy/deepseek-v41/h100-production.env new file mode 100644 index 0000000..e78e010 --- /dev/null +++ b/deploy/deepseek-v41/h100-production.env @@ -0,0 +1,18 @@ +# Qualification target, NOT release authorization. No credentials or host identifiers. +CHECKPOINT=deepseek-ai/DeepSeek-V4.1-Flash +CHECKPOINT_REVISION=dba1be0a40aa45a94ad051997016db3960a90277 +VLLM_SOURCE_REVISION=ac68c3087215e0a4f3cdfa218508c6aada57235d +VLLM_BASE_DIGEST=sha256:98adb2311118263a453ec0acd5d4b72bda3db26cecc0fe8ffc94b9ab03c951f9 +VLLM_IMAGE_ID=sha256:10b3c8fe9c38f6e87dfef21c8d0e457f76ab89b32892bb37a056375b25ddbf85 +TP=8 +GPU_MEMORY_UTILIZATION=0.85 +ENGRAM_CPU_OFFLOAD=true +MAX_MODEL_LEN=32768 +MAX_NUM_SEQS=2 +MAX_NUM_BATCHED_TOKENS=4096 +TOKENIZER_MODE=deepseek_v41 +REASONING_PARSER=deepseek_v41 +TOOL_CALL_PARSER=deepseek_v41 +LANGUAGE_MODEL_ONLY=true +HOST_RAM_LIMIT_GIB=1024 +MIN_GPU_FREE_MIB=3072 diff --git a/deploy/deepseek-v41/launch.py b/deploy/deepseek-v41/launch.py new file mode 100644 index 0000000..87863e1 --- /dev/null +++ b/deploy/deepseek-v41/launch.py @@ -0,0 +1,134 @@ +"""Canonical pinned launch; never stops workloads, pulls images, or reboots a node.""" +from __future__ import annotations + +import argparse +import hashlib +import json +import subprocess +from pathlib import Path + + +def load_config() -> dict[str, str]: + values: dict[str, str] = {} + for line in Path(__file__).with_name("h100-production.env").read_text().splitlines(): + if not line or line.startswith("#"): + continue + key, value = line.split("=", 1) + if key in values: + raise ValueError("duplicate canonical setting") + values[key] = value + return values + + +def server_args(config: dict[str, str]) -> list[str]: + if config["LANGUAGE_MODEL_ONLY"] != "true" or config["ENGRAM_CPU_OFFLOAD"] != "true": + raise ValueError("qualification requires text-only Engram CPU offload") + return [ + "--model", "/model", "--served-model-name", "/model", "deepseek-v4.1-flash", + "--tensor-parallel-size", config["TP"], "--language-model-only", + "--tokenizer-mode", config["TOKENIZER_MODE"], + "--reasoning-parser", config["REASONING_PARSER"], + "--tool-call-parser", config["TOOL_CALL_PARSER"], "--enable-auto-tool-choice", + "--engram-config", '{"cpu_offload":true}', + "--max-model-len", config["MAX_MODEL_LEN"], + "--max-num-seqs", config["MAX_NUM_SEQS"], + "--max-num-batched-tokens", config["MAX_NUM_BATCHED_TOKENS"], + "--gpu-memory-utilization", config["GPU_MEMORY_UTILIZATION"], + "--generation-config", "vllm", + "--host", "0.0.0.0", "--port", "8000", "--no-enable-log-requests", + ] + + +def verify_checkpoint(model: Path, config: dict[str, str]) -> None: + manifest = json.loads(Path(__file__).with_name("checkpoint-manifest.json").read_text()) + if (manifest["model_id"] != config["CHECKPOINT"] + or manifest["revision"] != config["CHECKPOINT_REVISION"]): + raise RuntimeError("upstream manifest/config identity mismatch") + expected_paths = {item["path"] for item in manifest["files"]} + for path in model.rglob("*"): + relative = path.relative_to(model) + if path.is_symlink(): + raise RuntimeError("checkpoint symlinks are not admitted") + if (path.is_file() and relative.as_posix() not in expected_paths + and relative.parts[:2] != (".cache", "huggingface")): + raise RuntimeError("unexpected unpinned checkpoint file") + for item in manifest["files"]: + relative = Path(item["path"]) + if relative.is_absolute() or ".." in relative.parts: + raise ValueError("unsafe checkpoint path") + path = model / relative + if (path.is_symlink() or not path.is_file() or not path.resolve().is_relative_to(model) + or path.stat().st_size != item["size"]): + raise RuntimeError("checkpoint file admission failed") + digest = hashlib.sha256() if item["lfs_sha256"] else hashlib.sha1() + if not item["lfs_sha256"]: + digest.update(b"blob " + str(item["size"]).encode() + b"\0") + with path.open("rb") as handle: + while chunk := handle.read(8 * 1024 * 1024): + digest.update(chunk) + if digest.hexdigest() != (item["lfs_sha256"] or item["git_blob_id"]): + raise RuntimeError("checkpoint content differs from pinned upstream revision") + print(json.dumps({"checkpoint_verified": True, "files": len(manifest["files"])}), flush=True) + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--model-path", required=True, type=Path) + parser.add_argument("--cache-path", required=True, type=Path) + parser.add_argument("--name", default="c3r-deepseek-qualified") + parser.add_argument("--port", type=int, default=18000, + help="Loopback qualification port; production cutover is separately approved") + parser.add_argument("--execute", action="store_true") + parser.add_argument("--server-args", action="store_true", + help="Print canonical serving arguments for real upstream parser validation") + args = parser.parse_args() + config = load_config() + if args.server_args: + print(json.dumps(server_args(config))) + return + model, cache = args.model_path.resolve(), args.cache_path.resolve() + if (not model.is_dir() or not cache.is_dir() or model.is_relative_to(cache) + or cache.is_relative_to(model)): + raise ValueError("distinct existing checkpoint and private cache directories required") + if not 1 <= args.port <= 65535 or not args.name.startswith("c3r-deepseek-"): + raise ValueError("invalid scoped name or loopback port") + command = [ + "docker", "run", "-d", "--name", args.name, "--gpus", "all", + "--network", "bridge", "-p", f"127.0.0.1:{args.port}:8000", + "--read-only", "--cap-drop", "ALL", "--security-opt", "no-new-privileges", + "--shm-size", "16g", "--memory", config["HOST_RAM_LIMIT_GIB"] + "g", + "--memory-swap", config["HOST_RAM_LIMIT_GIB"] + "g", + "--tmpfs", "/tmp:rw,mode=1777,size=2g", + "--mount", f"type=bind,src={model},dst=/model,readonly", + "--mount", f"type=bind,src={cache},dst=/runtime", + "-e", "VLLM_CACHE_ROOT=/runtime/vllm", "-e", "XDG_CACHE_HOME=/runtime/.cache", + "-e", "FLASHINFER_WORKSPACE_BASE=/runtime/flashinfer", + config["VLLM_IMAGE_ID"], *server_args(config), + ] + if not args.execute: + print(json.dumps({"command": command, "qualification_status": "pending"})) + return + verify_checkpoint(model, config) + # Fail before allocating: only the already built immutable image is eligible. + subprocess.run(["docker", "image", "inspect", config["VLLM_IMAGE_ID"]], + check=True, stdout=subprocess.DEVNULL, timeout=20) + gpu_rows = subprocess.check_output([ + "nvidia-smi", "--query-gpu=memory.total,memory.free", "--format=csv,noheader,nounits" + ], text=True, timeout=15).splitlines() + if len(gpu_rows) != int(config["TP"]): + raise RuntimeError("expected eight measured GPUs") + for row in gpu_rows: + total, free = map(int, row.split(",")) + if free < total * float(config["GPU_MEMORY_UTILIZATION"]) + int(config["MIN_GPU_FREE_MIB"]): + raise RuntimeError("insufficient co-resident headroom; do not overlap model engines") + memory = {line.split(":")[0]: int(line.split()[1]) + for line in Path("/proc/meminfo").read_text().splitlines() + if line.split(":")[0] in {"MemTotal", "MemAvailable", "SwapTotal", "SwapFree"}} + if (memory["MemAvailable"] < int(config["HOST_RAM_LIMIT_GIB"]) * 1024**2 + memory["MemTotal"] * .2 + or memory["SwapTotal"] != memory["SwapFree"]): + raise RuntimeError("host-RAM reserve or no-swap gate failed") + subprocess.run(command, check=True, timeout=60) + + +if __name__ == "__main__": + main() diff --git a/deploy/deepseek-v41/start.sh b/deploy/deepseek-v41/start.sh new file mode 100644 index 0000000..1a7ac4c --- /dev/null +++ b/deploy/deepseek-v41/start.sh @@ -0,0 +1,3 @@ +#!/bin/sh +set -eu +exec python3 "$(dirname "$0")/launch.py" "$@" diff --git a/docs/runtime-integrations.md b/docs/runtime-integrations.md index 6a2e9bb..d36123d 100644 --- a/docs/runtime-integrations.md +++ b/docs/runtime-integrations.md @@ -12,16 +12,19 @@ C3R is disabled. ## Deliberative providers -The release default is DeepSeek V4.1 Flash. Canonical identifiers are: +The stateless production target is self-hosted DeepSeek V4.1 Flash alongside +Qwen3-8B/CLM on the existing eight-H100 node. Canonical identifiers are: - Hugging Face checkpoint: `deepseek-ai/DeepSeek-V4.1-Flash` -- OpenRouter: `deepseek/deepseek-v4.1-flash` -- DeepSeek API: `deepseek-flash` +- Local OpenAI-compatible endpoint: `http://127.0.0.1:8000/v1`, alias `/model` This is a provider default, not execution authority. The official checkpoint has passed a private 8×H100 GCP localhost inference smoke test and has a verified private GCS backup; neither result qualifies a public C3R endpoint. `c3r.deliberative.default_provider_config` -constructs hosted profiles without embedding credentials. +constructs this local profile by default. Explicit remote profiles remain +legacy compatibility contracts only, not production routing or fallback. +See [`deploy/deepseek-v41`](../deploy/deepseek-v41/README.md) for the pinned +qualification target, startup correction, and recovery/candidate distinction. `c3r.adapters.providers.ProviderAdapter` supports four protocol shapes: @@ -37,7 +40,7 @@ invalid JSON, oversize content, or non-success HTTP status fail closed. `Provide converts transport, outage, timeout, and malformed-response failures into the deterministic action selected by host policy. API keys are supplied by the host and are absent from results and telemetry. -These are protocol-level contracts. Single credentialed, fixed-prompt OpenRouter probes for +These are protocol-level contracts. Historical credentialed, fixed-prompt remote probes for DeepSeek, Qwen3.8 Flash, and Claude Sonnet 4.6 capture limited latency and usage evidence. A provider becomes release-qualified only after multi-case tests capture model revision, region, latency, usage, failure, and fallback evidence in the integrated C3R path. diff --git a/tests/test_default_language_model.py b/tests/test_default_language_model.py index bb3f5e1..18cb261 100644 --- a/tests/test_default_language_model.py +++ b/tests/test_default_language_model.py @@ -2,26 +2,26 @@ from c3r.deliberative import ( DEFAULT_LANGUAGE_MODEL, - DEFAULT_OPENROUTER_MODEL, default_provider_config, ) class DefaultLanguageModelTests(unittest.TestCase): - def test_openrouter_is_the_default_gateway(self) -> None: - config = default_provider_config(api_key="test-key") + def test_self_hosted_deepseek_is_the_default_gateway(self) -> None: + config = default_provider_config() self.assertEqual(DEFAULT_LANGUAGE_MODEL, "deepseek-ai/DeepSeek-V4.1-Flash") - self.assertEqual(config.model, DEFAULT_OPENROUTER_MODEL) - self.assertEqual(config.base_url, "https://openrouter.ai/api/v1") + self.assertEqual(config.model, "/model") + self.assertEqual(config.base_url, "http://127.0.0.1:8000/v1") + self.assertIsNone(config.api_key) def test_direct_deepseek_uses_official_alias(self) -> None: config = default_provider_config(api_key="test-key", gateway="deepseek") self.assertEqual(config.model, "deepseek-flash") self.assertEqual(config.base_url, "https://api.deepseek.com") - def test_credentials_are_never_optional(self) -> None: + def test_explicit_remote_compatibility_requires_credentials(self) -> None: with self.assertRaises(ValueError): - default_provider_config(api_key="") + default_provider_config(api_key="", gateway="deepseek") if __name__ == "__main__": diff --git a/tests/test_ingress_proxy.py b/tests/test_ingress_proxy.py index 08f6655..db317af 100644 --- a/tests/test_ingress_proxy.py +++ b/tests/test_ingress_proxy.py @@ -164,4 +164,4 @@ def test_upstream_route_must_be_loopback_and_secrets_distinct(self): if __name__ == "__main__": unittest.main() - + diff --git a/tests/test_serve.py b/tests/test_serve.py index 20d49ed..dc29a53 100644 --- a/tests/test_serve.py +++ b/tests/test_serve.py @@ -167,4 +167,4 @@ def test_production_mode_requires_system_one_path(self): if __name__ == "__main__": unittest.main() - + diff --git a/tests/test_stateless_api.py b/tests/test_stateless_api.py index 7feebc4..9a59bae 100644 --- a/tests/test_stateless_api.py +++ b/tests/test_stateless_api.py @@ -212,6 +212,55 @@ def test_positive_cvoc_generation_still_requires_independent_verification(self): self.assertEqual((status, body["error"]), (503, "service_unavailable")) self.assertEqual(calls, []) + def test_hosted_text_generation_cannot_transfer_local_only_state(self): + calls = [] + def hosted_transport(*args): + calls.append(args) + return TransportResponse(200, {"choices": [{"message": {"content": "ready"}, + "finish_reason": "stop"}]}, 10) + provider = ProviderAdapter(ProviderConfig( + "hosted", ProviderKind.OPENAI_COMPATIBLE, "https://openrouter.ai/api/v1", + "deepseek/deepseek-v4.1-flash", "test-not-a-secret"), transport=hosted_transport) + runtime, _ = controller(deliberative=True) + factory = ReadOnlyRequestFactory(definitions=(ActionDefinition( + "DELIBERATE", ActionFamily.DELIBERATE, "compute", "generate", + RiskClass.READ_ONLY, ((),), ("local",), ("policy",), 0, 0, + ),), policy=AuthorityPolicy(frozenset({ActionFamily.DELIBERATE}), + frozenset({RiskClass.READ_ONLY})), + estimate_source=lambda _: {}, remaining_usd=0, data_boundary="local") + self.server.responses = ResponsesService(runtime, factory, provider) + status, body = self.call("/v1/responses", payload={ + "model": "c3r-core", "input": "Local-only synthetic state"}) + self.assertEqual((status, body["error"]), (503, "service_unavailable")) + self.assertEqual(calls, []) + + def test_approved_hosted_responses_enforces_zero_retention_routing(self): + observed = [] + def hosted_transport(url, headers, payload): + observed.append(payload) + return TransportResponse(200, {"choices": [{"message": {"content": "ready"}, + "finish_reason": "stop"}], + "usage": {"prompt_tokens": 5, "completion_tokens": 2, + "cost": 0.000001}}, 10) + provider = ProviderAdapter(ProviderConfig( + "openrouter-deepseek", ProviderKind.OPENAI_COMPATIBLE, "https://openrouter.ai/api/v1", + "deepseek/deepseek-v4.1-flash", "test-not-a-secret"), transport=hosted_transport) + runtime, _ = controller(deliberative=True) + factory = ReadOnlyRequestFactory(definitions=(ActionDefinition( + "DELIBERATE", ActionFamily.DELIBERATE, "compute", "generate", + RiskClass.READ_ONLY, ((),), ("hosted",), ("policy",), 0, 0, + data_boundary="approved_remote", + ),), policy=AuthorityPolicy(frozenset({ActionFamily.DELIBERATE}), + frozenset({RiskClass.READ_ONLY})), + estimate_source=lambda _: {}, remaining_usd=0, data_boundary="approved_remote") + self.server.responses = ResponsesService(runtime, factory, provider) + status, body = self.call("/v1/responses", payload={"model": "c3r-core", "input": "ready"}) + self.assertEqual(status, 200) + self.assertEqual(observed[0]["provider"], {"zdr": True, "data_collection": "deny", + "require_parameters": True, "allow_fallbacks": False}) + self.assertEqual(body["c3r"]["observed_provider_cost_usd"], 0.000001) + self.assertFalse(body["store"]) + def test_new_paths_require_authentication(self): for path in ("/v1/c3r/decide", "/v1/c3r/execute", "/v1/responses"): status, body = self.call(path, token="wrong", payload={"goal": "Find record"}) From a9e91f9c280742e7e7d9c18315c64c2f2136b755 Mon Sep 17 00:00:00 2001 From: wilkont Date: Thu, 1 Oct 2026 12:42:59 -0500 Subject: [PATCH 52/55] Pin self-hosted H100 qualification config and correct startup flag Reviewed local source: 9adf2754ace906306aff1a0a47b7b89c70c54607. Qualification remains pending. --- cloudbuild.qualify.yaml | 95 +++++++++++++++++++++++++++++++++++++++++ 1 file changed, 95 insertions(+) create mode 100644 cloudbuild.qualify.yaml diff --git a/cloudbuild.qualify.yaml b/cloudbuild.qualify.yaml new file mode 100644 index 0000000..876bfa0 --- /dev/null +++ b/cloudbuild.qualify.yaml @@ -0,0 +1,95 @@ +# No local source upload: fetch and verify the exact public commit/tree. +# This is build/scan qualification, NOT GPU startup or release approval. +substitutions: + _SOURCE_COMMIT: '' + _SOURCE_TREE: '' +steps: + - name: gcr.io/cloud-builders/git + id: exact-source + entrypoint: sh + args: + - -c + - >- + test "$${#COMMIT}" = 40 + && case "$$COMMIT" in *[!0-9a-f]*) exit 1;; esac + && git init /workspace/source + && git -C /workspace/source remote add origin https://github.com/ColomboAI-com/c3r.git + && git -C /workspace/source fetch --depth=1 origin "$$COMMIT" + && git -C /workspace/source checkout --detach FETCH_HEAD + && test "$$(git -C /workspace/source rev-parse HEAD)" = "$$COMMIT" + && test "$$(git -C /workspace/source rev-parse 'HEAD^{tree}')" = "$$TREE" + env: ['COMMIT=${_SOURCE_COMMIT}', 'TREE=${_SOURCE_TREE}'] + - name: python:3.11-alpine3.24 + id: unit-311 + dir: source + waitFor: [exact-source] + entrypoint: python + args: [-m, unittest, discover, -s, tests, -v] + - name: python:3.12-alpine3.23 + id: unit-312 + dir: source + waitFor: [exact-source] + entrypoint: python + args: [-m, unittest, discover, -s, tests, -v] + - name: python:3.13-alpine3.23 + id: unit-313 + dir: source + waitFor: [exact-source] + entrypoint: python + args: [-m, unittest, discover, -s, tests, -v] + - name: gcr.io/cloud-builders/git + id: clm-source + dir: source + waitFor: [exact-source] + entrypoint: sh + args: + - -c + - >- + git clone --no-checkout https://github.com/Contrastive-LM/CLM.git /workspace/clm-upstream + && test "$$(git -C /workspace/clm-upstream rev-parse bb42c6c5bf914fd449bed2f6ca65be80602cb1f7^{commit})" = bb42c6c5bf914fd449bed2f6ca65be80602cb1f7 + && git -C /workspace/clm-upstream archive --format=tar + --output=/workspace/source/deploy/clm-source-bb42c6c.tar bb42c6c5bf914fd449bed2f6ca65be80602cb1f7 + - name: gcr.io/cloud-builders/docker + id: api-image + dir: source + waitFor: [unit-311, unit-312, unit-313] + args: [build, --pull, --file, Dockerfile, --tag, 'c3r-api:qualification', .] + - name: gcr.io/cloud-builders/docker + id: clm-image + dir: source + waitFor: [clm-source] + args: [build, --pull, --file, deploy/Dockerfile.clm-hardened, --tag, 'c3r-clm:qualification', deploy] + - name: gcr.io/cloud-builders/docker + id: deepseek-image + dir: source + waitFor: [exact-source] + args: [build, --pull, --file, deploy/deepseek-v41/Dockerfile, --tag, 'c3r-deepseek:qualification', deploy/deepseek-v41] + - name: python:3.13-alpine3.23 + id: serving-config-validation + dir: source + waitFor: [exact-source] + entrypoint: python + args: [deploy/deepseek-v41/launch.py, --model-path, /private/model, --cache-path, /private/cache, --server-args] + - name: gcr.io/cloud-builders/docker + id: api-import + waitFor: [api-image] + args: [run, --rm, 'c3r-api:qualification', python, -c, 'import c3r.serve'] + - name: gcr.io/cloud-builders/docker + id: clm-import + waitFor: [clm-image] + args: [run, --rm, --entrypoint, python3, 'c3r-clm:qualification', -c, 'from clm.engine import Engine; from clm.server import create_app'] + - name: aquasec/trivy:0.74.0 + id: api-security + waitFor: [api-image] + args: [image, --no-progress, --exit-code, '1', --severity, 'HIGH,CRITICAL', 'c3r-api:qualification'] + - name: aquasec/trivy:0.74.0 + id: clm-security + waitFor: [clm-image] + args: [image, --no-progress, --exit-code, '1', --severity, 'HIGH,CRITICAL', 'c3r-clm:qualification'] + - name: aquasec/trivy:0.74.0 + id: deepseek-security + waitFor: [deepseek-image] + args: [image, --no-progress, --exit-code, '1', --severity, 'HIGH,CRITICAL', 'c3r-deepseek:qualification'] +options: + diskSizeGb: 200 +timeout: 3600s From e8b82c30cd8f13937df4642cd8f6873d93ead8b0 Mon Sep 17 00:00:00 2001 From: wilkont Date: Thu, 1 Oct 2026 12:59:59 -0500 Subject: [PATCH 53/55] Pin self-hosted H100 qualification config and correct startup flag Reviewed local source: a7acb8c6127e708ad83e134c7988ff66fa5319e6. Qualification remains pending. --- cloudbuild.qualify.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/cloudbuild.qualify.yaml b/cloudbuild.qualify.yaml index 876bfa0..89f6ee5 100644 --- a/cloudbuild.qualify.yaml +++ b/cloudbuild.qualify.yaml @@ -84,11 +84,11 @@ steps: args: [image, --no-progress, --exit-code, '1', --severity, 'HIGH,CRITICAL', 'c3r-api:qualification'] - name: aquasec/trivy:0.74.0 id: clm-security - waitFor: [clm-image] + waitFor: [clm-image, api-security] args: [image, --no-progress, --exit-code, '1', --severity, 'HIGH,CRITICAL', 'c3r-clm:qualification'] - name: aquasec/trivy:0.74.0 id: deepseek-security - waitFor: [deepseek-image] + waitFor: [deepseek-image, clm-security] args: [image, --no-progress, --exit-code, '1', --severity, 'HIGH,CRITICAL', 'c3r-deepseek:qualification'] options: diskSizeGb: 200 From 197b02613458d92e0a507ab718683f774b8dc6fd Mon Sep 17 00:00:00 2001 From: wilkont Date: Thu, 1 Oct 2026 13:16:18 -0500 Subject: [PATCH 54/55] Pin self-hosted H100 qualification config and correct startup flag Reviewed local source: 5a06bc79a663996183723e5fb39ef73630929d15. Qualification remains pending. --- deploy/deepseek-v41/Dockerfile | 9 +++++++- deploy/deepseek-v41/README.md | 5 +++++ deploy/deepseek-v41/compiler_probe.c | 9 ++++++++ deploy/deepseek-v41/launch.py | 2 ++ tests/test_deepseek_launch.py | 33 ++++++++++++++++++++++++++++ 5 files changed, 57 insertions(+), 1 deletion(-) create mode 100644 deploy/deepseek-v41/compiler_probe.c create mode 100644 tests/test_deepseek_launch.py diff --git a/deploy/deepseek-v41/Dockerfile b/deploy/deepseek-v41/Dockerfile index 049aa48..2f22e96 100644 --- a/deploy/deepseek-v41/Dockerfile +++ b/deploy/deepseek-v41/Dockerfile @@ -5,10 +5,17 @@ LABEL org.opencontainers.image.source="https://github.com/vllm-project/vllm" \ org.opencontainers.image.revision="ac68c3087215e0a4f3cdfa218508c6aada57235d" \ com.colomboai.c3r.qualification="pending" RUN apt-get update && apt-get upgrade -y \ - && apt-get purge -y linux-libc-dev python3-msgpack python3-setuptools python3-urllib3 \ + && apt-get install -y --no-install-recommends libc6-dev python3.12-dev \ + && apt-get purge -y python3-msgpack python3-setuptools python3-urllib3 \ && rm -rf /var/lib/apt/lists/* \ && python3 -m pip uninstall --break-system-packages -y opentelemetry-exporter-otlp-common COPY check_upstream_dependencies.py /opt/c3r/check_upstream_dependencies.py +COPY compiler_probe.c /opt/c3r/compiler_probe.c +# Triton compiles CUDA helpers at runtime. Removing development headers can make +# a vulnerability scan clean while breaking the actual model startup. +RUN gcc -fsyntax-only -I/usr/include/python3.12 \ + -I/usr/local/lib/python3.12/dist-packages/triton/backends/nvidia/include \ + /opt/c3r/compiler_probe.c RUN python3 /opt/c3r/check_upstream_dependencies.py \ && python3 -m pip uninstall --break-system-packages -y pip ENV HOME=/runtime \ diff --git a/deploy/deepseek-v41/README.md b/deploy/deepseek-v41/README.md index 81e8a44..75b725a 100644 --- a/deploy/deepseek-v41/README.md +++ b/deploy/deepseek-v41/README.md @@ -5,6 +5,11 @@ System Two, and C3R Core on the existing eight-H100 node. No new GPU, OpenRouter, or hosted inference provider is part of this deployment. `h100-production.env` is a **qualification target**, not a production approval. +The currently pinned candidate failed Triton runtime compilation because its +hardening removed required C headers. **Do not execute that image.** The repaired +Dockerfile preserves patched development headers and runs a compile-only probe; +its rebuilt image must independently pass security scans and receive a new pin +before another supervised GPU load. A clean scanner result alone is insufficient. It pins the locally built immutable candidate image, upstream source/base image, checkpoint revision, TP8, 85% reservation, Engram CPU offload, 32K context, two sequences, batch limit, and both parsers. Registry publication of that local diff --git a/deploy/deepseek-v41/compiler_probe.c b/deploy/deepseek-v41/compiler_probe.c new file mode 100644 index 0000000..c9a85a6 --- /dev/null +++ b/deploy/deepseek-v41/compiler_probe.c @@ -0,0 +1,9 @@ +/* Build-time preflight for headers required by Triton's runtime C helpers. + * Passing this probe is not proof of GPU kernels, model startup, or quality. */ +#include +#include +#include + +int main(void) { + return EXIT_SUCCESS; +} diff --git a/deploy/deepseek-v41/launch.py b/deploy/deepseek-v41/launch.py index 87863e1..703b7e7 100644 --- a/deploy/deepseek-v41/launch.py +++ b/deploy/deepseek-v41/launch.py @@ -108,6 +108,8 @@ def main() -> None: if not args.execute: print(json.dumps({"command": command, "qualification_status": "pending"})) return + if config["VLLM_IMAGE_ID"] == "sha256:10b3c8fe9c38f6e87dfef21c8d0e457f76ab89b32892bb37a056375b25ddbf85": + raise RuntimeError("known failed compiler preflight; rebuild, scan and repin before execution") verify_checkpoint(model, config) # Fail before allocating: only the already built immutable image is eligible. subprocess.run(["docker", "image", "inspect", config["VLLM_IMAGE_ID"]], diff --git a/tests/test_deepseek_launch.py b/tests/test_deepseek_launch.py new file mode 100644 index 0000000..8b4a8a3 --- /dev/null +++ b/tests/test_deepseek_launch.py @@ -0,0 +1,33 @@ +"""Regression checks at the canonical launcher CLI boundary.""" + +from pathlib import Path +import shutil +import subprocess +import sys +import tempfile +import unittest + + +class DeepSeekLaunchTests(unittest.TestCase): + def test_known_broken_image_cannot_allocate(self): + launcher = Path(__file__).resolve().parents[1] / "deploy/deepseek-v41/launch.py" + with tempfile.TemporaryDirectory() as directory: + fixture = Path(directory) / "launch.py" + shutil.copy2(launcher, fixture) + config = launcher.with_name("h100-production.env").read_text() + lines = [ + "VLLM_IMAGE_ID=sha256:10b3c8fe9c38f6e87dfef21c8d0e457f76ab89b32892bb37a056375b25ddbf85" + if line.startswith("VLLM_IMAGE_ID=") else line + for line in config.splitlines() + ] + fixture.with_name("h100-production.env").write_text("\n".join(lines)) + model, cache = Path(directory) / "model", Path(directory) / "cache" + model.mkdir() + cache.mkdir() + result = subprocess.run( + [sys.executable, str(fixture), "--model-path", str(model), + "--cache-path", str(cache), "--execute"], + capture_output=True, text=True, timeout=10, check=False, + ) + self.assertNotEqual(result.returncode, 0) + self.assertIn("known failed compiler preflight", result.stderr) From 0333653f24c04355550450df8ecd6c5fd1aee2d7 Mon Sep 17 00:00:00 2001 From: wilkont Date: Thu, 1 Oct 2026 13:20:23 -0500 Subject: [PATCH 55/55] Pin self-hosted H100 qualification config and correct startup flag Reviewed local source: 45f2a864e95892ea8ed465fd0a49bcda4d29e7b9. Qualification remains pending. --- tests/test_deepseek_launch.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_deepseek_launch.py b/tests/test_deepseek_launch.py index 8b4a8a3..b4e56f6 100644 --- a/tests/test_deepseek_launch.py +++ b/tests/test_deepseek_launch.py @@ -1,11 +1,11 @@ """Regression checks at the canonical launcher CLI boundary.""" -from pathlib import Path import shutil import subprocess import sys import tempfile import unittest +from pathlib import Path class DeepSeekLaunchTests(unittest.TestCase):