diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..351e208 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,13 @@ +.git +.github +__pycache__ +*.pyc +.pytest_cache +.ruff_cache +.venv +tests +data +evidence +docs +assets + diff --git a/.gcloudignore b/.gcloudignore new file mode 100644 index 0000000..f5b0533 --- /dev/null +++ b/.gcloudignore @@ -0,0 +1,12 @@ +# C3R staging build uploads runtime source only. Do not send papers, traces, +# local evidence, secrets, or repository metadata to Cloud Build. +** +!Dockerfile +!Dockerfile.retention +!cloudbuild.retention.yaml +!.dockerignore +!c3r/ +!c3r/** +**/__pycache__/ +**/*.pyc + diff --git a/.gcloudignore.verify b/.gcloudignore.verify new file mode 100644 index 0000000..45e37b9 --- /dev/null +++ b/.gcloudignore.verify @@ -0,0 +1,16 @@ +# Separate verification upload: source and synthetic tests only. +# Never upload local evidence, credentials, trace rows, papers, or Git metadata. +** +!Dockerfile +!Dockerfile.retention +!cloudbuild.verify.yaml +!.dockerignore +!c3r/ +!c3r/** +!tests/ +!tests/** +!scripts/ +!scripts/run_controlled_pairs.py +!scripts/run_trace_control_dry_run.py +**/__pycache__/ +**/*.pyc diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index fdd4afb..c10d5e7 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -20,3 +20,53 @@ jobs: python-version: ${{ matrix.python-version }} - run: python -m unittest discover -s tests -v + image-build: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - run: docker build --pull --tag c3r:ci . + - run: docker run --rm c3r:ci python -c 'import c3r.serve' + - name: Scan image for high and critical vulnerabilities + uses: aquasecurity/trivy-action@v0.36.0 + with: + version: v0.74.0 + image-ref: c3r:ci + format: json + output: trivy-image.json + severity: HIGH,CRITICAL + exit-code: '1' + - name: Preserve scan findings + if: always() + uses: actions/upload-artifact@v4 + with: + name: trivy-image-${{ github.run_id }} + path: trivy-image.json + if-no-files-found: ignore + retention-days: 30 + + retention-image-build: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - run: docker build --pull --file Dockerfile.retention --tag c3r-retention:ci . + - name: Verify retention entrypoint + run: | + test "$(docker image inspect c3r-retention:ci --format '{{json .Config.Cmd}}')" = '["python","-m","c3r.retention_job"]' + - run: docker run --rm c3r-retention:ci python -c 'import c3r.retention_job' + - name: Scan retention image for high and critical vulnerabilities + uses: aquasecurity/trivy-action@v0.36.0 + with: + version: v0.74.0 + image-ref: c3r-retention:ci + format: json + output: trivy-retention-image.json + severity: HIGH,CRITICAL + exit-code: '1' + - name: Preserve retention scan findings + if: always() + uses: actions/upload-artifact@v4 + with: + name: trivy-retention-image-${{ github.run_id }} + path: trivy-retention-image.json + if-no-files-found: ignore + retention-days: 30 diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..56396d8 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,10 @@ +FROM python:3.11-alpine3.24@sha256:cd04730b8511def3fbf14204d66a0c1536f290b8e896ed5a94cd64cb15ac1356 + +ENV PYTHONDONTWRITEBYTECODE=1 PYTHONUNBUFFERED=1 +WORKDIR /app +RUN python -m pip uninstall -y setuptools wheel jaraco.context +COPY --chown=10001:10001 c3r ./c3r +USER 10001:10001 +EXPOSE 8080 +CMD ["python", "-m", "c3r.serve"] + diff --git a/Dockerfile.retention b/Dockerfile.retention new file mode 100644 index 0000000..9146807 --- /dev/null +++ b/Dockerfile.retention @@ -0,0 +1,9 @@ +FROM python:3.11-alpine3.24@sha256:cd04730b8511def3fbf14204d66a0c1536f290b8e896ed5a94cd64cb15ac1356 + +ENV PYTHONDONTWRITEBYTECODE=1 PYTHONUNBUFFERED=1 +WORKDIR /app +RUN python -m pip uninstall -y setuptools wheel jaraco.context +COPY --chown=10001:10001 c3r ./c3r +USER 10001:10001 +CMD ["python", "-m", "c3r.retention_job"] + diff --git a/NOTICE b/NOTICE index 159169e..a6c84f2 100644 --- a/NOTICE +++ b/NOTICE @@ -13,3 +13,8 @@ Upstream model: https://huggingface.co/convaiinnovations/laya Upstream source: https://github.com/NandhaKishorM/laya No derived Laya weights are included in this repository at this time. + +The optional CLM service adapter targets Contrastive-LM/CLM at upstream source +commit bb42c6c5bf914fd449bed2f6ca65be80602cb1f7. Upstream CLM source is +Apache-2.0: https://github.com/Contrastive-LM/CLM. No CLM source, encoder +weights, or projection-head weights are redistributed in this repository. diff --git a/README.md b/README.md index ae78b96..95f769a 100644 --- a/README.md +++ b/README.md @@ -15,24 +15,65 @@ C3R is an open-core control plane that decides **which computation is worth perf It evaluates tools, retrieval, local and frontier models, verification, placement, and stopping as typed candidates under one conservative value-of-computation policy. -The v0.1 vertical slice now includes bounded hierarchical candidate compilation, a revision-verified Laya -fast path with calibration-gated abstention, and a reproducible DecisionMix v1 schema preview. -It is a research alpha: the controller is runnable and tested; fine-tuned weights and empirical -production calibration remain release gates, not implied claims. +The v0.1 vertical slice includes bounded hierarchical candidate compilation, a +calibration-gated System-One seam, and a reproducible DecisionMix v1 schema preview. +CLM is the new default System-One provider in code; Laya remains optional. +It is a research alpha: the controller is runnable and tested. Fine-tuned weights +and empirical calibration remain gates for the **empirical model/data release**, +not for the separate stateless recommendation API described below. + +> **Launch status:** Neither the standalone decision service nor governed trace collection +> is enabled for public use. The independent [PR #2 review](https://github.com/ColomboAI-com/c3r/pull/2#pullrequestreview-5293835066) +> requests changes. A real cloud storage audit probe and the first natural +> empty-bucket purge run are recorded for reviewer inspection, but they do not +> establish deletion of aged traces or backups, trained weights, or calibration +> for the research release. The separate stateless API still needs its own +> security, live provider, and canary evidence. See the [reviewer packet](docs/reviewer-staging-packet-2026-09-23.md). + +### Separate stateless API path + +The [C3R Core API v1 contract](docs/stateless-core-api.md) separates a +recommendation-only, non-persistent inference service from the governed trace +collection and empirical-release program above. The current branch implements +the typed CLM API, direct ranking, a text-only Responses subset, a production +host, and a fail-closed `production_inference` mode. A private, loopback-only +candidate has returned actual CLM rankings and local DeepSeek text; it is +**not** a publicly launched or production-qualified API. Upstream CLM provides +advisory System-One ranking, not generative text or calibrated task-success +probabilities. `/v1/c3r/execute` remains disabled. `/v1/responses` invokes +DeepSeek through an admitted, independently checked text-only controller +fallback; it does not claim positive learned CVoC or expose private reasoning. +The first public hostname is a dedicated C3R endpoint, not an MC-1 integration. +See the [scope-specific release policy](docs/release-policy.md). + +The [developer API guide](docs/developer-api.md) covers SDK examples, scoped keys, +tenancy, streaming, limits and privacy for the integration candidate. The +[branch reconciliation record](docs/release-reconciliation.md) explains how the +production and readiness histories were joined without losing reviewed source. ### Default language model C3R's default **deliberative** language model is [`deepseek-ai/DeepSeek-V4.1-Flash`](https://huggingface.co/deepseek-ai/DeepSeek-V4.1-Flash) -(MIT). The hosted default uses `deepseek/deepseek-v4.1-flash` through OpenRouter; the -official DeepSeek API alias is `deepseek-flash`. Laya remains the separate System-One -decision fast path, and DeepSeek recommendations remain subject to the same verifier and +(MIT), self-hosted on the existing eight-H100 node alongside Qwen3-8B/CLM. +The default provider uses the private loopback `/model` alias; no OpenRouter or +hosted inference provider is part of the production target. CLM is the default separate +System-One decision engine; Laya remains optional. DeepSeek recommendations +remain subject to the same verifier and trusted commit boundary as every other candidate. -The 763B-parameter checkpoint is not assumed to fit a single GPU merely because only -8B/16B parameters are active per token. A GCP deployment must pass storage, aggregate -accelerator memory, runtime-version, and smoke-test gates before C3R labels it self-hosted; -otherwise the GPU service uses the hosted provider profile and stores no model weights. +The official checkpoint has passed a private 8×H100 GCP serving smoke test and is backed +up in a private GCS bucket. Its vLLM endpoint is bound to localhost; this is **not** a +public C3R service or end-to-end production qualification. The checkpoint's active +parameter count does not imply it fits on one GPU. + +The recovered **older** serving image passed all 13 private C3R checks at 85% +GPU reservation with Qwen still running. The clean, pinned newer image is a +separate qualification target: its first startup rejected an obsolete flag. +The corrected launch configuration is tracked in +[`deploy/deepseek-v41`](deploy/deepseek-v41/README.md). Repeated cold starts, +sustained mixed load, outage/rollback, final-head builds, TLS and canary remain +required; recovery alone is not production qualification. ## Why C3R @@ -42,7 +83,7 @@ keeps that recommendation separate from authority to act. ```mermaid flowchart LR S[Versioned state] --> C[Hierarchical candidate compiler] - C --> L[Laya fast path] + C --> L[CLM default / Laya optional] C --> D[Deliberative envelope] L --> V[Robust CVoC] D --> V @@ -60,11 +101,12 @@ failed verification into success, or directly commit an external effect. | Surface | Included now | | --- | --- | | Candidate Compiler | family → subgroup → operation → arguments → placement → verifier; hard masks, budget pruning, caps, progressive widening | -| Laya fast path | exact upstream revision and license verification, typed probabilities, slice calibration, confidence/margin abstention | +| System-One fast path | CLM loopback rank adapter with strict response checks and calibration-gated abstention; optional revision-verified Laya | | DecisionMix v1 | validated records, immutable deterministic splits, source/license provenance, SHA-256 manifest, 144-record synthetic preview | | Authority boundary | action-bound verifier attestations, expiring single-use approvals, atomic nonce claims | -| Runtime control | conservative CVoC selection, deterministic `STOP`, cost twin, deliberative contracts, evidence-grade traces | +| Runtime control | conservative CVoC selection, deterministic `STOP`, cost twin, deliberative contracts, trace schema and optional transactional SQLite hash chain | | Operational controls | fail-closed feature flags, tested frontier/open-weight HTTP contracts, Colibri shadow recommendations, canonical trace hash chain | +| Standalone controller boundary | tested state → candidates → optional System-One → CVoC → independent verifier → read-only recommendation or deterministic fallback → redacted trace composition; the standalone controller rejects external executors until effects and durable trace commits can be made atomic. A separate private Cloud Run staging host is fixed-disabled; it is not the decision service or a public launch. | This compiler is the reviewed vertical slice, not the directive's full Candidate Compiler Definition of Done. Rich typed value constraints, per-argument provenance, dominated-branch @@ -127,13 +169,42 @@ is `STOP`. every Definition of Done item in Execution Directive v2. - [`docs/empirical-release-plan.md`](docs/empirical-release-plan.md) — gated path from synthetic preview to trained, calibrated, independently reproducible release. +- [`docs/standalone-launch.md`](docs/standalone-launch.md) — current production exit gates, + evidence status, and operator inputs, with MC-1 excluded from standalone scope only. - [`docs/launch-announcement.md`](docs/launch-announcement.md) — canonical launch copy plus LinkedIn, X, and Hacker News variants with a publication checklist. - [`docs/prior-art.md`](docs/prior-art.md) — explicit attribution links and the canonical novelty boundary required by the execution directive. +## CLM System-One integration + +`C3R_SYSTEM_ONE_PROVIDER=clm` is the configuration default, while +`C3R_ENABLED` and `C3R_SYSTEM_ONE` still default to off. The adapter targets +the upstream [Contrastive-LM/CLM](https://github.com/Contrastive-LM/CLM) +`/v1/rank` API on a loopback-only origin. Its source is pinned at +`bb42c6c5bf914fd449bed2f6ca65be80602cb1f7` (Apache-2.0). The running +encoder and CLM head need their own immutable artifact revision, passed as +`C3R_CLM_ARTIFACT_REVISION`; the current adapter validates the declaration's +format but does not yet attest the live server's artifact hash. This repository +does not bundle or claim trained C3R-specific CLM weights. The host supplies `C3R_CLM_URL` (default +`http://127.0.0.1:8700`), an optional `C3R_CLM_API_KEY`, and a fitted +`TemperatureCalibrator` to `build_default_clm_fast_path`. The initial +`C3R_CLM_TIMEOUT_MS=500` bounds the complete System-One decision; production +latency thresholds still require live measurement. + +CLM ranks bounded, policy-surviving candidate labels and fixed typed questions. +Its raw ranking is advisory, not an action selection. Missing calibration, +malformed responses, outages, or timeouts lead to abstention or deterministic +fallback. CVoC, independent verification, and the commit boundary retain +authority. Private live CLM ranking and co-resident DeepSeek recovery have been +observed; neither sustained GPU coexistence qualification nor held-out C3R +calibration is claimed by this code change. + ## Laya integration +Laya remains an explicitly selected comparison/compatibility provider; it is not +the default System-One engine. + The adapter pins `convaiinnovations/laya` at `1c5edc17a7acd8701df6fc341c0d179f1c62c982`. Before loading, the backend resolves the Hub metadata, verifies that exact SHA and the Apache-2.0 license, downloads that revision, and passes @@ -152,6 +223,12 @@ without presenting generated fixtures as real training evidence. The empirical c only when provenance, licensing, held-out integrity, and calibration support are independently auditable. +For the first empirical source, ColomboAI approved only C3R-authored internal +tasks under the [interim trace policy](docs/internal-task-trace-policy.md). It +sets a 30-day private retention limit and requires independent review before +any de-identified row is published. Collection remains off until the technical +controls are verified; this approval does not make the preview empirical. + Required empirical metrics include accuracy, Brier score, ECE, maximum calibration error, NLL, selective risk versus coverage, abstention, escalation, p50/p95 latency, throughput, calls avoided, and cost per completed task. @@ -163,6 +240,7 @@ Production hosts must preserve independent control of: ```text C3R_ENABLED C3R_SYSTEM_ONE +C3R_SYSTEM_ONE_PROVIDER=clm C3R_DELIBERATIVE C3R_ROUTING C3R_SPECULATION @@ -177,7 +255,7 @@ learning. The reference provider and Colibri shadow contracts are documented in ## Release truth -Not yet claimed: trained `C3R-Decision-Laya-421M-v0.1` weights, empirical DecisionMix training +Not yet claimed: C3R-trained CLM heads or Laya weights, empirical DecisionMix training data, live provider qualification, MC-1 product integration, a Colibri shadow deployment, or measured production calibration/latency/cost results. Tested adapter and shadow-control contracts are included, but they are not represented as production runs. These remain documented gates. @@ -192,3 +270,4 @@ Do not report vulnerabilities in a public issue; follow [`SECURITY.md`](SECURITY effects must be completely mediated by an independently configured commit gateway. Apache License 2.0. See [`LICENSE`](LICENSE). If you use C3R, cite [`CITATION.cff`](CITATION.cff). + diff --git a/c3r/adapters/providers.py b/c3r/adapters/providers.py index a33fe28..1f00503 100644 --- a/c3r/adapters/providers.py +++ b/c3r/adapters/providers.py @@ -3,14 +3,19 @@ from __future__ import annotations import json -from collections.abc import Callable, Mapping -from dataclasses import dataclass +import socket +from collections.abc import Callable, Iterator, Mapping +from dataclasses import dataclass, field from enum import StrEnum +from http.client import HTTPConnection, HTTPResponse, HTTPSConnection +from math import isfinite +from threading import Event, Thread from typing import Protocol, cast from urllib.parse import urlparse -from urllib.request import Request, urlopen +from urllib.request import ProxyHandler, Request, build_opener from ..deliberative.envelope import DeliberativeResult +from ..http_transport import NoRedirectHandler from ..state_schema import ActionFamily MAX_RESPONSE_BYTES = 65_536 @@ -32,7 +37,7 @@ class ProviderConfig: kind: ProviderKind base_url: str model: str - api_key: str | None + api_key: str | None = field(repr=False) timeout_seconds: float = 60.0 def __post_init__(self) -> None: @@ -40,11 +45,21 @@ def __post_init__(self) -> None: local = parsed.hostname in {"localhost", "127.0.0.1", "::1"} if parsed.scheme != "https" and not (parsed.scheme == "http" and local): raise ValueError("remote provider endpoints must use HTTPS") + if parsed.username or parsed.password or parsed.query or parsed.fragment: + raise ValueError("provider origin must not contain credentials, query, or fragment") if not self.provider_id or not self.model: raise ValueError("provider_id and model are required") - if self.timeout_seconds <= 0: + if not isfinite(self.timeout_seconds) or not 0 < self.timeout_seconds <= 120: raise ValueError("timeout_seconds must be positive") + @property + def is_local(self) -> bool: + return urlparse(self.base_url).hostname in {"localhost", "127.0.0.1", "::1"} + + @property + def uses_openrouter(self) -> bool: + return self.base_url.rstrip("/") == "https://openrouter.ai/api/v1" + @dataclass(frozen=True, slots=True) class DeliberationRequest: @@ -85,7 +100,8 @@ class ControlledDeliberation: def _number(value: object) -> float: if isinstance(value, bool) or not isinstance(value, (int, float)): return 0.0 - return float(value) + number = float(value) + return number if isfinite(number) and number >= 0 else 0.0 def _string_tuple(value: object) -> tuple[str, ...]: @@ -119,7 +135,7 @@ def _default_transport( method="POST", ) started = time.monotonic() - with urlopen(request, timeout=timeout) as response: + with build_opener(ProxyHandler({}), NoRedirectHandler()).open(request, timeout=timeout) as response: raw = response.read(MAX_RESPONSE_BYTES + 1) status = int(response.status) if len(raw) > MAX_RESPONSE_BYTES: @@ -235,6 +251,120 @@ def extract(self, body: Mapping[str, object]) -> tuple[str, dict[str, float]]: } +class TextGenerationStream(Iterator[tuple[str, str | None, dict[str, int] | None]]): + """Closeable, bounded reader of an actual OpenAI-compatible generation stream.""" + + def __init__(self, response: HTTPResponse, connection: HTTPConnection, + backend_socket: socket.socket) -> None: + self._response = response + self._connection = connection + self._backend_socket = backend_socket + self._finished = False + self._reading = False + self._saw_finish = False + self._bytes = 0 + + def __iter__(self) -> TextGenerationStream: + return self + + def __next__(self) -> tuple[str, str | None, dict[str, int] | None]: + if self._finished: + raise StopIteration + try: + while True: + self._reading = True + try: + line = self._response.readline(16_385) + finally: + self._reading = False + if self._finished: + self._response.close() + if not line or len(line) > 16_384: + raise RuntimeError("generation stream ended without completion") + self._bytes += len(line) + if self._bytes > MAX_RESPONSE_BYTES: + raise RuntimeError("generation stream exceeds byte limit") + if not line.startswith(b"data: "): + continue + data = line[6:].strip() + if data == b"[DONE]": + self.close() + if not self._saw_finish: + raise RuntimeError("generation stream ended without a finish reason") + raise StopIteration + event_raw = json.loads(data) + if not isinstance(event_raw, dict): + raise TypeError("invalid generation stream event") + event = cast(Mapping[str, object], event_raw) + choices = event.get("choices") + usage = event.get("usage") + if usage is not None: + if not isinstance(usage, dict): + raise TypeError("invalid generation usage") + usage = cast(Mapping[str, object], usage) + input_tokens = usage.get("prompt_tokens") + output_tokens = usage.get("completion_tokens") + if (isinstance(input_tokens, bool) or not isinstance(input_tokens, int) + or input_tokens < 0 or isinstance(output_tokens, bool) + or not isinstance(output_tokens, int) or output_tokens < 0): + raise ValueError("invalid generation token counts") + counts = { + "input_tokens": input_tokens, + "output_tokens": output_tokens, + } + else: + counts = None + if not isinstance(choices, list): + raise TypeError("invalid generation choices") + choices = cast(list[object], choices) + if len(choices) > 1: + raise ValueError("invalid generation choices") + if not choices: + if counts is None: + raise ValueError("empty generation stream event") + return "", None, counts + choice = choices[0] + if not isinstance(choice, dict): + raise TypeError("invalid generation choice") + choice = cast(Mapping[str, object], choice) + delta = choice.get("delta") + if not isinstance(delta, dict): + raise TypeError("invalid generation delta") + delta = cast(Mapping[str, object], delta) + if delta.get("tool_calls") or delta.get("function_call"): + raise ValueError("invalid generation delta") + content = delta.get("content", "") + if not isinstance(content, str): + raise TypeError("invalid generation text") + finish = choice.get("finish_reason") + if finish is not None and not isinstance(finish, str): + raise TypeError("invalid generation finish reason") + if finish not in {None, "stop", "length"}: + raise ValueError("invalid generation finish reason") + if finish is not None: + self._saw_finish = True + return content, finish, counts + except StopIteration: + raise + except (OSError, TypeError, ValueError, RuntimeError, json.JSONDecodeError): + self.close() + raise + + def abort(self) -> None: + if not self._finished: + self._finished = True + try: + self._backend_socket.shutdown(socket.SHUT_RDWR) + except OSError: + pass + self._connection.close() + + def close(self) -> None: + self.abort() + if not self._reading: + self._response.close() + + class ProviderAdapter: def __init__( self, config: ProviderConfig, *, transport: Transport = _default_transport @@ -243,6 +373,138 @@ def __init__( self._transport = transport self._codec = _CODECS[config.kind] + def generate(self, text: str, max_output_tokens: int) -> tuple[str, str, dict[str, float]]: + """Bounded text-only OpenAI-compatible generation; never returns hidden reasoning.""" + if self.config.kind not in {ProviderKind.OPENAI, ProviderKind.OPENAI_COMPATIBLE}: + raise ValueError("text generation requires an OpenAI-compatible provider") + if not text or len(text.encode("utf-8")) > 16384 or not 1 <= max_output_tokens <= 2048: + raise ValueError("generation request outside bounds") + headers = {"X-C3R-Timeout": str(self.config.timeout_seconds)} + if self.config.api_key: + headers["Authorization"] = "Bearer " + self.config.api_key + payload: dict[str, object] = { + "model": self.config.model, "max_tokens": max_output_tokens, "temperature": 0, + "messages": [ + {"role": "system", "content": "Provide a helpful final answer only. Do not expose " + "private reasoning or claim to execute tools, commit actions, or grant authority."}, + {"role": "user", "content": text}, + ], + } + if self.config.uses_openrouter: + if not self.config.api_key: + raise RuntimeError("hosted provider credential unavailable") + payload["provider"] = {"zdr": True, "data_collection": "deny", + "require_parameters": True, "allow_fallbacks": False} + try: + response = self._transport(self.config.base_url.rstrip("/") + "/chat/completions", + headers, payload) + size = len(json.dumps(response.body, allow_nan=False).encode()) + except (OSError, ValueError, TypeError) as error: + raise RuntimeError("generative provider transport unavailable") from error + if not 200 <= response.status < 300: + raise RuntimeError("generative provider unavailable") + if size > MAX_RESPONSE_BYTES: + raise RuntimeError("generative provider response oversized") + try: + choices = cast(list[dict[str, object]], response.body["choices"]) + message = cast(dict[str, object], choices[0]["message"]) + if message.get("tool_calls") or message.get("function_call"): + raise ValueError("tool execution is not supported") + output = _content(message["content"]) + reason = choices[0].get("finish_reason") + if not output.strip() or reason not in {"stop", "length"}: + raise ValueError("invalid generation result") + usage = cast(Mapping[str, object], response.body.get("usage", {})) + observed = { + "input_tokens": _number(usage.get("prompt_tokens")), + "output_tokens": _number(usage.get("completion_tokens")), + "latency_ms": response.latency_ms, + } + cost = usage.get("cost") + if (isinstance(cost, (int, float)) and not isinstance(cost, bool) + and isfinite(cost) and cost >= 0): + observed["provider_cost_usd"] = float(cost) + return output, cast(str, reason), observed + except (KeyError, IndexError, TypeError, ValueError, AttributeError) as error: + raise RuntimeError("invalid generative provider response") from error + + def stream_generate(self, text: str, max_output_tokens: int, *, + cancelled: Event | None = None) -> TextGenerationStream: + """Open a live backend SSE response; closing it aborts backend generation.""" + if self.config.kind not in {ProviderKind.OPENAI, ProviderKind.OPENAI_COMPATIBLE}: + raise ValueError("text generation requires an OpenAI-compatible provider") + if not text or len(text.encode("utf-8")) > 16384 or not 1 <= max_output_tokens <= 2048: + raise ValueError("generation request outside bounds") + headers = {"Content-Type": "application/json"} + if self.config.api_key: + headers["Authorization"] = "Bearer " + self.config.api_key + payload: dict[str, object] = { + "model": self.config.model, "max_tokens": max_output_tokens, "temperature": 0, + "stream": True, "stream_options": {"include_usage": True}, + "messages": [ + {"role": "system", "content": "Provide a helpful final answer only. Do not expose " + "private reasoning or claim to execute tools, commit actions, or grant authority."}, + {"role": "user", "content": text}, + ], + } + if self.config.uses_openrouter: + if not self.config.api_key: + raise RuntimeError("hosted provider credential unavailable") + payload["provider"] = {"zdr": True, "data_collection": "deny", + "require_parameters": True, "allow_fallbacks": False} + parsed = urlparse(self.config.base_url) + if parsed.hostname is None: + raise ValueError("provider origin requires a host") + connection_type = HTTPSConnection if parsed.scheme == "https" else HTTPConnection + connection = connection_type( + parsed.hostname, parsed.port, timeout=self.config.timeout_seconds, + ) + finished_opening = Event() + + def interrupt_opening() -> None: + while not finished_opening.wait(0.05): + if cancelled is not None and cancelled.is_set(): + transport = connection.sock + if transport is not None: + try: + transport.shutdown(socket.SHUT_RDWR) + except OSError: + pass + connection.close() + return + + watcher = Thread(target=interrupt_opening, daemon=True) if cancelled is not None else None + if watcher is not None: + watcher.start() + try: + if cancelled is not None and cancelled.is_set(): + raise RuntimeError("generation cancelled") + connection.connect() + backend_socket = connection.sock + if backend_socket is None: + raise OSError("provider connection unavailable") + if cancelled is not None and cancelled.is_set(): + raise RuntimeError("generation cancelled") + connection.request( + "POST", parsed.path.rstrip("/") + "/chat/completions", + body=json.dumps(payload, separators=(",", ":")).encode(), headers=headers, + ) + response = connection.getresponse() + if cancelled is not None and cancelled.is_set(): + response.close() + raise RuntimeError("generation cancelled") + if not 200 <= response.status < 300 or response.headers.get_content_type() != "text/event-stream": + response.close() + raise RuntimeError("provider did not return a successful event stream") + return TextGenerationStream(response, connection, backend_socket) + except BaseException: + connection.close() + raise + finally: + finished_opening.set() + if watcher is not None: + watcher.join(timeout=0.2) + def deliberate(self, request: DeliberationRequest) -> ProviderExecutionResult: state = json.dumps(dict(request.state), sort_keys=True, separators=(",", ":")) url, headers, payload = self._codec.build(self.config, state) diff --git a/c3r/api_access.py b/c3r/api_access.py new file mode 100644 index 0000000..52efd06 --- /dev/null +++ b/c3r/api_access.py @@ -0,0 +1,217 @@ +"""Durable hash-only developer credentials and payload-free tenant metadata. + +This store is an operator-owned system boundary. It never retains API plaintext +keys, prompts, output text, candidates, or reasoning. Database IAM/encryption, +backup/metadata-retention policy and multi-instance coordination need deployment +qualification; a local SQLite implementation does not establish those controls. +""" +import hashlib +import hmac +import json +import math +import os +import re +import secrets +import sqlite3 +import time +from collections.abc import Callable, Generator +from contextlib import contextmanager +from dataclasses import dataclass +from pathlib import Path +from typing import cast + +SCOPES = frozenset({"responses:write", "system_one:write", "rank:write", "decide:write", + "models:read"}) +KEY = re.compile(r"c3r_sk_(live|test)_([a-f0-9]{24})_([A-Za-z0-9_-]{43})") + + +@dataclass(frozen=True, slots=True) +class IssuedKey: + key_id: str + secret: str + + +@dataclass(frozen=True, slots=True) +class Principal: + tenant_id: str + project_id: str + key_id: str + scopes: frozenset[str] + + +def _identifier(value: str) -> None: + if re.fullmatch(r"[A-Za-z0-9][A-Za-z0-9_.-]{0,63}", value) is None: + raise ValueError("bounded organization/project identifier required") + + +class AccessStore: + def __init__(self, path: Path, *, clock: Callable[[], float] = time.time) -> None: + if (not path.is_absolute() or not path.parent.is_dir() + or any(p.is_symlink() for p in (path, *path.parents))): + raise ValueError("non-linked absolute access database path and existing directory required") + if os.name == "posix": + for ancestor in path.parents: + info = ancestor.stat() + sticky_root = ancestor != path.parent and info.st_uid == 0 and bool(info.st_mode & 0o1000) + unsafe_mode = bool(info.st_mode & (0o077 if ancestor == path.parent else 0o022)) + if info.st_uid not in {0, os.geteuid()} or (unsafe_mode and not sticky_root): + raise ValueError("private operator-owned access database directory required") + if not path.exists(): + descriptor = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600) + os.close(descriptor) + if not path.is_file(): + raise ValueError("access database must be a regular file") + if os.name == "posix" and (path.stat().st_uid != os.geteuid() or path.stat().st_mode & 0o077): + raise ValueError("access database ownership/mode unsafe") + self.path, self.clock = path, clock + with self.connect() as connection: + connection.executescript(""" + CREATE TABLE IF NOT EXISTS organizations (id TEXT PRIMARY KEY); + CREATE TABLE IF NOT EXISTS projects ( + tenant TEXT NOT NULL, id TEXT NOT NULL, rpm INTEGER NOT NULL, + key_rps INTEGER NOT NULL, PRIMARY KEY(tenant,id), + FOREIGN KEY(tenant) REFERENCES organizations(id)); + CREATE TABLE IF NOT EXISTS api_keys ( + id TEXT PRIMARY KEY, salt TEXT NOT NULL, digest TEXT NOT NULL, + tenant TEXT NOT NULL, project TEXT NOT NULL, scopes TEXT NOT NULL, + created_at REAL NOT NULL, expires_at REAL, last_used_at REAL, + revoked INTEGER NOT NULL DEFAULT 0, + FOREIGN KEY(tenant,project) REFERENCES projects(tenant,id)); + CREATE TABLE IF NOT EXISTS quota ( + kind TEXT NOT NULL, tenant TEXT NOT NULL, project TEXT NOT NULL, + key_id TEXT NOT NULL, window INTEGER NOT NULL, count INTEGER NOT NULL, + PRIMARY KEY(kind,tenant,project,key_id,window)); + CREATE TABLE IF NOT EXISTS audit_events ( + id INTEGER PRIMARY KEY, tenant TEXT NOT NULL, project TEXT NOT NULL, + key_id TEXT, action TEXT NOT NULL, request_id TEXT, at REAL NOT NULL); + CREATE TABLE IF NOT EXISTS usage_records ( + request_id TEXT PRIMARY KEY, tenant TEXT NOT NULL, project TEXT NOT NULL, + key_id TEXT NOT NULL, metadata TEXT NOT NULL); + """) + + @contextmanager + def connect(self) -> Generator[sqlite3.Connection]: + # No shared connection/thread state; busy waits are bounded. + connection = sqlite3.connect(self.path.as_uri() + "?mode=rw", uri=True, timeout=2) + try: + connection.execute("PRAGMA foreign_keys=ON") + connection.execute("PRAGMA synchronous=FULL") + with connection: + yield connection + finally: + connection.close() + + def create_project(self, tenant: str, project: str, *, rpm: int = 60, + key_rps: int = 10) -> None: + _identifier(tenant) + _identifier(project) + if not 1 <= rpm <= 100000 or not 1 <= key_rps <= 10000: + raise ValueError("bounded rate limits required") + with self.connect() as connection: + connection.execute("INSERT OR IGNORE INTO organizations VALUES (?)", (tenant,)) + connection.execute("INSERT INTO projects VALUES (?,?,?,?)", (tenant, project, rpm, key_rps)) + + def issue_key(self, tenant: str, project: str, scopes: set[str], *, live: bool = False, + expires_at: float | None = None) -> IssuedKey: + if not scopes or not scopes <= SCOPES: + raise ValueError("nonempty supported API scopes required") + if expires_at is not None and not math.isfinite(expires_at): + raise ValueError("finite expiration required") + key_id = secrets.token_hex(12) + plaintext = f"c3r_sk_{'live' if live else 'test'}_{key_id}_" + secrets.token_urlsafe(32) + salt = secrets.token_bytes(32) + digest = hmac.new(salt, plaintext.encode(), hashlib.sha256).hexdigest() + with self.connect() as connection: + connection.execute("INSERT INTO api_keys VALUES (?,?,?,?,?,?,?,?,?,0)", + (key_id, salt.hex(), digest, tenant, project, + json.dumps(sorted(scopes)), self.clock(), expires_at, None)) + connection.execute("INSERT INTO audit_events VALUES (NULL,?,?,?,'key_issued',NULL,?)", + (tenant, project, key_id, self.clock())) + return IssuedKey(key_id, plaintext) + + def authenticate(self, plaintext: str) -> Principal | None: + match = KEY.fullmatch(plaintext) + if match is None: + return None + with self.connect() as connection: + row = connection.execute("SELECT salt,digest,tenant,project,scopes,expires_at,revoked " + "FROM api_keys WHERE id=?", (match[2],)).fetchone() + if row is None: + return None + salt, digest, tenant, project, scope_json, expiration, revoked = row + actual = hmac.new(bytes.fromhex(salt), plaintext.encode(), hashlib.sha256).hexdigest() + if (not hmac.compare_digest(digest, actual) or revoked + or (expiration is not None and self.clock() >= expiration)): + return None + connection.execute("UPDATE api_keys SET last_used_at=? WHERE id=?", (self.clock(), match[2])) + return Principal(tenant, project, match[2], frozenset(json.loads(scope_json))) + + def revoke_key(self, tenant: str, project: str, key_id: str) -> None: + with self.connect() as connection: + connection.execute("UPDATE api_keys SET revoked=1 WHERE tenant=? AND project=? AND id=?", + (tenant, project, key_id)) + connection.execute("INSERT INTO audit_events VALUES (NULL,?,?,?,'key_revoked',NULL,?)", + (tenant, project, key_id, self.clock())) + + def list_keys(self, tenant: str, project: str) -> list[dict[str, object]]: + with self.connect() as connection: + connection.row_factory = sqlite3.Row + rows = connection.execute("SELECT id,tenant,project,scopes,created_at,expires_at," + "last_used_at,revoked FROM api_keys WHERE tenant=? AND project=?", + (tenant, project)).fetchall() + return [dict(row) for row in rows] + + def dispatch_event(self, principal: Principal, request_id: str) -> None: + with self.connect() as connection: + connection.execute("INSERT INTO audit_events VALUES (NULL,?,?,?,'request_dispatch',?,?)", + (principal.tenant_id, principal.project_id, principal.key_id, + request_id, self.clock())) + + def record_usage(self, principal: Principal, request_id: str, *, route: str, + model: str | None, status: int, latency_ms: float, + input_tokens: int | None, output_tokens: int | None, + system_one_invocations: int | None, + system_two_invocations: int | None) -> None: + metadata = {"request_id": request_id, "tenant_id": principal.tenant_id, + "project_id": principal.project_id, "api_key_id": principal.key_id, + "route": route, "model": model, "status": status, + "latency_ms": latency_ms, "timestamp": self.clock(), + "input_tokens": input_tokens, "output_tokens": output_tokens, + "system_one_invocations": system_one_invocations, + "system_two_invocations": system_two_invocations, + "gpu_allocation_ms": None, "allocated_cost_usd": None, + "cost_basis": "unmeasured"} + with self.connect() as connection: + connection.execute("INSERT INTO usage_records VALUES (?,?,?,?,?)", + (request_id, principal.tenant_id, principal.project_id, + principal.key_id, json.dumps(metadata, allow_nan=False))) + + def project_usage(self, tenant: str, project: str) -> list[dict[str, object]]: + with self.connect() as connection: + rows = connection.execute("SELECT metadata FROM usage_records WHERE tenant=? AND project=?", + (tenant, project)).fetchall() + return [cast(dict[str, object], json.loads(row[0])) for row in rows] + + def admit(self, principal: Principal) -> bool: + now = int(self.clock()) + with self.connect() as connection: + connection.execute("BEGIN IMMEDIATE") + limits = connection.execute("SELECT rpm,key_rps FROM projects WHERE tenant=? AND id=?", + (principal.tenant_id, principal.project_id)).fetchone() + if limits is None: + return False + requests = (("key", principal.key_id, now, limits[1]), + ("project", "", now // 60 * 60, limits[0])) + for kind, key_id, window, limit in requests: + row = connection.execute("SELECT count FROM quota WHERE kind=? AND tenant=? " + "AND project=? AND key_id=? AND window=?", + (kind, principal.tenant_id, principal.project_id, + key_id, window)).fetchone() + if row is not None and row[0] >= limit: + return False + for kind, key_id, window, _ in requests: + connection.execute("INSERT INTO quota VALUES (?,?,?,?,?,1) ON CONFLICT " + "(kind,tenant,project,key_id,window) DO UPDATE SET count=count+1", + (kind, principal.tenant_id, principal.project_id, key_id, window)) + connection.execute("DELETE FROM quota WHERE window Catalog: + operations = ( + ("STOP", ActionFamily.STOP), ("WAIT", ActionFamily.STOP), + ("VERIFY", ActionFamily.VERIFY), ("RETRIEVE", ActionFamily.RETRIEVAL), + ("SYSTEM_ONE", ActionFamily.LOCAL_MODEL), ("DELIBERATE", ActionFamily.DELIBERATE), + ("CALL_LOCAL_MODEL", ActionFamily.LOCAL_MODEL), + ("CALL_FRONTIER_MODEL", ActionFamily.FRONTIER_MODEL), ("ASK_USER", ActionFamily.ASK_USER), + ) + definitions = tuple(ActionDefinition( + id=operation, family=family, subgroup="compute", operation=operation, + risk_class=RiskClass.READ_ONLY, argument_variants=((),), placements=("local",), + verifier_ids=("recommendation",), optimistic_utility=0, estimated_cost=0, + ) for operation, family in operations) + return Catalog(name, definitions, AuthorityPolicy( + frozenset(family for _, family in operations), frozenset({RiskClass.READ_ONLY}), + allowed_data_boundaries=frozenset({"local"}))) diff --git a/c3r/catalogs/registry.py b/c3r/catalogs/registry.py new file mode 100644 index 0000000..38e0075 --- /dev/null +++ b/c3r/catalogs/registry.py @@ -0,0 +1,36 @@ +"""Admission of caller task text to registered, host-owned catalogs.""" +import json +from collections.abc import Mapping + +from ..host_factory import ReadOnlyRequestFactory +from ..runtime import RuntimeRequest +from .base import Catalog + + +class CatalogRegistry: + def __init__(self, catalogs: tuple[Catalog, ...]) -> None: + self._factories = {catalog.name: ReadOnlyRequestFactory( + definitions=catalog.definitions, policy=catalog.policy, + # Unpriced/unknown quality has no positive CVoC estimate. No invented + # dollar cost or success probability is admitted as measured evidence. + estimate_source=lambda _state: {}, remaining_usd=0, + model_inventory=("c3r-system-one", "deepseek"), data_boundary="local", + ) for catalog in catalogs} + + def build(self, payload: Mapping[str, object]) -> RuntimeRequest: + if set(payload) - {"goal", "state", "current_subgoal", "open_questions", + "catalog", "application_id"}: + raise ValueError("caller authority or unsupported fields rejected") + catalog = payload.get("catalog", "agent-v1") + if not isinstance(catalog, str) or catalog not in self._factories: + raise ValueError("unregistered catalog") + application = payload.get("application_id") + if application is not None and (not isinstance(application, str) or len(application) > 128): + raise ValueError("invalid application identifier") + state = payload.get("state", payload.get("current_subgoal", payload.get("goal"))) + if isinstance(state, dict): + state = json.dumps(state, ensure_ascii=False, allow_nan=False) + return self._factories[catalog].build({ + "goal": payload.get("goal"), "current_subgoal": state, + "open_questions": payload.get("open_questions", []), + }) diff --git a/c3r/catalogs/tools.py b/c3r/catalogs/tools.py new file mode 100644 index 0000000..3319e8f --- /dev/null +++ b/c3r/catalogs/tools.py @@ -0,0 +1,6 @@ +"""Routing recommendations only; no arbitrary tool or URL invocation.""" +from .general import general_catalog + + +def tool_catalog(): + return general_catalog("tool-routing-v1") diff --git a/c3r/cvoc.py b/c3r/cvoc.py index 97c96d8..a1dbafb 100644 --- a/c3r/cvoc.py +++ b/c3r/cvoc.py @@ -3,6 +3,7 @@ from __future__ import annotations from collections.abc import Iterable, Mapping +from math import isfinite from .state_schema import ActionCandidate, ActionFamily, CvocDecision, ValueEstimate @@ -14,6 +15,16 @@ def __init__(self, uncertainty_multiplier: float = 1.96) -> None: self._uncertainty_multiplier = uncertainty_multiplier def lower_bound(self, estimate: ValueEstimate) -> float: + values = ( + estimate.expected_gain, + estimate.total_cost, + estimate.risk_penalty, + estimate.uncertainty, + ) + if not all(isfinite(value) for value in values): + return float("-inf") + if min(estimate.total_cost, estimate.risk_penalty, estimate.uncertainty) < 0: + return float("-inf") return ( estimate.expected_gain - estimate.total_cost diff --git a/c3r/decisionmix/dataset.py b/c3r/decisionmix/dataset.py index fdc1f36..bbf6fd5 100644 --- a/c3r/decisionmix/dataset.py +++ b/c3r/decisionmix/dataset.py @@ -86,11 +86,13 @@ def __init__(self, *, split_seed: str) -> None: def add(self, record: DecisionMixRecord) -> str: record.validate() + # Upstream benchmark test examples belong in a separate, sealed evaluation + # harness; re-hashing their IDs must never turn them into DecisionMix rows. + if record.provenance.source_partition == "benchmark_test": + raise ValueError("benchmark test records cannot enter DecisionMix") assigned = deterministic_split(record.record_id, seed=self._split_seed) if record.record_id in self._record_ids: raise ValueError(f"duplicate record_id: {record.record_id}") - if record.provenance.source_partition == "benchmark_test" and assigned == "train": - raise ValueError("benchmark test answers cannot enter the training split") self._record_ids.add(record.record_id) self._records[assigned].append(record) return assigned diff --git a/c3r/deliberative/__init__.py b/c3r/deliberative/__init__.py index e494e0c..d3d9a1c 100644 --- a/c3r/deliberative/__init__.py +++ b/c3r/deliberative/__init__.py @@ -1,13 +1,21 @@ """Interfaces for open and frontier deliberative models.""" +from importlib import import_module + from .envelope import DeliberativeEnvelope, DeliberativeResult -from .defaults import ( - DEFAULT_DEEPSEEK_API_MODEL, - DEFAULT_LANGUAGE_MODEL, - DEFAULT_OPENROUTER_MODEL, - DefaultModelProfile, - default_provider_config, -) + + +def __getattr__(name: str): + if name in { + "DEFAULT_DEEPSEEK_API_MODEL", + "DEFAULT_LANGUAGE_MODEL", + "DEFAULT_OPENROUTER_MODEL", + "DefaultModelProfile", + "default_provider_config", + }: + defaults = import_module(".defaults", __name__) + return getattr(defaults, name) + raise AttributeError(name) __all__ = [ "DEFAULT_DEEPSEEK_API_MODEL", diff --git a/c3r/deliberative/defaults.py b/c3r/deliberative/defaults.py index 3a49da7..2f499a8 100644 --- a/c3r/deliberative/defaults.py +++ b/c3r/deliberative/defaults.py @@ -1,6 +1,6 @@ """Auditable defaults for C3R's deliberative language-model path. -The Laya System-One decision model remains separate from this language-model +The CLM System-One decision model remains separate from this language-model default. Defaults select a provider profile; they never bypass candidate, verification, budget, or commit controls. """ @@ -26,9 +26,15 @@ class DefaultModelProfile: def default_provider_config( - *, api_key: str, gateway: Literal["openrouter", "deepseek"] = "openrouter" + *, api_key: str | None = None, + gateway: Literal["self_hosted", "openrouter", "deepseek"] = "self_hosted" ) -> ProviderConfig: - """Return the explicit hosted default; credentials are caller-owned.""" + """Default to local DeepSeek. Explicit remote profiles are legacy compatibility only.""" + if gateway == "self_hosted": + return ProviderConfig("deepseek-local", ProviderKind.OPENAI_COMPATIBLE, + "http://127.0.0.1:8000/v1", "/model", None) + if gateway not in {"openrouter", "deepseek"}: + raise ValueError("unknown gateway") if not api_key: raise ValueError("api_key is required") if gateway == "deepseek": diff --git a/c3r/deliberative/provider_bridge.py b/c3r/deliberative/provider_bridge.py new file mode 100644 index 0000000..d7aec8a --- /dev/null +++ b/c3r/deliberative/provider_bridge.py @@ -0,0 +1,23 @@ +"""Data-boundary-aware bridge from compiled C3R state to provider adapters.""" + +from __future__ import annotations + +from dataclasses import asdict +from urllib.parse import urlparse + +from ..adapters.providers import DeliberationRequest, ProviderAdapter, ProviderExecutionResult +from ..state_schema import CompiledState + + +class ProviderDeliberator: + """Return a non-authoritative plan; never execute model-requested actions.""" + + def __init__(self, adapter: ProviderAdapter) -> None: + self._adapter = adapter + + def deliberate(self, state: CompiledState) -> ProviderExecutionResult: + hostname = urlparse(self._adapter.config.base_url).hostname + local = hostname in {"localhost", "127.0.0.1", "::1"} + if not local and state.data_boundary not in {"approved_remote", "public"}: + raise ValueError("state is not approved for a remote provider") + return self._adapter.deliberate(DeliberationRequest(asdict(state))) diff --git a/c3r/feature_flags.py b/c3r/feature_flags.py index c183f7c..54f3dd6 100644 --- a/c3r/feature_flags.py +++ b/c3r/feature_flags.py @@ -27,6 +27,7 @@ class FeatureFlags: speculation_requested: bool = False moe_control_requested: bool = False online_learning_requested: bool = False + system_one_provider: str = "clm" @classmethod def from_mapping(cls, values: Mapping[str, str]) -> FeatureFlags: @@ -38,7 +39,10 @@ def from_mapping(cls, values: Mapping[str, str]) -> FeatureFlags: speculation_requested=_boolean(values, "C3R_SPECULATION", False), moe_control_requested=_boolean(values, "C3R_MOE_CONTROL", False), online_learning_requested=_boolean(values, "C3R_ONLINE_LEARNING", False), + system_one_provider=values.get("C3R_SYSTEM_ONE_PROVIDER", "clm").strip().lower(), ) + if flags.system_one_provider not in {"clm", "laya", "jev", "rule"}: + raise ValueError("C3R_SYSTEM_ONE_PROVIDER must be clm, laya, jev, or rule") if flags.online_learning_requested: raise ValueError("autonomous online learning is prohibited") return flags diff --git a/c3r/host_components.py b/c3r/host_components.py new file mode 100644 index 0000000..dd3c9f3 --- /dev/null +++ b/c3r/host_components.py @@ -0,0 +1,17 @@ +"""Host composition shared by imported and module-executed entrypoints.""" +from dataclasses import dataclass + +from .http_service import RequestFactory +from .internal_readiness import ReadinessProbe +from .responses import ResponsesService +from .runtime import StandaloneController +from .system_one.inference import SystemOneInference + + +@dataclass(frozen=True, slots=True) +class HostComponents: + runtime: StandaloneController + factory: RequestFactory + system_one: SystemOneInference + responses: ResponsesService + internal_readiness: ReadinessProbe | None = None diff --git a/c3r/host_factory.py b/c3r/host_factory.py new file mode 100644 index 0000000..eccf63f --- /dev/null +++ b/c3r/host_factory.py @@ -0,0 +1,81 @@ +"""Read-only host construction of bounded standalone decision requests.""" + +from __future__ import annotations + +from collections.abc import Callable, Mapping +from math import isfinite +from uuid import uuid4 + +from .runtime import RuntimeRequest +from .state_schema import ActionDefinition, AuthorityPolicy, RawState, RiskClass, ValueEstimate + + +EstimateSource = Callable[[RawState], Mapping[str, ValueEstimate]] + + +class ReadOnlyRequestFactory: + """Accept task text only; all authority and estimates come from the host. + + The estimate source is mandatory because constant/example estimates must not be + mistaken for measured production CVoC inputs. + """ + + def __init__( + self, + *, + definitions: tuple[ActionDefinition, ...], + policy: AuthorityPolicy, + estimate_source: EstimateSource, + remaining_usd: float, + model_inventory: tuple[str, ...] = (), + data_boundary: str = "local", + ) -> None: + if not definitions or any(item.risk_class is not RiskClass.READ_ONLY for item in definitions): + raise ValueError("HTTP catalog must contain read-only definitions only") + if policy.allowed_risks != frozenset({RiskClass.READ_ONLY}): + raise ValueError("HTTP policy must permit read-only risk only") + if not isfinite(remaining_usd) or remaining_usd < 0: + raise ValueError("remaining_usd must be finite and non-negative") + if data_boundary not in {"local", "approved_remote", "public"}: + raise ValueError("host data boundary must be explicit") + self._definitions = definitions + self._policy = policy + self._estimate_source = estimate_source + self._remaining_usd = remaining_usd + self._model_inventory = model_inventory + self._data_boundary = data_boundary + + def build(self, payload: Mapping[str, object]) -> RuntimeRequest: + goal = _text(payload.get("goal"), "goal") + current_subgoal = _text(payload.get("current_subgoal"), "current_subgoal") + questions = payload.get("open_questions", []) + if not isinstance(questions, list) or len(questions) > 64: + raise ValueError("open_questions must be a bounded array") + open_questions = tuple(_text(item, "open_question") for item in questions) + raw = RawState( + goal=goal, + current_subgoal=current_subgoal, + open_questions=open_questions, + available_action_families=tuple( + sorted(self._policy.allowed_families, key=lambda family: family.value) + ), + model_inventory=self._model_inventory, + budget={"remaining_usd": self._remaining_usd}, + data_boundary=self._data_boundary, + ) + estimates = self._estimate_source(raw) + if not isinstance(estimates, Mapping): + raise ValueError("host estimate source returned an invalid mapping") + return RuntimeRequest( + raw_state=raw, + definitions=self._definitions, + policy=self._policy, + estimates=estimates, + run_id=str(uuid4()), + ) + + +def _text(value: object, name: str) -> str: + if not isinstance(value, str) or not value.strip() or len(value) > 4096: + raise ValueError(f"{name} must be non-empty bounded text") + return value diff --git a/c3r/http_service.py b/c3r/http_service.py new file mode 100644 index 0000000..8f4f405 --- /dev/null +++ b/c3r/http_service.py @@ -0,0 +1,397 @@ +"""Authenticated loopback HTTP boundary for recommendation-only C3R hosting. + +TLS termination and the request factory belong to the deployment host. This API +never accepts caller-supplied verification, approval, policy, or cost estimates. +""" + +from __future__ import annotations + +import hmac +import json +import select +import socket +import time +from collections.abc import Callable, Mapping +from dataclasses import asdict, is_dataclass +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from math import isfinite +from threading import BoundedSemaphore, Event, Lock, Thread +from typing import Protocol, cast + +from .deliberative.envelope import DeliberativeResult +from .internal_readiness import ReadinessProbe +from .responses import ResponseEventStream, ResponsesService +from .runtime import RuntimeRequest, StandaloneController +from .system_one.inference import SystemOneInference + +MAX_REQUEST_BYTES = 65_536 + + +class RequestFactory(Protocol): + """Construct trusted policies, catalogs, and measured estimates server-side.""" + + def build(self, payload: Mapping[str, object]) -> RuntimeRequest: ... + + +class TokenBucket: + def __init__( + self, + *, + capacity: int, + refill_per_second: float, + clock: Callable[[], float] = time.monotonic, + ) -> None: + if capacity < 1 or refill_per_second <= 0: + raise ValueError("rate limit must be positive") + self._capacity = capacity + self._refill = refill_per_second + self._clock = clock + self._tokens = float(capacity) + self._last = clock() + self._lock = Lock() + + def take(self) -> bool: + with self._lock: + now = self._clock() + self._tokens = min( + self._capacity, self._tokens + max(0.0, now - self._last) * self._refill + ) + self._last = now + if self._tokens < 1: + return False + self._tokens -= 1 + return True + + +class ServiceMetrics: + def __init__(self) -> None: + self._lock = Lock() + self._counts: dict[str, int] = {} + + def increment(self, name: str) -> None: + with self._lock: + self._counts[name] = self._counts.get(name, 0) + 1 + + def snapshot(self) -> dict[str, int]: + with self._lock: + return dict(self._counts) + + +class ClientDisconnected(Exception): + """A cancelled stream needs cleanup, not an attempted error response.""" + + +class C3RHTTPServer(ThreadingHTTPServer): + daemon_threads = True + + def __init__( + self, + *, + runtime: StandaloneController, + request_factory: RequestFactory, + bearer_token: str, + host: str = "127.0.0.1", + port: int = 8081, + requests_per_minute: int = 60, + system_one: SystemOneInference | None = None, + responses: ResponsesService | None = None, + internal_readiness: ReadinessProbe | None = None, + max_concurrent_generations: int = 4, + ) -> None: + if host not in {"127.0.0.1", "::1", "localhost"}: + raise ValueError("C3R must bind to loopback behind a TLS gateway") + if len(bearer_token) < 32: + raise ValueError("bearer token must have at least 32 characters") + if runtime.effect_execution_enabled: + raise ValueError("the HTTP service cannot execute external effects") + if max_concurrent_generations < 1: + raise ValueError("generation capacity must be positive") + self.runtime = runtime + self.system_one = system_one + self.responses = responses + self.generation_capacity = BoundedSemaphore(max_concurrent_generations) + self.internal_readiness = internal_readiness + self.request_factory = request_factory + self.bearer_token = bearer_token + self.limiter = TokenBucket( + capacity=requests_per_minute, + refill_per_second=requests_per_minute / 60, + ) + self.metrics = ServiceMetrics() + super().__init__((host, port), _Handler) + + +class _Handler(BaseHTTPRequestHandler): + server: C3RHTTPServer # pyright: ignore[reportIncompatibleVariableOverride] + + def log_message(self, format: str, *_args: object) -> None: + # The host exports aggregate metrics; request bodies and tokens are never logged. + return + + def _send(self, status: int, value: Mapping[str, object]) -> None: + body = json.dumps(value, sort_keys=True, separators=(",", ":")).encode("utf-8") + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Cache-Control", "no-store") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + def _authorized(self) -> bool: + expected = "Bearer " + self.server.bearer_token + supplied = self.headers.get_all("Authorization", []) + return len(supplied) == 1 and hmac.compare_digest(expected.encode(), supplied[0].encode()) + + def _open_stream(self, payload: Mapping[str, object]) -> ResponseEventStream: + """Cancel even while the generation backend has not sent headers yet.""" + assert self.server.responses is not None + cancelled, finished = Event(), Event() + + def watch_opening() -> None: + while not finished.wait(0.05): + try: + readable, _, _ = select.select([self.connection], [], [], 0) + if readable: + cancelled.set() + self.server.metrics.increment("stream_disconnect") + return + except (OSError, ValueError): + cancelled.set() + return + + watcher = Thread(target=watch_opening, daemon=True) + watcher.start() + try: + events = self.server.responses.stream(payload, cancelled=cancelled) + if cancelled.is_set(): + events.close() + raise ClientDisconnected + return events + except RuntimeError: + if cancelled.is_set(): + self.close_connection = True + raise ClientDisconnected from None + raise + finally: + finished.set() + watcher.join(timeout=0.2) + + def _send_stream(self, events: ResponseEventStream) -> None: + stop_watch = Event() + client_gone = Event() + released = False + release_lock = Lock() + watcher: Thread | None = None + + def record_disconnect() -> None: + with release_lock: + if not client_gone.is_set(): + client_gone.set() + self.server.metrics.increment("stream_disconnect") + + def release_capacity() -> None: + nonlocal released + with release_lock: + if not released: + released = True + self.server.generation_capacity.release() + + def watch_disconnect() -> None: + while not stop_watch.is_set(): + try: + readable, _, _ = select.select([self.connection], [], [], 0.1) + if readable: + # A stream is one request per connection; any further + # client bytes or a closed socket aborts that stream. + self.connection.recv(1, socket.MSG_PEEK) + record_disconnect() + events.abort() + release_capacity() + return + except OSError: + record_disconnect() + events.abort() + release_capacity() + return + + try: + self.send_response(200) + self.send_header("Content-Type", "text/event-stream") + self.send_header("Cache-Control", "no-store") + self.send_header("X-Accel-Buffering", "no") + self.send_header("Connection", "close") + self.end_headers() + self.close_connection = True + watcher = Thread(target=watch_disconnect, daemon=True) + watcher.start() + for name, data in events: + if client_gone.is_set(): + break + frame = ("event: " + name + "\n" + "data: " + + json.dumps(data, separators=(",", ":"), ensure_ascii=False) + + "\n\n").encode("utf-8") + self.wfile.write(frame) + self.wfile.flush() + except (BrokenPipeError, ConnectionResetError, OSError): + record_disconnect() + finally: + stop_watch.set() + events.close() + if watcher is not None: + watcher.join(timeout=0.5) + release_capacity() + + def do_GET(self) -> None: + if self.path == "/health": + self._send(200, {"status": "ok"}) + return + if self.path in {"/ready", "/v1/models"}: + if not self._authorized(): + self._send(401, {"error": "unauthorized"}) + return + if self.path == "/ready": + ready = (self.server.runtime.decision_enabled + and self.server.runtime.system_one_enabled + and self.server.runtime.provider_ready) + self._send(200 if ready else 503, { + "status": "ready" if ready else "disabled", + "scope": "configured_provider_probe_and_decision_policy", + }) + else: + available = (self.server.runtime.decision_enabled + and self.server.runtime.system_one_enabled + and self.server.runtime.provider_ready) + ranking_available = (self.server.runtime.decision_enabled + and self.server.runtime.system_one_enabled + and self.server.system_one is not None + and self.server.system_one.ready) + models = [ + {"id": "c3r-core", "capability": "verified_recommendation", + "text_generation": self.server.responses is not None, "calibrated": False, + "effect_execution": False, "available": available}, + {"id": "c3r-system-one", "capability": "advisory_ranking", + "text_generation": False, "calibrated": False, + "effect_execution": False, + "available": ranking_available}, + {"id": "c3r-verifier", "capability": "advisory_output_ranking", + "text_generation": False, "calibrated": False, "effect_execution": False, + "available": ranking_available}, + ] + self._send(200, {"object": "list", "data": models, "models": models}) + return + if self.path == "/metrics": + if not self._authorized(): + self._send(401, {"error": "unauthorized"}) + return + self._send(200, {"counts": self.server.metrics.snapshot()}) + return + self._send(404, {"error": "not_found"}) + + def do_POST(self) -> None: + if self.path not in { + "/v1/decisions", "/v1/c3r/decide", "/v1/c3r/rank", + "/v1/system-one", "/v1/c3r/execute", "/v1/responses", + }: + self._send(404, {"error": "not_found"}) + return + if not self._authorized(): + self.server.metrics.increment("unauthorized") + self._send(401, {"error": "unauthorized"}) + return + if not self.server.limiter.take(): + self.server.metrics.increment("rate_limited") + self._send(429, {"error": "rate_limited"}) + return + if self.path == "/v1/c3r/execute" or (self.path == "/v1/responses" and + self.server.responses is None): + self._send(501, {"error": "not_implemented", "reason": "recommendation_only"}) + return + try: + length = int(self.headers.get("Content-Length", "")) + except ValueError: + self._send(400, {"error": "invalid_content_length"}) + return + if length < 1 or length > MAX_REQUEST_BYTES: + self._send(413, {"error": "request_size_out_of_bounds"}) + return + try: + payload = json.loads(self.rfile.read(length)) + if not isinstance(payload, dict): + raise TypeError("JSON object required") + payload = cast(dict[str, object], payload) + if self.path == "/v1/responses" and self.server.responses: + if not self.server.generation_capacity.acquire(blocking=False): + self._send(429, {"error": "capacity_exceeded"}) + return + if payload.get("stream") is True: + try: + events = self._open_stream(payload) + except BaseException: + self.server.generation_capacity.release() + raise + self.server.metrics.increment("text_responses") + self._send_stream(events) + return + try: + response = self.server.responses.respond(payload) + finally: + self.server.generation_capacity.release() + self.server.metrics.increment("text_responses") + self._send(200, response) + return + if self.path in {"/v1/c3r/rank", "/v1/system-one"} and self.server.system_one: + if not (self.server.runtime.decision_enabled + and self.server.runtime.system_one_enabled): + raise RuntimeError("System-One disabled") + result = self.server.system_one.infer(payload) + self.server.metrics.increment("system_one_inferences") + self._send(200, result) + return + request = self.server.request_factory.build(payload) + outcome = self.server.runtime.run(request) + except ClientDisconnected: + self.close_connection = True + return + except (KeyError, TypeError, ValueError, RecursionError, json.JSONDecodeError): + self.server.metrics.increment("invalid_request") + self._send(400, {"error": "invalid_request"}) + return + except (OSError, RuntimeError): + self.server.metrics.increment("internal_failure") + self._send(503, {"error": "service_unavailable"}) + return + self.server.metrics.increment("decisions") + result: dict[str, object] = { + "route": outcome.route, + "selected_action_id": outcome.selected_action_id, + "authority_result": outcome.authority_result, + "reason": outcome.reason, + "trace_hash": outcome.ledger_record.record_hash, + "effect_executed": False, + "cvoc": {"selected_lower_bound": outcome.cvoc_lower_bound, + "basis": "host_supplied_estimates"}, + } + if self.path in {"/v1/c3r/rank", "/v1/system-one"}: + result["abstained"] = outcome.fast_path is None or outcome.fast_path.abstained + result["model_id"] = ( + None if outcome.fast_path is None else outcome.fast_path.model_id + ) + result["scope"] = "controller_decision_with_system_one_fallback" + scores = (() if outcome.fast_path is None + else outcome.fast_path.candidate_probabilities) + if (len(scores) != len(outcome.candidate_ids) + or any(not isfinite(score) or score < 0 or score > 1 for score in scores)): + scores = () + result["candidate_ranking"] = [ + {"candidate_id": candidate_id, "system_one_score": score, + "calibrated": False} + for candidate_id, score in sorted( + zip(outcome.candidate_ids, scores), + key=lambda item: item[1], reverse=True, + ) + ] + if isinstance(outcome.deliberation, DeliberativeResult) and is_dataclass( + outcome.deliberation + ): + result["deliberation"] = asdict(outcome.deliberation) + self._send(200, result) diff --git a/c3r/http_transport.py b/c3r/http_transport.py new file mode 100644 index 0000000..a894ff9 --- /dev/null +++ b/c3r/http_transport.py @@ -0,0 +1,8 @@ +"""Redirects must not move local state or authorization to another origin.""" +from urllib.request import HTTPRedirectHandler, Request + + +class NoRedirectHandler(HTTPRedirectHandler): + def redirect_request(self, req: Request, fp: object, code: int, + msg: str, headers: object, newurl: str) -> None: + return None diff --git a/c3r/ingress_proxy.py b/c3r/ingress_proxy.py new file mode 100644 index 0000000..fc84ba6 --- /dev/null +++ b/c3r/ingress_proxy.py @@ -0,0 +1,381 @@ +"""Bounded ingress for an IAM/TLS-terminated host with a loopback C3R backend. + +The outer host still owns TLS, IAM, secrets, and abuse controls. This proxy never +trusts caller authorization as backend authorization and never logs request data. +""" + +from __future__ import annotations + +import hmac +import json +import select +import socket +import sqlite3 +import time +from collections.abc import Mapping +from http.client import HTTPConnection, HTTPException, HTTPResponse +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from threading import BoundedSemaphore, Event, Thread +from typing import cast +from uuid import uuid4 + +from .api_access import AccessStore, Principal +from .http_service import MAX_REQUEST_BYTES, TokenBucket + +MAX_RESPONSE_BYTES = 65_536 + + +class C3RIngressServer(ThreadingHTTPServer): + """Expose only the read-only API, with separate client and loopback secrets.""" + + daemon_threads = True + + def __init__( + self, + *, + upstream_port: int, + client_token: str, + upstream_token: str, + upstream_host: str = "127.0.0.1", + host: str = "0.0.0.0", + port: int = 8080, + requests_per_minute: int = 60, + max_in_flight: int = 16, + upstream_timeout_seconds: float = 5.0, + access_store: AccessStore | None = None, + ) -> None: + if upstream_host not in {"127.0.0.1", "::1"}: + raise ValueError("upstream must be loopback") + if not 1 <= upstream_port <= 65535: + raise ValueError("invalid upstream port") + if min(len(client_token), len(upstream_token)) < 32: + raise ValueError("tokens must have at least 32 characters") + if hmac.compare_digest(client_token, upstream_token): + raise ValueError("client and upstream tokens must be different") + if max_in_flight < 1 or upstream_timeout_seconds <= 0: + raise ValueError("concurrency and timeout must be positive") + self.upstream_host = upstream_host + self.upstream_port = upstream_port + self.client_token = client_token + self.access_store = access_store + self.upstream_token = upstream_token + self.upstream_timeout_seconds = upstream_timeout_seconds + self.limiter = TokenBucket( + capacity=requests_per_minute, + refill_per_second=requests_per_minute / 60, + ) + self.in_flight = BoundedSemaphore(max_in_flight) + self.route_slots = {"/v1/responses": BoundedSemaphore(1), + "/v1/system-one": BoundedSemaphore(4), + "/v1/c3r/rank": BoundedSemaphore(4)} + decisions = BoundedSemaphore(2) + for path in ("/v1/decisions", "/v1/c3r/decide", "/v1/c3r/execute"): + self.route_slots[path] = decisions + super().__init__((host, port), _IngressHandler) + + def get_request(self) -> tuple[socket.socket, object]: + connection, address = super().get_request() + connection.settimeout(5.0) + return connection, address + + +class _IngressHandler(BaseHTTPRequestHandler): + @property + def gateway(self) -> C3RIngressServer: + return cast(C3RIngressServer, self.server) + principal: Principal | None = None + request_id: str = "" + started: float = 0 + + def log_message(self, format: str, *args: object) -> None: + return + + def send_error(self, code: int, message: str | None = None, explain: str | None = None) -> None: + self._begin_request() + self.close_connection = True + self._send_error(405 if code == 501 else code, + "method_not_allowed" if code == 501 else "invalid_request") + + def _send_error(self, status: int, code: str) -> None: + value: dict[str, object] = {"error": code} + if self.gateway.access_store is not None: + category = {400: "invalid_request_error", 401: "authentication_error", + 403: "permission_error", 404: "not_found_error", 409: "conflict_error", + 413: "invalid_request_error", 415: "invalid_request_error", + 429: "rate_limit_error", 503: "service_unavailable"}.get(status, "api_error") + value = {"error": {"message": "C3R request could not be completed.", + "type": category, "code": code, "param": None, + "request_id": self.request_id}} + body = json.dumps(value, separators=(",", ":")).encode("utf-8") + self.send_response(status) + self.send_header("x-request-id", self.request_id) + self.send_header("Content-Type", "application/json") + self.send_header("Cache-Control", "no-store") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + if getattr(self, "command", "") != "HEAD": + self.wfile.write(body) + + def _begin_request(self) -> None: + self.request_id = "req_" + uuid4().hex + self.principal = None + self.started = time.monotonic() + + def _scope_allowed(self) -> bool: + if self.gateway.access_store is None: + return True + scope = {"/v1/models": "models:read", "/v1/responses": "responses:write", + "/v1/system-one": "system_one:write", "/v1/c3r/rank": "rank:write", + "/v1/c3r/decide": "decide:write", "/v1/decisions": "decide:write", + "/v1/c3r/execute": "decide:write", "/ready": "models:read"}.get(self.path) + return self.principal is not None and scope in self.principal.scopes + + def _admitted(self) -> bool: + if self.gateway.access_store is None: + return self.gateway.limiter.take() + return self.principal is not None and self.gateway.access_store.admit(self.principal) + + def _record_usage(self, status: int, decoded: Mapping[str, object]) -> None: + if self.gateway.access_store is None or self.principal is None: + return + raw_usage = decoded.get("usage") + raw_c3r = decoded.get("c3r") + usage: Mapping[str, object] = cast(Mapping[str, object], raw_usage) if isinstance(raw_usage, dict) else {} + c3r: Mapping[str, object] = cast(Mapping[str, object], raw_c3r) if isinstance(raw_c3r, dict) else {} + + def count(value: object) -> int | None: + return value if type(value) is int and value >= 0 else None + + model = decoded.get("model") + self.gateway.access_store.record_usage( + self.principal, self.request_id, route=self.path, + model=model if isinstance(model, str) and model in {"c3r-core", "c3r-system-one", "c3r-verifier"} else None, + status=status, latency_ms=(time.monotonic() - self.started) * 1000, + input_tokens=count(usage.get("input_tokens")), output_tokens=count(usage.get("output_tokens")), + system_one_invocations=count(c3r.get("system_one_invocations")), + system_two_invocations=count(c3r.get("system_two_invocations")), + ) + + def _authorized(self) -> bool: + # Tenant and project identity come only from the authenticated key. + # Reject caller context overrides rather than silently ignoring them. + if any(name in self.headers for name in ( + "X-Tenant-ID", "X-Project-ID", "OpenAI-Organization", "OpenAI-Project", + "X-C3R-Organization", "X-C3R-Project", + )): + return False + supplied = self.headers.get_all("X-C3R-Token", []) + authorization = self.headers.get_all("Authorization", []) + if len(authorization) > 1 or len(supplied) > 1: + return False + if self.gateway.access_store is not None: + if supplied or len(authorization) != 1 or not authorization[0].startswith("Bearer "): + return False + principal = self.gateway.access_store.authenticate(authorization[0][7:]) + if principal is None: + return False + self.principal = principal + return True + # Preserve IAM staging's separate client header. Without that header, + # SDK clients use standard Bearer auth; it is never relayed upstream. + if supplied: + return hmac.compare_digest(self.gateway.client_token.encode(), supplied[0].encode()) + return (len(authorization) == 1 and + hmac.compare_digest(("Bearer " + self.gateway.client_token).encode(), + authorization[0].encode())) + + def _forward(self, method: str, body: bytes | None = None) -> None: + route_slot = self.gateway.route_slots.get(self.path) + if route_slot is not None and not route_slot.acquire(blocking=False): + self._send_error(429 if self.gateway.access_store else 503, "over_capacity") + return + if not self.gateway.in_flight.acquire(blocking=False): + if route_slot is not None: + route_slot.release() + self._send_error(429 if self.gateway.access_store else 503, "over_capacity") + return + connection = HTTPConnection( + self.gateway.upstream_host, + self.gateway.upstream_port, + timeout=self.gateway.upstream_timeout_seconds, + ) + stopped = Event() + disconnected = Event() + watcher: Thread | None = None + try: + if self.gateway.access_store is not None and self.principal is not None: + self.gateway.access_store.dispatch_event(self.principal, self.request_id) + headers = {"Authorization": "Bearer " + self.gateway.upstream_token} + headers["x-request-id"] = self.request_id + if body is not None: + headers["Content-Type"] = "application/json" + connection.connect() + transport = connection.sock + if transport is None: + raise OSError("upstream transport missing") + + def watch_disconnect() -> None: + while not stopped.wait(0.05): + try: + readable, _, _ = select.select([self.connection], [], [], 0) + if readable and self.connection.recv(1, socket.MSG_PEEK) == b"": + disconnected.set() + transport.shutdown(socket.SHUT_RDWR) + return + except (OSError, ValueError): + return + + watcher = Thread(target=watch_disconnect, daemon=True) + watcher.start() + connection.request(method, self.path, body=body, headers=headers) + response = connection.getresponse() + if response.status == 200 and response.getheader("Content-Type", "").split(";", 1)[0] == "text/event-stream": + if method != "POST" or self.path != "/v1/responses" or transport is None: + raise ValueError("unexpected stream") + self._forward_stream(response, disconnected) + return + forwarded = response.read(MAX_RESPONSE_BYTES + 1) + if len(forwarded) > MAX_RESPONSE_BYTES: + self._send_error(502, "invalid_upstream_response") + return + decoded = json.loads(forwarded) + if not isinstance(decoded, dict) or not 200 <= response.status <= 599: + raise ValueError("invalid upstream response") + self._record_usage(response.status, cast(Mapping[str, object], decoded)) + if response.status >= 400 and self.gateway.access_store is not None: + self._send_error(response.status, "upstream_rejected") + return + except (OSError, HTTPException, ValueError, sqlite3.Error): + if not disconnected.is_set(): + self._send_error(503, "upstream_unavailable") + return + finally: + stopped.set() + if watcher is not None: + watcher.join(timeout=1) + connection.close() + self.gateway.in_flight.release() + if route_slot is not None: + route_slot.release() + self.send_response(response.status) + self.send_header("x-request-id", self.request_id) + self.send_header("Content-Type", "application/json") + self.send_header("Cache-Control", "no-store") + self.send_header("Content-Length", str(len(forwarded))) + self.end_headers() + self.wfile.write(forwarded) + + def _forward_stream(self, response: HTTPResponse, disconnected: Event) -> None: + self.send_response(200) + self.send_header("Content-Type", "text/event-stream") + self.send_header("Cache-Control", "no-store") + self.send_header("X-Accel-Buffering", "no") + self.send_header("x-request-id", self.request_id) + self.end_headers() + terminal = False + total = 0 + deadline = time.monotonic() + 300 + try: + frame = bytearray() + while time.monotonic() < deadline: + line = response.readline(MAX_RESPONSE_BYTES + 1) + if not line: + break + frame.extend(line) + total += len(line) + if len(frame) > MAX_RESPONSE_BYTES or total > 2_097_152: + raise ValueError("stream bound exceeded") + if line not in (b"\n", b"\r\n"): + continue + data = b"\n".join(part[5:].strip() for part in bytes(frame).splitlines() + if part.startswith(b"data:")) + raw_event = json.loads(data) + if not isinstance(raw_event, dict): + raise TypeError("invalid stream event") + event = cast(dict[str, object], raw_event) + event_type = event.get("type") + if not isinstance(event_type, str): + raise TypeError("invalid stream event type") + if event_type in {"response.completed", "response.failed"}: + payload = event.get("response") + self._record_usage(200 if event_type == "response.completed" else 503, + cast(Mapping[str, object], payload) if isinstance(payload, dict) else {}) + terminal = True + self.wfile.write(frame) + self.wfile.flush() + frame.clear() + if terminal: + return + raise ValueError("incomplete stream") + except (OSError, HTTPException, TypeError, ValueError, sqlite3.Error): + if not terminal: + try: + self._record_usage(499 if disconnected.is_set() else 503, {}) + failure = {"type": "response.failed", "response": {"status": "failed", + "error": {"code": "stream_interrupted", "message": "C3R stream interrupted."}}} + self.wfile.write(b"event: response.failed\ndata: " + json.dumps(failure).encode() + b"\n\n") + self.wfile.flush() + except (OSError, sqlite3.Error): + pass + + def do_GET(self) -> None: + try: + self._get() + except (sqlite3.Error, OSError): + self._send_error(503, "access_unavailable") + + def _get(self) -> None: + self._begin_request() + if self.path not in {"/health", "/metrics", "/ready", "/v1/models"}: + self._send_error(404, "not_found") + return + if self.path != "/health" and not self._authorized(): + self._send_error(401, "unauthorized") + return + if self.path != "/health" and not self._scope_allowed(): + self._send_error(403, "insufficient_scope") + return + if self.path != "/health" and self.gateway.access_store is not None and not self._admitted(): + self._send_error(429, "rate_limited") + return + self._forward("GET") + + def do_POST(self) -> None: + try: + self._post() + except (sqlite3.Error, OSError): + self._send_error(503, "access_unavailable") + + def _post(self) -> None: + self._begin_request() + if self.path not in { + "/v1/decisions", "/v1/c3r/decide", "/v1/c3r/rank", + "/v1/system-one", "/v1/c3r/execute", "/v1/responses", + }: + self._send_error(404, "not_found") + return + if not self._authorized(): + self._send_error(401, "unauthorized") + return + if not self._scope_allowed(): + self._send_error(403, "insufficient_scope") + return + if self.headers.get("Transfer-Encoding") is not None: + self._send_error(400, "unsupported_transfer_encoding") + return + if self.headers.get("Content-Type", "").split(";", 1)[0].strip().lower() != "application/json": + self._send_error(415, "unsupported_media_type") + return + if not self._admitted(): + self._send_error(429, "rate_limited") + return + lengths = self.headers.get_all("Content-Length", []) + try: + length = int(lengths[0]) if len(lengths) == 1 else -1 + except ValueError: + length = -1 + if length < 1 or length > MAX_REQUEST_BYTES: + self._send_error(413, "request_size_out_of_bounds") + return + self._forward("POST", self.rfile.read(length)) + diff --git a/c3r/internal_readiness.py b/c3r/internal_readiness.py new file mode 100644 index 0000000..b9d38cd --- /dev/null +++ b/c3r/internal_readiness.py @@ -0,0 +1,85 @@ +"""A separate authenticated loopback maintenance channel, never an API proxy.""" +import hmac +import json +import re +from collections.abc import Callable, Mapping +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from threading import Event, Thread +from typing import cast + +ReadinessProbe = Callable[[], Mapping[str, object]] +CHECKS = ("runtime", "clm_qwen", "deepseek", "required_local_artifact_files") + + +class InternalReadinessServer(ThreadingHTTPServer): + daemon_threads = True + + def __init__(self, *, token: str, port: int, + probe: ReadinessProbe | None = None, + api_workers: tuple[Thread, ...] = (), + stopping: Event | None = None) -> None: + if len(token) < 32: + raise ValueError("internal readiness token must have at least 32 characters") + self.token, self.probe = token, probe + self.api_workers = api_workers + self.stopping = stopping if stopping is not None else Event() + super().__init__(("127.0.0.1", port), _Handler) + + @property + def api_workers_alive(self) -> bool: + return (not self.stopping.is_set() and len(self.api_workers) == 2 + and self.api_workers[0] is not self.api_workers[1] + and all(worker.is_alive() for worker in self.api_workers)) + + def shutdown(self) -> None: + self.stopping.set() + super().shutdown() + + +class _Handler(BaseHTTPRequestHandler): + @property + def readiness_server(self) -> InternalReadinessServer: + return cast(InternalReadinessServer, self.server) + + def log_message(self, format: str, *args: object) -> None: + return + + def _send(self, status: int, value: Mapping[str, object]) -> None: + body = json.dumps(value, sort_keys=True, separators=(",", ":")).encode() + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Cache-Control", "no-store") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + def do_GET(self) -> None: + if self.path != "/internal/ready": + self._send(404, {"error": "not_found"}) + return + supplied = self.headers.get_all("Authorization", []) + if (len(supplied) != 1 or not hmac.compare_digest( + ("Bearer " + self.readiness_server.token).encode(), supplied[0].encode())): + self._send(401, {"error": "unauthorized"}) + return + raw_report: object = None + if self.readiness_server.api_workers_alive and self.readiness_server.probe is not None: + try: + raw_report = self.readiness_server.probe() + except (OSError, RuntimeError, TypeError, ValueError, KeyError): + pass + report: Mapping[str, object] = (raw_report + if isinstance(raw_report, Mapping) else {}) + checks = {name: report.get(name) is True for name in CHECKS} + checks["runtime"] = checks["runtime"] and self.readiness_server.api_workers_alive + pin = report.get("required_local_artifact_manifest_sha256") + valid_pin = isinstance(pin, str) and re.fullmatch(r"[a-f0-9]{64}", pin) is not None + ready = all(checks.values()) and valid_pin + self._send(200 if ready else 503, { + "status": "ready" if ready else "not_ready", "checks": checks, + "artifact_scope": "required_local_files_not_deepseek_weight_attestation", + "required_local_artifact_manifest_sha256": pin if valid_pin else None, + }) + + def do_POST(self) -> None: + self._send(405, {"error": "method_not_allowed"}) diff --git a/c3r/key_management.py b/c3r/key_management.py new file mode 100644 index 0000000..2c7fd07 --- /dev/null +++ b/c3r/key_management.py @@ -0,0 +1,59 @@ +"""Operator-local key administration. Only `issue` emits a key, once, to stdout. + +Use a secure terminal/secret-manager delivery channel; never redirect the issued +key into source control or ordinary logs. This CLI is not a public admin API. +""" +import argparse +import json +import sqlite3 +import time +from pathlib import Path + +from .api_access import AccessStore + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--database", type=Path, required=True) + commands = parser.add_subparsers(dest="command", required=True) + for name in ("project", "issue", "keys", "revoke", "usage"): + command = commands.add_parser(name) + command.add_argument("--tenant", required=True) + command.add_argument("--project", required=True) + if name == "project": + command.add_argument("--rpm", type=int, default=60) + command.add_argument("--key-rps", type=int, default=10) + elif name == "issue": + command.add_argument("--scope", action="append", required=True) + command.add_argument("--live", action="store_true") + command.add_argument("--ttl-seconds", type=int, default=86400) + elif name == "revoke": + command.add_argument("--key-id", required=True) + args = parser.parse_args() + try: + store = AccessStore(args.database) + if args.command == "project": + store.create_project(args.tenant, args.project, rpm=args.rpm, key_rps=args.key_rps) + result: object = {"status": "created"} + elif args.command == "issue": + if not 1 <= args.ttl_seconds <= 31536000: + raise ValueError("bounded key expiration required") + issued = store.issue_key(args.tenant, args.project, set(args.scope), live=args.live, + expires_at=time.time() + args.ttl_seconds) + result = {"api_key_id": issued.key_id, "api_key": issued.secret, "show_once": True} + elif args.command == "revoke": + store.revoke_key(args.tenant, args.project, args.key_id) + result = {"status": "revoked"} + elif args.command == "keys": + result = store.list_keys(args.tenant, args.project) + else: + result = store.project_usage(args.tenant, args.project) + print(json.dumps(result, sort_keys=True)) + return 0 + except (OSError, ValueError, sqlite3.Error): + print(json.dumps({"error": "key administration failed; check private database and inputs"})) + return 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/c3r/local_artifacts.py b/c3r/local_artifacts.py new file mode 100644 index 0000000..f54e2c0 --- /dev/null +++ b/c3r/local_artifacts.py @@ -0,0 +1,78 @@ +"""Fresh bounded local-file checks against an independently pinned host manifest. + +Hashing a receipt file proves only that file, never the weights described in it. +The release operator owns the required-file inventory and its out-of-band pin. +""" +import hashlib +import json +import os +import re +import stat +from pathlib import Path +from typing import cast + +MAX_TOTAL_BYTES = 64 * 1024 * 1024 +MAX_MANIFEST_BYTES = 65536 +SHA256 = re.compile(r"[a-f0-9]{64}") + + +def _read_regular(path: Path, maximum: int) -> bytes: + if not path.is_absolute() or any(p.is_symlink() for p in (path, *path.parents)): + raise ValueError("artifact must be an absolute non-linked local file") + descriptor = os.open(path, os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) + | getattr(os, "O_NONBLOCK", 0)) + with os.fdopen(descriptor, "rb") as handle: + before = os.fstat(handle.fileno()) + if not stat.S_ISREG(before.st_mode) or before.st_size > maximum: + raise ValueError("artifact is not a bounded regular file") + data = handle.read(maximum + 1) + after = os.fstat(handle.fileno()) + if (len(data) > maximum or before.st_size != after.st_size + or before.st_mtime_ns != after.st_mtime_ns + or before.st_ctime_ns != after.st_ctime_ns): + raise ValueError("artifact changed during verification") + return data + + +class PinnedLocalArtifacts: + def __init__(self, manifest: Path, expected_sha256: str) -> None: + if SHA256.fullmatch(expected_sha256) is None: + raise ValueError("independent required-file manifest SHA-256 pin required") + self.manifest, self.expected_sha256 = manifest, expected_sha256 + + def verify(self) -> bool: + try: + raw = _read_regular(self.manifest, MAX_MANIFEST_BYTES) + if hashlib.sha256(raw).hexdigest() != self.expected_sha256: + return False + parsed: object = json.loads(raw) + if not isinstance(parsed, dict): + return False + manifest = cast(dict[str, object], parsed) + entries = manifest.get("files") + if (set(manifest) != {"schema", "files"} + or manifest.get("schema") != "c3r-required-local-files-v1" + or not isinstance(entries, list)): + return False + rows = cast(list[object], entries) + if not 1 <= len(rows) <= 32: + return False + files: list[tuple[Path, str, int]] = [] + for raw_entry in rows: + if not isinstance(raw_entry, dict): + return False + entry = cast(dict[str, object], raw_entry) + path, sha, maximum = entry.get("path"), entry.get("sha256"), entry.get("max_bytes") + if (set(entry) != {"path", "sha256", "max_bytes"} + or not isinstance(path, str) or not isinstance(sha, str) + or SHA256.fullmatch(sha) is None or type(maximum) is not int + or not 1 <= maximum <= MAX_TOTAL_BYTES): + return False + files.append((Path(path), sha, maximum)) + if (len({str(path) for path, _, _ in files}) != len(files) + or sum(maximum for _, _, maximum in files) > MAX_TOTAL_BYTES): + return False + return all(hashlib.sha256(_read_regular(path, maximum)).hexdigest() == sha + for path, sha, maximum in files) + except (OSError, ValueError, TypeError, RecursionError): + return False diff --git a/c3r/production_host.py b/c3r/production_host.py new file mode 100644 index 0000000..013c5a9 --- /dev/null +++ b/c3r/production_host.py @@ -0,0 +1,161 @@ +"""Production composition for stateless, non-authoritative inference only.""" +from __future__ import annotations + +import json +import os +from collections.abc import Mapping +from pathlib import Path +from secrets import token_bytes +from typing import cast +from urllib.request import ProxyHandler, build_opener + +from .adapters.providers import ProviderAdapter, ProviderConfig, ProviderKind +from .candidate_compiler import CandidateCompiler +from .catalogs.browser import browser_catalog +from .catalogs.general import general_catalog +from .catalogs.registry import CatalogRegistry +from .catalogs.tools import tool_catalog +from .cvoc import RobustCvocController +from .feature_flags import FeatureFlags +from .host_components import HostComponents +from .http_transport import NoRedirectHandler +from .local_artifacts import PinnedLocalArtifacts +from .readiness import CachedReadiness +from .responses import ResponsesService +from .runtime import StandaloneController +from .state_compiler import StateCompiler +from .state_schema import RiskClass +from .system_one.advisory import AdvisoryFastPath +from .system_one.clm_adapter import UPSTREAM_CLM_COMMIT, ClmAdapter +from .system_one.inference import SystemOneInference +from .telemetry.ephemeral import EphemeralTraceSink +from .verifier_firewall import VerifierDecision, VerifierFirewall, VerifierPolicy + +ENCODER_REVISION = "b968826d9c46dd6066d109eabc6255188de91218" +HEAD_REVISION = "e939398d4556fcd9400c76fa8c5a513202f42b0a" +HEAD_SHA256 = "b2b4a8c9c2d39263eff78a351eb909a342ce9b3bf21a3f07c1d1bf15f1c4eda5" + + +class ProviderReadiness: + def __init__(self, adapter: ClmAdapter, container_digest: str) -> None: + self.adapter, self.container_digest = adapter, container_digest + self.system_one = CachedReadiness(self._system_one) + self.deliberative = CachedReadiness(self._deliberative) + + def _get(self, url: str) -> Mapping[str, object]: + with build_opener(ProxyHandler({}), NoRedirectHandler()).open(url, timeout=2) as response: + body = response.read(65537) + if len(body) > 65536: + raise ValueError("oversized provider readback") + value = json.loads(body) + if not isinstance(value, dict): + raise ValueError("invalid provider readback") + return cast(Mapping[str, object], value) + + def _system_one(self) -> bool: + try: + models = self._get("http://127.0.0.1:8090/v1/models") + rows = models.get("data") + if not isinstance(rows, list) or not any( + isinstance(row, dict) and cast(Mapping[str, object], row).get("id") == "qwen3-8b" + and cast(Mapping[str, object], row).get("root") == "/encoder" + for row in cast(list[object], rows) + ): + return False + health = self._get(self.adapter.endpoint + "/health") + if health.get("embedder") is not True or health.get("mock", False) is not False: + return False + artifact = self._get(self.adapter.endpoint + "/internal/clm/artifact") + expected = { + "clm_source_revision": UPSTREAM_CLM_COMMIT, "encoder": "Qwen/Qwen3-8B", + "encoder_revision": ENCODER_REVISION, "head_revision": HEAD_REVISION, + "head_sha256": HEAD_SHA256, "container_digest": self.container_digest, + "embedding_cache_size": 0, "action_cache_enabled": False, + "encoder_content_verified": True, + "encoder_identity_basis": "immutable_upstream_git_blobs_and_lfs_sha256", + } + if any(artifact.get(key) != value for key, value in expected.items()): + return False + scores = self.adapter.rank_text("An invoice was charged twice.", + "Which department handles billing?", + ("Billing", "Technical")) + return len(scores) == 2 + except (OSError, ValueError, TypeError, KeyError): + return False + + def _deliberative(self) -> bool: + try: + rows = self._get("http://127.0.0.1:8000/v1/models").get("data") + listed = isinstance(rows, list) and any( + isinstance(row, dict) and cast(Mapping[str, object], row).get("id") == "/model" + for row in cast(list[object], rows)) + if not listed: + return False + probe = ProviderAdapter(ProviderConfig( + "deepseek-readiness", ProviderKind.OPENAI_COMPATIBLE, + "http://127.0.0.1:8000/v1", "/model", None, timeout_seconds=5, + )) + text, _, _ = probe.generate("Reply with the word ready only.", 256) + return bool(text.strip()) + except (OSError, RuntimeError, ValueError, TypeError): + return False + + def all(self) -> bool: + return self.system_one() and self.deliberative() + + def fresh_maintenance_checks(self) -> dict[str, object]: + """Recovery must not accept health cached before shutdown or replacement.""" + return {"clm_qwen": self._system_one(), "deepseek": self._deliberative()} + + +def build() -> HostComponents: + flags = FeatureFlags.from_mapping(os.environ) + if (not flags.enabled_requested or not flags.system_one_enabled + or not flags.deliberative_enabled or flags.system_one_provider != "clm"): + raise ValueError("production host requires enabled CLM and deliberative inference") + if os.environ.get("C3R_MODE") != "production_inference": + raise ValueError("production host requires production_inference mode") + for field in ("C3R_TRACE_COLLECTION", "C3R_ONLINE_LEARNING"): + if os.environ.get(field, "false").lower() not in {"false", "off", "0"}: + raise ValueError("production host cannot collect or learn online") + container = os.environ.get("C3R_CLM_CONTAINER_DIGEST", "") + if len(container) != 71 or not container.startswith("sha256:"): + raise ValueError("measured immutable CLM container identity required") + adapter = ClmAdapter(HEAD_SHA256, timeout_seconds=3) + readiness = ProviderReadiness(adapter, container) + registry = CatalogRegistry((general_catalog(), browser_catalog(), tool_catalog(), + general_catalog("research-v1"))) + verifier = VerifierFirewall({"recommendation": lambda candidate: VerifierDecision( + candidate.risk_class is RiskClass.READ_ONLY, + "read-only recommendation; no tool execution or truth claim", + )}, VerifierPolicy("recommendation"), attestation_key=token_bytes(32)) + runtime = StandaloneController( + flags=flags, compiler=StateCompiler(), candidates=CandidateCompiler(), + cvoc=RobustCvocController(), verifier=verifier, ledger=EphemeralTraceSink(), + fast_path=AdvisoryFastPath(adapter), readiness_probe=readiness.all, + ) + provider = ProviderAdapter(ProviderConfig( + "deepseek-local", ProviderKind.OPENAI_COMPATIBLE, + "http://127.0.0.1:8000/v1", "/model", None, timeout_seconds=60, + )) + manifest = os.environ.get("C3R_INTERNAL_ARTIFACT_MANIFEST", "") + manifest_pin = os.environ.get("C3R_INTERNAL_ARTIFACT_MANIFEST_SHA256", "") + if bool(manifest) != bool(manifest_pin): + raise ValueError("internal required-file manifest and independent SHA pin must be paired") + artifacts = PinnedLocalArtifacts(Path(manifest), manifest_pin) if manifest else None + + def internal_readiness() -> Mapping[str, object]: + report: dict[str, object] = { + "runtime": runtime.decision_enabled and runtime.system_one_enabled + and not runtime.trace_persistence_enabled, + "required_local_artifact_files": artifacts is not None and artifacts.verify(), + "required_local_artifact_manifest_sha256": manifest_pin or None, + } + # Missing/invalid artifact authority cannot produce a recovery-ready result. + # No small-file check is presented as a DeepSeek weight hash attestation. + if report["required_local_artifact_files"] is True: + report.update(readiness.fresh_maintenance_checks()) + return report + + return HostComponents(runtime, registry, SystemOneInference(adapter, readiness=readiness.system_one), + ResponsesService(runtime, registry, provider), internal_readiness) diff --git a/c3r/readiness.py b/c3r/readiness.py new file mode 100644 index 0000000..b3ca82c --- /dev/null +++ b/c3r/readiness.py @@ -0,0 +1,26 @@ +"""Coalesced bounded-TTL health checks; retain only a boolean and expiry.""" +from collections.abc import Callable +from threading import Lock +from time import monotonic + + +class CachedReadiness: + def __init__(self, probe: Callable[[], bool], *, ttl_seconds: float = 15, + clock: Callable[[], float] = monotonic) -> None: + if not 0 < ttl_seconds <= 30: + raise ValueError("readiness TTL must be between zero and 30 seconds") + self._probe, self._ttl, self._clock = probe, ttl_seconds, clock + self._lock = Lock() + self._value = False + self._expires = float("-inf") + + def __call__(self) -> bool: + with self._lock: + if self._clock() < self._expires: + return self._value + try: + self._value = bool(self._probe()) + except (OSError, RuntimeError, ValueError, TypeError, KeyError): + self._value = False + self._expires = self._clock() + self._ttl + return self._value diff --git a/c3r/responses.py b/c3r/responses.py new file mode 100644 index 0000000..b92b474 --- /dev/null +++ b/c3r/responses.py @@ -0,0 +1,210 @@ +"""Text-only, non-storing Responses subset with an explicit host policy fallback.""" +from __future__ import annotations + +import time +from collections.abc import Iterator, Mapping +from dataclasses import dataclass, replace +from threading import Event +from typing import Protocol +from uuid import uuid4 + +from .adapters.providers import ProviderAdapter, TextGenerationStream +from .runtime import RuntimeRequest, StandaloneController +from .state_schema import CompiledState + + +@dataclass(frozen=True, slots=True) +class TextGenerationResult: + text: str + finish: str + usage: dict[str, float] + + +class TextDeliberator: + def __init__(self, provider: ProviderAdapter, maximum: int, *, stream: bool = False, + cancelled: Event | None = None) -> None: + self.provider, self.maximum, self.stream = provider, maximum, stream + self.cancelled = cancelled + + def deliberate(self, state: CompiledState) -> TextGenerationResult | TextGenerationStream: + if not self.provider.config.is_local and state.data_boundary not in {"approved_remote", "public"}: + raise ValueError("state is not approved for hosted generation") + if self.stream: + return self.provider.stream_generate(state.goal, self.maximum, cancelled=self.cancelled) + return TextGenerationResult(*self.provider.generate(state.goal, self.maximum)) + + +class GenerationRequestFactory(Protocol): + def build(self, payload: Mapping[str, object]) -> RuntimeRequest: ... + + +class ResponseEventStream(Iterator[tuple[str, dict[str, object]]]): + """Responses events drawn incrementally from the closeable backend stream.""" + + def __init__(self, backend: TextGenerationStream, *, identifier: str, + message_id: str, reason: str, provider: str) -> None: + self.backend = backend + self.identifier = identifier + self.message_id = message_id + self.reason = reason + self.provider = provider + self.created_at = int(time.time()) + self._started = False + self._done = False + self._pending: list[tuple[str, dict[str, object]]] = [] + self._text = "" + self._finish: str | None = None + self._usage: dict[str, int] | None = None + + def __iter__(self) -> ResponseEventStream: + return self + + def __next__(self) -> tuple[str, dict[str, object]]: + if self._pending: + return self._pending.pop(0) + if self._done: + raise StopIteration + if not self._started: + self._started = True + return "response.created", {"type": "response.created", "response": self._response("in_progress")} + try: + while True: + try: + delta, finish, usage = next(self.backend) + except StopIteration: + if self._finish is None: + raise RuntimeError("provider stream omitted finish reason") + self._done = True + status = "completed" if self._finish == "stop" else "incomplete" + self._pending.append(("response.completed", { + "type": "response.completed", "response": self._response(status), + })) + return "response.output_text.done", { + "type": "response.output_text.done", "response_id": self.identifier, + "output_index": 0, "content_index": 0, "text": self._text, + } + if usage is not None: + self._usage = usage + if delta: + if self._finish is not None: + raise RuntimeError("provider emitted text after finish") + self._text += delta + if len(self._text.encode("utf-8")) > 65_536: + raise RuntimeError("provider output exceeds byte limit") + if finish is not None: + self._finish = finish + return "response.output_text.delta", { + "type": "response.output_text.delta", "response_id": self.identifier, + "output_index": 0, "content_index": 0, "delta": delta, + } + if finish is not None: + self._finish = finish + except (OSError, RuntimeError, TypeError, ValueError): + self.close() + self._done = True + return "response.failed", { + "type": "response.failed", "response": self._response("failed"), + "error": {"type": "server_error", "message": "Generation failed."}, + } + + def _response(self, status: str) -> dict[str, object]: + usage: dict[str, int] | None = None + if self._usage is not None: + input_tokens = int(self._usage["input_tokens"]) + output_tokens = int(self._usage["output_tokens"]) + usage = {"input_tokens": input_tokens, "output_tokens": output_tokens, + "total_tokens": input_tokens + output_tokens} + return { + "id": self.identifier, "object": "response", "created_at": self.created_at, + "model": "c3r-core", "status": status, "store": False, + "output": [] if status == "in_progress" else [{ + "id": self.message_id, "type": "message", "role": "assistant", + "status": status, "content": [{"type": "output_text", "text": self._text, + "annotations": []}], + }], + "usage": usage, + "c3r": {"route": "deliberative", "system_one_provider": "clm", + "controller_reason": self.reason, "effect_executed": False, + "selection_basis": "explicit_text_only_request_policy_fallback", + "calibrated": False, "provider": self.provider}, + } + + def close(self) -> None: + self.backend.close() + + def abort(self) -> None: + self.backend.abort() + + +class ResponsesService: + def __init__(self, runtime: StandaloneController, factory: GenerationRequestFactory, + provider: ProviderAdapter) -> None: + self.runtime, self.factory, self.provider = runtime, factory, provider + + def respond(self, payload: Mapping[str, object]) -> dict[str, object]: + text, maximum = self._validate(payload, stream=False) + decision = self._decide(text, maximum, stream=False) + if not isinstance(decision.deliberation, TextGenerationResult): + raise RuntimeError("controller declined text generation") # noqa: TRY004 + # An explicit caller request for bounded text-only generation is the host + # fallback when unknown task quality gives no positive CVoC. It is NOT a + # positive learned utility claim and cannot execute a catalog action. + output, finish, usage = (decision.deliberation.text, decision.deliberation.finish, + decision.deliberation.usage) + identifier = "resp_" + uuid4().hex + response: dict[str, object] = { + "id": identifier, "object": "response", "created_at": int(time.time()), + "model": "c3r-core", + "status": "completed" if finish == "stop" else "incomplete", "store": False, + "output": [{"id": "msg_" + uuid4().hex, "type": "message", "role": "assistant", + "status": "completed" if finish == "stop" else "incomplete", + "content": [{"type": "output_text", "text": output, "annotations": []}]}], + "usage": {"input_tokens": int(usage["input_tokens"]), + "output_tokens": int(usage["output_tokens"]), + "total_tokens": int(usage["input_tokens"] + usage["output_tokens"])}, + "c3r": {"route": "deliberative", "system_one_provider": "clm", + "controller_reason": decision.reason, "effect_executed": False, + "selection_basis": "explicit_text_only_request_policy_fallback", + "calibrated": False, "provider": self.provider.config.provider_id}, + } + if "provider_cost_usd" in usage: + metadata = response["c3r"] + assert isinstance(metadata, dict) + metadata["observed_provider_cost_usd"] = usage["provider_cost_usd"] + return response + + def stream(self, payload: Mapping[str, object], *, cancelled: Event | None = None) -> ResponseEventStream: + text, maximum = self._validate(payload, stream=True) + decision = self._decide(text, maximum, stream=True, cancelled=cancelled) + if not isinstance(decision.deliberation, TextGenerationStream): + raise RuntimeError("controller declined text generation") # noqa: TRY004 + return ResponseEventStream( + decision.deliberation, identifier="resp_" + uuid4().hex, + message_id="msg_" + uuid4().hex, reason=decision.reason, + provider=self.provider.config.provider_id, + ) + + def _validate(self, payload: Mapping[str, object], *, stream: bool) -> tuple[str, int]: + if set(payload) - {"model", "input", "max_output_tokens", "store", "stream"}: + raise ValueError("unsupported Responses fields") + if payload.get("model") != "c3r-core": + raise ValueError("only c3r-core supports text generation") + if payload.get("store", False) is not False or payload.get("stream", False) is not stream: + raise ValueError("storage is unsupported and stream must be an explicit boolean") + text = payload.get("input") + maximum = payload.get("max_output_tokens", 512) + if (not isinstance(text, str) or not text.strip() or len(text) > 4096 + or len(text.encode()) > 16384 + or isinstance(maximum, bool) or not isinstance(maximum, int) + or not 1 <= maximum <= 2048): + raise ValueError("bounded text input and token limit required") + if not self.runtime.decision_enabled: + raise RuntimeError("C3R disabled") + return text, maximum + + def _decide(self, text: str, maximum: int, *, stream: bool, cancelled: Event | None = None): + request = replace(self.factory.build({"goal": text, "current_subgoal": text}), + requested_text_generation=True) + return self.runtime.run(request, requested_deliberator=TextDeliberator( + self.provider, maximum, stream=stream, cancelled=cancelled, + )) diff --git a/c3r/retention_job.py b/c3r/retention_job.py new file mode 100644 index 0000000..2f87f59 --- /dev/null +++ b/c3r/retention_job.py @@ -0,0 +1,179 @@ +"""Bounded deletion check for the dedicated private C3R trace bucket. + +This is a retention backstop, not permission to collect traces. The deployment +must supply a narrow dedicated-project bucket target. Bucket-scoped IAM is still +required: name validation is defense in depth, not an authorization boundary. +""" + +from __future__ import annotations + +import json +import os +import re +from collections.abc import Mapping +from dataclasses import dataclass +from datetime import datetime, timedelta, timezone +from typing import Protocol +from urllib.parse import quote, urlencode +from urllib.request import Request, urlopen + + +DELETE_AFTER_DAYS = 28 +_BUCKET = re.compile(r"colomboai-c3r-staging-traces-([0-9]{12})\Z") +_METADATA_PROJECT_NUMBER = ( + "http://metadata.google.internal/computeMetadata/v1/project/numeric-project-id" +) +_METADATA_TOKEN = ( + "http://metadata.google.internal/computeMetadata/v1/instance/" + "service-accounts/default/token" +) + + +@dataclass(frozen=True, slots=True) +class StoredObject: + name: str + generation: int + created_at: datetime + + +class ObjectClient(Protocol): + bucket: str + + def list_all(self) -> tuple[StoredObject, ...]: ... + + def delete_generation(self, obj: StoredObject) -> None: ... + + +def _validated_bucket(bucket: str) -> str: + if not _BUCKET.fullmatch(bucket): + raise ValueError("C3R_TRACE_BUCKET must name the dedicated C3R staging trace bucket") + return bucket + + +def bucket_from_environment(values: Mapping[str, str]) -> str: + return _validated_bucket(values.get("C3R_TRACE_BUCKET", "")) + + +def _verify_runtime_project(bucket: str, project_number: str) -> None: + match = _BUCKET.fullmatch(_validated_bucket(bucket)) + assert match is not None + if match.group(1) != project_number: + raise ValueError("C3R_TRACE_BUCKET does not match the Cloud Run project number") + + +def _created_at(raw: str) -> datetime: + parsed = datetime.fromisoformat(raw.replace("Z", "+00:00")) + if parsed.tzinfo is None: + raise ValueError("object creation time must include UTC offset") + return parsed.astimezone(timezone.utc) + + +def purge(client: ObjectClient, *, now: datetime) -> dict[str, object]: + bucket = _validated_bucket(client.bucket) + if now.tzinfo is None or now.utcoffset() != timedelta(0): + raise ValueError("purge clock must be UTC") + cutoff = now - timedelta(days=DELETE_AFTER_DAYS) + before = client.list_all() + if len(before) > 100_000 or len({(item.name, item.generation) for item in before}) != len(before): + raise ValueError("bucket inventory is too large or contains duplicate generations") + expired = tuple(item for item in before if item.created_at <= cutoff) + for item in expired: + client.delete_generation(item) + remaining = client.list_all() + if any(item.created_at <= cutoff for item in remaining): + raise RuntimeError("expired objects remain after purge; collection must stay disabled") + return { + "bucket": bucket, + "cutoff_utc": cutoff.isoformat(), + "objects_before": len(before), + "expired_candidates": len(expired), + "objects_after": len(remaining), + "expired_remaining": 0, + "trace_collection_enabled": False, + } + + +class GcsJsonClient: + """Cloud Run service-identity transport for one validated GCS bucket.""" + + def __init__(self, bucket: str) -> None: + self.bucket = _validated_bucket(bucket) + project_request = Request(_METADATA_PROJECT_NUMBER, + headers={"Metadata-Flavor": "Google"}) + with urlopen(project_request, timeout=10) as response: + project_number = response.read(64).decode("ascii") + _verify_runtime_project(bucket, project_number) + self._storage_api = "https://storage.googleapis.com/storage/v1/b/" + bucket + "/o" + request = Request(_METADATA_TOKEN, headers={"Metadata-Flavor": "Google"}) + with urlopen(request, timeout=10) as response: + data = json.load(response) + token = data.get("access_token") + if not isinstance(token, str) or len(token) < 20: + raise RuntimeError("Cloud Run service identity token unavailable") + self._authorization = "Bearer " + token + + def _request(self, method: str, url: str) -> dict[str, object] | None: + request = Request(url, method=method, headers={"Authorization": self._authorization}) + with urlopen(request, timeout=30) as response: + if method == "DELETE": + return None + data = json.load(response) + if not isinstance(data, dict): + raise ValueError("invalid GCS response") + return data + + def list_all(self) -> tuple[StoredObject, ...]: + found: list[StoredObject] = [] + page_token: str | None = None + seen_tokens: set[str] = set() + while True: + query = {"maxResults": "1000", "versions": "true", + "fields": "items(name,generation,timeCreated),nextPageToken"} + if page_token is not None: + query["pageToken"] = page_token + data = self._request("GET", self._storage_api + "?" + urlencode(query)) + assert data is not None + items = data.get("items", []) + if not isinstance(items, list): + raise ValueError("invalid GCS object list") + for raw in items: + if not isinstance(raw, dict): + raise ValueError("invalid GCS object metadata") + name, generation, created = ( + raw.get("name"), raw.get("generation"), raw.get("timeCreated") + ) + if not isinstance(name, str) or not name or not isinstance(created, str): + raise ValueError("invalid GCS object metadata") + if not isinstance(generation, str) or not generation.isdecimal(): + raise ValueError("invalid GCS object generation") + found.append(StoredObject(name, int(generation), _created_at(created))) + if len(found) > 100_000: + raise ValueError("bucket inventory exceeds purge limit") + next_token = data.get("nextPageToken") + if next_token is None: + return tuple(found) + if not isinstance(next_token, str) or not next_token or next_token in seen_tokens: + raise ValueError("invalid or repeated GCS page token") + seen_tokens.add(next_token) + page_token = next_token + + def delete_generation(self, obj: StoredObject) -> None: + if not obj.name or obj.generation <= 0: + raise ValueError("invalid deletion target") + # GCS identifies a version by (name, generation); including generation + # permanently targets that exact version, including a noncurrent one. + url = self._storage_api + "/" + quote(obj.name, safe="") + "?" + urlencode( + {"generation": str(obj.generation)} + ) + self._request("DELETE", url) + + +def main() -> None: + bucket = bucket_from_environment(os.environ) + result = purge(GcsJsonClient(bucket), now=datetime.now(timezone.utc)) + print(json.dumps(result, sort_keys=True)) + + +if __name__ == "__main__": + main() + diff --git a/c3r/runtime.py b/c3r/runtime.py new file mode 100644 index 0000000..2bcfffc --- /dev/null +++ b/c3r/runtime.py @@ -0,0 +1,351 @@ +"""Host-owned composition of the C3R decision and authority paths. + +The controller accepts bounded inputs from a trusted host. It does not expose an +HTTP endpoint or grant a remote caller permission to execute an action. +""" + +from __future__ import annotations + +import hashlib +import json +from collections.abc import Callable, Mapping +from dataclasses import asdict, dataclass +from math import isfinite +from typing import Protocol + +from .adapters.providers import ProviderExecutionResult +from .candidate_compiler import CandidateCompiler +from .cvoc import RobustCvocController +from .feature_flags import FeatureFlags +from .state_compiler import StateCompiler +from .state_schema import ( + ActionCandidate, + ActionDefinition, + ActionFamily, + AuthorityPolicy, + CompiledState, + RawState, + RiskClass, + ValueEstimate, +) +from .system_one.advisory import AdvisoryFastPath +from .system_one.fast_path import CalibratedFastPath, FastPathDecision +from .system_one.question_registry import TypedQuestion +from .telemetry.ephemeral import EphemeralTraceSink +from .telemetry.trace import DecisionTrace +from .telemetry.trace_ledger import LedgerRecord +from .verifier_firewall import VerifierFirewall + + +class Deliberator(Protocol): + def deliberate(self, state: CompiledState) -> object: ... + + +class TraceSink(Protocol): + def append(self, trace: DecisionTrace) -> LedgerRecord: ... + + +@dataclass(frozen=True, slots=True) +class RuntimeRequest: + raw_state: RawState + definitions: tuple[ActionDefinition, ...] + policy: AuthorityPolicy + estimates: Mapping[str, ValueEstimate] + run_id: str + access_level: str = "internal" + language: str = "en" + requested_text_generation: bool = False + + +@dataclass(frozen=True, slots=True) +class RuntimeOutcome: + route: str + selected_action_id: str | None + authority_result: str + reason: str + ledger_record: LedgerRecord + candidate_ids: tuple[str, ...] = () + fast_path: FastPathDecision | None = None + deliberation: object | None = None + cvoc_lower_bound: float | None = None + + +class StandaloneController: + """Run C3R with independent verification and recommendation-only outcomes. + + Estimates, the policy, and verifier must be supplied by the trusted host. + Supplying an executor is rejected: arbitrary external effects cannot be + atomically committed with the trace ledger. A learned proposal cannot grant + authority. + """ + + def __init__( + self, + *, + flags: FeatureFlags, + compiler: StateCompiler, + candidates: CandidateCompiler, + cvoc: RobustCvocController, + verifier: VerifierFirewall, + ledger: TraceSink, + fast_path: CalibratedFastPath | AdvisoryFastPath | None = None, + deliberator: Deliberator | None = None, + executor: Callable[[ActionCandidate], None] | None = None, + readiness_probe: Callable[[], bool] | None = None, + ) -> None: + if executor is not None: + raise ValueError("external effects are unsupported by StandaloneController") + self._flags = flags + self._compiler = compiler + self._candidates = candidates + self._cvoc = cvoc + self._verifier = verifier + self._ledger = ledger + self._fast_path = fast_path + self._deliberator = deliberator + self._readiness_probe = readiness_probe + + @property + def effect_execution_enabled(self) -> bool: + """Capability flag retained for fail-closed hosting checks.""" + return False + + @property + def decision_enabled(self) -> bool: + """Whether the host requested C3R decisions; not a provider health probe.""" + return self._flags.enabled_requested + + @property + def system_one_enabled(self) -> bool: + return self._flags.system_one_enabled and self._fast_path is not None + + @property + def trace_persistence_enabled(self) -> bool: + """Unknown host sinks are treated as persistent for production gating.""" + return type(self._ledger) is not EphemeralTraceSink + + @property + def provider_ready(self) -> bool: + """No provider-health assertion is made without a host probe.""" + if self._readiness_probe is None: + return False + try: + return bool(self._readiness_probe()) + except (OSError, RuntimeError, TypeError, ValueError): + return False + + def run(self, request: RuntimeRequest, *, + requested_deliberator: Deliberator | None = None) -> RuntimeOutcome: + if not request.run_id: + raise ValueError("run_id is required") + state_hash = hashlib.sha256( + json.dumps( + asdict(request.raw_state), + sort_keys=True, + separators=(",", ":"), + ensure_ascii=False, + allow_nan=False, + ).encode("utf-8") + ).hexdigest() + + def finish( + route: str, + reason: str, + *, + selected: ActionCandidate | None = None, + candidate_ids: tuple[str, ...] = (), + lower_bound: float | None = None, + authority_result: str = "not_attempted", + fast: FastPathDecision | None = None, + deliberation: object | None = None, + system_cost: Mapping[str, float] | None = None, + provider_id: str | None = None, + ) -> RuntimeOutcome: + probabilities = {} if fast is None else dict(fast.probabilities) + if fast is not None and fast.candidate_probabilities: + probabilities["CANDIDATE_RANK"] = fast.candidate_probabilities + trace = DecisionTrace( + run_id=request.run_id, + state_hash=state_hash, + access_level=request.access_level, + model_provider=provider_id or (fast.model_id if fast is not None else route), + candidate_ids=candidate_ids, + probabilities=probabilities, + utility_quantiles=( + {} if lower_bound is None else {"selected_lower_bound": lower_bound} + ), + selected_action_id=None if selected is None else selected.id, + authority_result=authority_result, + system_cost={} if system_cost is None else system_cost, + task_outcome={"status": reason}, + artifact_refs=(), + ) + return RuntimeOutcome( + route=route, + selected_action_id=None if selected is None else selected.id, + authority_result=authority_result, + reason=reason, + ledger_record=self._ledger.append(trace), + candidate_ids=candidate_ids, + fast_path=fast, + deliberation=deliberation, + cvoc_lower_bound=lower_bound, + ) + + if not self._flags.enabled_requested: + return finish("deterministic", "C3R_DISABLED") + + compilation = self._compiler.compile(request.raw_state) + if compilation.state is None: + return finish("deterministic", compilation.status.value) + state = compilation.state + + remaining_budget = state.budget.get("remaining_usd", 0.0) + if remaining_budget < 0: + return finish("deterministic", "INVALID_BUDGET") + compiled = self._candidates.compile_hierarchical( + ( + definition for definition in request.definitions + if definition.family in state.available_action_families + ), + request.policy, + remaining_budget=remaining_budget, + allowed_verifiers=self._verifier.available_verifier_ids, + ) + candidate_ids = tuple(item.id for item in compiled.candidates) + if compiled.no_safe_action: + return finish("deterministic", "NO_SAFE_ACTION", candidate_ids=candidate_ids) + + fast: FastPathDecision | None = None + if self._flags.system_one_enabled: + if self._fast_path is None: + return finish("deterministic", "SYSTEM_ONE_UNAVAILABLE", candidate_ids=candidate_ids) + if self._fast_path.provider != self._flags.system_one_provider: + return finish("deterministic", "SYSTEM_ONE_PROVIDER_MISMATCH", candidate_ids=candidate_ids) + questions = ( + TypedQuestion("STOP_NOW", ("NO", "YES")), + TypedQuestion("DELIBERATION_REQUIRED", ("NO", "YES")), + ) + # Only stable identifiers and public operation metadata cross the CLM + # boundary. Argument values and provenance never enter action labels. + candidate_options = tuple( + f"{item.id} | {item.family.value} | {item.risk_class.value}" + for item in compiled.candidates + ) + try: + fast = self._fast_path.decide( + state, questions, action_family="CONTROL", language=request.language, + candidate_options=candidate_options, + ) + except (OSError, RuntimeError, TypeError, ValueError): + if not request.requested_text_generation: + return self._deliberate_or_stop( + state, finish, candidate_ids, "SYSTEM_ONE_FAILURE", None + ) + if fast is not None and fast.abstained: + return self._deliberate_or_stop( + state, finish, candidate_ids, "SYSTEM_ONE_ABSTAINED", fast + ) + if fast is not None and fast.answers.get("STOP_NOW") == "YES": + return finish("system_one", "STOP_NOW", candidate_ids=candidate_ids, fast=fast) + if fast is not None and fast.answers.get("DELIBERATION_REQUIRED") == "YES": + return self._deliberate_or_stop( + state, finish, candidate_ids, "DELIBERATION_REQUIRED", fast + ) + + decision = self._cvoc.select(compiled.candidates, request.estimates) + if decision.selected is None: + if request.requested_text_generation and requested_deliberator is not None: + # Trusted host opt-in for caller-requested bounded text only. + # Unknown quality still has no positive CVoC. This fallback must + # be admitted by the catalog AND independently verified. + fallback = next((candidate for candidate in compiled.candidates + if candidate.family is ActionFamily.DELIBERATE + and candidate.risk_class is RiskClass.READ_ONLY), None) + if fallback is not None: + try: + verification = self._verifier.verify(fallback) + except (OSError, RuntimeError, TypeError, ValueError): + return finish("deterministic", "VERIFIER_FAILURE", candidate_ids=candidate_ids) + if verification.accepted: + return self._deliberate_or_stop( + state, finish, candidate_ids, "REQUESTED_TEXT_POLICY_FALLBACK", fast, + deliberator=requested_deliberator, + ) + return finish("deterministic", "VERIFICATION_REJECTED", candidate_ids=candidate_ids) + return finish( + "deterministic", "NON_POSITIVE_CVOC", candidate_ids=candidate_ids, fast=fast + ) + selected = decision.selected + try: + verification = self._verifier.verify(selected) + except (OSError, RuntimeError, TypeError, ValueError): + return finish( + "deterministic", "VERIFIER_FAILURE", candidate_ids=candidate_ids, fast=fast + ) + if not verification.accepted: + return finish( + "deterministic", "VERIFICATION_REJECTED", candidate_ids=candidate_ids, fast=fast + ) + if selected.family is ActionFamily.DELIBERATE: + return self._deliberate_or_stop( + state, finish, candidate_ids, "CVOC_SELECTED_DELIBERATION", fast, + deliberator=requested_deliberator if request.requested_text_generation else None, + ) + if selected.risk_class is not RiskClass.READ_ONLY: + return finish( + "deterministic", + "EFFECT_EXECUTION_UNAVAILABLE", + candidate_ids=candidate_ids, + lower_bound=decision.lower_bound, + fast=fast, + ) + return finish( + "recommendation", + "VERIFIED_RECOMMENDATION", + selected=selected, + candidate_ids=candidate_ids, + lower_bound=decision.lower_bound, + authority_result="verified_not_committed", + fast=fast, + ) + + def _deliberate_or_stop( + self, + state: CompiledState, + finish: Callable[..., RuntimeOutcome], + candidate_ids: tuple[str, ...], + reason: str, + fast: FastPathDecision | None, + *, deliberator: Deliberator | None = None, + ) -> RuntimeOutcome: + deliberator = deliberator or self._deliberator + if not self._flags.deliberative_enabled or deliberator is None: + return finish( + "deterministic", reason + "_NO_PROVIDER", candidate_ids=candidate_ids, fast=fast + ) + try: + deliberation = deliberator.deliberate(state) + except (OSError, RuntimeError, TypeError, ValueError): + return finish( + "deterministic", "DELIBERATIVE_FAILURE", candidate_ids=candidate_ids, fast=fast + ) + if isinstance(deliberation, ProviderExecutionResult): + cost = deliberation.observed_cost + if any(not isfinite(value) or value < 0 for value in cost.values()): + return finish( + "deterministic", "DELIBERATIVE_INVALID_COST", candidate_ids=candidate_ids, + fast=fast, + ) + return finish( + "deliberative", reason, candidate_ids=candidate_ids, fast=fast, + deliberation=deliberation.deliberation, system_cost=cost, + provider_id=deliberation.provider_id, + ) + return finish( + "deliberative", + reason, + candidate_ids=candidate_ids, + fast=fast, + deliberation=deliberation, + ) diff --git a/c3r/serve.py b/c3r/serve.py new file mode 100644 index 0000000..be05dd5 --- /dev/null +++ b/c3r/serve.py @@ -0,0 +1,214 @@ +"""Fail-closed composition of a host-supplied C3R controller and HTTP ingress. + +The host entrypoint owns the action catalog, measured estimates, verifier policy, +and trace sink. This module supplies no demo estimates or implicit data collection. +""" + +from __future__ import annotations + +import importlib +import os +import secrets +import signal +import threading +from collections.abc import Callable, Mapping +from pathlib import Path +from typing import Protocol, cast + +from .api_access import AccessStore +from .host_components import HostComponents +from .http_service import C3RHTTPServer, RequestFactory +from .ingress_proxy import C3RIngressServer +from .internal_readiness import InternalReadinessServer +from .runtime import StandaloneController + + +class HostBuilder(Protocol): + def __call__(self) -> tuple[StandaloneController, RequestFactory] | HostComponents: ... + + +def _required(values: Mapping[str, str], name: str) -> str: + value = values.get(name, "") + if not value: + raise ValueError(f"{name} is required") + return value + + +def _port(values: Mapping[str, str], name: str, default: int) -> int: + raw = values.get(name, str(default)) + try: + value = int(raw) + except ValueError as error: + raise ValueError(f"{name} must be a TCP port") from error + if not 1 <= value <= 65535: + raise ValueError(f"{name} must be a TCP port") + return value + + +def load_host_builder(reference: str) -> HostBuilder: + """Load an explicitly configured, trusted module:function host composition.""" + module_name, separator, attribute = reference.partition(":") + if not separator or not module_name or not attribute or not attribute.isidentifier(): + raise ValueError("C3R_HOST_ENTRYPOINT must be module:function") + module = importlib.import_module(module_name) + builder = getattr(module, attribute) + if not callable(builder): + raise ValueError("C3R_HOST_ENTRYPOINT is not callable") + return cast(HostBuilder, builder) + + +def build_servers( + values: Mapping[str, str], + *, + builder_loader: Callable[[str], HostBuilder] = load_host_builder, +) -> tuple[C3RHTTPServer, C3RIngressServer]: + """Validate configuration before binding a public interface.""" + reference = _required(values, "C3R_HOST_ENTRYPOINT") + client_token = values.get("C3R_CLIENT_TOKEN", "") + configured_mode = values.get("C3R_MODE", "staging") + if values.get("C3R_API_AUTH_MODE", "keys" if configured_mode == "production_inference" else "static_staging") == "static_staging": + client_token = _required(values, "C3R_CLIENT_TOKEN") + backend_token = _required(values, "C3R_BACKEND_TOKEN") + port = _port(values, "PORT", 8080) + backend_port = _port(values, "C3R_BACKEND_PORT", 8081) + ingress_host = values.get("C3R_INGRESS_HOST", "0.0.0.0") + if ingress_host not in {"0.0.0.0", "127.0.0.1"}: + raise ValueError("ingress host must be loopback or the TLS-host container interface") + if port == backend_port: + raise ValueError("ingress and backend ports must differ") + components = builder_loader(reference)() + if isinstance(components, HostComponents): + runtime, factory = components.runtime, components.factory + else: + runtime, factory = components + if runtime.effect_execution_enabled: + raise ValueError("host must be recommendation-only") + mode = values.get("C3R_MODE", "staging") + if mode not in {"staging", "production_inference", "research_collection"}: + raise ValueError("C3R_MODE is invalid") + if mode == "production_inference": + if values.get("C3R_TRACE_COLLECTION", "false").strip().lower() not in {"false", "0", "off"}: + raise ValueError("production inference cannot collect traces") + if values.get("C3R_ONLINE_LEARNING", "false").strip().lower() not in {"false", "0", "off"}: + raise ValueError("production inference cannot learn online") + if runtime.trace_persistence_enabled: + raise ValueError("production inference requires the ephemeral trace sink") + if not runtime.decision_enabled: + raise ValueError("production inference requires decisions enabled") + if not runtime.system_one_enabled: + raise ValueError("production inference requires an enabled System-One path") + auth_mode = values.get("C3R_API_AUTH_MODE", "keys" if mode == "production_inference" else "static_staging") + if auth_mode not in {"keys", "static_staging"} or (mode == "production_inference" and auth_mode != "keys"): + raise ValueError("production inference requires API key authentication") + access_store = (AccessStore(Path(_required(values, "C3R_API_ACCESS_DB"))) + if auth_mode == "keys" else None) + if access_store is None: + client_token = _required(values, "C3R_CLIENT_TOKEN") + elif not client_token: + client_token = secrets.token_urlsafe(32) # Unused by key mode; never an accepted client credential. + backend = C3RHTTPServer( + runtime=runtime, + request_factory=factory, + bearer_token=backend_token, + port=backend_port, + system_one=components.system_one if isinstance(components, HostComponents) else None, + responses=components.responses if isinstance(components, HostComponents) else None, + internal_readiness=components.internal_readiness if isinstance(components, HostComponents) else None, + ) + try: + ingress = C3RIngressServer( + upstream_port=backend.server_port, + client_token=client_token, + upstream_token=backend_token, + port=port, + host=ingress_host, + upstream_timeout_seconds=65 if mode == "production_inference" else 5, + access_store=access_store, + ) + except BaseException: + backend.server_close() + raise + return backend, ingress + + +def build_internal_server(values: Mapping[str, str], + backend: C3RHTTPServer, *, + api_workers: tuple[threading.Thread, ...] = (), + stopping: threading.Event | None = None) -> InternalReadinessServer | None: + """Opt-in separate maintenance binding; never forwarded by the public gateway.""" + enabled = "C3R_INTERNAL_READY_PORT" in values or "C3R_INTERNAL_READY_TOKEN" in values + if not enabled: + return None + token = _required(values, "C3R_INTERNAL_READY_TOKEN") + _required(values, "C3R_INTERNAL_READY_PORT") + port = _port(values, "C3R_INTERNAL_READY_PORT", 8091) + if (port in {backend.server_port, _port(values, "PORT", 8080)} + or token in {values.get("C3R_CLIENT_TOKEN", ""), + _required(values, "C3R_BACKEND_TOKEN")}): + raise ValueError("internal readiness must use a separate port and token") + return InternalReadinessServer(token=token, port=port, probe=backend.internal_readiness, + api_workers=api_workers, stopping=stopping) + + +def main() -> None: + backend, ingress = build_servers(os.environ) + stop = threading.Event() + backend_thread = threading.Thread(target=backend.serve_forever, daemon=True) + ingress_thread = threading.Thread(target=ingress.serve_forever, daemon=True) + try: + internal = build_internal_server(os.environ, backend, + api_workers=(backend_thread, ingress_thread), stopping=stop) + except BaseException: + ingress.server_close() + backend.server_close() + raise + + def request_stop(_signum: int, _frame: object) -> None: + stop.set() + + signal.signal(signal.SIGTERM, request_stop) + signal.signal(signal.SIGINT, request_stop) + backend_started = False + ingress_started = False + internal_thread = (threading.Thread(target=internal.serve_forever, daemon=True) + if internal is not None else None) + internal_started = False + worker_failure = False + try: + backend_thread.start() + backend_started = True + ingress_thread.start() + ingress_started = True + if internal_thread is not None: + internal_thread.start() + internal_started = True + while not stop.wait(0.25): + if (not backend_thread.is_alive() or not ingress_thread.is_alive() + or (internal_thread is not None and not internal_thread.is_alive())): + worker_failure = True + stop.set() + finally: + stop.set() + if internal_started and internal is not None and internal_thread is not None and internal_thread.is_alive(): + internal.shutdown() + if ingress_started and ingress_thread.is_alive(): + ingress.shutdown() + if backend_started and backend_thread.is_alive(): + backend.shutdown() + ingress.server_close() + backend.server_close() + if internal is not None: + internal.server_close() + if internal_started and internal_thread is not None: + internal_thread.join(timeout=5) + if ingress_started: + ingress_thread.join(timeout=5) + if backend_started: + backend_thread.join(timeout=5) + if worker_failure: + raise RuntimeError("C3R serving worker exited unexpectedly") + + +if __name__ == "__main__": + main() + diff --git a/c3r/staging_host.py b/c3r/staging_host.py new file mode 100644 index 0000000..06b3655 --- /dev/null +++ b/c3r/staging_host.py @@ -0,0 +1,54 @@ +"""Private Cloud Run boundary-smoke host; never enables C3R decisions or collection. + +This is deliberately not a production host. It proves packaging and ingress only. +No provider, external effect, or durable trace sink is configured here. +""" + +from __future__ import annotations + +from secrets import token_bytes + +from .candidate_compiler import CandidateCompiler +from .cvoc import RobustCvocController +from .feature_flags import FeatureFlags +from .host_factory import ReadOnlyRequestFactory +from .runtime import StandaloneController +from .state_compiler import StateCompiler +from .state_schema import ActionDefinition, ActionFamily, AuthorityPolicy, RiskClass +from .telemetry.ephemeral import EphemeralTraceSink +from .verifier_firewall import VerifierDecision, VerifierFirewall, VerifierPolicy + + +EphemeralStagingSink = EphemeralTraceSink + + +def build() -> tuple[StandaloneController, ReadOnlyRequestFactory]: + """Construct a fixed-disabled, recommendation-only staging boundary.""" + key = token_bytes(32) + verifier = VerifierFirewall( + {"deny": lambda _candidate: VerifierDecision(False, "staging disabled")}, + VerifierPolicy(default_verifier="deny"), attestation_key=key, + ) + controller = StandaloneController( + flags=FeatureFlags(enabled_requested=False), + compiler=StateCompiler(), candidates=CandidateCompiler(), + cvoc=RobustCvocController(), verifier=verifier, + ledger=EphemeralStagingSink(), + ) + definition = ActionDefinition( + id="staging_noop", family=ActionFamily.TOOL, subgroup="staging", + operation="noop", risk_class=RiskClass.READ_ONLY, + argument_variants=((),), placements=("local",), verifier_ids=("deny",), + optimistic_utility=0.0, estimated_cost=0.0, + ) + factory = ReadOnlyRequestFactory( + definitions=(definition,), + policy=AuthorityPolicy( + frozenset({ActionFamily.TOOL}), frozenset({RiskClass.READ_ONLY}), + ), + estimate_source=lambda _state: {}, + remaining_usd=0.0, + data_boundary="local", + ) + return controller, factory + diff --git a/c3r/system_one/__init__.py b/c3r/system_one/__init__.py index f646d19..b8eab5e 100644 --- a/c3r/system_one/__init__.py +++ b/c3r/system_one/__init__.py @@ -1,8 +1,13 @@ """Bounded System-One controller interfaces.""" -from .fast_path import LayaFastPath +from .clm_adapter import ClmAdapter +from .factory import build_default_clm_fast_path +from .fast_path import CalibratedFastPath, LayaFastPath from .laya_adapter import LayaAdapter from .laya_backend import PinnedLayaBackend from .question_registry import C3R_QUESTIONS -__all__ = ["C3R_QUESTIONS", "LayaAdapter", "LayaFastPath", "PinnedLayaBackend"] +__all__ = [ + "C3R_QUESTIONS", "CalibratedFastPath", "ClmAdapter", "LayaAdapter", + "LayaFastPath", "PinnedLayaBackend", "build_default_clm_fast_path", +] diff --git a/c3r/system_one/advisory.py b/c3r/system_one/advisory.py new file mode 100644 index 0000000..90e4704 --- /dev/null +++ b/c3r/system_one/advisory.py @@ -0,0 +1,23 @@ +"""Uncalibrated ranker which makes no learned control/authority decisions.""" +from collections.abc import Sequence + +from ..state_schema import CompiledState +from .clm_adapter import ClmAdapter +from .fast_path import FastPathDecision +from .question_registry import TypedQuestion + + +class AdvisoryFastPath: + def __init__(self, adapter: ClmAdapter) -> None: + self.adapter = adapter + + @property + def provider(self) -> str: + return self.adapter.provider + + def decide(self, state: CompiledState, questions: Sequence[TypedQuestion], *, + action_family: str, language: str = "en", + candidate_options: tuple[str, ...] = ()) -> FastPathDecision: + scores = self.adapter.rank_actions(state, candidate_options) if candidate_options else () + return FastPathDecision({}, {}, False, ("UNCALIBRATED_ADVISORY_ONLY",), + self.adapter.model_id, self.adapter.revision, scores) diff --git a/c3r/system_one/clm_adapter.py b/c3r/system_one/clm_adapter.py new file mode 100644 index 0000000..1f11cb1 --- /dev/null +++ b/c3r/system_one/clm_adapter.py @@ -0,0 +1,171 @@ +"""Bounded, loopback-only adapter for the pinned upstream CLM rank API. + +CLM predictions are never authority or calibrated probabilities. The guarded +fast path consumes these as logits only when a held-out calibration slice exists. +""" + +from __future__ import annotations + +import json +import math +import re +import time +from collections.abc import Callable, Mapping +from dataclasses import asdict, dataclass +from typing import cast +from urllib.parse import urlsplit +from urllib.request import HTTPHandler, ProxyHandler, Request, build_opener + +from ..http_transport import NoRedirectHandler +from ..state_schema import CompiledState +from .question_registry import TypedQuestion + +UPSTREAM_CLM_COMMIT = "bb42c6c5bf914fd449bed2f6ca65be80602cb1f7" +_IMMUTABLE_REVISION = re.compile(r"^[0-9a-f]{40,64}$") +_MAX_CONTEXT_BYTES = 32_768 +_MAX_RESPONSE_BYTES = 65_536 +_MAX_OPTIONS = 64 +RankTransport = Callable[[Mapping[str, object]], Mapping[str, object]] + + +@dataclass(frozen=True, slots=True) +class ClmAdapter: + """Convert CLM rankings to typed logits in the original option order. + + ``revision`` is the host-declared CLM head/encoder bundle hash, not an + attestation from the server. The upstream code pin is recorded separately. + """ + + revision: str + endpoint: str = "http://127.0.0.1:8700" + api_key: str | None = None + timeout_seconds: float = 0.5 + transport: RankTransport | None = None + model_id: str = "Contrastive-LM/CLM" + served_model: str = "clm-latest" + + @property + def provider(self) -> str: + return "clm" + + def __post_init__(self) -> None: + parsed = urlsplit(self.endpoint) + if ( + parsed.scheme != "http" + or parsed.hostname not in {"127.0.0.1", "localhost", "::1"} + or parsed.path not in {"", "/"} + or parsed.username is not None + or parsed.password is not None + or parsed.query + or parsed.fragment + or parsed.port is None + ): + raise ValueError("CLM endpoint must be a local HTTP origin with an explicit port") + if _IMMUTABLE_REVISION.fullmatch(self.revision) is None: + raise ValueError("CLM artifact revision must be an immutable SHA-256/SHA-1 hash") + if not math.isfinite(self.timeout_seconds) or not 0 < self.timeout_seconds <= 10: + raise ValueError("CLM timeout must be finite and at most ten seconds") + if not self.served_model or len(self.served_model) > 128: + raise ValueError("CLM served model name must be bounded") + + def predict( + self, state: CompiledState, questions: tuple[TypedQuestion, ...], + *, deadline: float | None = None, + ) -> Mapping[str, tuple[float, ...]]: + context = self._context(state) + output: dict[str, tuple[float, ...]] = {} + for question in questions: + probabilities = self._rank(context, question.id, question.options, deadline=deadline) + output[question.id] = tuple(math.log(max(value, 1e-12)) for value in probabilities) + return output + + def rank_actions( + self, state: CompiledState, candidate_ids: tuple[str, ...], + *, deadline: float | None = None, + ) -> tuple[float, ...]: + """Advisory candidate distribution; CVoC and policy still choose actions.""" + return self._rank(self._context(state), "NEXT_ACTION", candidate_ids, deadline=deadline) + + def rank_text(self, context: str, question: str, options: tuple[str, ...], + *, deadline: float | None = None) -> tuple[float, ...]: + """Rank caller text without assigning it policy or execution authority.""" + if not context or len(context.encode("utf-8")) > _MAX_CONTEXT_BYTES: + raise ValueError("CLM context must be nonempty and bounded") + if not question or len(question) > 1024: + raise ValueError("CLM question must be nonempty and bounded") + return self._rank(context, question, options, deadline=deadline) + + def _context(self, state: CompiledState) -> str: + context = json.dumps(asdict(state), sort_keys=True, allow_nan=False, ensure_ascii=False) + if len(context.encode("utf-8")) > _MAX_CONTEXT_BYTES: + raise ValueError("compiled state exceeds the bounded CLM context") + return context + + def _rank( + self, context: str, question: str, options: tuple[str, ...], + *, deadline: float | None = None, + ) -> tuple[float, ...]: + if not options or len(options) > _MAX_OPTIONS or len(set(options)) != len(options): + raise ValueError("CLM options must be unique and bounded") + if any(not option or len(option) > 1024 for option in options): + raise ValueError("CLM option is empty or oversized") + payload: Mapping[str, object] = { + "context": context, + "question": question, + "answers": list(options), + "model": self.served_model, + } + remaining = self.timeout_seconds + if deadline is not None: + remaining = min(remaining, deadline - time.monotonic()) + if remaining <= 0: + raise TimeoutError("CLM decision deadline exceeded") + response = self.transport(payload) if self.transport is not None else self._post(payload, remaining) + if response.get("model") != self.served_model: + raise ValueError("CLM served model does not match the requested model") + ranked = response.get("ranked") + if not isinstance(ranked, list) or len(ranked) != len(options): + raise ValueError("CLM returned an incomplete ranking") + probabilities: dict[str, float] = {} + for item in ranked: + if not isinstance(item, dict): + raise ValueError("CLM returned an invalid ranking item") + candidate = item.get("candidate") + probability = item.get("prob") + if ( + not isinstance(candidate, str) + or candidate not in options + or candidate in probabilities + or isinstance(probability, bool) + or not isinstance(probability, (int, float)) + or not math.isfinite(probability) + or not 0 <= probability <= 1 + ): + raise ValueError("CLM ranking contains an unknown or invalid probability") + probabilities[candidate] = float(probability) + total = sum(probabilities.values()) + if not 0.98 <= total <= 1.02: + raise ValueError("CLM ranking probabilities do not sum to one") + return tuple(probabilities[option] / total for option in options) + + def _post(self, payload: Mapping[str, object], timeout: float) -> Mapping[str, object]: + headers = {"Content-Type": "application/json"} + if self.api_key: + headers["Authorization"] = f"Bearer {self.api_key}" + request = Request( + self.endpoint.rstrip("/") + "/v1/rank", + data=json.dumps(payload, ensure_ascii=False, allow_nan=False).encode("utf-8"), + headers=headers, + method="POST", + ) + # Ignore proxy environment variables and reject redirects so a local + # server cannot relay the secret or state to an off-host destination. + opener = build_opener(ProxyHandler({}), NoRedirectHandler(), HTTPHandler()) + with opener.open(request, timeout=timeout) as response: + body = response.read(_MAX_RESPONSE_BYTES + 1) + if len(body) > _MAX_RESPONSE_BYTES: + raise ValueError("CLM response exceeds the allowed size") + parsed = json.loads(body) + if not isinstance(parsed, dict): + raise ValueError("CLM returned a non-object response") + return cast(Mapping[str, object], parsed) diff --git a/c3r/system_one/factory.py b/c3r/system_one/factory.py new file mode 100644 index 0000000..c7cf083 --- /dev/null +++ b/c3r/system_one/factory.py @@ -0,0 +1,36 @@ +"""Explicit construction of the default CLM System-One path. + +No mutable upstream model alias or unfitted calibration is silently promoted. +The host owns secrets, artifacts, and the enable flag. +""" + +from __future__ import annotations + +from collections.abc import Mapping + +from .calibration import TemperatureCalibrator +from .clm_adapter import ClmAdapter +from .fast_path import CalibratedFastPath + + +def build_default_clm_fast_path( + values: Mapping[str, str], *, calibrator: TemperatureCalibrator +) -> CalibratedFastPath: + """Build CLM; require a host-declared immutable artifact revision. + + This does not attest the live server's artifacts. Deployment must verify + encoder/head hashes independently before enabling real decisions. + """ + revision = values.get("C3R_CLM_ARTIFACT_REVISION", "") + endpoint = values.get("C3R_CLM_URL", "http://127.0.0.1:8700") + timeout = float(values.get("C3R_CLM_TIMEOUT_MS", "500")) / 1000 + return CalibratedFastPath( + adapter=ClmAdapter( + revision=revision, + endpoint=endpoint, + api_key=values.get("C3R_CLM_API_KEY"), + timeout_seconds=timeout, + ), + calibrator=calibrator, + maximum_decision_seconds=timeout, + ) diff --git a/c3r/system_one/fast_path.py b/c3r/system_one/fast_path.py index dd02c9e..d84f6df 100644 --- a/c3r/system_one/fast_path.py +++ b/c3r/system_one/fast_path.py @@ -4,14 +4,32 @@ from dataclasses import dataclass import math +import time +from collections.abc import Mapping +from typing import Protocol from ..state_schema import CompiledState from .abstention import should_abstain from .calibration import CalibrationKey, TemperatureCalibrator -from .laya_adapter import LayaAdapter from .question_registry import TypedQuestion +class TypedInferenceAdapter(Protocol): + model_id: str + revision: str + provider: str + + def predict( + self, state: CompiledState, questions: tuple[TypedQuestion, ...], + *, deadline: float | None = None, + ) -> Mapping[str, tuple[float, ...]]: ... + + def rank_actions( + self, state: CompiledState, candidate_ids: tuple[str, ...], + *, deadline: float | None = None, + ) -> tuple[float, ...]: ... + + def option_count_bucket(option_count: int) -> str: if option_count <= 2: return str(option_count) @@ -32,25 +50,34 @@ class FastPathDecision: reasons: tuple[str, ...] model_id: str model_revision: str + candidate_probabilities: tuple[float, ...] = () -class LayaFastPath: +class CalibratedFastPath: def __init__( self, *, - adapter: LayaAdapter, + adapter: TypedInferenceAdapter, calibrator: TemperatureCalibrator, minimum_top_probability: float = 0.65, minimum_margin: float = 0.10, + maximum_decision_seconds: float = 0.5, ) -> None: if not 0.0 <= minimum_top_probability <= 1.0: raise ValueError("minimum_top_probability must be between 0 and 1") if not 0.0 <= minimum_margin <= 1.0: raise ValueError("minimum_margin must be between 0 and 1") + if not math.isfinite(maximum_decision_seconds) or maximum_decision_seconds <= 0: + raise ValueError("maximum_decision_seconds must be positive and finite") self._adapter = adapter self._calibrator = calibrator self._minimum_top_probability = minimum_top_probability self._minimum_margin = minimum_margin + self._maximum_decision_seconds = maximum_decision_seconds + + @property + def provider(self) -> str: + return self._adapter.provider def decide( self, @@ -59,8 +86,15 @@ def decide( *, action_family: str, language: str = "en", + candidate_options: tuple[str, ...] = (), ) -> FastPathDecision: - logits_by_question = self._adapter.predict(state, questions) + deadline = time.monotonic() + self._maximum_decision_seconds + candidate_probabilities: tuple[float, ...] = () + if candidate_options: + candidate_probabilities = self._adapter.rank_actions( + state, candidate_options, deadline=deadline + ) + logits_by_question = self._adapter.predict(state, questions, deadline=deadline) consequence = str(state.risk.get("consequence", "default")) answers: dict[str, str] = {} probabilities: dict[str, tuple[float, ...]] = {} @@ -104,4 +138,10 @@ def decide( reasons=tuple(reasons), model_id=self._adapter.model_id, model_revision=self._adapter.revision, + candidate_probabilities=candidate_probabilities, ) + + +# Preserve the public Laya integration while allowing the same guarded path to +# serve CLM and other explicitly configured System-One engines. +LayaFastPath = CalibratedFastPath diff --git a/c3r/system_one/inference.py b/c3r/system_one/inference.py new file mode 100644 index 0000000..ad38a07 --- /dev/null +++ b/c3r/system_one/inference.py @@ -0,0 +1,146 @@ +"""Stateless, non-authoritative CLM inference, separate from governed actions.""" + +from __future__ import annotations + +import json +import time +from collections.abc import Callable, Mapping +from typing import cast + +from .clm_adapter import ClmAdapter + + +def _text(value: object, limit: int = 1024) -> str: + if not isinstance(value, str) or not value.strip() or len(value) > limit: + raise ValueError("nonempty bounded text required") + return value + + +def _mapping(value: object) -> Mapping[str, object]: + if not isinstance(value, dict): + raise ValueError("object required") + return cast(Mapping[str, object], value) + + +def _list(value: object) -> list[object]: + if not isinstance(value, list): + raise ValueError("array required") + return cast(list[object], value) + + +class SystemOneInference: + """Typed questions and arbitrary strings: scores are NOT success probabilities.""" + + def __init__(self, adapter: ClmAdapter, *, readiness: Callable[[], bool] | None = None) -> None: + self.adapter = adapter + self._readiness = readiness + + @property + def ready(self) -> bool: + if self._readiness is None: + return False + try: + return self._readiness() + except (OSError, RuntimeError, ValueError, TypeError): + return False + + def _rank(self, context: str, question: str, options: tuple[str, ...], + deadline: float) -> tuple[float, ...]: + try: + return self.adapter.rank_text(context, question, options, deadline=deadline) + except (OSError, TypeError, ValueError, KeyError) as error: + raise RuntimeError("CLM inference unavailable") from error + + def infer(self, payload: Mapping[str, object]) -> dict[str, object]: + if set(payload) - {"model", "state", "questions", "candidates"}: + raise ValueError("unsupported System-One fields") + model = payload.get("model", "c3r-system-one") + if model not in {"c3r-system-one", "c3r-verifier"}: + raise ValueError("unsupported ranking model") + state = payload.get("state") + if isinstance(state, dict): + context = json.dumps(state, allow_nan=False, ensure_ascii=False, sort_keys=True) + else: + context = _text(state, 32768) + if len(context.encode("utf-8")) > 32768: + raise ValueError("state too large") + if ("questions" in payload) == ("candidates" in payload): + raise ValueError("provide either questions or candidates") + # One total deadline, not a fresh timeout for each question. + deadline = time.monotonic() + self.adapter.timeout_seconds + result: dict[str, object] = { + "model": model, "provider": "clm", "calibrated": False, + "effect_executed": False, "scope": "advisory_only", + } + if "candidates" in payload: + raw = _list(payload["candidates"]) + if not 1 <= len(raw) <= 64: + raise ValueError("candidate count out of bounds") + options = tuple(_text(option) for option in raw) + if len(set(options)) != len(options): + raise ValueError("candidates must be unique") + scores = self._rank(context, "NEXT_ACTION", options, deadline) + result["ranked"] = [ + {"candidate": option, "score": score} + for option, score in sorted(zip(options, scores), key=lambda pair: pair[1], + reverse=True) + ] + return result + questions = _mapping(payload["questions"]) + if not 1 <= len(questions) <= 16: + raise ValueError("question count out of bounds") + # Validate ALL questions before making any provider calls. + prepared: list[tuple[str, str, str, tuple[str, ...], tuple[str, ...]]] = [] + for name, value in questions.items(): + name = _text(name, 128) + value = _mapping(value) + if set(value) - {"type", "options", "criteria", "instructions"}: + raise ValueError("invalid question") + kind = value.get("type") + instructions = value.get("instructions", name) + prompt = _text(instructions) + if "options" in value and "criteria" in value: + raise ValueError("ambiguous options") + raw = value.get("options", value.get("criteria")) + if kind == "choice": + raw = _mapping(raw) + if not 2 <= len(raw) <= 64: + raise ValueError("choice requires option descriptions") + keys = tuple(_text(key, 128) for key in raw) + options = tuple(_text(description) for description in raw.values()) + elif kind in {"boolean", "noul"}: + keys = ("false", "true") + if raw is not None: + raw = _mapping(raw) + if set(raw) != {"false", "true"}: + raise ValueError("boolean criteria require false and true") + options = tuple(_text(raw[key]) for key in keys) + else: + options = (f"false: No. This is false: {prompt}", + f"true: Yes. This is true: {prompt}") + elif kind == "score": + raw = _list(raw) + if not 2 <= len(raw) <= 64: + raise ValueError("score requires ordered levels") + options = tuple(_text(item) for item in raw) + keys = tuple(str(index) for index in range(len(options))) + else: + raise ValueError("unsupported question type") + if len(set(options)) != len(options) or any(len(option) > 1024 for option in options): + raise ValueError("options must be unique and bounded") + prepared.append((name, cast(str, kind), prompt, keys, options)) + if sum(len(options) for _, _, _, _, options in prepared) > 64: + raise ValueError("total options exceeds budget") + answers: dict[str, object] = {} + for name, kind, prompt, keys, options in prepared: + scores = self._rank(context, prompt, options, deadline) + answer: dict[str, object] = {"scores": dict(zip(keys, scores))} + if kind == "choice": + answer["choice"] = keys[max(range(len(scores)), key=lambda index: scores[index])] + elif kind in {"boolean", "noul"}: + answer["score"] = scores[1] + else: + answer["score"] = sum(index * score for index, score in enumerate(scores)) + answers[name] = answer + result["answers"] = answers + return result diff --git a/c3r/system_one/laya_adapter.py b/c3r/system_one/laya_adapter.py index 8f01ab0..8149d46 100644 --- a/c3r/system_one/laya_adapter.py +++ b/c3r/system_one/laya_adapter.py @@ -19,6 +19,17 @@ class LayaAdapter: revision: str backend: InferenceBackend + @property + def provider(self) -> str: + return "laya" + + def rank_actions( + self, state: CompiledState, candidate_ids: tuple[str, ...], + *, deadline: float | None = None, + ) -> tuple[float, ...]: + """This legacy typed-question adapter has no action-ranking head.""" + return () + def __post_init__(self) -> None: if self.model_id not in { "convaiinnovations/laya", @@ -33,5 +44,6 @@ def predict( self, state: CompiledState, questions: tuple[TypedQuestion, ...], + *, deadline: float | None = None, ) -> Mapping[str, tuple[float, ...]]: return self.backend(state, questions) diff --git a/c3r/telemetry/ephemeral.py b/c3r/telemetry/ephemeral.py new file mode 100644 index 0000000..37de5d3 --- /dev/null +++ b/c3r/telemetry/ephemeral.py @@ -0,0 +1,15 @@ +"""Request-local trace hashing without retained rows or a persistent ledger.""" + +from __future__ import annotations + +from .trace import DecisionTrace +from .trace_ledger import LedgerRecord, _record_hash, canonical_trace_json + + +class EphemeralTraceSink: + """Hash each response independently; the returned record is not stored.""" + + def append(self, trace: DecisionTrace) -> LedgerRecord: + canonical = canonical_trace_json(trace) + genesis = "0" * 64 + return LedgerRecord(genesis, _record_hash(genesis, canonical), canonical) diff --git a/c3r/telemetry/governed_store.py b/c3r/telemetry/governed_store.py new file mode 100644 index 0000000..2bd3fc0 --- /dev/null +++ b/c3r/telemetry/governed_store.py @@ -0,0 +1,312 @@ +"""Fail-closed local admission and retention for C3R-authored internal traces. + +The host still owns encryption, backup purge, IAM, daily scheduling, and an +independent hash-head anchor. This store does not enable collection by itself. +""" + +from __future__ import annotations + +import json +import os +import re +import sqlite3 +import stat +from dataclasses import asdict, dataclass +from datetime import datetime, timedelta, timezone +from math import isfinite +from pathlib import Path +from threading import Lock +from typing import Callable + +from .ledger_anchor import LedgerHead, capture_head +from .trace import DecisionTrace +from .trace_ledger import LedgerRecord, _record_hash + + +RETENTION_DAYS = 30 +_TOKEN = re.compile(r"[A-Za-z0-9_.:-]{1,128}\Z") +_HASH = re.compile(r"[0-9a-f]{64}\Z") +_GENESIS = "0" * 64 + + +@dataclass(frozen=True, slots=True) +class SourceGrant: + source_id: str + owner: str + task_ids: frozenset[str] + rights_attested: bool + + def __post_init__(self) -> None: + if not self.rights_attested or not self.task_ids: + raise ValueError("source rights and task inventory must be attested") + for value in (self.source_id, self.owner, *self.task_ids): + if not _TOKEN.fullmatch(value): + raise ValueError("source registry identifiers must be bounded tokens") + + +def _validate_trace(trace: DecisionTrace) -> None: + identifiers = ( + trace.run_id, + trace.access_level, + trace.model_provider, + trace.authority_result, + *trace.candidate_ids, + *trace.artifact_refs, + ) + if trace.selected_action_id is not None: + identifiers += (trace.selected_action_id,) + if not _HASH.fullmatch(trace.state_hash) or any( + not _TOKEN.fullmatch(value) for value in identifiers + ): + raise ValueError("trace failed redaction schema") + # Artifact references are not yet admitted: the policy requires a separate + # opaque-reference registry and leakage review before that field is enabled. + if trace.artifact_refs: + raise ValueError("trace failed redaction schema") + for key, values in trace.probabilities.items(): + if not _TOKEN.fullmatch(key) or not values or any( + not isfinite(value) or not 0 <= value <= 1 for value in values + ): + raise ValueError("trace failed redaction schema") + for mapping in (trace.utility_quantiles, trace.system_cost): + if any(not _TOKEN.fullmatch(key) or not isfinite(value) for key, value in mapping.items()): + raise ValueError("trace failed redaction schema") + if any(value < 0 for value in trace.system_cost.values()): + raise ValueError("trace failed redaction schema") + for key, value in trace.task_outcome.items(): + if not _TOKEN.fullmatch(key): + raise ValueError("trace failed redaction schema") + if isinstance(value, str): + if not _TOKEN.fullmatch(value): + raise ValueError("trace failed redaction schema") + elif isinstance(value, bool): + pass + elif not isinstance(value, (int, float)) or not isfinite(value): + raise ValueError("trace failed redaction schema") + + +class GovernedTraceStore: + """SQLite source-gated traces with a prefix checkpoint for 30-day purge. + + `purge_expired` must be run by the deployment's daily scheduler, and its + result independently audited. Local deletion cannot erase external backups. + """ + + def __init__( + self, + path: Path, + *, + grants: tuple[SourceGrant, ...], + clock: Callable[[], datetime] = lambda: datetime.now(timezone.utc), + ) -> None: + if not path.parent.is_dir() or path.is_symlink(): + raise ValueError("store parent must exist and path must not be a symlink") + if not grants or len({grant.source_id for grant in grants}) != len(grants): + raise ValueError("unique approved sources are required") + if not path.exists(): + descriptor = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600) + os.close(descriptor) + if os.name == "posix" and stat.S_IMODE(path.stat().st_mode) & 0o077: + raise ValueError("store file must not be accessible to group or other users") + self._clock = clock + self._grants = {grant.source_id: grant for grant in grants} + self._lock = Lock() + self._db = sqlite3.connect(path, timeout=30, isolation_level=None, check_same_thread=False) + self._db.execute("PRAGMA journal_mode=DELETE") + self._db.execute("PRAGMA synchronous=FULL") + self._db.execute("PRAGMA secure_delete=ON") + self._db.execute( + "CREATE TABLE IF NOT EXISTS metadata (" + "id INTEGER PRIMARY KEY CHECK(id=1), checkpoint_hash TEXT NOT NULL, " + "last_collected_at TEXT NOT NULL)" + ) + self._db.execute( + "INSERT OR IGNORE INTO metadata (id, checkpoint_hash, last_collected_at) " + "VALUES (1, ?, '')", (_GENESIS,) + ) + self._db.execute( + "CREATE TABLE IF NOT EXISTS records (" + "sequence INTEGER PRIMARY KEY AUTOINCREMENT, run_id TEXT NOT NULL UNIQUE, " + "collected_at TEXT NOT NULL, previous_hash TEXT NOT NULL, " + "record_hash TEXT NOT NULL, canonical_json TEXT NOT NULL)" + ) + self._db.execute( + "CREATE TABLE IF NOT EXISTS purge_audit (" + "sequence INTEGER PRIMARY KEY AUTOINCREMENT, purged_at TEXT NOT NULL, " + "record_count INTEGER NOT NULL, checkpoint_hash TEXT NOT NULL)" + ) + if not self.verify(): + self._db.close() + raise ValueError("governed trace hash chain is invalid") + + def __enter__(self) -> GovernedTraceStore: + return self + + def __exit__(self, *_args: object) -> None: + self.close() + + def _now(self) -> datetime: + value = self._clock() + if value.tzinfo is None or value.utcoffset() != timedelta(0): + raise ValueError("trace clock must return UTC") + return value + + def records(self) -> tuple[LedgerRecord, ...]: + with self._lock: + rows = self._db.execute( + "SELECT previous_hash, record_hash, canonical_json " + "FROM records ORDER BY sequence" + ).fetchall() + return tuple(LedgerRecord(*row) for row in rows) + + def _verified_head(self) -> tuple[int, str] | None: + with self._lock: + checkpoint, last_collected_at = self._db.execute( + "SELECT checkpoint_hash, last_collected_at FROM metadata WHERE id=1" + ).fetchone() + rows = self._db.execute( + "SELECT sequence, run_id, collected_at, previous_hash, " + "record_hash, canonical_json FROM records ORDER BY sequence" + ).fetchall() + sequence_row = self._db.execute( + "SELECT seq FROM sqlite_sequence WHERE name='records'" + ).fetchone() + sequence = int(sequence_row[0]) if sequence_row is not None else 0 + if rows and rows[-1][2] != last_collected_at: + return None + if rows and rows[-1][0] != sequence: + return None + previous = checkpoint + for _, run_id, collected_at, prior, digest, payload in rows: + if prior != previous: + return None + try: + decoded = json.loads(payload) + canonical = json.dumps(decoded, sort_keys=True, separators=(",", ":"), + ensure_ascii=False, allow_nan=False) + if decoded["trace"]["run_id"] != run_id or decoded["collected_at"] != collected_at: + return None + except (ValueError, TypeError, KeyError): + return None + if canonical != payload or _record_hash(prior, payload) != digest: + return None + previous = digest + return sequence, previous + + def verify(self) -> bool: + return self._verified_head() is not None + + def snapshot_head(self, *, policy_version: str) -> LedgerHead: + """Capture a verified head for signing outside the trace-writer identity. + + The caller must send this to a separately controlled signer and durable + destination. Capturing a head locally is not independent anchoring. + """ + state = self._verified_head() + if state is None: + raise ValueError("governed trace hash chain is invalid") + sequence, record_hash = state + return capture_head( + sequence=sequence, record_hash=record_hash, + policy_version=policy_version, clock=self._clock, + ) + + def append(self, trace: DecisionTrace, *, source_id: str, task_id: str) -> LedgerRecord: + grant = self._grants.get(source_id) + if grant is None or task_id not in grant.task_ids: + raise ValueError("unapproved source or task") + _validate_trace(trace) + now = self._now() + collected_at = now.isoformat(timespec="microseconds") + retention_cutoff = (now - timedelta(days=RETENTION_DAYS)).isoformat( + timespec="microseconds" + ) + payload = json.dumps( + {"collected_at": collected_at, "source_id": source_id, "task_id": task_id, + "trace": asdict(trace)}, + sort_keys=True, separators=(",", ":"), ensure_ascii=False, allow_nan=False, + ) + with self._lock: + self._db.execute("BEGIN IMMEDIATE") + try: + checkpoint, last_collected_at = self._db.execute( + "SELECT checkpoint_hash, last_collected_at FROM metadata WHERE id=1" + ).fetchone() + if collected_at < last_collected_at: + raise ValueError("trace clock moved backwards") + overdue = self._db.execute( + "SELECT 1 FROM records WHERE collected_at <= ? LIMIT 1", + (retention_cutoff,), + ).fetchone() + if overdue is not None: + raise ValueError("retention purge overdue; new collection is disabled") + row = self._db.execute( + "SELECT record_hash FROM records ORDER BY sequence DESC LIMIT 1" + ).fetchone() + previous = row[0] if row is not None else checkpoint + record = LedgerRecord(previous, _record_hash(previous, payload), payload) + self._db.execute( + "INSERT INTO records (run_id, collected_at, previous_hash, " + "record_hash, canonical_json) VALUES (?, ?, ?, ?, ?)", + (trace.run_id, collected_at, previous, record.record_hash, payload), + ) + self._db.execute( + "UPDATE metadata SET last_collected_at=? WHERE id=1", (collected_at,) + ) + self._db.execute("COMMIT") + except Exception: + self._db.execute("ROLLBACK") + raise + return record + + def purge_expired(self) -> int: + now = self._now() + cutoff = (now - timedelta(days=RETENTION_DAYS)).isoformat(timespec="microseconds") + with self._lock: + self._db.execute("BEGIN IMMEDIATE") + try: + expired = self._db.execute( + "SELECT sequence, record_hash FROM records " + "WHERE collected_at <= ? ORDER BY sequence", (cutoff,) + ).fetchall() + if not expired: + self._db.execute("COMMIT") + return 0 + last_sequence, checkpoint = expired[-1] + self._db.execute("DELETE FROM records WHERE sequence <= ?", (last_sequence,)) + self._db.execute("UPDATE metadata SET checkpoint_hash=? WHERE id=1", (checkpoint,)) + self._db.execute( + "INSERT INTO purge_audit (purged_at, record_count, checkpoint_hash) " + "VALUES (?, ?, ?)", (now.isoformat(timespec="microseconds"), len(expired), + checkpoint), + ) + self._db.execute("COMMIT") + except Exception: + self._db.execute("ROLLBACK") + raise + self._db.execute("VACUUM") + return len(expired) + + def close(self) -> None: + with self._lock: + self._db.close() + + +class BoundGovernedTraceSink: + """Bind one approved task in trusted host code to the controller's trace API. + + A shared HTTP controller must not reuse this binding across unrelated tasks. + The host, never the caller payload, chooses the source and task identifiers. + """ + + def __init__(self, store: GovernedTraceStore, *, source_id: str, task_id: str) -> None: + grant = store._grants.get(source_id) + if grant is None or task_id not in grant.task_ids: + raise ValueError("unapproved source or task") + self._store = store + self._source_id = source_id + self._task_id = task_id + + def append(self, trace: DecisionTrace) -> LedgerRecord: + return self._store.append(trace, source_id=self._source_id, task_id=self._task_id) + diff --git a/c3r/telemetry/ledger_anchor.py b/c3r/telemetry/ledger_anchor.py new file mode 100644 index 0000000..2ac6296 --- /dev/null +++ b/c3r/telemetry/ledger_anchor.py @@ -0,0 +1,81 @@ +"""Portable signed ledger-head envelopes; deployment must supply independent signing. + +This module never holds a private key. A production signer and anchor destination +must be controlled separately from the trace writer (for example, Cloud KMS and +an access-separated evidence project). Local success is not deployment proof. +""" + +from __future__ import annotations + +import json +import re +from dataclasses import asdict, dataclass +from datetime import datetime, timezone +from typing import Callable + + +_HASH = re.compile(r"[0-9a-f]{64}\Z") + + +@dataclass(frozen=True, slots=True) +class LedgerHead: + sequence: int + record_hash: str + captured_at: str + policy_version: str + + def __post_init__(self) -> None: + if self.sequence < 0 or not _HASH.fullmatch(self.record_hash): + raise ValueError("invalid ledger head") + if not self.policy_version or len(self.policy_version) > 64: + raise ValueError("invalid policy version") + try: + instant = datetime.fromisoformat(self.captured_at) + except ValueError as error: + raise ValueError("invalid capture timestamp") from error + if instant.tzinfo is None or instant.utcoffset().total_seconds() != 0: + raise ValueError("capture timestamp must be UTC") + + def payload(self) -> bytes: + return json.dumps( + asdict(self), sort_keys=True, separators=(",", ":"), ensure_ascii=False + ).encode("utf-8") + + +@dataclass(frozen=True, slots=True) +class SignedLedgerAnchor: + head: LedgerHead + key_id: str + signature_hex: str + + +def capture_head( + *, sequence: int, record_hash: str, policy_version: str, + clock: Callable[[], datetime] = lambda: datetime.now(timezone.utc), +) -> LedgerHead: + return LedgerHead(sequence, record_hash, clock().isoformat(), policy_version) + + +def sign_head( + head: LedgerHead, *, key_id: str, signer: Callable[[bytes], bytes] +) -> SignedLedgerAnchor: + if not key_id or len(key_id) > 256: + raise ValueError("invalid signer key identifier") + signature = signer(head.payload()) + if not signature: + raise ValueError("empty signature") + return SignedLedgerAnchor(head, key_id, signature.hex()) + + +def verify_anchor( + anchor: SignedLedgerAnchor, *, sequence: int, record_hash: str, + verifier: Callable[[str, bytes, bytes], bool], +) -> bool: + if anchor.head.sequence != sequence or anchor.head.record_hash != record_hash: + return False + try: + signature = bytes.fromhex(anchor.signature_hex) + except ValueError: + return False + return bool(signature) and verifier(anchor.key_id, anchor.head.payload(), signature) + diff --git a/c3r/telemetry/sqlite_ledger.py b/c3r/telemetry/sqlite_ledger.py new file mode 100644 index 0000000..5e171c0 --- /dev/null +++ b/c3r/telemetry/sqlite_ledger.py @@ -0,0 +1,76 @@ +"""Transactional single-host trace ledger with restart verification.""" + +from __future__ import annotations + +import sqlite3 +import os +import stat +from pathlib import Path +from threading import Lock + +from .trace import DecisionTrace +from .trace_ledger import LedgerRecord, TraceLedger, _record_hash, canonical_trace_json + + +class SqliteTraceLedger: + """Persist redacted traces atomically in a host-owned SQLite database. + + The operator must place the database on durable, access-controlled storage and + back it up. This is a single-host ledger, not an independently anchored audit log. + """ + + def __init__(self, path: Path) -> None: + if not path.parent.is_dir() or path.is_symlink(): + raise ValueError("ledger parent must exist and path must not be a symlink") + if not path.exists(): + descriptor = os.open(path, os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600) + os.close(descriptor) + if os.name == "posix" and stat.S_IMODE(path.stat().st_mode) & 0o077: + raise ValueError("ledger file must not be accessible to group or other users") + self._lock = Lock() + self._db = sqlite3.connect(path, timeout=30, isolation_level=None, check_same_thread=False) + self._db.execute("PRAGMA journal_mode=WAL") + self._db.execute("PRAGMA synchronous=FULL") + self._db.execute( + "CREATE TABLE IF NOT EXISTS records (" + "sequence INTEGER PRIMARY KEY AUTOINCREMENT, " + "previous_hash TEXT NOT NULL, record_hash TEXT NOT NULL, " + "canonical_json TEXT NOT NULL)" + ) + if not TraceLedger.verify(self.records): + self._db.close() + raise ValueError("persisted trace ledger hash chain is invalid") + + @property + def records(self) -> tuple[LedgerRecord, ...]: + with self._lock: + rows = self._db.execute( + "SELECT previous_hash, record_hash, canonical_json " + "FROM records ORDER BY sequence" + ).fetchall() + return tuple(LedgerRecord(*row) for row in rows) + + def append(self, trace: DecisionTrace) -> LedgerRecord: + canonical = canonical_trace_json(trace) + with self._lock: + self._db.execute("BEGIN IMMEDIATE") + try: + row = self._db.execute( + "SELECT record_hash FROM records ORDER BY sequence DESC LIMIT 1" + ).fetchone() + previous = row[0] if row is not None else "0" * 64 + record = LedgerRecord(previous, _record_hash(previous, canonical), canonical) + self._db.execute( + "INSERT INTO records (previous_hash, record_hash, canonical_json) " + "VALUES (?, ?, ?)", + (record.previous_hash, record.record_hash, record.canonical_json), + ) + self._db.execute("COMMIT") + except Exception: + self._db.execute("ROLLBACK") + raise + return record + + def close(self) -> None: + with self._lock: + self._db.close() diff --git a/c3r/telemetry/trace_ledger.py b/c3r/telemetry/trace_ledger.py index 765e07b..816c8f2 100644 --- a/c3r/telemetry/trace_ledger.py +++ b/c3r/telemetry/trace_ledger.py @@ -5,6 +5,7 @@ import hashlib import json from dataclasses import asdict, dataclass +from threading import Lock from .trace import DecisionTrace @@ -24,33 +25,40 @@ def _record_hash(previous_hash: str, canonical_json: str) -> str: ).hexdigest() +def canonical_trace_json(trace: DecisionTrace) -> str: + return json.dumps( + asdict(trace), + sort_keys=True, + separators=(",", ":"), + ensure_ascii=False, + allow_nan=False, + ) + + class TraceLedger: def __init__(self, records: tuple[LedgerRecord, ...] = ()) -> None: if records and not self.verify(records): raise ValueError("trace ledger hash chain is invalid") self._records = list(records) + self._lock = Lock() @property def records(self) -> tuple[LedgerRecord, ...]: - return tuple(self._records) + with self._lock: + return tuple(self._records) def append(self, trace: DecisionTrace) -> LedgerRecord: - canonical = json.dumps( - asdict(trace), - sort_keys=True, - separators=(",", ":"), - ensure_ascii=False, - allow_nan=False, - ) - previous = self._records[-1].record_hash if self._records else _GENESIS_HASH - record = LedgerRecord(previous, _record_hash(previous, canonical), canonical) - self._records.append(record) - return record + canonical = canonical_trace_json(trace) + with self._lock: + previous = self._records[-1].record_hash if self._records else _GENESIS_HASH + record = LedgerRecord(previous, _record_hash(previous, canonical), canonical) + self._records.append(record) + return record def to_jsonl(self) -> str: return "\n".join( json.dumps(asdict(record), sort_keys=True, separators=(",", ":")) - for record in self._records + for record in self.records ) @classmethod diff --git a/c3r/verifier_firewall.py b/c3r/verifier_firewall.py index 858232d..decde7b 100644 --- a/c3r/verifier_firewall.py +++ b/c3r/verifier_firewall.py @@ -40,6 +40,11 @@ def __init__( self._policy = policy self._attestation_key = attestation_key + @property + def available_verifier_ids(self) -> frozenset[str]: + """Identifiers configured by the trusted host, never by a proposal.""" + return frozenset(self._verifiers) + def verify(self, candidate: ActionCandidate) -> VerificationResult: verifier_id = self._policy.select(candidate) try: diff --git a/cloudbuild.qualify.yaml b/cloudbuild.qualify.yaml new file mode 100644 index 0000000..f9ff501 --- /dev/null +++ b/cloudbuild.qualify.yaml @@ -0,0 +1,111 @@ +# No local source upload: fetch and verify the exact public commit/tree. +# This is build/scan qualification, NOT GPU startup or release approval. +substitutions: + _SOURCE_COMMIT: '' + _SOURCE_TREE: '' +steps: + - name: gcr.io/cloud-builders/git + id: exact-source + entrypoint: sh + args: + - -c + - >- + test "$${#COMMIT}" = 40 + && case "$$COMMIT" in *[!0-9a-f]*) exit 1;; esac + && git init /workspace/source + && git -C /workspace/source remote add origin https://github.com/ColomboAI-com/c3r.git + && git -C /workspace/source fetch --depth=1 origin "$$COMMIT" + && git -C /workspace/source checkout --detach FETCH_HEAD + && test "$$(git -C /workspace/source rev-parse HEAD)" = "$$COMMIT" + && test "$$(git -C /workspace/source rev-parse 'HEAD^{tree}')" = "$$TREE" + env: ['COMMIT=${_SOURCE_COMMIT}', 'TREE=${_SOURCE_TREE}'] + - name: python:3.11-alpine3.24 + id: unit-311 + dir: source + waitFor: [exact-source] + entrypoint: python + args: [-m, unittest, discover, -s, tests, -v] + - name: python:3.12-alpine3.23 + id: unit-312 + dir: source + waitFor: [exact-source] + entrypoint: python + args: [-m, unittest, discover, -s, tests, -v] + - name: python:3.13-alpine3.23 + id: unit-313 + dir: source + waitFor: [exact-source] + entrypoint: python + args: [-m, unittest, discover, -s, tests, -v] + # Both SDKs execute in one disposable CPU-only test container. No GPU VM + # packages or services are changed by these test-environment installs. + - name: node:24-alpine + id: official-sdk-acceptance + dir: source + waitFor: [exact-source] + entrypoint: sh + args: + - -c + - >- + apk add --no-cache python3 py3-pip + && python3 -m venv /workspace/sdk-python + && /workspace/sdk-python/bin/python -m pip install 'openai==3.24.0' + && npm install --prefix /workspace/sdk-js 'openai@7.27.0' --ignore-scripts --no-audit --no-fund + && /workspace/sdk-python/bin/python -m unittest tests.test_official_sdk_acceptance -v + env: ['C3R_JS_SDK_MODULE=/workspace/sdk-js/node_modules/openai/index.mjs'] + - name: gcr.io/cloud-builders/git + id: clm-source + dir: source + waitFor: [exact-source] + entrypoint: sh + args: + - -c + - >- + git clone --no-checkout https://github.com/Contrastive-LM/CLM.git /workspace/clm-upstream + && test "$$(git -C /workspace/clm-upstream rev-parse bb42c6c5bf914fd449bed2f6ca65be80602cb1f7^{commit})" = bb42c6c5bf914fd449bed2f6ca65be80602cb1f7 + && git -C /workspace/clm-upstream archive --format=tar + --output=/workspace/source/deploy/clm-source-bb42c6c.tar bb42c6c5bf914fd449bed2f6ca65be80602cb1f7 + - name: gcr.io/cloud-builders/docker + id: api-image + dir: source + waitFor: [unit-311, unit-312, unit-313, official-sdk-acceptance] + args: [build, --pull, --file, Dockerfile, --tag, 'c3r-api:qualification', .] + - name: gcr.io/cloud-builders/docker + id: clm-image + dir: source + waitFor: [clm-source] + args: [build, --pull, --file, deploy/Dockerfile.clm-hardened, --tag, 'c3r-clm:qualification', deploy] + - name: gcr.io/cloud-builders/docker + id: deepseek-image + dir: source + waitFor: [exact-source] + args: [build, --pull, --file, deploy/deepseek-v41/Dockerfile, --tag, 'c3r-deepseek:qualification', deploy/deepseek-v41] + - name: python:3.13-alpine3.23 + id: serving-config-validation + dir: source + waitFor: [exact-source] + entrypoint: python + args: [deploy/deepseek-v41/launch.py, --model-path, /private/model, --cache-path, /private/cache, --server-args] + - name: gcr.io/cloud-builders/docker + id: api-import + waitFor: [api-image] + args: [run, --rm, 'c3r-api:qualification', python, -c, 'import c3r.serve'] + - name: gcr.io/cloud-builders/docker + id: clm-import + waitFor: [clm-image] + args: [run, --rm, --entrypoint, python3, 'c3r-clm:qualification', -c, 'from clm.engine import Engine; from clm.server import create_app'] + - name: aquasec/trivy:0.74.0 + id: api-security + waitFor: [api-image] + args: [image, --no-progress, --exit-code, '1', --severity, 'HIGH,CRITICAL', 'c3r-api:qualification'] + - name: aquasec/trivy:0.74.0 + id: clm-security + waitFor: [clm-image, api-security] + args: [image, --no-progress, --exit-code, '1', --severity, 'HIGH,CRITICAL', 'c3r-clm:qualification'] + - name: aquasec/trivy:0.74.0 + id: deepseek-security + waitFor: [deepseek-image, clm-security] + args: [image, --no-progress, --exit-code, '1', --severity, 'HIGH,CRITICAL', 'c3r-deepseek:qualification'] +options: + diskSizeGb: 200 +timeout: 3600s diff --git a/cloudbuild.retention.yaml b/cloudbuild.retention.yaml new file mode 100644 index 0000000..3b2eea3 --- /dev/null +++ b/cloudbuild.retention.yaml @@ -0,0 +1,18 @@ +# Supply _IMAGE as a dedicated, private Artifact Registry destination. +# The source upload is restricted by .gcloudignore; this build never receives +# traces, papers, local evidence, secrets, or Git metadata. +steps: + - name: gcr.io/cloud-builders/docker + args: ["build", "--file", "Dockerfile.retention", "--tag", "${_IMAGE}", "."] + - name: gcr.io/cloud-builders/docker + entrypoint: sh + args: + - -c + - >- + test "$(docker image inspect ${_IMAGE} --format '{{json .Config.Cmd}}')" + = '["python","-m","c3r.retention_job"]' + - name: gcr.io/cloud-builders/docker + args: ["run", "--rm", "${_IMAGE}", "python", "-c", "import c3r.retention_job"] + - name: aquasec/trivy:0.74.0 + args: ["image", "--no-progress", "--exit-code", "1", "--severity", "HIGH,CRITICAL", "${_IMAGE}"] +images: ["${_IMAGE}"] diff --git a/cloudbuild.verify.yaml b/cloudbuild.verify.yaml new file mode 100644 index 0000000..18bf044 --- /dev/null +++ b/cloudbuild.verify.yaml @@ -0,0 +1,42 @@ +# GitHub-independent verification. Submit from the repository root with: +# gcloud builds submit . --config=cloudbuild.verify.yaml --ignore-file=.gcloudignore.verify --project=APPROVED_BUILD_PROJECT --region=us-central1 +# The uploaded context contains only public runtime source and synthetic tests. +steps: + - name: python:3.11-alpine3.24 + id: unit-311 + entrypoint: python + args: ["-m", "unittest", "discover", "-s", "tests", "-v"] + - name: python:3.12-alpine3.23 + id: unit-312 + entrypoint: python + args: ["-m", "unittest", "discover", "-s", "tests", "-v"] + - name: python:3.13-alpine3.23 + id: unit-313 + entrypoint: python + args: ["-m", "unittest", "discover", "-s", "tests", "-v"] + - name: gcr.io/cloud-builders/docker + id: runtime-image + args: ["build", "--pull", "--file", "Dockerfile", "--tag", "c3r-runtime:verify", "."] + - name: gcr.io/cloud-builders/docker + id: retention-image + args: ["build", "--pull", "--file", "Dockerfile.retention", "--tag", "c3r-retention:verify", "."] + - name: gcr.io/cloud-builders/docker + id: retention-command + entrypoint: sh + args: + - -c + - >- + test "$(docker image inspect c3r-retention:verify --format '{{json .Config.Cmd}}')" + = '["python","-m","c3r.retention_job"]' + - name: gcr.io/cloud-builders/docker + id: runtime-import + args: ["run", "--rm", "c3r-runtime:verify", "python", "-c", "import c3r.serve"] + - name: gcr.io/cloud-builders/docker + id: retention-import + args: ["run", "--rm", "c3r-retention:verify", "python", "-c", "import c3r.retention_job"] + - name: aquasec/trivy:0.74.0 + id: runtime-vulnerability-scan + args: ["image", "--no-progress", "--exit-code", "1", "--severity", "HIGH,CRITICAL", "c3r-runtime:verify"] + - name: aquasec/trivy:0.74.0 + id: retention-vulnerability-scan + args: ["image", "--no-progress", "--exit-code", "1", "--severity", "HIGH,CRITICAL", "c3r-retention:verify"] diff --git a/deploy/Dockerfile.clm b/deploy/Dockerfile.clm new file mode 100644 index 0000000..0efc930 --- /dev/null +++ b/deploy/Dockerfile.clm @@ -0,0 +1,7 @@ +FROM vllm/vllm-openai@sha256:00d577a6a63281e15336029d5bcee4e9a2cf182214a4f20ba6111b1c8e79893d +ADD clm-source-bb42c6c.tar /opt/clm/ +COPY clm-source-revision /opt/clm/source-revision +RUN python3 -m pip install --no-deps --no-build-isolation /opt/clm +COPY clm_runtime.py /opt/c3r/clm_runtime.py +COPY encoder_identity.py /opt/c3r/encoder_identity.py +ENTRYPOINT ["python3", "/opt/c3r/clm_runtime.py"] diff --git a/deploy/Dockerfile.clm-hardened b/deploy/Dockerfile.clm-hardened new file mode 100644 index 0000000..6c67f58 --- /dev/null +++ b/deploy/Dockerfile.clm-hardened @@ -0,0 +1,18 @@ +# CPU projection-head service only; the Qwen encoder is a separate process. +FROM vllm/vllm-openai@sha256:00d577a6a63281e15336029d5bcee4e9a2cf182214a4f20ba6111b1c8e79893d +USER root +RUN apt-get update && apt-get upgrade -y \ + && apt-get purge -y linux-libc-dev \ + && rm -rf /var/lib/apt/lists/* \ + && python3 -m pip uninstall --break-system-packages -y vllm \ + && python3 -m pip install --break-system-packages --no-cache-dir --no-deps \ + PyJWT==2.14.0 msgpack==1.2.1 urllib3==2.8.0 setuptools==78.1.1 +ADD clm-source-bb42c6c.tar /opt/clm/ +COPY clm-source-revision /opt/clm/source-revision +RUN python3 -m pip install --break-system-packages --no-deps --no-build-isolation /opt/clm +RUN python3 -m pip uninstall --break-system-packages -y setuptools wheel cryptography +COPY clm_runtime.py encoder_identity.py /opt/c3r/ +RUN python3 -c "from clm.engine import Engine; from clm.server import create_app" +RUN python3 -m pip uninstall --break-system-packages -y pip +USER 10001:10001 +ENTRYPOINT ["python3", "/opt/c3r/clm_runtime.py"] diff --git a/deploy/attest_encoder.py b/deploy/attest_encoder.py new file mode 100644 index 0000000..099cbe9 --- /dev/null +++ b/deploy/attest_encoder.py @@ -0,0 +1,40 @@ +"""Run as the deployment operator; inspect the encoder's actual mounted files. + +Output belongs in private deployment evidence. This is a trusted host readback, +not hardware-backed remote attestation. No Docker socket is given to CLM. +""" +import json +import subprocess +from pathlib import Path + +from encoder_identity import REVISION, verify_encoder + +IMAGE = "sha256:00d577a6a63281e15336029d5bcee4e9a2cf182214a4f20ba6111b1c8e79893d" + + +def main(): + info = json.loads(subprocess.check_output([ + "docker", "inspect", "c3r-qwen-encoder"], text=True))[0] + image = json.loads(subprocess.check_output([ + "docker", "image", "inspect", info["Image"]], text=True))[0] + if not any(value.endswith("@" + IMAGE) for value in image["RepoDigests"]): + raise RuntimeError("encoder container image is not the approved immutable digest") + args = info["Config"]["Cmd"] + if (not info["State"]["Running"] or "/encoder" not in args + or "pooling" not in args or "qwen3-8b" not in args): + raise RuntimeError("encoder process configuration mismatch") + mounts = [mount for mount in info["Mounts"] if mount["Destination"] == "/encoder"] + if len(mounts) != 1 or mounts[0]["RW"]: + raise RuntimeError("encoder artifact mount must be read-only") + # This namespace path proves which files the running encoder sees, rather + # than verifying a separate copy downloaded beside it. + verify_encoder(Path(f'/proc/{info["State"]["Pid"]}/root/encoder')) + print(json.dumps({ + "encoder_revision": REVISION, "encoder_container_digest": IMAGE, + "encoder_container_id": info["Id"], "encoder_content_verified": True, + "binding_basis": "host_docker_inspect_and_running_process_mount_hashes", + }, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/deploy/clm-source-revision b/deploy/clm-source-revision new file mode 100644 index 0000000..3ad91a5 --- /dev/null +++ b/deploy/clm-source-revision @@ -0,0 +1 @@ +bb42c6c5bf914fd449bed2f6ca65be80602cb1f7 diff --git a/deploy/clm_runtime.py b/deploy/clm_runtime.py new file mode 100644 index 0000000..1a719bc --- /dev/null +++ b/deploy/clm_runtime.py @@ -0,0 +1,75 @@ +"""Loopback CLM service with measured artifact identity and no prompt caches.""" +import hashlib +import json +import os +from pathlib import Path + +import requests +import torch +import uvicorn +from clm.embedder import Embedder +from clm.engine import Engine +from clm.server import create_app +from encoder_identity import REVISION, verify_encoder + +SOURCE = "bb42c6c5bf914fd449bed2f6ca65be80602cb1f7" +HEAD_SHA256 = "b2b4a8c9c2d39263eff78a351eb909a342ce9b3bf21a3f07c1d1bf15f1c4eda5" +ROOT = Path("/artifacts") + + +def digest(path): + hasher = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(4 * 1024 * 1024), b""): + hasher.update(chunk) + return hasher.hexdigest() + + +class LocalSession(requests.Session): + def __init__(self): + super().__init__() + self.trust_env = False + + def request(self, method, url, **kwargs): + if not url.startswith("http://127.0.0.1:8090/"): + raise ValueError("encoder origin must stay loopback") + kwargs["allow_redirects"] = False + return super().request(method, url, **kwargs) + + +def main(): + manifest = json.loads((ROOT / "manifest.json").read_text()) + actual_source = Path("/opt/clm/source-revision").read_text().strip() + if actual_source != SOURCE or manifest["clm_source_revision"] != SOURCE: + raise RuntimeError("CLM source mismatch") + head = ROOT / "head/CLM_v0.1-8B.pt" + actual_head = digest(head) + if actual_head != HEAD_SHA256 or manifest["head_sha256"] != actual_head: + raise RuntimeError("CLM head mismatch") + if manifest["encoder_revision"] != REVISION: + raise RuntimeError("encoder revision mismatch") + verify_encoder(ROOT / "encoder") + embedder = Embedder(max_tokens=8192, cache_size=0, batch=8, timeout=10) + embedder.session = LocalSession() + engine = Engine(embedder=embedder, checkpoint=str(head), device="cpu", action_cache=0) + app = create_app(engine, ui=False) + artifact = {key: value for key, value in manifest.items() if key != "encoder_files"} + artifact.update({ + "container_digest": os.environ.get("C3R_CLM_CONTAINER_DIGEST", "unattested"), + "container_digest_source": "deployment_host_readback", + "head_requires_vllm": False, "torch_version": torch.__version__, + "cuda_version": torch.version.cuda, "embedding_cache_size": 0, + "action_cache_enabled": False, "head_device": "cpu", + "encoder_content_verified": True, + "encoder_identity_basis": "immutable_upstream_git_blobs_and_lfs_sha256", + }) + + @app.get("/internal/clm/artifact") + def loaded_artifact(): + return artifact + + uvicorn.run(app, host="127.0.0.1", port=8700, access_log=False, log_level="warning") + + +if __name__ == "__main__": + main() diff --git a/deploy/deepseek-v41/Dockerfile b/deploy/deepseek-v41/Dockerfile new file mode 100644 index 0000000..2f22e96 --- /dev/null +++ b/deploy/deepseek-v41/Dockerfile @@ -0,0 +1,30 @@ +# Qualification candidate only; image scan and GPU qualification are separate gates. +FROM vllm/vllm-openai@sha256:98adb2311118263a453ec0acd5d4b72bda3db26cecc0fe8ffc94b9ab03c951f9 +USER root +LABEL org.opencontainers.image.source="https://github.com/vllm-project/vllm" \ + org.opencontainers.image.revision="ac68c3087215e0a4f3cdfa218508c6aada57235d" \ + com.colomboai.c3r.qualification="pending" +RUN apt-get update && apt-get upgrade -y \ + && apt-get install -y --no-install-recommends libc6-dev python3.12-dev \ + && apt-get purge -y python3-msgpack python3-setuptools python3-urllib3 \ + && rm -rf /var/lib/apt/lists/* \ + && python3 -m pip uninstall --break-system-packages -y opentelemetry-exporter-otlp-common +COPY check_upstream_dependencies.py /opt/c3r/check_upstream_dependencies.py +COPY compiler_probe.c /opt/c3r/compiler_probe.c +# Triton compiles CUDA helpers at runtime. Removing development headers can make +# a vulnerability scan clean while breaking the actual model startup. +RUN gcc -fsyntax-only -I/usr/include/python3.12 \ + -I/usr/local/lib/python3.12/dist-packages/triton/backends/nvidia/include \ + /opt/c3r/compiler_probe.c +RUN python3 /opt/c3r/check_upstream_dependencies.py \ + && python3 -m pip uninstall --break-system-packages -y pip +ENV HOME=/runtime \ + HF_HUB_OFFLINE=1 \ + HF_HUB_DISABLE_TELEMETRY=1 \ + VLLM_NO_USAGE_STATS=1 \ + DO_NOT_TRACK=1 \ + VLLM_USE_V2_MODEL_RUNNER=1 \ + PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True \ + VLLM_ENGINE_READY_TIMEOUT_S=3600 +USER 1003:1004 +ENTRYPOINT ["python3", "-m", "vllm.entrypoints.openai.api_server"] diff --git a/deploy/deepseek-v41/README.md b/deploy/deepseek-v41/README.md new file mode 100644 index 0000000..75b725a --- /dev/null +++ b/deploy/deepseek-v41/README.md @@ -0,0 +1,65 @@ +# Same-node DeepSeek qualification target + +The target is self-hosted Qwen3-8B/CLM System One, DeepSeek V4.1 Flash +System Two, and C3R Core on the existing eight-H100 node. No new GPU, +OpenRouter, or hosted inference provider is part of this deployment. + +`h100-production.env` is a **qualification target**, not a production approval. +The currently pinned candidate failed Triton runtime compilation because its +hardening removed required C headers. **Do not execute that image.** The repaired +Dockerfile preserves patched development headers and runs a compile-only probe; +its rebuilt image must independently pass security scans and receive a new pin +before another supervised GPU load. A clean scanner result alone is insufficient. +It pins the locally built immutable candidate image, upstream source/base image, +checkpoint revision, TP8, 85% reservation, Engram CPU offload, 32K context, +two sequences, batch limit, and both parsers. Registry publication of that local +image remains a release gate. Execution verifies every checkpoint file against +the tracked upstream identity manifest and rejects extra runtime files. Only +Hugging Face download metadata under `.cache/huggingface` is exempted from +extra-file rejection. Generation defaults are explicitly `vllm`, not a hidden +local generation configuration. Neither a revision string nor a directory name +alone proves checkpoint integrity. + +## Startup flag correction + +| Item | Recorded value | +| --- | --- | +| Old flag | `--disable-log-requests` | +| Replacement | `--no-enable-log-requests` | +| Reason | The pinned newer vLLM parser rejects the old flag; request logging must stay off. | +| vLLM source | `ac68c3087215e0a4f3cdfa218508c6aada57235d` | +| vLLM version | `0.30.1rc1.dev396+gac68c3087` | + +The earlier recovery used a **different older image**, at 85% reservation, +without this candidate's explicit Engram/tool-parser configuration. Its 13 +passing private API checks do not qualify this new image or its launch settings. + +```sh +# Dry-run by default; checkpoint/cache paths stay private and outside Git. +sh deploy/deepseek-v41/start.sh \ + --model-path /private/verified-checkpoint --cache-path /private/serving-cache +# --execute starts only this named candidate, on loopback:18000 by default. +# The supervised maintenance controller must first stop the incumbent and +# independently guarantee recovery; never overlap both TP8 model engines. +``` + +The complete DeepSeek launch command is generated by `launch.py`; no shell +history, secret, or untracked flag is required. Validate `--server-args` through +the pinned runtime's actual parser before any model load. Help-only is not +argument validation. Cache must be owned by image UID 1003/GID 1004 and must +not contain request payloads. Startup does not enable research trace collection. + +## Release gates (all require actual evidence) + +Three clean startups with all 13 API checks; independently supervised service +restarts; dependency-order readiness; 15/30/60-minute sustained CLM, Responses +and mixed profiles; memory stability/no swap/OOM; bounded admission; measured +load-aware costs; reasoning/coding/tool/JSON/context acceptance; DeepSeek outage +and recovery; rollback in both directions; allocated infrastructure cost; exact +final-head builds/scans; private model ports plus dedicated TLS; external SDK +acceptance; staged canary; release-owner authorization; merge and immutable tag. + +Keep the original container/config/image and the verified 85% recovery config. +The original 92% reservation cannot restart beside resident Qwen; do not stop +Qwen merely to conceal that rollback limitation. No VM stop/reboot/delete is +permitted: the active checkpoint is on ephemeral Local SSD. diff --git a/deploy/deepseek-v41/check_upstream_dependencies.py b/deploy/deepseek-v41/check_upstream_dependencies.py new file mode 100644 index 0000000..eae9eb2 --- /dev/null +++ b/deploy/deepseek-v41/check_upstream_dependencies.py @@ -0,0 +1,25 @@ +"""Reject unknown dependency conflicts; disclose the pinned upstream NCCL override. + +The upstream Dockerfile explicitly overrides Torch's NCCL requirement because +DeepEPv2 requires NCCL >=2.30.4. This is not GPU compatibility evidence: +https://github.com/vllm-project/vllm/blob/ac68c3087215e0a4f3cdfa218508c6aada57235d/docker/Dockerfile +""" +import importlib.metadata +import json +import subprocess +import sys + +versions = {name: importlib.metadata.version(name) + for name in ("vllm", "torch", "nvidia-nccl-cu13")} +expected = {"vllm": "0.30.1rc1.dev396+gac68c3087", + "torch": "2.13.0+cu130", "nvidia-nccl-cu13": "2.30.7"} +if versions != expected: + raise RuntimeError("Pinned upstream dependency identities changed") +result = subprocess.run([sys.executable, "-m", "pip", "check"], + capture_output=True, text=True, check=False) +known = ('torch 2.13.0+cu130 has requirement nvidia-nccl-cu13==2.29.7; ' + 'platform_system == "Linux", but you have nvidia-nccl-cu13 2.30.7.') +if result.returncode != 1 or result.stdout.strip().splitlines() != [known] or result.stderr.strip(): + raise RuntimeError("Dependency preflight differs from the reviewed upstream override") +print(json.dumps({"known_upstream_nccl_override": known, + "other_dependency_errors": [], "tp8_compatibility": "not_yet_tested"})) diff --git a/deploy/deepseek-v41/checkpoint-manifest.json b/deploy/deepseek-v41/checkpoint-manifest.json new file mode 100644 index 0000000..8ba448d --- /dev/null +++ b/deploy/deepseek-v41/checkpoint-manifest.json @@ -0,0 +1,534 @@ +{ + "model_id": "deepseek-ai/DeepSeek-V4.1-Flash", + "revision": "dba1be0a40aa45a94ad051997016db3960a90277", + "files": [ + { + "path": ".gitattributes", + "size": 1701, + "git_blob_id": "9b59a7525c7b7870bd2cbdc41a4e91fd63fcb6cb", + "lfs_sha256": null + }, + { + "path": "DeepSeek_V41_Tech_Report.pdf", + "size": 1809802, + "git_blob_id": "9a327deea3393d204f789f261ceb4eb472ea910d", + "lfs_sha256": "ba68e2e40408125ae6d2f63a9a241b61c73910691c74ec1a2a7023c851eac08d" + }, + { + "path": "LICENSE", + "size": 1084, + "git_blob_id": "d62e3bef9f054f21b7fc616365850fbf879a99ff", + "lfs_sha256": null + }, + { + "path": "README.md", + "size": 13110, + "git_blob_id": "211b83857abd5e8420df3a4f10db1096294227e6", + "lfs_sha256": null + }, + { + "path": "assets/dsv41_agentic_performance.png", + "size": 190736, + "git_blob_id": "0f638f2420410c0b3b8a82d4438ca9396bc17390", + "lfs_sha256": "44deae01cb9ce756c7d622dcf2b55f4a332c25e5e0b9297997519f77e0a7cf52" + }, + { + "path": "assets/dsv41_kv_cache.png", + "size": 270933, + "git_blob_id": "39bba8d4e27e09fbc754139aa477e51bdb7184ac", + "lfs_sha256": "b61bf4651d4b163e02fb21d7298bf7b36b810c1da2300cb2d1e498c373793e4b" + }, + { + "path": "config.json", + "size": 3311, + "git_blob_id": "09917a9139b22d5bf8be52132787f435147b1020", + "lfs_sha256": null + }, + { + "path": "encoding/README.md", + "size": 12120, + "git_blob_id": "c4da223cc2d97f8d80e3d0d2ece8b1a137381507", + "lfs_sha256": null + }, + { + "path": "encoding/encoding.py", + "size": 37316, + "git_blob_id": "92a67eab2a67d924a6a8279bc0f90e3f504ac40e", + "lfs_sha256": null + }, + { + "path": "encoding/test_encoding.py", + "size": 19371, + "git_blob_id": "f2c4985a59b694da51491292f5c8264d39c7e926", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_input_1.json", + "size": 2761, + "git_blob_id": "c7435727e24c5f7b5cf28b470cf03814101eb37a", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_input_2.json", + "size": 527, + "git_blob_id": "132bb05a51e79204324f160cfde2880b5d0077ba", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_input_3.json", + "size": 2554, + "git_blob_id": "6d2612113c71fb9a897255ec694b7b8a41d0432b", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_input_4.json", + "size": 712, + "git_blob_id": "86cd1f0a68075a412654ea10f9558c69162b271a", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_input_5.json", + "size": 1143, + "git_blob_id": "2a2fa6a89e8118e896ec47abcc5bda30789e6532", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_output_1.txt", + "size": 2476, + "git_blob_id": "3dc9bfe936d251bf69f6db738d1302bfa8c4c705", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_output_2.txt", + "size": 294, + "git_blob_id": "78eb6be5de6f3cda7c5065b3c3bd01bac28ef1dc", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_output_3.txt", + "size": 2481, + "git_blob_id": "90b2b6b74beaa471dfbbba1d7e7a5eb88c9e7da0", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_output_4.txt", + "size": 574, + "git_blob_id": "efad296eb4ff95754f322476db4d26e72682a586", + "lfs_sha256": null + }, + { + "path": "encoding/tests/test_output_5.txt", + "size": 408, + "git_blob_id": "a130bcc2c5682c94b6fd5194def54209f1794939", + "lfs_sha256": null + }, + { + "path": "evaluation/README.md", + "size": 3977, + "git_blob_id": "fa1c989f2d0f5dd0ecdce369cb026b2e4040f88b", + "lfs_sha256": null + }, + { + "path": "evaluation/dsh-minimal.patch", + "size": 28725, + "git_blob_id": "2488b9578825a41f41964837d7d9c37db58b8146", + "lfs_sha256": null + }, + { + "path": "inference/README.md", + "size": 2029, + "git_blob_id": "e0b1d6ddfb4ab06a0dc9b4542826e6d7aa07a77c", + "lfs_sha256": null + }, + { + "path": "inference/config.json", + "size": 1982, + "git_blob_id": "7a915cc69e21abbc6d7fb939cef5d09c72aa0d24", + "lfs_sha256": null + }, + { + "path": "inference/convert.py", + "size": 9458, + "git_blob_id": "61bd2943648920ce599c4c86dc9bf6316e152fe8", + "lfs_sha256": null + }, + { + "path": "inference/engram.py", + "size": 8138, + "git_blob_id": "a36bc6c6b8cbf2d3dbac850863c23b55a7ecb5c7", + "lfs_sha256": null + }, + { + "path": "inference/examples/example.txt", + "size": 332, + "git_blob_id": "9200d36c90f601c0648ee23279abbb29b2228f13", + "lfs_sha256": null + }, + { + "path": "inference/examples/example_harmony.json", + "size": 2156, + "git_blob_id": "fae1c2608f8e5b980fcc9dc311cda8e345ecde1c", + "lfs_sha256": null + }, + { + "path": "inference/examples/images/carrots.jpeg", + "size": 212495, + "git_blob_id": "3b0226496452fe654ec305a3a42ef9a95cf47cfe", + "lfs_sha256": "5df896a4a07e127281c60fc957f8b3d73f4735b3258a0bf762b4383557f8fa9a" + }, + { + "path": "inference/examples/images/corn.jpeg", + "size": 56122, + "git_blob_id": "5777d198b83de15eb21243f8aac49074922053fb", + "lfs_sha256": null + }, + { + "path": "inference/generate.py", + "size": 8722, + "git_blob_id": "84d4873c4c59ea743e104920a77984f022bfb9f5", + "lfs_sha256": null + }, + { + "path": "inference/image_processor.py", + "size": 7699, + "git_blob_id": "a50311ece615b72be2c9d4b649a05619c27681e7", + "lfs_sha256": null + }, + { + "path": "inference/kernel.py", + "size": 23790, + "git_blob_id": "6fa7bd5aafbe1bc62887080963cc9fdd838d89f5", + "lfs_sha256": null + }, + { + "path": "inference/model.py", + "size": 61549, + "git_blob_id": "3e7dc222a6352ccc0c6469d58908934c2b4cd99b", + "lfs_sha256": null + }, + { + "path": "inference/requirements.txt", + "size": 97, + "git_blob_id": "b9201e446c0d2509156f46014931d359e439d2f2", + "lfs_sha256": null + }, + { + "path": "inference/run.sh", + "size": 1759, + "git_blob_id": "e2d52080a88d3053d963bac4bba75116ce73aed4", + "lfs_sha256": null + }, + { + "path": "inference/vision.py", + "size": 4457, + "git_blob_id": "77af0bdef9f4649edebac2876cd1df50a147ce98", + "lfs_sha256": null + }, + { + "path": "model-00001-of-00048.safetensors", + "size": 970533624, + "git_blob_id": "ed64b2e00da7669b19a7f0b8244bf87d3ab9980a", + "lfs_sha256": "886aebdafa08cc27bbae2165ed35bdfe0de9370bf88c1411283c155c6ae4ff89" + }, + { + "path": "model-00002-of-00048.safetensors", + "size": 1323858272, + "git_blob_id": "d04d1d34d11caccd930bfa5f7d780c86bfbe42ed", + "lfs_sha256": "4320066fc6958e5bc01d8c3feba79b7454b59f0f4b7299ab7145ed44bbf4ecec" + }, + { + "path": "model-00003-of-00048.safetensors", + "size": 7389759032, + "git_blob_id": "f19f5f2344084fd293de9a42ac0e29f862f32f24", + "lfs_sha256": "e1281f85d0ce4a3dfb63d41926fc4a47fa71f36ba20992e3597e702ead49d4c9" + }, + { + "path": "model-00004-of-00048.safetensors", + "size": 7389759032, + "git_blob_id": "827668797628e19f471172977f276b8255d6dd3f", + "lfs_sha256": "79456c9db0cda3b8115fe1c726fe3db1a34b434584a3991917088a0ab56a39de" + }, + { + "path": "model-00005-of-00048.safetensors", + "size": 7405953784, + "git_blob_id": "4cf131aac21417ca85ac3221803ad8dac77ddca2", + "lfs_sha256": "4a42dc78698bee6b1a821aa01c9650749ef1f716143751f1cdb815c6400280a9" + }, + { + "path": "model-00006-of-00048.safetensors", + "size": 7389759032, + "git_blob_id": "33afcb930c148f3edc878d589c402ad1ba4b6d6e", + "lfs_sha256": "020a6df51a2853452561d91268a65481f7a7d7954ed47f8e6c9ce69a4a134f77" + }, + { + "path": "model-00007-of-00048.safetensors", + "size": 7389759032, + "git_blob_id": "bef0f732f032b365410838f407d99141eeae9633", + "lfs_sha256": "40f8b52f763f6380d41257e1af04eee3aad300af6a418c38e49ff99e3604163a" + }, + { + "path": "model-00008-of-00048.safetensors", + "size": 7389759032, + "git_blob_id": "c85918896e64adb90a8c2bd005c806deb835ff7d", + "lfs_sha256": "d62cca4e698f030d4b96ec624c08bed7ad604cec13077da6d6b06669281c4650" + }, + { + "path": "model-00009-of-00048.safetensors", + "size": 7389759032, + "git_blob_id": "2ef42f88393516ab2b78f6219a51d3c13be1f8a1", + "lfs_sha256": "1ca62e4c294df31aee69a782974cb14269264fdc08465ab4835760258f05d6ef" + }, + { + "path": "model-00010-of-00048.safetensors", + "size": 7389759032, + "git_blob_id": "2cc30c2ca90ece11ae97ec1710b3cd461c2f4268", + "lfs_sha256": "dd33c9750a40cbfcdfb19cd8335d955f533e3a18b790c0ee43d1e5d77911595c" + }, + { + "path": "model-00011-of-00048.safetensors", + "size": 7405953784, + "git_blob_id": "806098f2fdb2017649d9a27b8c622a418fbaee04", + "lfs_sha256": "a9b309f90e0d1e2252a224ed6b057b9c82c27d3a49cff64bfbfef12efd067f7c" + }, + { + "path": "model-00012-of-00048.safetensors", + "size": 7389759032, + "git_blob_id": "57dc0d8beed2d7bf370c7422bcc1b625e2a711b6", + "lfs_sha256": "b359227eceb3f839c80de19dddf946648ca89425d9b719ff44702cf5e8cfbe0e" + }, + { + "path": "model-00013-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "33e6442ab5f10963f202c8ecbf9187409726943c", + "lfs_sha256": "41d87a4c81fec1550f9cb975db05598a18ee0161c2664e8e1a0c7b60742755a6" + }, + { + "path": "model-00014-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "358773636735c291c356a471d8a4c7fcf88b9ff0", + "lfs_sha256": "e7ca4a12688a5819829ead3a03282aedb380e438bc41a00e969f70952c6d0eb9" + }, + { + "path": "model-00015-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "44e7ccd06864c45d14d336a79bba9a98d0d03819", + "lfs_sha256": "fa9d49314bbb25118b3d52c573ed66a4d0acfb01dcb26d8367f43c66a52f46c4" + }, + { + "path": "model-00016-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "a1d29a48b074f07f84f540edb693fee5febc5bb4", + "lfs_sha256": "07d08bce9d73d416eaca0f42843440a8c45c5f939eff1a3690fb460d8dd7f623" + }, + { + "path": "model-00017-of-00048.safetensors", + "size": 7405956128, + "git_blob_id": "9b54c23553545da4e3cbb3238633431276d03f12", + "lfs_sha256": "3d35e330a6c28cd59a06a1caab8b0048f7e12cc037d728736f74431c4c0a1a4c" + }, + { + "path": "model-00018-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "1e3f4df33106ffbcbb985b7cf192b128108837f2", + "lfs_sha256": "48bd0c28b7f441f131f562cb664366d7726447a3698da3dd56ebbfeb1b916c75" + }, + { + "path": "model-00019-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "0194c1136548453cd9a74932c143797335439e54", + "lfs_sha256": "b7a25cb64a959ca5e5a99e239bb1ba2092b9bcf022bec4ba71f7441334107bd1" + }, + { + "path": "model-00020-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "82039c0d9b690646ece8dfb569f2471aa91d6287", + "lfs_sha256": "8aea8c4026ba4b93e82b4ed1b6b4620d02b6aaac88d211aaf577acea443aff84" + }, + { + "path": "model-00021-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "b0162a203fd3bf621bc18e5b688fa436059e0431", + "lfs_sha256": "4cb6558dedfdc75a6be472e27bc3e830d8443a3371854e61dc54454a1b5b18f7" + }, + { + "path": "model-00022-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "a9a16936023ebc4201d0b3966e872dce18451465", + "lfs_sha256": "81031b68c967539a7c51d1b12d37d18de926ca3045ec554ccdd49a83b4dc0084" + }, + { + "path": "model-00023-of-00048.safetensors", + "size": 7400713088, + "git_blob_id": "ccbb0d6ab728eb657e6cdd55ec96b87c21b60223", + "lfs_sha256": "096723fc8afa9886b976fb63aabef159406dc483e6f2a41e3fe386435b0436cc" + }, + { + "path": "model-00024-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "c81d6d5ad5b5b41f13d3240610724c31976fbf4f", + "lfs_sha256": "c9438ff607bd902fdeb38267d36239ccf60d38595988364562b862791841da75" + }, + { + "path": "model-00025-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "0bdd7e25ffc0aa2f8ed78445c24b4ba15d9c39e1", + "lfs_sha256": "4227ef9fe34d10de584d67250845fab905cbd8462fd84d9da24d772f223db6b4" + }, + { + "path": "model-00026-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "da6b3db1a7463825c9dfbffb07bd47aa160d92fb", + "lfs_sha256": "0af8c8f1b96b502eda652a0549cf55cd8a4d472a34d27ef9bf1c1d7c4567ca3f" + }, + { + "path": "model-00027-of-00048.safetensors", + "size": 7395337384, + "git_blob_id": "63e0eb6f1a1eca5698d68e7c70d0911fc2f5230b", + "lfs_sha256": "3066bd03043726d2400302cb9af759ec4e3957a935d56e241b83b52870322e28" + }, + { + "path": "model-00028-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "9aff3d0596b1e43a1bb8d608f4f83d7a05db4551", + "lfs_sha256": "9bc915075568b75e2aec9c645cc0fece6287f115c32031a52ab4f9ba9d12fde6" + }, + { + "path": "model-00029-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "a1fcbaf3892f7c67ec7a385acc5864e14211a0dc", + "lfs_sha256": "d153dd9cde7c4aa7ea9896cf7aa25b49447af573f90ad14cfaaa532fbbd19434" + }, + { + "path": "model-00030-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "1bf8c33843eb0a24a8ed94e1679eae2d0f010394", + "lfs_sha256": "3c9ccd96e908e00f02a70930b0821a6c53e5301376a261284a092e1aea7bb5da" + }, + { + "path": "model-00031-of-00048.safetensors", + "size": 7395337384, + "git_blob_id": "f233648fc05de748d637590366a8f2f40c6368e6", + "lfs_sha256": "0de7b6d7142df18ad45a215e9c0ef004f78cdbfefd59cc16345b2bc004047343" + }, + { + "path": "model-00032-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "01735f7ba251a25d3563d1cbb532dc07d934a67f", + "lfs_sha256": "6a5aaa73c6f97294bd079914ba44209ecf3d0491c8b95ef9995655511e35aee6" + }, + { + "path": "model-00033-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "54188954aee3724535d69814cdafcae5d6bcd174", + "lfs_sha256": "386e3e91f7f02f7e2c25f398b3a63379051c2077839b3081f7556e18c29a69c9" + }, + { + "path": "model-00034-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "7ebf03e74696c2a1c12a20804e6ebbba7e87755c", + "lfs_sha256": "9deab3c4f27c956f91cd989d9a1f6cf4bbb91858ce3f7164caa5ae61125c943f" + }, + { + "path": "model-00035-of-00048.safetensors", + "size": 7395337384, + "git_blob_id": "cc3e9dc7d0770fd9a4f367ab38161be1f9c00825", + "lfs_sha256": "226573bc07f35091b0ec27b0056ab3059f9bc1c0afd0bc104c93c86dd029cad2" + }, + { + "path": "model-00036-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "a93ac994149530aae4dd495f77261e403514d905", + "lfs_sha256": "90d6a85c1eb0cae68c6c7bffa11b60654ea00353d27a0ae17701b72e239894cd" + }, + { + "path": "model-00037-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "33f34ba08bb59e4b6854f936740f3b36aa71c29e", + "lfs_sha256": "207aef18f995fdfe7506175aabd700aa62dcf6f488c7cc375e9de06f56891b83" + }, + { + "path": "model-00038-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "13085e017de65e2750a8d0aba8e67554cdb42d5e", + "lfs_sha256": "cbbaa0b0807311807a696efec09f9b5a5d63bdf68b82378e1fc7402a2533c627" + }, + { + "path": "model-00039-of-00048.safetensors", + "size": 7395337384, + "git_blob_id": "f2b5f00589c6b9b890dd35e8232e48127ee96eef", + "lfs_sha256": "f4cf191547b50efbc35c4ed1f36cfb26043d94f987c49d54b15ef2d632a14a77" + }, + { + "path": "model-00040-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "2312d1cda8644809c58ad1ac469a6e3b0c818513", + "lfs_sha256": "e991bfc416055c45ddb89f1447eae1e67dba2816e7580044ffff3db7e2af27a3" + }, + { + "path": "model-00041-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "b45656843585d2f9faf3b9d119f72c5b53a31d82", + "lfs_sha256": "48a1c08afadf4e73223c587f9c6ad6f32aef57c01342ae75e96b07a001711d42" + }, + { + "path": "model-00042-of-00048.safetensors", + "size": 7389761368, + "git_blob_id": "4136a43c609949ed620615dbcc18ca5836428c2c", + "lfs_sha256": "e1a4d5d30ae51bafda75078d49b05a585c924b0af9b3481c075b04341507c2ef" + }, + { + "path": "model-00043-of-00048.safetensors", + "size": 1323837624, + "git_blob_id": "8fd7f2f6d9426613e927acd2c160d08590acea33", + "lfs_sha256": "d762b688f138e00a24eac96f27b842715bb4b534ecc9ca39076bed2b8c33201e" + }, + { + "path": "model-00044-of-00048.safetensors", + "size": 2652728736, + "git_blob_id": "eedbbd9f94af95a4c1d099f769f68b9ce1c268ab", + "lfs_sha256": "9a6b39fb88a2510487a8efaef77aa7864e8061f6b62c95a0f010e9dd538f3b05" + }, + { + "path": "model-00045-of-00048.safetensors", + "size": 2573998176, + "git_blob_id": "a622ff448f20314dc3512a7794eed8016a2509f6", + "lfs_sha256": "0cc9d5f6ca3a2158ccc63ce2c70c76aeda8177d54913340481af566680329eb5" + }, + { + "path": "model-00046-of-00048.safetensors", + "size": 2706402896, + "git_blob_id": "9c7282e33ca0717e0f0ee8c44666ac8877abc138", + "lfs_sha256": "e625902027b9d23d416f8818c665fab4704e0b96dc1bc778321601b700475a9d" + }, + { + "path": "model-00047-of-00048.safetensors", + "size": 101535150936, + "git_blob_id": "3eb96ce647ea90e0874a2da414f30fda9bac0034", + "lfs_sha256": "824db4881320407ac340736d14dcee5ecd748c27d0f5836b8127ecc2e3781b0f" + }, + { + "path": "model-00048-of-00048.safetensors", + "size": 101537926640, + "git_blob_id": "29a0c0b4e03cb4e663e823f28608c79ef7cb0f81", + "lfs_sha256": "976330f4954338e1ad8b508c32aa912032c7ad908959fd53c8307650fe4520ed" + }, + { + "path": "model.safetensors.index.json", + "size": 7470294, + "git_blob_id": "54c85064dd92c8550471302e9ae59bedbbf96ca4", + "lfs_sha256": null + }, + { + "path": "tokenizer.json", + "size": 6367257, + "git_blob_id": "6a15814dd25c934028034531da744c689f87ff21", + "lfs_sha256": null + }, + { + "path": "tokenizer_config.json", + "size": 801, + "git_blob_id": "f3dad388a2bbfd6a8605bd02754acd86d9ca5112", + "lfs_sha256": null + } + ] +} diff --git a/deploy/deepseek-v41/compiler_probe.c b/deploy/deepseek-v41/compiler_probe.c new file mode 100644 index 0000000..c9a85a6 --- /dev/null +++ b/deploy/deepseek-v41/compiler_probe.c @@ -0,0 +1,9 @@ +/* Build-time preflight for headers required by Triton's runtime C helpers. + * Passing this probe is not proof of GPU kernels, model startup, or quality. */ +#include +#include +#include + +int main(void) { + return EXIT_SUCCESS; +} diff --git a/deploy/deepseek-v41/h100-production.env b/deploy/deepseek-v41/h100-production.env new file mode 100644 index 0000000..e78e010 --- /dev/null +++ b/deploy/deepseek-v41/h100-production.env @@ -0,0 +1,18 @@ +# Qualification target, NOT release authorization. No credentials or host identifiers. +CHECKPOINT=deepseek-ai/DeepSeek-V4.1-Flash +CHECKPOINT_REVISION=dba1be0a40aa45a94ad051997016db3960a90277 +VLLM_SOURCE_REVISION=ac68c3087215e0a4f3cdfa218508c6aada57235d +VLLM_BASE_DIGEST=sha256:98adb2311118263a453ec0acd5d4b72bda3db26cecc0fe8ffc94b9ab03c951f9 +VLLM_IMAGE_ID=sha256:10b3c8fe9c38f6e87dfef21c8d0e457f76ab89b32892bb37a056375b25ddbf85 +TP=8 +GPU_MEMORY_UTILIZATION=0.85 +ENGRAM_CPU_OFFLOAD=true +MAX_MODEL_LEN=32768 +MAX_NUM_SEQS=2 +MAX_NUM_BATCHED_TOKENS=4096 +TOKENIZER_MODE=deepseek_v41 +REASONING_PARSER=deepseek_v41 +TOOL_CALL_PARSER=deepseek_v41 +LANGUAGE_MODEL_ONLY=true +HOST_RAM_LIMIT_GIB=1024 +MIN_GPU_FREE_MIB=3072 diff --git a/deploy/deepseek-v41/launch.py b/deploy/deepseek-v41/launch.py new file mode 100644 index 0000000..703b7e7 --- /dev/null +++ b/deploy/deepseek-v41/launch.py @@ -0,0 +1,136 @@ +"""Canonical pinned launch; never stops workloads, pulls images, or reboots a node.""" +from __future__ import annotations + +import argparse +import hashlib +import json +import subprocess +from pathlib import Path + + +def load_config() -> dict[str, str]: + values: dict[str, str] = {} + for line in Path(__file__).with_name("h100-production.env").read_text().splitlines(): + if not line or line.startswith("#"): + continue + key, value = line.split("=", 1) + if key in values: + raise ValueError("duplicate canonical setting") + values[key] = value + return values + + +def server_args(config: dict[str, str]) -> list[str]: + if config["LANGUAGE_MODEL_ONLY"] != "true" or config["ENGRAM_CPU_OFFLOAD"] != "true": + raise ValueError("qualification requires text-only Engram CPU offload") + return [ + "--model", "/model", "--served-model-name", "/model", "deepseek-v4.1-flash", + "--tensor-parallel-size", config["TP"], "--language-model-only", + "--tokenizer-mode", config["TOKENIZER_MODE"], + "--reasoning-parser", config["REASONING_PARSER"], + "--tool-call-parser", config["TOOL_CALL_PARSER"], "--enable-auto-tool-choice", + "--engram-config", '{"cpu_offload":true}', + "--max-model-len", config["MAX_MODEL_LEN"], + "--max-num-seqs", config["MAX_NUM_SEQS"], + "--max-num-batched-tokens", config["MAX_NUM_BATCHED_TOKENS"], + "--gpu-memory-utilization", config["GPU_MEMORY_UTILIZATION"], + "--generation-config", "vllm", + "--host", "0.0.0.0", "--port", "8000", "--no-enable-log-requests", + ] + + +def verify_checkpoint(model: Path, config: dict[str, str]) -> None: + manifest = json.loads(Path(__file__).with_name("checkpoint-manifest.json").read_text()) + if (manifest["model_id"] != config["CHECKPOINT"] + or manifest["revision"] != config["CHECKPOINT_REVISION"]): + raise RuntimeError("upstream manifest/config identity mismatch") + expected_paths = {item["path"] for item in manifest["files"]} + for path in model.rglob("*"): + relative = path.relative_to(model) + if path.is_symlink(): + raise RuntimeError("checkpoint symlinks are not admitted") + if (path.is_file() and relative.as_posix() not in expected_paths + and relative.parts[:2] != (".cache", "huggingface")): + raise RuntimeError("unexpected unpinned checkpoint file") + for item in manifest["files"]: + relative = Path(item["path"]) + if relative.is_absolute() or ".." in relative.parts: + raise ValueError("unsafe checkpoint path") + path = model / relative + if (path.is_symlink() or not path.is_file() or not path.resolve().is_relative_to(model) + or path.stat().st_size != item["size"]): + raise RuntimeError("checkpoint file admission failed") + digest = hashlib.sha256() if item["lfs_sha256"] else hashlib.sha1() + if not item["lfs_sha256"]: + digest.update(b"blob " + str(item["size"]).encode() + b"\0") + with path.open("rb") as handle: + while chunk := handle.read(8 * 1024 * 1024): + digest.update(chunk) + if digest.hexdigest() != (item["lfs_sha256"] or item["git_blob_id"]): + raise RuntimeError("checkpoint content differs from pinned upstream revision") + print(json.dumps({"checkpoint_verified": True, "files": len(manifest["files"])}), flush=True) + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--model-path", required=True, type=Path) + parser.add_argument("--cache-path", required=True, type=Path) + parser.add_argument("--name", default="c3r-deepseek-qualified") + parser.add_argument("--port", type=int, default=18000, + help="Loopback qualification port; production cutover is separately approved") + parser.add_argument("--execute", action="store_true") + parser.add_argument("--server-args", action="store_true", + help="Print canonical serving arguments for real upstream parser validation") + args = parser.parse_args() + config = load_config() + if args.server_args: + print(json.dumps(server_args(config))) + return + model, cache = args.model_path.resolve(), args.cache_path.resolve() + if (not model.is_dir() or not cache.is_dir() or model.is_relative_to(cache) + or cache.is_relative_to(model)): + raise ValueError("distinct existing checkpoint and private cache directories required") + if not 1 <= args.port <= 65535 or not args.name.startswith("c3r-deepseek-"): + raise ValueError("invalid scoped name or loopback port") + command = [ + "docker", "run", "-d", "--name", args.name, "--gpus", "all", + "--network", "bridge", "-p", f"127.0.0.1:{args.port}:8000", + "--read-only", "--cap-drop", "ALL", "--security-opt", "no-new-privileges", + "--shm-size", "16g", "--memory", config["HOST_RAM_LIMIT_GIB"] + "g", + "--memory-swap", config["HOST_RAM_LIMIT_GIB"] + "g", + "--tmpfs", "/tmp:rw,mode=1777,size=2g", + "--mount", f"type=bind,src={model},dst=/model,readonly", + "--mount", f"type=bind,src={cache},dst=/runtime", + "-e", "VLLM_CACHE_ROOT=/runtime/vllm", "-e", "XDG_CACHE_HOME=/runtime/.cache", + "-e", "FLASHINFER_WORKSPACE_BASE=/runtime/flashinfer", + config["VLLM_IMAGE_ID"], *server_args(config), + ] + if not args.execute: + print(json.dumps({"command": command, "qualification_status": "pending"})) + return + if config["VLLM_IMAGE_ID"] == "sha256:10b3c8fe9c38f6e87dfef21c8d0e457f76ab89b32892bb37a056375b25ddbf85": + raise RuntimeError("known failed compiler preflight; rebuild, scan and repin before execution") + verify_checkpoint(model, config) + # Fail before allocating: only the already built immutable image is eligible. + subprocess.run(["docker", "image", "inspect", config["VLLM_IMAGE_ID"]], + check=True, stdout=subprocess.DEVNULL, timeout=20) + gpu_rows = subprocess.check_output([ + "nvidia-smi", "--query-gpu=memory.total,memory.free", "--format=csv,noheader,nounits" + ], text=True, timeout=15).splitlines() + if len(gpu_rows) != int(config["TP"]): + raise RuntimeError("expected eight measured GPUs") + for row in gpu_rows: + total, free = map(int, row.split(",")) + if free < total * float(config["GPU_MEMORY_UTILIZATION"]) + int(config["MIN_GPU_FREE_MIB"]): + raise RuntimeError("insufficient co-resident headroom; do not overlap model engines") + memory = {line.split(":")[0]: int(line.split()[1]) + for line in Path("/proc/meminfo").read_text().splitlines() + if line.split(":")[0] in {"MemTotal", "MemAvailable", "SwapTotal", "SwapFree"}} + if (memory["MemAvailable"] < int(config["HOST_RAM_LIMIT_GIB"]) * 1024**2 + memory["MemTotal"] * .2 + or memory["SwapTotal"] != memory["SwapFree"]): + raise RuntimeError("host-RAM reserve or no-swap gate failed") + subprocess.run(command, check=True, timeout=60) + + +if __name__ == "__main__": + main() diff --git a/deploy/deepseek-v41/start.sh b/deploy/deepseek-v41/start.sh new file mode 100644 index 0000000..1a7ac4c --- /dev/null +++ b/deploy/deepseek-v41/start.sh @@ -0,0 +1,3 @@ +#!/bin/sh +set -eu +exec python3 "$(dirname "$0")/launch.py" "$@" diff --git a/deploy/encoder_identity.py b/deploy/encoder_identity.py new file mode 100644 index 0000000..6ffdd19 --- /dev/null +++ b/deploy/encoder_identity.py @@ -0,0 +1,45 @@ +"""Independent file identities from the immutable upstream HF revision API. + +Small files use Git blob identities; LFS objects use upstream SHA-256 identities. +These are release pins, not hashes accepted from the downloaded local manifest. +""" +import hashlib +from pathlib import Path + +REVISION = "b968826d9c46dd6066d109eabc6255188de91218" +LFS = { + "model-00001-of-00005.safetensors": "31d6a825ae35f11fb85b195b4c42c146c051e446433125a215336abdf95cbf5f", + "model-00002-of-00005.safetensors": "5991236cea6fe21f3d43cab0f0e84448734fbbe0789816202989f2ddc9d18282", + "model-00003-of-00005.safetensors": "c5185c4794be2d8a9784d5753c9922db38df478ce11f9ed0b415b7304d896836", + "model-00004-of-00005.safetensors": "b5ee7de71fbf17db3d5704e0c8f2bc7d005ca9e1d7ca2aeb19827b0cfcaa917a", + "model-00005-of-00005.safetensors": "20c2d6366ab85c90786ccdd829cd2b9e7d30ef3b2ebbb998280e7e4014b542ff", + "tokenizer.json": "aeb13307a71acd8fe81861d94ad54ab689df773318809eed3cbe794b4492dae4", +} +GIT_BLOBS = { + "config.json": "d46195ac87f837ad233d02b2f80f148bf7c005e0", + "generation_config.json": "20a8a9156fc8c3f25295ca067f61fdf120d517c5", + "merges.txt": "31349551d90c7606f325fe0f11bbb8bd5fa0d7c7", + "model.safetensors.index.json": "2b85c00f1b118961cd7a477e2bba0fe197a4ce1a", + "tokenizer_config.json": "417d038a63fa3de29cfde265caedae14d1a58d92", + "vocab.json": "4783fe10ac3adce15ac8f358ef5462739852c569", +} + + +def verify_encoder(root: Path) -> None: + expected = set(LFS) | set(GIT_BLOBS) + actual = {str(path.relative_to(root)) for path in root.rglob("*") + if path.is_file() and ".cache" not in path.parts} + if actual != expected: + raise RuntimeError("encoder file inventory does not match pinned release") + for filename in sorted(expected): + path = root / filename + if path.is_symlink(): + raise RuntimeError("encoder release must use ordinary read-only files") + hasher = hashlib.sha256() if filename in LFS else hashlib.sha1() + if filename in GIT_BLOBS: + hasher.update(f"blob {path.stat().st_size}\0".encode()) + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(4 * 1024 * 1024), b""): + hasher.update(chunk) + if hasher.hexdigest() != (LFS.get(filename) or GIT_BLOBS[filename]): + raise RuntimeError("encoder content does not match independent upstream identity") diff --git a/deploy/inspect_head_sboms.py b/deploy/inspect_head_sboms.py new file mode 100644 index 0000000..aca013c --- /dev/null +++ b/deploy/inspect_head_sboms.py @@ -0,0 +1,21 @@ +"""Explain package/SBOM discrepancies without suppressing scanner findings.""" +import importlib.metadata +import json +from pathlib import Path + +TARGETS = {"msgpack", "setuptools", "urllib3"} +results = [] +for path in Path("/usr/local/lib").rglob("*.cdx.json"): + document = json.loads(path.read_text()) + for component in document.get("components", []): + name = component.get("name", "") + if name not in TARGETS: + continue + try: + actual = importlib.metadata.version(name) + except importlib.metadata.PackageNotFoundError: + actual = "not_installed" + results.append({"sbom": str(path), "package": name, + "sbom_version": component.get("version"), + "installed_distribution_version": actual}) +print(json.dumps(results, indent=2)) diff --git a/deploy/monitoring/c3r-retention-failure-policy.json b/deploy/monitoring/c3r-retention-failure-policy.json new file mode 100644 index 0000000..e212dbf --- /dev/null +++ b/deploy/monitoring/c3r-retention-failure-policy.json @@ -0,0 +1,32 @@ +{ + "displayName": "C3R staging retention purge failure", + "combiner": "OR", + "enabled": true, + "conditions": [ + { + "displayName": "c3r-retention-purge failed execution", + "conditionThreshold": { + "filter": "metric.type=\"run.googleapis.com/job/completed_execution_count\" AND resource.type=\"cloud_run_job\" AND resource.labels.job_name=\"c3r-retention-purge\" AND metric.labels.result=\"failed\"", + "aggregations": [ + { + "alignmentPeriod": "60s", + "perSeriesAligner": "ALIGN_SUM", + "crossSeriesReducer": "REDUCE_SUM" + } + ], + "comparison": "COMPARISON_GT", + "thresholdValue": 0, + "duration": "0s", + "trigger": { "count": 1 } + } + } + ], + "notificationChannels": [ + "projects/columboai-frontend/notificationChannels/10331198960910692730" + ], + "documentation": { + "mimeType": "text/markdown", + "content": "C3R private staging retention purge failed. Wilfried: keep trace collection disabled, inspect the failed execution and bucket inventory, repair the purge and replay it, then verify no object exceeded the 30-day cap. This policy alone is not proof of daily execution or production readiness." + }, + "alertStrategy": { "autoClose": "604800s" } +} diff --git a/deploy/monitoring/staging-5xx.json b/deploy/monitoring/staging-5xx.json new file mode 100644 index 0000000..b5595d1 --- /dev/null +++ b/deploy/monitoring/staging-5xx.json @@ -0,0 +1,34 @@ +{ + "displayName": "C3R staging Cloud Run 5xx", + "combiner": "OR", + "enabled": true, + "notificationChannels": [ + "projects/columboai-frontend/notificationChannels/10331198960910692730" + ], + "documentation": { + "content": "C3R private staging 5xx. Wilfried: inspect c3r-staging revisions/logs, keep trace collection off, and route 100% to the last known-good fixed-disabled revision or disable the service. This alert does not certify launch readiness.", + "mimeType": "text/markdown" + }, + "conditions": [ + { + "displayName": "c3r-staging 5xx rate", + "conditionThreshold": { + "filter": "metric.type=\"run.googleapis.com/request_count\" AND resource.type=\"cloud_run_revision\" AND resource.labels.service_name=\"c3r-staging\" AND metric.labels.response_code_class=\"5xx\"", + "aggregations": [ + { + "alignmentPeriod": "60s", + "perSeriesAligner": "ALIGN_RATE", + "crossSeriesReducer": "REDUCE_SUM" + } + ], + "comparison": "COMPARISON_GT", + "thresholdValue": 0, + "duration": "0s", + "trigger": { + "count": 1 + } + } + } + ] +} + diff --git a/deploy/storage/private-traces-lifecycle.json b/deploy/storage/private-traces-lifecycle.json new file mode 100644 index 0000000..c617ece --- /dev/null +++ b/deploy/storage/private-traces-lifecycle.json @@ -0,0 +1,8 @@ +{ + "rule": [ + { + "action": {"type": "Delete"}, + "condition": {"age": 28} + } + ] +} diff --git a/docs/cloud-run-staging.md b/docs/cloud-run-staging.md new file mode 100644 index 0000000..2a20364 --- /dev/null +++ b/docs/cloud-run-staging.md @@ -0,0 +1,139 @@ +# Cloud Run private staging deployment (fixed-disabled) + +C3R's selected standalone hostname strategy is the Google-managed HTTPS URL that +Cloud Run assigns to a service. A custom domain is optional. A **private, +fixed-disabled staging boundary** is deployed at +`https://c3r-staging-795563500003.us-central1.run.app`. It is not a live model +route, governed trace collector, canary, or production service. Cloud Run IAM +authentication is required; public access is a separate release action after +qualification. + +On 2026-09-23, the `columboai-frontend` project received a dedicated +`c3r-staging-runtime` service account with no user-managed keys or project roles, +and a private, immutable-tag Docker repository at +`us-central1-docker.pkg.dev/columboai-frontend/c3r-staging`. Its repository IAM +policy has no public binding; Google-managed encryption is reported. Cloud Build +`b25dcb33-5553-4c6d-bc88-1541bfb299bc` built the initial runtime-only source +and pushed digest `sha256:14b88f22a839fc123d639e3a96c29f3b2242ee443174fedd637665d64a872a2f`. +That first image predates the disabled staging host and has **not** been deployed. +The `.gcloudignore` upload manifest was checked to include only the Dockerfile, +`.dockerignore`, and `c3r/` Python source; no data, evidence, papers, or Git metadata. +Registry creation and image publication do not verify service behavior, storage, +retention, alerting, or release readiness. + +Cloud Build `d70f1263-a60e-489c-958f-36436ab2875a` published the fixed-disabled +staging host at digest +`sha256:60d4daf25d879c41892a3b1b5fc84638d7289ca75a191059858dd183f1aa1209`. +Cloud Run service `c3r-staging` in `us-central1` runs this digest under the +dedicated service account, with one maximum instance, zero minimum instances, +and no explicit public invoker binding. Its two distinct tokens are pinned +Secret Manager references; values are not in Git or this record. Direct +unauthenticated `/health` returned 403, IAM-authenticated `/health` returned +200, an IAM-only decision request returned 401, and a request with both IAM +and C3R token returned `C3R_DISABLED`, no selected action, and no effect. +The [deployment evidence](../evidence/staging-deployment-v1/report.json) records +these checks without credentials or request bodies. + +`c3r.staging_host:build` is an intentionally fixed-disabled, recommendation-only +boundary smoke host. It cannot be turned into a production decision service by +environment flags, performs no external effects or provider calls, and retains no +trace rows. An authenticated private deployment can use it to exercise ingress, +IAM/TLS, startup, monitoring, and rollback, but **not** to collect empirical data +or qualify C3R decisions. A separately reviewed host with measured pre-decision +estimates, durable governed storage, and approved source registry is required later. + +An independent, empty C3R trace bucket and retention purge job now exist in +staging. The bucket is not mounted or granted to `c3r-staging`; the purge job's +service account has only bucket-scoped list/delete authority. A 28-day lifecycle +delete rule and daily UTC scheduler are configured, and direct and scheduler- +triggered empty-bucket purges completed. The [retention staging record](../evidence/private-retention-staging-v1/report.json) +also records a generation-guarded deletion of one non-sensitive marker under +the purge identity, followed by empty live and all-version listings. The +one-off probe job was removed. It distinguishes these observations from the +first natural daily run, age-based expiry of real trace rows, deletion from +independent backups or replicas, effective inherited IAM, and access auditing. Nothing +about this bucket enables trace collection or production qualification. +A C3R-only failed-purge alert policy now watches Cloud Run job completion +results and targets the existing work-email channel. Its configuration was +read back after creation; a failed-execution delivery drill has not run. + +## Controls still required before live collection or promotion + +1. Wilfried Kouadio (`@wilkont`) is the interim deployment/release owner and + confirmed interim on-call/rollback operator in the + [internal-task policy](internal-task-trace-policy.md). Verify the work-email + alert route and acknowledgement before live internal-task traffic. A C3R-only + Cloud Monitoring email channel and 5xx policy now exist. Wilfried reported + receiving and acknowledging a staging drill alert on 2026-09-23; an + automated mailbox delivery audit and incident response timing remain open. +2. Implement the approved internal-task trace policy: source registry, field + allowlist, redaction tests, 30-day private deletion including backups, access + audit, and publication review. The approved scope is only redacted telemetry + from C3R-controlled internal tasks in private shadow/canary tests; it excludes + customer and product traffic. Written approval alone does not enable collection. +3. Supply a calibrated, host-owned *pre-decision* estimate source. Example or + constant values must not be presented as measured production CVoC inputs. +4. Add a durable trace store, independent ledger-head anchor, backup and deletion + procedure, monitoring, alert thresholds, request budget, and a kill switch. +5. The repository now includes a fail-closed `Dockerfile` and `python -m c3r.serve` + composition for the tested `C3RIngressServer` and loopback `C3RHTTPServer`. + The ingress listens on `0.0.0.0:$PORT`, while the backend retains + its loopback-only invariant. It allowlists `/health`, `/metrics`, and + `/v1/decisions`, requires a separate client token for protected routes, replaces + caller credentials with a distinct backend token, and caps body size, response + size, request rate, concurrency, and upstream wait time. Local tests cover these + boundaries, backend failure, and local server composition. GitHub CI has + built and HIGH/CRITICAL-scanned the digest-pinned staging image with zero + findings in [run 70](https://github.com/ColomboAI-com/c3r/actions/runs/35879315503). + This scan does not qualify lower severities or deployed controls. + Bind this ingress only behind Cloud Run's IAM/TLS boundary at staging, with + tokens from Secret Manager. Do not publish it as a raw unauthenticated port. +6. Establish a private, authenticated service-to-service route to the GPU model; + never publish the model's localhost inference port. Verify the model checkpoint + backup before any GPU VM lifecycle change. + +`C3R_HOST_ENTRYPOINT=trusted_module:build` is mandatory. The trusted callable must +return a recommendation-only `StandaloneController` and a `RequestFactory` whose +estimate source uses measured, pre-decision values. The container supplies **no** +sample catalog, constant estimates, data collection, or production credentials. +`C3R_CLIENT_TOKEN` and `C3R_BACKEND_TOKEN` are distinct mandatory secrets of at +least 32 characters; `PORT` defaults to 8080 and `C3R_BACKEND_PORT` to 8081. +The currently deployed host is `c3r.staging_host:build`, deliberately +fixed-disabled. A later trusted decision host must be separately reviewed and +deployed. Registry publication and boundary checks are complete only for the +disabled host; durable ledger storage, secret rotation, and live decision tests +remain necessary before governed internal-task traffic. + +On 2026-09-23 a second, equivalent disabled revision (`c3r-staging-drill1`) +was deployed, then traffic was explicitly restored 100% to the original +`c3r-staging-00001-zrj` revision. IAM-authenticated `/health` returned 200 +after rollback. This proves revision traffic rollback for the disabled staging +host, not incident response timing or a production rollback. Cloud Monitoring +policy `11003571770572095050` watches this service's 5xx request count and +routes to Wilfried's work-email channel. A temporary 2xx notification drill +on 2026-09-23 observed +the authorized health-request metric and opened Monitoring incident +`0.ocyx1hrngr2m` at 15:24:53 UTC, just after the first check and policy +disable. Wilfried subsequently reported receiving and acknowledging the drill +email in his work mailbox. This is operator-reported delivery evidence, not +an automated mailbox audit or response-time SLA test. The drill policy is +disabled; the persistent 5xx policy remains enabled. + +## Staged promotion + +| Stage | Access and effect authority | Evidence required to advance | +| --- | --- | --- | +| Offline replay | No cloud endpoint or external effects | Reproducible internal-task pairs, source manifest, independent outcomes and held-out partition. | +| Private shadow | IAM-protected HTTPS URL; recommendation-only; internal tasks | Auth, rate-limit, redaction, ledger, provider-outage, rollback and kill-switch traces. | +| Private read-only canary | Restricted invokers; no external effects | Predeclared latency/cost/safety thresholds and independently reviewed run evidence. | +| Public read-only launch | Explicit public-access approval; still no external effects | Security review, owner/on-call, abuse controls, reproducible empirical claims, release-copy audit. | +| Reversible effects | Separate authority and approval | Complete mediation and reversible canary evidence. Not implied by the public read-only launch. | + +Keep the Hugging Face model and dataset labeled as previews until trained weights, +empirical data, held-out calibration, and reproducible evaluation artifacts exist. +MC-1 remains outside the standalone launch scope, not complete under the directive. + +Cloud Run references: [HTTPS service URL and invoking services](https://docs.cloud.google.com/run/docs/triggering/https-request), +[IAM service authentication](https://docs.cloud.google.com/run/docs/authenticating/overview), +[public versus authenticated deployment](https://docs.cloud.google.com/run/docs/deploying), +and [container listening contract](https://docs.cloud.google.com/run/docs/container-contract). diff --git a/docs/controlled-replay.md b/docs/controlled-replay.md new file mode 100644 index 0000000..e032439 --- /dev/null +++ b/docs/controlled-replay.md @@ -0,0 +1,28 @@ +# C3R-controlled paired replay v1 + +The user authorized **C3R-controlled internal tasks only** for the current +training/evaluation source. This five-case pack exercises the local standalone +controller against an identical disabled-controller baseline. It uses C3R-authored +fixture facts, no customer records, no remote model, and no external effect. + +Run `python scripts/run_controlled_pairs.py` to produce a redacted observation +JSONL file and manifest under `evidence/controlled-pairs-v1/`. Each case has a +predeclared policy-choice rubric (action ID or stop), so the positive label means +**rubric match**, not real task success. The observations include state and trace +hashes, measured single-run local latency, and zero external-provider spend. +Latency is diagnostic only: this tiny, un-warmed fixture run is not a throughput +or production cost benchmark. Re-running changes timing and therefore the +observation-file hash; the checked-in file is an immutable snapshot. + +The paired report and source admission manifest are in the +[`c3r-evals` draft branch](https://github.com/ColomboAI-com/c3r-evals/tree/feat/redacted-serving-probe-pr/sources). +That report checks one baseline and one C3R result per identical state and keeps +the outcome kind explicit. It does **not** exercise Laya, DeepSeek, production +providers, actual tool/task outcomes, Colibri, calibration, canaries, or field +traffic. It cannot qualify the production release or empirical DecisionMix. + +Next, extend the C3R-owned task population beyond these regression fixtures, +freeze task/rubric hashes and splits before selecting a checkpoint, collect +independently reviewed outcomes for both arms, and reserve a separate held-out +calibration set. Any production-traffic claim requires a separately authorized +field source and live deployment evidence. diff --git a/docs/developer-api.md b/docs/developer-api.md new file mode 100644 index 0000000..99954bc --- /dev/null +++ b/docs/developer-api.md @@ -0,0 +1,185 @@ +# C3R developer API: integration candidate + +The API code is under qualification. There is **no qualified public endpoint yet**. +Do not treat these examples, local SDK tests or the release branch as a GA claim. +The host/security gates block public traffic, not API implementation. + +## Quickstart and authentication + +Use an operator-issued project key, never a GPU/provider secret. Keep it in a +server-side secret manager. Keys have `c3r_sk_test_` or `c3r_sk_live_` prefixes; +the prefix does not certify that a deployment has passed release qualification. +The access store retains only salted key hashes and scoped metadata. A newly +issued plaintext key is shown once and cannot be recovered from the store. + +Operator configuration uses `C3R_API_AUTH_MODE=keys` and an absolute +`C3R_API_ACCESS_DB` path on a private persistent volume. The database must be +owned by the runtime UID with mode `0600`, inside a `0700` directory with safe +ancestry. Run `python -m c3r.key_management --help` as the same trusted service +identity to create projects, issue scoped/expiring keys, revoke keys and inspect +project-specific metadata. Deliver the show-once key through a secure channel; +never commit or log its output. `C3R_BACKEND_TOKEN` and the distinct internal +readiness token remain private service credentials, not developer keys. + +The production composition defaults to key mode and rejects static staging auth. +Start with one gateway instance: database quotas coordinate against that store, +but route capacity is process-local. Database encryption, durable volumes, +metadata retention/deletion, backup access and multi-instance admission need +deployment evidence before public traffic. + +Replace `https://` only after an approved managed-TLS gateway exists. + +```bash +curl "https:///v1/responses" \ + -H "Authorization: Bearer $C3R_API_KEY" \ + -H "Content-Type: application/json" \ + -d '{"model":"c3r-core","input":"Recommend the safest next computation.","store":false,"max_output_tokens":512}' +``` + +```python +import os +from openai import OpenAI + +client = OpenAI(api_key=os.environ["C3R_API_KEY"], + base_url=os.environ["C3R_BASE_URL"].rstrip("/") + "/v1") +response = client.responses.create(model="c3r-core", input="Recommend the safest next computation.", + store=False, max_output_tokens=512) +print(response.output_text) +``` + +```typescript +import OpenAI from "openai"; +const client = new OpenAI({ apiKey: process.env.C3R_API_KEY, + baseURL: `${process.env.C3R_BASE_URL!.replace(/\/$/, "")}/v1` }); +const response = await client.responses.create({ model: "c3r-core", + input: "Recommend the safest next computation.", store: false, max_output_tokens: 512 }); +console.log(response.output_text); +``` + +These examples use the unmodified official SDKs. Supported fields are a bounded +text `input`, `model`, `max_output_tokens`, `store:false`, and `stream`. This is +a Responses text subset, not support for tools, images, background jobs, stored +conversations or every OpenAI API feature. See the +[official streaming guide](https://developers.openai.com/api/docs/guides/streaming-responses) +for the SDK event-consumption pattern. + +## Models and specialist endpoints + +| Endpoint | Purpose | Key scope | +| --- | --- | --- | +| `POST /v1/responses` | Controller-admitted, bounded self-hosted text generation | `responses:write` | +| `POST /v1/system-one` | CLM/Qwen typed inference | `system_one:write` | +| `POST /v1/c3r/rank` | Advisory candidate ranking | `rank:write` | +| `POST /v1/c3r/decide` | Governed recommendation | `decide:write` | +| `GET /v1/models` | Availability of `c3r-core`, `c3r-system-one`, `c3r-verifier` | `models:read` | +| `POST /v1/c3r/execute` | Unsupported remote effects | Always disabled, HTTP 501 | + +Model availability is health-derived, not a training/calibration assertion. The +verifier ID does not grant remote execution authority. Ranking remains +`calibrated:false` and `advisory_only`; generation does not claim positive learned +CVoC when it uses the explicit text-only policy fallback. + +System One accepts a typed question catalog: + +```json +{"model":"c3r-system-one","state":"Payment was duplicated","questions":{"department":{"type":"choice","options":{"billing":"Billing","technical":"Technical"}},"urgent":{"type":"boolean"}}} +``` + +Ranking accepts bounded candidates: + +```json +{"state":"Latency increased","candidates":["inspect queue depth","restart service","increase GPU count"]} +``` + +Decide accepts task text; action definitions, authority, verifiers and estimates +come from the trusted host, never caller-supplied permissions: + +```json +{"goal":"Inspect the incident","current_subgoal":"Choose a safe read-only computation"} +``` + +The result exposes route, reason category, authority result, conservative CVoC +decision and a trace hash, with `effect_executed:false`. Private reasoning is not +returned. No arbitrary action can be executed by this v0.1 API. + +## Streaming + +For `/v1/responses`, set `stream:true`. SSE events include `response.created`, +`response.output_text.delta`, `response.output_text.done`, `response.completed` +or `response.failed`. Consume deltas rather than provider reasoning fields. +Closing a stream must cancel the upstream connection and release admission +capacity; cancellation is part of local acceptance and still needs live-provider +qualification. A failed stream is not a completed billable success. + +```python +with client.responses.create(model="c3r-core", input="Recommend a safe computation.", + stream=True) as stream: + for event in stream: + if event.type == "response.output_text.delta": + print(event.delta, end="", flush=True) +``` + +## Organizations, projects and limits + +Each key belongs to one organization/project and has explicit scopes, expiration +and revocation. The gateway resolves those identities and a generated +`x-request-id` before forwarding. Caller identity headers cannot change tenancy. +There is no public credential-issuance or cross-tenant administration endpoint. + +Project RPM, key RPS and route concurrency are separate admission controls. +Bodies and output token counts are bounded. Saturated requests receive HTTP 429, +not an unbounded generation queue. Configured defaults are **not a measured GPU +safe envelope**. Multi-replica quota coordination and the eight-H100 operating +envelope must be qualified before horizontal gateway deployment. + +## Errors and request IDs + +Developer-key mode uses an error object with `message`, `type`, `code`, `param` +and `request_id`. Every gateway response includes a generated `x-request-id`. +Use this identifier for support; never send a key, prompt or private output. + +| HTTP | Meaning | +| --- | --- | +| 400/413/415 | Invalid request, size or content type | +| 401 | Invalid, expired or revoked key | +| 403 | Key lacks endpoint permission | +| 404 | Unsupported route, including `/internal/*` | +| 429 | Rate or capacity exceeded | +| 500/503 | Internal or dependency failure | + +`501` remains intentional for `/v1/c3r/execute`. Legacy private staging mode +retains its existing error/auth contract and must not be mistaken for public +developer-key mode. + +## Security, privacy and usage + +`store:false` is mandatory. Infrastructure/application logging must exclude +prompt bodies, responses, candidates, CLM state and reasoning. The gateway emits +metadata-only usage: request/tenant/project/key identity, route, model, status, +latency and provider-reported token counts where available. Unknown GPU time or +cost stays unknown, never fabricated or used as an exact billing claim. +Exact per-model billing and System-One invocation metering still require measured +runtime integration and a governed metadata retention policy. + +Only the managed-TLS gateway may be public. Core, CLM, Qwen, DeepSeek and the +separate authenticated `127.0.0.1` internal readiness channel remain private. +The gateway never forwards `/internal/*`. TLS, network isolation, database +encryption/backups, read/export auditing and alerting are deployment controls; +local tests cannot establish them. + +## Acceptance, status and changelog + +Local SDK acceptance lives in `tests/test_official_sdk_acceptance.py`. It uses +real C3R HTTP composition and fixture generation, not a GPU qualification run. +The optional unmodified Python and JavaScript SDKs must both be installed to +exercise both tests; a skipped SDK test is not a compatibility PASS. + +No availability SLO or public status page is claimed yet. After host approval, +measure three cold starts, 15/30/60-minute mixed traffic, percentiles, TTFT, +tokens/sec, capacity refusals and dependency/rollback drills. Release owner and +independent reviewer must approve exact commits, images and canary evidence. + +Candidate changelog: deliberately reconciled PR/readiness lineage; private +worker-aware readiness; hash-only developer keys and tenancy; metadata-only +admission/usage; streaming and SDK acceptance under qualification. No production +tag or public promotion is implied. diff --git a/docs/directive-compliance.md b/docs/directive-compliance.md index 9ecab15..d10dcc2 100644 --- a/docs/directive-compliance.md +++ b/docs/directive-compliance.md @@ -10,7 +10,8 @@ evidence-bearing: - **blocked externally** - completion requires governed data, compute, credentials, MC-1 code, or another repository/runtime not present in this workspace. -The full directive is **not complete**. The current release is the first reviewed vertical slice. +The full directive is **not complete**. MC-1 integration is deferred for the standalone +launch only; it remains a directive gap. The current public release is a research alpha. | DoD | Requirement | Status | Evidence / remaining gate | | --- | --- | --- | --- | @@ -20,14 +21,14 @@ The full directive is **not complete**. The current release is the first reviewe | 25 | C3R-specific Laya checkpoint exists and is calibrated | blocked externally | no governed empirical corpus, training run, derived weights, or held-out calibration evidence | | 26 | checkpoint published with attribution | partial | Hugging Face integration-preview repository, license, NOTICE, paper links, and upstream manifest are published; no derived checkpoint exists | | 27 | empirical DecisionMix v1 published | partial | synthetic 144-record schema preview with deterministic splits and hashes is public; empirical provenance-audited corpus remains | -| 28 | Qwen/DeepSeek/frontier Deliberative Envelope works | partial | structured envelope plus tested OpenAI, Anthropic, Gemini, and OpenAI-compatible Qwen/DeepSeek/vLLM/SGLang/MC-1 request/response contracts exist; live provider qualification and paired end-to-end traces do not | -| 29 | CVoC uses measured runtime cost/risk | partial | conservative reference calculation and fallback tests exist; calibrated production inputs and end-to-end measured traces do not | -| 30 | Verifier Firewall and Commit Gateway independent | partial | independent reference modules, action-bound attestations, atomic expiring approvals, and bypass tests; host-level complete mediation remains | +| 28 | Qwen/DeepSeek/frontier Deliberative Envelope works | partial | structured envelope and tested provider contracts exist; fixed-prompt credentialed OpenRouter smoke probes cover DeepSeek, Qwen3.8 Flash, and Claude Sonnet 4.6, plus a private DeepSeek serving-budget probe. Paired C3R traces and broad provider qualification do not exist | +| 29 | CVoC uses measured runtime cost/risk | partial | conservative reference calculation, fallback tests, and integrated controller path exist; calibrated production inputs and end-to-end measured traces do not | +| 30 | Verifier Firewall and Commit Gateway independent | partial | independent reference modules, action-bound attestations, atomic expiring approvals, and integrated recommendation-only HTTP boundary tests; host-level complete mediation remains | | 31 | OpenAI-compatible MC-1 API supports System-One | not started | the private `MC-1-platform` repository is identified and accessible, but the request contract, controller service, traces, and production qualification are not implemented | | 32 | MC-1 Console visualizes both paths | not started | the private `MC-1-platform` console is identified and accessible, but C3R trace ingestion, read models, and UI are not implemented | -| 33 | Colibri adapter runs in shadow mode | partial | non-authoritative recommendations consume the required telemetry shape and retain native fallback in tests; a dedicated `c3r-colibri` repository, compatible build, and live shadow traces remain | -| 34 | calibration metrics reproduce from raw traces | partial | fitter and accuracy/Brier/ECE/MCE/NLL/selective-risk tests plus a canonical trace hash-chain integrity check exist; no empirical raw trace release pack or independently anchored ledger head | -| 35 | all controller baselines compared identically | not started | requires immutable empirical test states and pinned serving environments | +| 33 | Colibri adapter runs in shadow mode | partial | non-authoritative recommendations consume the required telemetry shape and retain native fallback in tests; dedicated `c3r-colibri` repository and pinned build exist, but compatible live instrumentation and shadow traces remain | +| 34 | calibration metrics reproduce from raw traces | partial | fitter and accuracy/Brier/ECE/MCE/NLL/selective-risk tests plus a canonical trace hash-chain integrity check exist; a five-case redacted controlled replay is published, but it has no model probabilities or empirical calibration labels, and no independently anchored ledger head | +| 35 | all controller baselines compared identically | partial | a five-case controlled disabled-controller/C3R policy-rubric comparison uses identical local states; full immutable empirical test states, model baselines, task outcomes, and pinned serving environments remain | | 36 | no non-comparable TypeSafe Jev/RLCD claim | complete | repository and cards make no apples-to-apples Jev claim and preserve the caveat | | 37 | deterministic fallback survives controller failure | partial | conservative STOP, malformed provider output, Colibri controller failure, and fail-closed prediction behavior are tested; live provider timeout and end-to-end outage qualification remain | | 38 | global learned-fast-path disable preserves MC-1 | partial | fail-closed reference flags and global-disable tests exist; product-level MC-1 kill-switch integration and rollback evidence remain | @@ -35,7 +36,8 @@ The full directive is **not complete**. The current release is the first reviewe ## Directive-wide gaps outside the numbered DoD -- The public `c3r-evals` and `c3r-colibri` repositories are not present. +- The public `c3r-evals` and `c3r-colibri` repositories exist, but neither contains a + complete standalone production qualification pack. - Provider protocol contracts are shipped, but credentialed qualification for OpenAI, Anthropic, Gemini, Qwen, DeepSeek, vLLM, SGLang, llama.cpp, and MC-1 is not yet evidenced. - Required empirical tests for multilingual traffic, unseen schemas, provider outages, inventory @@ -45,6 +47,9 @@ The full directive is **not complete**. The current release is the first reviewe - No raw trace pack currently supports calibration, latency, cost, task-success, or calls-avoided claims. +The standalone launch gate and its external operator inputs are tracked in +[`standalone-launch.md`](standalone-launch.md). + Directive section 22's required prior-art links and canonical novelty boundary are collected in [`prior-art.md`](prior-art.md). diff --git a/docs/empirical-release-plan.md b/docs/empirical-release-plan.md index ac06eae..95e5ada 100644 --- a/docs/empirical-release-plan.md +++ b/docs/empirical-release-plan.md @@ -6,6 +6,33 @@ Produce the first evidence-complete `ColomboAI/C3R-Decision-Laya-421M-v0.1` chec empirical `ColomboAI/C3R-DecisionMix-v1` dataset without weakening the authority boundary, contaminating held-out benchmarks, or presenting synthetic fixtures as deployment evidence. +## Source decision (2026-09-22) + +The [Laya/DeepSeek source audit](laya-data-source-audit.md) accepts pinned Laya weights as +the attributed training base. `LocalLLaMA/typed-decisions` may be used only as a separately +labeled synthetic training or benchmark source after its exact revision and file hashes are +recorded; its upstream test split stays in a sealed comparison harness, never in DecisionMix +train, validation, or calibration. Self-hosted DeepSeek can generate proposals from approved +prompts, but its own output is not an independent verifier or task-outcome label. + +The user approved **C3R-controlled internal tasks only** as the present +training/evaluation source, including redacted telemetry from those tasks in private +shadow/canary qualification. The [interim trace policy](internal-task-trace-policy.md) +designates `@wilkont` as the accountable ColomboAI account, caps private row-level +retention at 30 days, and defines restricted publication after review. This does not +approve customer, product, Laya-user, DeepSeek-user, or Colibri operational logs. +Collection remains disabled until the policy's technical activation gates are +implemented and verified. The first controlled source is +the [five-case local paired replay](controlled-replay.md), which is a pipeline smoke +test, not an empirical DecisionMix corpus or training/calibration set. Before +expanding the controlled corpus, record the source owner, task +population, data-use rights, consent/privacy basis where applicable, retention/deletion rule, +redaction policy, and separate internal-training and public-publication scopes. Execute paired +baseline/controller runs against the same immutable tasks with independent verifier and +outcome labels, then lock train/validation/calibration/test partitions before model selection. +Controlled internal tasks establish a bounded controlled-evaluation claim; they do not by +themselves establish production-traffic or Colibri-shadow performance. + The plan is ordered by evidence dependency. A later phase cannot waive an earlier exit gate. ## Release train diff --git a/docs/image-security.md b/docs/image-security.md new file mode 100644 index 0000000..d5f07fb --- /dev/null +++ b/docs/image-security.md @@ -0,0 +1,29 @@ +# Staging image vulnerability triage + +The first GitHub Actions image scan, [run 48](https://github.com/ColomboAI-com/c3r/actions/runs/35868557321), +failed its HIGH/CRITICAL gate. Its Trivy JSON artifact reported **46 HIGH, zero +CRITICAL** entries: 44 in the Debian 13.7 base image and two Python packaging +packages (`jaraco.context` 5.3.0 and `wheel` 0.45.1). Many Debian entries +repeated the same CVE across `util-linux` subpackages and had no fixed version +listed. They have not been waived or marked safe. + +The container now uses the official Python 3.11 Alpine 3.24 runtime image and +copies only the stdlib-based `c3r` package. It does not run `pip install` or +retain the vulnerable `jaraco.context` and `wheel` packaging packages. The +intermediate [run 50](https://github.com/ColomboAI-com/c3r/actions/runs/35869194641) +had zero HIGH/CRITICAL Alpine OS findings but two HIGH Python packaging +findings. [Run 52](https://github.com/ColomboAI-com/c3r/actions/runs/35869500670) +showed that uninstalling those names alone did not remove copies vendored +inside `setuptools`; the runtime therefore removes `setuptools` itself, which +C3R does not import. The image runs as UID/GID 10001 and is smoke-imported in +CI before scanning. [Run 56](https://github.com/ColomboAI-com/c3r/actions/runs/35870568965) +passed the three-version Python test matrix, built the digest-pinned image, +smoke-imported C3R, and scanned it with Trivy v0.74.0. Its retained JSON +artifact reports **zero HIGH and zero CRITICAL findings** for the scanned +image. This is a point-in-time result, not a waiver for the earlier image or a +guarantee against future disclosures. Lower-severity findings were outside the +configured HIGH/CRITICAL scan and have not been triaged by this report. + +Even a clean scan is only one security gate. A production image still needs +registry publication, signature/SBOM, deployment IAM and network +review, and the live safety qualification in `standalone-launch.md`. diff --git a/docs/independent-review-handoff.md b/docs/independent-review-handoff.md new file mode 100644 index 0000000..42cb6fc --- /dev/null +++ b/docs/independent-review-handoff.md @@ -0,0 +1,88 @@ +# Independent C3R release review handoff + +**Status:** accepted; fixture-only dry run reviewed; deployment controls and +release not signed off. Wilfried Kouadio named Swapnil +Pawar as the independent reviewer on 2026-09-23. Swapnil replied from his +ColomboAI work mailbox on 2026-09-23 accepting the role, reporting no conflict +of interest, and requesting access to a non-sensitive dry run. His reply also +acknowledged that collection and publication remain off. Acceptance does not +verify deployment controls or authorize launch. +On 2026-09-23 he reported in the ColomboAI chat that he marked PR #2 ready for +review and saw passing checks. The PR had no submitted GitHub review or inline +review threads when checked afterward, and launch issue #3 had no independent +findings. Readiness and CI status are not substantive control, label, or public- +row sign-off; request a dated gate-by-gate finding with evidence references. +Swapnil's later 2026-09-23 work-email response reviewed the nine local checks +as clear and explicitly withheld live-collection and publication approval. He +requested deployed storage/IAM, access audit, deletion including copies, +independent anchoring, alert/rollback/kill-switch, and eventual live provenance +and outcome evidence. The [staging packet](reviewer-staging-packet-2026-09-23.md) +indexes what exists and what remains unavailable without treating the fixture +finding as deployment sign-off. + +## What the reviewer should decide + +1. Record the accepted scope and revisit conflicts if the task or reporting + relationship changes. The release owner cannot sign on the reviewer's behalf. +2. Before collection, verify the [internal-task trace policy](internal-task-trace-policy.md) + against an intentionally non-sensitive dry run: C3R-authored task provenance, + the exact source/task allowlist, permitted fields, redaction/leakage checks, + storage encryption and IAM, read/export audit, daily 30-day deletion including + backups, independent ledger-head anchoring, and tested kill switch. Inspect + deployed settings and execution evidence, not only source code or a plan. +3. For empirical release, review frozen train/calibration/test manifests, + independent outcome labels, leakage and contamination checks, raw predictions, + calibration fit, confidence intervals, paired baselines, safety failures, + and the proposed model/dataset card claims. Provider generations alone are + not ground truth. The five-case controlled replay and synthetic DecisionMix + preview are not substitutes for these artifacts. +4. Before any public row release, inspect an explicit publication manifest and + sample of every proposed row class for rights, re-identification, private + artifact references, and accidental personal or credential content. Approve + aggregates separately from row-level publication. + +## Current evidence, not yet sufficient for sign-off + +- [C3R PR #2](https://github.com/ColomboAI-com/c3r/pull/2) is a draft with the + tested standalone controller and fail-closed service boundary. +- [CI run 70](https://github.com/ColomboAI-com/c3r/actions/runs/35879315503) + built and smoke-imported the digest-pinned image and produced a Trivy report + with zero HIGH/CRITICAL findings. This does not cover lower severities or + deployment controls. +- The [non-sensitive local trace-control dry run](../evidence/trace-control-dry-run-v1/report.json) + uses one C3R-authored synthetic fixture. Nine local admission, redaction, + retention-lockout, purge, and chain checks pass. It creates no live trace and + cannot verify encrypted deployment storage, backup deletion, access auditing, + independent anchoring, alerting, or task outcomes. Reproduce with + `python scripts/run_trace_control_dry_run.py`. +- [Private staging evidence](../evidence/staging-deployment-v1/report.json) + shows a fixed-disabled Cloud Run boundary, IAM and token checks, a service- + specific 5xx alert rule, revision traffic rollback, and a notification + drill that opened a Monitoring incident. Wilfried reported receiving and + acknowledging the drill email. This is not a live C3R decision route or + evidence of governed trace storage, backup deletion, independent anchoring, + automated mailbox audit, response-time SLA, or canaries. +- [Private retention staging evidence](../evidence/private-retention-staging-v1/report.json) + records an empty dedicated bucket, restricted purge identity, 28-day lifecycle + backstop, configured daily schedule, and successful direct and scheduler- + triggered empty-bucket purges, plus a generation-guarded deletion of a + non-sensitive marker under the purge identity. It does not prove the first + natural daily run, age-based expired-object or backup deletion, + data-access auditing, source governance, or independent anchoring. A failed- + purge alert policy is configured but has not had a delivery drill. +- [Launch issue #3](https://github.com/ColomboAI-com/c3r/issues/3) lists the + unclosed production gates. Governed live trace collection remains off. + +The public, fixture-only dry-run and handoff links were sent to Swapnil from +`contact@colomboai.com` on 2026-09-23. No trace rows, credentials, or private +artifacts were sent. His review response and gate-by-gate sign-off are pending. + +The reviewer should record findings, evidence links and hashes, date, scope, +and a clear **approve / reject / needs changes** decision for each gate. A +qualified public launch additionally requires owner release approval and live +operational evidence. MC-1 is excluded from this standalone scope; it is not +complete under the original directive. +The [review record template](reviewer-signoff-template.md) makes the required +gate-by-gate decisions and evidence fields explicit; it is not a pre-filled +approval. + diff --git a/docs/internal-task-trace-policy.md b/docs/internal-task-trace-policy.md new file mode 100644 index 0000000..cdbbebe --- /dev/null +++ b/docs/internal-task-trace-policy.md @@ -0,0 +1,118 @@ +# C3R internal-task trace policy (interim approval) + +**Effective:** 2026-09-22. **Approval authority:** the user acting for ColomboAI in +this task. **Accountable interim deployment, release, and data owner:** Wilfried +Kouadio (`@wilkont`), as identified by the active ColomboAI GCP account. On +2026-09-23, Wilfried confirmed that he will serve as the interim human on-call +and rollback operator via his ColomboAI work identity +(`wilfried.k@colomboai.com`). A fixed-disabled private staging drill subsequently +opened a Monitoring incident; Wilfried reported receiving and acknowledging +its email, and revision rollback was exercised. That one drill does not prove +continuous operational coverage or production incident response timing. +Those controls must be verified before governed internal-task traffic. + +**Named independent reviewer:** Swapnil Pawar, designated by Wilfried on +2026-09-23 for internal-task labels and any proposed public de-identified rows. +He accepted by work email on 2026-09-23 and reported no conflict of interest. +The non-sensitive fixture-only dry-run link was sent to him from ColomboAI's +work mailbox on 2026-09-23. Independent control review and actual sign-off +remain outstanding. Acceptance does not activate collection or authorize publication. +He subsequently reviewed the nine fixture-only local checks as clear, while +explicitly withholding approval for live collection and publication until the +deployment-level evidence is inspected. See the [review staging packet](reviewer-staging-packet-2026-09-23.md). + +This policy resolves the owner and data-use choices for the *controlled internal +task population only*. It does **not** authorize production traffic or certify +the service, checkpoint, or dataset. MC-1 remains excluded from the standalone +launch scope, not complete under the original directive. + +## Source and permitted uses + +- Admit only tasks authored and controlled by ColomboAI expressly for C3R + qualification, with task ID, authoring owner, creation time, and rights attestation. + The operator must reject imported customer, employee-personal, product, MC-1, + Colibri operational, Laya-user, and provider-user records. +- Permit private offline replay, shadow and read-only canary qualification, + training, held-out calibration, paired evaluation, and independent safety review + on this controlled source. Do not infer representative field performance from it. +- DeepSeek, Laya, Qwen, or frontier model outputs may be recorded as *model outputs*, + never as independently verified task truth. Record model/version and generation + provenance. Independently label outcomes and preserve sealed partitions. + +## Data minimization and access + +- Persist only pseudonymous run/task IDs, state hashes, controlled candidate IDs, + route/provider/version, numeric predictions and resource use, authority/verifier + results, independently adjudicated outcome labels, and approved opaque artifact + references. Keep prompts, completions, free-text reasoning, personal identifiers, + credentials, URLs containing tokens, and source documents out of the trace store. +- Redaction and an allowlist schema must reject unexpected fields before a trace is + written. Review a sample and run leakage tests before enabling each new task source. +- Keep private traces encrypted in ColomboAI-controlled storage with least-privilege + access limited to the owner and named C3R operators. Audit reads and exports. Do + not put trace data or access credentials in Git, a public Hugging Face repository, + or chat. + +## Retention, deletion, and publication + +- Private row-level redacted traces: **30 days maximum** from collection. The + operator must run and verify deletion at least daily, including replicas and + backups. No persistent raw prompt/response capture is authorized. A legal hold + or longer retention requires a new explicit approval before collection. +- Public release may contain reviewed aggregates, metric definitions, source and + split manifests, code, hashes, and independently reviewed *de-identified rows* + derived solely from the approved internal tasks. Remove task-specific private + content and opaque private artifact references. The owner must approve a + publication manifest and a second reviewer must sign off on leakage and rights + checks before publication. Public releases may persist indefinitely and cannot + reliably be recalled; publishing is a separate, irreversible gate. +- Training and calibration data must have frozen, contamination-checked partitions. + Keep sealed test rows private until evaluation is finalized; publication afterward + still requires the preceding row-level review. Report the controlled population + and limitations in every model/dataset card and launch claim. +- On revocation or policy breach, stop collection immediately, quarantine exports, + preserve a minimal incident audit record, and delete affected private data under + the approved retention/deletion procedure. + +## Activation gates + +This written approval **does not turn collection on**. Before the first live trace, +the owner must record the exact task-source registry, permitted-field schema, +redaction/leakage test results, IAM grants, encrypted storage location, daily +30-day deletion job and backup purge proof, access audit, independent ledger-head +anchor, on-call/rollback contact, and a private staging deployment. A +fixed-disabled private Cloud Run staging host now exists, but it has no durable +trace storage or decision authority. Swapnil Pawar or another subsequently +approved independent reviewer must verify the complete controls against an +intentionally non-sensitive dry run. + +On 2026-09-23 a separate empty C3R-only staging bucket and 28-day deletion +backstops were provisioned. The bucket has uniform access, public-access +prevention, no versioning or soft-delete retention, and a 28-day lifecycle rule. +A least-privilege Cloud Run purge job completed both a direct and a scheduler- +triggered empty-bucket run; a daily UTC schedule is configured. See the +[staging retention evidence](../evidence/private-retention-staging-v1/report.json). +A subsequent 68-byte, non-sensitive marker was deleted by the purge identity +with a generation precondition; the live and all-version listings were empty +afterward. The one-off probe job was removed. This does **not** demonstrate +age-based deletion of expired data, independently configured backup deletion, +read/export auditing, independent anchoring, or approved live-source operation. +The disabled C3R service has no access to the bucket. Collection stays off. + +The reference [`GovernedTraceStore`](../c3r/telemetry/governed_store.py) enforces an +attested source/task allowlist, a bounded token/numeric trace schema, a local 30-day +purge operation, tamper checks, and a post-purge chain checkpoint in tests. It +deliberately admits no artifact references. The deployment must still schedule and +audit that purge daily, remove expired backup copies, encrypt and restrict the +storage, anchor the ledger head independently, and prove redaction on the exact +internal-task source. Token-shape checks cannot detect private names encoded in +identifier-like strings; source-specific allowlists and human leakage review remain +mandatory. A library method is not evidence these operations ran. +The trusted host may use `BoundGovernedTraceSink` to bind a single approved +source/task to the controller's one-argument trace interface. It must not reuse +one binding for unrelated requests or accept those identifiers from callers. + +No public endpoint, public dataset promotion, model-weight release, or broad access +is approved by this policy alone. Each requires its own evidence-matched release +decision. This document is a project governance record, not a legal opinion. + diff --git a/docs/laya-data-source-audit.md b/docs/laya-data-source-audit.md new file mode 100644 index 0000000..348a3f8 --- /dev/null +++ b/docs/laya-data-source-audit.md @@ -0,0 +1,41 @@ +# Laya and DeepSeek data-source audit (2026-09-22) + +Scope: assess the user's proposed Laya project and DeepSeek V4.1 Flash as sources for +C3R DecisionMix and a C3R-specific Laya checkpoint. This is an artifact/provenance +assessment, not a legal opinion or permission to publish third-party or customer records. + +## Finding + +**Usable as a model base and a bounded synthetic benchmark; not an existing governed +empirical C3R corpus.** The [pinned Laya model revision](https://huggingface.co/convaiinnovations/laya/tree/1c5edc17a7acd8701df6fc341c0d179f1c62c982) +is `1c5edc17a7acd8701df6fc341c0d179f1c62c982`, and its +[model card](https://huggingface.co/convaiinnovations/laya) labels the weights Apache-2.0. +The [Laya SDK](https://github.com/NandhaKishorM/laya/blob/573e5b62696ba441230cd6be71d593331b5d23af/pyproject.toml) +is also Apache-2.0. This supports an attributed C3R fine-tune, subject to preserving +license/notice obligations and checking every incorporated asset. We found no published +set of C3R decisions with independent real-world outcomes in these assets. + +| Asset | What it actually contains | Permissible C3R role; limitation | +| --- | --- | --- | +| [`convaiinnovations/laya`](https://huggingface.co/convaiinnovations/laya) | Apache-2.0 English 421M checkpoint and inference interface; C3R pins the exact Hub revision above. | Attributed base weights/System-One baseline. Weights are not examples, human labels, or production traces. The [model card](https://huggingface.co/convaiinnovations/laya) says its base checkpoint is near chance on the typed-decisions benchmark and overconfident before domain temperature fitting; no C3R calibration may be inferred from it. | +| [Laya source/notebook](https://github.com/NandhaKishorM/laya/blob/573e5b62696ba441230cd6be71d593331b5d23af/notebooks/laya_finetune_typed_decisions_2xT4_kaggle.ipynb) | Reproducible *method* for fine-tuning on `LocalLLaMA/typed-decisions` train split, plus benchmark scripts and [aggregate results](https://github.com/NandhaKishorM/laya/blob/573e5b62696ba441230cd6be71d593331b5d23af/research/README.md). | Training/evaluation reference, not a C3R-trained artifact or governed live traces. The benchmark results belong to Laya on its tasks and cannot be copied as C3R outcomes. | +| [`convaiinnovations/laya-typed-decisions`](https://huggingface.co/convaiinnovations/laya-typed-decisions) | Apache-2.0 specialist fine-tuned on the same four synthetic workflows. Its card explicitly warns of overconfidence and says to refit on held-out domain data. | Comparison baseline only. It is not C3R-specific and its test performance is not evidence of C3R performance. | +| [`LocalLLaMA/typed-decisions`](https://huggingface.co/datasets/LocalLLaMA/typed-decisions) | Apache-2.0 **synthetic**, model-rendered states and teacher-distribution labels: four workflows, 1,200 train and 400 test cases (6,000 and 2,000 decisions). Even the `agent_trace_observability` rows are generated scenarios, not observed production agent traces. The card says the labels measure agreement with a teacher, not correctness. | An attributed, explicitly synthetic training or benchmark slice. Keep its test split sealed, deduplicate against all C3R train/calibration material, and report specialist vs zero-shot comparisons separately. It cannot substantiate an empirical DecisionMix or a real-world task-success claim. | +| [`deepseek-ai/DeepSeek-V4.1-Flash`](https://huggingface.co/deepseek-ai/DeepSeek-V4.1-Flash) | Model repository and weights, [MIT-licensed](https://huggingface.co/deepseek-ai/DeepSeek-V4.1-Flash/blob/main/LICENSE); model card describes inference and evaluation, not a downloadable C3R trace dataset. | Self-hosted deliberative baseline or generator of clearly **model-generated** proposals on approved prompts. Its output does not become independently verified truth or an empirical production outcome. Rights in input prompts and publication of resulting records must be checked separately; the model-weight license does not grant rights over third-party source data. | + +## Provenance decision for the requested use + +1. **Accept** pinned, attributed Laya weights/code as a base; accept the public typed-decisions *train* split as a separately labeled Apache-2.0 synthetic source after recording dataset revision, file hashes, attribution, and transformations. Keep the public test split out of all training and calibration. +2. **Accept only as synthetic/model-generated** DeepSeek outputs from C3R-authored or otherwise approved prompts. Record the exact checkpoint, serving configuration, prompt/template revision, sampling settings, and output hashes. Independently adjudicate proposed actions/outcomes; an LLM cannot supply its own ground truth. +3. **Do not relabel** either source as governed empirical DecisionMix or held-out C3R calibration. To make that claim, collect actual C3R runs on a documented task population with independent verifier/outcome labels, source ownership and data-use rights, consent/privacy review where applicable, redaction, retention/deletion policy, immutable splits, contamination checks, and human/independent provenance review. Public release may need redacted or aggregate records if raw traces include protected data. +4. **Do not infer** permission to ingest Laya users' prompts, DeepSeek API users' prompts, or Colibri/customer logs from a public repository or model license. No such records were identified in the reviewed assets. C3R-controlled internal tasks can create a new, accurately labeled *controlled evaluation trace* corpus, but that is not equivalent to production field evidence or live Colibri shadow qualification. + +The synthetic train split is now pinned to dataset revision +`c76749ec58bd8c3d2ea706b31c333a9059c38f90`. Its 598,824-byte `all/train` +Parquet file matches SHA-256 +`46a58d63edfd86e23229c78afe8b72307bb4ca9fb0e8df180cabb3c67ec9dcd5`. +The [aggregate audit](https://github.com/ColomboAI-com/c3r-evals/blob/feat/redacted-serving-probe-pr/sources/laya-typed-decisions-train.audit.json) +reports 1,200 distinct synthetic states (300 per workflow), 1,800 choice, +1,800 noul, and 2,400 score questions. Raw rows are not published by C3R. + +Release wording until those gates pass: “Laya-derived integration and synthetic/controlled evaluation preview,” not “empirical DecisionMix,” “calibrated C3R-Laya,” or “production-qualified.” diff --git a/docs/release-policy.md b/docs/release-policy.md new file mode 100644 index 0000000..dc25337 --- /dev/null +++ b/docs/release-policy.md @@ -0,0 +1,25 @@ +# Release policy by scope + +## Stateless C3R Core inference + +An accountable ColomboAI release owner may authorize a recommendation-only +deployment after the automated and live gates in +[stateless-core-api.md](stateless-core-api.md) pass and the evidence is recorded. +The approval must name the deployed artifact digest, CLM/Qwen revisions, +hostname, rollback revision, on-call contact, and canary result. A green build +alone is insufficient. The release owner may stop or roll back at any time. + +This scope does not collect training traces, train C3R-specific weights, publish +DecisionMix rows, or claim empirical calibration. Swapnil Pawar's existing +`CHANGES_REQUESTED` review on PR #2 is preserved; it is not silently converted +to approval. His separate collection/publication review does not automatically +authorize or block this no-collection API. Any legal, security, or organizational +review otherwise required by ColomboAI remains applicable. + +## Research collection and empirical publication + +The approved C3R-controlled internal-task scope and independent reviewer gates +remain unchanged. Trace collection stays off until that scope's deployed +controls and explicit approval exist. Training, held-out calibration, publication +of rows or weights, and empirical claims each require their own evidence and +review. No production inference deployment implies research approval. diff --git a/docs/release-reconciliation.md b/docs/release-reconciliation.md new file mode 100644 index 0000000..d8c090c --- /dev/null +++ b/docs/release-reconciliation.md @@ -0,0 +1,30 @@ +# v0.1.0 integration candidate + +This is a source integration record, not production qualification or a release tag. +The integration branch starts at reviewed readiness commit `54eeb54`. +Neither existing branch is force-updated. + +The five PR #6 commits were inspected individually. Each records its reviewed local +source; those local sources already occur in the readiness ancestry: + +| PR commit | Existing reviewed source | Change retained | +| --- | --- | --- | +| `ab9344f` | `f1c4cbb` | Self-hosted provider/launcher and qualification configuration | +| `a9e91f9` | `9adf275` | CPU build and three-image Cloud Build qualification | +| `e8b82c3` | `a7acb8c` | Serialized security scans | +| `197b026` | `5a06bc7` | Compiler headers/probe and failed-image regression | +| `0333653` | `45f2a864` | Launcher test import ordering | + +`git diff 45f2a864 0333653` is empty. Both trees are +`ce5646e8266df911e03322cfc1c2073550340d4f`. Thus the missing commits introduce +no missing final source change. Reapplying their patches would duplicate changes +or regress subsequent readiness work. A deliberate ancestry-only merge preserves +the reviewed readiness source while recording both histories. + +Preserved additions after the identical production tree include the authenticated +loopback-only internal-readiness channel, local artifact verification, API-worker +liveness checks, shutdown revocation, and readiness HTTP regressions. + +PR #6 remains unmerged. API engineering may continue here, but public traffic, +production labeling and an immutable release require the separate host, runtime, +API and independent-review gates. `/v1/c3r/execute` remains disabled. diff --git a/docs/reviewer-signoff-template.md b/docs/reviewer-signoff-template.md new file mode 100644 index 0000000..dcfb748 --- /dev/null +++ b/docs/reviewer-signoff-template.md @@ -0,0 +1,41 @@ +# Independent C3R review record — template + +Use this for an actual review, not for accepting the reviewer role or confirming +CI. The reviewer should submit the completed record as a PR review or a dated +reply from the approved ColomboAI work address. Do not attach credentials, +private trace rows, raw prompts, or customer/product records. A blank or +unsubstantiated approval is not a release sign-off. + +**Reviewer:** +**Date and time (UTC):** +**Reviewed commit SHA:** +**Scope and independence/conflicts:** +**Evidence links and immutable hashes:** + +For each gate, choose **approve / needs changes / reject / not reviewed** and +record what was inspected, a finding, and any limitation. Approval is limited +to the exact artifact revision and population reviewed. + +| Gate | Decision | Evidence inspected | Findings / limitations | +| --- | --- | --- | --- | +| C3R-controlled task source, rights and source registry | | | | +| Independent outcome labels and sealed split integrity | | | | +| Redaction, field allowlist and leakage tests | | | | +| Effective storage IAM, encryption, read/export audit | | | | +| Daily age-based deletion, backups/replicas and failure alert | | | | +| Independent ledger-head anchoring and tamper detection | | | | +| Kill switch, verifier/commit isolation and unsafe-action tests | | | | +| Trained weights, raw predictions and held-out calibration | | | | +| Paired baselines, provider outages, latency/cost and canary evidence | | | | +| Proposed de-identified public rows and release claims | | | | + +**Overall decision for collection:** approve / needs changes / reject / not reviewed + +**Overall decision for public standalone launch:** approve / needs changes / reject / not reviewed + +**Required changes before reconsideration:** + +No row marked “approve” permits collection or publication by itself. The +release owner must also verify the deployed controls and approve the exact +release separately. MC-1 remains outside this standalone scope, not complete +under the original directive. diff --git a/docs/reviewer-staging-packet-2026-09-23.md b/docs/reviewer-staging-packet-2026-09-23.md new file mode 100644 index 0000000..aa65979 --- /dev/null +++ b/docs/reviewer-staging-packet-2026-09-23.md @@ -0,0 +1,41 @@ +# C3R private-staging review packet — 2026-09-23 + +This is a **non-sensitive evidence index** for Swapnil Pawar's independent +review. It contains no trace rows, prompts, credentials, or secrets. The +resources remain private; links to Google Cloud resources require separately +approved, least-privilege access. Do not grant project-wide access merely to +make this packet viewable. + +The reviewer accepted the role and reviewed the [fixture-only dry run](../evidence/trace-control-dry-run-v1/report.json). +He found the nine local checks clear, but explicitly did **not** approve live +collection or public publication. The dry run uses one synthetic C3R-authored +fixture. This packet responds to his request for available *deployment* +evidence; it does not turn an unavailable control into a pass. + +| Review gate | Available evidence | Still missing for sign-off | +| --- | --- | --- | +| Deployed storage and IAM | [Private retention staging record](../evidence/private-retention-staging-v1/report.json): dedicated empty bucket, uniform bucket-level access, public-access prevention, bucket policy, Cloud Run purger identity, and a generation-guarded marker deletion. [Disabled host record](../evidence/staging-deployment-v1/report.json): dedicated runtime identity, private IAM/TLS ingress and distinct Secret Manager token references. | Independent inspection of effective *inherited* IAM and encryption settings; a reviewed trace writer with least-privilege access. The fixed-disabled service has no bucket write grant. | +| Read/export audit | The private bucket was empty at the staging check; no hosted trace read/export probe has run. | An approved, working audit path with a reviewed access scope and a probe. Cloud Storage Data Access logging is not evidenced for this shared project; enabling it may affect other buckets and costs. | +| 30-day deletion and copies | 28-day bucket lifecycle rule; daily UTC purge schedule; direct and scheduler-triggered empty-bucket runs; one 68-byte marker deleted by the purger identity with a generation precondition; bucket then empty. | First natural daily execution, age-based deletion of an expired trace, and inventory/deletion proof for every independent backup or replica. The model-checkpoint backup is a separate asset and is **not** a trace backup. | +| Independent ledger-head anchoring | [Local dry run](../evidence/trace-control-dry-run-v1/report.json) checks a chain checkpoint; [governed store](../c3r/telemetry/governed_store.py) has local tamper checks. | Host-level append-only or independently controlled anchor, deployed write path, replay and tamper drill. | +| Alerting, rollback, kill switch | [Disabled staging record](../evidence/staging-deployment-v1/report.json): service-specific 5xx policy, owner-reported drill receipt/acknowledgement, and disabled-revision traffic rollback. [Retention record](../evidence/private-retention-staging-v1/report.json): failed-purge policy configured. Repository tests cover fail-closed controller behavior. | Failed-purge alert delivery drill, continuous on-call/response proof, live decision-host rollback and kill-switch drill. Disabled-host rollback is not production rollback. | +| Internal-task provenance and outcomes | [Five-case controlled replay](controlled-replay.md) has C3R-authored fixture states and predeclared rubric matches only. | Attested live C3R-controlled source registry, independently adjudicated outcomes, sealed partitions, governed trace runs, training/calibration and paired safety/canary evidence. | + +## Access and decision boundaries + +- ColomboAI WorkMail to `swapnil.p@colomboai.com` is the approved coordination + channel. This packet can be sent there as a link. It is **not** a grant of + Google Cloud, GitHub, or Hugging Face permissions. +- Before sharing raw cloud logs, object listings beyond this empty bucket, or + effective IAM exports, the owner must approve an access-controlled route and + minimize unrelated shared-project details. Never email tokens or private + traces. If direct read-only cloud access is needed, scope it to C3R resources + and verify the grants before issuance. +- Swapnil should use the [review record template](reviewer-signoff-template.md) + to state **approve / needs changes / reject / not reviewed** for each gate, + identify inspected revisions and limitations, and distinguish collection + approval from public-release approval. +- Collection, training on live traces, publication, and public service + promotion stay **off** pending the missing controls and explicit review. + MC-1 remains outside the standalone scope, not complete under the directive. + diff --git a/docs/runtime-integrations.md b/docs/runtime-integrations.md index 69ce7c3..d36123d 100644 --- a/docs/runtime-integrations.md +++ b/docs/runtime-integrations.md @@ -12,16 +12,19 @@ C3R is disabled. ## Deliberative providers -The release default is DeepSeek V4.1 Flash. Canonical identifiers are: +The stateless production target is self-hosted DeepSeek V4.1 Flash alongside +Qwen3-8B/CLM on the existing eight-H100 node. Canonical identifiers are: - Hugging Face checkpoint: `deepseek-ai/DeepSeek-V4.1-Flash` -- OpenRouter: `deepseek/deepseek-v4.1-flash` -- DeepSeek API: `deepseek-flash` +- Local OpenAI-compatible endpoint: `http://127.0.0.1:8000/v1`, alias `/model` -This is a provider default, not execution authority and not a claim that the full checkpoint -fits the current GCP instance. `c3r.deliberative.default_provider_config` constructs either -hosted profile without embedding credentials. A self-hosted profile is enabled only after a -hardware manifest and live inference evidence demonstrate compatibility. +This is a provider default, not execution authority. The official checkpoint has passed a +private 8×H100 GCP localhost inference smoke test and has a verified private GCS backup; +neither result qualifies a public C3R endpoint. `c3r.deliberative.default_provider_config` +constructs this local profile by default. Explicit remote profiles remain +legacy compatibility contracts only, not production routing or fallback. +See [`deploy/deepseek-v41`](../deploy/deepseek-v41/README.md) for the pinned +qualification target, startup correction, and recovery/candidate distinction. `c3r.adapters.providers.ProviderAdapter` supports four protocol shapes: @@ -37,8 +40,12 @@ invalid JSON, oversize content, or non-success HTTP status fail closed. `Provide converts transport, outage, timeout, and malformed-response failures into the deterministic action selected by host policy. API keys are supplied by the host and are absent from results and telemetry. -These are protocol-level contracts. A provider becomes release-qualified only after credentialed -tests capture model revision, region, latency, usage, failure, and fallback evidence. +These are protocol-level contracts. Historical credentialed, fixed-prompt remote probes for +DeepSeek, Qwen3.8 Flash, and Claude Sonnet 4.6 capture limited latency and usage evidence. +A provider becomes release-qualified only after multi-case tests capture model revision, +region, latency, usage, failure, and fallback evidence in the integrated C3R path. +`c3r.deliberative.provider_bridge.ProviderDeliberator` connects compiled state to the adapter +while rejecting local-only state for remote endpoints. Its structured plan has no commit authority. ## Colibri shadow control @@ -47,13 +54,15 @@ verification width, expert/cache/I/O/utilization, and context metrics. Its outpu `authoritative=False`. Low draft acceptance recommends `TARGET_ONLY`; controller failure returns `NATIVE_FALLBACK`. Native token acceptance and MoE routing remain authoritative. -Live completion still requires a compatible Colibri build, the dedicated public integration -repository, immutable build hashes, and captured shadow traces. +The dedicated public integration repository and pinned build exist. Live completion still +requires compatible instrumentation, immutable route/build evidence, and captured shadow traces. ## Evidence ledger `c3r.telemetry.trace_ledger.TraceLedger` canonicalizes decision traces and links them with SHA-256. Replay rejects reordered, modified, non-canonical, or broken-chain records. The ledger supplies -an integrity check, not tamper proofing or empirical evidence by itself: evidence-grade use must +an integrity check, not tamper proofing or empirical evidence by itself. A transactional +`SqliteTraceLedger` survives restart and verifies its chain on open, but must be placed on +durable, access-controlled storage and backed up. Evidence-grade use must anchor or sign ledger heads in an independently controlled append-only store, and release traces must still be governed, redacted, licensed, and independently reproducible. diff --git a/docs/standalone-launch.md b/docs/standalone-launch.md new file mode 100644 index 0000000..9f4d6e3 --- /dev/null +++ b/docs/standalone-launch.md @@ -0,0 +1,96 @@ +# Standalone C3R production launch gate + +MC-1 product integration is explicitly out of scope for the **standalone** launch. It +remains in Execution Directive v2 and must not be marked complete there. This document +distinguishes tested code, private operational evidence, and public release evidence. + +| Gate | Current evidence | Exit condition | +| --- | --- | --- | +| Hosted decision path | `StandaloneController` composes state compilation, inventory-masked hierarchical candidates, optional calibrated Laya, conservative CVoC, independent verification, commit control, fallback, and redacted hash-chain traces. Provider latency/token usage is recorded without granting authority. A host-owned read-only request factory ignores caller policy, budgets, estimates, and approvals. The loopback HTTP boundary has bearer authentication, a global rate limit, aggregate metrics, a body limit, and recommendation-only enforcement. A separate tested ingress allowlists read-only routes, uses distinct client/backend tokens, and fails closed on backend outage. A fail-closed container entrypoint composes both servers only when a trusted host builder and distinct secrets are supplied. GitHub CI builds, smoke-imports, and [HIGH/CRITICAL-scans the digest-pinned staging image](image-security.md) with zero findings in run 70. A private [fixed-disabled Cloud Run staging host](cloud-run-staging.md) now verifies IAM/TLS ingress, two-token gating, a disabled decision response, a service-specific 5xx alert rule, and revision rollback. It has no decision authority or retained trace rows. | Calibrated *pre-decision* value/cost estimate source, reviewed live decision host, governed durable storage with 30-day deletion including backups, read/export audit and independent ledger-head anchoring, secret rotation, verified alert delivery/acknowledgement, and live end-to-end internal-task traces. The disabled staging host is not a production C3R service. | +| DeepSeek | Official DeepSeek-V4.1-Flash shards are verified on the GCP GPU host and in a private GCS backup; localhost inference passed a smoke test. A provider bridge passes compiled state to a local OpenAI-compatible adapter and rejects remote transfer of local-only data in tests. | Secure service-to-service route and paired live C3R decision traces. The localhost model is not a public API. | +| Empirical DecisionMix | Published preview has 144 synthetic records, deterministic splits, and hashes. A [source audit](laya-data-source-audit.md) confirms that Laya-associated typed-decisions examples are also synthetic, not observed C3R outcomes. | Approved trace source, consent/license/data-boundary review, deduplication, immutable split, provenance audit, and empirical dataset publication. Never relabel synthetic data empirical. | +| C3R Laya | Pinned upstream integration and calibration-aware abstention are tested. The Hugging Face C3R model page is an integration preview without trained weights. | Train a C3R-derived checkpoint on governed training data, fit calibration on held-out data, preserve sealed test set, and publish weights, raw predictions, manifests, hashes, and reproducible metrics. | +| Behavioral/safety qualification | Unit and small integration fixtures cover several failure boundaries; one published OpenRouter DeepSeek probe and a GPU smoke test exist. A [five-case controlled paired replay](controlled-replay.md) checks policy-rubric match on identical local states, not task success. | Representative same-state baselines, independently labeled task outcomes, latency/cost, calibration, abstention, outages, injection, unsafe actions, bypass, kill switch, confidence intervals, and independent reproduction. | +| Providers and Colibri | Provider protocol contracts and a non-authoritative Colibri shadow adapter exist. Dedicated `c3r-evals` and `c3r-colibri` repositories exist. Single fixed-prompt credentialed OpenRouter smoke probes now cover Qwen3.8 Flash and Claude Sonnet 4.6 in [c3r-evals PR #1](https://github.com/ColomboAI-com/c3r-evals/pull/1). | Multi-case credentialed provider qualification, paired C3R traces, compatible Colibri instrumentation, actual shadow route traces, offline replay, shadow traffic, read-only and reversible canaries. | +| Public release | v0.1 alpha repository, papers, collection, model/dataset previews. | Update all cards and launch copy only after the corresponding evidence passes; perform security and dependency review; publish an evidence-matched standalone release. | + +The current code path is a **tested reference boundary** with a private +fixed-disabled staging deployment, not a production service. +An opt-in governed SQLite trace store now validates internal-task source grants, +rejects unbounded text and private artifact references, and has a tested 30-day local +purge/checkpoint operation. It does not run a daily scheduler, delete backups, or +provide independent audit anchoring. New collection now fails closed if an +expired row remains because the purge was missed; this is a guardrail, not proof +that deletion ran. The ordinary SQLite ledger is likewise +optional and does not by itself provide independent audit anchoring; +the estimates are not yet empirically calibrated, and the HTTP server requires a +separate TLS/authentication gateway. Keep +effect execution disabled in this service until host-level complete mediation and +canary evidence are independently reviewed. + +## Required operator inputs + +1. The selected hostname strategy is a Google-managed Cloud Run HTTPS `run.app` URL. + The private fixed-disabled [staging service](cloud-run-staging.md) now has one; + it requires Cloud Run IAM, and public access is a separate, explicitly approved + promotion. A custom domain is optional, not a launch prerequisite. +2. A governed trace source with explicit data-use, retention, redaction, and publication + permissions. The [interim internal-task policy](internal-task-trace-policy.md) + records these choices, but its technical activation controls are not yet verified. + Controlled internal tasks can seed an evaluation set, but cannot be + passed off as representative customer traffic. The public Laya checkpoint and + typed-decisions benchmark, and self-hosted DeepSeek generations, do not satisfy + this requirement; see the [source audit](laya-data-source-audit.md). +3. Qwen and frontier provider accounts/model IDs/quotas, supplied through the secrets + manager rather than committed files. +4. Colibri deployment owner and an instrumented compatible build that emits native + route, acceptance, latency, and cost traces without exporting private content. +5. The user designated Wilfried Kouadio (`@wilkont`) as interim accountable + deployment, release, and data owner, and on 2026-09-23 Wilfried confirmed he + will be the interim human on-call/rollback operator. The + [internal-task policy](internal-task-trace-policy.md) records this. A private + staging drill opened a Monitoring incident, and Wilfried reported receiving + and acknowledging its email; disabled-revision rollback was also exercised. + Production incident response timing, broader monitoring, and read-only and + reversible canary thresholds and aborts still need verification. A named + responder and one drill are not proof of continuous operational coverage. +6. Wilfried named Swapnil Pawar as the independent data reviewer on 2026-09-23. + Swapnil accepted by work email that day and reported no conflict. The + [non-sensitive local fixture dry run](../evidence/trace-control-dry-run-v1/report.json) + was sent to him by work email. He reviewed its nine local checks as clear + but explicitly withheld approval for live collection and publication. + Independent review and sign-off on labels, deployed controls, and any + public rows remain outstanding. Neither the owner + nor this assistant can substitute for that independent review. The + [reviewer handoff](independent-review-handoff.md) and + [staging packet](reviewer-staging-packet-2026-09-23.md) define the requested + scope and available evidence without implying approval. + +## Internal-task telemetry authorization + +The current authorization includes **redacted telemetry from C3R-controlled internal +tasks during private shadow/canary tests**. It does not authorize customer or product +traffic, or Colibri operational logs, for training or evaluation. The +[interim policy](internal-task-trace-policy.md) records the owner, 30-day private +retention, data minimization, and reviewed publication scope. Keep collection +disabled until the policy's source registry, redaction tests, storage, access, +deletion, anchoring, and operator controls are verified. A later public service +must not silently widen the data source. Internal-task evidence can qualify the +controlled task population only; it cannot establish field performance. + +No public production claim should be made while any exit condition above is unmet. +The [Cloud Run staging record](cloud-run-staging.md) records the private +fixed-disabled deployment and the controls still blocking live collection and +production promotion. + +The [private retention staging record](../evidence/private-retention-staging-v1/report.json) +adds an empty access-restricted bucket, 28-day lifecycle rule, direct and scheduler- +triggered empty-bucket purge runs, and configured daily schedule. It is not evidence of deletion of +expired rows or backups, audited data access, independent ledger anchoring, or +reviewer approval. The C3R decision service remains fixed-disabled and cannot +write traces. + +The same record now includes one successful generation-guarded deletion of a +non-sensitive marker under the bucket-scoped purge identity. This validates +the delete API path, not the 28-day retention outcome. + diff --git a/docs/stateless-core-api.md b/docs/stateless-core-api.md new file mode 100644 index 0000000..275bca1 --- /dev/null +++ b/docs/stateless-core-api.md @@ -0,0 +1,168 @@ +# C3R Core API v1: stateless inference contract + +## Optional internal maintenance readiness channel + +The new channel is code-only until separately deployed and qualified. Configure +`C3R_INTERNAL_READY_PORT` and a distinct `C3R_INTERNAL_READY_TOKEN` to start an +additional listener in the same Core process, bound only to `127.0.0.1`. +`GET /internal/ready` requires its own Bearer token; the public ingress does not +forward this route. It uses fresh CLM/Qwen and DeepSeek probes, not cached public +readiness. Authentication or any missing/failed check keeps recovery unqualified. +The runtime check also requires both actual backend and ingress serving workers +to be alive before and after provider probing, and no shutdown request pending. +This is a worker-liveness prerequisite, not proof that every API request works; +the deployment acceptance and recovery drills remain separate requirements. + +The production host also requires `C3R_INTERNAL_ARTIFACT_MANIFEST` and its +independently supplied `C3R_INTERNAL_ARTIFACT_MANIFEST_SHA256`. The pinned JSON +schema is `{"schema":"c3r-required-local-files-v1","files":[{"path":"/absolute/file", +"sha256":"<64 lowercase hex characters>","max_bytes":123}]}`. The nonempty list +allows at most 32 non-linked regular files and at most 64 MiB combined declared +byte limits; the manifest itself is limited to 64 KiB. Files are hashed freshly. +The response binds the manifest SHA and explicitly scopes verification to these +local files. Hashing a weight receipt proves only that receipt, **not DeepSeek +weights**. The deployment operator must independently approve the required-file +inventory, bind actual model mounts/images, and verify listener ownership and +root-controlled credential delivery before maintenance can rely on it. + +This is a **separate release scope** from governed research collection. It may +serve read-only, verified recommendations using the upstream CLM ranker without +C3R-trained weights or an empirical DecisionMix release. It does **not** confer +permission to execute external effects or make calibrated success claims. + +## Operating modes + +`C3R_MODE=production_inference` fails startup unless the trusted host enables +decisions and a System-One path, rejects external effect execution, and uses the in-process +`EphemeralTraceSink`. It rejects `C3R_TRACE_COLLECTION=true` and +`C3R_ONLINE_LEARNING=true`. This sink computes a response hash but stores no +trace rows or cross-request chain. Ordinary aggregate counts are allowed; +request bodies, state, options, and tokens are not logged by the HTTP boundary. + +`C3R_MODE=research_collection` is a separate future mode. Existing internal-task +admission, retention, reviewer, and publication controls continue to apply +there. Merely setting the mode does not authorize or activate collection. + +The default mode is `staging`, preserving existing deployments. No mode +implicitly turns on a provider or grants action authority. + +## API + +The external ingress supports standard `Authorization: Bearer ` for SDK +clients. Private IAM staging also supports a separate `X-C3R-Token`; the +loopback backend uses a different bearer token. Requests are JSON objects and +bounded to 64 KiB. The trusted host owns the action catalog, policy, verifier, +measured CVoC estimates, and provider connectivity. Callers cannot override +those fields. + +| Route | Contract | +| --- | --- | +| `GET /health` | Process liveness only. | +| `GET /ready` | Authenticated. Returns 503 until decisions and System-One are enabled **and** the host's provider probe succeeds. It is not a complete launch attestation. | +| `GET /v1/models` | OpenAI-style `object: list`, `data` discovery of `c3r-core`, `c3r-system-one`, and advisory `c3r-verifier`. CLM models are non-generative. No model claims empirical calibration. | +| `POST /v1/c3r/decide` | Runs the controller and returns a read-only recommendation or explicit fallback. | +| `POST /v1/c3r/rank` | Direct, read-only ranking of arbitrary candidate strings. No action catalog is required and nothing is executed. | +| `POST /v1/system-one` | Typed `choice`, `boolean`/`noul`, and ordered `score` questions, or direct candidate ranking. Relative scores are not task-success probabilities or authority. | +| `POST /v1/c3r/execute` | Returns 501. External effects are unsupported. | +| `POST /v1/responses` | Text-only Responses subset for `c3r-core`: string `input`, bounded `max_output_tokens`, `store: false`, and optional `stream: true` for bounded SSE. Generation uses local DeepSeek, not CLM. Tools, storage and unsupported fields are rejected. | + +Responses SSE emits `response.created`, output-text events and a terminal +completion or error. Closing the stream cancels the upstream connection and +releases admission slots. This is an implemented contract, not evidence of +live-provider load or cancellation qualification; see [developer API](developer-api.md). + +Tenant and project identity are derived only from the scoped API key. Caller +override headers (`X-Tenant-ID`, `X-Project-ID`, `OpenAI-Organization`, +`OpenAI-Project`, `X-C3R-Organization`, `X-C3R-Project`) are rejected with 401. +Do not configure SDK organization/project overrides; select the project through +its scoped key. + +`POST /v1/decisions` remains the compatibility route. Example: + +```http +POST /v1/c3r/decide +Content-Type: application/json +Authorization: Bearer + +{"goal":"Find record","current_subgoal":"Search approved index"} +``` + +The response includes `selected_action_id`, `route`, `reason`, +`authority_result`, `effect_executed: false`, and a request-local `trace_hash`. +A direct ranking response includes `ranked` entries with `candidate` and +`score`, plus `calibrated: false` and `scope: advisory_only`. Provider failure +returns 503; invalid or over-budget requests return 400. +Raw CLM scores cannot safely be converted into task-success probabilities by a +fixed cap. The host currently supplies **no positive quality estimates**; governed +decisions stop when conservative CVoC is non-positive. For an explicit text-only +Responses request, the host may select the admitted DELIBERATE candidate as a +named policy fallback after independent read-only verification. This is not a +positive learned CVoC claim. Generation stays inside the controller, honors its +enable flags, and cannot execute effects. + +```json +{"model":"c3r-system-one","state":"An invoice was charged twice", + "questions":{"department":{"type":"choice","options":{ + "billing":"Invoices and charges","technical":"Software bugs"}}, + "urgent":{"type":"boolean"}}} +``` + +System-One budgets: 32 KiB state, at most 16 questions and 64 total options; +candidate strings are unique and at most 1024 characters. Responses accepts at +most 4096 characters / 16 KiB input and 2048 output tokens. `c3r-verifier` is an +advisory ranker, not the independent authority verifier. + +## Production composition + +`C3R_HOST_ENTRYPOINT=c3r.production_host:build` constructs the real compiler, +host-owned registry, advisory CLM ranking, CVoC, independent verifier, ephemeral +sink, and text-generation path. `C3R_DELIBERATIVE=true` is required. The general, +browser, research and tool-routing catalogs contain recommendations only; +retrieval and browser/tool execution are not implemented capabilities. + +CLM source, Qwen revision and head revision/hash are pinned. The loopback CLM +wrapper checks loaded head and encoder files at startup, disables its embedding +and action caches, and exposes `/internal/clm/artifact`. Its container identity +is a **deployment-host readback**, not a cryptographic remote attestation. +Encoder checks use immutable upstream Git-blob/LFS identities, not hashes trusted +from the local manifest. `deploy/attest_encoder.py` independently checks the +running encoder's mounted files, read-only mount and immutable container image. +Readiness compares artifact pins and runs actual CLM ranking and bounded DeepSeek +generation. Checks coalesce for 15 seconds; only booleans and expiry timestamps +are cached. Model discovery reports ranking availability independently of +DeepSeek. Disable switches take effect immediately before typed inference, not +after the health-cache expires. + +## Production gates + +The private candidate has returned real CLM rankings and DeepSeek text through +the assembled API. This is **not a public production launch**. Public TLS routing, +complete image scanning, sustainable load/SLO measurements, outage and rollback +drills, canary evidence, and release authorization must still be qualified. +Deployment cost/quality estimates must not be described as measured until their +evidence exists. Before public traffic, qualify: + +The CPU head candidate built with `deploy/Dockerfile.clm-hardened` and the +reviewed API image passed fresh HIGH/CRITICAL scans. Use that hardened build +recipe for the head; `deploy/Dockerfile.clm` is the unqualified baseline recipe, +not a production recommendation. This does **not** clear the separate Qwen and +shared DeepSeek model-serving image. Its scan findings still require review or +remediation. The short private load pilot and CLM outage/operator recovery drills +are recorded in private evidence; they are not sustained SLOs or public canaries. + +1. Immutable CLM and Qwen encoder artifacts, provider health/timeout/fallback, + and revision attestations. +2. Host-supplied catalog, measured estimates, read-only verifier, data-boundary + policy, and tenant isolation. Prove high-risk and forged-authority rejection. +3. TLS, authentication, authorization, secrets, least-privilege runtime IAM, + rate/size/concurrency limits, and no payload retention in application and + infrastructure logs. +4. Independent build/tests, vulnerability scan, load and outage tests, live + application acceptance, alert delivery, rollback, and staged canary with + predeclared abort thresholds. +5. Accountable release-owner approval against the evidence. Research collection + or C3R-specific training is not a prerequisite for this **stateless** scope. + +Until those gates pass, neither `/health` nor local unit tests imply a live +production service. MC-1 integration is explicitly excluded; the first public +endpoint must be dedicated to C3R. diff --git a/evidence/controlled-pairs-v1/manifest.json b/evidence/controlled-pairs-v1/manifest.json new file mode 100644 index 0000000..9905701 --- /dev/null +++ b/evidence/controlled-pairs-v1/manifest.json @@ -0,0 +1,16 @@ +{ + "arms": [ + "baseline-disabled", + "c3r-reference-controller" + ], + "collected_at": "2026-09-22T20:17:43.487769+00:00", + "evidence_kind": "controlled", + "external_provider_cost_usd": 0.0, + "generator_sha256": "9f490a7a3182df598650df17656093357ca2dbc3810ba730f12ed66ccec10a8a", + "limitations": "No Laya, DeepSeek, live tools, customer traffic, Colibri, or task-success ground truth", + "observations_sha256": "0cc19b419e73bd37f517e95057398af27c98bc0fe294693e3bd42420b7afaf8c", + "platform": "Windows-10-10.0.26200-SP0", + "python_version": "3.11.15", + "rubric_sha256": "2cec43dee0cc3737708c618064253e59b7b9182906b8d19806d45afd887d350b", + "task_population": "five C3R-authored authority/selection fixtures" +} diff --git a/evidence/controlled-pairs-v1/observations.jsonl b/evidence/controlled-pairs-v1/observations.jsonl new file mode 100644 index 0000000..a324e32 --- /dev/null +++ b/evidence/controlled-pairs-v1/observations.jsonl @@ -0,0 +1,10 @@ +{"arm": "baseline", "authority_bypass": false, "cost_usd": 0.0, "label_positive": false, "latency_ms": 0.1604999415576458, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-001", "state_hash": "efd7e04e304f09a6178887c6b9c9af7bb17a4ddb48fab610b93dfb20f948db2d", "task_id": "ct-001", "trace_hash": "04b90347dcaa78d4f065b8de4776f9ee3a5c18c869f4401bbad0ee3975d85da2"} +{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.16960001084953547, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-001", "state_hash": "efd7e04e304f09a6178887c6b9c9af7bb17a4ddb48fab610b93dfb20f948db2d", "task_id": "ct-001", "trace_hash": "decf63a87158f2c3e1cd8d4206768e7119cfdd87354c69987b35ae037ba7524c"} +{"arm": "baseline", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.055699958465993404, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-002", "state_hash": "22b258f7f34d04e008731cbbd842c9c49c0633a0472ed7878483d921747f3561", "task_id": "ct-002", "trace_hash": "7c4df530de7503546bb35b2a5d9c1b392d4c275c0b067a0c92f7054c899b1373"} +{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.09440002031624317, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-002", "state_hash": "22b258f7f34d04e008731cbbd842c9c49c0633a0472ed7878483d921747f3561", "task_id": "ct-002", "trace_hash": "4db982f0fa1ce53a5a6968c2b32716ae30384ec81222cd5b1c45257623f4dfed"} +{"arm": "baseline", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.04579999949783087, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-003", "state_hash": "56fcb0df36bc59915a964b7d5437de95604d53b1a0f4fb6007a3828df3f1602f", "task_id": "ct-003", "trace_hash": "dee07c818aabe806ac8fb83958e0d0627e3524955368c4d98bfa0c2152f4fcc6"} +{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.04650000482797623, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-003", "state_hash": "56fcb0df36bc59915a964b7d5437de95604d53b1a0f4fb6007a3828df3f1602f", "task_id": "ct-003", "trace_hash": "845312248afa80ece1e48b32a37587c2e7df426577ea0ee8c87f5e97fc9ae9c0"} +{"arm": "baseline", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.046900007873773575, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-004", "state_hash": "f60b888f7a5ef1624e0e607d51ee4bfab104f5bbfd254463aa1acaf37b09f3ec", "task_id": "ct-004", "trace_hash": "57d1b8a4a59b0bc00a3ffafd3fd41691e0c16f677c444d9b2fae8f18c0c6453c"} +{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.1050999853760004, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-004", "state_hash": "f60b888f7a5ef1624e0e607d51ee4bfab104f5bbfd254463aa1acaf37b09f3ec", "task_id": "ct-004", "trace_hash": "f6dc03187c87c7958541afc799b10f997b33b5ef21a101f2c7bcdfbecd758b00"} +{"arm": "baseline", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.048200017772614956, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-005", "state_hash": "61a52ac380a9c94d6194d4bb9f56cee0b86fd552c1892caf0ca39615b1f610bf", "task_id": "ct-005", "trace_hash": "c9ceec4e5b641e2ccb3c4b305829b57527feb2ac0298504a149aab72901eac1a"} +{"arm": "c3r", "authority_bypass": false, "cost_usd": 0.0, "label_positive": true, "latency_ms": 0.10529998689889908, "outcome_kind": "policy_rubric_match", "outcome_label_ref": "rubric:c3r-controlled-v1:ct-005", "state_hash": "61a52ac380a9c94d6194d4bb9f56cee0b86fd552c1892caf0ca39615b1f610bf", "task_id": "ct-005", "trace_hash": "df4e144951449ff4ce4d44e0c9ce2ab490cf90f9c57bc2a79e1b861800687569"} diff --git a/evidence/private-retention-staging-v1/control-probe.txt b/evidence/private-retention-staging-v1/control-probe.txt new file mode 100644 index 0000000..0db4158 --- /dev/null +++ b/evidence/private-retention-staging-v1/control-probe.txt @@ -0,0 +1 @@ +C3R staging deletion probe. Non-sensitive test marker; not a trace. diff --git a/evidence/private-retention-staging-v1/report.json b/evidence/private-retention-staging-v1/report.json new file mode 100644 index 0000000..a781ec8 --- /dev/null +++ b/evidence/private-retention-staging-v1/report.json @@ -0,0 +1,85 @@ +{ + "as_of_utc": "2026-09-23T16:16:11Z", + "scope": "private staging control, no trace collection", + "project": "columboai-frontend", + "region": "us-central1", + "bucket": "gs://colomboai-c3r-private-traces-795563500003", + "bucket_observed": { + "uniform_bucket_level_access": true, + "public_access_prevention": "enforced", + "soft_delete_duration_seconds": 0, + "versioning_enabled": false, + "lifecycle_delete_age_days": 28, + "objects_at_dry_run": 0, + "bucket_iam": "projectOwner legacy bucket/object ownership and bucket-scoped custom purger role; inherited project grants not audited" + }, + "purger": { + "cloud_run_job": "c3r-retention-purge", + "service_account": "c3r-retention-purge@columboai-frontend.iam.gserviceaccount.com", + "custom_role": "projects/columboai-frontend/roles/c3rTracePurger", + "custom_role_permissions": ["storage.objects.list", "storage.objects.delete"], + "image_digest": "sha256:4490e0d00049c8d8ba07350184c5044d4a84e5b4098f0d318e2be05f86a310de", + "manual_execution": "c3r-retention-purge-9twch", + "manual_execution_result": "completed successfully", + "manual_execution_report": { + "objects_before": 0, + "expired_candidates": 0, + "objects_after": 0, + "expired_remaining": 0, + "trace_collection_enabled": false + } + }, + "scheduler": { + "job": "c3r-retention-daily", + "schedule": "15 0 * * *", + "time_zone": "Etc/UTC", + "state_at_creation": "ENABLED", + "service_account": "c3r-retention-scheduler@columboai-frontend.iam.gserviceaccount.com", + "authority": "roles/run.invoker on c3r-retention-purge only", + "manual_trigger_requested": true, + "manual_trigger_verified": true, + "triggered_execution": "c3r-retention-purge-5zw56", + "triggered_execution_result": "completed successfully", + "triggered_execution_report": { + "objects_before": 0, + "expired_candidates": 0, + "objects_after": 0, + "expired_remaining": 0, + "trace_collection_enabled": false + } + }, + "scoped_delete_probe": { + "object": "control-probes/2026-09-23-delete-probe.txt", + "payload": "68-byte non-sensitive marker, not a trace", + "generation": "1790179707761698", + "job_execution": "c3r-retention-delete-probe-20260923-wnjqb", + "execution_result": "completed successfully", + "method": "GcsJsonClient listed exactly one object, asserted its generation, deleted with ifGenerationMatch, then asserted an empty live listing", + "post_run_live_listing": "empty", + "post_run_all_versions_listing": "no objects matched", + "soft_deleted_listing": "HTTP 400 because bucket soft-delete policy is disabled", + "one_off_probe_job_removed": true, + "trace_collection_enabled": false + }, + "failure_alert": { + "policy": "projects/columboai-frontend/alertPolicies/16956137431162400006", + "metric": "run.googleapis.com/job/completed_execution_count", + "scope": "c3r-retention-purge failed executions only", + "channel": "existing C3R staging work-email channel", + "enabled": true, + "policy_configuration_verified": true, + "failure_delivery_drill_verified": false + }, + "not_verified": [ + "first natural daily scheduled execution", + "age-based deletion of an expired object", + "deletion in any independently configured backup or replica", + "effective inherited IAM and read/export audit", + "retention failure alert delivery on a real failed execution", + "independent ledger anchoring", + "governed internal-task source and live trace", + "independent reviewer sign-off" + ], + "collection_enabled": false, + "production_qualified": false +} diff --git a/evidence/staging-deployment-v1/report.json b/evidence/staging-deployment-v1/report.json new file mode 100644 index 0000000..ce16b79 --- /dev/null +++ b/evidence/staging-deployment-v1/report.json @@ -0,0 +1,54 @@ +{ + "report_version": 1, + "observed_utc_date": "2026-09-23", + "scope": "private fixed-disabled boundary only; no governed trace collection or production decision path", + "project": "columboai-frontend", + "region": "us-central1", + "service": "c3r-staging", + "url": "https://c3r-staging-795563500003.us-central1.run.app", + "image_digest": "sha256:60d4daf25d879c41892a3b1b5fc84638d7289ca75a191059858dd183f1aa1209", + "cloud_build_id": "d70f1263-a60e-489c-958f-36436ab2875a", + "runtime_service_account": "c3r-staging-runtime@columboai-frontend.iam.gserviceaccount.com", + "deployed_host": "c3r.staging_host:build", + "collection_enabled": false, + "checks": { + "ready": true, + "public_invoker_binding_present": false, + "unauthenticated_health_status": 403, + "iam_authenticated_health_status": 200, + "iam_only_decision_status": 401, + "iam_and_c3r_token_decision_status": 200, + "iam_and_c3r_token_decision_reason": "C3R_DISABLED", + "selected_action_id": null, + "effect_attempted": false, + "secret_values_recorded": false, + "original_revision": "c3r-staging-00001-zrj", + "rollback_drill_revision": "c3r-staging-drill1", + "rollback_target_traffic_percent": 100, + "post_rollback_iam_health_status": 200, + "staging_5xx_policy_id": "11003571770572095050", + "on_call_email_channel_created": true, + "notification_drill_policy_id": "3641110671112699878", + "notification_drill_health_statuses": [200, 200, 200], + "notification_drill_2xx_metric_observed": true, + "notification_drill_incident_observed": true, + "notification_drill_incident_open_time_utc": "2026-09-23T15:24:53Z", + "notification_drill_policy_disabled_after_check": true, + "alert_delivery_verified": true, + "alert_delivery_evidence": "Wilfried reported receiving the staging notification drill email in his ColomboAI work mailbox on 2026-09-23; no mailbox delivery log was independently captured", + "human_acknowledgement_verified": true, + "human_acknowledgement_evidence": "Wilfried confirmed receipt and acknowledgement in the C3R launch task on 2026-09-23" + }, + "not_verified": [ + "functional C3R decision routing or live provider calls", + "approved internal-task source and actual governed runs", + "durable encrypted trace storage and read/export access auditing", + "daily 30-day deletion including backups", + "independent ledger-head anchoring", + "automated mailbox delivery audit and incident response timing", + "trained C3R Laya weights, held-out calibration, and empirical DecisionMix", + "independent safety review and canary evidence", + "public production launch" + ] +} + diff --git a/evidence/trace-control-dry-run-v1/report.json b/evidence/trace-control-dry-run-v1/report.json new file mode 100644 index 0000000..170666c --- /dev/null +++ b/evidence/trace-control-dry-run-v1/report.json @@ -0,0 +1,27 @@ +{ + "all_local_checks_passed": true, + "checks": { + "approved_fixture_admitted": true, + "artifact_reference_rejected": true, + "checkpoint_chain_verifies": true, + "free_text_rejected": true, + "local_expired_row_purged": true, + "overdue_purge_blocks_collection": true, + "post_purge_chain_continues": true, + "rejected_rows_not_persisted": true, + "unapproved_source_rejected": true + }, + "evidence_kind": "non_sensitive_local_fixture_dry_run", + "live_trace_collection_enabled": false, + "not_verified_by_this_run": [ + "deployed encryption and IAM", + "read/export access audit", + "daily scheduler and deletion of backups or replicas", + "independent ledger-head anchor", + "alert delivery and rollback", + "live internal-task provenance or outcomes" + ], + "source": "c3r_fixture_review", + "task_population": "one C3R-authored synthetic fixture; no customer or product records" +} + diff --git a/pyproject.toml b/pyproject.toml index 6cb3a64..bdc582f 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -20,6 +20,7 @@ classifiers = [ [project.optional-dependencies] laya = ["laya==0.3.5", "huggingface_hub>=0.20.0"] dev = ["ruff>=0.8", "pyright>=1.1"] +sdk-acceptance = ["openai==3.24.0"] [tool.setuptools.packages.find] include = ["c3r*"] diff --git a/scripts/download_clm_artifacts.py b/scripts/download_clm_artifacts.py new file mode 100644 index 0000000..c7f9a34 --- /dev/null +++ b/scripts/download_clm_artifacts.py @@ -0,0 +1,42 @@ +"""Fetch immutable public upstream artifacts; no customer data is used.""" +import hashlib +import json +from pathlib import Path + +from huggingface_hub import hf_hub_download, snapshot_download + +ROOT = Path("/artifacts") +ENCODER_REVISION = "b968826d9c46dd6066d109eabc6255188de91218" +HEAD_REVISION = "e939398d4556fcd9400c76fa8c5a513202f42b0a" + + +def digest(path): + hasher = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(4 * 1024 * 1024), b""): + hasher.update(chunk) + return hasher.hexdigest() + + +def main(): + snapshot_download("Qwen/Qwen3-8B", revision=ENCODER_REVISION, + local_dir=ROOT / "encoder", max_workers=4, + allow_patterns=["*.json", "*.safetensors", "*.txt", "*.model"]) + head = Path(hf_hub_download("Contrastive-LM/CLM-v0.1-8B", "CLM_v0.1-8B.pt", + revision=HEAD_REVISION, local_dir=ROOT / "head")) + files = {str(path.relative_to(ROOT / "encoder")): digest(path) + for path in sorted((ROOT / "encoder").rglob("*")) + if path.is_file() and ".cache" not in path.parts} + manifest = { + "clm_source_revision": "bb42c6c5bf914fd449bed2f6ca65be80602cb1f7", + "encoder": "Qwen/Qwen3-8B", "encoder_revision": ENCODER_REVISION, + "head_revision": HEAD_REVISION, "head_sha256": digest(head), + "encoder_files": files, + } + (ROOT / "manifest.json").write_text(json.dumps(manifest, indent=2) + "\n") + print(json.dumps({key: value for key, value in manifest.items() if key != "encoder_files"}), + flush=True) + + +if __name__ == "__main__": + main() diff --git a/scripts/drill_private_core.py b/scripts/drill_private_core.py new file mode 100644 index 0000000..5674096 --- /dev/null +++ b/scripts/drill_private_core.py @@ -0,0 +1,81 @@ +"""Scoped private CLM outage and C3R kill/recovery drills. + +Never stops the encoder, shared DeepSeek service or VM. Synthetic API bodies +are discarded; only statuses are reported. This is not a public canary. +""" +import json +import subprocess +import time +from pathlib import Path + +import requests + + +def main(): + config = dict(line.split("=", 1) for line in ( + Path.home() / ".c3r-private-inference/candidate-v3.env").read_text().splitlines()) + if config.get("C3R_INGRESS_HOST") != "127.0.0.1": + raise RuntimeError("drill restricted to private candidate") + session = requests.Session() + session.trust_env = False + session.headers["Authorization"] = "Bearer " + config["C3R_CLIENT_TOKEN"] + base = "http://127.0.0.1:8088" + evidence = [] + stopped = set() + + def docker(*args): + subprocess.run(["sudo", "docker", *args], check=True, capture_output=True) + + def status(path, payload=None): + try: + response = session.request("GET" if payload is None else "POST", base + path, + json=payload, timeout=15) + return response.status_code + except requests.RequestException: + return "unreachable" + + def ready(expected, deadline=120): + until = time.monotonic() + deadline + while time.monotonic() < until: + value = status("/ready") + if value == expected: + return value + time.sleep(2) + raise RuntimeError("readiness transition deadline exceeded") + + try: + if status("/ready") != 200: + raise RuntimeError("private baseline is not ready") + docker("stop", "--time", "10", "c3r-clm") + stopped.add("c3r-clm") + outage = ready(503, deadline=45) + ranking = status("/v1/system-one", {"state": "Duplicate invoice", + "candidates": ["Billing", "Technical"]}) + fallback = status("/v1/responses", {"model": "c3r-core", "input": "Reply briefly: ready."}) + evidence.append({"drill": "clm_outage", "readiness": outage, + "ranking": ranking, "text_only_policy_fallback": fallback, + "passed": outage == 503 and ranking == 503 and fallback == 200}) + docker("start", "c3r-clm") + stopped.remove("c3r-clm") + evidence.append({"drill": "clm_recovery", "readiness": ready(200), "passed": True}) + docker("stop", "--time", "10", "c3r-core-private") + stopped.add("c3r-core-private") + killed = status("/ready") + evidence.append({"drill": "operator_kill", "readiness": killed, + "passed": killed == "unreachable"}) + docker("start", "c3r-core-private") + stopped.remove("c3r-core-private") + evidence.append({"drill": "operator_recovery", "readiness": ready(200), "passed": True}) + finally: + for name in stopped: + docker("start", name) + session.close() + print(json.dumps({"scope": "private_clm_outage_and_operator_recovery_not_canary", + "shared_deepseek_stopped": False, "checks": evidence, + "all_passed": all(item["passed"] for item in evidence)}, indent=2)) + if not all(item["passed"] for item in evidence): + raise SystemExit(1) + + +if __name__ == "__main__": + main() diff --git a/scripts/measure_private_core.py b/scripts/measure_private_core.py new file mode 100644 index 0000000..f3a712a --- /dev/null +++ b/scripts/measure_private_core.py @@ -0,0 +1,67 @@ +"""Pilot load probe: aggregate synthetic-request timings, never training traces. + +Admission responses are reported, not misrepresented as successful inference. +This short run is neither a sustained SLO test nor a public canary. +""" +import concurrent.futures +import json +import math +import os +import time +from collections import Counter +from pathlib import Path + +import requests + + +def main(): + config = dict(line.split("=", 1) for line in ( + Path.home() / ".c3r-private-inference" / + os.environ.get("C3R_PROBE_CONFIG", "candidate-v3.env")).read_text().splitlines()) + token = config["C3R_CLIENT_TOKEN"] + base = "http://127.0.0.1:" + config.get("PORT", "8088") + + def call(_): + session = requests.Session() + session.trust_env = False + started = time.monotonic() + try: + response = session.post(base + "/v1/c3r/rank", headers={ + "Authorization": "Bearer " + token}, json={ + "state": "An invoice was charged twice.", + "candidates": ["Billing", "Technical"]}, timeout=15) + status = str(response.status_code) + except requests.RequestException: + status = "transport_failure" + finally: + session.close() + return status, (time.monotonic() - started) * 1000 + + profiles = [] + for concurrency in (1, 10, 25, 50, 100): + started = time.monotonic() + with concurrent.futures.ThreadPoolExecutor(max_workers=concurrency) as executor: + results = list(executor.map(call, range(concurrency))) + elapsed = time.monotonic() - started + successful = sorted(latency for status, latency in results if status == "200") + counts = dict(Counter(status for status, _ in results)) + profile = {"concurrency": concurrency, "requests": len(results), + "statuses": counts, "elapsed_seconds": round(elapsed, 3), + "successful_requests_per_second": round(len(successful) / elapsed, 3), + "successful_p50_ms": None if not successful else round( + successful[(len(successful) - 1) // 2], 2), + "successful_p95_ms": None if not successful else round( + successful[math.ceil(len(successful) * .95) - 1], 2)} + profiles.append(profile) + # Predeclared abort: transport failure or unexpected server failures. + if any(status not in {"200", "429", "503"} for status in counts): + print(json.dumps({"scope": "private_synthetic_load_pilot", "aborted": True, + "profiles": profiles}, indent=2)) + raise SystemExit(1) + print(json.dumps({"scope": "private_synthetic_load_pilot_not_slo_or_canary", + "currency_cost": "unqualified_no_billing_allocation", + "profiles": profiles}, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/scripts/probe_live_models.py b/scripts/probe_live_models.py new file mode 100644 index 0000000..88d4d50 --- /dev/null +++ b/scripts/probe_live_models.py @@ -0,0 +1,38 @@ +"""Non-sensitive loopback inference drill. Does not store task traces.""" +import json +import time + +import requests + + +def main(): + session = requests.Session() + session.trust_env = False + probes = ( + ("clm", "http://127.0.0.1:8700/v1/rank", { + "model": "clm-latest", "context": "The customer was charged twice for an invoice.", + "question": "Which department handles this request?", + "answers": ["Billing: invoices and charges", "Technical: software bugs"], + }), + ("deepseek", "http://127.0.0.1:8000/v1/chat/completions", { + "model": "/model", "max_tokens": 512, "temperature": 0, + "messages": [{"role": "user", "content": "Reply with one sentence: what is an invoice?"}], + }), + ) + for name, url, payload in probes: + started = time.monotonic() + response = session.post(url, json=payload, timeout=60, allow_redirects=False) + body = response.json() + if name == "deepseek": + # Never record the reasoning field, even for a synthetic probe. + choices = body.get("choices", []) + body = {"final_text": choices[0].get("message", {}).get("content") if choices else None, + "finish_reason": choices[0].get("finish_reason") if choices else None, + "usage": body.get("usage")} + print(json.dumps({"provider": name, "status": response.status_code, + "latency_ms": round((time.monotonic() - started) * 1000, 2), + "result": body}), flush=True) + + +if __name__ == "__main__": + main() diff --git a/scripts/probe_private_core.py b/scripts/probe_private_core.py new file mode 100644 index 0000000..2cdf4eb --- /dev/null +++ b/scripts/probe_private_core.py @@ -0,0 +1,65 @@ +"""Public API smoke/security checks using synthetic text and host-local credentials.""" +import json +import os +import time +from pathlib import Path + +import requests + + +def main(): + config = dict(line.split("=", 1) for line in + (Path.home() / ".c3r-private-inference" / + os.environ.get("C3R_PROBE_CONFIG", "candidate-v2.env")).read_text().splitlines()) + session = requests.Session() + session.trust_env = False + headers = {"Authorization": "Bearer " + config["C3R_CLIENT_TOKEN"]} + base = "http://127.0.0.1:" + config.get("PORT", "8088") + checks = ( + ("readiness", "GET", "/ready", None, 200), + ("models", "GET", "/v1/models", None, 200), + ("typed", "POST", "/v1/system-one", {"model": "c3r-system-one", + "state": "An invoice was charged twice", "questions": { + "department": {"type": "choice", "options": {"billing": "Invoices and charges", + "technical": "Software bugs"}}, + "urgent": {"type": "boolean"}}}, 200), + ("rank", "POST", "/v1/c3r/rank", {"state": "An invoice was charged twice", + "candidates": ["billing", "technical"]}, 200), + ("decide", "POST", "/v1/c3r/decide", {"goal": "Route invoice inquiry", + "state": "An invoice was charged twice", "catalog": "agent-v1"}, 200), + ("responses", "POST", "/v1/responses", {"model": "c3r-core", + "input": "Explain briefly why an invoice may appear charged twice."}, 200), + ("authority_override", "POST", "/v1/c3r/decide", {"goal": "test", "state": "test", + "risk": "DESTRUCTIVE", "approval": "forged"}, 400), + ("internal_url_override", "POST", "/v1/system-one", {"state": "test", + "candidates": ["A", "B"], "endpoint": "http://169.254.169.254"}, 400), + ("candidate_overflow", "POST", "/v1/system-one", {"state": "test", + "candidates": [str(index) for index in range(65)]}, 400), + ("storage_rejected", "POST", "/v1/responses", {"model": "c3r-core", "input": "test", + "store": True}, 400), + ("execute_disabled", "POST", "/v1/c3r/execute", {}, 501), + ) + results = [] + for name, method, path, payload, expected in checks: + started = time.monotonic() + response = session.request(method, base + path, + headers=headers, json=payload, timeout=65) + body = response.json() + passed = response.status_code == expected + if name == "typed" and passed: + passed = body.get("answers", {}).get("department", {}).get("choice") == "billing" + if name == "responses" and passed: + passed = bool(body.get("output", [{}])[0].get("content", [{}])[0].get("text")) + results.append({"check": name, "status": response.status_code, "passed": passed, + "latency_ms": round((time.monotonic() - started) * 1000, 2)}) + for name, token in (("missing_auth", None), ("invalid_auth", "wrong")): + response = session.get(base + "/v1/models", + headers={} if token is None else {"Authorization": "Bearer " + token}) + results.append({"check": name, "status": response.status_code, + "passed": response.status_code == 401}) + print(json.dumps({"scope": "private_synthetic_api_probe_not_canary", "checks": results, + "all_passed": all(result["passed"] for result in results)}, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/scripts/replace_private_candidate.py b/scripts/replace_private_candidate.py new file mode 100644 index 0000000..8199061 --- /dev/null +++ b/scripts/replace_private_candidate.py @@ -0,0 +1,80 @@ +"""Replace only task-created private C3R containers, restoring them on failure. + +Run on the deployment host after build, artifact verification and scans. This +never stops the shared DeepSeek service or the VM, nor exposes an interface. +""" +import json +import os +import subprocess +import time +import urllib.error +import urllib.request +from pathlib import Path + + +def docker(*args): + return subprocess.check_output(["sudo", "docker", *args], text=True).strip() + + +def main(): + folder = Path.home() / ".c3r-private-inference" + previous = folder / "candidate-v2.env" + current = folder / "candidate-v3.env" + if current.exists(): + raise RuntimeError("v3 config exists; do not implicitly rotate or replace") + config = dict(line.split("=", 1) for line in previous.read_text().splitlines()) + if (config.get("C3R_INGRESS_HOST") != "127.0.0.1" + or config.get("C3R_TRACE_COLLECTION") != "false" + or config.get("C3R_ONLINE_LEARNING") != "false"): + raise RuntimeError("replacement is restricted to non-collecting private inference") + clm_image = docker("image", "inspect", "c3r-clm:hardened-clean-bb42c6c", "--format", "{{.Id}}") + core_image = docker("image", "inspect", "c3r-core:reviewed-20260930", "--format", "{{.Id}}") + config["C3R_CLM_CONTAINER_DIGEST"] = clm_image + descriptor = os.open(current, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + with os.fdopen(descriptor, "w") as handle: + handle.write("".join(f"{key}={value}\n" for key, value in config.items())) + moved = [] + started = [] + try: + for name in ("c3r-core-private", "c3r-clm"): + docker("stop", "--time", "10", name) + docker("rename", name, name + "-before-v3") + moved.append(name) + docker("run", "-d", "--name", "c3r-clm", "--network", "host", "--read-only", + "--tmpfs", "/tmp:rw,size=256m", "-v", "/mnt/c3r-models/c3r-clm:/artifacts:ro", + "-e", "C3R_CLM_CONTAINER_DIGEST=" + clm_image, clm_image) + started.append("c3r-clm") + docker("run", "-d", "--name", "c3r-core-private", "--network", "host", + "--env-file", str(current), "--read-only", "--cap-drop", "ALL", + "--security-opt", "no-new-privileges", "--pids-limit", "128", + "--memory", "512m", "--cpus", "2", core_image) + started.append("c3r-core-private") + opener = urllib.request.build_opener(urllib.request.ProxyHandler({})) + deadline = time.monotonic() + 120 + while time.monotonic() < deadline: + request = urllib.request.Request("http://127.0.0.1:8088/ready", headers={ + "Authorization": "Bearer " + config["C3R_CLIENT_TOKEN"]}) + try: + with opener.open(request, timeout=15) as response: + if response.status == 200: + print(json.dumps({"status": "private_candidate_ready", + "core_image": core_image, "clm_image": clm_image, + "trace_collection": False, "public_access": False})) + return + except (OSError, urllib.error.URLError): + pass + time.sleep(3) + raise RuntimeError("private candidate readiness deadline exceeded") + except BaseException: + for name in reversed(started): + docker("stop", "--time", "10", name) + docker("rename", name, name + "-failed-v3") + for name in reversed(moved): + docker("rename", name + "-before-v3", name) + docker("start", name) + print(json.dumps({"status": "previous_private_candidate_restored"})) + raise + + +if __name__ == "__main__": + main() diff --git a/scripts/run_controlled_pairs.py b/scripts/run_controlled_pairs.py new file mode 100644 index 0000000..0ff8068 --- /dev/null +++ b/scripts/run_controlled_pairs.py @@ -0,0 +1,156 @@ +"""Run public, C3R-authored read-only/authority fixtures as paired controller traces. + +This is a pipeline smoke test. It does not measure production task success, Laya, +DeepSeek, calibrated CVoC, or live Colibri behavior. +""" + +from __future__ import annotations + +import json +import platform +import sys +from dataclasses import asdict, dataclass +from datetime import UTC, datetime +from hashlib import sha256 +from pathlib import Path +from time import perf_counter + +REPOSITORY_ROOT = Path(__file__).resolve().parents[1] +if str(REPOSITORY_ROOT) not in sys.path: + sys.path.insert(0, str(REPOSITORY_ROOT)) + +from c3r.candidate_compiler import CandidateCompiler +from c3r.cvoc import RobustCvocController +from c3r.feature_flags import FeatureFlags +from c3r.runtime import RuntimeRequest, StandaloneController +from c3r.state_compiler import StateCompiler +from c3r.state_schema import ( + ActionDefinition, + ActionFamily, + AuthorityPolicy, + Provenance, + RawState, + RiskClass, + ValueEstimate, +) +from c3r.telemetry.trace_ledger import TraceLedger +from c3r.verifier_firewall import VerifierDecision, VerifierFirewall, VerifierPolicy + + +@dataclass(frozen=True) +class ControlledCase: + task_id: str + fact_provenance: bool + verifier_accepts: bool + estimate_gain: float + risk: RiskClass + expected_action_id: str | None + + +ACTION_ID = "lookup:0:local:policy" +CASES = ( + ControlledCase("ct-001", True, True, 0.9, RiskClass.READ_ONLY, ACTION_ID), + ControlledCase("ct-002", True, True, 0.0, RiskClass.READ_ONLY, None), + ControlledCase("ct-003", False, True, 0.9, RiskClass.READ_ONLY, None), + ControlledCase("ct-004", True, False, 0.9, RiskClass.READ_ONLY, None), + ControlledCase("ct-005", True, True, 0.9, RiskClass.EXTERNAL_WRITE, None), +) + + +def _request(case: ControlledCase, arm: str) -> RuntimeRequest: + fact = f"fixture fact {case.task_id}" + raw = RawState( + goal="Choose a bounded records action", + current_subgoal="Check one fixture record", + verified_facts=(fact,), + available_action_families=(ActionFamily.TOOL,), + budget={"remaining_usd": 1.0}, + provenance={fact: Provenance("c3r-controlled-fixture", "2026-09-22T00:00:00Z")} + if case.fact_provenance else {}, + ) + definition = ActionDefinition( + id="lookup", family=ActionFamily.TOOL, subgroup="records", + operation="get" if case.risk is RiskClass.READ_ONLY else "update", + risk_class=case.risk, argument_variants=((('record_id', case.task_id),),), + placements=("local",), verifier_ids=("policy",), optimistic_utility=1.0, + estimated_cost=0.1, data_boundary="local", + ) + return RuntimeRequest( + raw_state=raw, + definitions=(definition,), + policy=AuthorityPolicy(frozenset({ActionFamily.TOOL}), frozenset({case.risk})), + estimates={ACTION_ID: ValueEstimate(case.estimate_gain, 0.1, 0.0, 0.1)}, + run_id=f"{case.task_id}-{arm}", + ) + + +def _run(case: ControlledCase, arm: str) -> dict[str, object]: + key = b"controlled-verifier-test-key" + verifier = VerifierFirewall( + {"policy": lambda _: VerifierDecision(case.verifier_accepts, "fixture policy")}, + VerifierPolicy(default_verifier="policy"), attestation_key=key, + ) + ledger = TraceLedger() + controller = StandaloneController( + flags=FeatureFlags(enabled_requested=arm == "c3r"), + compiler=StateCompiler(), candidates=CandidateCompiler(), + cvoc=RobustCvocController(), verifier=verifier, + ledger=ledger, + ) + start = perf_counter() + outcome = controller.run(_request(case, arm)) + latency_ms = (perf_counter() - start) * 1000 + trace = json.loads(outcome.ledger_record.canonical_json) + success = outcome.selected_action_id == case.expected_action_id + if case.expected_action_id is not None: + success = success and outcome.authority_result == "verified_not_committed" + return { + "task_id": case.task_id, + "state_hash": trace["state_hash"], + "arm": arm, + "label_positive": success, + "outcome_kind": "policy_rubric_match", + "outcome_label_ref": f"rubric:c3r-controlled-v1:{case.task_id}", + "latency_ms": latency_ms, + "cost_usd": 0.0, + "authority_bypass": False, # effects are structurally unavailable in this controller + "trace_hash": outcome.ledger_record.record_hash, + } + + +def collect() -> tuple[list[dict[str, object]], dict[str, object]]: + observations = [_run(case, arm) for case in CASES for arm in ("baseline", "c3r")] + cases_json = json.dumps([asdict(case) for case in CASES], sort_keys=True, default=str) + manifest: dict[str, object] = { + "evidence_kind": "controlled", + "task_population": "five C3R-authored authority/selection fixtures", + "rubric_sha256": sha256(cases_json.encode("utf-8")).hexdigest(), + "arms": ["baseline-disabled", "c3r-reference-controller"], + "external_provider_cost_usd": 0.0, + "limitations": "No Laya, DeepSeek, live tools, customer traffic, Colibri, or task-success ground truth", + } + return observations, manifest + + +def main() -> None: + output = Path(__file__).resolve().parents[1] / "evidence" / "controlled-pairs-v2" + output.mkdir(parents=True, exist_ok=True) + observations, manifest = collect() + observations_bytes = ( + "\n".join(json.dumps(row, sort_keys=True) for row in observations) + "\n" + ).encode("utf-8") + (output / "observations.jsonl").write_bytes(observations_bytes) + manifest.update({ + "observations_sha256": sha256(observations_bytes).hexdigest(), + "generator_sha256": sha256(Path(__file__).read_bytes()).hexdigest(), + "collected_at": datetime.now(UTC).isoformat(), + "python_version": platform.python_version(), + "platform": platform.platform(), + }) + (output / "manifest.json").write_text( + json.dumps(manifest, indent=2, sort_keys=True) + "\n", encoding="utf-8", newline="\n", + ) + + +if __name__ == "__main__": + main() diff --git a/scripts/run_trace_control_dry_run.py b/scripts/run_trace_control_dry_run.py new file mode 100644 index 0000000..28000de --- /dev/null +++ b/scripts/run_trace_control_dry_run.py @@ -0,0 +1,122 @@ +"""Reproducible local fixture check for independent trace-control review. + +This never enables live collection and never claims deployed storage controls. +""" + +from __future__ import annotations + +import json +import sys +from collections.abc import Callable +from datetime import datetime, timedelta, timezone +from pathlib import Path +from tempfile import TemporaryDirectory + +REPOSITORY_ROOT = Path(__file__).resolve().parents[1] +if str(REPOSITORY_ROOT) not in sys.path: + sys.path.insert(0, str(REPOSITORY_ROOT)) + +from c3r.telemetry.governed_store import GovernedTraceStore, SourceGrant +from c3r.telemetry.trace import DecisionTrace + + +START = datetime(2026, 9, 23, 0, 0, tzinfo=timezone.utc) +SOURCE_ID = "c3r_fixture_review" +TASK_ID = "fixture_task_001" + + +def _trace(run_id: str, **changes: object) -> DecisionTrace: + fields: dict[str, object] = { + "run_id": run_id, + "state_hash": "a" * 64, + "access_level": "internal", + "model_provider": "fixture_only", + "candidate_ids": ("recommend",), + "probabilities": {"route": (0.8, 0.2)}, + "utility_quantiles": {"selected_lower_bound": 0.3}, + "selected_action_id": "recommend", + "authority_result": "verified", + "system_cost": {"latency_ms": 1.0}, + "task_outcome": {"status": "fixture_only"}, + "artifact_refs": (), + } + fields.update(changes) + return DecisionTrace(**fields) + + +def _rejects(callback: Callable[[], object], expected_reason: str) -> bool: + try: + callback() + except ValueError as error: + return expected_reason in str(error) + return False + + +def run() -> dict[str, object]: + grant = SourceGrant( + source_id=SOURCE_ID, owner="c3r_review_fixture", + task_ids=frozenset({TASK_ID}), rights_attested=True, + ) + clock = [START] + with TemporaryDirectory(prefix="c3r-trace-review-") as directory: + with GovernedTraceStore( + Path(directory) / "trace.sqlite3", grants=(grant,), clock=lambda: clock[0], + ) as store: + first = store.append(_trace("fixture_run_001"), source_id=SOURCE_ID, task_id=TASK_ID) + checks = { + "approved_fixture_admitted": len(store.records()) == 1, + "unapproved_source_rejected": _rejects(lambda: store.append( + _trace("fixture_run_002"), source_id="customer_logs", task_id=TASK_ID, + ), "unapproved source or task"), + "free_text_rejected": _rejects(lambda: store.append( + _trace("fixture_run_003", task_outcome={"status": "email me at a@example.com"}), + source_id=SOURCE_ID, task_id=TASK_ID, + ), "redaction"), + "artifact_reference_rejected": _rejects(lambda: store.append( + _trace("fixture_run_004", artifact_refs=("private_artifact",)), + source_id=SOURCE_ID, task_id=TASK_ID, + ), "redaction"), + } + checks["rejected_rows_not_persisted"] = len(store.records()) == 1 + clock[0] = START + timedelta(days=31) + checks["overdue_purge_blocks_collection"] = _rejects(lambda: store.append( + _trace("fixture_run_005"), source_id=SOURCE_ID, task_id=TASK_ID, + ), "retention purge overdue") + checks["local_expired_row_purged"] = store.purge_expired() == 1 + checks["checkpoint_chain_verifies"] = store.verify() and not store.records() + second = store.append( + _trace("fixture_run_006"), source_id=SOURCE_ID, task_id=TASK_ID, + ) + checks["post_purge_chain_continues"] = ( + second.previous_hash == first.record_hash and store.verify() + ) + return { + "evidence_kind": "non_sensitive_local_fixture_dry_run", + "live_trace_collection_enabled": False, + "source": SOURCE_ID, + "task_population": "one C3R-authored synthetic fixture; no customer or product records", + "checks": checks, + "all_local_checks_passed": all(checks.values()), + "not_verified_by_this_run": [ + "deployed encryption and IAM", + "read/export access audit", + "daily scheduler and deletion of backups or replicas", + "independent ledger-head anchor", + "alert delivery and rollback", + "live internal-task provenance or outcomes", + ], + } + + +def main() -> None: + report = run() + output = REPOSITORY_ROOT / "evidence" / "trace-control-dry-run-v1" / "report.json" + output.parent.mkdir(parents=True, exist_ok=True) + output.write_text(json.dumps(report, indent=2, sort_keys=True) + "\n", encoding="utf-8") + if not report["all_local_checks_passed"]: + raise SystemExit("local trace-control dry run failed") + + +if __name__ == "__main__": + main() + diff --git a/scripts/start_private_core.py b/scripts/start_private_core.py new file mode 100644 index 0000000..93005b1 --- /dev/null +++ b/scripts/start_private_core.py @@ -0,0 +1,41 @@ +"""Start a loopback-only candidate. Tokens stay in a mode-0600 host file.""" +import json +import os +import secrets +import subprocess +from pathlib import Path + + +def main(): + destination = Path.home() / ".c3r-private-inference" + destination.mkdir(mode=0o700, exist_ok=True) + config = destination / "candidate-v2.env" + if config.exists(): + raise RuntimeError("candidate config already exists; do not rotate implicitly") + env = { + "C3R_HOST_ENTRYPOINT": "c3r.production_host:build", + "C3R_MODE": "production_inference", "C3R_ENABLED": "true", + "C3R_SYSTEM_ONE": "true", "C3R_SYSTEM_ONE_PROVIDER": "clm", + "C3R_DELIBERATIVE": "true", "C3R_TRACE_COLLECTION": "false", + "C3R_ONLINE_LEARNING": "false", "C3R_INGRESS_HOST": "127.0.0.1", + "PORT": "8088", "C3R_BACKEND_PORT": "8089", + "C3R_CLM_CONTAINER_DIGEST": "sha256:922fe094c0804fc2b4e327f2dbbe97b749e47de7d8806b2f467cae7ecd0d2dda", + "C3R_CLIENT_TOKEN": secrets.token_urlsafe(48), + "C3R_BACKEND_TOKEN": secrets.token_urlsafe(48), + } + descriptor = os.open(config, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + with os.fdopen(descriptor, "w") as handle: + handle.write("".join(f"{key}={value}\n" for key, value in env.items())) + result = subprocess.run([ + "sudo", "docker", "run", "-d", "--name", "c3r-core-private", + "--network", "host", "--env-file", str(config), "--read-only", + "--cap-drop", "ALL", "--security-opt", "no-new-privileges", + "--pids-limit", "128", "--memory", "512m", "--cpus", "2", + "c3r-core:working-20260930", + ], check=True, capture_output=True, text=True) + print(json.dumps({"container_id": result.stdout.strip(), "bind": "127.0.0.1:8088", + "trace_collection": False})) + + +if __name__ == "__main__": + main() diff --git a/tests/official_sdk_acceptance.mjs b/tests/official_sdk_acceptance.mjs new file mode 100644 index 0000000..b79d4a2 --- /dev/null +++ b/tests/official_sdk_acceptance.mjs @@ -0,0 +1,21 @@ +import assert from "node:assert/strict"; +import { pathToFileURL } from "node:url"; +const { default: OpenAI } = await import(pathToFileURL(process.env.C3R_JS_SDK_MODULE).href); +const url = new URL(process.env.C3R_SDK_TEST_URL); +assert.equal(url.hostname, "127.0.0.1"); +const client = new OpenAI({ apiKey: process.env.C3R_SDK_TEST_KEY, + baseURL: url.href, maxRetries: 0, timeout: 5000 }); +const response = await client.responses.create({ model: "c3r-core", input: "Inspect", store: false }); +assert.equal(response.output_text, "Inspect safely."); +assert.equal(response.usage.total_tokens, 7); +assert.equal(typeof response.created_at, "number"); +assert.ok(!JSON.stringify(response).includes("NEVER_PUBLIC")); +const stream = await client.responses.create({ model: "c3r-core", input: "Inspect", stream: true }); +let text = "", terminal; +for await (const event of stream) { + if (event.type === "response.output_text.delta") text += event.delta; + terminal = event.type; +} +assert.equal(text, "Inspect safely."); +assert.equal(terminal, "response.completed"); +console.log("SDK_ACCEPTANCE_PASS"); diff --git a/tests/test_api_access_http.py b/tests/test_api_access_http.py new file mode 100644 index 0000000..c2878bb --- /dev/null +++ b/tests/test_api_access_http.py @@ -0,0 +1,264 @@ +"""Developer authentication and admission through the real gateway HTTP boundary.""" +import json +import select +import socket +import subprocess +import sys +import tempfile +import threading +import time +import unittest +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path +from typing import cast +from urllib.error import HTTPError +from urllib.request import Request, urlopen + +from c3r.api_access import AccessStore +from c3r.ingress_proxy import C3RIngressServer + +BACKEND_TOKEN = "backend-test-token-distinct-and-long-enough" + + +class Upstream(BaseHTTPRequestHandler): + def log_message(self, format: str, *args: object) -> None: + return + + def do_GET(self) -> None: + body = b'{"object":"list","data":[{"id":"c3r-system-one"}]}' + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + def do_POST(self) -> None: + payload = json.loads(self.rfile.read(int(self.headers["Content-Length"]))) + if payload.get("input") == "wait_headers": + server = cast(UpstreamServer, self.server) + server.pending_headers.set() + deadline = time.monotonic() + 2 + while time.monotonic() < deadline: + readable, _, _ = select.select([self.connection], [], [], 0.05) + if readable and self.connection.recv(1, socket.MSG_PEEK) == b"": + server.cancelled.set() + return + return + if payload.get("input") == "reject": + body = b'{"error":"private upstream diagnostic"}' + self.send_response(400) + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + return + if payload.get("stream") is True: + self.send_response(200) + self.send_header("Content-Type", "text/event-stream") + self.end_headers() + self.wfile.write(b'event: response.created\ndata: {"type":"response.created"}\n\n') + self.wfile.flush() + server = cast(UpstreamServer, self.server) + deadline = time.monotonic() + 2 + while not server.stream_release.wait(0.05) and time.monotonic() < deadline: + readable, _, _ = select.select([self.connection], [], [], 0) + if readable and self.connection.recv(1, socket.MSG_PEEK) == b"": + server.cancelled.set() + return + self.wfile.write(b'event: response.output_text.delta\ndata: {"type":"response.output_text.delta","delta":"safe"}\n\n') + self.wfile.write(b'event: response.completed\ndata: {"type":"response.completed","response":{"model":"c3r-core","usage":{"input_tokens":13,"output_tokens":7}}}\n\n') + self.wfile.flush() + return + body = json.dumps({"model": "c3r-core", "output_text": "PRIVATE_RETURN_MARKER", + "usage": {"input_tokens": 13, "output_tokens": 7}, + "c3r": {"system_one_invocations": 1, "system_two_invocations": 1}}).encode() + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + +class UpstreamServer(ThreadingHTTPServer): + stream_release: threading.Event + cancelled: threading.Event + pending_headers: threading.Event + + +class APIAccessHTTPTests(unittest.TestCase): + def setUp(self) -> None: + self.scratch = tempfile.TemporaryDirectory() + self.store = AccessStore(Path(self.scratch.name) / "access.sqlite3") + self.store.create_project("tenant-a", "project-a") + self.key = self.store.issue_key("tenant-a", "project-a", {"models:read"}) + self.upstream = UpstreamServer(("127.0.0.1", 0), Upstream) + self.upstream.stream_release = threading.Event() + self.upstream.cancelled = threading.Event() + self.upstream.pending_headers = threading.Event() + self.gateway = C3RIngressServer( + upstream_port=self.upstream.server_port, upstream_token=BACKEND_TOKEN, + client_token="unused-staging-token-with-at-least-32-characters", + host="127.0.0.1", port=0, access_store=self.store, + ) + self.threads = [threading.Thread(target=server.serve_forever, daemon=True) + for server in (self.upstream, self.gateway)] + for thread in self.threads: + thread.start() + + def tearDown(self) -> None: + self.gateway.shutdown() + self.upstream.shutdown() + self.gateway.server_close() + self.upstream.server_close() + for thread in self.threads: + thread.join(timeout=2) + self.scratch.cleanup() + + def call(self, path: str, *, key: str | None = None, method: str = "GET", + payload: dict[str, object] | None = None) -> tuple[int, dict[str, object], str | None]: + request = Request(f"http://127.0.0.1:{self.gateway.server_port}" + path, + headers={"Authorization": "Bearer " + (key or self.key.secret), + "Content-Type": "application/json"}, + data=json.dumps(payload).encode() if payload is not None else None, + method=method) + try: + with urlopen(request, timeout=2) as response: + return response.status, cast(dict[str, object], json.load(response)), response.headers.get("x-request-id") + except HTTPError as error: + return error.code, cast(dict[str, object], json.load(error)), error.headers.get("x-request-id") + + def test_scoped_developer_key_can_discover_models(self) -> None: + status, body, request_id = self.call("/v1/models") + self.assertEqual((status, body["object"]), (200, "list")) + self.assertIsNotNone(request_id) + + def test_key_scope_blocks_generation_with_standard_error_and_request_id(self) -> None: + status, body, request_id = self.call("/v1/responses", method="POST", payload={ + "model": "c3r-core", "input": "sensitive text must not be retained"}) + error = cast(dict[str, object], body["error"]) + self.assertEqual((status, error["type"]), (403, "permission_error")) + self.assertEqual(error["request_id"], request_id) + + def test_revoked_and_expired_keys_cannot_infer(self) -> None: + self.store.revoke_key("tenant-a", "project-a", self.key.key_id) + self.assertEqual(self.call("/v1/models")[0], 401) + expired = self.store.issue_key("tenant-a", "project-a", {"models:read"}, expires_at=1) + self.assertEqual(self.call("/v1/models", key=expired.secret)[0], 401) + + def test_project_rate_limit_is_shared_by_its_keys_not_other_tenants(self) -> None: + self.store.create_project("tenant-b", "project-a", rpm=1) + first = self.store.issue_key("tenant-b", "project-a", {"models:read"}) + second = self.store.issue_key("tenant-b", "project-a", {"models:read"}) + self.assertEqual(self.call("/v1/models", key=first.secret)[0], 200) + status, body, _ = self.call("/v1/models", key=second.secret) + self.assertEqual((status, cast(dict[str, object], body["error"])["type"]), + (429, "rate_limit_error")) + self.assertEqual(self.call("/v1/models")[0], 200) + + def test_operator_cli_issues_show_once_key_accepted_by_gateway(self) -> None: + result = subprocess.run([sys.executable, "-m", "c3r.key_management", "--database", + str(self.store.path), "issue", "--tenant", "tenant-a", + "--project", "project-a", "--scope", "models:read"], + capture_output=True, text=True, timeout=5, check=False) + self.assertEqual(result.returncode, 0, result.stderr) + issued = cast(dict[str, str], json.loads(result.stdout)) + self.assertEqual(self.call("/v1/models", key=issued["api_key"])[0], 200) + listed = subprocess.run([sys.executable, "-m", "c3r.key_management", "--database", + str(self.store.path), "keys", "--tenant", "tenant-a", + "--project", "project-a"], capture_output=True, text=True, timeout=5, check=False) + self.assertEqual(listed.returncode, 0) + self.assertNotIn(issued["api_key"], listed.stdout) + + def test_usage_cli_reports_actual_counts_without_payloads_or_fabricated_gpu_cost(self) -> None: + key = self.store.issue_key("tenant-a", "project-a", {"responses:write"}) + status, _, request_id = self.call("/v1/responses", key=key.secret, method="POST", payload={ + "model": "c3r-core", "input": "PRIVATE_INPUT_MARKER"}) + self.assertEqual(status, 200) + result = subprocess.run([sys.executable, "-m", "c3r.key_management", "--database", + str(self.store.path), "usage", "--tenant", "tenant-a", + "--project", "project-a"], capture_output=True, text=True, timeout=5, check=False) + self.assertEqual(result.returncode, 0) + rows = cast(list[dict[str, object]], json.loads(result.stdout)) + self.assertEqual((rows[0]["request_id"], rows[0]["input_tokens"], rows[0]["output_tokens"]), + (request_id, 13, 7)) + self.assertIsNone(rows[0]["gpu_allocation_ms"]) + self.assertIsNone(rows[0]["allocated_cost_usd"]) + self.assertNotIn("PRIVATE_", result.stdout) + self.assertNotIn(key.secret, self.store.path.read_bytes().decode("latin1")) + + def test_stream_is_forwarded_before_generation_finishes(self) -> None: + key = self.store.issue_key("tenant-a", "project-a", {"responses:write"}) + request = Request(f"http://127.0.0.1:{self.gateway.server_port}/v1/responses", + headers={"Authorization": "Bearer " + key.secret, + "Content-Type": "application/json"}, + data=b'{"model":"c3r-core","input":"test","stream":true}') + with urlopen(request, timeout=3) as response: + self.assertEqual(response.headers.get_content_type(), "text/event-stream") + self.assertEqual(response.readline(), b"event: response.created\n") + status, rejected, _ = self.call("/v1/responses", key=key.secret, method="POST", + payload={"model": "c3r-core", "input": "test"}) + self.assertEqual(status, 429) + self.assertEqual(cast(dict[str, object], rejected["error"])["type"], "rate_limit_error") + self.upstream.stream_release.set() + remaining = response.read() + self.assertIn(b"event: response.completed", remaining) + + def test_upstream_rejection_has_canonical_redacted_error(self) -> None: + key = self.store.issue_key("tenant-a", "project-a", {"responses:write"}) + status, body, request_id = self.call("/v1/responses", key=key.secret, method="POST", + payload={"model": "c3r-core", "input": "reject"}) + self.assertEqual(status, 400) + self.assertEqual(cast(dict[str, object], body["error"])["request_id"], request_id) + self.assertNotIn("private", json.dumps(body)) + + def test_disconnect_cancels_idle_upstream_and_releases_core_capacity(self) -> None: + key = self.store.issue_key("tenant-a", "project-a", {"responses:write"}) + request = Request(f"http://127.0.0.1:{self.gateway.server_port}/v1/responses", + headers={"Authorization": "Bearer " + key.secret, + "Content-Type": "application/json"}, + data=b'{"model":"c3r-core","input":"test","stream":true}') + with urlopen(request, timeout=3) as response: + self.assertEqual(response.readline(), b"event: response.created\n") + self.assertTrue(self.upstream.cancelled.wait(1), "idle generation was not cancelled") + deadline = time.monotonic() + 1 + status = 429 + while time.monotonic() < deadline: + status, _, _ = self.call("/v1/responses", key=key.secret, method="POST", + payload={"model": "c3r-core", "input": "test"}) + if status == 200: + break + time.sleep(0.05) + self.assertEqual(status, 200) + + def test_unavailable_access_database_fails_closed_with_sanitized_error(self) -> None: + self.store.path.unlink() + status, body, _ = self.call("/v1/models") + self.assertEqual(status, 503) + self.assertEqual(cast(dict[str, object], body["error"])["code"], "access_unavailable") + + def test_unsupported_method_returns_json_error_and_request_id(self) -> None: + status, body, request_id = self.call("/v1/models", method="PUT") + self.assertEqual(status, 405) + self.assertEqual(cast(dict[str, object], body["error"])["request_id"], request_id) + + def test_disconnect_before_upstream_headers_cancels_and_releases_capacity(self) -> None: + key = self.store.issue_key("tenant-a", "project-a", {"responses:write"}) + payload = b'{"model":"c3r-core","input":"wait_headers","stream":true}' + with socket.create_connection(("127.0.0.1", self.gateway.server_port), timeout=2) as client: + client.sendall(("POST /v1/responses HTTP/1.1\r\nHost: localhost\r\nAuthorization: Bearer " + + key.secret + "\r\nContent-Type: application/json\r\nContent-Length: " + + str(len(payload)) + "\r\n\r\n").encode() + payload) + self.assertTrue(self.upstream.pending_headers.wait(1)) + self.assertTrue(self.upstream.cancelled.wait(1)) + status = 429 + deadline = time.monotonic() + 1 + while time.monotonic() < deadline: + status, _, _ = self.call("/v1/responses", key=key.secret, method="POST", + payload={"model": "c3r-core", "input": "test"}) + if status == 200: + break + time.sleep(0.05) + self.assertEqual(status, 200) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_clm_adapter.py b/tests/test_clm_adapter.py new file mode 100644 index 0000000..0bba50a --- /dev/null +++ b/tests/test_clm_adapter.py @@ -0,0 +1,113 @@ +import math +import time +import unittest + +from c3r.feature_flags import FeatureFlags +from c3r.system_one.calibration import CalibrationKey, TemperatureCalibrator +from c3r.system_one.clm_adapter import ClmAdapter, UPSTREAM_CLM_COMMIT +from c3r.system_one.fast_path import CalibratedFastPath +from c3r.system_one.question_registry import TypedQuestion +from tests.test_laya_fast_path import compiled_state + + +REVISION = "a" * 64 + + +def ranked(payload: dict[str, object]) -> dict[str, object]: + options = payload["answers"] + assert isinstance(options, list) + return { + "model": "clm-latest", + "ranked": [ + {"rank": index + 1, "candidate": option, "prob": probability} + for index, (option, probability) in enumerate( + zip(reversed(options), (0.9, 0.1), strict=True) + ) + ] + } + + +class ClmAdapterTests(unittest.TestCase): + def test_default_is_clm_but_global_enable_is_off(self) -> None: + flags = FeatureFlags.from_mapping({}) + self.assertEqual(flags.system_one_provider, "clm") + self.assertFalse(flags.system_one_enabled) + + def test_rejects_remote_endpoint_and_mutable_revision(self) -> None: + with self.assertRaisesRegex(ValueError, "local HTTP"): + ClmAdapter(revision=REVISION, endpoint="https://example.com:8700") + with self.assertRaisesRegex(ValueError, "immutable"): + ClmAdapter(revision="latest") + self.assertEqual(len(UPSTREAM_CLM_COMMIT), 40) + + def test_expired_decision_budget_never_calls_clm(self) -> None: + def unreachable(_payload: object) -> dict[str, object]: + self.fail("expired request reached CLM") + + adapter = ClmAdapter(revision=REVISION, transport=unreachable) + with self.assertRaises(TimeoutError): + adapter.rank_actions( + compiled_state(), ("a", "b"), deadline=time.monotonic() - 1 + ) + + def test_maps_ranked_probabilities_back_to_fixed_option_order(self) -> None: + calls: list[object] = [] + + def transport(payload: object) -> dict[str, object]: + calls.append(payload) + assert isinstance(payload, dict) + return ranked(payload) + + adapter = ClmAdapter(revision=REVISION, transport=transport) + prediction = adapter.predict( + compiled_state(), (TypedQuestion("STOP_NOW", ("NO", "YES")),) + ) + self.assertAlmostEqual(math.exp(prediction["STOP_NOW"][0]), 0.1) + self.assertAlmostEqual(math.exp(prediction["STOP_NOW"][1]), 0.9) + self.assertEqual(len(calls), 1) + self.assertEqual(adapter.rank_actions(compiled_state(), ("a", "b")), (0.1, 0.9)) + + def test_fails_closed_on_unknown_duplicate_and_nonfinite_candidates(self) -> None: + invalid_rows = ( + (("NO", 0.5), ("EXTRA", 0.5)), + (("NO", 0.5), ("NO", 0.5)), + (("NO", float("nan")), ("YES", 0.5)), + (("NO", 0.1), ("YES", 0.1)), + ) + for rows in invalid_rows: + response = { + "model": "clm-latest", + "ranked": [{"candidate": candidate, "prob": prob} for candidate, prob in rows], + } + with self.subTest(response=response), self.assertRaises(ValueError): + ClmAdapter(revision=REVISION, transport=lambda _payload: response).predict( + compiled_state(), (TypedQuestion("STOP_NOW", ("NO", "YES")),) + ) + + def test_no_calibration_means_abstention_even_with_high_raw_rank(self) -> None: + adapter = ClmAdapter(revision=REVISION, transport=lambda payload: ranked(dict(payload))) + fast = CalibratedFastPath(adapter=adapter, calibrator=TemperatureCalibrator({})) + outcome = fast.decide( + compiled_state(), (TypedQuestion("STOP_NOW", ("NO", "YES")),), + action_family="CONTROL", candidate_options=("stop", "continue"), + ) + self.assertTrue(outcome.abstained) + self.assertEqual(outcome.answers, {}) + self.assertEqual(outcome.candidate_probabilities, (0.1, 0.9)) + + def test_held_out_calibration_enables_typed_answer(self) -> None: + adapter = ClmAdapter(revision=REVISION, transport=lambda payload: ranked(dict(payload))) + key = CalibrationKey("STOP_NOW", "CONTROL", "2", "en", "low") + fast = CalibratedFastPath( + adapter=adapter, calibrator=TemperatureCalibrator({key: 1.0}) + ) + outcome = fast.decide( + compiled_state(), (TypedQuestion("STOP_NOW", ("NO", "YES")),), + action_family="CONTROL", + ) + self.assertFalse(outcome.abstained) + self.assertEqual(outcome.answers["STOP_NOW"], "YES") + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_controlled_pairs.py b/tests/test_controlled_pairs.py new file mode 100644 index 0000000..caa5c23 --- /dev/null +++ b/tests/test_controlled_pairs.py @@ -0,0 +1,21 @@ +import unittest + +from scripts.run_controlled_pairs import CASES, collect + + +class ControlledPairTests(unittest.TestCase): + def test_pairs_are_same_state_and_non_effectful(self) -> None: + rows, manifest = collect() + self.assertEqual(len(rows), 2 * len(CASES)) + self.assertEqual(manifest["evidence_kind"], "controlled") + self.assertEqual(sum(bool(row["label_positive"]) for row in rows if row["arm"] == "c3r"), 5) + for index in range(0, len(rows), 2): + baseline, c3r = rows[index : index + 2] + self.assertEqual({baseline["arm"], c3r["arm"]}, {"baseline", "c3r"}) + self.assertEqual(baseline["state_hash"], c3r["state_hash"]) + self.assertFalse(baseline["authority_bypass"]) + self.assertFalse(c3r["authority_bypass"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_cvoc.py b/tests/test_cvoc.py index 5117dad..82b3e62 100644 --- a/tests/test_cvoc.py +++ b/tests/test_cvoc.py @@ -37,7 +37,18 @@ def test_stops_when_all_lower_bounds_are_non_positive(self) -> None: self.assertIsNone(decision.selected) self.assertEqual(decision.fallback, "STOP") + def test_invalid_or_negative_cost_estimates_fail_closed(self) -> None: + candidate = ActionCandidate("local", ActionFamily.LOCAL_MODEL, RiskClass.READ_ONLY, 1.0) + for estimate in ( + ValueEstimate(float("nan"), 0.0, 0.0, 0.0), + ValueEstimate(1.0, -1.0, 0.0, 0.0), + ValueEstimate(1.0, 0.0, float("inf"), 0.0), + ): + with self.subTest(estimate=estimate): + decision = RobustCvocController().select((candidate,), {"local": estimate}) + self.assertIsNone(decision.selected) + self.assertEqual(decision.fallback, ActionFamily.STOP) + if __name__ == "__main__": unittest.main() - diff --git a/tests/test_decisionmix.py b/tests/test_decisionmix.py index 6714915..73b58c6 100644 --- a/tests/test_decisionmix.py +++ b/tests/test_decisionmix.py @@ -44,16 +44,16 @@ def test_deterministic_split_is_stable(self) -> None: self.assertEqual(first, second) self.assertIn(first, {"train", "validation", "test"}) - def test_forbids_benchmark_test_answers_in_training(self) -> None: + def test_forbids_benchmark_test_answers_in_all_decisionmix_splits(self) -> None: builder = DecisionMixBuilder(split_seed="c3r-v1") - record_id = next( - f"leak-{index}" - for index in range(1000) - if deterministic_split(f"leak-{index}", seed="c3r-v1") == "train" - ) - - with self.assertRaisesRegex(ValueError, "benchmark test"): - builder.add(record(record_id, source_partition="benchmark_test")) + for split in ("train", "validation", "test"): + record_id = next( + f"leak-{split}-{index}" + for index in range(1000) + if deterministic_split(f"leak-{split}-{index}", seed="c3r-v1") == split + ) + with self.subTest(split=split), self.assertRaisesRegex(ValueError, "benchmark test"): + builder.add(record(record_id, source_partition="benchmark_test")) def test_rejects_mutable_generator_revision(self) -> None: valid = record("mutable") diff --git a/tests/test_deepseek_launch.py b/tests/test_deepseek_launch.py new file mode 100644 index 0000000..b4e56f6 --- /dev/null +++ b/tests/test_deepseek_launch.py @@ -0,0 +1,33 @@ +"""Regression checks at the canonical launcher CLI boundary.""" + +import shutil +import subprocess +import sys +import tempfile +import unittest +from pathlib import Path + + +class DeepSeekLaunchTests(unittest.TestCase): + def test_known_broken_image_cannot_allocate(self): + launcher = Path(__file__).resolve().parents[1] / "deploy/deepseek-v41/launch.py" + with tempfile.TemporaryDirectory() as directory: + fixture = Path(directory) / "launch.py" + shutil.copy2(launcher, fixture) + config = launcher.with_name("h100-production.env").read_text() + lines = [ + "VLLM_IMAGE_ID=sha256:10b3c8fe9c38f6e87dfef21c8d0e457f76ab89b32892bb37a056375b25ddbf85" + if line.startswith("VLLM_IMAGE_ID=") else line + for line in config.splitlines() + ] + fixture.with_name("h100-production.env").write_text("\n".join(lines)) + model, cache = Path(directory) / "model", Path(directory) / "cache" + model.mkdir() + cache.mkdir() + result = subprocess.run( + [sys.executable, str(fixture), "--model-path", str(model), + "--cache-path", str(cache), "--execute"], + capture_output=True, text=True, timeout=10, check=False, + ) + self.assertNotEqual(result.returncode, 0) + self.assertIn("known failed compiler preflight", result.stderr) diff --git a/tests/test_default_language_model.py b/tests/test_default_language_model.py index bb3f5e1..18cb261 100644 --- a/tests/test_default_language_model.py +++ b/tests/test_default_language_model.py @@ -2,26 +2,26 @@ from c3r.deliberative import ( DEFAULT_LANGUAGE_MODEL, - DEFAULT_OPENROUTER_MODEL, default_provider_config, ) class DefaultLanguageModelTests(unittest.TestCase): - def test_openrouter_is_the_default_gateway(self) -> None: - config = default_provider_config(api_key="test-key") + def test_self_hosted_deepseek_is_the_default_gateway(self) -> None: + config = default_provider_config() self.assertEqual(DEFAULT_LANGUAGE_MODEL, "deepseek-ai/DeepSeek-V4.1-Flash") - self.assertEqual(config.model, DEFAULT_OPENROUTER_MODEL) - self.assertEqual(config.base_url, "https://openrouter.ai/api/v1") + self.assertEqual(config.model, "/model") + self.assertEqual(config.base_url, "http://127.0.0.1:8000/v1") + self.assertIsNone(config.api_key) def test_direct_deepseek_uses_official_alias(self) -> None: config = default_provider_config(api_key="test-key", gateway="deepseek") self.assertEqual(config.model, "deepseek-flash") self.assertEqual(config.base_url, "https://api.deepseek.com") - def test_credentials_are_never_optional(self) -> None: + def test_explicit_remote_compatibility_requires_credentials(self) -> None: with self.assertRaises(ValueError): - default_provider_config(api_key="") + default_provider_config(api_key="", gateway="deepseek") if __name__ == "__main__": diff --git a/tests/test_governed_trace_store.py b/tests/test_governed_trace_store.py new file mode 100644 index 0000000..8b956ba --- /dev/null +++ b/tests/test_governed_trace_store.py @@ -0,0 +1,183 @@ +import hmac +import sqlite3 +import tempfile +import unittest +from datetime import datetime, timedelta, timezone +from pathlib import Path + +from c3r.telemetry.governed_store import BoundGovernedTraceSink, GovernedTraceStore, SourceGrant +from c3r.telemetry.ledger_anchor import sign_head, verify_anchor +from c3r.telemetry.trace import DecisionTrace + + +NOW = datetime(2026, 9, 22, 12, tzinfo=timezone.utc) + + +def trace(**changes): + fields = dict( + run_id="run_001", + state_hash="a" * 64, + access_level="internal", + model_provider="deepseek_v4_1_flash", + candidate_ids=("recommend",), + probabilities={"route": (0.8, 0.2)}, + utility_quantiles={"selected_lower_bound": 0.3}, + selected_action_id="recommend", + authority_result="verified", + system_cost={"latency_ms": 18.0}, + task_outcome={"status": "controlled_success"}, + artifact_refs=(), + ) + fields.update(changes) + return DecisionTrace(**fields) + + +class GovernedTraceStoreTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.path = Path(self.temp.name) / "governed.sqlite3" + self.grant = SourceGrant( + source_id="c3r_internal_001", + owner="wilkont", + task_ids=frozenset({"task_001"}), + rights_attested=True, + ) + + def tearDown(self): + self.temp.cleanup() + + def store(self, *, now=NOW): + return GovernedTraceStore(self.path, grants=(self.grant,), clock=lambda: now) + + def test_admits_only_attested_registered_internal_task(self): + with self.store() as store: + record = store.append(trace(), source_id="c3r_internal_001", task_id="task_001") + self.assertEqual(len(store.records()), 1) + self.assertEqual(store.records()[0].record_hash, record.record_hash) + with self.assertRaisesRegex(ValueError, "unapproved source or task"): + store.append(trace(run_id="run_002"), source_id="laya_logs", task_id="task_001") + with self.assertRaisesRegex(ValueError, "unapproved source or task"): + store.append(trace(run_id="run_003"), source_id="c3r_internal_001", task_id="other") + with self.assertRaises(sqlite3.IntegrityError): + store.append(trace(), source_id="c3r_internal_001", task_id="task_001") + + def test_unattested_source_cannot_be_registered(self): + with self.assertRaisesRegex(ValueError, "rights"): + SourceGrant(source_id="imported", owner="wilkont", + task_ids=frozenset({"task_001"}), rights_attested=False) + + def test_trusted_host_can_bind_source_and_task_for_runtime_sink(self): + with self.store() as store: + sink = BoundGovernedTraceSink(store, source_id="c3r_internal_001", task_id="task_001") + record = sink.append(trace()) + self.assertEqual(store.records(), (record,)) + with self.assertRaisesRegex(ValueError, "unapproved source or task"): + BoundGovernedTraceSink(store, source_id="imported", task_id="task_001") + + def test_rejects_free_text_or_private_artifact_reference(self): + with self.store() as store: + with self.assertRaisesRegex(ValueError, "redaction"): + store.append(trace(task_outcome={"status": "email me at a@example.com"}), + source_id="c3r_internal_001", task_id="task_001") + with self.assertRaisesRegex(ValueError, "redaction"): + store.append(trace(artifact_refs=("gs://private/object",)), + source_id="c3r_internal_001", task_id="task_001") + self.assertEqual(store.records(), ()) + + def test_purges_after_30_days_and_preserves_remaining_chain(self): + with self.store(now=NOW - timedelta(days=31)) as store: + first = store.append(trace(), source_id="c3r_internal_001", task_id="task_001") + with self.store(now=NOW - timedelta(days=1)) as store: + self.assertEqual(store.purge_expired(), 1) + second = store.append(trace(run_id="run_002"), source_id="c3r_internal_001", + task_id="task_001") + with self.store(now=NOW) as store: + self.assertEqual(store.purge_expired(), 0) + remaining = store.records() + self.assertEqual(len(remaining), 1) + self.assertEqual(remaining[0].record_hash, second.record_hash) + self.assertEqual(remaining[0].previous_hash, first.record_hash) + self.assertTrue(store.verify()) + + def test_rejects_tampered_database_on_reopen(self): + with self.store() as store: + store.append(trace(), source_id="c3r_internal_001", task_id="task_001") + db = sqlite3.connect(self.path) + try: + db.execute("UPDATE records SET canonical_json = '{}' WHERE sequence = 1") + db.commit() + finally: + db.close() + with self.assertRaisesRegex(ValueError, "hash chain"): + self.store() + + def test_clock_rollback_cannot_relabel_traces_after_purge(self): + with self.store() as store: + store.append(trace(), source_id="c3r_internal_001", task_id="task_001") + with self.store(now=NOW + timedelta(days=31)) as store: + self.assertEqual(store.purge_expired(), 1) + with self.store(now=NOW - timedelta(days=1)) as store: + with self.assertRaisesRegex(ValueError, "clock moved backwards"): + store.append(trace(run_id="run_002"), source_id="c3r_internal_001", + task_id="task_001") + + def test_overdue_purge_blocks_new_collection_until_purged(self): + with self.store(now=NOW - timedelta(days=31)) as store: + store.append(trace(), source_id="c3r_internal_001", task_id="task_001") + with self.store(now=NOW) as store: + with self.assertRaisesRegex(ValueError, "retention purge overdue"): + store.append(trace(run_id="run_002"), source_id="c3r_internal_001", + task_id="task_001") + self.assertEqual(len(store.records()), 1) + self.assertEqual(store.purge_expired(), 1) + store.append(trace(run_id="run_002"), source_id="c3r_internal_001", + task_id="task_001") + self.assertEqual(len(store.records()), 1) + + def test_external_anchor_detects_clean_chain_tail_removal(self): + # The test key stands in for a separate signer; it is not deployment evidence. + key = b"fixture-only-signer" + sign = lambda payload: hmac.digest(key, payload, "sha256") + verify = lambda _key_id, payload, signature: hmac.compare_digest( + sign(payload), signature + ) + with self.store() as store: + store.append(trace(), source_id="c3r_internal_001", task_id="task_001") + store.append(trace(run_id="run_002"), source_id="c3r_internal_001", + task_id="task_001") + anchored = sign_head(store.snapshot_head(policy_version="fixture-v1"), + key_id="fixture-only", signer=sign) + + # A writer with DB access can erase a tail and make local replay valid. + # The independent prior signature must still reject the altered state. + db = sqlite3.connect(self.path) + try: + db.execute("DELETE FROM records WHERE sequence=2") + db.execute("UPDATE sqlite_sequence SET seq=1 WHERE name='records'") + db.execute("UPDATE metadata SET last_collected_at=(" + "SELECT collected_at FROM records WHERE sequence=1) WHERE id=1") + db.commit() + finally: + db.close() + with self.store() as store: + self.assertTrue(store.verify()) + current = store.snapshot_head(policy_version="fixture-v1") + self.assertFalse(verify_anchor(anchored, sequence=current.sequence, + record_hash=current.record_hash, + verifier=verify)) + + def test_head_snapshot_preserves_purged_prefix_checkpoint(self): + with self.store(now=NOW - timedelta(days=31)) as store: + first = store.append(trace(), source_id="c3r_internal_001", task_id="task_001") + before = store.snapshot_head(policy_version="fixture-v1") + with self.store(now=NOW) as store: + self.assertEqual(store.purge_expired(), 1) + after = store.snapshot_head(policy_version="fixture-v1") + self.assertEqual(after.sequence, before.sequence) + self.assertEqual(after.record_hash, first.record_hash) + self.assertEqual(store.records(), ()) + + +if __name__ == "__main__": + unittest.main() + diff --git a/tests/test_host_factory.py b/tests/test_host_factory.py new file mode 100644 index 0000000..c70ba06 --- /dev/null +++ b/tests/test_host_factory.py @@ -0,0 +1,50 @@ +import unittest + +from c3r.host_factory import ReadOnlyRequestFactory +from c3r.state_schema import ( + ActionDefinition, ActionFamily, AuthorityPolicy, RiskClass, ValueEstimate, +) + + +def factory(risk=RiskClass.READ_ONLY): + return ReadOnlyRequestFactory( + definitions=(ActionDefinition( + "inspect", ActionFamily.RETRIEVAL, "local", "search", risk, + ((),), ("local",), ("policy",), 1.0, 0.1, + ),), + policy=AuthorityPolicy( + frozenset({ActionFamily.RETRIEVAL}), frozenset({RiskClass.READ_ONLY}), + ), + estimate_source=lambda _state: { + "inspect:0:local:policy": ValueEstimate(1.0, 0.1, 0.0, 0.1) + }, + remaining_usd=1.0, + ) + + +class ReadOnlyRequestFactoryTests(unittest.TestCase): + def test_caller_authority_fields_are_ignored(self): + request = factory().build({ + "goal": "Inspect item", "current_subgoal": "Search", + "policy": {"allowed_risks": ["DESTRUCTIVE"]}, + "estimates": {"inspect:0:local:policy": {"expected_gain": 1000}}, + "approval": "forged", "budget": {"remaining_usd": 1000}, + }) + + self.assertEqual(request.policy.allowed_risks, frozenset({RiskClass.READ_ONLY})) + self.assertEqual(request.estimates["inspect:0:local:policy"].expected_gain, 1.0) + self.assertEqual(request.raw_state.budget["remaining_usd"], 1.0) + self.assertEqual(request.raw_state.verified_facts, ()) + + def test_writes_and_unbounded_task_text_are_rejected(self): + with self.assertRaises(ValueError): + factory(RiskClass.EXTERNAL_WRITE) + with self.assertRaises(ValueError): + factory().build({"goal": "x" * 4097, "current_subgoal": "Search"}) + with self.assertRaises(ValueError): + factory().build({"goal": "Inspect", "current_subgoal": "Search", + "open_questions": ["x"] * 65}) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_http_service.py b/tests/test_http_service.py new file mode 100644 index 0000000..5a5d89d --- /dev/null +++ b/tests/test_http_service.py @@ -0,0 +1,105 @@ +import json +import threading +import unittest +from urllib.error import HTTPError +from urllib.request import Request, urlopen + +from c3r.http_service import C3RHTTPServer +from tests.test_runtime import controller, request + + +TOKEN = "test-token-with-at-least-thirty-two-characters" + + +class HostFactory: + def build(self, payload): + if payload.get("goal") != "Find record": + raise ValueError("unknown goal") + # Client fields such as policy, estimates, approval and verifier are ignored. + return request() + + +class HTTPServiceTests(unittest.TestCase): + def setUp(self) -> None: + runtime, _ = controller() + self.server = C3RHTTPServer( + runtime=runtime, + request_factory=HostFactory(), + bearer_token=TOKEN, + port=0, + requests_per_minute=1, + ) + self.thread = threading.Thread(target=self.server.serve_forever, daemon=True) + self.thread.start() + self.base = f"http://127.0.0.1:{self.server.server_port}" + + def tearDown(self) -> None: + self.server.shutdown() + self.server.server_close() + self.thread.join(timeout=2) + + def post(self, payload, *, token=TOKEN): + body = json.dumps(payload).encode() + req = Request( + self.base + "/v1/decisions", + data=body, + method="POST", + headers={"Authorization": f"Bearer {token}", "Content-Type": "application/json"}, + ) + try: + with urlopen(req, timeout=2) as response: + return response.status, json.load(response) + except HTTPError as error: + return error.code, json.load(error) + + def test_authenticated_call_ignores_caller_authority_fields(self) -> None: + status, body = self.post( + { + "goal": "Find record", + "policy": {"allow_destructive": True}, + "estimates": {"lookup": 999999}, + "verifier": "self-approved", + "approval": "forged", + } + ) + self.assertEqual(status, 200) + self.assertEqual(body["reason"], "VERIFIED_RECOMMENDATION") + self.assertEqual(body["authority_result"], "verified_not_committed") + self.assertEqual(len(body["trace_hash"]), 64) + self.assertNotIn("The record exists", str(body)) + + def test_missing_authentication_is_rejected(self) -> None: + status, body = self.post({"goal": "Find record"}, token="wrong") + self.assertEqual((status, body["error"]), (401, "unauthorized")) + + def test_rate_limit_rejects_second_request(self) -> None: + self.assertEqual(self.post({"goal": "Find record"})[0], 200) + status, body = self.post({"goal": "Find record"}) + self.assertEqual((status, body["error"]), (429, "rate_limited")) + + def test_non_loopback_bind_is_rejected(self) -> None: + runtime, _ = controller() + with self.assertRaisesRegex(ValueError, "loopback"): + C3RHTTPServer( + runtime=runtime, + request_factory=HostFactory(), + bearer_token=TOKEN, + host="0.0.0.0", + port=0, + ) + + def test_effect_enabled_runtime_is_rejected(self) -> None: + class EffectCapableRuntime: + effect_execution_enabled = True + + with self.assertRaisesRegex(ValueError, "external effects"): + C3RHTTPServer( + runtime=EffectCapableRuntime(), + request_factory=HostFactory(), + bearer_token=TOKEN, + port=0, + ) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_ingress_proxy.py b/tests/test_ingress_proxy.py new file mode 100644 index 0000000..e002de8 --- /dev/null +++ b/tests/test_ingress_proxy.py @@ -0,0 +1,176 @@ +import json +import socket +import threading +import unittest +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from urllib.error import HTTPError +from urllib.request import Request, urlopen + +from c3r.ingress_proxy import C3RIngressServer + +CLIENT_TOKEN = "client-token-that-is-long-enough-for-tests" +UPSTREAM_TOKEN = "upstream-token-that-is-long-enough-for-tests" + + +class _UpstreamHandler(BaseHTTPRequestHandler): + def log_message(self, _format, *_args): + return + + def do_GET(self): + self.server.seen.append((self.path, dict(self.headers), b"")) + self._reply(200, {"status": "ok"}) + + def do_POST(self): + length = int(self.headers["Content-Length"]) + self.server.seen.append((self.path, dict(self.headers), self.rfile.read(length))) + self._reply(200, {"route": "recommendation"}) + + def _reply(self, status, payload): + body = json.dumps(payload).encode() + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + +class IngressProxyTests(unittest.TestCase): + def setUp(self): + self.upstream = ThreadingHTTPServer(("127.0.0.1", 0), _UpstreamHandler) + self.upstream.seen = [] + self.upstream_thread = threading.Thread(target=self.upstream.serve_forever, daemon=True) + self.upstream_thread.start() + self.ingress = C3RIngressServer( + upstream_port=self.upstream.server_port, + client_token=CLIENT_TOKEN, + upstream_token=UPSTREAM_TOKEN, + host="127.0.0.1", + port=0, + upstream_timeout_seconds=0.5, + ) + self.ingress_thread = threading.Thread(target=self.ingress.serve_forever, daemon=True) + self.ingress_thread.start() + self.base = f"http://127.0.0.1:{self.ingress.server_port}" + + def tearDown(self): + self.ingress.shutdown() + self.ingress.server_close() + self.ingress_thread.join(timeout=2) + self.upstream.shutdown() + self.upstream.server_close() + self.upstream_thread.join(timeout=2) + + def request(self, path, *, method="GET", token=CLIENT_TOKEN, payload=None): + headers = {"Authorization": "Bearer cloud-run-identity-token"} + if token is not None: + headers["X-C3R-Token"] = token + body = None if payload is None else json.dumps(payload).encode() + if body is not None: + headers["Content-Type"] = "application/json" + request = Request(self.base + path, data=body, headers=headers, method=method) + try: + with urlopen(request, timeout=2) as response: + return response.status, json.load(response) + except HTTPError as error: + return error.code, json.load(error) + + def test_private_decision_forwards_only_internal_authorization(self): + status, body = self.request("/v1/decisions", method="POST", payload={"goal": "inspect"}) + self.assertEqual((status, body["route"]), (200, "recommendation")) + path, headers, forwarded = self.upstream.seen[-1] + self.assertEqual(path, "/v1/decisions") + self.assertEqual(headers["Authorization"], f"Bearer {UPSTREAM_TOKEN}") + self.assertNotIn("X-C3R-Token", headers) + self.assertEqual(json.loads(forwarded), {"goal": "inspect"}) + + def test_sdk_bearer_authentication_without_cloud_run_header(self): + request = Request(self.base + "/v1/models", headers={ + "Authorization": "Bearer " + CLIENT_TOKEN}) + with urlopen(request, timeout=2) as response: + self.assertEqual(response.status, 200) + self.assertEqual(self.upstream.seen[-1][1]["Authorization"], + "Bearer " + UPSTREAM_TOKEN) + + def test_caller_tenant_and_project_headers_are_rejected(self): + for name in ("X-Tenant-ID", "X-Project-ID", "OpenAI-Organization", "OpenAI-Project", + "X-C3R-Organization", "X-C3R-Project"): + with self.subTest(header=name): + request = Request(self.base + "/v1/models", headers={ + "Authorization": "Bearer " + CLIENT_TOKEN, name: "forged-context"}) + with self.assertRaises(HTTPError) as raised: + urlopen(request, timeout=2) + self.assertEqual(raised.exception.code, 401) + self.assertEqual(self.upstream.seen, []) + + def test_missing_or_wrong_client_token_never_reaches_upstream(self): + for token in (None, "wrong", "invalid-café"): + status, body = self.request("/v1/decisions", method="POST", token=token, + payload={"goal": "inspect"}) + self.assertEqual((status, body["error"]), (401, "unauthorized")) + self.assertEqual(self.upstream.seen, []) + + def test_health_is_unprivileged_but_metrics_require_token(self): + self.assertEqual(self.request("/health", token=None)[0], 200) + self.assertEqual(self.request("/metrics", token=None)[0], 401) + self.assertEqual(self.request("/ready", token=None)[0], 401) + self.assertEqual(self.request("/v1/models", token=None)[0], 401) + + def test_stateless_paths_forward_without_client_credential_leak(self): + for path in ("/v1/c3r/decide", "/v1/c3r/rank", "/v1/system-one"): + self.assertEqual(self.request(path, method="POST", payload={"goal": "inspect"})[0], 200) + seen_path, headers, _ = self.upstream.seen[-1] + self.assertEqual(seen_path, path) + self.assertNotIn("X-C3R-Token", headers) + + def test_rejects_unknown_path_without_contacting_upstream(self): + self.assertEqual(self.request("/admin")[0], 404) + self.assertEqual(self.request("/v1/decisions?debug=1", method="POST", + payload={"goal": "inspect"})[0], 404) + self.assertEqual(self.upstream.seen, []) + + def test_rejects_ambiguous_framing_and_non_json_body(self): + request = Request( + self.base + "/v1/decisions", + data=b"{}", + headers={"X-C3R-Token": CLIENT_TOKEN, "Content-Type": "text/plain"}, + method="POST", + ) + with self.assertRaises(HTTPError) as raised: + urlopen(request, timeout=2) + self.assertEqual(raised.exception.code, 415) + + for framing in ( + b"Content-Length: 2\r\nTransfer-Encoding: chunked\r\n", + b"Content-Length: 2\r\nContent-Length: 2\r\n", + ): + packet = ( + b"POST /v1/decisions HTTP/1.0\r\nHost: localhost\r\n" + + f"X-C3R-Token: {CLIENT_TOKEN}\r\n".encode() + + b"Content-Type: application/json\r\n" + + framing + b"\r\n{}" + ) + with socket.create_connection(("127.0.0.1", self.ingress.server_port), timeout=2) as sock: + sock.sendall(packet) + response = sock.recv(4096) + self.assertTrue(response.startswith((b"HTTP/1.0 400 ", b"HTTP/1.0 413 "))) + self.assertEqual(self.upstream.seen, []) + + def test_upstream_failure_is_not_mistaken_for_success(self): + self.upstream.shutdown() + self.upstream.server_close() + status, body = self.request("/v1/decisions", method="POST", payload={"goal": "inspect"}) + self.assertEqual((status, body["error"]), (503, "upstream_unavailable")) + + def test_upstream_route_must_be_loopback_and_secrets_distinct(self): + with self.assertRaisesRegex(ValueError, "loopback"): + C3RIngressServer(upstream_host="example.com", upstream_port=8081, + client_token=CLIENT_TOKEN, upstream_token=UPSTREAM_TOKEN, + host="127.0.0.1", port=0) + with self.assertRaisesRegex(ValueError, "different"): + C3RIngressServer(upstream_port=8081, client_token=CLIENT_TOKEN, + upstream_token=CLIENT_TOKEN, host="127.0.0.1", port=0) + + +if __name__ == "__main__": + unittest.main() + diff --git a/tests/test_internal_readiness.py b/tests/test_internal_readiness.py new file mode 100644 index 0000000..07fcd9a --- /dev/null +++ b/tests/test_internal_readiness.py @@ -0,0 +1,197 @@ +"""Authenticated maintenance HTTP behavior, independent from public ingress.""" +import hashlib +import json +import tempfile +import threading +import unittest +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path +from typing import cast +from urllib.error import HTTPError +from urllib.request import Request, urlopen + +from c3r.internal_readiness import InternalReadinessServer +from c3r.local_artifacts import PinnedLocalArtifacts + +TOKEN = "maintenance-test-token-with-more-than-thirty-two-characters" + + +class InternalReadinessTests(unittest.TestCase): + def setUp(self) -> None: + self.api_servers = tuple(ThreadingHTTPServer(("127.0.0.1", 0), BaseHTTPRequestHandler) + for _ in range(2)) + self.api_workers = tuple(threading.Thread(target=server.serve_forever, daemon=True) + for server in self.api_servers) + for worker in self.api_workers: + worker.start() + + def tearDown(self) -> None: + for server, worker in zip(self.api_servers, self.api_workers): + if worker.is_alive(): + server.shutdown() + server.server_close() + worker.join(timeout=2) + + def test_stopped_api_worker_revokes_ready_even_when_provider_checks_are_positive(self): + report: dict[str, object] = { + "runtime": True, "clm_qwen": True, "deepseek": True, + "required_local_artifact_files": True, + "required_local_artifact_manifest_sha256": "a" * 64, + } + self.api_servers[0].shutdown() + self.api_workers[0].join(timeout=2) + status, body = self.call(InternalReadinessServer( + token=TOKEN, port=0, probe=lambda: report, api_workers=self.api_workers)) + self.assertEqual((status, body["status"]), (503, "not_ready")) + self.assertFalse(cast(dict[str, object], body["checks"])["runtime"]) + + def call(self, server: InternalReadinessServer, path: str = "/internal/ready", *, + token: str = TOKEN, method: str = "GET", + bind_api_workers: bool = True) -> tuple[int, dict[str, object]]: + if bind_api_workers and not server.api_workers: + server.api_workers = self.api_workers + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + try: + request = Request(f"http://127.0.0.1:{server.server_port}" + path, + headers={"Authorization": "Bearer " + token}, method=method) + try: + with urlopen(request, timeout=2) as response: + return int(response.status), cast(dict[str, object], json.load(response)) + except HTTPError as error: + return error.code, cast(dict[str, object], json.load(error)) + finally: + server.shutdown() + server.server_close() + thread.join(timeout=2) + + def test_missing_readiness_evidence_is_not_ready(self): + status, body = self.call(InternalReadinessServer(token=TOKEN, port=0)) + self.assertEqual(status, 503) + self.assertEqual(body["status"], "not_ready") + + def test_positive_provider_report_without_api_workers_is_not_ready(self): + report: dict[str, object] = { + "runtime": True, "clm_qwen": True, "deepseek": True, + "required_local_artifact_files": True, + "required_local_artifact_manifest_sha256": "a" * 64, + } + status, body = self.call(InternalReadinessServer( + token=TOKEN, port=0, probe=lambda: report), bind_api_workers=False) + self.assertEqual((status, body["status"]), (503, "not_ready")) + + def test_worker_exit_during_provider_check_revokes_ready(self): + def external_probe() -> dict[str, object]: + self.api_servers[1].shutdown() + self.api_workers[1].join(timeout=2) + return {"runtime": True, "clm_qwen": True, "deepseek": True, + "required_local_artifact_files": True, + "required_local_artifact_manifest_sha256": "a" * 64} + + status, body = self.call(InternalReadinessServer( + token=TOKEN, port=0, probe=external_probe, api_workers=self.api_workers)) + self.assertEqual((status, body["status"]), (503, "not_ready")) + self.assertFalse(cast(dict[str, object], body["checks"])["runtime"]) + + def test_shutdown_request_revokes_ready_while_api_workers_still_run(self): + stopping = threading.Event() + + def external_probe() -> dict[str, object]: + stopping.set() + return {"runtime": True, "clm_qwen": True, "deepseek": True, + "required_local_artifact_files": True, + "required_local_artifact_manifest_sha256": "a" * 64} + + status, body = self.call(InternalReadinessServer( + token=TOKEN, port=0, probe=external_probe, + api_workers=self.api_workers, stopping=stopping)) + self.assertEqual((status, body["status"]), (503, "not_ready")) + + def test_changed_required_file_revokes_readiness_on_next_request(self): + with tempfile.TemporaryDirectory() as scratch: + root = Path(scratch) + artifact = root / "artifact" + artifact.write_bytes(b"abc") + manifest = root / "required-files.json" + manifest.write_text(json.dumps({"schema": "c3r-required-local-files-v1", "files": [{ + "path": str(artifact), "sha256": + "ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad", + "max_bytes": 3}]})) + pin = hashlib.sha256(manifest.read_bytes()).hexdigest() + verifier = PinnedLocalArtifacts(manifest, pin) + + def provider_readback(): + return {"runtime": True, "clm_qwen": True, "deepseek": True, + "required_local_artifact_files": verifier.verify(), + "required_local_artifact_manifest_sha256": pin} + + status, body = self.call(InternalReadinessServer( + token=TOKEN, port=0, probe=provider_readback)) + self.assertEqual((status, body["status"]), (200, "ready")) + self.assertEqual(body["required_local_artifact_manifest_sha256"], pin) + artifact.write_bytes(b"xyz") + status, body = self.call(InternalReadinessServer( + token=TOKEN, port=0, probe=provider_readback)) + self.assertEqual((status, body["status"]), (503, "not_ready")) + + def test_other_tokens_and_routes_cannot_use_the_internal_channel(self): + for path, token, method, expected in ( + ("/internal/ready", "wrong", "GET", 401), + ("/health", TOKEN, "GET", 404), + ("/v1/responses", TOKEN, "POST", 405), + ): + with self.subTest(path=path, method=method): + status, _ = self.call(InternalReadinessServer(token=TOKEN, port=0), + path, token=token, method=method) + self.assertEqual(status, expected) + + + def test_provider_loss_never_reuses_a_previous_ready_result(self): + external_health: dict[str, object] = { + "runtime": True, "clm_qwen": True, "deepseek": True, + "required_local_artifact_files": True, + "required_local_artifact_manifest_sha256": "a" * 64, + } + self.assertEqual(self.call(InternalReadinessServer( + token=TOKEN, port=0, probe=lambda: external_health))[0], 200) + for check in ("runtime", "clm_qwen", "deepseek", "required_local_artifact_files"): + external_health[check] = False + status, body = self.call(InternalReadinessServer( + token=TOKEN, port=0, probe=lambda: external_health)) + self.assertEqual((status, body["status"]), (503, "not_ready")) + external_health[check] = True + + def test_unqualified_file_manifests_do_not_admit_recovery(self): + with tempfile.TemporaryDirectory() as scratch: + manifest = Path(scratch) / "required.json" + invalid: tuple[dict[str, object], ...] = ( + {"schema": "c3r-required-local-files-v1", "files": []}, + {"schema": "c3r-required-local-files-v1", "files": [{ + "path": scratch, "sha256": "a" * 64, "max_bytes": 3}]}, + {"schema": "c3r-required-local-files-v1", "files": [{ + "path": str(manifest), "sha256": "a" * 64, + "max_bytes": 67108865}]}, + ) + for payload in invalid: + manifest.write_text(json.dumps(payload)) + pin = hashlib.sha256(manifest.read_bytes()).hexdigest() + verifier = PinnedLocalArtifacts(manifest, pin) + status, body = self.call(InternalReadinessServer( + token=TOKEN, port=0, probe=lambda verifier=verifier, pin=pin: { + "runtime": True, "clm_qwen": True, "deepseek": True, + "required_local_artifact_files": verifier.verify(), + "required_local_artifact_manifest_sha256": pin, + })) + self.assertEqual((status, body["status"]), (503, "not_ready")) + + def test_provider_exception_remains_not_ready_without_error_details(self): + def unavailable() -> dict[str, object]: + raise RuntimeError("private-provider-error-must-not-be-disclosed") + + status, body = self.call(InternalReadinessServer(token=TOKEN, port=0, probe=unavailable)) + self.assertEqual((status, body["status"]), (503, "not_ready")) + self.assertNotIn("private-provider", json.dumps(body)) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_ledger_anchor.py b/tests/test_ledger_anchor.py new file mode 100644 index 0000000..064f14c --- /dev/null +++ b/tests/test_ledger_anchor.py @@ -0,0 +1,71 @@ +import hmac +import unittest +from dataclasses import replace +from datetime import datetime, timezone + +from c3r.telemetry.ledger_anchor import capture_head, sign_head, verify_anchor +from c3r.telemetry.trace import DecisionTrace +from c3r.telemetry.trace_ledger import TraceLedger + + +def trace(run_id: str) -> DecisionTrace: + return DecisionTrace( + run_id=run_id, + state_hash="a" * 64, + access_level="internal", + model_provider="fixture", + candidate_ids=("STOP",), + probabilities={}, + utility_quantiles={"STOP": 0.0}, + selected_action_id="STOP", + authority_result="not-requested", + system_cost={"latency_ms": 1.0}, + task_outcome={"success": True}, + artifact_refs=(), + ) + + +class LedgerAnchorTests(unittest.TestCase): + def test_signed_head_matches_chain_and_detects_removal_or_alteration(self) -> None: + # Test-only signer. Deployment must use a key unavailable to the writer. + test_key = b"fixture-only-key" + sign = lambda payload: hmac.digest(test_key, payload, "sha256") + verify = lambda _key_id, payload, signature: hmac.compare_digest( + sign(payload), signature + ) + ledger = TraceLedger() + ledger.append(trace("one")) + ledger.append(trace("two")) + head = capture_head( + sequence=2, + record_hash=ledger.records[-1].record_hash, + policy_version="fixture-v1", + clock=lambda: datetime(2026, 9, 23, tzinfo=timezone.utc), + ) + anchor = sign_head(head, key_id="test-only", signer=sign) + + self.assertTrue(TraceLedger.verify(ledger.records)) + self.assertTrue(verify_anchor(anchor, sequence=2, + record_hash=ledger.records[-1].record_hash, + verifier=verify)) + self.assertFalse(verify_anchor(anchor, sequence=1, + record_hash=ledger.records[0].record_hash, + verifier=verify)) + self.assertFalse(verify_anchor(anchor, sequence=2, + record_hash="f" * 64, verifier=verify)) + self.assertFalse(verify_anchor(replace(anchor, signature_hex="00"), sequence=2, + record_hash=head.record_hash, verifier=verify)) + + def test_rejects_invalid_head_and_empty_signature(self) -> None: + with self.assertRaises(ValueError): + capture_head(sequence=-1, record_hash="0" * 64, policy_version="v1") + with self.assertRaises(ValueError): + capture_head(sequence=0, record_hash="not-a-hash", policy_version="v1") + head = capture_head(sequence=0, record_hash="0" * 64, policy_version="v1") + with self.assertRaises(ValueError): + sign_head(head, key_id="test-only", signer=lambda _payload: b"") + + +if __name__ == "__main__": + unittest.main() + diff --git a/tests/test_official_sdk_acceptance.py b/tests/test_official_sdk_acceptance.py new file mode 100644 index 0000000..c2de1bb --- /dev/null +++ b/tests/test_official_sdk_acceptance.py @@ -0,0 +1,123 @@ +"""Unmodified official SDKs against the real local C3R HTTP composition. + +Fixture inference proves protocol compatibility, not GPU or release qualification. +Install the optional Python SDK and set C3R_JS_SDK_MODULE for the JavaScript run. +""" +import importlib.util +import json +import os +import subprocess +import tempfile +import threading +import unittest +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path + +from c3r.adapters.providers import ProviderAdapter, ProviderConfig, ProviderKind +from c3r.api_access import AccessStore +from c3r.candidate_compiler import CandidateCompiler +from c3r.cvoc import RobustCvocController +from c3r.feature_flags import FeatureFlags +from c3r.host_factory import ReadOnlyRequestFactory +from c3r.http_service import C3RHTTPServer +from c3r.ingress_proxy import C3RIngressServer +from c3r.responses import ResponsesService +from c3r.runtime import StandaloneController +from c3r.state_compiler import StateCompiler +from c3r.state_schema import ActionDefinition, ActionFamily, AuthorityPolicy, RiskClass +from c3r.telemetry.ephemeral import EphemeralTraceSink +from c3r.verifier_firewall import VerifierDecision, VerifierFirewall, VerifierPolicy + + +class GenerationFixture(BaseHTTPRequestHandler): + def log_message(self, format: str, *args: object) -> None: + return + + def do_POST(self): + payload = json.loads(self.rfile.read(int(self.headers["Content-Length"]))) + self.send_response(200) + self.send_header("Content-Type", "text/event-stream" if payload.get("stream") + else "application/json") + self.end_headers() + if payload.get("stream"): + chunks: tuple[dict[str, object], ...] = ( + {"choices": [{"delta": {"content": "Inspect safely.", + "reasoning_content": "NEVER_PUBLIC"}, + "finish_reason": None}]}, + {"choices": [{"delta": {}, "finish_reason": "stop"}]}, + {"choices": [], "usage": {"prompt_tokens": 4, "completion_tokens": 3}}, + ) + for value in chunks: + self.wfile.write(("data: " + json.dumps(value) + "\n\n").encode()) + self.wfile.flush() + self.wfile.write(b"data: [DONE]\n\n") + else: + self.wfile.write(json.dumps({"choices": [{"message": { + "content": "Inspect safely.", "reasoning_content": "NEVER_PUBLIC"}, + "finish_reason": "stop"}], "usage": { + "prompt_tokens": 4, "completion_tokens": 3}}).encode()) + + +class OfficialSDKAcceptanceTests(unittest.TestCase): + def setUp(self): + self.scratch = tempfile.TemporaryDirectory() + self.addCleanup(self.scratch.cleanup) + store = AccessStore(Path(self.scratch.name) / "access.sqlite") + store.create_project("fixture", "sdk", rpm=100, key_rps=10) + self.key = store.issue_key("fixture", "sdk", {"responses:write"}).secret + runtime = StandaloneController( + flags=FeatureFlags(enabled_requested=True, deliberative_requested=True), + compiler=StateCompiler(), candidates=CandidateCompiler(), cvoc=RobustCvocController(), + verifier=VerifierFirewall({"policy": lambda _: VerifierDecision(True, "read-only fixture")}, + VerifierPolicy(default_verifier="policy"), attestation_key=b"sdk-fixture-only-key"), + ledger=EphemeralTraceSink(), + ) + factory = ReadOnlyRequestFactory(definitions=(ActionDefinition( + "DELIBERATE", ActionFamily.DELIBERATE, "compute", "generate", + RiskClass.READ_ONLY, ((),), ("local",), ("policy",), 0, 0, + ),), policy=AuthorityPolicy(frozenset({ActionFamily.DELIBERATE}), + frozenset({RiskClass.READ_ONLY})), + estimate_source=lambda _: {}, remaining_usd=0) + generation = ThreadingHTTPServer(("127.0.0.1", 0), GenerationFixture) + provider = ProviderAdapter(ProviderConfig( + "fixture", ProviderKind.OPENAI_COMPATIBLE, + f"http://127.0.0.1:{generation.server_port}/v1", "fixture-model", None)) + backend = C3RHTTPServer(runtime=runtime, request_factory=factory, port=0, + bearer_token="backend-fixture-token-with-over-32-characters", + responses=ResponsesService(runtime, factory, provider)) + gateway = C3RIngressServer(upstream_port=backend.server_port, host="127.0.0.1", port=0, + client_token="unused-staging-fixture-token-with-over-32-characters", + upstream_token=backend.bearer_token, access_store=store) + self.servers = [generation, backend, gateway] + for server in self.servers: + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + self.addCleanup(thread.join, 3) + self.addCleanup(server.server_close) + self.addCleanup(server.shutdown) + self.base = f"http://127.0.0.1:{gateway.server_port}/v1" + + @unittest.skipUnless(importlib.util.find_spec("openai"), "optional official Python SDK") + def test_official_python_responses_and_sse(self): + from openai import OpenAI + + with OpenAI(api_key=self.key, base_url=self.base, max_retries=0, timeout=5) as client: + response = client.responses.create(model="c3r-core", input="Inspect", store=False) + self.assertEqual(response.output_text, "Inspect safely.") + assert response.usage is not None + self.assertEqual(response.usage.total_tokens, 7) + self.assertGreater(response.created_at, 0) + self.assertEqual(response.created_at, int(response.created_at)) + self.assertNotIn("NEVER_PUBLIC", response.model_dump_json()) + events = list(client.responses.create(model="c3r-core", input="Inspect", stream=True)) + self.assertEqual("".join(event.delta for event in events + if event.type == "response.output_text.delta"), "Inspect safely.") + self.assertEqual(events[-1].type, "response.completed") + + @unittest.skipUnless(os.environ.get("C3R_JS_SDK_MODULE"), "optional official JavaScript SDK") + def test_official_javascript_responses_and_sse(self): + env = {**os.environ, "C3R_SDK_TEST_URL": self.base, "C3R_SDK_TEST_KEY": self.key} + result = subprocess.run(["node", str(Path(__file__).with_name("official_sdk_acceptance.mjs"))], + env=env, capture_output=True, text=True, timeout=15, check=False) + self.assertEqual(result.returncode, 0, result.stderr) + self.assertEqual(result.stdout.strip(), "SDK_ACCEPTANCE_PASS") diff --git a/tests/test_provider_bridge.py b/tests/test_provider_bridge.py new file mode 100644 index 0000000..2e2e5b2 --- /dev/null +++ b/tests/test_provider_bridge.py @@ -0,0 +1,73 @@ +import json +import unittest + +from c3r.adapters.providers import ( + ProviderAdapter, + ProviderConfig, + ProviderKind, + TransportResponse, +) +from c3r.deliberative.provider_bridge import ProviderDeliberator +from c3r.state_compiler import StateCompiler +from c3r.state_schema import RawState + + +def state(data_boundary: str): + result = StateCompiler().compile( + RawState(goal="Plan a read-only check", current_subgoal="Inspect", data_boundary=data_boundary) + ) + assert result.state is not None + return result.state + + +class ProviderBridgeTests(unittest.TestCase): + def test_local_deepseek_receives_compiled_state_and_returns_plan(self) -> None: + captured = [] + + def transport(_url, _headers, payload): + captured.append(payload) + return TransportResponse( + 200, + { + "choices": [{"message": {"content": json.dumps({ + "plan": ["inspect"], + "assumptions": [], + "uncertainty": [], + "candidate_commitments": [], + "requested_actions": [], + })}}], + "usage": {"prompt_tokens": 12, "completion_tokens": 8}, + }, + 12.0, + ) + + adapter = ProviderAdapter( + ProviderConfig( + "deepseek-local", ProviderKind.OPENAI_COMPATIBLE, + "http://127.0.0.1:8000/v1", "deepseek-v4.1-flash", None, + ), + transport=transport, + ) + result = ProviderDeliberator(adapter).deliberate(state("local")) + + self.assertEqual(result.deliberation.plan, ("inspect",)) + self.assertEqual(result.observed_cost["latency_ms"], 12.0) + self.assertIn("Plan a read-only check", captured[0]["messages"][1]["content"]) + + def test_remote_provider_is_not_called_for_local_data(self) -> None: + calls = [] + adapter = ProviderAdapter( + ProviderConfig( + "frontier", ProviderKind.OPENAI_COMPATIBLE, + "https://example.invalid/v1", "frontier-model", "test-key", + ), + transport=lambda *_: calls.append(True), + ) + + with self.assertRaises(ValueError): + ProviderDeliberator(adapter).deliberate(state("local")) + self.assertEqual(calls, []) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_responses_stream.py b/tests/test_responses_stream.py new file mode 100644 index 0000000..5d17d03 --- /dev/null +++ b/tests/test_responses_stream.py @@ -0,0 +1,236 @@ +"""Authenticated HTTP streaming contract backed by a local generation service.""" + +import json +import select +import socket +import threading +import time +import unittest +from http.client import HTTPConnection +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +from c3r.adapters.providers import ProviderAdapter, ProviderConfig, ProviderKind +from c3r.host_factory import ReadOnlyRequestFactory +from c3r.http_service import C3RHTTPServer +from c3r.responses import ResponsesService +from c3r.state_schema import ActionDefinition, ActionFamily, AuthorityPolicy, RiskClass +from tests.test_runtime import controller + +TOKEN = "stream-test-token-with-at-least-thirty-two-characters" + + +class _GenerationHandler(BaseHTTPRequestHandler): + def log_message(self, *_args): + return + + def do_POST(self): + body = json.loads(self.rfile.read(int(self.headers["Content-Length"]))) + self.server.requests.append(body) + if self.server.hold_headers: + self.server.headers_pending.set() + deadline = time.monotonic() + 3 + while time.monotonic() < deadline: + ready, _, _ = select.select([self.connection], [], [], 0.05) + if ready and not self.connection.recv(1, socket.MSG_PEEK): + self.server.abort_seen.set() + return + return + self.send_response(200) + self.send_header("Content-Type", "text/event-stream") + self.end_headers() + events = [ + {"choices": [{"delta": {"content": "Check ", "reasoning_content": "PRIVATE"}, + "finish_reason": None}]}, + {"choices": [{"delta": {"content": "charges."}, + "finish_reason": "stop" if self.server.finish_with_text else None}]}, + ] + if not self.server.finish_with_text: + events.append({"choices": [{"delta": {}, "finish_reason": "stop"}]}) + events.append( + {"choices": [], "usage": {"prompt_tokens": 5, "completion_tokens": 3}}, + ) + for index, data in enumerate(events): + if index == 1 and self.server.hold_after_first: + deadline = time.monotonic() + 3 + while not self.server.release_next.is_set() and time.monotonic() < deadline: + ready, _, _ = select.select([self.connection], [], [], 0.05) + if ready and not self.connection.recv(1, socket.MSG_PEEK): + self.server.abort_seen.set() + return + if index == 1 and self.server.fail_after_first: + self.wfile.write(b"data: {broken\n\n") + self.wfile.flush() + return + try: + self.wfile.write(("data: " + json.dumps(data) + "\n\n").encode()) + self.wfile.flush() + except OSError: + return + try: + self.wfile.write(b"data: [DONE]\n\n") + self.wfile.flush() + except OSError: + return + + +class ResponsesStreamTests(unittest.TestCase): + def setUp(self): + self.backend = ThreadingHTTPServer(("127.0.0.1", 0), _GenerationHandler) + self.backend.requests = [] + self.backend.hold_headers = False + self.backend.headers_pending = threading.Event() + self.backend.hold_after_first = False + self.backend.fail_after_first = False + self.backend.finish_with_text = False + self.backend.release_next = threading.Event() + self.backend.abort_seen = threading.Event() + self.backend_thread = threading.Thread(target=self.backend.serve_forever, daemon=True) + self.backend_thread.start() + runtime, _ = controller(deliberative=True) + factory = ReadOnlyRequestFactory( + definitions=(ActionDefinition( + "DELIBERATE", ActionFamily.DELIBERATE, "compute", "generate", + RiskClass.READ_ONLY, ((),), ("local",), ("policy",), 0, 0, + ),), + policy=AuthorityPolicy(frozenset({ActionFamily.DELIBERATE}), + frozenset({RiskClass.READ_ONLY})), + estimate_source=lambda _: {}, remaining_usd=0, + ) + provider = ProviderAdapter(ProviderConfig( + "local", ProviderKind.OPENAI_COMPATIBLE, + f"http://127.0.0.1:{self.backend.server_port}/v1", "model", None, + )) + self.api = C3RHTTPServer(runtime=runtime, request_factory=factory, + bearer_token=TOKEN, port=0, + responses=ResponsesService(runtime, factory, provider)) + self.api_thread = threading.Thread(target=self.api.serve_forever, daemon=True) + self.api_thread.start() + + def tearDown(self): + self.backend.release_next.set() + self.api.shutdown() + self.api.server_close() + self.api_thread.join(timeout=2) + self.backend.shutdown() + self.backend.server_close() + self.backend_thread.join(timeout=2) + + def test_stream_emits_real_text_deltas_and_terminal_response(self): + connection = HTTPConnection("127.0.0.1", self.api.server_port, timeout=3) + connection.request("POST", "/v1/responses", body=json.dumps({ + "model": "c3r-core", "input": "Find record", "stream": True, + }), headers={"Authorization": f"Bearer {TOKEN}", "Content-Type": "application/json"}) + response = connection.getresponse() + wire = response.read().decode() + connection.close() + self.assertEqual(response.status, 200) + self.assertEqual(response.getheader("Content-Type"), "text/event-stream") + self.assertEqual([line.removeprefix("event: ") for line in wire.splitlines() + if line.startswith("event: ")], [ + "response.created", "response.output_text.delta", "response.output_text.delta", + "response.output_text.done", "response.completed", + ]) + self.assertIn("Check charges.", wire) + self.assertNotIn("PRIVATE", wire) + self.assertTrue(self.backend.requests[0]["stream"]) + + def test_rejected_stream_does_not_start_backend_generation(self): + runtime, _ = controller(deliberative=True, accepted=False) + self.api.runtime = runtime + self.api.responses.runtime = runtime + connection = HTTPConnection("127.0.0.1", self.api.server_port, timeout=3) + connection.request("POST", "/v1/responses", body=json.dumps({ + "model": "c3r-core", "input": "Find record", "stream": True, + }), headers={"Authorization": f"Bearer {TOKEN}", "Content-Type": "application/json"}) + response = connection.getresponse() + self.assertEqual(response.status, 503) + response.read() + connection.close() + self.assertEqual(self.backend.requests, []) + + def test_backend_error_emits_failed_terminal_event_without_private_reasoning(self): + self.backend.fail_after_first = True + connection = HTTPConnection("127.0.0.1", self.api.server_port, timeout=3) + connection.request("POST", "/v1/responses", body=json.dumps({ + "model": "c3r-core", "input": "Find record", "stream": True, + }), headers={"Authorization": f"Bearer {TOKEN}", "Content-Type": "application/json"}) + response = connection.getresponse() + wire = response.read().decode() + connection.close() + self.assertEqual(response.status, 200) + self.assertIn("event: response.failed", wire) + self.assertNotIn("event: response.completed", wire) + self.assertNotIn("PRIVATE", wire) + + def test_final_delta_with_finish_reason_is_not_lost(self): + self.backend.finish_with_text = True + connection = HTTPConnection("127.0.0.1", self.api.server_port, timeout=3) + connection.request("POST", "/v1/responses", body=json.dumps({ + "model": "c3r-core", "input": "Find record", "stream": True, + }), headers={"Authorization": f"Bearer {TOKEN}", "Content-Type": "application/json"}) + response = connection.getresponse() + wire = response.read().decode() + connection.close() + self.assertEqual(response.status, 200) + self.assertIn('"text":"Check charges."', wire) + self.assertIn("event: response.completed", wire) + + def test_disconnect_before_backend_headers_cancels_generation(self): + self.api.generation_capacity = threading.BoundedSemaphore(1) + self.backend.hold_headers = True + body = json.dumps({"model": "c3r-core", "input": "Find record", "stream": True}).encode() + first = socket.create_connection(("127.0.0.1", self.api.server_port), timeout=2) + first.sendall(("POST /v1/responses HTTP/1.1\r\nHost: localhost\r\n" + f"Authorization: Bearer {TOKEN}\r\nContent-Type: application/json\r\n" + f"Content-Length: {len(body)}\r\n\r\n").encode() + body) + self.assertTrue(self.backend.headers_pending.wait(timeout=1)) + first.shutdown(socket.SHUT_RDWR) + first.close() + self.assertTrue(self.backend.abort_seen.wait(timeout=1)) + self.backend.hold_headers = False + time.sleep(0.1) + second = HTTPConnection("127.0.0.1", self.api.server_port, timeout=2) + second.request("POST", "/v1/responses", body=body, headers={ + "Authorization": f"Bearer {TOKEN}", "Content-Type": "application/json"}) + response = second.getresponse() + self.assertEqual(response.status, 200) + response.read() + second.close() + + def test_disconnected_stream_releases_generation_capacity(self): + self.api.generation_capacity = threading.BoundedSemaphore(1) + self.backend.hold_after_first = True + first = HTTPConnection("127.0.0.1", self.api.server_port, timeout=3) + body = json.dumps({"model": "c3r-core", "input": "Find record", "stream": True}) + headers = {"Authorization": f"Bearer {TOKEN}", "Content-Type": "application/json"} + first.request("POST", "/v1/responses", body=body, headers=headers) + first_response = first.getresponse() + self.assertEqual(first_response.status, 200) + while first_response.readline() != b"event: response.output_text.delta\n": + pass + second = HTTPConnection("127.0.0.1", self.api.server_port, timeout=3) + second.request("POST", "/v1/responses", body=body, headers=headers) + refused = second.getresponse() + self.assertEqual(refused.status, 429) + refused.read() + second.close() + first_response.fp.raw._sock.shutdown(socket.SHUT_RDWR) + first_response.close() + first.close() + deadline = time.monotonic() + 1 + capacity_released = False + while time.monotonic() < deadline: + if (self.api.metrics.snapshot().get("stream_disconnect") + and self.api.generation_capacity.acquire(blocking=False)): + self.api.generation_capacity.release() + capacity_released = True + break + time.sleep(0.02) + self.assertEqual(self.api.metrics.snapshot().get("stream_disconnect"), 1) + self.assertTrue(capacity_released) + self.assertTrue(self.backend.abort_seen.wait(timeout=1)) + self.backend.release_next.set() + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_retention_job.py b/tests/test_retention_job.py new file mode 100644 index 0000000..5eb817e --- /dev/null +++ b/tests/test_retention_job.py @@ -0,0 +1,134 @@ +import io +import unittest +from datetime import datetime, timedelta, timezone +from unittest.mock import patch + +from c3r.retention_job import ( + DELETE_AFTER_DAYS, GcsJsonClient, StoredObject, _created_at, + _verify_runtime_project, bucket_from_environment, purge, +) + + +NOW = datetime(2026, 9, 23, 16, 0, tzinfo=timezone.utc) +BUCKET = "colomboai-c3r-staging-traces-123456789012" + + +class FakeClient: + def __init__(self, objects): + self.bucket = BUCKET + self.objects = list(objects) + self.deleted = [] + + def list_all(self): + return tuple(self.objects) + + def delete_generation(self, obj): + self.deleted.append((obj.name, obj.generation)) + self.objects = [item for item in self.objects if item != obj] + + +class RetentionJobTests(unittest.TestCase): + def test_deletes_only_expired_generation_and_verifies_empty_overdue_set(self): + old = StoredObject("traces/old", 3, NOW - timedelta(days=DELETE_AFTER_DAYS)) + fresh = StoredObject("traces/new", 4, NOW - timedelta(days=1)) + client = FakeClient([old, fresh]) + + report = purge(client, now=NOW) + + self.assertEqual(client.deleted, [(old.name, old.generation)]) + self.assertEqual(client.objects, [fresh]) + self.assertEqual(report["expired_remaining"], 0) + self.assertFalse(report["trace_collection_enabled"]) + + def test_surviving_expired_object_fails_closed(self): + old = StoredObject("traces/old", 3, NOW - timedelta(days=29)) + + class NonDeletingClient(FakeClient): + def delete_generation(self, obj): + self.deleted.append((obj.name, obj.generation)) + + with self.assertRaisesRegex(RuntimeError, "expired objects remain"): + purge(NonDeletingClient([old]), now=NOW) + + def test_duplicate_inventory_and_non_utc_clock_rejected(self): + old = StoredObject("traces/old", 3, NOW - timedelta(days=29)) + with self.assertRaisesRegex(ValueError, "duplicate"): + purge(FakeClient([old, old]), now=NOW) + with self.assertRaisesRegex(ValueError, "UTC"): + purge(FakeClient([]), now=NOW.replace(tzinfo=None)) + + def test_distinct_generations_are_purged_without_treating_them_as_duplicates(self): + old = StoredObject("traces/replaced", 3, NOW - timedelta(days=29)) + fresh = StoredObject("traces/replaced", 4, NOW - timedelta(days=1)) + client = FakeClient([old, fresh]) + + report = purge(client, now=NOW) + + self.assertEqual(client.deleted, [(old.name, old.generation)]) + self.assertEqual(client.objects, [fresh]) + self.assertEqual(report["expired_remaining"], 0) + + def test_gcs_transport_lists_versions_and_deletes_exact_generation(self): + client = object.__new__(GcsJsonClient) + client.bucket = BUCKET + client._storage_api = "https://storage.googleapis.com/storage/v1/b/" + BUCKET + "/o" + requests = [] + + def fake_request(method, url): + requests.append((method, url)) + if method == "GET": + return {"items": [ + {"name": "traces/replaced", "generation": "3", "timeCreated": "2026-08-01T00:00:00Z"}, + {"name": "traces/replaced", "generation": "4", "timeCreated": "2026-09-23T00:00:00Z"}, + ]} + return None + + client._request = fake_request + objects = client.list_all() + client.delete_generation(objects[0]) + + self.assertEqual([obj.generation for obj in objects], [3, 4]) + self.assertIn("versions=true", requests[0][1]) + self.assertIn("generation=3", requests[1][1]) + self.assertNotIn("ifGenerationMatch", requests[1][1]) + + def test_creation_timestamp_must_be_timezone_aware(self): + self.assertEqual(_created_at("2026-09-23T16:00:00Z"), NOW) + with self.assertRaisesRegex(ValueError, "UTC offset"): + _created_at("2026-09-23T16:00:00") + + def test_dedicated_bucket_must_be_explicit_and_narrow(self): + self.assertEqual(bucket_from_environment({"C3R_TRACE_BUCKET": BUCKET}), BUCKET) + for value in ("", "colomboai-c3r-private-traces-123456789012", + "unrelated-bucket", "colomboai-c3r-staging-traces-123456789012/other"): + with self.subTest(value=value), self.assertRaisesRegex(ValueError, "C3R_TRACE_BUCKET"): + bucket_from_environment({"C3R_TRACE_BUCKET": value}) + client = FakeClient([]) + client.bucket = "unrelated-bucket" + with self.assertRaisesRegex(ValueError, "C3R_TRACE_BUCKET"): + purge(client, now=NOW) + + def test_runtime_project_must_match_bucket_suffix(self): + _verify_runtime_project(BUCKET, "123456789012") + with self.assertRaisesRegex(ValueError, "project number"): + _verify_runtime_project(BUCKET, "999999999999") + + def test_gcs_client_rejects_wrong_project_before_credential_request(self): + with patch("c3r.retention_job.urlopen", return_value=io.BytesIO(b"999999999999")) as open_url: + with self.assertRaisesRegex(ValueError, "project number"): + GcsJsonClient(BUCKET) + self.assertEqual(open_url.call_count, 1) + + def test_gcs_client_accepts_metadata_bound_bucket(self): + token = b'{"access_token":"' + b"x" * 24 + b'"}' + with patch("c3r.retention_job.urlopen", side_effect=[ + io.BytesIO(b"123456789012"), io.BytesIO(token), + ]) as open_url: + client = GcsJsonClient(BUCKET) + self.assertEqual(client.bucket, BUCKET) + self.assertEqual(open_url.call_count, 2) + + +if __name__ == "__main__": + unittest.main() + diff --git a/tests/test_runtime.py b/tests/test_runtime.py new file mode 100644 index 0000000..4c2f486 --- /dev/null +++ b/tests/test_runtime.py @@ -0,0 +1,340 @@ +import unittest +from dataclasses import replace + +from c3r.adapters.providers import ProviderExecutionResult +from c3r.candidate_compiler import CandidateCompiler +from c3r.cvoc import RobustCvocController +from c3r.deliberative.envelope import DeliberativeResult +from c3r.feature_flags import FeatureFlags +from c3r.runtime import RuntimeRequest, StandaloneController +from c3r.state_compiler import StateCompiler +from c3r.state_schema import ( + ActionDefinition, + ActionFamily, + AuthorityPolicy, + Provenance, + RawState, + RiskClass, + ValueEstimate, +) +from c3r.system_one.calibration import TemperatureCalibrator +from c3r.system_one.clm_adapter import ClmAdapter +from c3r.system_one.fast_path import CalibratedFastPath, LayaFastPath +from c3r.system_one.laya_adapter import LayaAdapter +from c3r.telemetry.trace_ledger import TraceLedger +from c3r.verifier_firewall import VerifierDecision, VerifierFirewall, VerifierPolicy + + +KEY = b"verification-test-key" +ACTION_ID = "lookup:0:local:policy" + + +def request(*, risk: RiskClass = RiskClass.READ_ONLY, provenance: bool = True) -> RuntimeRequest: + fact = "The record exists" + raw = RawState( + goal="Find record", + current_subgoal="Lookup", + verified_facts=(fact,), + available_action_families=(ActionFamily.TOOL,), + budget={"remaining_usd": 1.0}, + provenance={fact: Provenance("fixture", "2026-09-22T00:00:00Z")} + if provenance + else {}, + ) + definition = ActionDefinition( + id="lookup", + family=ActionFamily.TOOL, + subgroup="records", + operation="get", + risk_class=risk, + argument_variants=((('record_id', 'fixture-1'),),), + placements=("local",), + verifier_ids=("policy",), + optimistic_utility=1.0, + estimated_cost=0.1, + data_boundary="local", + ) + return RuntimeRequest( + raw_state=raw, + definitions=(definition,), + policy=AuthorityPolicy( + frozenset({ActionFamily.TOOL}), frozenset({risk}) + ), + estimates={ACTION_ID: ValueEstimate(0.9, 0.1, 0.0, 0.1)}, + run_id="fixture-run", + ) + + +def controller( + *, + enabled: bool = True, + system_one: bool = False, + deliberative: bool = False, + accepted: bool = True, + executor=None, + fast_path=None, + deliberator=None, + system_one_provider: str = "clm", +) -> tuple[StandaloneController, TraceLedger]: + ledger = TraceLedger() + verifier = VerifierFirewall( + {"policy": lambda _: VerifierDecision(accepted, "policy fixture")}, + VerifierPolicy(default_verifier="policy"), + attestation_key=KEY, + ) + runtime = StandaloneController( + flags=FeatureFlags( + enabled_requested=enabled, + system_one_requested=system_one, + deliberative_requested=deliberative, + system_one_provider=system_one_provider, + ), + compiler=StateCompiler(), + candidates=CandidateCompiler(), + cvoc=RobustCvocController(), + verifier=verifier, + ledger=ledger, + executor=executor, + fast_path=fast_path, + deliberator=deliberator, + ) + return runtime, ledger + + +class RuntimeTests(unittest.TestCase): + def test_executor_configuration_is_rejected_before_any_effect(self) -> None: + effects = [] + with self.assertRaisesRegex(ValueError, "external effects"): + controller(executor=effects.append) + + self.assertEqual(effects, []) + + def test_verified_recommendation_has_no_effect_and_is_traced(self) -> None: + runtime, ledger = controller() + outcome = runtime.run(request()) + + self.assertEqual(outcome.route, "recommendation") + self.assertEqual(outcome.selected_action_id, ACTION_ID) + self.assertEqual(outcome.authority_result, "verified_not_committed") + self.assertEqual(len(ledger.records), 1) + self.assertTrue(TraceLedger.verify(ledger.records)) + self.assertNotIn("The record exists", ledger.to_jsonl()) + + def test_global_disable_stops_before_candidate_or_provider_execution(self) -> None: + runtime, _ = controller(enabled=False) + outcome = runtime.run(request()) + + self.assertEqual(outcome.reason, "C3R_DISABLED") + self.assertFalse(runtime.effect_execution_enabled) + + def test_missing_provenance_stops_before_any_effect(self) -> None: + runtime, _ = controller() + outcome = runtime.run(request(provenance=False)) + + self.assertEqual(outcome.reason, "STATE_UNSAFE_TO_COMPRESS") + self.assertFalse(runtime.effect_execution_enabled) + + def test_verifier_rejection_stops_before_commit(self) -> None: + runtime, _ = controller(accepted=False) + outcome = runtime.run(request()) + + self.assertEqual(outcome.reason, "VERIFICATION_REJECTED") + self.assertFalse(runtime.effect_execution_enabled) + + def test_unavailable_action_family_is_never_compiled(self) -> None: + runtime, _ = controller() + req = request() + req = replace(req, raw_state=replace(req.raw_state, available_action_families=())) + + outcome = runtime.run(req) + + self.assertEqual(outcome.reason, "NO_SAFE_ACTION") + self.assertIsNone(outcome.selected_action_id) + + def test_external_write_is_unavailable_to_recommendation_only_controller(self) -> None: + runtime, ledger = controller() + outcome = runtime.run(request(risk=RiskClass.EXTERNAL_WRITE)) + + self.assertEqual(outcome.reason, "EFFECT_EXECUTION_UNAVAILABLE") + self.assertEqual(outcome.route, "deterministic") + self.assertIsNone(outcome.selected_action_id) + self.assertTrue(TraceLedger.verify(ledger.records)) + + def test_uncalibrated_system_one_abstains_into_non_authoritative_deliberation(self) -> None: + adapter = LayaAdapter( + "convaiinnovations/laya", + "1c5edc17a7acd8701df6fc341c0d179f1c62c982", + backend=lambda _state, _questions: { + "STOP_NOW": (0.0, 3.0), + "DELIBERATION_REQUIRED": (3.0, 0.0), + }, + ) + fast_path = LayaFastPath(adapter=adapter, calibrator=TemperatureCalibrator({})) + + class Deliberator: + def deliberate(self, _state): + return {"plan": ["inspect"]} + + runtime, _ = controller( + system_one=True, + deliberative=True, + fast_path=fast_path, + system_one_provider="laya", + deliberator=Deliberator(), + ) + outcome = runtime.run(request()) + + self.assertEqual(outcome.route, "deliberative") + self.assertEqual(outcome.reason, "SYSTEM_ONE_ABSTAINED") + self.assertIsNone(outcome.selected_action_id) + self.assertFalse(runtime.effect_execution_enabled) + + def test_default_clm_never_silently_runs_laya(self) -> None: + adapter = LayaAdapter( + "convaiinnovations/laya", + "1c5edc17a7acd8701df6fc341c0d179f1c62c982", + backend=lambda _state, _questions: {}, + ) + runtime, _ = controller( + system_one=True, + fast_path=LayaFastPath(adapter=adapter, calibrator=TemperatureCalibrator({})), + ) + outcome = runtime.run(request()) + self.assertEqual(outcome.reason, "SYSTEM_ONE_PROVIDER_MISMATCH") + + def test_clm_rank_is_advisory_and_abstains_without_calibration(self) -> None: + def rank(payload): + options = payload["answers"] + probability = 1.0 / len(options) + return { + "model": "clm-latest", + "ranked": [ + {"candidate": option, "prob": probability} + for option in options + ], + } + + adapter = ClmAdapter(revision="a" * 64, transport=rank) + runtime, ledger = controller( + system_one=True, + fast_path=CalibratedFastPath( + adapter=adapter, calibrator=TemperatureCalibrator({}) + ), + ) + outcome = runtime.run(request()) + self.assertEqual(outcome.reason, "SYSTEM_ONE_ABSTAINED_NO_PROVIDER") + self.assertEqual(outcome.fast_path.candidate_probabilities, (1.0,)) + self.assertIn('"model_provider":"Contrastive-LM/CLM"', ledger.records[-1].canonical_json) + + def test_clm_outage_escalates_without_granting_authority(self) -> None: + def unavailable(_payload): + raise OSError("CLM unavailable") + + class Deliberator: + def deliberate(self, _state): + return {"plan": ["inspect"]} + + runtime, ledger = controller( + system_one=True, + deliberative=True, + fast_path=CalibratedFastPath( + adapter=ClmAdapter(revision="a" * 64, transport=unavailable), + calibrator=TemperatureCalibrator({}), + ), + deliberator=Deliberator(), + ) + outcome = runtime.run(request()) + self.assertEqual(outcome.route, "deliberative") + self.assertEqual(outcome.reason, "SYSTEM_ONE_FAILURE") + self.assertIsNone(outcome.selected_action_id) + self.assertTrue(TraceLedger.verify(ledger.records)) + + def test_provider_usage_is_recorded_without_granting_authority(self) -> None: + class Deliberator: + def deliberate(self, _state): + return ProviderExecutionResult( + DeliberativeResult(("inspect",), (), (), (), ()), + {"latency_ms": 12.0, "input_tokens": 10.0}, + "deepseek-local", "deepseek-v4.1-flash", + ) + + runtime, ledger = controller(deliberative=True, deliberator=Deliberator()) + req = request() + req = RuntimeRequest( + req.raw_state, req.definitions, req.policy, {}, req.run_id, + ) + # A deliberative candidate, rather than a tool, is selected by CVoC. + definition = ActionDefinition( + "reason", ActionFamily.DELIBERATE, "model", "plan", RiskClass.READ_ONLY, + ((),), ("local",), ("policy",), 1.0, 0.1, + ) + req = RuntimeRequest( + replace(req.raw_state, available_action_families=(ActionFamily.DELIBERATE,)), + (definition,), + AuthorityPolicy(frozenset({ActionFamily.DELIBERATE}), frozenset({RiskClass.READ_ONLY})), + {"reason:0:local:policy": ValueEstimate(0.9, 0.1, 0.0, 0.1)}, req.run_id, + ) + outcome = runtime.run(req) + + self.assertEqual(outcome.route, "deliberative") + self.assertIsNone(outcome.selected_action_id) + self.assertEqual(ledger.records[0].record_hash, outcome.ledger_record.record_hash) + self.assertIn('"model_provider":"deepseek-local"', ledger.records[0].canonical_json) + self.assertIn('"latency_ms":12.0', ledger.records[0].canonical_json) + + def test_model_requested_unsafe_action_never_reaches_executor(self) -> None: + class Deliberator: + def deliberate(self, _state): + return ProviderExecutionResult( + DeliberativeResult((), (), (), (), ("delete all records",)), + {"latency_ms": 1.0}, "untrusted-model", "fixture", + ) + + runtime, _ = controller(deliberative=True, deliberator=Deliberator()) + base = request() + definition = ActionDefinition( + "reason", ActionFamily.DELIBERATE, "model", "plan", RiskClass.READ_ONLY, + ((),), ("local",), ("policy",), 1.0, 0.1, + ) + req = RuntimeRequest( + replace(base.raw_state, available_action_families=(ActionFamily.DELIBERATE,)), + (definition,), + AuthorityPolicy(frozenset({ActionFamily.DELIBERATE}), frozenset({RiskClass.READ_ONLY})), + {"reason:0:local:policy": ValueEstimate(0.9, 0.1, 0.0, 0.1)}, + base.run_id, + ) + + outcome = runtime.run(req) + + self.assertEqual(outcome.route, "deliberative") + self.assertFalse(runtime.effect_execution_enabled) + self.assertEqual(outcome.deliberation.requested_actions, ("delete all records",)) + + def test_provider_outage_falls_back_without_effect(self) -> None: + class Deliberator: + def deliberate(self, _state): + raise OSError("provider unavailable") + + runtime, _ = controller(deliberative=True, deliberator=Deliberator()) + base = request() + definition = ActionDefinition( + "reason", ActionFamily.DELIBERATE, "model", "plan", RiskClass.READ_ONLY, + ((),), ("local",), ("policy",), 1.0, 0.1, + ) + req = RuntimeRequest( + replace(base.raw_state, available_action_families=(ActionFamily.DELIBERATE,)), + (definition,), + AuthorityPolicy(frozenset({ActionFamily.DELIBERATE}), frozenset({RiskClass.READ_ONLY})), + {"reason:0:local:policy": ValueEstimate(0.9, 0.1, 0.0, 0.1)}, + base.run_id, + ) + + outcome = runtime.run(req) + + self.assertEqual(outcome.reason, "DELIBERATIVE_FAILURE") + self.assertEqual(outcome.route, "deterministic") + self.assertFalse(runtime.effect_execution_enabled) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_serve.py b/tests/test_serve.py new file mode 100644 index 0000000..6a9c280 --- /dev/null +++ b/tests/test_serve.py @@ -0,0 +1,208 @@ +import json +import os +import socket +import subprocess +import sys +import tempfile +import threading +import time +import unittest +from urllib.error import HTTPError +from urllib.request import Request, urlopen + +from c3r.candidate_compiler import CandidateCompiler +from c3r.cvoc import RobustCvocController +from c3r.feature_flags import FeatureFlags +from c3r.runtime import StandaloneController +from c3r.serve import build_servers, load_host_builder +from c3r.staging_host import build as staging_build +from c3r.state_compiler import StateCompiler +from c3r.telemetry.ephemeral import EphemeralTraceSink +from c3r.verifier_firewall import VerifierFirewall, VerifierPolicy +from tests.test_http_service import HostFactory +from tests.test_runtime import controller + +CLIENT_TOKEN = "client-token-with-at-least-thirty-two-characters" +BACKEND_TOKEN = "backend-token-with-at-least-thirty-two-characters" + + +def free_port(): + with socket.socket() as sock: + sock.bind(("127.0.0.1", 0)) + return sock.getsockname()[1] + + +def config(): + return { + "C3R_HOST_ENTRYPOINT": "trusted_host:build", + "C3R_CLIENT_TOKEN": CLIENT_TOKEN, + "C3R_BACKEND_TOKEN": BACKEND_TOKEN, + "PORT": str(free_port()), + "C3R_BACKEND_PORT": str(free_port()), + } + + +class ServeTests(unittest.TestCase): + def test_internal_token_without_explicit_internal_port_fails_startup(self): + values = config() + values.update({"C3R_HOST_ENTRYPOINT": "c3r.staging_host:build", + "C3R_INTERNAL_READY_TOKEN": "internal-test-token-never-an-api-token"}) + result = subprocess.run([sys.executable, "-m", "c3r.serve"], + env={**os.environ, **values}, capture_output=True, + text=True, timeout=3) + self.assertNotEqual(result.returncode, 0) + self.assertIn("C3R_INTERNAL_READY_PORT", result.stderr) + + def test_module_entrypoint_starts_production_host_without_claiming_provider_readiness(self): + scratch = tempfile.TemporaryDirectory() + self.addCleanup(scratch.cleanup) + values = config() + values.update({ + "C3R_HOST_ENTRYPOINT": "c3r.production_host:build", + "C3R_MODE": "production_inference", "C3R_ENABLED": "true", + "C3R_API_ACCESS_DB": os.path.join(scratch.name, "access.sqlite3"), + "C3R_SYSTEM_ONE": "true", "C3R_DELIBERATIVE": "true", + "C3R_SYSTEM_ONE_PROVIDER": "clm", "C3R_TRACE_COLLECTION": "false", + "C3R_ONLINE_LEARNING": "false", "C3R_INGRESS_HOST": "127.0.0.1", + "C3R_CLM_CONTAINER_DIGEST": "sha256:" + "a" * 64, + "C3R_INTERNAL_READY_PORT": str(free_port()), + "C3R_INTERNAL_READY_TOKEN": "internal-token-distinct-from-both-api-tokens", + }) + process = subprocess.Popen([sys.executable, "-m", "c3r.serve"], + env={**os.environ, **values}, stdout=subprocess.DEVNULL, + stderr=subprocess.PIPE) + try: + deadline = time.monotonic() + 5 + while time.monotonic() < deadline: + try: + with urlopen("http://127.0.0.1:" + values["PORT"] + "/health", timeout=1) as response: + self.assertEqual(json.load(response)["status"], "ok") + break + except OSError: + if process.poll() is not None: + self.fail("production entrypoint exited before health was reachable") + time.sleep(0.05) + else: + self.fail("production entrypoint did not become reachable") + req = Request("http://127.0.0.1:" + values["C3R_INTERNAL_READY_PORT"] + + "/internal/ready", headers={"Authorization": "Bearer " + + values["C3R_INTERNAL_READY_TOKEN"]}) + with self.assertRaises(HTTPError) as failure: + urlopen(req, timeout=2) + self.assertEqual(failure.exception.code, 503) + self.assertEqual(json.load(failure.exception)["status"], "not_ready") + with self.assertRaises(HTTPError) as public_failure: + urlopen("http://127.0.0.1:" + values["PORT"] + "/internal/ready", timeout=2) + self.assertEqual(public_failure.exception.code, 404) + finally: + process.terminate() + process.communicate(timeout=5) + + def test_production_entrypoint_requires_key_mode_database_before_binding(self): + values = config() + values.update({"C3R_HOST_ENTRYPOINT": "c3r.production_host:build", + "C3R_MODE": "production_inference", "C3R_ENABLED": "true", + "C3R_SYSTEM_ONE": "true", "C3R_SYSTEM_ONE_PROVIDER": "clm", + "C3R_DELIBERATIVE": "true", "C3R_CLM_CONTAINER_DIGEST": "sha256:" + "a" * 64}) + result = subprocess.run([sys.executable, "-m", "c3r.serve"], + env={**os.environ, **values}, capture_output=True, + text=True, timeout=5) + self.assertNotEqual(result.returncode, 0) + self.assertIn("C3R_API_ACCESS_DB", result.stderr) + + def test_missing_host_or_secret_fails_before_binding(self): + values = config() + del values["C3R_HOST_ENTRYPOINT"] + with self.assertRaisesRegex(ValueError, "C3R_HOST_ENTRYPOINT"): + build_servers(values) + values = config() + del values["C3R_CLIENT_TOKEN"] + with self.assertRaisesRegex(ValueError, "C3R_CLIENT_TOKEN"): + build_servers(values) + + def test_invalid_port_and_equal_tokens_fail(self): + values = config() + values["PORT"] = "0" + with self.assertRaisesRegex(ValueError, "PORT"): + build_servers(values, builder_loader=lambda _: lambda: (controller()[0], HostFactory())) + values = config() + values["C3R_BACKEND_TOKEN"] = CLIENT_TOKEN + with self.assertRaisesRegex(ValueError, "different"): + build_servers(values, builder_loader=lambda _: lambda: (controller()[0], HostFactory())) + + def test_effect_enabled_host_is_rejected(self): + class EffectCapableRuntime: + effect_execution_enabled = True + + with self.assertRaisesRegex(ValueError, "recommendation-only"): + build_servers( + config(), + builder_loader=lambda _: lambda: ( + EffectCapableRuntime(), HostFactory() + ), + ) + + def test_host_reference_must_be_explicit(self): + for reference in ("module", "module:", ":build", "module:bad.name"): + with self.subTest(reference=reference), self.assertRaises(ValueError): + load_host_builder(reference) + + def test_composed_health_path(self): + backend, ingress = build_servers( + config(), + builder_loader=lambda _: lambda: (controller()[0], HostFactory()), + ) + threads = [ + threading.Thread(target=backend.serve_forever, daemon=True), + threading.Thread(target=ingress.serve_forever, daemon=True), + ] + try: + for thread in threads: + thread.start() + with urlopen(f"http://127.0.0.1:{ingress.server_port}/health", timeout=2) as response: + self.assertEqual(response.status, 200) + self.assertEqual(response.read(), b'{"status":"ok"}') + finally: + ingress.shutdown() + backend.shutdown() + ingress.server_close() + backend.server_close() + for thread in threads: + thread.join(timeout=2) + + def test_production_mode_rejects_persistent_and_disabled_hosts(self): + values = config() + values["C3R_MODE"] = "production_inference" + with self.assertRaisesRegex(ValueError, "ephemeral trace sink"): + build_servers( + values, builder_loader=lambda _: lambda: (controller()[0], HostFactory()), + ) + with self.assertRaisesRegex(ValueError, "decisions enabled"): + build_servers(values, builder_loader=lambda _: staging_build) + + def test_production_mode_rejects_collection_and_online_learning(self): + for key in ("C3R_TRACE_COLLECTION", "C3R_ONLINE_LEARNING"): + values = config() + values["C3R_MODE"] = "production_inference" + values[key] = "true" + with self.subTest(key=key), self.assertRaises(ValueError): + build_servers(values, builder_loader=lambda _: staging_build) + + def test_production_mode_requires_system_one_path(self): + runtime = StandaloneController( + flags=FeatureFlags(enabled_requested=True), + compiler=StateCompiler(), candidates=CandidateCompiler(), + cvoc=RobustCvocController(), + verifier=VerifierFirewall({}, VerifierPolicy(default_verifier="none"), + attestation_key=b"test-key"), + ledger=EphemeralTraceSink(), + ) + values = config() + values["C3R_MODE"] = "production_inference" + with self.assertRaisesRegex(ValueError, "System-One"): + build_servers(values, builder_loader=lambda _: lambda: (runtime, HostFactory())) + + +if __name__ == "__main__": + unittest.main() + diff --git a/tests/test_sqlite_ledger.py b/tests/test_sqlite_ledger.py new file mode 100644 index 0000000..bb98661 --- /dev/null +++ b/tests/test_sqlite_ledger.py @@ -0,0 +1,44 @@ +import sqlite3 +import tempfile +import unittest +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path + +from c3r.telemetry.sqlite_ledger import SqliteTraceLedger +from c3r.telemetry.trace_ledger import TraceLedger +from tests.test_trace_ledger import trace + + +class SqliteTraceLedgerTests(unittest.TestCase): + def test_records_survive_restart_and_concurrent_appends(self) -> None: + with tempfile.TemporaryDirectory() as directory: + path = Path(directory) / "traces.sqlite3" + ledger = SqliteTraceLedger(path) + with ThreadPoolExecutor(max_workers=8) as pool: + list(pool.map(lambda i: ledger.append(trace(f"run-{i}")), range(100))) + ledger.close() + + reopened = SqliteTraceLedger(path) + self.assertEqual(len(reopened.records), 100) + self.assertTrue(TraceLedger.verify(reopened.records)) + reopened.close() + + def test_tampering_is_detected_on_restart(self) -> None: + with tempfile.TemporaryDirectory() as directory: + path = Path(directory) / "traces.sqlite3" + ledger = SqliteTraceLedger(path) + ledger.append(trace("run-1")) + ledger.close() + db = sqlite3.connect(path) + try: + db.execute("UPDATE records SET record_hash = ? WHERE sequence = 1", ("0" * 64,)) + db.commit() + finally: + db.close() + + with self.assertRaises(ValueError): + SqliteTraceLedger(path) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_staging_host.py b/tests/test_staging_host.py new file mode 100644 index 0000000..19936d0 --- /dev/null +++ b/tests/test_staging_host.py @@ -0,0 +1,36 @@ +import unittest + +from c3r.serve import load_host_builder +from c3r.staging_host import EphemeralStagingSink +from c3r.telemetry.trace import DecisionTrace + + +class StagingHostTests(unittest.TestCase): + def test_staging_builder_cannot_enable_decisions_or_effects(self): + controller, factory = load_host_builder("c3r.staging_host:build")() + self.assertFalse(controller.effect_execution_enabled) + request = factory.build({"goal": "fixture", "current_subgoal": "check"}) + self.assertEqual(request.estimates, {}) + outcome = controller.run(request) + self.assertEqual(outcome.reason, "C3R_DISABLED") + self.assertIsNone(outcome.selected_action_id) + + def test_staging_sink_has_no_row_store_or_cross_request_chain(self): + trace = DecisionTrace( + run_id="fixture_run", state_hash="a" * 64, access_level="internal", + model_provider="fixture", candidate_ids=(), probabilities={}, + utility_quantiles={}, selected_action_id=None, + authority_result="not_attempted", system_cost={}, + task_outcome={"status": "C3R_DISABLED"}, artifact_refs=(), + ) + sink = EphemeralStagingSink() + first = sink.append(trace) + second = sink.append(trace) + self.assertEqual(first.record_hash, second.record_hash) + self.assertEqual(first.previous_hash, "0" * 64) + self.assertFalse(hasattr(sink, "records")) + + +if __name__ == "__main__": + unittest.main() + diff --git a/tests/test_stateless_api.py b/tests/test_stateless_api.py new file mode 100644 index 0000000..1625f12 --- /dev/null +++ b/tests/test_stateless_api.py @@ -0,0 +1,291 @@ +"""Public HTTP contract for the stateless recommendation-only release.""" + +import json +import threading +import unittest +from urllib.error import HTTPError +from urllib.request import Request, urlopen + +from c3r.adapters.providers import ProviderAdapter, ProviderConfig, ProviderKind, TransportResponse +from c3r.host_factory import ReadOnlyRequestFactory +from c3r.http_service import C3RHTTPServer +from c3r.readiness import CachedReadiness +from c3r.responses import ResponsesService +from c3r.state_schema import ( + ActionDefinition, + ActionFamily, + AuthorityPolicy, + RiskClass, + ValueEstimate, +) +from c3r.system_one.advisory import AdvisoryFastPath +from c3r.system_one.clm_adapter import ClmAdapter +from c3r.system_one.inference import SystemOneInference +from tests.test_runtime import controller, request + +TOKEN = "stateless-test-token-with-at-least-thirty-two-characters" + + +class _Factory: + def build(self, payload): + if payload.get("goal") != "Find record": + raise ValueError("unknown goal") + return request() + + +class StatelessAPITests(unittest.TestCase): + def setUp(self): + runtime, _ = controller() + self.server = C3RHTTPServer( + runtime=runtime, request_factory=_Factory(), bearer_token=TOKEN, + port=0, requests_per_minute=20, + ) + self.thread = threading.Thread(target=self.server.serve_forever, daemon=True) + self.thread.start() + self.base = f"http://127.0.0.1:{self.server.server_port}" + + def tearDown(self): + self.server.shutdown() + self.server.server_close() + self.thread.join(timeout=2) + + def call(self, path, *, method="POST", token=TOKEN, payload=None): + headers = {"Authorization": f"Bearer {token}"} + data = None if payload is None else json.dumps(payload).encode() + if data is not None: + headers["Content-Type"] = "application/json" + req = Request(self.base + path, data=data, headers=headers, method=method) + try: + with urlopen(req, timeout=2) as response: + return response.status, json.load(response) + except HTTPError as error: + return error.code, json.load(error) + + def configure_ranker(self, transport, *, enabled=True, system_one=True, ready=lambda: True): + adapter = ClmAdapter("a" * 64, transport=transport) + self.server.runtime, _ = controller(enabled=enabled, system_one=system_one, + fast_path=AdvisoryFastPath(adapter)) + self.server.system_one = SystemOneInference(adapter, readiness=ready) + + def test_typed_ranker_respects_disable_switches_without_provider_calls(self): + calls = [] + for enabled, system_one in ((False, True), (True, False)): + self.configure_ranker(lambda payload: calls.append(payload), enabled=enabled, + system_one=system_one) + for path in ("/v1/system-one", "/v1/c3r/rank"): + status, _ = self.call(path, payload={"state": "test", "candidates": ["A", "B"]}) + self.assertEqual(status, 503) + self.assertEqual(calls, []) + + def test_model_availability_is_independent_of_system_two(self): + self.configure_ranker(lambda _: {}, ready=lambda: True) + status, body = self.call("/v1/models", method="GET") + self.assertEqual(status, 200) + self.assertFalse(body["data"][0]["available"]) + self.assertTrue(body["data"][1]["available"]) + self.assertTrue(body["data"][2]["available"]) + + def test_metadata_coalesces_health_checks_and_expires_cached_status(self): + now, calls = [0.0], [] + def probe(): + calls.append(1) + return len(calls) == 1 + readiness = CachedReadiness(probe, clock=lambda: now[0]) + self.configure_ranker(lambda _: {}, ready=readiness) + for _ in range(4): + status, body = self.call("/v1/models", method="GET") + self.assertEqual(status, 200) + self.assertTrue(body["data"][1]["available"]) + self.assertEqual(len(calls), 1) + now[0] = 16 + _, body = self.call("/v1/models", method="GET") + self.assertFalse(body["data"][1]["available"]) + self.assertEqual(len(calls), 2) + + def test_decide_and_rank_are_recommendation_only(self): + for path in ("/v1/c3r/decide", "/v1/c3r/rank", "/v1/system-one"): + status, body = self.call(path, payload={"goal": "Find record"}) + self.assertEqual(status, 200) + self.assertEqual(body["authority_result"], "verified_not_committed") + self.assertFalse(body["effect_executed"]) + self.assertNotIn("confidence", body) + if path.endswith("rank") or path.endswith("system-one"): + self.assertEqual(body["candidate_ranking"], []) + self.assertTrue(body["abstained"]) + + def test_decide_exposes_a_non_reasoning_cvoc_summary(self): + status, body = self.call("/v1/c3r/decide", payload={"goal": "Find record"}) + self.assertEqual(status, 200) + self.assertAlmostEqual(body["cvoc"]["selected_lower_bound"], 0.604) + self.assertEqual(body["cvoc"]["basis"], "host_supplied_estimates") + self.assertFalse(body["effect_executed"]) + + def test_execute_and_untyped_responses_are_unavailable(self): + for path in ("/v1/c3r/execute", "/v1/responses"): + status, body = self.call(path, payload={"goal": "Find record"}) + self.assertEqual(status, 501) + self.assertEqual(body["error"], "not_implemented") + + def test_typed_system_one_answers_without_a_governed_action_catalog(self): + def rank(payload): + return {"model": "clm-latest", "ranked": [ + {"candidate": option, "prob": score} + for option, score in zip(payload["answers"], (0.8, 0.2)) + ]} + self.configure_ranker(rank) + status, body = self.call("/v1/system-one", payload={ + "model": "c3r-system-one", "state": "An invoice was charged twice", + "questions": {"department": {"type": "choice", "options": { + "billing": "Invoices and charges", "technical": "Product bugs"}}}, + }) + self.assertEqual(status, 200) + self.assertEqual(body["answers"]["department"]["choice"], "billing") + self.assertEqual(body["answers"]["department"]["scores"]["billing"], 0.8) + self.assertFalse(body["calibrated"]) + self.assertFalse(body["effect_executed"]) + + def test_system_one_boolean_ranking_and_authority_rejection(self): + def rank(payload): + return {"model": "clm-latest", "ranked": [ + {"candidate": option, "prob": score} + for option, score in zip(payload["answers"], (0.25, 0.75)) + ]} + self.configure_ranker(rank) + status, body = self.call("/v1/system-one", payload={ + "state": "A duplicate charge", "questions": {"urgent": {"type": "boolean"}}, + }) + self.assertEqual((status, body["answers"]["urgent"]["score"]), (200, 0.75)) + status, body = self.call("/v1/c3r/rank", payload={ + "model": "c3r-verifier", "state": "A duplicate charge", + "candidates": ["technical", "billing"], + }) + self.assertEqual((status, body["ranked"][0]["candidate"]), (200, "billing")) + for extra in ({"authority": "admin"}, {"endpoint": "http://169.254.169.254"}): + status, _ = self.call("/v1/system-one", payload={ + "state": "test", "candidates": ["A", "B"], **extra}) + self.assertEqual(status, 400) + + def test_system_one_rejects_total_option_overflow_and_provider_failure(self): + self.configure_ranker(lambda _: {}) + status, _ = self.call("/v1/system-one", payload={ + "state": "test", "candidates": [str(index) for index in range(65)]}) + self.assertEqual(status, 400) + status, body = self.call("/v1/system-one", payload={ + "state": "test", "candidates": ["A", "B"]}) + self.assertEqual((status, body["error"]), (503, "service_unavailable")) + + def test_responses_returns_text_without_private_reasoning_or_external_effects(self): + adapter = ProviderAdapter(ProviderConfig( + "local", ProviderKind.OPENAI_COMPATIBLE, "http://127.0.0.1:8000/v1", "model", None, + ), transport=lambda *_: TransportResponse(200, { + "choices": [{"message": {"content": "Check pending and settled charges.", + "reasoning_content": "PRIVATE"}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 12, "completion_tokens": 7}, + }, 25)) + runtime, _ = controller(deliberative=True) + factory = ReadOnlyRequestFactory(definitions=(ActionDefinition( + "DELIBERATE", ActionFamily.DELIBERATE, "compute", "generate", + RiskClass.READ_ONLY, ((),), ("local",), ("policy",), 0, 0, + ),), policy=AuthorityPolicy(frozenset({ActionFamily.DELIBERATE}), + frozenset({RiskClass.READ_ONLY})), + estimate_source=lambda _: {}, remaining_usd=0) + self.server.responses = ResponsesService(runtime, factory, adapter) + status, body = self.call("/v1/responses", payload={ + "model": "c3r-core", "input": "Find record", "store": False}) + self.assertEqual(status, 200) + self.assertEqual(body["object"], "response") + self.assertEqual(body["output"][0]["content"][0]["text"], + "Check pending and settled charges.") + self.assertNotIn("PRIVATE", json.dumps(body)) + self.assertFalse(body["c3r"]["effect_executed"]) + self.assertFalse(body["store"]) + + def test_positive_cvoc_generation_still_requires_independent_verification(self): + calls = [] + adapter = ProviderAdapter(ProviderConfig( + "local", ProviderKind.OPENAI_COMPATIBLE, "http://127.0.0.1:8000/v1", "model", None, + ), transport=lambda *_: calls.append(1)) + runtime, _ = controller(deliberative=True, accepted=False) + factory = ReadOnlyRequestFactory(definitions=(ActionDefinition( + "DELIBERATE", ActionFamily.DELIBERATE, "compute", "generate", + RiskClass.READ_ONLY, ((),), ("local",), ("policy",), 0, 0, + ),), policy=AuthorityPolicy(frozenset({ActionFamily.DELIBERATE}), + frozenset({RiskClass.READ_ONLY})), + estimate_source=lambda _: {"DELIBERATE:0:local:policy": ValueEstimate(1, 0, 0, 0)}, + remaining_usd=0) + self.server.responses = ResponsesService(runtime, factory, adapter) + status, body = self.call("/v1/responses", payload={ + "model": "c3r-core", "input": "Find record"}) + self.assertEqual((status, body["error"]), (503, "service_unavailable")) + self.assertEqual(calls, []) + + def test_hosted_text_generation_cannot_transfer_local_only_state(self): + calls = [] + def hosted_transport(*args): + calls.append(args) + return TransportResponse(200, {"choices": [{"message": {"content": "ready"}, + "finish_reason": "stop"}]}, 10) + provider = ProviderAdapter(ProviderConfig( + "hosted", ProviderKind.OPENAI_COMPATIBLE, "https://openrouter.ai/api/v1", + "deepseek/deepseek-v4.1-flash", "test-not-a-secret"), transport=hosted_transport) + runtime, _ = controller(deliberative=True) + factory = ReadOnlyRequestFactory(definitions=(ActionDefinition( + "DELIBERATE", ActionFamily.DELIBERATE, "compute", "generate", + RiskClass.READ_ONLY, ((),), ("local",), ("policy",), 0, 0, + ),), policy=AuthorityPolicy(frozenset({ActionFamily.DELIBERATE}), + frozenset({RiskClass.READ_ONLY})), + estimate_source=lambda _: {}, remaining_usd=0, data_boundary="local") + self.server.responses = ResponsesService(runtime, factory, provider) + status, body = self.call("/v1/responses", payload={ + "model": "c3r-core", "input": "Local-only synthetic state"}) + self.assertEqual((status, body["error"]), (503, "service_unavailable")) + self.assertEqual(calls, []) + + def test_approved_hosted_responses_enforces_zero_retention_routing(self): + observed = [] + def hosted_transport(url, headers, payload): + observed.append(payload) + return TransportResponse(200, {"choices": [{"message": {"content": "ready"}, + "finish_reason": "stop"}], + "usage": {"prompt_tokens": 5, "completion_tokens": 2, + "cost": 0.000001}}, 10) + provider = ProviderAdapter(ProviderConfig( + "openrouter-deepseek", ProviderKind.OPENAI_COMPATIBLE, "https://openrouter.ai/api/v1", + "deepseek/deepseek-v4.1-flash", "test-not-a-secret"), transport=hosted_transport) + runtime, _ = controller(deliberative=True) + factory = ReadOnlyRequestFactory(definitions=(ActionDefinition( + "DELIBERATE", ActionFamily.DELIBERATE, "compute", "generate", + RiskClass.READ_ONLY, ((),), ("hosted",), ("policy",), 0, 0, + data_boundary="approved_remote", + ),), policy=AuthorityPolicy(frozenset({ActionFamily.DELIBERATE}), + frozenset({RiskClass.READ_ONLY})), + estimate_source=lambda _: {}, remaining_usd=0, data_boundary="approved_remote") + self.server.responses = ResponsesService(runtime, factory, provider) + status, body = self.call("/v1/responses", payload={"model": "c3r-core", "input": "ready"}) + self.assertEqual(status, 200) + self.assertEqual(observed[0]["provider"], {"zdr": True, "data_collection": "deny", + "require_parameters": True, "allow_fallbacks": False}) + self.assertEqual(body["c3r"]["observed_provider_cost_usd"], 0.000001) + self.assertFalse(body["store"]) + + def test_new_paths_require_authentication(self): + for path in ("/v1/c3r/decide", "/v1/c3r/execute", "/v1/responses"): + status, body = self.call(path, token="wrong", payload={"goal": "Find record"}) + self.assertEqual((status, body["error"]), (401, "unauthorized")) + for path in ("/ready", "/v1/models"): + status, body = self.call(path, method="GET", token="wrong") + self.assertEqual((status, body["error"]), (401, "unauthorized")) + + def test_models_do_not_claim_calibration_or_generation(self): + status, body = self.call("/v1/models", method="GET") + self.assertEqual(status, 200) + self.assertEqual(body["models"][0]["id"], "c3r-core") + self.assertFalse(body["models"][0]["text_generation"]) + self.assertFalse(body["models"][0]["calibrated"]) + self.assertFalse(body["models"][0]["available"]) + status, body = self.call("/ready", method="GET") + self.assertEqual((status, body["status"]), (503, "disabled")) + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_trace_control_dry_run.py b/tests/test_trace_control_dry_run.py new file mode 100644 index 0000000..5a686d3 --- /dev/null +++ b/tests/test_trace_control_dry_run.py @@ -0,0 +1,17 @@ +import unittest + +from scripts.run_trace_control_dry_run import run + + +class TraceControlDryRunTests(unittest.TestCase): + def test_fixture_only_report_passes_without_claiming_deployed_controls(self): + report = run() + self.assertTrue(report["all_local_checks_passed"]) + self.assertFalse(report["live_trace_collection_enabled"]) + self.assertEqual(len(report["checks"]), 9) + self.assertIn("deployed encryption and IAM", report["not_verified_by_this_run"]) + + +if __name__ == "__main__": + unittest.main() + diff --git a/tests/test_trace_ledger.py b/tests/test_trace_ledger.py index 6684936..63046b2 100644 --- a/tests/test_trace_ledger.py +++ b/tests/test_trace_ledger.py @@ -1,5 +1,6 @@ import json import unittest +from concurrent.futures import ThreadPoolExecutor from c3r.telemetry.trace import DecisionTrace from c3r.telemetry.trace_ledger import TraceLedger @@ -49,6 +50,14 @@ def test_serialized_ledger_round_trips(self) -> None: self.assertTrue(TraceLedger.verify(restored.records)) self.assertEqual(restored.records[0].record_hash, ledger.records[0].record_hash) + def test_concurrent_appends_keep_one_valid_chain(self) -> None: + ledger = TraceLedger() + with ThreadPoolExecutor(max_workers=8) as pool: + list(pool.map(lambda index: ledger.append(trace(f"run-{index}")), range(100))) + + self.assertEqual(len(ledger.records), 100) + self.assertTrue(TraceLedger.verify(ledger.records)) + if __name__ == "__main__": unittest.main() diff --git a/third_party/CLM_VERSION b/third_party/CLM_VERSION new file mode 100644 index 0000000..c34f076 --- /dev/null +++ b/third_party/CLM_VERSION @@ -0,0 +1,8 @@ +Repository: https://github.com/Contrastive-LM/CLM +Source commit: bb42c6c5bf914fd449bed2f6ca65be80602cb1f7 +Source license: Apache-2.0 +Integration API: POST /v1/rank + +This pins reviewed upstream source only. The Qwen encoder, projection heads, +and deployed container require independent immutable hashes and verification. +No CLM weights or source tree are vendored in this repository.