diff --git a/examples/control_plane/cli-output-probe-runner.py b/examples/control_plane/cli-output-probe-runner.py index ead3818ec4..c1b078d9ec 100644 --- a/examples/control_plane/cli-output-probe-runner.py +++ b/examples/control_plane/cli-output-probe-runner.py @@ -101,9 +101,8 @@ def _receipt_row( if isinstance(payload, dict) else [] ), - "runtime_root_command_route_count": ( - semantics.runtime_root_command_route_count(text) - ), + **{f"{option}_command_route_count": count + for option, count in semantics.command_route_counts(text).items()}, "host_prompt_static_safety_revision": semantics.host_prompt_static_safety_revision(text), "heartbeat_user_language_prompt_revision": ( semantics.heartbeat_user_language_prompt_revision(text) diff --git a/loopx/control_plane/testing/cli_output_budget.py b/loopx/control_plane/testing/cli_output_budget.py index 7ae6103640..473d7bc5c2 100644 --- a/loopx/control_plane/testing/cli_output_budget.py +++ b/loopx/control_plane/testing/cli_output_budget.py @@ -169,12 +169,14 @@ class CliOutputCommandClassification: # TurnEnvelope intentionally carries the complete authoring schema # that the validator accepts, plus the typed executor and selection # facts needed to decide whether execution is authorized. The - # latest-main fixture measures 14,159 chars, so 14,500 retains a - # narrow 341-char regression margin without relaxing Todo growth. + # Explicit registry routing adds 75 necessary command characters: + # the same fixture measured 14,482 before routing and 14,557 after. + # Keep the executable authority binding intact; 14,600 leaves a + # 43-character margin without relaxing line or per-Todo growth. # The over-target TurnEnvelope diagnostic remains visible instead # of hiding authority overflow; latest main renders it in 542 # characters, leaving a narrow 58-character presentation margin. - "crowded": {"json": 14_500, "markdown": 600}, + "crowded": {"json": 14_600, "markdown": 600}, "multi_agent": {"json": 12_000, "markdown": 300}, }, max_lines={ diff --git a/loopx/control_plane/testing/cli_output_differential.py b/loopx/control_plane/testing/cli_output_differential.py index b883abcb80..183a56cd93 100644 --- a/loopx/control_plane/testing/cli_output_differential.py +++ b/loopx/control_plane/testing/cli_output_differential.py @@ -163,11 +163,11 @@ class GrowthAllowance: "compact_payload_chars": 192, } -# Explicit runtime-root command routing repeats one bounded command prefix per +# Explicit registry/runtime-root routing repeats one bounded command argument per # executable action. The allowance covers the prefix and its JSON projection; # it is per newly observed route, not per row, so unrelated output growth still # fails under the normal hot-path policy. -_RUNTIME_ROOT_COMMAND_ROUTE_GROWTH_PER_ROUTE: dict[Metric, int] = { +_COMMAND_ROUTE_GROWTH_PER_ROUTE: dict[Metric, int] = { "chars": 160, "utf8_bytes": 160, "lines": 0, @@ -309,19 +309,21 @@ def _heartbeat_user_language_migration_allowance( } -def _runtime_root_route_growth_allowances( +def _command_route_growth_allowances( base: dict[str, Any], candidate: dict[str, Any] -) -> tuple[int, dict[Metric, int]]: - base_routes = base.get("runtime_root_command_route_count") - candidate_routes = candidate.get("runtime_root_command_route_count") - if type(base_routes) is not int or type(candidate_routes) is not int: - return 0, {} - added_routes = max(0, candidate_routes - base_routes) - if not added_routes: - return 0, {} - return added_routes, { - metric: added_routes * allowance - for metric, allowance in _RUNTIME_ROOT_COMMAND_ROUTE_GROWTH_PER_ROUTE.items() +) -> tuple[dict[str, int], dict[Metric, int]]: + additions: dict[str, int] = {} + for option in ("runtime_root", "registry"): + field = f"{option}_command_route_count" + before, after = base.get(field), candidate.get(field) + # Missing, malformed or negative observations never grant an allowance. + if type(before) is int and type(after) is int and 0 <= before < after: + additions[option] = after - before + if not additions: + return {}, {} + return additions, { + metric: sum(additions.values()) * allowance + for metric, allowance in _COMMAND_ROUTE_GROWTH_PER_ROUTE.items() } @@ -679,8 +681,8 @@ def _compare_row(base: dict[str, Any], candidate: dict[str, Any]) -> dict[str, A candidate, output_format=output_format, ) - added_runtime_root_routes, runtime_root_route_allowances = ( - _runtime_root_route_growth_allowances(base, candidate) + added_command_routes, command_route_allowances = ( + _command_route_growth_allowances(base, candidate) ) projection_allowance, projection_failures, projection_signals = _projection_envelope_migration( @@ -759,10 +761,10 @@ def _compare_row(base: dict[str, Any], candidate: dict[str, Any]) -> dict[str, A "compact_payload_chars": 512, }[metric], ) - if runtime_root_route_allowances: + if command_route_allowances: allowance = max( allowance, - runtime_root_route_allowances[metric], + command_route_allowances[metric], ) deltas[metric] = delta allowances[metric] = allowance @@ -826,10 +828,10 @@ def _compare_row(base: dict[str, Any], candidate: dict[str, Any]) -> dict[str, A "planning inventory detail schema migrated: " f"{migration.inventory_detail_schema_migration}" ) - if runtime_root_route_allowances: + for option, count in added_command_routes.items(): review_signals.append( - "runtime-root command route coverage added: " - f"{added_runtime_root_routes} executable route(s)" + f"{option.replace('_', '-')} command route coverage added: " + f"{count} executable route(s)" ) if migration.guided_todo_delta_schema_changed: if migration.guided_todo_delta_schema_migration is None: diff --git a/loopx/control_plane/testing/cli_output_semantics.py b/loopx/control_plane/testing/cli_output_semantics.py index 7b04a8e2c1..0b7abc47fd 100644 --- a/loopx/control_plane/testing/cli_output_semantics.py +++ b/loopx/control_plane/testing/cli_output_semantics.py @@ -3,6 +3,7 @@ import hashlib import json import re +import shlex from typing import Any @@ -79,10 +80,7 @@ def managed_executor_binding_revision(text: str) -> str | None: ) _MARKDOWN_HEADING = re.compile(r"^#{1,6}\s+.+$") -_RUNTIME_ROOT_COMMAND_ROUTE = re.compile( - r"(?m)(?:^|[\"'`])[^\r\n\S]*loopx\s+--runtime-root\s+" - r"(?:\"[^\"\r\n]+\"|'[^'\r\n]+'|\S+)" -) + def json_shape_paths(value: Any, *, path: str = "$") -> list[str]: @@ -202,8 +200,65 @@ def markdown_headings(text: str) -> list[str]: return [line.strip() for line in text.splitlines() if _MARKDOWN_HEADING.match(line)] -def runtime_root_command_route_count(text: str) -> int: - return len(_RUNTIME_ROOT_COMMAND_ROUTE.findall(text)) +def command_route_counts(text: str) -> dict[str, int]: + """Measure well-formed rendered routes, never grant runtime authority. + + Decode JSON strings before shell parsing, or read standalone/Markdown code + commands. Prose mentioning an option and malformed argv earn no allowance. + Both bindings are measured in one pass; duplicates within a command count once. + """ + counts = {"runtime_root": 0, "registry": 0} + pending: list[Any] = [text] + commands: list[str] = [] + while pending: + value = pending.pop() + if isinstance(value, dict): + pending.extend(value.values()) + elif isinstance(value, list): + pending.extend(value) + elif isinstance(value, str): + try: + decoded = json.loads(value) + except ValueError: + for line in value.splitlines(): + stripped = line.strip() + if stripped.startswith("loopx "): + commands.append(stripped) + continue + try: + pending.append(json.loads(line)) + except ValueError: + commands.extend(re.findall(r"`(loopx [^`\r\n]+)`", line)) + else: + if isinstance(decoded, (dict, list, str)): + pending.append(decoded) + + for command in commands: + try: + argv = shlex.split(command) + except ValueError: + continue + bindings: set[str] = set() + index = 1 + while index < len(argv) and argv[index].startswith("-"): + option = argv[index] + if (option not in {"--registry", "--runtime-root", "--format"} + or index + 1 >= len(argv) + or not argv[index + 1] or argv[index + 1].startswith("-")): + break + if option == "--format": + if argv[index + 1] not in {"json", "markdown"}: + break + else: + bindings.add(option[2:].replace("-", "_")) + index += 2 + # Reject an incomplete/invalid option prefix, or one with no subcommand. + if (index == len(argv) or not argv[index].strip() + or argv[index].startswith("-")): + continue + for binding in bindings: + counts[binding] += 1 + return counts def projection_envelope_schema_versions(value: Any) -> list[str]: diff --git a/tests/control_plane/test_cli_output_budget.py b/tests/control_plane/test_cli_output_budget.py index 962d59ab4a..acf9fa01fe 100644 --- a/tests/control_plane/test_cli_output_budget.py +++ b/tests/control_plane/test_cli_output_budget.py @@ -506,6 +506,14 @@ def _measure_scenario(root: Path, scenario: Scenario) -> dict[str, dict[str, dic text, output_format=output_format, ) + if (surface_id, scenario.name, output_format) == ( + "loopx_turn_plan", "crowded", "json" + ): + # Budget compaction must not discard the writeback target. + action = measurement["payload"]["turn_envelope"]["writeback"]["next_cli_actions"][0] + argv = shlex.split(action) + assert argv[argv.index("--registry") + 1] == str(registry_path) + assert argv[argv.index("--runtime-root") + 1] == str(runtime) spec = CLI_OUTPUT_BUDGET_BY_ID[surface_id] assert_cli_output_baseline( spec, @@ -1123,7 +1131,7 @@ def test_crowded_turn_plan_budget_preserves_executable_vision_authoring( ) # This fixed executable schema legitimately crosses the old 12k/320 # ceiling; retain bounded headroom without relaxing Todo-scale growth. - assert 12_000 < len(text) <= 14_500 + assert 12_000 < len(text) <= CLI_OUTPUT_BUDGET_BY_ID["loopx_turn_plan"].max_chars["crowded"]["json"] assert len(text.splitlines()) <= 400 diff --git a/tests/control_plane/test_cli_output_differential.py b/tests/control_plane/test_cli_output_differential.py index eb65804fe7..1174830433 100644 --- a/tests/control_plane/test_cli_output_differential.py +++ b/tests/control_plane/test_cli_output_differential.py @@ -18,7 +18,7 @@ guided_todo_delta_schema_versions, planning_horizon_schema_versions, planning_inventory_detail_schema_versions, - runtime_root_command_route_count, + command_route_counts, todo_work_counts_schema_versions, ) @@ -47,6 +47,7 @@ def _row(**overrides: object) -> dict[str, object]: "planning_inventory_detail_schema_versions": [], "todo_work_counts_schema_versions": [], "runtime_root_command_route_count": 0, + "registry_command_route_count": 0, } row.update(overrides) return row @@ -803,7 +804,8 @@ def test_planning_inventory_detail_migration_is_bounded_and_fail_closed() -> Non ] -def test_runtime_root_route_growth_has_per_route_budget() -> None: +@pytest.mark.parametrize("option", ["runtime_root", "registry"]) +def test_command_route_growth_has_per_route_budget(option: str) -> None: base = _row( chars=1_000, utf8_bytes=1_000, @@ -819,7 +821,7 @@ def test_runtime_root_route_growth_has_per_route_budget() -> None: compact_payload_chars=1_320, action_signature_sha256=None, action_signature_coverages=[], - runtime_root_command_route_count=2, + **{f"{option}_command_route_count": 2}, ) result = compare_cli_output_receipts(_receipt(base), _receipt(candidate)) @@ -833,11 +835,12 @@ def test_runtime_root_route_growth_has_per_route_budget() -> None: "compact_payload_chars": 320, } assert result["rows"][0]["review_signals"] == [ - "runtime-root command route coverage added: 2 executable route(s)" + f"{option.replace('_', '-')} command route coverage added: 2 executable route(s)" ] -def test_runtime_root_route_growth_still_fails_above_per_route_budget() -> None: +@pytest.mark.parametrize("option", ["runtime_root", "registry"]) +def test_command_route_growth_still_fails_above_per_route_budget(option: str) -> None: base = _row( chars=1_000, utf8_bytes=1_000, @@ -853,7 +856,7 @@ def test_runtime_root_route_growth_still_fails_above_per_route_budget() -> None: compact_payload_chars=1_321, action_signature_sha256=None, action_signature_coverages=[], - runtime_root_command_route_count=2, + **{f"{option}_command_route_count": 2}, ) result = compare_cli_output_receipts(_receipt(base), _receipt(candidate)) @@ -862,19 +865,22 @@ def test_runtime_root_route_growth_still_fails_above_per_route_budget() -> None: assert "chars grew by 321; allowance is 320" in result["rows"][0]["failures"] -def test_invalid_runtime_root_route_count_does_not_grant_budget() -> None: +@pytest.mark.parametrize("option", ["runtime_root", "registry"]) +@pytest.mark.parametrize("before, after", [(0, True), (0, "2"), (None, 2), (-1, 2), (0, -1), (2, 2)]) +def test_invalid_or_unchanged_command_route_count_does_not_grant_budget(option: str, before: object, after: object) -> None: base = _row( chars=1_000, utf8_bytes=1_000, lines=10, compact_payload_chars=1_000, + **{f"{option}_command_route_count": before}, ) candidate = _row( chars=1_097, utf8_bytes=1_097, lines=10, compact_payload_chars=1_097, - runtime_root_command_route_count=True, + **{f"{option}_command_route_count": after}, ) result = compare_cli_output_receipts(_receipt(base), _receipt(candidate)) @@ -899,23 +905,24 @@ def test_runtime_root_route_count_only_matches_executable_command_prefixes() -> "loopx --runtime-root" ) - assert runtime_root_command_route_count(text) == 4 + assert command_route_counts(text)["runtime_root"] == 4 -def test_runtime_root_route_allowance_is_fail_closed_for_invalid_counts() -> None: +@pytest.mark.parametrize("option", ["runtime_root", "registry"]) +def test_command_route_allowance_is_fail_closed_for_invalid_counts(option: str) -> None: base = _row( chars=1_000, utf8_bytes=1_000, lines=10, compact_payload_chars=1_000, - runtime_root_command_route_count=0, + **{f"{option}_command_route_count": 0}, ) candidate = _row( chars=1_000, utf8_bytes=1_000, lines=10, compact_payload_chars=1_000, - runtime_root_command_route_count="2", + **{f"{option}_command_route_count": "2"}, ) result = compare_cli_output_receipts(_receipt(base), _receipt(candidate)) @@ -1109,3 +1116,38 @@ def test_projection_envelope_migration_is_status_only_bounded_and_one_time(outpu outside_base = {**base, "row_id": f"surface/{surface}/small/{output_format}"} outside = {**candidate, "row_id": outside_base["row_id"]} assert not compare_cli_output_receipts(_receipt(outside_base), _receipt(outside))["ok"] + + +def test_both_command_routes_are_counted_in_either_order() -> None: + text = ( + "loopx --registry '/tmp/registry path' --runtime-root /tmp/root refresh-state\n" + "loopx --runtime-root /tmp/root --registry /tmp/registry quota should-run\n" + "Use --registry PATH, or say loopx --registry /tmp/path.\n" + "loopx --registry" + ) + assert command_route_counts(text) == {"registry": 2, "runtime_root": 2} + + +@pytest.mark.parametrize("command", [ + 'loopx --registry "" turn plan', + "loopx --registry --runtime-root /tmp/root turn plan", + 'loopx --registry "/tmp/unclosed turn plan', + "loopx --registry /tmp/registry", + "loopx --registry /tmp/registry --runtime-root", + "loopx --registry /tmp/registry --format invalid turn plan", + 'loopx --registry /tmp/registry ""', + 'loopx --registry /tmp/registry " "', +]) +@pytest.mark.parametrize("render", [str, lambda command: json.dumps({"command": command}), lambda command: f"- execute: `{command}`"]) +def test_malformed_command_never_grants_route_growth(command, render) -> None: + counts = command_route_counts(render(command)) + assert counts == {"registry": 0, "runtime_root": 0} + base = _row(chars=1_000, utf8_bytes=1_000, compact_payload_chars=1_000) + candidate = _row(chars=1_160, utf8_bytes=1_160, compact_payload_chars=1_160, + **{f"{key}_command_route_count": value for key, value in counts.items()}) + assert compare_cli_output_receipts(_receipt(base), _receipt(candidate))["ok"] is False + + +def test_json_escaped_paths_and_duplicate_arguments_are_counted_once() -> None: + command = """loopx --format json --registry '/tmp/a \"quoted\" path' --registry /tmp/final --runtime-root '/tmp/root path' turn plan""" + assert command_route_counts(json.dumps({"command": command})) == {"registry": 1, "runtime_root": 1} diff --git a/tests/control_plane/test_cli_output_probe_runner.py b/tests/control_plane/test_cli_output_probe_runner.py index a3290cc20c..ab5da964b0 100644 --- a/tests/control_plane/test_cli_output_probe_runner.py +++ b/tests/control_plane/test_cli_output_probe_runner.py @@ -1,6 +1,7 @@ from __future__ import annotations import runpy +import re from pathlib import Path import pytest @@ -86,5 +87,6 @@ def oversized_stdout(command): return rc, text + " " * 15_000 monkeypatch.setattr(probe, "_invoke_cli", oversized_stdout) - with pytest.raises(AssertionError, match="baseline ceiling is 14500"): + ceiling = probe.CLI_OUTPUT_BUDGET_BY_ID["loopx_turn_plan"].max_chars["crowded"]["json"] + with pytest.raises(AssertionError, match=re.escape(f"baseline ceiling is {ceiling}")): crowded_turn_probe(probe, cli_output_semantics, tmp_path / "growth")