Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 2 additions & 3 deletions examples/control_plane/cli-output-probe-runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -101,9 +101,8 @@ def _receipt_row(
if isinstance(payload, dict)
else []
),
"runtime_root_command_route_count": (
semantics.runtime_root_command_route_count(text)
),
**{f"{option}_command_route_count": count
for option, count in semantics.command_route_counts(text).items()},
"host_prompt_static_safety_revision": semantics.host_prompt_static_safety_revision(text),
"heartbeat_user_language_prompt_revision": (
semantics.heartbeat_user_language_prompt_revision(text)
Expand Down
8 changes: 5 additions & 3 deletions loopx/control_plane/testing/cli_output_budget.py
Original file line number Diff line number Diff line change
Expand Up @@ -169,12 +169,14 @@ class CliOutputCommandClassification:
# TurnEnvelope intentionally carries the complete authoring schema
# that the validator accepts, plus the typed executor and selection
# facts needed to decide whether execution is authorized. The
# latest-main fixture measures 14,159 chars, so 14,500 retains a
# narrow 341-char regression margin without relaxing Todo growth.
# Explicit registry routing adds 75 necessary command characters:
# the same fixture measured 14,482 before routing and 14,557 after.
# Keep the executable authority binding intact; 14,600 leaves a
# 43-character margin without relaxing line or per-Todo growth.
# The over-target TurnEnvelope diagnostic remains visible instead
# of hiding authority overflow; latest main renders it in 542
# characters, leaving a narrow 58-character presentation margin.
"crowded": {"json": 14_500, "markdown": 600},
"crowded": {"json": 14_600, "markdown": 600},
"multi_agent": {"json": 12_000, "markdown": 300},
},
max_lines={
Expand Down
44 changes: 23 additions & 21 deletions loopx/control_plane/testing/cli_output_differential.py
Original file line number Diff line number Diff line change
Expand Up @@ -163,11 +163,11 @@ class GrowthAllowance:
"compact_payload_chars": 192,
}

# Explicit runtime-root command routing repeats one bounded command prefix per
# Explicit registry/runtime-root routing repeats one bounded command argument per
# executable action. The allowance covers the prefix and its JSON projection;
# it is per newly observed route, not per row, so unrelated output growth still
# fails under the normal hot-path policy.
_RUNTIME_ROOT_COMMAND_ROUTE_GROWTH_PER_ROUTE: dict[Metric, int] = {
_COMMAND_ROUTE_GROWTH_PER_ROUTE: dict[Metric, int] = {
"chars": 160,
"utf8_bytes": 160,
"lines": 0,
Expand Down Expand Up @@ -309,19 +309,21 @@ def _heartbeat_user_language_migration_allowance(
}


def _runtime_root_route_growth_allowances(
def _command_route_growth_allowances(
base: dict[str, Any], candidate: dict[str, Any]
) -> tuple[int, dict[Metric, int]]:
base_routes = base.get("runtime_root_command_route_count")
candidate_routes = candidate.get("runtime_root_command_route_count")
if type(base_routes) is not int or type(candidate_routes) is not int:
return 0, {}
added_routes = max(0, candidate_routes - base_routes)
if not added_routes:
return 0, {}
return added_routes, {
metric: added_routes * allowance
for metric, allowance in _RUNTIME_ROOT_COMMAND_ROUTE_GROWTH_PER_ROUTE.items()
) -> tuple[dict[str, int], dict[Metric, int]]:
additions: dict[str, int] = {}
for option in ("runtime_root", "registry"):
field = f"{option}_command_route_count"
before, after = base.get(field), candidate.get(field)
# Missing, malformed or negative observations never grant an allowance.
if type(before) is int and type(after) is int and 0 <= before < after:
additions[option] = after - before
if not additions:
return {}, {}
return additions, {
metric: sum(additions.values()) * allowance
for metric, allowance in _COMMAND_ROUTE_GROWTH_PER_ROUTE.items()
}


Expand Down Expand Up @@ -679,8 +681,8 @@ def _compare_row(base: dict[str, Any], candidate: dict[str, Any]) -> dict[str, A
candidate,
output_format=output_format,
)
added_runtime_root_routes, runtime_root_route_allowances = (
_runtime_root_route_growth_allowances(base, candidate)
added_command_routes, command_route_allowances = (
_command_route_growth_allowances(base, candidate)
)

projection_allowance, projection_failures, projection_signals = _projection_envelope_migration(
Expand Down Expand Up @@ -759,10 +761,10 @@ def _compare_row(base: dict[str, Any], candidate: dict[str, Any]) -> dict[str, A
"compact_payload_chars": 512,
}[metric],
)
if runtime_root_route_allowances:
if command_route_allowances:
allowance = max(
allowance,
runtime_root_route_allowances[metric],
command_route_allowances[metric],
)
deltas[metric] = delta
allowances[metric] = allowance
Expand Down Expand Up @@ -826,10 +828,10 @@ def _compare_row(base: dict[str, Any], candidate: dict[str, Any]) -> dict[str, A
"planning inventory detail schema migrated: "
f"{migration.inventory_detail_schema_migration}"
)
if runtime_root_route_allowances:
for option, count in added_command_routes.items():
review_signals.append(
"runtime-root command route coverage added: "
f"{added_runtime_root_routes} executable route(s)"
f"{option.replace('_', '-')} command route coverage added: "
f"{count} executable route(s)"
)
if migration.guided_todo_delta_schema_changed:
if migration.guided_todo_delta_schema_migration is None:
Expand Down
67 changes: 61 additions & 6 deletions loopx/control_plane/testing/cli_output_semantics.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
import hashlib
import json
import re
import shlex
from typing import Any


Expand Down Expand Up @@ -79,10 +80,7 @@ def managed_executor_binding_revision(text: str) -> str | None:
)

_MARKDOWN_HEADING = re.compile(r"^#{1,6}\s+.+$")
_RUNTIME_ROOT_COMMAND_ROUTE = re.compile(
r"(?m)(?:^|[\"'`])[^\r\n\S]*loopx\s+--runtime-root\s+"
r"(?:\"[^\"\r\n]+\"|'[^'\r\n]+'|\S+)"
)



def json_shape_paths(value: Any, *, path: str = "$") -> list[str]:
Expand Down Expand Up @@ -202,8 +200,65 @@ def markdown_headings(text: str) -> list[str]:
return [line.strip() for line in text.splitlines() if _MARKDOWN_HEADING.match(line)]


def runtime_root_command_route_count(text: str) -> int:
return len(_RUNTIME_ROOT_COMMAND_ROUTE.findall(text))
def command_route_counts(text: str) -> dict[str, int]:
"""Measure well-formed rendered routes, never grant runtime authority.

Decode JSON strings before shell parsing, or read standalone/Markdown code
commands. Prose mentioning an option and malformed argv earn no allowance.
Both bindings are measured in one pass; duplicates within a command count once.
"""
counts = {"runtime_root": 0, "registry": 0}
pending: list[Any] = [text]
commands: list[str] = []
while pending:
value = pending.pop()
if isinstance(value, dict):
pending.extend(value.values())
elif isinstance(value, list):
pending.extend(value)
elif isinstance(value, str):
try:
decoded = json.loads(value)
except ValueError:
for line in value.splitlines():
stripped = line.strip()
if stripped.startswith("loopx "):
commands.append(stripped)
continue
try:
pending.append(json.loads(line))
except ValueError:
commands.extend(re.findall(r"`(loopx [^`\r\n]+)`", line))
else:
if isinstance(decoded, (dict, list, str)):
pending.append(decoded)

for command in commands:
try:
argv = shlex.split(command)
except ValueError:
continue
bindings: set[str] = set()
index = 1
while index < len(argv) and argv[index].startswith("-"):
option = argv[index]
if (option not in {"--registry", "--runtime-root", "--format"}
or index + 1 >= len(argv)
or not argv[index + 1] or argv[index + 1].startswith("-")):
break
if option == "--format":
if argv[index + 1] not in {"json", "markdown"}:
break
else:
bindings.add(option[2:].replace("-", "_"))
index += 2
# Reject an incomplete/invalid option prefix, or one with no subcommand.
if (index == len(argv) or not argv[index].strip()
or argv[index].startswith("-")):
continue
for binding in bindings:
counts[binding] += 1
return counts


def projection_envelope_schema_versions(value: Any) -> list[str]:
Expand Down
10 changes: 9 additions & 1 deletion tests/control_plane/test_cli_output_budget.py
Original file line number Diff line number Diff line change
Expand Up @@ -506,6 +506,14 @@ def _measure_scenario(root: Path, scenario: Scenario) -> dict[str, dict[str, dic
text,
output_format=output_format,
)
if (surface_id, scenario.name, output_format) == (
"loopx_turn_plan", "crowded", "json"
):
# Budget compaction must not discard the writeback target.
action = measurement["payload"]["turn_envelope"]["writeback"]["next_cli_actions"][0]
argv = shlex.split(action)
assert argv[argv.index("--registry") + 1] == str(registry_path)
assert argv[argv.index("--runtime-root") + 1] == str(runtime)
spec = CLI_OUTPUT_BUDGET_BY_ID[surface_id]
assert_cli_output_baseline(
spec,
Expand Down Expand Up @@ -1123,7 +1131,7 @@ def test_crowded_turn_plan_budget_preserves_executable_vision_authoring(
)
# This fixed executable schema legitimately crosses the old 12k/320
# ceiling; retain bounded headroom without relaxing Todo-scale growth.
assert 12_000 < len(text) <= 14_500
assert 12_000 < len(text) <= CLI_OUTPUT_BUDGET_BY_ID["loopx_turn_plan"].max_chars["crowded"]["json"]
assert len(text.splitlines()) <= 400


Expand Down
66 changes: 54 additions & 12 deletions tests/control_plane/test_cli_output_differential.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,7 @@
guided_todo_delta_schema_versions,
planning_horizon_schema_versions,
planning_inventory_detail_schema_versions,
runtime_root_command_route_count,
command_route_counts,
todo_work_counts_schema_versions,
)

Expand Down Expand Up @@ -47,6 +47,7 @@ def _row(**overrides: object) -> dict[str, object]:
"planning_inventory_detail_schema_versions": [],
"todo_work_counts_schema_versions": [],
"runtime_root_command_route_count": 0,
"registry_command_route_count": 0,
}
row.update(overrides)
return row
Expand Down Expand Up @@ -803,7 +804,8 @@ def test_planning_inventory_detail_migration_is_bounded_and_fail_closed() -> Non
]


def test_runtime_root_route_growth_has_per_route_budget() -> None:
@pytest.mark.parametrize("option", ["runtime_root", "registry"])
def test_command_route_growth_has_per_route_budget(option: str) -> None:
base = _row(
chars=1_000,
utf8_bytes=1_000,
Expand All @@ -819,7 +821,7 @@ def test_runtime_root_route_growth_has_per_route_budget() -> None:
compact_payload_chars=1_320,
action_signature_sha256=None,
action_signature_coverages=[],
runtime_root_command_route_count=2,
**{f"{option}_command_route_count": 2},
)

result = compare_cli_output_receipts(_receipt(base), _receipt(candidate))
Expand All @@ -833,11 +835,12 @@ def test_runtime_root_route_growth_has_per_route_budget() -> None:
"compact_payload_chars": 320,
}
assert result["rows"][0]["review_signals"] == [
"runtime-root command route coverage added: 2 executable route(s)"
f"{option.replace('_', '-')} command route coverage added: 2 executable route(s)"
]


def test_runtime_root_route_growth_still_fails_above_per_route_budget() -> None:
@pytest.mark.parametrize("option", ["runtime_root", "registry"])
def test_command_route_growth_still_fails_above_per_route_budget(option: str) -> None:
base = _row(
chars=1_000,
utf8_bytes=1_000,
Expand All @@ -853,7 +856,7 @@ def test_runtime_root_route_growth_still_fails_above_per_route_budget() -> None:
compact_payload_chars=1_321,
action_signature_sha256=None,
action_signature_coverages=[],
runtime_root_command_route_count=2,
**{f"{option}_command_route_count": 2},
)

result = compare_cli_output_receipts(_receipt(base), _receipt(candidate))
Expand All @@ -862,19 +865,22 @@ def test_runtime_root_route_growth_still_fails_above_per_route_budget() -> None:
assert "chars grew by 321; allowance is 320" in result["rows"][0]["failures"]


def test_invalid_runtime_root_route_count_does_not_grant_budget() -> None:
@pytest.mark.parametrize("option", ["runtime_root", "registry"])
@pytest.mark.parametrize("before, after", [(0, True), (0, "2"), (None, 2), (-1, 2), (0, -1), (2, 2)])
def test_invalid_or_unchanged_command_route_count_does_not_grant_budget(option: str, before: object, after: object) -> None:
base = _row(
chars=1_000,
utf8_bytes=1_000,
lines=10,
compact_payload_chars=1_000,
**{f"{option}_command_route_count": before},
)
candidate = _row(
chars=1_097,
utf8_bytes=1_097,
lines=10,
compact_payload_chars=1_097,
runtime_root_command_route_count=True,
**{f"{option}_command_route_count": after},
)

result = compare_cli_output_receipts(_receipt(base), _receipt(candidate))
Expand All @@ -899,23 +905,24 @@ def test_runtime_root_route_count_only_matches_executable_command_prefixes() ->
"loopx --runtime-root"
)

assert runtime_root_command_route_count(text) == 4
assert command_route_counts(text)["runtime_root"] == 4


def test_runtime_root_route_allowance_is_fail_closed_for_invalid_counts() -> None:
@pytest.mark.parametrize("option", ["runtime_root", "registry"])
def test_command_route_allowance_is_fail_closed_for_invalid_counts(option: str) -> None:
base = _row(
chars=1_000,
utf8_bytes=1_000,
lines=10,
compact_payload_chars=1_000,
runtime_root_command_route_count=0,
**{f"{option}_command_route_count": 0},
)
candidate = _row(
chars=1_000,
utf8_bytes=1_000,
lines=10,
compact_payload_chars=1_000,
runtime_root_command_route_count="2",
**{f"{option}_command_route_count": "2"},
)

result = compare_cli_output_receipts(_receipt(base), _receipt(candidate))
Expand Down Expand Up @@ -1109,3 +1116,38 @@ def test_projection_envelope_migration_is_status_only_bounded_and_one_time(outpu
outside_base = {**base, "row_id": f"surface/{surface}/small/{output_format}"}
outside = {**candidate, "row_id": outside_base["row_id"]}
assert not compare_cli_output_receipts(_receipt(outside_base), _receipt(outside))["ok"]


def test_both_command_routes_are_counted_in_either_order() -> None:
text = (
"loopx --registry '/tmp/registry path' --runtime-root /tmp/root refresh-state\n"
"loopx --runtime-root /tmp/root --registry /tmp/registry quota should-run\n"
"Use --registry PATH, or say loopx --registry /tmp/path.\n"
"loopx --registry"
)
assert command_route_counts(text) == {"registry": 2, "runtime_root": 2}


@pytest.mark.parametrize("command", [
'loopx --registry "" turn plan',
"loopx --registry --runtime-root /tmp/root turn plan",
'loopx --registry "/tmp/unclosed turn plan',
"loopx --registry /tmp/registry",
"loopx --registry /tmp/registry --runtime-root",
"loopx --registry /tmp/registry --format invalid turn plan",
'loopx --registry /tmp/registry ""',
'loopx --registry /tmp/registry " "',
])
@pytest.mark.parametrize("render", [str, lambda command: json.dumps({"command": command}), lambda command: f"- execute: `{command}`"])
def test_malformed_command_never_grants_route_growth(command, render) -> None:
counts = command_route_counts(render(command))
assert counts == {"registry": 0, "runtime_root": 0}
base = _row(chars=1_000, utf8_bytes=1_000, compact_payload_chars=1_000)
candidate = _row(chars=1_160, utf8_bytes=1_160, compact_payload_chars=1_160,
**{f"{key}_command_route_count": value for key, value in counts.items()})
assert compare_cli_output_receipts(_receipt(base), _receipt(candidate))["ok"] is False


def test_json_escaped_paths_and_duplicate_arguments_are_counted_once() -> None:
command = """loopx --format json --registry '/tmp/a \"quoted\" path' --registry /tmp/final --runtime-root '/tmp/root path' turn plan"""
assert command_route_counts(json.dumps({"command": command})) == {"registry": 1, "runtime_root": 1}
4 changes: 3 additions & 1 deletion tests/control_plane/test_cli_output_probe_runner.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
from __future__ import annotations

import runpy
import re
from pathlib import Path

import pytest
Expand Down Expand Up @@ -86,5 +87,6 @@ def oversized_stdout(command):
return rc, text + " " * 15_000

monkeypatch.setattr(probe, "_invoke_cli", oversized_stdout)
with pytest.raises(AssertionError, match="baseline ceiling is 14500"):
ceiling = probe.CLI_OUTPUT_BUDGET_BY_ID["loopx_turn_plan"].max_chars["crowded"]["json"]
with pytest.raises(AssertionError, match=re.escape(f"baseline ceiling is {ceiling}")):
crowded_turn_probe(probe, cli_output_semantics, tmp_path / "growth")
Loading