Skip to content

Commit 5fa910d

Browse files
committed
Merge origin/main into codex/fix-usage-notice-version-test
Signed-off-by: duanjialing.777 <duanjialing.777@bytedance.com>
2 parents 85b87e9 + f49b4a0 commit 5fa910d

7 files changed

Lines changed: 157 additions & 47 deletions

File tree

‎examples/control_plane/cli-output-probe-runner.py‎

Lines changed: 2 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -101,9 +101,8 @@ def _receipt_row(
101101
if isinstance(payload, dict)
102102
else []
103103
),
104-
"runtime_root_command_route_count": (
105-
semantics.runtime_root_command_route_count(text)
106-
),
104+
**{f"{option}_command_route_count": count
105+
for option, count in semantics.command_route_counts(text).items()},
107106
"host_prompt_static_safety_revision": semantics.host_prompt_static_safety_revision(text),
108107
"heartbeat_user_language_prompt_revision": (
109108
semantics.heartbeat_user_language_prompt_revision(text)

‎loopx/control_plane/testing/cli_output_budget.py‎

Lines changed: 5 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -169,12 +169,14 @@ class CliOutputCommandClassification:
169169
# TurnEnvelope intentionally carries the complete authoring schema
170170
# that the validator accepts, plus the typed executor and selection
171171
# facts needed to decide whether execution is authorized. The
172-
# latest-main fixture measures 14,159 chars, so 14,500 retains a
173-
# narrow 341-char regression margin without relaxing Todo growth.
172+
# Explicit registry routing adds 75 necessary command characters:
173+
# the same fixture measured 14,482 before routing and 14,557 after.
174+
# Keep the executable authority binding intact; 14,600 leaves a
175+
# 43-character margin without relaxing line or per-Todo growth.
174176
# The over-target TurnEnvelope diagnostic remains visible instead
175177
# of hiding authority overflow; latest main renders it in 542
176178
# characters, leaving a narrow 58-character presentation margin.
177-
"crowded": {"json": 14_500, "markdown": 600},
179+
"crowded": {"json": 14_600, "markdown": 600},
178180
"multi_agent": {"json": 12_000, "markdown": 300},
179181
},
180182
max_lines={

‎loopx/control_plane/testing/cli_output_differential.py‎

Lines changed: 23 additions & 21 deletions
Original file line numberDiff line numberDiff line change
@@ -163,11 +163,11 @@ class GrowthAllowance:
163163
"compact_payload_chars": 192,
164164
}
165165

166-
# Explicit runtime-root command routing repeats one bounded command prefix per
166+
# Explicit registry/runtime-root routing repeats one bounded command argument per
167167
# executable action. The allowance covers the prefix and its JSON projection;
168168
# it is per newly observed route, not per row, so unrelated output growth still
169169
# fails under the normal hot-path policy.
170-
_RUNTIME_ROOT_COMMAND_ROUTE_GROWTH_PER_ROUTE: dict[Metric, int] = {
170+
_COMMAND_ROUTE_GROWTH_PER_ROUTE: dict[Metric, int] = {
171171
"chars": 160,
172172
"utf8_bytes": 160,
173173
"lines": 0,
@@ -309,19 +309,21 @@ def _heartbeat_user_language_migration_allowance(
309309
}
310310

311311

312-
def _runtime_root_route_growth_allowances(
312+
def _command_route_growth_allowances(
313313
base: dict[str, Any], candidate: dict[str, Any]
314-
) -> tuple[int, dict[Metric, int]]:
315-
base_routes = base.get("runtime_root_command_route_count")
316-
candidate_routes = candidate.get("runtime_root_command_route_count")
317-
if type(base_routes) is not int or type(candidate_routes) is not int:
318-
return 0, {}
319-
added_routes = max(0, candidate_routes - base_routes)
320-
if not added_routes:
321-
return 0, {}
322-
return added_routes, {
323-
metric: added_routes * allowance
324-
for metric, allowance in _RUNTIME_ROOT_COMMAND_ROUTE_GROWTH_PER_ROUTE.items()
314+
) -> tuple[dict[str, int], dict[Metric, int]]:
315+
additions: dict[str, int] = {}
316+
for option in ("runtime_root", "registry"):
317+
field = f"{option}_command_route_count"
318+
before, after = base.get(field), candidate.get(field)
319+
# Missing, malformed or negative observations never grant an allowance.
320+
if type(before) is int and type(after) is int and 0 <= before < after:
321+
additions[option] = after - before
322+
if not additions:
323+
return {}, {}
324+
return additions, {
325+
metric: sum(additions.values()) * allowance
326+
for metric, allowance in _COMMAND_ROUTE_GROWTH_PER_ROUTE.items()
325327
}
326328

327329

@@ -679,8 +681,8 @@ def _compare_row(base: dict[str, Any], candidate: dict[str, Any]) -> dict[str, A
679681
candidate,
680682
output_format=output_format,
681683
)
682-
added_runtime_root_routes, runtime_root_route_allowances = (
683-
_runtime_root_route_growth_allowances(base, candidate)
684+
added_command_routes, command_route_allowances = (
685+
_command_route_growth_allowances(base, candidate)
684686
)
685687

686688
projection_allowance, projection_failures, projection_signals = _projection_envelope_migration(
@@ -759,10 +761,10 @@ def _compare_row(base: dict[str, Any], candidate: dict[str, Any]) -> dict[str, A
759761
"compact_payload_chars": 512,
760762
}[metric],
761763
)
762-
if runtime_root_route_allowances:
764+
if command_route_allowances:
763765
allowance = max(
764766
allowance,
765-
runtime_root_route_allowances[metric],
767+
command_route_allowances[metric],
766768
)
767769
deltas[metric] = delta
768770
allowances[metric] = allowance
@@ -826,10 +828,10 @@ def _compare_row(base: dict[str, Any], candidate: dict[str, Any]) -> dict[str, A
826828
"planning inventory detail schema migrated: "
827829
f"{migration.inventory_detail_schema_migration}"
828830
)
829-
if runtime_root_route_allowances:
831+
for option, count in added_command_routes.items():
830832
review_signals.append(
831-
"runtime-root command route coverage added: "
832-
f"{added_runtime_root_routes} executable route(s)"
833+
f"{option.replace('_', '-')} command route coverage added: "
834+
f"{count} executable route(s)"
833835
)
834836
if migration.guided_todo_delta_schema_changed:
835837
if migration.guided_todo_delta_schema_migration is None:

‎loopx/control_plane/testing/cli_output_semantics.py‎

Lines changed: 61 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -3,6 +3,7 @@
33
import hashlib
44
import json
55
import re
6+
import shlex
67
from typing import Any
78

89

@@ -79,10 +80,7 @@ def managed_executor_binding_revision(text: str) -> str | None:
7980
)
8081

8182
_MARKDOWN_HEADING = re.compile(r"^#{1,6}\s+.+$")
82-
_RUNTIME_ROOT_COMMAND_ROUTE = re.compile(
83-
r"(?m)(?:^|[\"'`])[^\r\n\S]*loopx\s+--runtime-root\s+"
84-
r"(?:\"[^\"\r\n]+\"|'[^'\r\n]+'|\S+)"
85-
)
83+
8684

8785

8886
def json_shape_paths(value: Any, *, path: str = "$") -> list[str]:
@@ -202,8 +200,65 @@ def markdown_headings(text: str) -> list[str]:
202200
return [line.strip() for line in text.splitlines() if _MARKDOWN_HEADING.match(line)]
203201

204202

205-
def runtime_root_command_route_count(text: str) -> int:
206-
return len(_RUNTIME_ROOT_COMMAND_ROUTE.findall(text))
203+
def command_route_counts(text: str) -> dict[str, int]:
204+
"""Measure well-formed rendered routes, never grant runtime authority.
205+
206+
Decode JSON strings before shell parsing, or read standalone/Markdown code
207+
commands. Prose mentioning an option and malformed argv earn no allowance.
208+
Both bindings are measured in one pass; duplicates within a command count once.
209+
"""
210+
counts = {"runtime_root": 0, "registry": 0}
211+
pending: list[Any] = [text]
212+
commands: list[str] = []
213+
while pending:
214+
value = pending.pop()
215+
if isinstance(value, dict):
216+
pending.extend(value.values())
217+
elif isinstance(value, list):
218+
pending.extend(value)
219+
elif isinstance(value, str):
220+
try:
221+
decoded = json.loads(value)
222+
except ValueError:
223+
for line in value.splitlines():
224+
stripped = line.strip()
225+
if stripped.startswith("loopx "):
226+
commands.append(stripped)
227+
continue
228+
try:
229+
pending.append(json.loads(line))
230+
except ValueError:
231+
commands.extend(re.findall(r"`(loopx [^`\r\n]+)`", line))
232+
else:
233+
if isinstance(decoded, (dict, list, str)):
234+
pending.append(decoded)
235+
236+
for command in commands:
237+
try:
238+
argv = shlex.split(command)
239+
except ValueError:
240+
continue
241+
bindings: set[str] = set()
242+
index = 1
243+
while index < len(argv) and argv[index].startswith("-"):
244+
option = argv[index]
245+
if (option not in {"--registry", "--runtime-root", "--format"}
246+
or index + 1 >= len(argv)
247+
or not argv[index + 1] or argv[index + 1].startswith("-")):
248+
break
249+
if option == "--format":
250+
if argv[index + 1] not in {"json", "markdown"}:
251+
break
252+
else:
253+
bindings.add(option[2:].replace("-", "_"))
254+
index += 2
255+
# Reject an incomplete/invalid option prefix, or one with no subcommand.
256+
if (index == len(argv) or not argv[index].strip()
257+
or argv[index].startswith("-")):
258+
continue
259+
for binding in bindings:
260+
counts[binding] += 1
261+
return counts
207262

208263

209264
def projection_envelope_schema_versions(value: Any) -> list[str]:

‎tests/control_plane/test_cli_output_budget.py‎

Lines changed: 9 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -506,6 +506,14 @@ def _measure_scenario(root: Path, scenario: Scenario) -> dict[str, dict[str, dic
506506
text,
507507
output_format=output_format,
508508
)
509+
if (surface_id, scenario.name, output_format) == (
510+
"loopx_turn_plan", "crowded", "json"
511+
):
512+
# Budget compaction must not discard the writeback target.
513+
action = measurement["payload"]["turn_envelope"]["writeback"]["next_cli_actions"][0]
514+
argv = shlex.split(action)
515+
assert argv[argv.index("--registry") + 1] == str(registry_path)
516+
assert argv[argv.index("--runtime-root") + 1] == str(runtime)
509517
spec = CLI_OUTPUT_BUDGET_BY_ID[surface_id]
510518
assert_cli_output_baseline(
511519
spec,
@@ -1123,7 +1131,7 @@ def test_crowded_turn_plan_budget_preserves_executable_vision_authoring(
11231131
)
11241132
# This fixed executable schema legitimately crosses the old 12k/320
11251133
# ceiling; retain bounded headroom without relaxing Todo-scale growth.
1126-
assert 12_000 < len(text) <= 14_500
1134+
assert 12_000 < len(text) <= CLI_OUTPUT_BUDGET_BY_ID["loopx_turn_plan"].max_chars["crowded"]["json"]
11271135
assert len(text.splitlines()) <= 400
11281136

11291137

‎tests/control_plane/test_cli_output_differential.py‎

Lines changed: 54 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -18,7 +18,7 @@
1818
guided_todo_delta_schema_versions,
1919
planning_horizon_schema_versions,
2020
planning_inventory_detail_schema_versions,
21-
runtime_root_command_route_count,
21+
command_route_counts,
2222
todo_work_counts_schema_versions,
2323
)
2424

@@ -47,6 +47,7 @@ def _row(**overrides: object) -> dict[str, object]:
4747
"planning_inventory_detail_schema_versions": [],
4848
"todo_work_counts_schema_versions": [],
4949
"runtime_root_command_route_count": 0,
50+
"registry_command_route_count": 0,
5051
}
5152
row.update(overrides)
5253
return row
@@ -803,7 +804,8 @@ def test_planning_inventory_detail_migration_is_bounded_and_fail_closed() -> Non
803804
]
804805

805806

806-
def test_runtime_root_route_growth_has_per_route_budget() -> None:
807+
@pytest.mark.parametrize("option", ["runtime_root", "registry"])
808+
def test_command_route_growth_has_per_route_budget(option: str) -> None:
807809
base = _row(
808810
chars=1_000,
809811
utf8_bytes=1_000,
@@ -819,7 +821,7 @@ def test_runtime_root_route_growth_has_per_route_budget() -> None:
819821
compact_payload_chars=1_320,
820822
action_signature_sha256=None,
821823
action_signature_coverages=[],
822-
runtime_root_command_route_count=2,
824+
**{f"{option}_command_route_count": 2},
823825
)
824826

825827
result = compare_cli_output_receipts(_receipt(base), _receipt(candidate))
@@ -833,11 +835,12 @@ def test_runtime_root_route_growth_has_per_route_budget() -> None:
833835
"compact_payload_chars": 320,
834836
}
835837
assert result["rows"][0]["review_signals"] == [
836-
"runtime-root command route coverage added: 2 executable route(s)"
838+
f"{option.replace('_', '-')} command route coverage added: 2 executable route(s)"
837839
]
838840

839841

840-
def test_runtime_root_route_growth_still_fails_above_per_route_budget() -> None:
842+
@pytest.mark.parametrize("option", ["runtime_root", "registry"])
843+
def test_command_route_growth_still_fails_above_per_route_budget(option: str) -> None:
841844
base = _row(
842845
chars=1_000,
843846
utf8_bytes=1_000,
@@ -853,7 +856,7 @@ def test_runtime_root_route_growth_still_fails_above_per_route_budget() -> None:
853856
compact_payload_chars=1_321,
854857
action_signature_sha256=None,
855858
action_signature_coverages=[],
856-
runtime_root_command_route_count=2,
859+
**{f"{option}_command_route_count": 2},
857860
)
858861

859862
result = compare_cli_output_receipts(_receipt(base), _receipt(candidate))
@@ -862,19 +865,22 @@ def test_runtime_root_route_growth_still_fails_above_per_route_budget() -> None:
862865
assert "chars grew by 321; allowance is 320" in result["rows"][0]["failures"]
863866

864867

865-
def test_invalid_runtime_root_route_count_does_not_grant_budget() -> None:
868+
@pytest.mark.parametrize("option", ["runtime_root", "registry"])
869+
@pytest.mark.parametrize("before, after", [(0, True), (0, "2"), (None, 2), (-1, 2), (0, -1), (2, 2)])
870+
def test_invalid_or_unchanged_command_route_count_does_not_grant_budget(option: str, before: object, after: object) -> None:
866871
base = _row(
867872
chars=1_000,
868873
utf8_bytes=1_000,
869874
lines=10,
870875
compact_payload_chars=1_000,
876+
**{f"{option}_command_route_count": before},
871877
)
872878
candidate = _row(
873879
chars=1_097,
874880
utf8_bytes=1_097,
875881
lines=10,
876882
compact_payload_chars=1_097,
877-
runtime_root_command_route_count=True,
883+
**{f"{option}_command_route_count": after},
878884
)
879885

880886
result = compare_cli_output_receipts(_receipt(base), _receipt(candidate))
@@ -899,23 +905,24 @@ def test_runtime_root_route_count_only_matches_executable_command_prefixes() ->
899905
"loopx --runtime-root"
900906
)
901907

902-
assert runtime_root_command_route_count(text) == 4
908+
assert command_route_counts(text)["runtime_root"] == 4
903909

904910

905-
def test_runtime_root_route_allowance_is_fail_closed_for_invalid_counts() -> None:
911+
@pytest.mark.parametrize("option", ["runtime_root", "registry"])
912+
def test_command_route_allowance_is_fail_closed_for_invalid_counts(option: str) -> None:
906913
base = _row(
907914
chars=1_000,
908915
utf8_bytes=1_000,
909916
lines=10,
910917
compact_payload_chars=1_000,
911-
runtime_root_command_route_count=0,
918+
**{f"{option}_command_route_count": 0},
912919
)
913920
candidate = _row(
914921
chars=1_000,
915922
utf8_bytes=1_000,
916923
lines=10,
917924
compact_payload_chars=1_000,
918-
runtime_root_command_route_count="2",
925+
**{f"{option}_command_route_count": "2"},
919926
)
920927

921928
result = compare_cli_output_receipts(_receipt(base), _receipt(candidate))
@@ -1109,3 +1116,38 @@ def test_projection_envelope_migration_is_status_only_bounded_and_one_time(outpu
11091116
outside_base = {**base, "row_id": f"surface/{surface}/small/{output_format}"}
11101117
outside = {**candidate, "row_id": outside_base["row_id"]}
11111118
assert not compare_cli_output_receipts(_receipt(outside_base), _receipt(outside))["ok"]
1119+
1120+
1121+
def test_both_command_routes_are_counted_in_either_order() -> None:
1122+
text = (
1123+
"loopx --registry '/tmp/registry path' --runtime-root /tmp/root refresh-state\n"
1124+
"loopx --runtime-root /tmp/root --registry /tmp/registry quota should-run\n"
1125+
"Use --registry PATH, or say loopx --registry /tmp/path.\n"
1126+
"loopx --registry"
1127+
)
1128+
assert command_route_counts(text) == {"registry": 2, "runtime_root": 2}
1129+
1130+
1131+
@pytest.mark.parametrize("command", [
1132+
'loopx --registry "" turn plan',
1133+
"loopx --registry --runtime-root /tmp/root turn plan",
1134+
'loopx --registry "/tmp/unclosed turn plan',
1135+
"loopx --registry /tmp/registry",
1136+
"loopx --registry /tmp/registry --runtime-root",
1137+
"loopx --registry /tmp/registry --format invalid turn plan",
1138+
'loopx --registry /tmp/registry ""',
1139+
'loopx --registry /tmp/registry " "',
1140+
])
1141+
@pytest.mark.parametrize("render", [str, lambda command: json.dumps({"command": command}), lambda command: f"- execute: `{command}`"])
1142+
def test_malformed_command_never_grants_route_growth(command, render) -> None:
1143+
counts = command_route_counts(render(command))
1144+
assert counts == {"registry": 0, "runtime_root": 0}
1145+
base = _row(chars=1_000, utf8_bytes=1_000, compact_payload_chars=1_000)
1146+
candidate = _row(chars=1_160, utf8_bytes=1_160, compact_payload_chars=1_160,
1147+
**{f"{key}_command_route_count": value for key, value in counts.items()})
1148+
assert compare_cli_output_receipts(_receipt(base), _receipt(candidate))["ok"] is False
1149+
1150+
1151+
def test_json_escaped_paths_and_duplicate_arguments_are_counted_once() -> None:
1152+
command = """loopx --format json --registry '/tmp/a \"quoted\" path' --registry /tmp/final --runtime-root '/tmp/root path' turn plan"""
1153+
assert command_route_counts(json.dumps({"command": command})) == {"registry": 1, "runtime_root": 1}

‎tests/control_plane/test_cli_output_probe_runner.py‎

Lines changed: 3 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,7 @@
11
from __future__ import annotations
22

33
import runpy
4+
import re
45
from pathlib import Path
56

67
import pytest
@@ -86,5 +87,6 @@ def oversized_stdout(command):
8687
return rc, text + " " * 15_000
8788

8889
monkeypatch.setattr(probe, "_invoke_cli", oversized_stdout)
89-
with pytest.raises(AssertionError, match="baseline ceiling is 14500"):
90+
ceiling = probe.CLI_OUTPUT_BUDGET_BY_ID["loopx_turn_plan"].max_chars["crowded"]["json"]
91+
with pytest.raises(AssertionError, match=re.escape(f"baseline ceiling is {ceiling}")):
9092
crowded_turn_probe(probe, cli_output_semantics, tmp_path / "growth")

0 commit comments

Comments
 (0)