Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 14 additions & 0 deletions docs/development/testing-and-quality.md
Original file line number Diff line number Diff line change
Expand Up @@ -683,6 +683,20 @@ override or merge bypass.
断言,但必须保留解析、必需字段、锚点和语义差异检查;candidate 始终执行当前绝对
预算。measurement-only 不能用于 candidate,也不是合并旁路。

The real-CLI differential runner and pytest use the same fixed-width fixture
alias **per default scenario**. Alias the scenario root, not just its parent:
otherwise scenario-name suffixes change repeated absolute command paths and
can create a size failure unrelated to output growth. Measure unmodified
stdout, keep fixture populations and budgets unchanged, and retain the separate
real-long-path command-integrity check. This aligns measurement layouts; it does
not shorten production commands or qualify long-path output under short-path caps.

独立 real-CLI 对照和 pytest 对每个默认场景使用相同的固定宽度 fixture 别名。
别名应指向场景根目录,而非仅指向父目录;否则场景名会改变多处绝对命令路径,
产生与输出增长无关的尺寸失败。仍测量未经改写的 stdout,保留原负载、预算及
独立的真实长路径命令完整性检查。这仅统一测量布局,不缩短生产命令,也不将
长路径输出冒充短路径预算已通过。

The PR-review packet's `semantic_alignment` rule consumes this evidence through
the existing `validation_matrix` and `observable_semantics` rows. It does not
add a separate budget receipt or approval gate. The result checker verifies
Expand Down
103 changes: 53 additions & 50 deletions examples/control_plane/cli-output-probe-runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -163,61 +163,64 @@ def _default_rows(
) -> list[dict]:
rows: list[dict] = []
for scenario in probe.SCENARIOS:
project, runtime, registry_path, state_file = probe._write_fixture(
fixture_root / scenario.name,
scenario,
)
for output_format in ("json", "markdown"):
commands = probe._surface_commands(
project=project,
runtime=runtime,
registry_path=registry_path,
state_file=state_file,
output_format=output_format,
# Match _measure_scenario: alias each scenario, not its parent. Otherwise
# emitted command paths include the scenario suffix only in this runner.
with probe._stable_budget_fixture_root(fixture_root / scenario.name) as root:
project, runtime, registry_path, state_file = probe._write_fixture(
root,
scenario,
)
for surface_id, command in commands.items():
exit_code, text = probe._invoke_cli(command)
if exit_code != 0:
raise AssertionError(f"{surface_id}/{output_format} failed")
measurement = probe.measure_cli_output(
text, output_format=output_format
for output_format in ("json", "markdown"):
commands = probe._surface_commands(
project=project,
runtime=runtime,
registry_path=registry_path,
state_file=state_file,
output_format=output_format,
)
surface = probe.CLI_OUTPUT_BUDGET_BY_ID[surface_id]
if enforce_budget:
probe.assert_cli_output_baseline(
surface,
scenario=scenario.name,
output_format=output_format,
text=text,
measurement=measurement,
for surface_id, command in commands.items():
exit_code, text = probe._invoke_cli(command)
if exit_code != 0:
raise AssertionError(f"{surface_id}/{output_format} failed")
measurement = probe.measure_cli_output(
text, output_format=output_format
)
else:
_assert_output_contract(
output_format=output_format,
text=text,
measurement=measurement,
semantic_json_keys=surface.semantic_json_keys,
markdown_anchor=surface.markdown_anchor,
)
rows.append(
_receipt_row(
semantics=semantics,
row_id=f"surface/{surface_id}/{scenario.name}/{output_format}",
surface_id=surface_id,
scenario=scenario.name,
output_format=output_format,
qualification_policy=surface.qualification_policy,
semantic_json_keys=surface.semantic_json_keys,
markdown_anchor=surface.markdown_anchor,
measurement=measurement,
text=text,
output_contract_version=(
getattr(surface, "output_contract_version", None)
if output_format == "json"
else None
surface = probe.CLI_OUTPUT_BUDGET_BY_ID[surface_id]
if enforce_budget:
probe.assert_cli_output_baseline(
surface,
scenario=scenario.name,
output_format=output_format,
text=text,
measurement=measurement,
)
else:
_assert_output_contract(
output_format=output_format,
text=text,
measurement=measurement,
semantic_json_keys=surface.semantic_json_keys,
markdown_anchor=surface.markdown_anchor,
)
rows.append(
_receipt_row(
semantics=semantics,
row_id=f"surface/{surface_id}/{scenario.name}/{output_format}",
surface_id=surface_id,
scenario=scenario.name,
output_format=output_format,
qualification_policy=surface.qualification_policy,
semantic_json_keys=surface.semantic_json_keys,
markdown_anchor=surface.markdown_anchor,
measurement=measurement,
text=text,
output_contract_version=(
getattr(surface, "output_contract_version", None)
if output_format == "json"
else None
),
),
)
)
return rows


Expand Down
90 changes: 90 additions & 0 deletions tests/control_plane/test_cli_output_probe_runner.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,90 @@
from __future__ import annotations

import runpy
from pathlib import Path

import pytest

from loopx.control_plane.testing import cli_output_semantics
from tests.control_plane import test_cli_output_budget as probe


RUNNER = (
Path(__file__).resolve().parents[2]
/ "examples/control_plane/cli-output-probe-runner.py"
)


@pytest.fixture
def crowded_turn_probe(monkeypatch: pytest.MonkeyPatch):
commands = probe._surface_commands
monkeypatch.setattr(probe, "SCENARIOS", (probe.SCENARIOS[1],))

def turn_json_only(**kwargs):
if kwargs["output_format"] != "json":
return {}
return {"loopx_turn_plan": commands(**kwargs)["loopx_turn_plan"]}

monkeypatch.setattr(probe, "_surface_commands", turn_json_only)
return runpy.run_path(str(RUNNER))["_default_rows"]


@pytest.mark.parametrize("layout", ["short", "long-runner-layout" * 8])
def test_runner_uses_the_pytest_scenario_alias_without_rewriting_stdout(
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
crowded_turn_probe,
layout: str,
) -> None:
fixture_roots: list[Path] = []
write_fixture = probe._write_fixture
invoke_cli = probe._invoke_cli
emitted: list[str] = []

def capture_fixture(root, scenario):
fixture_roots.append(root)
assert (scenario.todo_count, scenario.agent_count, scenario.run_count) == (
36,
1,
12,
)
return write_fixture(root, scenario)

def capture_stdout(command):
rc, text = invoke_cli(command)
emitted.append(text)
return rc, text

monkeypatch.setattr(probe, "_write_fixture", capture_fixture)
monkeypatch.setattr(probe, "_invoke_cli", capture_stdout)
rows = crowded_turn_probe(probe, cli_output_semantics, tmp_path / layout)

assert len(rows) == len(emitted) == len(fixture_roots) == 1
root = fixture_roots[0]
assert root.parent == Path("/tmp")
assert root.name.startswith("loopx-cli-budget-")
assert len(root.name.removeprefix("loopx-cli-budget-")) == 12
assert not root.exists() # The shared context cleans up its alias.
assert str(root) in emitted[0] # Full CLI command paths are still emitted.
assert rows[0]["row_id"] == "surface/loopx_turn_plan/crowded/json"
assert rows[0]["chars"] == len(emitted[0])
assert (
"$.turn_envelope.replan_action_packet.writeback_contract.vision_authoring"
in (rows[0]["json_shape_paths"])
)


def test_runner_still_rejects_actual_stdout_growth(
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
crowded_turn_probe,
) -> None:
invoke_cli = probe._invoke_cli

def oversized_stdout(command):
rc, text = invoke_cli(command)
return rc, text + " " * 15_000

monkeypatch.setattr(probe, "_invoke_cli", oversized_stdout)
with pytest.raises(AssertionError, match="baseline ceiling is 14500"):
crowded_turn_probe(probe, cli_output_semantics, tmp_path / "growth")
Loading