From 30ca03d32bba1da678a681c8c95b0737e64851eb Mon Sep 17 00:00:00 2001 From: yilin-succeed <204474593+yilin-succeed@users.noreply.github.com> Date: Wed, 16 Sep 2026 19:06:12 +0800 Subject: [PATCH] fix(quota): settle an autonomous replan spend from its own terminal frontier An autonomous replan whose accepted semantic outcome is a coverage-backed `no_followup` derives `terminal_no_followup` from its durable writeback. The strict terminal guard then rejected the quota spend of that exact settlement, which stranded the settlement and forced an unnecessary control-plane self-repair turn after the benchmark work was already complete (#4501). The terminal guard is correct for new and unrelated work. What it must not do is reject the remaining step of the typed settlement that produced the terminal state. `build_quota_slot_preview_for_decision()` now admits a delivery-completion spend from `terminal_no_followup` only when the request is bound to an autonomous-replan settlement that already owns a committed durable-writeback receipt and does not yet own a committed spend receipt. Every other terminal spend - unbound, mismatched, or already accounted - keeps failing closed. Post-spend terminal closeout needs no new mutation step for the replan path: `receiptBoundReplayPhase` already reports an autonomous-replan settlement as `settled` only once writeback and spend receipts both exist, so ordering the spend before closeout is what makes the existing rule observable. Adds tests/control_plane/test_replan_terminal_spend_ordering.py covering the coverage-backed replan round trip and the unbound terminal rejection. Signed-off-by: YZJF,YCDG,DJLY,ZZZB Signed-off-by: yilin-succeed <204474593+yilin-succeed@users.noreply.github.com> --- loopx/control_plane/quota/slot_accounting.py | 64 +++- .../test_replan_terminal_spend_ordering.py | 289 ++++++++++++++++++ 2 files changed, 351 insertions(+), 2 deletions(-) create mode 100644 tests/control_plane/test_replan_terminal_spend_ordering.py diff --git a/loopx/control_plane/quota/slot_accounting.py b/loopx/control_plane/quota/slot_accounting.py index bc9088e840..4166d99b4a 100644 --- a/loopx/control_plane/quota/slot_accounting.py +++ b/loopx/control_plane/quota/slot_accounting.py @@ -2,7 +2,7 @@ from .effective_action import EffectiveAction import json -from collections.abc import Callable, Iterable +from collections.abc import Callable, Iterable, Mapping from copy import deepcopy from pathlib import Path from typing import Any @@ -22,6 +22,7 @@ ) from .monitor_poll import QUOTA_MONITOR_POLL_CLASSIFICATION from .scheduler_ack import QUOTA_SCHEDULER_ACK_CLASSIFICATION +from ..effect_program import SettlementBindingKind from .settlement import ( SettlementFailureKind, SettlementIdentity, @@ -119,6 +120,20 @@ def _todo_binding_error( ) +def _receipt_committed( + result: SettlementResult[Any] | None, + step_kind: SettlementStepKind, +) -> bool: + """Report whether one settlement step already owns a committed receipt.""" + + if result is None or result.failure is not None: + return False + return any( + receipt.step_kind is step_kind and receipt.status == "committed" + for receipt in result.receipts + ) + + def _resolve_preview_settlement( *, raw_runtime_root: Any, @@ -171,6 +186,12 @@ def _resolve_preview_settlement( "delivery_run": result.value if result.failure is None else None, "delivery_workspace_causality": readback.workspace_causality, "reason": result.failure.reason if result.failure is not None else None, + "writeback_committed": _receipt_committed( + readback.writeback, SettlementStepKind.DURABLE_WRITEBACK + ), + "spend_committed": _receipt_committed( + readback.spend, SettlementStepKind.QUOTA_SPEND + ), } @@ -490,6 +511,41 @@ def _missing_delivery_workspace_preview( } +DELIVERY_COMPLETION_SPEND_STATES = frozenset( + {"waiting", "focus_wait", "operator_gate", "eligible"} +) +TERMINAL_NO_FOLLOWUP_DECISION_STATE = "terminal_no_followup" + + +def _admits_delivery_completion_spend_state( + *, + before: Mapping[str, Any], + identity: SettlementIdentity | None, + settlement: Mapping[str, Any], +) -> bool: + """Admit the decision states one delivery-completion spend may settle from. + + ``terminal_no_followup`` is admitted only for the autonomous-replan + settlement whose own durable writeback derived that frontier and which has + not yet recorded its spend. The terminal guard therefore stays strict for + every new, unrelated, or already-accounted spend, while the remaining step + of the settlement that produced the terminal frontier is no longer + stranded. See the ``terminal_settlement_ordering_gap`` repair pattern. + """ + + state = str(before.get("state") or "") + if state in DELIVERY_COMPLETION_SPEND_STATES: + return True + if state != TERMINAL_NO_FOLLOWUP_DECISION_STATE: + return False + return ( + identity is not None + and identity.binding_kind is SettlementBindingKind.AUTONOMOUS_REPLAN + and settlement.get("writeback_committed") is True + and settlement.get("spend_committed") is not True + ) + + def build_quota_slot_preview_for_decision( status_payload: dict[str, Any], *, @@ -754,7 +810,11 @@ def build_quota_slot_preview_for_decision( ) and before.get("effective_action") != EffectiveAction.AUTOMATION_PROMPT_UPGRADE_REQUIRED.value and not safe_bypass_spend - and str(before.get("state") or "") in {"waiting", "focus_wait", "operator_gate", "eligible"} + and _admits_delivery_completion_spend_state( + before=before, + identity=settlement_identity, + settlement=settlement, + ) ) if delivery_completion_spend: capability_repair_spend = False diff --git a/tests/control_plane/test_replan_terminal_spend_ordering.py b/tests/control_plane/test_replan_terminal_spend_ordering.py new file mode 100644 index 0000000000..35bc91dcf9 --- /dev/null +++ b/tests/control_plane/test_replan_terminal_spend_ordering.py @@ -0,0 +1,289 @@ +"""Regression coverage for issue #4501. + +An autonomous replan whose accepted semantic outcome is a coverage-backed +``no_followup`` derives ``terminal_no_followup`` from its durable writeback. +The strict terminal guard must keep rejecting new and unbound spends, but it +must not reject the remaining quota spend of the very settlement that produced +the terminal frontier. +""" + +from __future__ import annotations + +import json +import shlex +from pathlib import Path +from typing import Any + +from tests.control_plane.test_quota_settlement_cli import ( + AGENT_ID, + GOAL_ID, + _configure_autonomous_replan_fixture, + _projected_cli_args, + _run_cli, + _spend_run_count, + _write_fixture, +) + +TURN_INSTANCE_ID = "turn-terminal-replan-1" + +TERMINAL_STATE = "terminal_no_followup" + + +def _write_terminal_frontier_state(project: Path) -> None: + """Close every Todo source so a coverage-backed replan resolves terminal.""" + + state_path = project / f".codex/goals/{GOAL_ID}/ACTIVE_GOAL_STATE.md" + state_path.write_text( + "---\n" + "status: active\n" + "owner_mode: goal\n" + 'objective: "Settle one coverage-backed terminal replan."\n' + "updated_at: 2026-01-01T00:00:00+00:00\n" + "---\n\n" + "# Terminal Replan Settlement Fixture\n\n" + "## Objective\n\n" + "Settle one coverage-backed terminal replan.\n\n" + "## Next Action\n\n" + "- Close the covered frontier.\n\n" + "## User Todo / Owner Review Reading Queue\n\n" + "## Agent Todo\n\n" + "- [x] [P1-monitor] Observe the stable public fixture.\n" + " \n", + encoding="utf-8", + ) + + +def _coverage_backed_no_followup_refresh_args( + command: str, + *, + vision_path: Path, +) -> tuple[str, ...]: + """Fill the projected refresh template with a coverage-backed no_followup.""" + + command = ( + command.replace( + "", + "no_followup", + ) + .replace("", "surface-terminal") + .replace("", "hypothesis-terminal") + .replace("", "probe-terminal") + .replace("", "evidence-terminal") + ) + command += " --progress-coverage-scope-id coverage-terminal" + command += " --progress-coverage-complete" + command += f" --agent-vision-json {shlex.quote(str(vision_path))}" + return _projected_cli_args(command, turn_instance_id=TURN_INSTANCE_ID) + + +def _write_no_followup_vision(root: Path) -> Path: + vision = { + "schema_version": "goal_vision_replan_contract_v0", + "agent_id": AGENT_ID, + "state": "no_followup", + "vision_patch": { + "acceptance_summary": "The covered frontier is complete.", + }, + "path_delta": { + "schema_version": "goal_path_delta_v0", + "outcome": "stop", + "prior_assumption": "The frontier still needs another probe.", + "observed_reality": "Coverage is complete and no successor exists.", + "retained": ["bounded coverage read model"], + "changed": ["frontier resolution"], + "stopped": ["repeat probe of the covered frontier"], + "evidence_refs": ["evidence-terminal"], + }, + } + vision_path = root / "coverage-backed-no-followup-vision.json" + vision_path.write_text(json.dumps(vision), encoding="utf-8") + return vision_path + + +def _should_run( + registry_path: Path, + runtime: Path, + project: Path, +) -> dict[str, Any]: + rc, payload = _run_cli( + registry_path, + runtime, + "quota", + "should-run", + "--codex-app", + "--goal-id", + GOAL_ID, + "--agent-id", + AGENT_ID, + "--turn-instance-id", + TURN_INSTANCE_ID, + "--scan-path", + str(project), + ) + assert rc == 0, payload + return payload + + +def _dry_run_spend_preview( + registry_path: Path, + runtime: Path, + project: Path, + *, + spend_command: str, +) -> dict[str, Any]: + """Read the decision state the bound spend would settle from. + + Dropping ``--execute`` keeps the read non-mutating, so the terminal state + the writeback produced can be observed without consuming the slot. + """ + + args = [ + arg + for arg in _projected_cli_args( + spend_command, + turn_instance_id=TURN_INSTANCE_ID, + ) + if arg != "--execute" + ] + rc, preview = _run_cli( + registry_path, + runtime, + *args, + "--scan-path", + str(project), + ) + assert rc == 0, preview + return preview + + +def test_coverage_backed_no_followup_replan_does_not_strand_its_quota_spend( + tmp_path: Path, +) -> None: + project, runtime, registry_path = _write_fixture(tmp_path) + _configure_autonomous_replan_fixture(project, runtime, registry_path) + _write_terminal_frontier_state(project) + vision_path = _write_no_followup_vision(tmp_path) + + guard = _should_run(registry_path, runtime, project) + assert guard["decision"] == "autonomous_replan_required", guard + obligation_id = guard["replan_action_packet"]["obligation_id"] + assert guard["heartbeat_receipt"]["settlement_identity"]["binding_kind"] == ( + "autonomous_replan" + ) + actions = guard["interaction_contract"]["cli_channel"]["next_cli_actions"] + refresh_command = next(action for action in actions if "refresh-state" in action) + spend_command = next(action for action in actions if "spend-slot" in action) + for command in (refresh_command, spend_command): + assert f"--replan-obligation-id {obligation_id}" in command + assert f"--turn-instance-id {TURN_INSTANCE_ID}" in command + assert "--todo-id" not in command + + refresh_rc, refresh = _run_cli( + registry_path, + runtime, + *_coverage_backed_no_followup_refresh_args( + refresh_command, + vision_path=vision_path, + ), + ) + assert refresh_rc == 0, refresh + assert refresh["settlement_result"]["ok"] is True + assert [ + receipt["step_kind"] for receipt in refresh["settlement_result"]["receipts"] + ] == ["validation", "durable_writeback"] + + # The writeback makes the Goal terminal before the spend of the same + # settlement has been recorded. That is the #4501 ordering window. + preview = _dry_run_spend_preview( + registry_path, + runtime, + project, + spend_command=spend_command, + ) + assert preview["before"]["state"] == TERMINAL_STATE, preview + assert preview["before"]["effective_action"] == TERMINAL_STATE, preview + + spend_args = _projected_cli_args( + spend_command, + turn_instance_id=TURN_INSTANCE_ID, + ) + spend_rc, spend = _run_cli( + registry_path, + runtime, + *spend_args, + "--scan-path", + str(project), + ) + assert spend_rc == 0, spend + assert spend["settlement_result"]["ok"] is True + assert spend["appended"] is True + receipts = spend["settlement_result"]["receipts"] + assert [receipt["step_kind"] for receipt in receipts] == [ + "validation", + "durable_writeback", + "quota_spend", + ] + effect_ids = {receipt["effect_id"] for receipt in receipts} + assert len(effect_ids) == 1 + assert obligation_id in next(iter(effect_ids)) + + replay_rc, replay = _run_cli( + registry_path, + runtime, + *spend_args, + "--scan-path", + str(project), + ) + assert replay_rc == 0, replay + assert replay["idempotent_replay"] is True + assert replay["appended"] is False + assert _spend_run_count(runtime) == 1 + + +def test_unbound_terminal_spend_stays_rejected(tmp_path: Path) -> None: + """The terminal guard keeps rejecting a spend that owns no settlement.""" + + project, runtime, registry_path = _write_fixture(tmp_path) + _configure_autonomous_replan_fixture(project, runtime, registry_path) + _write_terminal_frontier_state(project) + vision_path = _write_no_followup_vision(tmp_path) + + guard = _should_run(registry_path, runtime, project) + actions = guard["interaction_contract"]["cli_channel"]["next_cli_actions"] + refresh_command = next(action for action in actions if "refresh-state" in action) + refresh_rc, refresh = _run_cli( + registry_path, + runtime, + *_coverage_backed_no_followup_refresh_args( + refresh_command, + vision_path=vision_path, + ), + ) + assert refresh_rc == 0, refresh + + rc, spend = _run_cli( + registry_path, + runtime, + "quota", + "spend-slot", + "--goal-id", + GOAL_ID, + "--slots", + "1", + "--source", + "heartbeat", + "--execute", + "--scan-path", + str(project), + ) + assert spend["ok"] is False + assert spend["appended"] is False + # The Goal is terminal; the guard still rejects a spend that owns no + # settlement binding of its own. + assert spend["before"]["state"] == TERMINAL_STATE, spend + assert _spend_run_count(runtime) == 0 + assert rc in {0, 1}