Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
64 changes: 62 additions & 2 deletions loopx/control_plane/quota/slot_accounting.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@
from .effective_action import EffectiveAction

import json
from collections.abc import Callable, Iterable
from collections.abc import Callable, Iterable, Mapping
from copy import deepcopy
from pathlib import Path
from typing import Any
Expand All @@ -22,6 +22,7 @@
)
from .monitor_poll import QUOTA_MONITOR_POLL_CLASSIFICATION
from .scheduler_ack import QUOTA_SCHEDULER_ACK_CLASSIFICATION
from ..effect_program import SettlementBindingKind
from .settlement import (
SettlementFailureKind,
SettlementIdentity,
Expand Down Expand Up @@ -119,6 +120,20 @@ def _todo_binding_error(
)


def _receipt_committed(
result: SettlementResult[Any] | None,
step_kind: SettlementStepKind,
) -> bool:
"""Report whether one settlement step already owns a committed receipt."""

if result is None or result.failure is not None:
return False
return any(
receipt.step_kind is step_kind and receipt.status == "committed"
for receipt in result.receipts
)


def _resolve_preview_settlement(
*,
raw_runtime_root: Any,
Expand Down Expand Up @@ -171,6 +186,12 @@ def _resolve_preview_settlement(
"delivery_run": result.value if result.failure is None else None,
"delivery_workspace_causality": readback.workspace_causality,
"reason": result.failure.reason if result.failure is not None else None,
"writeback_committed": _receipt_committed(
readback.writeback, SettlementStepKind.DURABLE_WRITEBACK
),
"spend_committed": _receipt_committed(
readback.spend, SettlementStepKind.QUOTA_SPEND
),
}


Expand Down Expand Up @@ -490,6 +511,41 @@ def _missing_delivery_workspace_preview(
}


DELIVERY_COMPLETION_SPEND_STATES = frozenset(
{"waiting", "focus_wait", "operator_gate", "eligible"}
)
TERMINAL_NO_FOLLOWUP_DECISION_STATE = "terminal_no_followup"


def _admits_delivery_completion_spend_state(
*,
before: Mapping[str, Any],
identity: SettlementIdentity | None,
settlement: Mapping[str, Any],
) -> bool:
"""Admit the decision states one delivery-completion spend may settle from.

``terminal_no_followup`` is admitted only for the autonomous-replan
settlement whose own durable writeback derived that frontier and which has
not yet recorded its spend. The terminal guard therefore stays strict for
every new, unrelated, or already-accounted spend, while the remaining step
of the settlement that produced the terminal frontier is no longer
stranded. See the ``terminal_settlement_ordering_gap`` repair pattern.
"""

state = str(before.get("state") or "")
if state in DELIVERY_COMPLETION_SPEND_STATES:
return True
if state != TERMINAL_NO_FOLLOWUP_DECISION_STATE:
return False
return (
identity is not None
and identity.binding_kind is SettlementBindingKind.AUTONOMOUS_REPLAN
and settlement.get("writeback_committed") is True
and settlement.get("spend_committed") is not True
)


def build_quota_slot_preview_for_decision(
status_payload: dict[str, Any],
*,
Expand Down Expand Up @@ -754,7 +810,11 @@ def build_quota_slot_preview_for_decision(
)
and before.get("effective_action") != EffectiveAction.AUTOMATION_PROMPT_UPGRADE_REQUIRED.value
and not safe_bypass_spend
and str(before.get("state") or "") in {"waiting", "focus_wait", "operator_gate", "eligible"}
and _admits_delivery_completion_spend_state(
before=before,
identity=settlement_identity,
settlement=settlement,
)
)
if delivery_completion_spend:
capability_repair_spend = False
Expand Down
289 changes: 289 additions & 0 deletions tests/control_plane/test_replan_terminal_spend_ordering.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,289 @@
"""Regression coverage for issue #4501.

An autonomous replan whose accepted semantic outcome is a coverage-backed
``no_followup`` derives ``terminal_no_followup`` from its durable writeback.
The strict terminal guard must keep rejecting new and unbound spends, but it
must not reject the remaining quota spend of the very settlement that produced
the terminal frontier.
"""

from __future__ import annotations

import json
import shlex
from pathlib import Path
from typing import Any

from tests.control_plane.test_quota_settlement_cli import (
AGENT_ID,
GOAL_ID,
_configure_autonomous_replan_fixture,
_projected_cli_args,
_run_cli,
_spend_run_count,
_write_fixture,
)

TURN_INSTANCE_ID = "turn-terminal-replan-1"

TERMINAL_STATE = "terminal_no_followup"


def _write_terminal_frontier_state(project: Path) -> None:
"""Close every Todo source so a coverage-backed replan resolves terminal."""

state_path = project / f".codex/goals/{GOAL_ID}/ACTIVE_GOAL_STATE.md"
state_path.write_text(
"---\n"
"status: active\n"
"owner_mode: goal\n"
'objective: "Settle one coverage-backed terminal replan."\n'
"updated_at: 2026-01-01T00:00:00+00:00\n"
"---\n\n"
"# Terminal Replan Settlement Fixture\n\n"
"## Objective\n\n"
"Settle one coverage-backed terminal replan.\n\n"
"## Next Action\n\n"
"- Close the covered frontier.\n\n"
"## User Todo / Owner Review Reading Queue\n\n"
"## Agent Todo\n\n"
"- [x] [P1-monitor] Observe the stable public fixture.\n"
" <!-- loopx:todo todo_id=todo_replan_monitor status=done "
"task_class=continuous_monitor action_kind=observe "
f"claimed_by={AGENT_ID} target_key=replan-settlement-fixture "
"cadence=1d next_due_at=2999-01-01T00%3A00%3A00Z "
"no_followup=true evidence=covered -->\n",
encoding="utf-8",
)


def _coverage_backed_no_followup_refresh_args(
command: str,
*,
vision_path: Path,
) -> tuple[str, ...]:
"""Fill the projected refresh template with a coverage-backed no_followup."""

command = (
command.replace(
"<advanced|blocked|exploration_exhausted|no_followup>",
"no_followup",
)
.replace("<surface-id>", "surface-terminal")
.replace("<hypothesis-id>", "hypothesis-terminal")
.replace("<probe-kind>", "probe-terminal")
.replace("<evidence-id>", "evidence-terminal")
)
command += " --progress-coverage-scope-id coverage-terminal"
command += " --progress-coverage-complete"
command += f" --agent-vision-json {shlex.quote(str(vision_path))}"
return _projected_cli_args(command, turn_instance_id=TURN_INSTANCE_ID)


def _write_no_followup_vision(root: Path) -> Path:
vision = {
"schema_version": "goal_vision_replan_contract_v0",
"agent_id": AGENT_ID,
"state": "no_followup",
"vision_patch": {
"acceptance_summary": "The covered frontier is complete.",
},
"path_delta": {
"schema_version": "goal_path_delta_v0",
"outcome": "stop",
"prior_assumption": "The frontier still needs another probe.",
"observed_reality": "Coverage is complete and no successor exists.",
"retained": ["bounded coverage read model"],
"changed": ["frontier resolution"],
"stopped": ["repeat probe of the covered frontier"],
"evidence_refs": ["evidence-terminal"],
},
}
vision_path = root / "coverage-backed-no-followup-vision.json"
vision_path.write_text(json.dumps(vision), encoding="utf-8")
return vision_path


def _should_run(
registry_path: Path,
runtime: Path,
project: Path,
) -> dict[str, Any]:
rc, payload = _run_cli(
registry_path,
runtime,
"quota",
"should-run",
"--codex-app",
"--goal-id",
GOAL_ID,
"--agent-id",
AGENT_ID,
"--turn-instance-id",
TURN_INSTANCE_ID,
"--scan-path",
str(project),
)
assert rc == 0, payload
return payload


def _dry_run_spend_preview(
registry_path: Path,
runtime: Path,
project: Path,
*,
spend_command: str,
) -> dict[str, Any]:
"""Read the decision state the bound spend would settle from.

Dropping ``--execute`` keeps the read non-mutating, so the terminal state
the writeback produced can be observed without consuming the slot.
"""

args = [
arg
for arg in _projected_cli_args(
spend_command,
turn_instance_id=TURN_INSTANCE_ID,
)
if arg != "--execute"
]
rc, preview = _run_cli(
registry_path,
runtime,
*args,
"--scan-path",
str(project),
)
assert rc == 0, preview
return preview


def test_coverage_backed_no_followup_replan_does_not_strand_its_quota_spend(
tmp_path: Path,
) -> None:
project, runtime, registry_path = _write_fixture(tmp_path)
_configure_autonomous_replan_fixture(project, runtime, registry_path)
_write_terminal_frontier_state(project)
vision_path = _write_no_followup_vision(tmp_path)

guard = _should_run(registry_path, runtime, project)
assert guard["decision"] == "autonomous_replan_required", guard
obligation_id = guard["replan_action_packet"]["obligation_id"]
assert guard["heartbeat_receipt"]["settlement_identity"]["binding_kind"] == (
"autonomous_replan"
)
actions = guard["interaction_contract"]["cli_channel"]["next_cli_actions"]
refresh_command = next(action for action in actions if "refresh-state" in action)
spend_command = next(action for action in actions if "spend-slot" in action)
for command in (refresh_command, spend_command):
assert f"--replan-obligation-id {obligation_id}" in command
assert f"--turn-instance-id {TURN_INSTANCE_ID}" in command
assert "--todo-id" not in command

refresh_rc, refresh = _run_cli(
registry_path,
runtime,
*_coverage_backed_no_followup_refresh_args(
refresh_command,
vision_path=vision_path,
),
)
assert refresh_rc == 0, refresh
assert refresh["settlement_result"]["ok"] is True
assert [
receipt["step_kind"] for receipt in refresh["settlement_result"]["receipts"]
] == ["validation", "durable_writeback"]

# The writeback makes the Goal terminal before the spend of the same
# settlement has been recorded. That is the #4501 ordering window.
preview = _dry_run_spend_preview(
registry_path,
runtime,
project,
spend_command=spend_command,
)
assert preview["before"]["state"] == TERMINAL_STATE, preview
assert preview["before"]["effective_action"] == TERMINAL_STATE, preview

spend_args = _projected_cli_args(
spend_command,
turn_instance_id=TURN_INSTANCE_ID,
)
spend_rc, spend = _run_cli(
registry_path,
runtime,
*spend_args,
"--scan-path",
str(project),
)
assert spend_rc == 0, spend
assert spend["settlement_result"]["ok"] is True
assert spend["appended"] is True
receipts = spend["settlement_result"]["receipts"]
assert [receipt["step_kind"] for receipt in receipts] == [
"validation",
"durable_writeback",
"quota_spend",
]
effect_ids = {receipt["effect_id"] for receipt in receipts}
assert len(effect_ids) == 1
assert obligation_id in next(iter(effect_ids))

replay_rc, replay = _run_cli(
registry_path,
runtime,
*spend_args,
"--scan-path",
str(project),
)
assert replay_rc == 0, replay
assert replay["idempotent_replay"] is True
assert replay["appended"] is False
assert _spend_run_count(runtime) == 1


def test_unbound_terminal_spend_stays_rejected(tmp_path: Path) -> None:
"""The terminal guard keeps rejecting a spend that owns no settlement."""

project, runtime, registry_path = _write_fixture(tmp_path)
_configure_autonomous_replan_fixture(project, runtime, registry_path)
_write_terminal_frontier_state(project)
vision_path = _write_no_followup_vision(tmp_path)

guard = _should_run(registry_path, runtime, project)
actions = guard["interaction_contract"]["cli_channel"]["next_cli_actions"]
refresh_command = next(action for action in actions if "refresh-state" in action)
refresh_rc, refresh = _run_cli(
registry_path,
runtime,
*_coverage_backed_no_followup_refresh_args(
refresh_command,
vision_path=vision_path,
),
)
assert refresh_rc == 0, refresh

rc, spend = _run_cli(
registry_path,
runtime,
"quota",
"spend-slot",
"--goal-id",
GOAL_ID,
"--slots",
"1",
"--source",
"heartbeat",
"--execute",
"--scan-path",
str(project),
)
assert spend["ok"] is False
assert spend["appended"] is False
# The Goal is terminal; the guard still rejects a spend that owns no
# settlement binding of its own.
assert spend["before"]["state"] == TERMINAL_STATE, spend
assert _spend_run_count(runtime) == 0
assert rc in {0, 1}
Loading