From dd23a621d823817a63d456fb23bde688c00fde5d Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 06:23:12 -0400 Subject: [PATCH 01/54] feat(turn-driver): read Turn lane liveness without taking the lane Add `lock_holder_liveness` to the file-lock owner and `turn_lane_liveness` on top of it. Both classify a lane's last executing Turn from its holder record alone: `released` on a clean exit, `dead` when the record names this machine and the pid is gone, `foreign_host` when the pid cannot be checked here, `unreadable` when a lock file carries no parseable record, and `live` only when a same-host pid is still alive. The probe never touches the kernel lock. A probe that acquired it for an instant would refuse a real `run-once --execute` racing that instant with `turn_lane_in_flight` for nothing; the new test drives the real fence wrapper concurrently with a continuous probe and proves the Turn is admitted exactly once. The holder host label is now single-sourced so the writer and the readers cannot disagree on what "this machine" is. Co-Authored-By: Claude Fable 5.1 Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- loopx/control_plane/turn_driver/lane_fence.py | 67 +++++--- loopx/file_lock.py | 91 +++++++++-- tests/test_turn_lane_fence.py | 144 +++++++++++++++++- 3 files changed, 274 insertions(+), 28 deletions(-) diff --git a/loopx/control_plane/turn_driver/lane_fence.py b/loopx/control_plane/turn_driver/lane_fence.py index 5ac99e8f75..9f8ef860e6 100644 --- a/loopx/control_plane/turn_driver/lane_fence.py +++ b/loopx/control_plane/turn_driver/lane_fence.py @@ -17,12 +17,20 @@ from contextlib import contextmanager from functools import wraps import hashlib -import json from pathlib import Path import re from typing import Any -from ...file_lock import lock_holder_path, try_exclusive_file_lock +from ...file_lock import ( + LOCK_HOLDER_ABSENT, + LOCK_HOLDER_DEAD, + LOCK_HOLDER_FOREIGN_HOST, + LOCK_HOLDER_LIVE, + LOCK_HOLDER_RELEASED, + LOCK_HOLDER_UNREADABLE, + lock_holder_liveness, + try_exclusive_file_lock, +) # Typed refusal for a lane whose single executor is already busy. The reason is # a fact about this lane, so a caller can retry it unchanged once it clears. @@ -103,6 +111,18 @@ def turn_lane_singleflight( yield lock_path +def _public_holder(record: Mapping[str, Any]) -> dict[str, Any]: + projection: dict[str, Any] = {} + for field in TURN_LANE_HOLDER_TEXT_FIELDS: + value = record.get(field) + if isinstance(value, str) and value: + projection[field] = value + pid = record.get("pid") + if isinstance(pid, int): + projection["pid"] = pid + return projection + + def turn_lane_holder_readback(target: Path) -> dict[str, Any]: """Return the public-safe identity of the Turn holding one lane, else ``{}``. @@ -113,21 +133,34 @@ def turn_lane_holder_readback(target: Path) -> dict[str, Any]: hosts share one runtime root. """ - try: - record = json.loads(lock_holder_path(target).read_text(encoding="utf-8")) - except (OSError, ValueError): - return {} - if not isinstance(record, Mapping): - return {} - projection: dict[str, Any] = {} - for field in TURN_LANE_HOLDER_TEXT_FIELDS: - value = record.get(field) - if isinstance(value, str) and value: - projection[field] = value - pid = record.get("pid") - if isinstance(pid, int): - projection["pid"] = pid - return projection + _state, record = lock_holder_liveness(target) + return _public_holder(record) + + +# Lane liveness vocabulary: the lock owner's holder states, named here so a +# projection can switch on them without learning the lock record format. +TURN_LANE_LIVE = LOCK_HOLDER_LIVE +TURN_LANE_RELEASED = LOCK_HOLDER_RELEASED +TURN_LANE_DEAD = LOCK_HOLDER_DEAD +TURN_LANE_FOREIGN_HOST = LOCK_HOLDER_FOREIGN_HOST +TURN_LANE_UNREADABLE = LOCK_HOLDER_UNREADABLE +TURN_LANE_ABSENT = LOCK_HOLDER_ABSENT + + +def turn_lane_liveness(target: Path) -> dict[str, Any]: + """Say whether one lane's last executing Turn is still running, read-only. + + The answer comes from the holder record alone: ``released_at`` for a clean + exit, the machine name for whether the pid can be checked here, and pid + liveness for a holder that never released. This never takes the lane lock, + not even for an instant: a probe that did would refuse a real Turn racing + the same instant with ``turn_lane_in_flight`` for no reason. ``live`` is + the only state that is evidence of execution; ``foreign_host`` and + ``unreadable`` are unknowns a consumer must fail closed on. + """ + + state, record = lock_holder_liveness(target) + return {"state": state, "holder": _public_holder(record)} def turn_lane_in_flight_record( diff --git a/loopx/file_lock.py b/loopx/file_lock.py index 4e25c05f96..7ffa396822 100644 --- a/loopx/file_lock.py +++ b/loopx/file_lock.py @@ -175,7 +175,7 @@ def _identity( # hosts sharing one runtime root can both read the holder, so the record # names its own machine and a reader never has to guess which host a pid # belongs to. The name is a sanitized label, not a path or a secret. - "host": _safe_label(socket.gethostname(), fallback="unknown"), + "host": lock_holder_host_label(), "agent_id": _safe_label( agent_id or os.environ.get("LOOPX_AGENT_ID"), fallback="unknown", @@ -268,14 +268,8 @@ def _mark_released( pass -def _read_holder_record(lock_path: Path) -> dict[str, object]: - try: - payload = json.loads(lock_path.read_text(encoding="utf-8")) - except (OSError, json.JSONDecodeError): - return {} - if not isinstance(payload, dict): - return {} - allowed = { +_HOLDER_RECORD_FIELDS = frozenset( + { "schema_version", "lock_id", "policy", @@ -286,7 +280,84 @@ def _read_holder_record(lock_path: Path) -> dict[str, object]: "acquired_at", "released_at", } - return {key: payload[key] for key in allowed if key in payload} +) + + +def _filter_holder_record(payload: object) -> dict[str, object]: + if not isinstance(payload, dict): + return {} + return {key: payload[key] for key in _HOLDER_RECORD_FIELDS if key in payload} + + +def _read_holder_record(lock_path: Path) -> dict[str, object]: + try: + payload = json.loads(lock_path.read_text(encoding="utf-8")) + except (OSError, json.JSONDecodeError): + return {} + return _filter_holder_record(payload) + + +def lock_holder_host_label() -> str: + """The machine label a holder record carries; readers compare against it.""" + + return _safe_label(socket.gethostname(), fallback="unknown") + + +# Liveness of a lock's last holder, read from its record alone. The kernel lock +# is never probed: a probe would hold the lock for an instant, and a real +# single-flight acquisition racing that instant would be refused for nothing. +LOCK_HOLDER_LIVE = "live" +LOCK_HOLDER_RELEASED = "released" +LOCK_HOLDER_DEAD = "dead" +LOCK_HOLDER_FOREIGN_HOST = "foreign_host" +LOCK_HOLDER_UNREADABLE = "unreadable" +LOCK_HOLDER_ABSENT = "absent" +LOCK_HOLDER_LIVENESS_STATES = ( + LOCK_HOLDER_LIVE, + LOCK_HOLDER_RELEASED, + LOCK_HOLDER_DEAD, + LOCK_HOLDER_FOREIGN_HOST, + LOCK_HOLDER_UNREADABLE, + LOCK_HOLDER_ABSENT, +) + + +def lock_holder_liveness(path: Path) -> tuple[str, dict[str, object]]: + """Classify the last holder of one lock without touching the kernel lock. + + Returns the liveness state and the filtered holder record. ``released`` + means the holder wrote ``released_at`` on a clean exit; ``dead`` means the + record names this machine and the pid is gone, which is what a crashed or + killed holder leaves behind; ``foreign_host`` means the pid cannot be + checked from here; ``unreadable`` means a lock file exists but carries no + parseable record, for example mid-acquisition. Only ``live`` is evidence of + a running holder, and even that is pid liveness, not the kernel lock: a + reused pid can keep a crashed holder looking alive until the next holder + overwrites the record. + """ + + holder_path = lock_holder_path(path) + try: + text = holder_path.read_text(encoding="utf-8") + except FileNotFoundError: + return LOCK_HOLDER_ABSENT, {} + except OSError: + return LOCK_HOLDER_UNREADABLE, {} + try: + record = _filter_holder_record(json.loads(text)) + except ValueError: + return LOCK_HOLDER_UNREADABLE, {} + if not record: + return LOCK_HOLDER_UNREADABLE, {} + released_at = record.get("released_at") + if isinstance(released_at, str) and released_at: + return LOCK_HOLDER_RELEASED, record + if record.get("host") != lock_holder_host_label(): + return LOCK_HOLDER_FOREIGN_HOST, record + pid = record.get("pid") + if isinstance(pid, bool) or not isinstance(pid, int): + return LOCK_HOLDER_UNREADABLE, record + return (LOCK_HOLDER_LIVE if process_is_alive(pid) else LOCK_HOLDER_DEAD), record def _operator_action(holder: dict[str, object], *, retry_mode: str) -> dict[str, object]: diff --git a/tests/test_turn_lane_fence.py b/tests/test_turn_lane_fence.py index 0341a6666e..2f5179f4b8 100644 --- a/tests/test_turn_lane_fence.py +++ b/tests/test_turn_lane_fence.py @@ -8,16 +8,29 @@ from __future__ import annotations +import json +import os import socket +import threading +import time from pathlib import Path from loopx.control_plane.turn_driver.executor import run_loopx_turn_once -from loopx.file_lock import _safe_label +from loopx.file_lock import _safe_label, lock_holder_path +from loopx.control_plane.turn_driver import lane_fence from loopx.control_plane.turn_driver.lane_fence import ( REMEDY_WAIT_FOR_IN_FLIGHT_TURN, + single_executor_per_turn_lane, + TURN_LANE_ABSENT, + TURN_LANE_DEAD, + TURN_LANE_FOREIGN_HOST, TURN_LANE_IN_FLIGHT, + TURN_LANE_LIVE, TURN_LANE_OPERATION, + TURN_LANE_RELEASED, + TURN_LANE_UNREADABLE, turn_lane_holder_readback, + turn_lane_liveness, turn_lane_singleflight, turn_lane_target, ) @@ -128,3 +141,132 @@ def test_the_holder_readback_stays_public_safe(tmp_path: Path) -> None: # The private lock identity and the runtime path never leave the process. assert str(tmp_path) not in str(holder) assert turn_lane_holder_readback(tmp_path / "absent.lane") == {} + + +def _lane(tmp_path: Path) -> Path: + return turn_lane_target( + runtime_root=tmp_path / "runtime", goal_id=GOAL_ID, plan=_plan() + ) + + +def _rewrite_holder(target: Path, **changes: object) -> None: + """Edit the holder record the way a crash or another machine would leave it.""" + + holder_path = lock_holder_path(target) + record = json.loads(holder_path.read_text(encoding="utf-8")) + record.pop("released_at", None) + record.update(changes) + holder_path.write_text(json.dumps(record), encoding="utf-8") + + +def test_liveness_follows_the_lane_from_absent_to_live_to_released( + tmp_path: Path, +) -> None: + target = _lane(tmp_path) + assert turn_lane_liveness(target) == {"state": TURN_LANE_ABSENT, "holder": {}} + + with turn_lane_singleflight( + runtime_root=tmp_path / "runtime", goal_id=GOAL_ID, plan=_plan() + ) as held: + assert held is not None + live = turn_lane_liveness(target) + + assert live["state"] == TURN_LANE_LIVE + # The holder is the same public-safe readback a refusal names. + assert live["holder"] == turn_lane_holder_readback(target) | {"pid": os.getpid()} + assert live["holder"]["pid"] == os.getpid() + assert str(tmp_path) not in json.dumps(live) + # A clean exit is a release, whatever the pid does afterwards. + assert turn_lane_liveness(target)["state"] == TURN_LANE_RELEASED + + +def test_liveness_fails_closed_on_dead_foreign_and_unreadable_holders( + tmp_path: Path, +) -> None: + target = _lane(tmp_path) + with turn_lane_singleflight( + runtime_root=tmp_path / "runtime", goal_id=GOAL_ID, plan=_plan() + ): + pass + + # A killed Turn never writes released_at; its pid is gone on this machine. + dead_pid = os.getpid() + while True: + dead_pid += 1 + try: + os.kill(dead_pid, 0) + except ProcessLookupError: + break + except OSError: + continue + if dead_pid > os.getpid() + 100_000: + raise AssertionError("no free pid found near this process") + _rewrite_holder(target, pid=dead_pid) + assert turn_lane_liveness(target)["state"] == TURN_LANE_DEAD + + # A holder on another machine cannot be pid-checked here, even if that pid + # happens to be alive on this one. + _rewrite_holder(target, pid=os.getpid(), host="another-machine") + foreign = turn_lane_liveness(target) + assert foreign["state"] == TURN_LANE_FOREIGN_HOST + assert foreign["holder"]["host"] == "another-machine" + + # A lock file with no parseable record is mid-acquisition or corrupt: not + # absent, and not evidence of anything. + lock_holder_path(target).write_text("", encoding="utf-8") + assert turn_lane_liveness(target) == {"state": TURN_LANE_UNREADABLE, "holder": {}} + lock_holder_path(target).write_text("{}", encoding="utf-8") + assert turn_lane_liveness(target)["state"] == TURN_LANE_UNREADABLE + + +def test_the_liveness_probe_never_refuses_a_concurrent_executing_turn( + tmp_path: Path, monkeypatch +) -> None: + """The probe reads a record; it never takes the lane, not even for an instant.""" + + fence_calls: list[str] = [] + real_fence = lane_fence.try_exclusive_file_lock + + def counting_fence(*args, **kwargs): + fence_calls.append(str(kwargs.get("operation"))) + return real_fence(*args, **kwargs) + + monkeypatch.setattr(lane_fence, "try_exclusive_file_lock", counting_fence) + target = _lane(tmp_path) + turn_lane_liveness(target) + turn_lane_holder_readback(target) + assert fence_calls == [] + + # The executing entry is the real fence wrapper run-once --execute goes + # through; only the Turn body is a stand-in that holds the lane a moment. + @single_executor_per_turn_lane( + lambda plan, record, **kwargs: {**record, "effects": kwargs["effects"]} + ) + def executing_turn(plan, *, runtime_root, goal_id, execute): + time.sleep(0.3) + return {"status": "committed", "held": turn_lane_liveness(target)["state"]} + + observed: set[str] = set() + stop = threading.Event() + + def probe() -> None: + while not stop.is_set(): + observed.add(turn_lane_liveness(target)["state"]) + + prober = threading.Thread(target=probe, daemon=True) + prober.start() + try: + payload = executing_turn( + _plan(), runtime_root=tmp_path / "runtime", goal_id=GOAL_ID, execute=True + ) + finally: + stop.set() + prober.join(timeout=5) + + # The Turn took the fence exactly once and was never told the lane was busy. + assert fence_calls == [TURN_LANE_OPERATION] + assert payload == {"status": "committed", "held": TURN_LANE_LIVE} + assert payload.get("reason") != TURN_LANE_IN_FLIGHT + assert observed <= {TURN_LANE_ABSENT, TURN_LANE_LIVE, TURN_LANE_RELEASED, TURN_LANE_UNREADABLE} + assert TURN_LANE_LIVE in observed + assert turn_lane_liveness(target)["state"] == TURN_LANE_RELEASED From 0b1a93f32f8f30cc65f1d6748edd104320c51d84 Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 06:16:04 -0400 Subject: [PATCH 02/54] feat(delegation): add stopped observation and typed stop decision Delegated members had no stopped observation: prepared, running and turn_returned could only end in accepted or rejected. Add "stopped" as a terminal observation reachable from the three open states and keep the inventory check driven by the same transition table. Add the exported collaboration.delegation.stop decision: a stop request settles only with an acknowledgement from a process that held the operation lock plus a free operation lock and a free Turn lane lock. Free locks with no acknowledgement are "unknown", never a fabricated settlement, and a grace timeout alone changes nothing while a lock is still held. Register the RPC handler next to the existing delegation observation handlers. Co-Authored-By: Claude Fable 5.1 Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- .../control_plane/collaboration/delegation.ts | 39 +++++++++++++++++-- .../control_plane/effect_runtime_handlers.ts | 3 +- 2 files changed, 38 insertions(+), 4 deletions(-) diff --git a/loopx/control_plane/collaboration/delegation.ts b/loopx/control_plane/collaboration/delegation.ts index f661435365..9ba1f172da 100644 --- a/loopx/control_plane/collaboration/delegation.ts +++ b/loopx/control_plane/collaboration/delegation.ts @@ -81,7 +81,7 @@ export function selectDelegationBinding(params: JsonObject): JsonObject { return binding; } -type Observation = "prepared" | "running" | "turn_returned" | "accepted" | "rejected"; +type Observation = "prepared" | "running" | "turn_returned" | "accepted" | "rejected" | "stopped"; function boundedReason(value: unknown, fallback: string): string { if (typeof value !== "string") return fallback; @@ -277,8 +277,8 @@ export function delegationPreflight(params: JsonObject): JsonObject { }; } const transitions: Record = { - prepared: ["running", "rejected"], running: ["turn_returned", "rejected"], - turn_returned: ["accepted", "rejected"], accepted: [], rejected: [], + prepared: ["running", "rejected", "stopped"], running: ["turn_returned", "rejected", "stopped"], + turn_returned: ["accepted", "rejected", "stopped"], accepted: [], rejected: [], stopped: [], }; /** Page only the caller's existing journal. A cursor is not a fleet snapshot. */ @@ -341,6 +341,39 @@ export function transitionDelegationObservation(params: JsonObject): JsonObject return {status: to}; } +type StopPhase = "requested" | "acknowledged" | "settled" | "unknown"; +const openStopPhases: readonly StopPhase[] = ["requested", "acknowledged"]; + +/** Advance one stop request from host lock facts; a receipt is never inferred from time. + * + * ``settled`` needs the acknowledgement of a process that held the operation + * lock plus both the operation lock and the Turn lane lock free: only then is + * the worker, its Turn child and its lane provably gone. Free locks without an + * acknowledgement mean the holder vanished before recording what it observed, + * which is ``unknown`` rather than a fake settlement. A grace timeout on its + * own moves nothing: a worker that is still holding a lock is still running. + */ +export function decideDelegationStop(params: JsonObject): JsonObject { + const phase = params.phase as StopPhase; + requireThat(openStopPhases.includes(phase), "delegation stop decision requires an open stop phase"); + requireThat(typeof params.acknowledged === "boolean", "delegation stop acknowledgement fact required"); + requireThat(typeof params.operation_lock_free === "boolean" && typeof params.lane_lock_free === "boolean", + "delegation stop lock facts required"); + requireThat(params.timed_out === undefined || typeof params.timed_out === "boolean", + "delegation stop timeout fact must be boolean"); + requireThat(phase !== "acknowledged" || params.acknowledged === true, + "an acknowledged stop cannot lose its acknowledgement"); + const locksFree = params.operation_lock_free === true && params.lane_lock_free === true; + if (params.acknowledged === true) { + if (locksFree) return {phase: "settled", terminal: true, reason: "acknowledged_and_locks_released"}; + return {phase: "acknowledged", terminal: false, reason: params.operation_lock_free === true + ? "turn_lane_still_held" : "operation_lock_still_held"}; + } + if (locksFree) return {phase: "unknown", terminal: true, reason: "holder_gone_without_acknowledgement"}; + return {phase: "requested", terminal: false, reason: params.timed_out === true + ? "holder_still_running_after_grace" : "awaiting_acknowledgement"}; +} + /** Repair only a false terminal observation after the exact Turn validated. * * This does not retry model work. The host boundary must prove that the diff --git a/loopx/control_plane/effect_runtime_handlers.ts b/loopx/control_plane/effect_runtime_handlers.ts index c239fa3c49..5eb9df9079 100644 --- a/loopx/control_plane/effect_runtime_handlers.ts +++ b/loopx/control_plane/effect_runtime_handlers.ts @@ -16,7 +16,7 @@ import {evaluateUserCompletion} from "./todos/user_completion.ts"; import {projectTodoSuccession} from "./todos/succession.ts"; import {projectLegacyTodoWorkCounts} from "./todos/summary_lanes.ts"; import {sealProjectionEnvelope} from "./projection_envelope.ts"; -import {recordDelegationAdoption, delegationInventoryItem, delegationInventoryQuery, delegationPreflight, delegationTurnPlanDecision, delegationValidationPlan, recoverValidatedDelegationSettlement, selectDelegationBinding, transitionDelegationObservation} from "./collaboration/delegation.ts"; +import {recordDelegationAdoption, decideDelegationStop, delegationInventoryItem, delegationInventoryQuery, delegationPreflight, delegationTurnPlanDecision, delegationValidationPlan, recoverValidatedDelegationSettlement, selectDelegationBinding, transitionDelegationObservation} from "./collaboration/delegation.ts"; import {resolveConversationTrigger} from "./collaboration/conversation_trigger.ts"; import {admitGoalDraft} from "./collaboration/goal_draft.ts"; import {planChatMode} from "./collaboration/chat_mode.ts"; @@ -744,6 +744,7 @@ export function createEffectRuntimeHandlers( ["chat.turn.accept", planChatTurnAcceptance], ["collaboration.delegation.observe", transitionDelegationObservation], ["collaboration.delegation.recover_validated_settlement", recoverValidatedDelegationSettlement], + ["collaboration.delegation.stop", decideDelegationStop], ["collaboration.delegation.adoption", recordDelegationAdoption], [ "collaboration.request.normalize", From 62d4c920e7d8f8018ace8b84510f4bc21a6dcd96 Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 06:29:48 -0400 Subject: [PATCH 03/54] feat(delegation): stop delegated members with acknowledged, settled receipts Delegated members could not be stopped: the worker held the operation lock for the whole run and rewrote the execution record from memory, so nothing written into that record could reach it or survive it, and a killed run left its Turn and hard lease unaccounted for. Add a stop receipt beside the execution record (executions//.stop.json) written only under the existing .dispatch lock, never into the record. Its phases are requested -> acknowledged -> settled with the terminals unknown and noop, and it records the requester, the lock-holding worker (pid, process group, host), the acknowledgement (pid, observed status, Turn key), the lease release outcome and the settlement facts (operation lock, Turn lane lock, Turn journal status). Delegations.stop: a terminal record returns an identical noop receipt on every call. Otherwise the request is written; when no worker holds the operation the requester takes the lock, marks the record stopped, releases the hard lease and settles. A same-host holder is SIGTERMed by process group (its run-once child and host bridge follow) and SIGKILLed only if it still holds the operation after the grace; another host's holder is never signalled and finds the request itself. Settlement is the typed collaboration.delegation.stop decision over lock facts: elapsed time is never a receipt. Worker: a SIGTERM handler raises DelegationStopRequested once; checkpoints before running, before each run-once and before completing the Todo raise on a stop file; the record is written only through a fenced write that re-reads the stop file under .dispatch and raises DelegationFenced for a stop this process did not acknowledge, so a late-returning or other-host worker records no Turn result and completes no Todo. Acknowledgement marks the record stopped, releases the hard lease and leaves the in_progress Turn journal for inspection. execute records the worker's pid/pgid/host at entry and acknowledges from under the lock when a stop already exists. resume refuses stopped work ("start a new operation id"), wait returns on stopped, and read exposes the stop phase. file_lock gains local_lock_host and read_lock_holder so the stop names the holder exactly as lock records do. Co-Authored-By: Claude Fable 5.1 Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- loopx/collaboration_mcp.py | 414 ++++++++++++++++++++++++++++++++++--- loopx/file_lock.py | 11 + 2 files changed, 400 insertions(+), 25 deletions(-) diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index c3f2b9304b..64e192ff00 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -14,6 +14,7 @@ import hashlib import json import os +import signal import stat import subprocess import sys @@ -26,7 +27,10 @@ if TYPE_CHECKING: from mcp.server.fastmcp import FastMCP -from .file_lock import exclusive_file_lock, LockAcquisitionPolicy, LockAcquireTimeoutError +from .file_lock import ( + exclusive_file_lock, lock_holder_host_label, read_lock_holder, try_exclusive_file_lock, + LockAcquisitionPolicy, LockAcquireTimeoutError, +) from .control_plane.effect_runtime import ( effect_runtime_request_scope, effect_runtime_result, EffectRuntimeRemoteError, ) @@ -38,6 +42,8 @@ turn_journal_path, ) from .control_plane.turn_driver.host_binding import turn_host_arg_option +from .control_plane.turn_driver.lane_fence import turn_lane_target +from .control_plane.work_items.task_lease import release_task_lease from .control_plane.collaboration.inbox import _hash, _read, _write, _root, _receipt from .control_plane.collaboration.peers import return_result from .control_plane.collaboration.inbox import acknowledge, _entry, normalize_request @@ -300,6 +306,65 @@ def consume_peer_result(request_id: str) -> dict: ) +DELEGATION_STOP_SCHEMA_VERSION = "loopx_delegation_stop_v0" +# Observations that no worker may reopen; a stop against one is a no-op receipt. +DELEGATION_TERMINAL_STATUSES = frozenset({"accepted", "rejected", "stopped"}) +DELEGATION_STOP_OPEN_PHASES = frozenset({"requested", "acknowledged"}) +# How long a signalled same-host worker may take to acknowledge before SIGKILL. +DELEGATION_STOP_GRACE_SECONDS = 10.0 +DELEGATION_STOPPED_MESSAGE = "delegation operation was stopped; start a new operation id" + + +class DelegationStopRequested(Exception): + """A stop reached the worker that owns this operation; it must acknowledge, not finish.""" + + def __init__(self, source: str) -> None: + super().__init__(source) + self.source = source + + +class DelegationFenced(DelegationStopRequested): + """A stop this process never acknowledged fences its execution-record write. + + The write is refused before it happens: a late-returning or other-host + worker records no Turn result, completes no Todo and publishes nothing. + """ + + def __init__(self) -> None: + super().__init__("fenced") + + +class _WorkerStopSignal: + """Turn the first SIGTERM into a stop request; absorb later ones during the acknowledgement.""" + + def __init__(self) -> None: + self.armed = True + + def __call__(self, signum: int, frame: object) -> None: + if not self.armed: + return + self.armed = False + raise DelegationStopRequested("SIGTERM") + + def disarm(self) -> None: + self.armed = False + + +def install_worker_stop_signal() -> _WorkerStopSignal | None: + """Install the detached worker's SIGTERM handler; ``None`` where signals are unsupported.""" + + if not hasattr(signal, "SIGTERM"): + return None + handler = _WorkerStopSignal() + try: + signal.signal(signal.SIGTERM, handler) + except (ValueError, OSError): + # Not the main thread, or a platform without handler support: the + # worker still honours stop files at every checkpoint and fenced write. + return None + return handler + + class Delegations: """Host IO for bound peer work; typed grants and observations stay in TS. @@ -311,6 +376,7 @@ def __init__(self, root: Path, registry: Path, goal_id: str, agent_id: str, conf self.root, self.registry = root.resolve(), registry.resolve() self.goal_id, self.agent_id, self.config = goal_id, agent_id, config.resolve() self._goal_ref_lock = Lock() + self._stop_signal: _WorkerStopSignal | None = None try: self.goal_ref = capture_collaboration_goal_ref( self.registry, @@ -352,6 +418,22 @@ def directory(self) -> dict: def path(self, operation_id: str) -> Path: return _root(self.root) / "executions" / _hash([self.goal_id, self.agent_id]) / (_hash(operation_id) + ".json") + @staticmethod + def _stop_path(path: Path) -> Path: + """The stop receipt sits beside its execution record and is never merged into it.""" + return path.with_name(path.stem + ".stop.json") + + @staticmethod + def _dispatch_lock(path: Path) -> Path: + return path.with_suffix(".dispatch") + + @staticmethod + def _read_stop(path: Path) -> dict | None: + stop_path = Delegations._stop_path(path) + if not stop_path.exists(): + return None + return _read(stop_path) + def operations(self, *, limit: int = 20, cursor: str | None = None) -> dict: from .control_plane.collaboration.delegation_inventory import read_delegation_inventory @@ -499,15 +581,19 @@ def _spawn(self, operation_id: str) -> None: def resume(self, operation_id: str) -> dict: path = self.path(operation_id) + if self._read_stop(path) is not None: + raise ValueError(DELEGATION_STOPPED_MESSAGE) try: with exclusive_file_lock( path, policy=LockAcquisitionPolicy.SINGLE_FLIGHT ): row = _read(path) binding = self._bound(row) + if row["status"] == "stopped": + raise ValueError(DELEGATION_STOPPED_MESSAGE) if row["status"] == "rejected": self._recover_validated_settlement(path, row, binding) - should_spawn = row["status"] not in {"accepted", "rejected"} + should_spawn = row["status"] not in DELEGATION_TERMINAL_STATUSES if should_spawn: self.binding( row["identity"]["binding"]["id"], require_active=True @@ -524,7 +610,7 @@ def wait(self, operation_id: str) -> dict: """Observe for at most 15 seconds; waiting neither starts nor resumes work.""" for _ in range(5): result = self.read(operation_id) - if result["status"] in {"accepted", "rejected"} or result["recovery_required"]: + if result["status"] in DELEGATION_TERMINAL_STATUSES or result["recovery_required"]: return result time.sleep(3) return self.read(operation_id) @@ -611,7 +697,7 @@ def _recover_validated_settlement( "error": None, } row.pop("error", None) - _write(path, row) + self._fenced_write(path, row) return True def adopt_result(self, operation_id: str, consumer_operation_id: str) -> dict: @@ -637,8 +723,11 @@ def _read_current(self, operation_id: str) -> dict: result = {"operation_id": operation_id, "request_id": row["identity"]["request_id"], "agent_id": binding["agent_id"], "todo_id": binding["todo_id"], "status": row["status"], "worker_active": active, - "recovery_required": not active and row["status"] not in {"accepted", "rejected"} + "recovery_required": not active and row["status"] not in DELEGATION_TERMINAL_STATUSES and time.time() - row.get("created_at", 0) > 15} + stop = self._read_stop(path) + if stop is not None: + result["stop"] = {"stop_id": stop["stop_id"], "phase": stop["phase"]} if row["status"] == "accepted": # A saved receipt cannot hide an amended task, verifier or output. artifacts = self._accepted(binding) @@ -654,7 +743,261 @@ def _observe(self, path: Path, row: dict, status: str, **facts) -> None: "from": row["status"], "to": status, **facts, }) row.update(status=decision["status"]) - _write(path, row) + self._fenced_write(path, row) + + def _fenced_write(self, path: Path, row: dict) -> None: + """Write the execution record only while no unacknowledged stop fences this process. + + The stop receipt is re-read under the dispatch lock on every write, so a + worker that returns after a stop it never saw writes nothing at all. + """ + + with exclusive_file_lock(self._dispatch_lock(path)): + stop = self._read_stop(path) + if stop is not None and not self._acknowledged_here(stop): + raise DelegationFenced() + _write(path, row) + + @staticmethod + def _acknowledged_here(stop: dict) -> bool: + ack = stop.get("ack") + return (isinstance(ack, dict) and ack.get("pid") == os.getpid() + and ack.get("host") == lock_holder_host_label()) + + def _raise_if_stop_requested(self, path: Path) -> None: + """Worker checkpoint: leave before the next host launch or Todo effect.""" + if self._read_stop(path) is not None: + raise DelegationStopRequested("stop_file") + + @staticmethod + def _worker_identity() -> dict: + return { + "pid": os.getpid(), + "pgid": os.getpgid(0) if hasattr(os, "getpgid") else None, + "host": lock_holder_host_label(), + } + + def _lane_target(self, binding: dict) -> Path: + return turn_lane_target(runtime_root=self.root, goal_id=self.goal_id, + plan={"turn_envelope": {"agent_id": binding["agent_id"]}}) + + def _operation_lock_free(self, path: Path) -> bool: + try: + with exclusive_file_lock(path, policy=LockAcquisitionPolicy.SINGLE_FLIGHT): + return True + except LockAcquireTimeoutError: + return False + + def _lane_lock_free(self, binding: dict) -> bool: + # The kernel lock is the only proof; the holder record is advisory. The + # probe holds the lane for an instant, which is the same observation a + # status read makes on the operation lock. + with try_exclusive_file_lock(self._lane_target(binding), agent_id=self.agent_id, + operation="loopx_delegation_stop_probe") as held: + return held is not None + + def _turn_journal_status(self, row: dict, binding: dict) -> str | None: + turn_key = row.get("turn_key") or self._matching_turn_key(row, binding) + if not turn_key: + return None + journal = load_turn_journal(turn_journal_path(self.root, goal_id=self.goal_id, turn_key=turn_key)) + status = journal.get("status") if isinstance(journal, dict) else None + return str(status) if status else None + + def _new_stop_record(self, row: dict, *, requested_by: str, worker: dict | None) -> dict: + requested_at = time.time() + return { + "schema_version": DELEGATION_STOP_SCHEMA_VERSION, + "stop_id": _hash([row["identity"]["operation_id"], requested_by, requested_at])[:32], + "operation_id": row["identity"]["operation_id"], + "request_id": row["identity"]["request_id"], + "phase": "requested", + "reason": "awaiting_acknowledgement", + "requested_by": requested_by, + "requested_at": requested_at, + "requested_status": row["status"], + "worker": worker, + "ack": None, + "lease": None, + "settled": None, + } + + def _stop_receipt(self, row: dict, binding: dict, stop: dict | None) -> dict: + receipt = { + "operation_id": row["identity"]["operation_id"], + "request_id": row["identity"]["request_id"], + "agent_id": binding["agent_id"], "todo_id": binding["todo_id"], + "status": row["status"], + } + if stop is None: + # Nothing was written: a terminal observation cannot be stopped, and + # repeating the request returns exactly this receipt again. + receipt.update(phase="noop", reason="delegation already " + row["status"], stop=None) + else: + receipt.update(phase=stop["phase"], reason=stop.get("reason"), stop=stop) + return receipt + + def stop(self, operation_id: str, *, execute: bool) -> dict: + """Stop one bounded member and return a receipt that says what was proven. + + ``requested`` is written beside the execution record, never into it. + When no worker holds the operation, this caller takes the lock, marks + the record stopped and releases the hard lease itself. A same-host + holder is signalled by process group and given a bounded grace to + acknowledge; another host's holder is left to find the request at its + next checkpoint or fenced write. ``settled`` and ``unknown`` come from + the typed decision over lock facts; elapsed time proves nothing. + """ + + require_operation_id(operation_id) + if not execute: + raise ValueError("delegation stop requires execute") + path = self.path(operation_id) + if not path.exists(): + raise ValueError("unknown delegation operation; start_delegation returns the operation_id to stop") + with exclusive_file_lock(self._dispatch_lock(path)): + row = _read(path) + binding = self._bound(row) + stop = self._read_stop(path) + if stop is None: + if row["status"] in DELEGATION_TERMINAL_STATUSES: + return self._stop_receipt(row, binding, None) + stop = self._new_stop_record(row, requested_by=self.agent_id, + worker=self._lock_holder_worker(path, row)) + _write(self._stop_path(path), stop) + if stop["phase"] in DELEGATION_STOP_OPEN_PHASES and stop.get("ack") is None: + try: + with exclusive_file_lock(path, policy=LockAcquisitionPolicy.SINGLE_FLIGHT): + row = _read(path) + self._acknowledge_stop(path, row, binding, source="requester") + except LockAcquireTimeoutError: + self._signal_worker(path, stop) + return self._settle_stop(path) + + def _lock_holder_worker(self, path: Path, row: dict) -> dict | None: + """Name the live operation-lock holder, else ``None``; a released record is not a worker.""" + + if self._operation_lock_free(path): + return None + holder = read_lock_holder(path) + if "released_at" in holder or not isinstance(holder.get("pid"), int): + return None + recorded = row.get("worker") if isinstance(row.get("worker"), dict) else {} + same = recorded.get("pid") == holder["pid"] and recorded.get("host") == holder.get("host") + pgid = recorded.get("pgid") if same and isinstance(recorded.get("pgid"), int) else holder["pid"] + return {"pid": holder["pid"], "pgid": pgid, "host": holder.get("host")} + + def _signal_worker(self, path: Path, stop: dict) -> None: + """Terminate a same-host holder's process group; never signal across hosts.""" + + worker = stop.get("worker") + if (not isinstance(worker, dict) or worker.get("host") != lock_holder_host_label() + or not hasattr(os, "killpg") or not isinstance(worker.get("pid"), int)): + return + pid, pgid = worker["pid"], worker.get("pgid") or worker["pid"] + if pgid == os.getpgid(0): + raise ValueError("delegation stop refuses to signal its own process group") + try: + if os.getpgid(pid) != pgid: + return # the pid was reused by an unrelated process + os.killpg(pgid, signal.SIGTERM) + except ProcessLookupError: + return + deadline = time.monotonic() + DELEGATION_STOP_GRACE_SECONDS + while time.monotonic() < deadline: + if self._operation_lock_free(path): + return + time.sleep(0.2) + try: + os.killpg(pgid, signal.SIGKILL) + except ProcessLookupError: + return + deadline = time.monotonic() + 2.0 + while time.monotonic() < deadline and not self._operation_lock_free(path): + time.sleep(0.1) + + def _acknowledge_stop(self, path: Path, row: dict, binding: dict, *, source: str) -> None: + """Acknowledge from under the operation lock: mark stopped, then release the hard lease. + + Only the lock holder may acknowledge. A stop that another process already + acknowledged, or that already settled, is left untouched. + """ + + if self._stop_signal is not None: + self._stop_signal.disarm() + with exclusive_file_lock(self._dispatch_lock(path)): + stop = self._read_stop(path) + if stop is None: + stop = self._new_stop_record(row, requested_by="signal:" + source, + worker=self._worker_identity()) + if stop.get("ack") is not None or stop["phase"] not in DELEGATION_STOP_OPEN_PHASES: + return + observed = row["status"] + decision = effect_runtime_result("collaboration.delegation.observe", { + "from": observed, "to": "stopped", + }) + row.update(status=decision["status"]) + _write(path, row) + stop.update(phase="acknowledged", reason="awaiting_lock_release", ack={ + "pid": os.getpid(), "host": lock_holder_host_label(), "at": time.time(), + "source": source, "observed_status": observed, + "turn_key": row.get("turn_key") or self._matching_turn_key(row, binding), + }) + _write(self._stop_path(path), stop) + try: + self._clear_delegation_bootstrap(row, binding) + except (OSError, ValueError): + pass # the bootstrap is host input; its state never blocks the receipt + lease = self._release_delegation_lease(row, binding) + with exclusive_file_lock(self._dispatch_lock(path)): + stop = self._read_stop(path) or stop + stop["lease"] = lease + _write(self._stop_path(path), stop) + + def _release_delegation_lease(self, row: dict, binding: dict) -> dict: + lease = row.get("task_lease") + if not isinstance(lease, dict) or lease.get("required") is not True: + return {"required": False, "released": None} + try: + result = release_task_lease( + runtime_root=self.root, goal_id=self.goal_id, todo_id=binding["todo_id"], + owner=binding["agent_id"], idempotency_key=str(lease["idempotency_key"]), + expected_version=lease.get("version"), registry_path=self.registry, + ) + except (ValueError, OSError, RuntimeError, EffectRuntimeRemoteError) as exc: + return {"required": True, "released": False, "error": str(exc)[:180]} + return {"required": True, "released": result.get("released") is True, + "missing": result.get("missing") is True} + + def _settle_stop(self, path: Path) -> dict: + with exclusive_file_lock(self._dispatch_lock(path)): + row = _read(path) + binding = self._bound(row) + stop = self._read_stop(path) + if stop is None: + return self._stop_receipt(row, binding, None) + if stop["phase"] not in DELEGATION_STOP_OPEN_PHASES: + return self._stop_receipt(row, binding, stop) + facts = { + "operation_lock_free": self._operation_lock_free(path), + "lane_lock_free": self._lane_lock_free(binding), + } + decision = effect_runtime_result("collaboration.delegation.stop", { + "phase": stop["phase"], "acknowledged": stop.get("ack") is not None, + "timed_out": time.time() - stop["requested_at"] > DELEGATION_STOP_GRACE_SECONDS, + **facts, + }) + if decision["phase"] != stop["phase"] or decision.get("reason") != stop.get("reason"): + stop.update(phase=decision["phase"], reason=decision.get("reason")) + if decision["phase"] in {"settled", "unknown"}: + lease = stop.get("lease") if isinstance(stop.get("lease"), dict) else {} + stop["settled"] = { + "at": time.time(), **facts, + "lease_released": lease.get("released"), + "turn_journal_status": self._turn_journal_status(row, binding), + } + _write(self._stop_path(path), stop) + return self._stop_receipt(row, binding, stop) def _cli(self, binding: dict, *args: str, timeout: int = 60) -> dict: completed = subprocess.run([*_python_module_command("loopx.cli"), @@ -711,21 +1054,34 @@ def execute(self, operation_id: str) -> None: # The existing bounded mutation policy still excludes concurrent workers. with exclusive_file_lock(path): row = _read(path) - if row["status"] in {"accepted", "rejected"}: + if row["status"] in DELEGATION_TERMINAL_STATUSES: return - binding = self._bound(row, require_active=True) - row.pop("error", None) - _write(path, row) - # Different request ids cannot run the same assigned task concurrently. - task_lock = _root(self.root) / "execution-slots" / _hash([self.goal_id, binding["todo_id"]]) - with exclusive_file_lock(task_lock, policy=LockAcquisitionPolicy.SINGLE_FLIGHT): - try: - self._execute(path, row, binding) - except (ValueError, KeyError, subprocess.TimeoutExpired, EffectRuntimeRemoteError) as exc: - row["error"] = str(exc)[:180] if isinstance(exc, (ValueError, EffectRuntimeRemoteError)) else type(exc).__name__ - _write(path, row) - if row["status"] == "prepared": - self._observe(path, row, "rejected") + binding = self._bound(row) + if self._read_stop(path) is not None: + # The stop arrived before any worker owned the operation: this + # holder acknowledges it from under the lock and launches nothing. + self._acknowledge_stop(path, row, binding, source="worker_entry") + return + try: + binding = self._bound(row, require_active=True) + row.pop("error", None) + row["worker"] = self._worker_identity() + self._fenced_write(path, row) + # Different request ids cannot run the same assigned task concurrently. + task_lock = _root(self.root) / "execution-slots" / _hash([self.goal_id, binding["todo_id"]]) + with exclusive_file_lock(task_lock, policy=LockAcquisitionPolicy.SINGLE_FLIGHT): + try: + self._execute(path, row, binding) + except (ValueError, KeyError, subprocess.TimeoutExpired, EffectRuntimeRemoteError) as exc: + row["error"] = str(exc)[:180] if isinstance(exc, (ValueError, EffectRuntimeRemoteError)) else type(exc).__name__ + self._fenced_write(path, row) + if row["status"] == "prepared": + self._observe(path, row, "rejected") + except DelegationStopRequested as stop: + # SIGTERM, a checkpoint or a fenced write: the host child is already + # gone (its run-once exits with this process's exception), the + # bootstrap was cleared, and only the acknowledgement remains. + self._acknowledge_stop(path, row, binding, source=stop.source) def _execution_arguments(self, binding: dict, operation_id: str) -> list[str]: """Exactly the same profile, workspace and validation arguments for preview/run.""" @@ -790,7 +1146,7 @@ def _record_turn_result( if publish: self._observe(path, row, "turn_returned") else: - _write(path, row) + self._fenced_write(path, row) def _receiver_adopted(self, row: dict, binding: dict) -> bool: request_id = row["identity"]["request_id"] @@ -892,7 +1248,7 @@ def _acquire_delegation_lease( goal_id=self.goal_id, ): row["task_lease"] = {"required": False, "handoff_mode": "legacy"} - _write(path, row) + self._fenced_write(path, row) return row["task_lease"] handoff_mode = show_goal_handoff_mode( registry_path=self.registry, @@ -904,7 +1260,7 @@ def _acquire_delegation_lease( "required": False, "handoff_mode": handoff_mode, } - _write(path, row) + self._fenced_write(path, row) return row["task_lease"] lease_key = self._turn_instance_id(row) result = self._cli( @@ -946,7 +1302,7 @@ def _acquire_delegation_lease( "idempotency_key": lease_key, "version": lease["version"], } - _write(path, row) + self._fenced_write(path, row) return row["task_lease"] def _complete_delegated_todo(self, row: dict, binding: dict) -> None: @@ -991,6 +1347,7 @@ def _execute(self, path: Path, row: dict, binding: dict) -> None: common = ["--goal-id", self.goal_id, "--agent-id", binding["agent_id"]] execution = self._execution_arguments(binding, row["identity"]["operation_id"]) try: + self._raise_if_stop_requested(path) if row["status"] == "prepared": acceptance = delegation_validation.capture(self, binding) if acceptance["plan"]["state"] != "ready" or not acceptance["files_current"]: @@ -1000,6 +1357,7 @@ def _execute(self, path: Path, row: dict, binding: dict) -> None: self._acquire_delegation_lease(path, row, binding) self._observe(path, row, "running") if row["status"] == "running": + self._raise_if_stop_requested(path) turn_key = self._matching_turn_key(row, binding) selector = ( ["--resume-turn-key", turn_key] @@ -1042,8 +1400,10 @@ def _execute(self, path: Path, row: dict, binding: dict) -> None: ) if not isinstance(row.get("task_lease"), dict): self._acquire_delegation_lease(path, row, binding) + self._raise_if_stop_requested(path) self._complete_delegated_todo(row, binding) todo_completed_for_settlement = True + self._raise_if_stop_requested(path) result = self._cli( binding, "turn", @@ -1071,6 +1431,7 @@ def _execute(self, path: Path, row: dict, binding: dict) -> None: self._bound(row, require_active=True) # revocation or rebinding while the model ran delegation_results.require_dependencies(self, binding, delegation_results.operation_brief(self, row)) if not todo_completed_for_settlement: + self._raise_if_stop_requested(path) self._complete_delegated_todo(row, binding) row["artifacts"] = self._accepted(binding) if not (_root(self.root) / "replies" / request_id / "conclusion.json").exists(): @@ -1097,7 +1458,7 @@ def _execute(self, path: Path, row: dict, binding: dict) -> None: # Retain uncertain execution for explicit same-operation recovery. # No fresh Turn is ever created because its client timed out. row["error"] = str(exc)[:180] if isinstance(exc, (ValueError, EffectRuntimeRemoteError)) else type(exc).__name__ - _write(path, row) + self._fenced_write(path, row) def register_delegation_tools(server, delegations: Delegations) -> None: @@ -1185,10 +1546,13 @@ def main(): if args.delegation_action == "validate": service._validate(service._bound(_read(service.path(args.operation_id)))) else: + service._stop_signal = install_worker_stop_signal() try: service.execute(args.operation_id) except LockAcquireTimeoutError: pass # Another worker still owns the operation after the bounded wait. + except DelegationStopRequested: + pass # Stopped before owning the operation; the holder acknowledges. return if args.workspace is None: parser.error("--workspace is required when serving MCP") diff --git a/loopx/file_lock.py b/loopx/file_lock.py index 7ffa396822..29303c066e 100644 --- a/loopx/file_lock.py +++ b/loopx/file_lock.py @@ -360,6 +360,17 @@ def lock_holder_liveness(path: Path) -> tuple[str, dict[str, object]]: return (LOCK_HOLDER_LIVE if process_is_alive(pid) else LOCK_HOLDER_DEAD), record +def read_lock_holder(path: Path) -> dict[str, object]: + """Read the advisory holder record behind ``path``'s lock; ``{}`` when absent. + + The record names the last process that held the lock and carries + ``released_at`` after a clean release. It is advisory readback for signals + and operator inspection; the kernel lock stays the only proof of holding. + """ + + return _read_holder_record(lock_holder_path(path)) + + def _operator_action(holder: dict[str, object], *, retry_mode: str) -> dict[str, object]: return { "required": True, From 654c1fbf0063505132eea0d04c4be1abbc66e39d Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 06:29:48 -0400 Subject: [PATCH 04/54] feat(delegation): expose stop through the CLI and MCP surfaces Add `loopx delegation stop --operation-id ID --execute` next to start/resume/adopt (--execute is required for the same reason) and the `stop_delegation(operation_id)` MCP tool, which runs the blocking stop off the event loop like wait_delegation. Both return the same receipt and never resume or rerun work. The inventory reader skips `.stop.json` sidecars, which sit beside execution records but are not records, and the delegation context and subagent context projections count `stopped` receipts instead of dropping them. Co-Authored-By: Claude Fable 5.1 Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- loopx/cli_commands/delegation.py | 12 +++++++----- loopx/collaboration_mcp.py | 12 ++++++++++++ .../collaboration/delegation_context.py | 1 + .../collaboration/delegation_inventory.py | 4 ++-- loopx/control_plane/subagent_context.ts | 2 +- 5 files changed, 23 insertions(+), 8 deletions(-) diff --git a/loopx/cli_commands/delegation.py b/loopx/cli_commands/delegation.py index 9d2fb72a97..e422e68c64 100644 --- a/loopx/cli_commands/delegation.py +++ b/loopx/cli_commands/delegation.py @@ -22,7 +22,7 @@ def register_delegation( "delegation", help="Launch and recover authorized peer work; returns JSON." ) add_format(parser) - parser.add_argument("delegation_action", choices=("list", "operations", "inspect", "start", "read", "wait", "resume", "adopt")) + parser.add_argument("delegation_action", choices=("list", "operations", "inspect", "start", "read", "wait", "resume", "adopt", "stop")) parser.add_argument("--goal-id", required=True) parser.add_argument("--agent-id", required=True, help="Calling registered Agent, not the worker.") parser.add_argument("--execution-config", type=Path, required=True, @@ -34,7 +34,7 @@ def register_delegation( parser.add_argument("--parent-request-id", help="For start: the request received by this coordinator.") parser.add_argument("--limit", type=int, help="For operations: page size, 1–50 (default 20).") parser.add_argument("--cursor", help="For operations: next_cursor returned by the previous page.") - parser.add_argument("--execute", action="store_true", help="Required for start/resume/adopt; grants no additional authority.") + parser.add_argument("--execute", action="store_true", help="Required for start/resume/adopt/stop; grants no additional authority.") def handle_delegation( @@ -45,10 +45,10 @@ def handle_delegation( action = args.delegation_action try: - if action in {"start", "resume", "adopt"} and not args.execute: + if action in {"start", "resume", "adopt", "stop"} and not args.execute: raise ValueError(f"delegation {action} requires --execute") - if action not in {"start", "resume", "adopt"} and args.execute: - raise ValueError("--execute is only valid for start/resume/adopt") + if action not in {"start", "resume", "adopt", "stop"} and args.execute: + raise ValueError("--execute is only valid for start/resume/adopt/stop") if action not in {"list", "operations", "inspect"} and not args.operation_id: raise ValueError(f"delegation {action} requires --operation-id") if action in {"list", "operations", "inspect"} and args.operation_id: @@ -86,6 +86,8 @@ def handle_delegation( result = service.read(args.operation_id) elif action == "wait": result = service.wait(args.operation_id) + elif action == "stop": + result = service.stop(args.operation_id, execute=args.execute) else: result = service.resume(args.operation_id) payload = {"ok": True, **result} diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index 64e192ff00..0070601fb3 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -1525,6 +1525,18 @@ def resume_delegation(operation_id: str) -> dict: """Reconnect an interrupted original execution; never launch a replacement Turn.""" return delegations.resume(operation_id) + @server.tool() + async def stop_delegation(operation_id: str) -> dict: + """Stop one original operation and return what was proven, not what was hoped. + + settled: the worker acknowledged and its operation and Turn lane locks are + free. acknowledged/requested: still winding down; call again. unknown: the + holder vanished before acknowledging; inspect its Turn before reusing the + task. noop: already accepted/rejected/stopped. Stopped work is not resumed; + a new scope needs a new operation id. Elapsed time is never a receipt. + """ + return await asyncio.to_thread(delegations.stop, operation_id, execute=True) + def main(): parser = argparse.ArgumentParser(description=__doc__) diff --git a/loopx/control_plane/collaboration/delegation_context.py b/loopx/control_plane/collaboration/delegation_context.py index 1886379374..187297b6bb 100644 --- a/loopx/control_plane/collaboration/delegation_context.py +++ b/loopx/control_plane/collaboration/delegation_context.py @@ -152,6 +152,7 @@ def project_delegation_context( "turn_returned", "accepted", "rejected", + "stopped", "unavailable", ) if statuses[key] diff --git a/loopx/control_plane/collaboration/delegation_inventory.py b/loopx/control_plane/collaboration/delegation_inventory.py index 66bd66a9f2..f070107f80 100644 --- a/loopx/control_plane/collaboration/delegation_inventory.py +++ b/loopx/control_plane/collaboration/delegation_inventory.py @@ -24,8 +24,8 @@ def addresses(): try: entries = directory.iterdir() for path in entries: - if path.suffix != ".json": - continue + if path.suffix != ".json" or path.name.endswith(".stop.json"): + continue # stop receipts sit beside their execution record if not BARE_SHA256_PATTERN.fullmatch(path.stem): raise ValueError("unexpected delegation record address; reconcile inventory storage") if query["cursor"] is None or path.stem > query["cursor"]: diff --git a/loopx/control_plane/subagent_context.ts b/loopx/control_plane/subagent_context.ts index 56f86e81ff..c21fadd14c 100644 --- a/loopx/control_plane/subagent_context.ts +++ b/loopx/control_plane/subagent_context.ts @@ -174,7 +174,7 @@ function boundedDelegationContext(value: unknown): JsonObject | null { const operationReceipts: JsonObject = {}; if (rawReceipts) { for (const key of ["observed", "prepared", "running", "turn_returned", "accepted", - "rejected", "unavailable", "recovery_required"]) { + "rejected", "stopped", "unavailable", "recovery_required"]) { if (Number.isInteger(rawReceipts[key]) && Number(rawReceipts[key]) >= 0) { operationReceipts[key] = Math.min(Number(rawReceipts[key]), 10_000); } From f9356e5a98353d8eecbf061d3303645f1bc0d666 Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 06:29:48 -0400 Subject: [PATCH 05/54] test(delegation): cover stop receipts, fencing and refused resume Semantics-first coverage for stopping delegated members: - stop while a detached worker runs a sleeping fixture host: the worker acknowledges SIGTERM under its own lock, the receipt settles only with a free operation lock and a lockable Turn lane, the record bytes stay frozen afterwards, the Todo stays open, the host and worker processes are gone, the Turn journal stays in_progress, and resume is refused without spawning; - stop after accepted: noop, identical on repeat, no stop file and artifacts unchanged; stop without a holder is acknowledged by the requester; - a worker SIGKILLed before acknowledging settles as unknown, not stopped, and resume stays refused; - a fenced write after another process's acknowledged stop raises and writes nothing, while an unacknowledged stop is taken from under the lock at entry; - CLI stop requires --execute and repeats its receipt; inventory pages past stop sidecars and reads a stopped record as stopped; - TS: stopped is terminal and reachable only from open observations, and the stop decision settles only on acknowledgement plus free locks. Co-Authored-By: Claude Fable 5.1 Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- tests/control_plane_ts/delegation.test.ts | 39 ++++- tests/test_delegation_cli.py | 28 ++++ tests/test_delegation_inventory.py | 7 + tests/test_local_delegation.py | 172 +++++++++++++++++++++- 4 files changed, 242 insertions(+), 4 deletions(-) diff --git a/tests/control_plane_ts/delegation.test.ts b/tests/control_plane_ts/delegation.test.ts index 7ae9fe4d41..0aaf6686ed 100644 --- a/tests/control_plane_ts/delegation.test.ts +++ b/tests/control_plane_ts/delegation.test.ts @@ -1,6 +1,6 @@ import test from "node:test"; import assert from "node:assert/strict"; -import {recordDelegationAdoption, delegationInventoryItem, delegationInventoryQuery, delegationPreflight, delegationTurnPlanDecision, delegationValidationPlan, recoverValidatedDelegationSettlement, selectDelegationBinding, transitionDelegationObservation} from "../../loopx/control_plane/collaboration/delegation.ts"; +import {recordDelegationAdoption, decideDelegationStop, delegationInventoryItem, delegationInventoryQuery, delegationPreflight, delegationTurnPlanDecision, delegationValidationPlan, recoverValidatedDelegationSettlement, selectDelegationBinding, transitionDelegationObservation} from "../../loopx/control_plane/collaboration/delegation.ts"; import {canonicalAuthoritySha256} from "../../loopx/control_plane/coordination/authority_store_codec.ts"; import {projectTurnSelectionRejection} from "../../loopx/control_plane/turn_driver/selection_rejection.ts"; @@ -102,6 +102,43 @@ test("message receipt and model return do not imply accepted work", () => { canonical_done: true, acceptance_ready: true, artifacts_current: true}), {status: "accepted"}); }); +test("stopped is terminal and reachable only from open observations", () => { + for (const from of ["prepared", "running", "turn_returned"]) + assert.deepEqual(transitionDelegationObservation({from, to: "stopped"}), {status: "stopped"}); + assert.deepEqual(transitionDelegationObservation({from: "stopped", to: "stopped"}), {status: "stopped"}); + for (const from of ["accepted", "rejected"]) + assert.throws(() => transitionDelegationObservation({from, to: "stopped"}), /transition/); + for (const to of ["running", "turn_returned", "accepted", "rejected"]) + assert.throws(() => transitionDelegationObservation({from: "stopped", to}), /transition/); + const observation = {operation_id: "op-1", request_id: "req", agent_id: "reviewer", todo_id: "todo_review", + status: "stopped", worker_active: false, recovery_required: false}; + assert.equal(delegationInventoryItem({record: {record_id: "a".repeat(64), operation_id: "op-1"}, + observation}).status, "stopped"); +}); + +test("a stop settles only on an acknowledgement plus free locks; time alone proves nothing", () => { + const open = {phase: "requested", acknowledged: false, operation_lock_free: false, lane_lock_free: false}; + assert.deepEqual(decideDelegationStop(open), {phase: "requested", terminal: false, reason: "awaiting_acknowledgement"}); + assert.deepEqual(decideDelegationStop({...open, timed_out: true}), + {phase: "requested", terminal: false, reason: "holder_still_running_after_grace"}); + assert.deepEqual(decideDelegationStop({...open, operation_lock_free: true, timed_out: true}), + {phase: "requested", terminal: false, reason: "holder_still_running_after_grace"}); + assert.deepEqual(decideDelegationStop({...open, operation_lock_free: true, lane_lock_free: true}), + {phase: "unknown", terminal: true, reason: "holder_gone_without_acknowledgement"}); + const acked = {phase: "acknowledged", acknowledged: true, operation_lock_free: false, lane_lock_free: false}; + assert.deepEqual(decideDelegationStop(acked), {phase: "acknowledged", terminal: false, reason: "operation_lock_still_held"}); + assert.deepEqual(decideDelegationStop({...acked, operation_lock_free: true}), + {phase: "acknowledged", terminal: false, reason: "turn_lane_still_held"}); + assert.deepEqual(decideDelegationStop({...acked, phase: "requested", operation_lock_free: true, lane_lock_free: true}), + {phase: "settled", terminal: true, reason: "acknowledged_and_locks_released"}); + assert.deepEqual(decideDelegationStop({...acked, operation_lock_free: true, lane_lock_free: true, timed_out: true}), + {phase: "settled", terminal: true, reason: "acknowledged_and_locks_released"}); + for (const patch of [{phase: "settled"}, {phase: "unknown"}, {phase: "noop"}, {acknowledged: "yes"}, + {operation_lock_free: 1}, {lane_lock_free: undefined}, {timed_out: "later"}, + {phase: "acknowledged", acknowledged: false}]) + assert.throws(() => decideDelegationStop({...open, ...patch})); +}); + test("a false rejection can reopen only for exact validated settlement recovery", () => { const evidence = { from: "rejected", diff --git a/tests/test_delegation_cli.py b/tests/test_delegation_cli.py index ee751375ae..95355fbbf7 100644 --- a/tests/test_delegation_cli.py +++ b/tests/test_delegation_cli.py @@ -84,6 +84,34 @@ def test_attached_cli_disconnect_retry_and_verified_return(service): assert "artifacts" not in inventory["items"][0] +def test_cli_stop_settles_a_running_member_and_refuses_resume(service): + root, runner = service + (root / "hold").touch() + source = root / "brief.json" + source.write_text(json.dumps(brief())) + status, started = cli(runner, "start", "--binding-id", "analysis", "--operation-id", "cli-stop", + "--brief-file", str(source), "--execute") + assert status == 0, started + deadline = time.monotonic() + 45 + while not (root / "host-started").exists() and time.monotonic() < deadline: + time.sleep(0.1) + assert (root / "host-started").exists() + status, refused = cli(runner, "stop", "--operation-id", "cli-stop") + assert status == 1 and "--execute" in refused["error"] + assert not runner._stop_path(runner.path("cli-stop")).exists() + status, stopped = cli(runner, "stop", "--operation-id", "cli-stop", "--execute") + assert status == 0 and stopped["phase"] == "settled" and stopped["status"] == "stopped", stopped + assert stopped["stop"]["ack"]["source"] == "SIGTERM" + status, again = cli(runner, "stop", "--operation-id", "cli-stop", "--execute") + assert status == 0 and again == stopped + status, resumed = cli(runner, "resume", "--operation-id", "cli-stop", "--execute") + assert status == 1 and "start a new operation id" in resumed["error"] + status, observed = cli(runner, "read", "--operation-id", "cli-stop") + assert status == 0 and observed["status"] == "stopped" and observed["stop"]["phase"] == "settled" + assert (root / "analyst" / "initial" / "host-invocations").read_text() == "1" + assert not demo.canonical_tasks(root)["todo_analyst-initial"]["done"] + + def test_cli_invalid_inputs_do_not_launch_work(service): root, runner = service bad = root / "bad.json" diff --git a/tests/test_delegation_inventory.py b/tests/test_delegation_inventory.py index 8486eb4a55..b75dec65c9 100644 --- a/tests/test_delegation_inventory.py +++ b/tests/test_delegation_inventory.py @@ -56,6 +56,13 @@ def test_corruption_and_stopped_worker_do_not_hide_healthy_sibling(service, monk assert by_id["stopped"]["recovery_required"] assert len(page["items"]) == 4 and not page["page_readback_complete"] assert sum(row["status"] == "unavailable" for row in page["items"]) == 2 + # A stop receipt beside its record is not another record and reads back as stopped. + assert runner.stop("healthy", execute=True)["phase"] == "settled" + assert runner._stop_path(runner.path("healthy")).exists() + page = runner.operations() + by_id = {row["operation_id"]: row for row in page["items"] if row["operation_id"]} + assert len(page["items"]) == 4 and by_id["healthy"]["status"] == "stopped" + assert not by_id["healthy"]["recovery_required"] def test_unknown_requester_and_unreadable_source_are_not_empty_inventory(service, monkeypatch): diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index d29b13fd7e..faf0208698 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -1,6 +1,7 @@ """Production delegation/Turn/TS completion with an explicit fixture model host.""" import json import asyncio +import os from pathlib import Path import subprocess import sys @@ -16,13 +17,13 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "examples" / "managed-research-team")) import research_team as demo # noqa: E402 from test_managed_research_scenario import fixture # noqa: E402 -from loopx.collaboration_mcp import Delegations # noqa: E402 +from loopx.collaboration_mcp import DelegationFenced, Delegations # noqa: E402 from loopx.control_plane.collaboration.peers import returns # noqa: E402 from loopx.control_plane.collaboration.inbox import _read # noqa: E402 -from loopx.file_lock import exclusive_file_lock # noqa: E402 +from loopx.file_lock import exclusive_file_lock, try_exclusive_file_lock # noqa: E402 -HOST = '''import json, sys, time +HOST = '''import json, os, sys, time from pathlib import Path from loopx.control_plane.turn_driver.host_candidate import build_result from loopx.control_plane.collaboration.inbox import acknowledge @@ -35,6 +36,7 @@ counter = workspace / 'host-invocations' counter.write_text(str(int(counter.read_text()) + 1 if counter.exists() else 1)) if (root / 'hold').exists(): + (root / 'host-pid').write_text(str(os.getpid())) (root / 'host-started').touch() while not (root / 'release').exists(): time.sleep(0.1) delegation = json.loads((workspace / 'DELEGATION.json').read_text()) @@ -169,6 +171,7 @@ async def disconnect_requester(): assert not (root / "host-started").exists() inventory = await session.call_tool("list_delegations", {}) assert not inventory.isError and json.loads(inventory.content[0].text)["items"] == [] + assert "stop_delegation" in {tool.name for tool in (await session.list_tools()).tools} result = await session.call_tool("start_delegation", { "binding_id": "analysis", "operation_id": "analysis-1", "brief": brief()}) assert not result.isError @@ -197,6 +200,13 @@ async def disconnect_requester(): assert len(returned) == 1 assert returned[0]["decision"] == "adopt" assert wait(reconnected)["artifacts"] == result["artifacts"] + # Accepted work cannot be stopped: nothing is written and the receipt repeats exactly. + noop = reconnected.stop("analysis-1", execute=True) + assert noop["phase"] == "noop" and noop["status"] == "accepted" and noop["stop"] is None + assert reconnected.stop("analysis-1", execute=True) == noop + assert not reconnected._stop_path(reconnected.path("analysis-1")).exists() + assert wait(reconnected)["artifacts"] == result["artifacts"] + assert "stop" not in reconnected.read("analysis-1") changed_brief = {**brief(), "purpose": "Changed instruction"} with pytest.raises(ValueError, match="identity conflict"): reconnected.start("analysis", "analysis-1", changed_brief) @@ -265,3 +275,159 @@ def read_on_publish(path, row, status, **facts): assert len(terminal_reads) == 1 assert not demo.canonical_tasks(root)["todo_analyst-initial"]["done"] assert returns(runner.root, runner.goal_id, "lead")["items"] == [] + + +def until(predicate, timeout=45): + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + if predicate(): + return True + time.sleep(0.1) + return predicate() + + +def process_gone(pid): + try: + os.kill(pid, 0) + except ProcessLookupError: + return True + try: + return "State:\tZ" in Path(f"/proc/{pid}/status").read_text() + except OSError: + return True + + +def start_held_worker(service, operation="analysis-stop"): + root, runner = service + (root / "hold").touch() + runner.start("analysis", operation, brief()) + assert until(lambda: (root / "host-started").exists()), _read(runner.path(operation)) + return int((root / "host-pid").read_text()) + + +def test_stop_while_executing_is_acknowledged_by_the_worker_and_settles(service, monkeypatch): + """The detached worker acknowledges SIGTERM under its own lock; time proves nothing.""" + root, runner = service + host_pid = start_held_worker(service) + path = runner.path("analysis-stop") + before = _read(path) + assert before["status"] == "running" and before["worker"]["pid"] == before["worker"]["pgid"] + receipt = runner.stop("analysis-stop", execute=True) + assert receipt["phase"] == "settled" and receipt["status"] == "stopped", receipt + stop = receipt["stop"] + assert stop["requested_by"] == "lead" and stop["requested_status"] == "running" + assert stop["worker"]["pid"] == before["worker"]["pid"] + assert stop["ack"]["pid"] == before["worker"]["pid"] and stop["ack"]["source"] == "SIGTERM" + assert stop["ack"]["observed_status"] == "running" and stop["ack"]["turn_key"] + assert stop["settled"]["operation_lock_free"] and stop["settled"]["lane_lock_free"] + assert stop["settled"]["turn_journal_status"] == "in_progress" + assert stop["lease"] == {"required": False, "released": None} + # The acknowledged record is final: nobody writes it again, the Todo stays open, + # the host process group is gone and the member's Turn lane can be taken. + frozen = path.read_bytes() + assert until(lambda: process_gone(host_pid), timeout=20) + assert until(lambda: process_gone(before["worker"]["pid"]), timeout=20) + binding = runner.binding("analysis") + with try_exclusive_file_lock(runner._lane_target(binding)) as held: + assert held is not None + assert not demo.canonical_tasks(root)["todo_analyst-initial"]["done"] + assert not (root / "analyst" / "initial" / "DELEGATION.json").exists() + assert path.read_bytes() == frozen + assert runner.stop("analysis-stop", execute=True) == receipt + observed = runner.read("analysis-stop") + assert observed["status"] == "stopped" and not observed["recovery_required"] + assert observed["stop"] == {"stop_id": stop["stop_id"], "phase": "settled"} + assert runner.wait("analysis-stop")["status"] == "stopped" + monkeypatch.setattr(runner, "_spawn", lambda _: pytest.fail("stopped work must not respawn")) + with pytest.raises(ValueError, match="start a new operation id"): + runner.resume("analysis-stop") + assert path.read_bytes() == frozen + page = runner.operations() + assert page["page_readback_complete"] and page["items"][0]["status"] == "stopped" + assert returns(runner.root, runner.goal_id, "lead")["items"] == [] + + +def test_stop_without_a_holder_is_acknowledged_by_the_requester(service, monkeypatch): + root, runner = service + monkeypatch.setattr(runner, "_spawn", lambda _: None) + runner.start("analysis", "analysis-idle", brief()) + receipt = runner.stop("analysis-idle", execute=True) + assert receipt["phase"] == "settled" and receipt["status"] == "stopped" + assert receipt["stop"]["worker"] is None and receipt["stop"]["requested_status"] == "prepared" + assert receipt["stop"]["ack"]["pid"] == os.getpid() and receipt["stop"]["ack"]["source"] == "requester" + assert receipt["stop"]["settled"]["turn_journal_status"] is None + frozen = runner.path("analysis-idle").read_bytes() + with pytest.raises(ValueError, match="start a new operation id"): + runner.resume("analysis-idle") + runner.execute("analysis-idle") # a late worker finds terminal work and launches nothing + assert runner.path("analysis-idle").read_bytes() == frozen + assert not (root / "host-started").exists() + assert runner.stop("analysis-idle", execute=True) == receipt + with pytest.raises(ValueError, match="requires execute"): + runner.stop("analysis-idle", execute=False) + with pytest.raises(ValueError, match="unknown delegation operation"): + runner.stop("never-started", execute=True) + + +def test_worker_killed_before_acknowledging_is_unknown_not_settled(service, monkeypatch): + """A vanished holder never becomes a settlement; the stop still fences resume.""" + import signal + + root, runner = service + host_pid = start_held_worker(service) + path = runner.path("analysis-stop") + worker = _read(path)["worker"] + + def kill_without_grace(target, stop): + assert stop["worker"]["pgid"] == worker["pgid"] != os.getpgid(0) + os.killpg(worker["pgid"], signal.SIGKILL) + assert until(lambda: runner._operation_lock_free(target), timeout=20) + + monkeypatch.setattr(runner, "_signal_worker", kill_without_grace) + receipt = runner.stop("analysis-stop", execute=True) + assert receipt["phase"] == "unknown" and receipt["status"] == "running", receipt + assert receipt["stop"]["ack"] is None and receipt["stop"]["settled"]["operation_lock_free"] + assert receipt["stop"]["settled"]["turn_journal_status"] == "in_progress" + assert until(lambda: process_gone(host_pid), timeout=20) + assert runner.stop("analysis-stop", execute=True) == receipt + with pytest.raises(ValueError, match="start a new operation id"): + runner.resume("analysis-stop") + assert runner.read("analysis-stop")["stop"]["phase"] == "unknown" + assert not demo.canonical_tasks(root)["todo_analyst-initial"]["done"] + + +def test_fenced_write_after_another_process_stop_writes_nothing(service, monkeypatch): + root, runner = service + monkeypatch.setattr(runner, "_spawn", lambda _: None) + runner.start("analysis", "analysis-fenced", brief()) + path = runner.path("analysis-fenced") + row = _read(path) + foreign = runner._new_stop_record(row, requested_by="other-host-lead", worker=None) + foreign.update(phase="acknowledged", ack={"pid": 1, "host": "elsewhere", "at": 0.0, + "source": "requester", "observed_status": "running", + "turn_key": None}) + from loopx.control_plane.collaboration.inbox import _write + + _write(runner._stop_path(path), foreign) + frozen = path.read_bytes() + with pytest.raises(DelegationFenced): + runner._fenced_write(path, {**row, "status": "running"}) + with pytest.raises(DelegationFenced): + runner._observe(path, dict(row), "running") + with pytest.raises(DelegationFenced): + runner._record_turn_result(path, {**row, "status": "running"}, + {"status": "committed", "result_kind": "validated_progress"}) + runner.execute("analysis-fenced") # the foreign acknowledgement stands; nothing is rewritten + assert path.read_bytes() == frozen + assert _read(runner._stop_path(path)) == foreign + assert not (root / "host-started").exists() + assert not demo.canonical_tasks(root)["todo_analyst-initial"]["done"] + # A stop this process may acknowledge is taken from under the lock at entry. + runner.start("analysis", "analysis-entry", brief()) + entry = runner.path("analysis-entry") + _write(runner._stop_path(entry), runner._new_stop_record(_read(entry), requested_by="lead", worker=None)) + runner.execute("analysis-entry") + assert _read(entry)["status"] == "stopped" + acknowledged = _read(runner._stop_path(entry)) + assert acknowledged["phase"] == "acknowledged" and acknowledged["ack"]["source"] == "worker_entry" + assert runner.stop("analysis-entry", execute=True)["phase"] == "settled" From f2553451d40aeb5424e6e859415de2a796eae236 Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 06:29:48 -0400 Subject: [PATCH 06/54] docs(delegation): document stopping a member and reading its receipt Describe `delegation stop --execute` / `stop_delegation` in both reference documents, in English and Chinese: where the request lives, how a same-host worker is signalled and acknowledges, why another host's worker is left to find the request, what settled, unknown and noop prove, and that stopped work needs a new operation id while its Turn journal and open Todo remain for the coordinator to inspect. Co-Authored-By: Claude Fable 5.1 Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- docs/reference/goal-chat-continuation.md | 8 +++- docs/reference/local-delegation.md | 47 ++++++++++++++++++++++-- 2 files changed, 50 insertions(+), 5 deletions(-) diff --git a/docs/reference/goal-chat-continuation.md b/docs/reference/goal-chat-continuation.md index 00569ce34a..7c367d68d3 100644 --- a/docs/reference/goal-chat-continuation.md +++ b/docs/reference/goal-chat-continuation.md @@ -100,7 +100,9 @@ messages; images use the ordinary conversation after pausing. The Chat hard timeout remains in force; this is not an unattended daemon. - To roll back, pause/close the Chat service before installing an older build. Disabling mode or deleting a binding does not cancel already admitted children; - use their own execution/recovery controls and retain their evidence. + stop one with `loopx delegation stop --execute` (or `stop_delegation`), read + its receipt, and retain its evidence. Only `settled` proves the worker + acknowledged and released its locks; stopped work needs a new operation id. The ordinary native command path also remains available without delegation: @@ -143,6 +145,8 @@ For a disposable mixed-team setup, use the 协调员保留只读沙箱,成员权限来自各自执行绑定,不继承管家的扩大权限。 成员通过验收与协调员报告、整个 Goal 验收分别显示;本模式不直接完成报告 Todo 或整个 Goal。额度是含历史用量的总量,正在执行的请求可能超额,成员另行计量。 -回滚旧版本前先暂停或关闭 Chat 服务;退出或撤销绑定不自动取消已启动的成员。 +回滚旧版本前先暂停或关闭 Chat 服务;退出或撤销绑定不自动取消已启动的成员, +用 `loopx delegation stop --execute`(或 `stop_delegation`)停止单个成员并阅读回执: +只有 `settled` 证明 worker 已确认并释放锁;已停止的工作需要新的 operation id。 此按钮目前限本机 managed Codex Goal 对话,不宣称 Lark、挂接会话或其他主力 驱动等价。可用下方示例准备一次隔离的本地 DSH+云端 Ark 协作。 diff --git a/docs/reference/local-delegation.md b/docs/reference/local-delegation.md index 26015e161e..b5e3e39fa2 100644 --- a/docs/reference/local-delegation.md +++ b/docs/reference/local-delegation.md @@ -157,6 +157,7 @@ delegate start --binding-id independent-review --operation-id review-round-1 \ --brief-file request.json --execute delegate read --operation-id review-round-1 delegate wait --operation-id review-round-1 +delegate stop --operation-id review-round-1 --execute ``` Inspection uses the bound worker workspace as its actual safety scan root. If @@ -221,6 +222,38 @@ own authorized peers supplies `--parent-request-id` on start. CLI and MCP share grant validation, detached execution, wait/readback and recovery rather than maintaining separate rules. +`stop --execute` ends one member's bounded work and returns a receipt that +states what was proven. The request is written beside the execution record +(`.stop.json`), never into it, so a worker that is still holding the +operation cannot overwrite it. A worker on this machine receives `SIGTERM` for +its whole process group, which ends its Turn child and its host; it +acknowledges from under its own lock, marks the record `stopped` and releases +its hard task lease. When nobody holds the operation, the requester +acknowledges itself. A worker on another machine is never signalled; it finds +the request at its next checkpoint or at its next record write, which is +refused. The receipt `phase` is `settled` only when an acknowledgement exists +and both the operation lock and the member's Turn lane lock are free; +`unknown` means the holder vanished before acknowledging, and `noop` means the +work was already accepted, rejected or stopped. `requested` or `acknowledged` +means it is still winding down: call `stop` again. A grace timeout never turns +into a receipt. Stopped work is not resumed; `resume` refuses it and a new +scope needs a new operation id. The Turn journal keeps its `in_progress` entry +for inspection, and the record is never rewritten as a completion. A stopped +member's Todo stays open, so the coordinator decides what happens next. + +中文:`stop --execute` 结束一个成员的有界工作,并返回一份只陈述已证明事实的 +回执。停止请求写在执行记录旁边的 `.stop.json`,从不写进记录本身, +因此仍持有该 operation 的 worker 无法覆盖它。本机 worker 会收到整个进程组的 +`SIGTERM`,其 Turn 子进程和 host 一并结束;worker 在自己的锁下确认,把记录标为 +`stopped` 并释放硬任务租约。没有持有者时由请求方自行确认。另一台机器上的 +worker 不会被发信号,它在下一个检查点或下一次写记录时发现请求,写入被拒绝。 +只有存在确认且 operation 锁与成员 Turn lane 锁都已释放时,`phase` 才是 +`settled`;`unknown` 表示持有者在确认前消失;`noop` 表示工作已 accepted、 +rejected 或 stopped;`requested`/`acknowledged` 表示仍在收尾,再次调用 `stop`。 +宽限期超时永远不会变成回执。已停止的工作不能 `resume`,新范围需要新的 +operation id。Turn journal 保留 `in_progress` 条目供检查,记录不会被改写成完成; +成员的 Todo 仍然打开,由协调者决定下一步。 + This entrypoint does not create Agents, grant bindings or wake an idle Codex conversation. The existing host/LoopX continuation policy owns the next lead turn. The conversation remains persistent independently of whether autonomous @@ -574,7 +607,8 @@ operation and observed artifact hashes in its existing inbox. Pending and delive receipts stay distinct from application; a retry after an uncertain response reuses the exact message and operation id. **Pause coordinator** stays in the panel and reports its actual scope. Dispatched members continue independently; -this entrypoint cannot stop the whole team. Ordinary polling does not read artifact +this entrypoint cannot stop the whole team. Stop one member explicitly with +`delegation stop --execute` or `stop_delegation` and read its receipt. Ordinary polling does not read artifact bodies or run preflight. Closing the panel changes no work state. This local operator entrypoint does not grant a Lark audience access. @@ -584,7 +618,8 @@ operator entrypoint does not grant a Lark audience access. 可展开查看,返回列表保留位置与键盘焦点。协调员运行时,可把执行标识、看到的 产物哈希和反馈投递到原收件箱;等待投递、已交付和已应用不能混为一谈。不确定响应 后重试同一消息和标识,避免重复投递。面板内的「暂停协调员」显示实际反馈,但不会 -停止已派发成员,也不宣称整个团队停止。暂停时仍可检查证据;读取不启动模型。 +停止已派发成员,也不宣称整个团队停止。要停止某个成员,显式使用 +`delegation stop --execute` 或 `stop_delegation` 并阅读其回执。暂停时仍可检查证据;读取不启动模型。 Screenshots use isolated synthetic research data, not a live-model qualification: [desktop evidence](../assets/personal-workspace/team-evidence-desktop.png), @@ -614,6 +649,10 @@ unchanged and cannot launch workers. With it, the Agent can: response is normal. `read_delegation` reads the durable original operation. 4. If `recovery_required` is true, call `resume_delegation` with that same id. This cannot retarget the work or silently create a replacement Turn. +5. Call `stop_delegation(operation_id)` to end one member. Read its `phase`: + `settled` is the only receipt that the worker acknowledged and released its + locks; `unknown` means the holder vanished first; `noop` means the work had + already ended. Stopped work cannot be resumed; use a new operation id. Configure the member's host to expose its own identity-bound collaboration tools. It reads `DELEGATION.json`, independently calls `assess_request`, and @@ -654,6 +693,7 @@ concurrent executions still use the same kernel lock and original Turn journal. | Requesting MCP conversation closes | The detached bounded worker continues; another connection reads the original operation. | | Duplicate start/resume while work runs | Operation identity, task lock and Turn journal prevent another concurrent execution. | | Worker process or machine stops | Reconnect with the same operator configuration and credentials, then resume the original Turn. | +| Member stopped on request | The worker acknowledges under its lock, its host and Turn child are ended, its lease is released; `settled` needs that acknowledgement plus free locks, `unknown` means the holder vanished first. The record is `stopped`; resume refuses it. | | Ark is computing without local tools | The already-started cloud turn can continue. It is not dependent on the local conversation. | | Ark requests a local tool while the host is absent | It waits for the local tool result. Recovery observes the original input/session and executes only previously unstarted tool calls. | | Tool execution or send acknowledgement is uncertain | Do not repeat the effect. Preserve the receipt/session for explicit reconciliation. | @@ -669,7 +709,8 @@ or default executor change; those existing configuration surfaces are untouched. To disable new admission, remove the caller's grants or remove `--execution-config` from the host. A stopped Goal refuses new starts/resumes; existing completed results remain readable. Disabling does not kill work already -running. Retain receipts, stop or reconcile owned workers, and confirm cloud +running; `delegation stop --execute` ends one member and returns a receipt. +Retain receipts, stop or reconcile owned workers, and confirm cloud resource cleanup before deleting a disposable runtime. The optional adapter's cleanup command never grants task completion. From 9a4059676ab379c52cb450fa32b39f92b85dc6d3 Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 10:25:17 -0400 Subject: [PATCH 07/54] fix(delegation): settle a stop from holder records, never by taking a lock The stop settlement probed the member's Turn lane by acquiring its lock for an instant, which could refuse a legitimate Turn of the same member racing that instant with turn_lane_in_flight. It also carried its own host label helper next to the file-lock owner's. Settlement now reads the lane's last holder record through turn_lane_liveness: released, dead or absent frees the lane; a live same-host holder frees it only when it sits outside the recorded worker's process group; another host's holder, an unattributable holder or an unreadable record keep the stop open. The operation lock is read the same way, so a stop never refuses a legitimate resume or status read. The local host helper is deleted in favour of lock_holder_host_label. A regression test races a real lane acquisition against settlement and proves the Turn is admitted. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- loopx/collaboration_mcp.py | 202 +++++++++++++----- .../control_plane/collaboration/delegation.ts | 33 +-- loopx/file_lock.py | 11 - tests/control_plane_ts/delegation.test.ts | 30 +-- tests/test_local_delegation.py | 117 +++++++++- 5 files changed, 291 insertions(+), 102 deletions(-) diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index 0070601fb3..7f90c2f7ce 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -28,8 +28,8 @@ from mcp.server.fastmcp import FastMCP from .file_lock import ( - exclusive_file_lock, lock_holder_host_label, read_lock_holder, try_exclusive_file_lock, - LockAcquisitionPolicy, LockAcquireTimeoutError, + exclusive_file_lock, lock_holder_host_label, lock_holder_liveness, + LOCK_HOLDER_FOREIGN_HOST, LOCK_HOLDER_LIVE, LockAcquisitionPolicy, LockAcquireTimeoutError, ) from .control_plane.effect_runtime import ( effect_runtime_request_scope, effect_runtime_result, EffectRuntimeRemoteError, @@ -42,7 +42,10 @@ turn_journal_path, ) from .control_plane.turn_driver.host_binding import turn_host_arg_option -from .control_plane.turn_driver.lane_fence import turn_lane_target +from .control_plane.turn_driver.lane_fence import ( + TURN_LANE_ABSENT, TURN_LANE_DEAD, TURN_LANE_LIVE, TURN_LANE_RELEASED, + turn_lane_liveness, turn_lane_target, +) from .control_plane.work_items.task_lease import release_task_lease from .control_plane.collaboration.inbox import _hash, _read, _write, _root, _receipt from .control_plane.collaboration.peers import return_result @@ -310,13 +313,18 @@ def consume_peer_result(request_id: str) -> dict: # Observations that no worker may reopen; a stop against one is a no-op receipt. DELEGATION_TERMINAL_STATUSES = frozenset({"accepted", "rejected", "stopped"}) DELEGATION_STOP_OPEN_PHASES = frozenset({"requested", "acknowledged"}) +DELEGATION_STOP_TERMINAL_PHASES = frozenset({"settled", "unknown"}) # How long a signalled same-host worker may take to acknowledge before SIGKILL. DELEGATION_STOP_GRACE_SECONDS = 10.0 DELEGATION_STOPPED_MESSAGE = "delegation operation was stopped; start a new operation id" -class DelegationStopRequested(Exception): - """A stop reached the worker that owns this operation; it must acknowledge, not finish.""" +class DelegationStopRequested(BaseException): + """A stop reached the worker that owns this operation; it must acknowledge, not finish. + + A ``BaseException`` like ``KeyboardInterrupt``: a termination request must + not be swallowed by an ``except Exception`` and turned into further work. + """ def __init__(self, source: str) -> None: super().__init__(source) @@ -335,14 +343,28 @@ def __init__(self) -> None: class _WorkerStopSignal: - """Turn the first SIGTERM into a stop request; absorb later ones during the acknowledgement.""" + """Turn SIGTERM into a stop request only when a stop was written for this operation. - def __init__(self) -> None: + Without a stop receipt the signal keeps its default meaning, so a shutdown + still leaves the operation recoverable by ``resume`` instead of stopping it. + Later signals are absorbed while the acknowledgement is written. + """ + + def __init__(self, stop_path: Path) -> None: + self.stop_path = stop_path self.armed = True def __call__(self, signum: int, frame: object) -> None: if not self.armed: return + try: + requested = self.stop_path.exists() + except OSError: + requested = False + if not requested: + signal.signal(signum, signal.SIG_DFL) + os.kill(os.getpid(), signum) + return self.armed = False raise DelegationStopRequested("SIGTERM") @@ -350,12 +372,12 @@ def disarm(self) -> None: self.armed = False -def install_worker_stop_signal() -> _WorkerStopSignal | None: +def install_worker_stop_signal(stop_path: Path) -> _WorkerStopSignal | None: """Install the detached worker's SIGTERM handler; ``None`` where signals are unsupported.""" if not hasattr(signal, "SIGTERM"): return None - handler = _WorkerStopSignal() + handler = _WorkerStopSignal(stop_path) try: signal.signal(signal.SIGTERM, handler) except (ValueError, OSError): @@ -592,7 +614,10 @@ def resume(self, operation_id: str) -> dict: if row["status"] == "stopped": raise ValueError(DELEGATION_STOPPED_MESSAGE) if row["status"] == "rejected": - self._recover_validated_settlement(path, row, binding) + try: + self._recover_validated_settlement(path, row, binding) + except DelegationFenced: + raise ValueError(DELEGATION_STOPPED_MESSAGE) from None should_spawn = row["status"] not in DELEGATION_TERMINAL_STATUSES if should_spawn: self.binding( @@ -610,7 +635,8 @@ def wait(self, operation_id: str) -> dict: """Observe for at most 15 seconds; waiting neither starts nor resumes work.""" for _ in range(5): result = self.read(operation_id) - if result["status"] in DELEGATION_TERMINAL_STATUSES or result["recovery_required"]: + if (result["status"] in DELEGATION_TERMINAL_STATUSES or result["recovery_required"] + or result.get("stop", {}).get("phase") in DELEGATION_STOP_TERMINAL_PHASES): return result time.sleep(3) return self.read(operation_id) @@ -720,12 +746,13 @@ def _read_current(self, operation_id: str) -> dict: active = False except LockAcquireTimeoutError: active = True + # Resume refuses an operation with a stop receipt, so it never needs recovery. + stop = self._read_stop(path) result = {"operation_id": operation_id, "request_id": row["identity"]["request_id"], "agent_id": binding["agent_id"], "todo_id": binding["todo_id"], "status": row["status"], "worker_active": active, "recovery_required": not active and row["status"] not in DELEGATION_TERMINAL_STATUSES - and time.time() - row.get("created_at", 0) > 15} - stop = self._read_stop(path) + and stop is None and time.time() - row.get("created_at", 0) > 15} if stop is not None: result["stop"] = {"stop_id": stop["stop_id"], "phase": stop["phase"]} if row["status"] == "accepted": @@ -782,19 +809,59 @@ def _lane_target(self, binding: dict) -> Path: plan={"turn_envelope": {"agent_id": binding["agent_id"]}}) def _operation_lock_free(self, path: Path) -> bool: + """Probe this operation's own kernel lock; only ever called once its stop receipt exists. + + Unlike the Turn lane, this lock admits nothing but this operation, and a + probe holding it for an instant refuses no legitimate acquisition once + the receipt is written: ``resume``, its only single-flight acquirer, + refuses a stopped operation before it touches the lock; ``execute``, + adoption and the requester acknowledgement wait through brief holders + with the mutation policy; and a status read already makes this same + instant observation. Before the receipt exists a ``resume`` is still + legitimate, so the holder is then read from its record instead. + """ + try: with exclusive_file_lock(path, policy=LockAcquisitionPolicy.SINGLE_FLIGHT): return True except LockAcquireTimeoutError: return False - def _lane_lock_free(self, binding: dict) -> bool: - # The kernel lock is the only proof; the holder record is advisory. The - # probe holds the lane for an instant, which is the same observation a - # status read makes on the operation lock. - with try_exclusive_file_lock(self._lane_target(binding), agent_id=self.agent_id, - operation="loopx_delegation_stop_probe") as held: - return held is not None + @staticmethod + def _recorded_worker(row: dict, stop: dict) -> dict | None: + worker = row.get("worker") + if not isinstance(worker, dict): + worker = stop.get("worker") + return worker if isinstance(worker, dict) else None + + def _worker_lane_released(self, row: dict, stop: dict, binding: dict) -> tuple[bool, str]: + """Say whether the stopped worker's Turn has let go of the member's lane, read-only. + + This never takes the lane lock: a probe holding it for an instant would + refuse a legitimate Turn of the same member racing that instant with + ``turn_lane_in_flight``. The lane's last holder record decides instead. + Released, dead or absent is released. A live holder on this machine is + released only when it sits outside the recorded worker's process group, + because the worker's run-once child runs in that group; a holder that + cannot be attributed, another host's holder and an unreadable record + prove nothing, so the typed decision keeps the stop open. + """ + + lane = turn_lane_liveness(self._lane_target(binding)) + state = lane["state"] + if state in {TURN_LANE_RELEASED, TURN_LANE_DEAD, TURN_LANE_ABSENT}: + return True, state + worker = self._recorded_worker(row, stop) + if (state != TURN_LANE_LIVE or worker is None or not hasattr(os, "getpgid") + or worker.get("host") != lock_holder_host_label() + or not isinstance(worker.get("pgid"), int)): + return False, state + try: + return os.getpgid(lane["holder"]["pid"]) != worker["pgid"], state + except ProcessLookupError: + return True, TURN_LANE_DEAD # the holder exited between the two reads + except OSError: + return False, state def _turn_journal_status(self, row: dict, binding: dict) -> str | None: turn_key = row.get("turn_key") or self._matching_turn_key(row, binding) @@ -866,26 +933,38 @@ def stop(self, operation_id: str, *, execute: bool) -> dict: worker=self._lock_holder_worker(path, row)) _write(self._stop_path(path), stop) if stop["phase"] in DELEGATION_STOP_OPEN_PHASES and stop.get("ack") is None: - try: - with exclusive_file_lock(path, policy=LockAcquisitionPolicy.SINGLE_FLIGHT): - row = _read(path) - self._acknowledge_stop(path, row, binding, source="requester") - except LockAcquireTimeoutError: + if stop.get("worker") is None: + # No worker was named when the request was written, so whoever owns + # the operation acknowledges it: this caller once the lock is free. + # The wait rides out a status read's instant hold, which must not + # be mistaken for a holder that vanished. + try: + with exclusive_file_lock(path): + self._acknowledge_stop(path, _read(path), binding, source="requester") + except LockAcquireTimeoutError: + pass # an unnamed holder meets the request at its next checkpoint or write + else: + # Only the named worker acknowledges. If it vanishes first, the typed + # decision reports unknown instead of a requester settlement. self._signal_worker(path, stop) return self._settle_stop(path) def _lock_holder_worker(self, path: Path, row: dict) -> dict | None: - """Name the live operation-lock holder, else ``None``; a released record is not a worker.""" + """Name the recorded worker while it is the operation lock's unreleased holder. + + Read from the holder record, never the kernel lock: this runs before the + stop receipt exists, when a probe could refuse a legitimate ``resume``. + Only the worker identity the execution record names can become a signal + target, so a status reader's instant holder record is never taken for it. + """ - if self._operation_lock_free(path): + state, holder = lock_holder_liveness(path) + recorded = row.get("worker") + if state not in {LOCK_HOLDER_LIVE, LOCK_HOLDER_FOREIGN_HOST} or not isinstance(recorded, dict): return None - holder = read_lock_holder(path) - if "released_at" in holder or not isinstance(holder.get("pid"), int): + if holder.get("pid") != recorded.get("pid") or holder.get("host") != recorded.get("host"): return None - recorded = row.get("worker") if isinstance(row.get("worker"), dict) else {} - same = recorded.get("pid") == holder["pid"] and recorded.get("host") == holder.get("host") - pgid = recorded.get("pgid") if same and isinstance(recorded.get("pgid"), int) else holder["pid"] - return {"pid": holder["pid"], "pgid": pgid, "host": holder.get("host")} + return {key: recorded.get(key) for key in ("pid", "pgid", "host")} def _signal_worker(self, path: Path, stop: dict) -> None: """Terminate a same-host holder's process group; never signal across hosts.""" @@ -919,29 +998,38 @@ def _signal_worker(self, path: Path, stop: dict) -> None: def _acknowledge_stop(self, path: Path, row: dict, binding: dict, *, source: str) -> None: """Acknowledge from under the operation lock: mark stopped, then release the hard lease. - Only the lock holder may acknowledge. A stop that another process already - acknowledged, or that already settled, is left untouched. + Only the operation-lock holder calls this. The record is transitioned as + it is on disk, so state that a fenced write refused stays unwritten; the + lease is released from what this process acquired, which may be newer + than the record. A missing stop, one already acknowledged or finished, + and a record that already reached a terminal observation stay untouched. """ if self._stop_signal is not None: self._stop_signal.disarm() with exclusive_file_lock(self._dispatch_lock(path)): stop = self._read_stop(path) - if stop is None: - stop = self._new_stop_record(row, requested_by="signal:" + source, - worker=self._worker_identity()) - if stop.get("ack") is not None or stop["phase"] not in DELEGATION_STOP_OPEN_PHASES: + current = _read(path) + if (stop is None or stop.get("ack") is not None + or stop["phase"] not in DELEGATION_STOP_OPEN_PHASES + or current["status"] in DELEGATION_TERMINAL_STATUSES): return - observed = row["status"] - decision = effect_runtime_result("collaboration.delegation.observe", { + observed = current["status"] + transition = effect_runtime_result("collaboration.delegation.observe", { "from": observed, "to": "stopped", }) - row.update(status=decision["status"]) - _write(path, row) - stop.update(phase="acknowledged", reason="awaiting_lock_release", ack={ + # This process holds the operation lock and its lane read comes later. + phase = effect_runtime_result("collaboration.delegation.stop", { + "phase": stop["phase"], "acknowledged": True, + "operation_lock_free": False, "worker_lane_released": False, + }) + current["status"] = transition["status"] + _write(path, current) + stop.update(phase=phase["phase"], reason=phase["reason"], ack={ "pid": os.getpid(), "host": lock_holder_host_label(), "at": time.time(), "source": source, "observed_status": observed, - "turn_key": row.get("turn_key") or self._matching_turn_key(row, binding), + "turn_key": (current.get("turn_key") or row.get("turn_key") + or self._matching_turn_key(current, binding)), }) _write(self._stop_path(path), stop) try: @@ -978,10 +1066,8 @@ def _settle_stop(self, path: Path) -> dict: return self._stop_receipt(row, binding, None) if stop["phase"] not in DELEGATION_STOP_OPEN_PHASES: return self._stop_receipt(row, binding, stop) - facts = { - "operation_lock_free": self._operation_lock_free(path), - "lane_lock_free": self._lane_lock_free(binding), - } + facts = {"operation_lock_free": self._operation_lock_free(path)} + facts["worker_lane_released"], lane_state = self._worker_lane_released(row, stop, binding) decision = effect_runtime_result("collaboration.delegation.stop", { "phase": stop["phase"], "acknowledged": stop.get("ack") is not None, "timed_out": time.time() - stop["requested_at"] > DELEGATION_STOP_GRACE_SECONDS, @@ -989,10 +1075,10 @@ def _settle_stop(self, path: Path) -> dict: }) if decision["phase"] != stop["phase"] or decision.get("reason") != stop.get("reason"): stop.update(phase=decision["phase"], reason=decision.get("reason")) - if decision["phase"] in {"settled", "unknown"}: + if decision["phase"] in DELEGATION_STOP_TERMINAL_PHASES: lease = stop.get("lease") if isinstance(stop.get("lease"), dict) else {} stop["settled"] = { - "at": time.time(), **facts, + "at": time.time(), **facts, "lane_state": lane_state, "lease_released": lease.get("released"), "turn_journal_status": self._turn_journal_status(row, binding), } @@ -1529,11 +1615,12 @@ def resume_delegation(operation_id: str) -> dict: async def stop_delegation(operation_id: str) -> dict: """Stop one original operation and return what was proven, not what was hoped. - settled: the worker acknowledged and its operation and Turn lane locks are - free. acknowledged/requested: still winding down; call again. unknown: the - holder vanished before acknowledging; inspect its Turn before reusing the - task. noop: already accepted/rejected/stopped. Stopped work is not resumed; - a new scope needs a new operation id. Elapsed time is never a receipt. + settled: the worker acknowledged, released the operation and let go of its + Turn lane. acknowledged/requested: still winding down; call again. unknown: + the named worker vanished before acknowledging; inspect its Turn and task + lease before reusing the task. noop: already accepted/rejected/stopped. + Stopped work is not resumed; a new scope needs a new operation id. Elapsed + time is never a receipt. """ return await asyncio.to_thread(delegations.stop, operation_id, execute=True) @@ -1558,7 +1645,8 @@ def main(): if args.delegation_action == "validate": service._validate(service._bound(_read(service.path(args.operation_id)))) else: - service._stop_signal = install_worker_stop_signal() + service._stop_signal = install_worker_stop_signal( + service._stop_path(service.path(args.operation_id))) try: service.execute(args.operation_id) except LockAcquireTimeoutError: diff --git a/loopx/control_plane/collaboration/delegation.ts b/loopx/control_plane/collaboration/delegation.ts index 9ba1f172da..6612deaec5 100644 --- a/loopx/control_plane/collaboration/delegation.ts +++ b/loopx/control_plane/collaboration/delegation.ts @@ -344,34 +344,37 @@ export function transitionDelegationObservation(params: JsonObject): JsonObject type StopPhase = "requested" | "acknowledged" | "settled" | "unknown"; const openStopPhases: readonly StopPhase[] = ["requested", "acknowledged"]; -/** Advance one stop request from host lock facts; a receipt is never inferred from time. +/** Advance one stop request from host release facts; a receipt is never inferred from time. * * ``settled`` needs the acknowledgement of a process that held the operation - * lock plus both the operation lock and the Turn lane lock free: only then is - * the worker, its Turn child and its lane provably gone. Free locks without an - * acknowledgement mean the holder vanished before recording what it observed, - * which is ``unknown`` rather than a fake settlement. A grace timeout on its - * own moves nothing: a worker that is still holding a lock is still running. + * lock, that lock free again, and the member's Turn lane released by the + * stopped worker's process group. The host reads the lane from its holder + * record and never takes it, so a legitimate Turn is not refused, and a holder + * it cannot attribute is not released. Both released without an + * acknowledgement means the named holder vanished before recording what it + * observed, which is ``unknown`` rather than a fake settlement. A grace + * timeout on its own moves nothing: a worker still holding a lock still runs. */ export function decideDelegationStop(params: JsonObject): JsonObject { const phase = params.phase as StopPhase; requireThat(openStopPhases.includes(phase), "delegation stop decision requires an open stop phase"); requireThat(typeof params.acknowledged === "boolean", "delegation stop acknowledgement fact required"); - requireThat(typeof params.operation_lock_free === "boolean" && typeof params.lane_lock_free === "boolean", - "delegation stop lock facts required"); + requireThat(typeof params.operation_lock_free === "boolean" && typeof params.worker_lane_released === "boolean", + "delegation stop release facts required"); requireThat(params.timed_out === undefined || typeof params.timed_out === "boolean", "delegation stop timeout fact must be boolean"); requireThat(phase !== "acknowledged" || params.acknowledged === true, "an acknowledged stop cannot lose its acknowledgement"); - const locksFree = params.operation_lock_free === true && params.lane_lock_free === true; + const operationFree = params.operation_lock_free === true; + const released = operationFree && params.worker_lane_released === true; if (params.acknowledged === true) { - if (locksFree) return {phase: "settled", terminal: true, reason: "acknowledged_and_locks_released"}; - return {phase: "acknowledged", terminal: false, reason: params.operation_lock_free === true - ? "turn_lane_still_held" : "operation_lock_still_held"}; + if (released) return {phase: "settled", terminal: true, reason: "acknowledged_and_worker_released"}; + return {phase: "acknowledged", terminal: false, + reason: operationFree ? "worker_lane_release_unproven" : "operation_lock_still_held"}; } - if (locksFree) return {phase: "unknown", terminal: true, reason: "holder_gone_without_acknowledgement"}; - return {phase: "requested", terminal: false, reason: params.timed_out === true - ? "holder_still_running_after_grace" : "awaiting_acknowledgement"}; + if (released) return {phase: "unknown", terminal: true, reason: "holder_gone_without_acknowledgement"}; + return {phase: "requested", terminal: false, reason: operationFree ? "worker_lane_release_unproven" + : params.timed_out === true ? "holder_still_running_after_grace" : "awaiting_acknowledgement"}; } /** Repair only a false terminal observation after the exact Turn validated. diff --git a/loopx/file_lock.py b/loopx/file_lock.py index 29303c066e..7ffa396822 100644 --- a/loopx/file_lock.py +++ b/loopx/file_lock.py @@ -360,17 +360,6 @@ def lock_holder_liveness(path: Path) -> tuple[str, dict[str, object]]: return (LOCK_HOLDER_LIVE if process_is_alive(pid) else LOCK_HOLDER_DEAD), record -def read_lock_holder(path: Path) -> dict[str, object]: - """Read the advisory holder record behind ``path``'s lock; ``{}`` when absent. - - The record names the last process that held the lock and carries - ``released_at`` after a clean release. It is advisory readback for signals - and operator inspection; the kernel lock stays the only proof of holding. - """ - - return _read_holder_record(lock_holder_path(path)) - - def _operator_action(holder: dict[str, object], *, retry_mode: str) -> dict[str, object]: return { "required": True, diff --git a/tests/control_plane_ts/delegation.test.ts b/tests/control_plane_ts/delegation.test.ts index 0aaf6686ed..2352c26d8c 100644 --- a/tests/control_plane_ts/delegation.test.ts +++ b/tests/control_plane_ts/delegation.test.ts @@ -116,26 +116,32 @@ test("stopped is terminal and reachable only from open observations", () => { observation}).status, "stopped"); }); -test("a stop settles only on an acknowledgement plus free locks; time alone proves nothing", () => { - const open = {phase: "requested", acknowledged: false, operation_lock_free: false, lane_lock_free: false}; +test("a stop settles only on an acknowledgement plus released holders; time alone proves nothing", () => { + const open = {phase: "requested", acknowledged: false, operation_lock_free: false, worker_lane_released: false}; assert.deepEqual(decideDelegationStop(open), {phase: "requested", terminal: false, reason: "awaiting_acknowledgement"}); assert.deepEqual(decideDelegationStop({...open, timed_out: true}), {phase: "requested", terminal: false, reason: "holder_still_running_after_grace"}); - assert.deepEqual(decideDelegationStop({...open, operation_lock_free: true, timed_out: true}), + // A lane release without a free operation lock is not a vanished holder. + assert.deepEqual(decideDelegationStop({...open, worker_lane_released: true, timed_out: true}), {phase: "requested", terminal: false, reason: "holder_still_running_after_grace"}); - assert.deepEqual(decideDelegationStop({...open, operation_lock_free: true, lane_lock_free: true}), + // A free operation lock with an unattributed lane holder proves nothing yet. + assert.deepEqual(decideDelegationStop({...open, operation_lock_free: true, timed_out: true}), + {phase: "requested", terminal: false, reason: "worker_lane_release_unproven"}); + assert.deepEqual(decideDelegationStop({...open, operation_lock_free: true, worker_lane_released: true}), {phase: "unknown", terminal: true, reason: "holder_gone_without_acknowledgement"}); - const acked = {phase: "acknowledged", acknowledged: true, operation_lock_free: false, lane_lock_free: false}; + const acked = {phase: "acknowledged", acknowledged: true, operation_lock_free: false, worker_lane_released: false}; assert.deepEqual(decideDelegationStop(acked), {phase: "acknowledged", terminal: false, reason: "operation_lock_still_held"}); + assert.deepEqual(decideDelegationStop({...acked, worker_lane_released: true}), + {phase: "acknowledged", terminal: false, reason: "operation_lock_still_held"}); assert.deepEqual(decideDelegationStop({...acked, operation_lock_free: true}), - {phase: "acknowledged", terminal: false, reason: "turn_lane_still_held"}); - assert.deepEqual(decideDelegationStop({...acked, phase: "requested", operation_lock_free: true, lane_lock_free: true}), - {phase: "settled", terminal: true, reason: "acknowledged_and_locks_released"}); - assert.deepEqual(decideDelegationStop({...acked, operation_lock_free: true, lane_lock_free: true, timed_out: true}), - {phase: "settled", terminal: true, reason: "acknowledged_and_locks_released"}); + {phase: "acknowledged", terminal: false, reason: "worker_lane_release_unproven"}); + assert.deepEqual(decideDelegationStop({...acked, phase: "requested", operation_lock_free: true, worker_lane_released: true}), + {phase: "settled", terminal: true, reason: "acknowledged_and_worker_released"}); + assert.deepEqual(decideDelegationStop({...acked, operation_lock_free: true, worker_lane_released: true, timed_out: true}), + {phase: "settled", terminal: true, reason: "acknowledged_and_worker_released"}); for (const patch of [{phase: "settled"}, {phase: "unknown"}, {phase: "noop"}, {acknowledged: "yes"}, - {operation_lock_free: 1}, {lane_lock_free: undefined}, {timed_out: "later"}, - {phase: "acknowledged", acknowledged: false}]) + {operation_lock_free: 1}, {worker_lane_released: undefined}, {lane_lock_free: true, worker_lane_released: undefined}, + {timed_out: "later"}, {phase: "acknowledged", acknowledged: false}]) assert.throws(() => decideDelegationStop({...open, ...patch})); }); diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index faf0208698..90fa1c2ee3 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -20,6 +20,7 @@ from loopx.collaboration_mcp import DelegationFenced, Delegations # noqa: E402 from loopx.control_plane.collaboration.peers import returns # noqa: E402 from loopx.control_plane.collaboration.inbox import _read # noqa: E402 +from loopx.control_plane.turn_driver.lane_fence import turn_lane_liveness, turn_lane_singleflight # noqa: E402 from loopx.file_lock import exclusive_file_lock, try_exclusive_file_lock # noqa: E402 @@ -312,6 +313,12 @@ def test_stop_while_executing_is_acknowledged_by_the_worker_and_settles(service, path = runner.path("analysis-stop") before = _read(path) assert before["status"] == "running" and before["worker"]["pid"] == before["worker"]["pgid"] + # The real run-once child holds the member's lane from inside the worker's group, + # which is what lets settlement attribute the lane without ever taking it. + binding = runner.binding("analysis") + lane = turn_lane_liveness(runner._lane_target(binding)) + assert lane["state"] == "live" and lane["holder"]["pid"] != before["worker"]["pid"] + assert os.getpgid(lane["holder"]["pid"]) == before["worker"]["pgid"] receipt = runner.stop("analysis-stop", execute=True) assert receipt["phase"] == "settled" and receipt["status"] == "stopped", receipt stop = receipt["stop"] @@ -319,7 +326,8 @@ def test_stop_while_executing_is_acknowledged_by_the_worker_and_settles(service, assert stop["worker"]["pid"] == before["worker"]["pid"] assert stop["ack"]["pid"] == before["worker"]["pid"] and stop["ack"]["source"] == "SIGTERM" assert stop["ack"]["observed_status"] == "running" and stop["ack"]["turn_key"] - assert stop["settled"]["operation_lock_free"] and stop["settled"]["lane_lock_free"] + assert stop["settled"]["operation_lock_free"] and stop["settled"]["worker_lane_released"] + assert stop["settled"]["lane_state"] in {"dead", "released"} assert stop["settled"]["turn_journal_status"] == "in_progress" assert stop["lease"] == {"required": False, "released": None} # The acknowledged record is final: nobody writes it again, the Todo stays open, @@ -327,7 +335,6 @@ def test_stop_while_executing_is_acknowledged_by_the_worker_and_settles(service, frozen = path.read_bytes() assert until(lambda: process_gone(host_pid), timeout=20) assert until(lambda: process_gone(before["worker"]["pid"]), timeout=20) - binding = runner.binding("analysis") with try_exclusive_file_lock(runner._lane_target(binding)) as held: assert held is not None assert not demo.canonical_tasks(root)["todo_analyst-initial"]["done"] @@ -356,6 +363,7 @@ def test_stop_without_a_holder_is_acknowledged_by_the_requester(service, monkeyp assert receipt["stop"]["worker"] is None and receipt["stop"]["requested_status"] == "prepared" assert receipt["stop"]["ack"]["pid"] == os.getpid() and receipt["stop"]["ack"]["source"] == "requester" assert receipt["stop"]["settled"]["turn_journal_status"] is None + assert receipt["stop"]["settled"]["lane_state"] in {"absent", "released"} frozen = runner.path("analysis-idle").read_bytes() with pytest.raises(ValueError, match="start a new operation id"): runner.resume("analysis-idle") @@ -370,32 +378,127 @@ def test_stop_without_a_holder_is_acknowledged_by_the_requester(service, monkeyp def test_worker_killed_before_acknowledging_is_unknown_not_settled(service, monkeypatch): - """A vanished holder never becomes a settlement; the stop still fences resume.""" + """A vanished named holder never becomes a settlement; the stop still fences resume.""" import signal root, runner = service host_pid = start_held_worker(service) path = runner.path("analysis-stop") worker = _read(path)["worker"] + killed = [] def kill_without_grace(target, stop): - assert stop["worker"]["pgid"] == worker["pgid"] != os.getpgid(0) + if killed: + return # later calls find nothing left to signal + killed.append(stop["worker"]) + assert stop["worker"] == worker and worker["pgid"] != os.getpgid(0) os.killpg(worker["pgid"], signal.SIGKILL) assert until(lambda: runner._operation_lock_free(target), timeout=20) monkeypatch.setattr(runner, "_signal_worker", kill_without_grace) + first = runner.stop("analysis-stop", execute=True) + assert first["stop"]["ack"] is None and first["phase"] in {"requested", "unknown"}, first + # The operation lock is free now, yet the requester never acknowledges for a + # named worker: the outcome converges on unknown once its Turn child is reaped. + assert until(lambda: runner.stop("analysis-stop", execute=True)["phase"] == "unknown", timeout=20) receipt = runner.stop("analysis-stop", execute=True) - assert receipt["phase"] == "unknown" and receipt["status"] == "running", receipt - assert receipt["stop"]["ack"] is None and receipt["stop"]["settled"]["operation_lock_free"] + assert receipt["status"] == "running" and receipt["stop"]["ack"] is None, receipt + assert receipt["stop"]["settled"]["operation_lock_free"] and receipt["stop"]["settled"]["worker_lane_released"] assert receipt["stop"]["settled"]["turn_journal_status"] == "in_progress" + assert receipt["stop"]["lease"] is None and len(killed) == 1 assert until(lambda: process_gone(host_pid), timeout=20) assert runner.stop("analysis-stop", execute=True) == receipt with pytest.raises(ValueError, match="start a new operation id"): runner.resume("analysis-stop") - assert runner.read("analysis-stop")["stop"]["phase"] == "unknown" + observed = runner.read("analysis-stop") + assert observed["stop"]["phase"] == "unknown" and not observed["recovery_required"] + assert runner.wait("analysis-stop")["stop"]["phase"] == "unknown" assert not demo.canonical_tasks(root)["todo_analyst-initial"]["done"] +def test_sigterm_without_a_stop_keeps_the_operation_recoverable(service, monkeypatch): + """A shutdown signal is not a stop: the worker dies as before and resume stays available.""" + import signal + + root, runner = service + start_held_worker(service, "analysis-term") + path = runner.path("analysis-term") + worker = _read(path)["worker"] + target = runner._lane_target(runner.binding("analysis")) + turn_child = turn_lane_liveness(target)["holder"]["pid"] + try: + os.kill(worker["pid"], signal.SIGTERM) + assert until(lambda: runner._operation_lock_free(path), timeout=20) + # Default termination, exactly as before: only the worker died, and its + # orphaned Turn child still holds the member's lane. + lane = turn_lane_liveness(target) + assert lane["state"] == "live" and lane["holder"]["pid"] == turn_child + assert not runner._stop_path(path).exists() + assert _read(path)["status"] == "running" and "stop" not in runner.read("analysis-term") + spawned = [] + monkeypatch.setattr(runner, "_spawn", spawned.append) + runner.resume("analysis-term") + assert spawned == ["analysis-term"] + finally: + try: + os.killpg(worker["pgid"], signal.SIGKILL) # the orphaned Turn child + except ProcessLookupError: + pass + + +def test_stop_settlement_never_refuses_a_concurrent_turn_on_the_member_lane(service, monkeypatch): + """Settling reads the lane holder record; a real Turn racing it is always admitted.""" + import threading + from loopx import file_lock + + _, runner = service + monkeypatch.setattr(runner, "_spawn", lambda _: None) + runner.start("analysis", "analysis-race", brief()) + path = runner.path("analysis-race") + binding = runner.binding("analysis") + target = runner._lane_target(binding) + lane_lock = target.with_name(target.name + ".lock") + opened = [] + real_open = file_lock._open_lock_descriptor + + def recording_open(lock_path, **kwargs): + opened.append((threading.get_ident(), Path(lock_path))) + return real_open(lock_path, **kwargs) + + monkeypatch.setattr(file_lock, "_open_lock_descriptor", recording_open) + lane = {"runtime_root": runner.root, "goal_id": runner.goal_id, + "plan": {"turn_envelope": {"agent_id": binding["agent_id"]}}} + admitted, refused, phases, settlers = [], [], [], [] + done = Event() + + def settle_continuously(): + settlers.append(threading.get_ident()) + while not done.is_set(): + phases.append(runner.stop("analysis-race", execute=True)["phase"]) + + # An unnamed holder keeps the operation open, so every stop call settles again + # and reads the member's lane while other Turns of that member take it. + with exclusive_file_lock(path), ThreadPoolExecutor(max_workers=1) as pool: + future = pool.submit(settle_continuously) + try: + assert until(lambda: len(phases) >= 2, timeout=30) + for _ in range(200): + with turn_lane_singleflight(**lane) as held: + (admitted if held is not None else refused).append(held) + assert until(lambda: len(phases) >= 4, timeout=30) + finally: + done.set() + future.result(timeout=30) + assert refused == [] and len(admitted) == 200 + assert set(phases) == {"requested"} + assert [lock for ident, lock in opened if ident in settlers and lock == lane_lock] == [] + # Once the holder is gone the requester acknowledges; settling still never takes the lane. + opened.clear() + receipt = runner.stop("analysis-race", execute=True) + assert receipt["phase"] == "settled" and receipt["stop"]["settled"]["lane_state"] == "released" + assert lane_lock not in {lock for _, lock in opened} + + def test_fenced_write_after_another_process_stop_writes_nothing(service, monkeypatch): root, runner = service monkeypatch.setattr(runner, "_spawn", lambda _: None) From 5e7b7bfd8fb425312d25a771911984040ac9ee03 Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 12:34:46 -0400 Subject: [PATCH 08/54] chore(census): follow registry reads moved on main Regenerated with scripts/generate_project_registry_io_manifest.py after rebasing onto main. Site ids and classifications are unchanged. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- loopx/semantics/project_registry_io_manifest_v1.json | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/loopx/semantics/project_registry_io_manifest_v1.json b/loopx/semantics/project_registry_io_manifest_v1.json index 1d92d29240..9b9d446702 100644 --- a/loopx/semantics/project_registry_io_manifest_v1.json +++ b/loopx/semantics/project_registry_io_manifest_v1.json @@ -1799,7 +1799,7 @@ }, { "site": "loopx/history.py::.collect_history::codec_read:load_registry#1", - "line": 341, + "line": 342, "column": 20, "kind": "codec_read", "api": "load_registry", @@ -1807,7 +1807,7 @@ }, { "site": "loopx/history.py::.inspect_index_duplicates::codec_read:load_registry#1", - "line": 597, + "line": 598, "column": 16, "kind": "codec_read", "api": "load_registry", @@ -1815,7 +1815,7 @@ }, { "site": "loopx/history.py::.rebuild_index_artifact_collisions::codec_read:load_registry#1", - "line": 811, + "line": 812, "column": 16, "kind": "codec_read", "api": "load_registry", @@ -1823,7 +1823,7 @@ }, { "site": "loopx/history.py::.repair_index_duplicates::codec_read:load_registry#1", - "line": 701, + "line": 702, "column": 16, "kind": "codec_read", "api": "load_registry", From 5d67e4795ff3febe82149f15dd4ada5900b8887c Mon Sep 17 00:00:00 2001 From: song Date: Wed, 30 Sep 2026 13:22:19 +0800 Subject: [PATCH 09/54] fix(delegation): settle a stop only after the native Host and its group exit The worker and its Turn lane let go while the Host supervisor is still terminating the Host, so their release never proved that the old executor stopped. The Host transport now names the process group its supervisor owns in a record beside the operation, the drain is read back from that record without signalling anything, and the typed decision requires an exited group before a stop may settle. A group that cannot be attributed keeps an acknowledged stop open for a later same-identity read, and the TS supervisor stays the only owner that terminates a Host. Signed-off-by: song --- loopx/collaboration_mcp.py | 52 +++++++++-- .../control_plane/collaboration/delegation.ts | 44 ++++++--- .../collaboration/delegation_inventory.py | 10 +- .../control_plane/turn_driver/host_process.ts | 12 ++- .../turn_driver/host_process_bridge.ts | 2 +- .../turn_driver/host_process_transport.py | 93 ++++++++++++++++++- 6 files changed, 188 insertions(+), 25 deletions(-) diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index 7f90c2f7ce..8158261e58 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -42,12 +42,18 @@ turn_journal_path, ) from .control_plane.turn_driver.host_binding import turn_host_arg_option +from .control_plane.turn_driver.host_process_transport import ( + HOST_PROCESS_DRAINING, HOST_PROCESS_RECORD_ENV, host_process_drain, +) from .control_plane.turn_driver.lane_fence import ( TURN_LANE_ABSENT, TURN_LANE_DEAD, TURN_LANE_LIVE, TURN_LANE_RELEASED, turn_lane_liveness, turn_lane_target, ) from .control_plane.work_items.task_lease import release_task_lease from .control_plane.collaboration.inbox import _hash, _read, _write, _root, _receipt +from .control_plane.collaboration.delegation_inventory import ( + DELEGATION_HOST_PROCESS_SUFFIX, DELEGATION_STOP_RECEIPT_SUFFIX, +) from .control_plane.collaboration.peers import return_result from .control_plane.collaboration.inbox import acknowledge, _entry, normalize_request from .control_plane.collaboration.goal_instance_scope import ( @@ -443,7 +449,12 @@ def path(self, operation_id: str) -> Path: @staticmethod def _stop_path(path: Path) -> Path: """The stop receipt sits beside its execution record and is never merged into it.""" - return path.with_name(path.stem + ".stop.json") + return path.with_name(path.stem + DELEGATION_STOP_RECEIPT_SUFFIX) + + @staticmethod + def _host_process_record(path: Path) -> Path: + """Where this operation's Turn names the native Host it launched, for drain readback.""" + return path.with_name(path.stem + DELEGATION_HOST_PROCESS_SUFFIX) @staticmethod def _dispatch_lock(path: Path) -> Path: @@ -913,7 +924,9 @@ def stop(self, operation_id: str, *, execute: bool) -> dict: holder is signalled by process group and given a bounded grace to acknowledge; another host's holder is left to find the request at its next checkpoint or fenced write. ``settled`` and ``unknown`` come from - the typed decision over lock facts; elapsed time proves nothing. + the typed decision over lock facts and the drain of the native Host the + Turn launched, whose TS supervisor alone terminates it; elapsed time + proves nothing. """ require_operation_id(operation_id) @@ -943,6 +956,9 @@ def stop(self, operation_id: str, *, execute: bool) -> dict: self._acknowledge_stop(path, _read(path), binding, source="requester") except LockAcquireTimeoutError: pass # an unnamed holder meets the request at its next checkpoint or write + else: + # A Host left behind by an earlier worker may still be terminating. + self._await_host_drain(path, time.monotonic() + DELEGATION_STOP_GRACE_SECONDS) else: # Only the named worker acknowledges. If it vanishes first, the typed # decision reports unknown instead of a requester settlement. @@ -982,9 +998,12 @@ def _signal_worker(self, path: Path, stop: dict) -> None: os.killpg(pgid, signal.SIGTERM) except ProcessLookupError: return + # Poll the release facts the decision needs; the deadline only bounds this + # call, and a Host still draining when it passes leaves the stop open. deadline = time.monotonic() + DELEGATION_STOP_GRACE_SECONDS while time.monotonic() < deadline: if self._operation_lock_free(path): + self._await_host_drain(path, deadline) return time.sleep(0.2) try: @@ -994,6 +1013,14 @@ def _signal_worker(self, path: Path, stop: dict) -> None: deadline = time.monotonic() + 2.0 while time.monotonic() < deadline and not self._operation_lock_free(path): time.sleep(0.1) + # The Host supervisor sits outside the worker's group and cleans up on its own. + self._await_host_drain(path, time.monotonic() + DELEGATION_STOP_GRACE_SECONDS) + + def _await_host_drain(self, path: Path, deadline: float) -> None: + """Wait, never kill: the TS Host supervisor owns terminating its process group.""" + while (host_process_drain(self._host_process_record(path)) == HOST_PROCESS_DRAINING + and time.monotonic() < deadline): + time.sleep(0.1) def _acknowledge_stop(self, path: Path, row: dict, binding: dict, *, source: str) -> None: """Acknowledge from under the operation lock: mark stopped, then release the hard lease. @@ -1018,10 +1045,11 @@ def _acknowledge_stop(self, path: Path, row: dict, binding: dict, *, source: str transition = effect_runtime_result("collaboration.delegation.observe", { "from": observed, "to": "stopped", }) - # This process holds the operation lock and its lane read comes later. + # This process holds the operation lock; its lane and Host drain are read later. phase = effect_runtime_result("collaboration.delegation.stop", { "phase": stop["phase"], "acknowledged": True, "operation_lock_free": False, "worker_lane_released": False, + "host_process": HOST_PROCESS_DRAINING, }) current["status"] = transition["status"] _write(path, current) @@ -1068,6 +1096,8 @@ def _settle_stop(self, path: Path) -> dict: return self._stop_receipt(row, binding, stop) facts = {"operation_lock_free": self._operation_lock_free(path)} facts["worker_lane_released"], lane_state = self._worker_lane_released(row, stop, binding) + # Read last: a Host seen drained after its worker and lane let go stays drained. + facts["host_process"] = host_process_drain(self._host_process_record(path)) decision = effect_runtime_result("collaboration.delegation.stop", { "phase": stop["phase"], "acknowledged": stop.get("ack") is not None, "timed_out": time.time() - stop["requested_at"] > DELEGATION_STOP_GRACE_SECONDS, @@ -1085,12 +1115,17 @@ def _settle_stop(self, path: Path) -> dict: _write(self._stop_path(path), stop) return self._stop_receipt(row, binding, stop) - def _cli(self, binding: dict, *args: str, timeout: int = 60) -> dict: + def _cli(self, binding: dict, *args: str, timeout: int = 60, host_record: Path | None = None) -> dict: + environment = _pinned_release_environment() + environment.pop(HOST_PROCESS_RECORD_ENV, None) + if host_record is not None: + # The Turn's Host transport names the process group its supervisor owns. + environment[HOST_PROCESS_RECORD_ENV] = str(host_record) completed = subprocess.run([*_python_module_command("loopx.cli"), "--registry", str(self.registry), "--runtime-root", str(self.root), "--format", "json", *args, ], cwd=binding["workspace"], capture_output=True, text=True, encoding="utf-8", - timeout=timeout, env=_pinned_release_environment()) + timeout=timeout, env=environment) try: value = json.loads(completed.stdout) except ValueError as exc: @@ -1456,7 +1491,8 @@ def _execute(self, path: Path, row: dict, binding: dict) -> None: ] ) result = self._cli(binding, "turn", "run-once", *common, *selector, *execution, - "--execute", timeout=binding["timeout_seconds"] + 60) + "--execute", timeout=binding["timeout_seconds"] + 60, + host_record=self._host_process_record(path)) self._record_turn_result(path, row, result) finally: # The compatibility bootstrap is private host input. Keeping it @@ -1500,6 +1536,7 @@ def _execute(self, path: Path, row: dict, binding: dict) -> None: *execution, "--execute", timeout=binding["timeout_seconds"] + 60, + host_record=self._host_process_record(path), ) self._record_turn_result(path, row, result, publish=False) if result.get("status") != "committed" or result.get("result_kind") != "validated_progress": @@ -1616,7 +1653,8 @@ async def stop_delegation(operation_id: str) -> dict: """Stop one original operation and return what was proven, not what was hoped. settled: the worker acknowledged, released the operation and let go of its - Turn lane. acknowledged/requested: still winding down; call again. unknown: + Turn lane, and the native host and its process group exited. + acknowledged/requested: still winding down; call again. unknown: the named worker vanished before acknowledging; inspect its Turn and task lease before reusing the task. noop: already accepted/rejected/stopped. Stopped work is not resumed; a new scope needs a new operation id. Elapsed diff --git a/loopx/control_plane/collaboration/delegation.ts b/loopx/control_plane/collaboration/delegation.ts index 6612deaec5..00d1214df0 100644 --- a/loopx/control_plane/collaboration/delegation.ts +++ b/loopx/control_plane/collaboration/delegation.ts @@ -343,17 +343,25 @@ export function transitionDelegationObservation(params: JsonObject): JsonObject type StopPhase = "requested" | "acknowledged" | "settled" | "unknown"; const openStopPhases: readonly StopPhase[] = ["requested", "acknowledged"]; +/** What the host read back about the native Host process the operation launched. */ +type HostProcessDrain = "not_launched" | "drained" | "draining" | "unattributable"; +const hostProcessDrains: readonly HostProcessDrain[] = ["not_launched", "drained", "draining", "unattributable"]; /** Advance one stop request from host release facts; a receipt is never inferred from time. * * ``settled`` needs the acknowledgement of a process that held the operation - * lock, that lock free again, and the member's Turn lane released by the - * stopped worker's process group. The host reads the lane from its holder - * record and never takes it, so a legitimate Turn is not refused, and a holder - * it cannot attribute is not released. Both released without an - * acknowledgement means the named holder vanished before recording what it - * observed, which is ``unknown`` rather than a fake settlement. A grace - * timeout on its own moves nothing: a worker still holding a lock still runs. + * lock, that lock free again, the member's Turn lane released by the stopped + * worker's process group, and the native Host the operation launched drained + * together with its process group. A worker and its lane can let go while the + * Host supervisor is still terminating the Host, so their release proves + * nothing about the Host. The host reads the lane from its holder record and + * never takes it, so a legitimate Turn is not refused, and a holder it cannot + * attribute is not released. A Host drain that cannot be attributed keeps an + * acknowledged stop open so a later read with the same identity can still + * settle it. Everything released without an acknowledgement means the named + * holder vanished before recording what it observed, which is ``unknown`` + * rather than a fake settlement. A grace timeout on its own moves nothing: a + * worker still holding a lock still runs. */ export function decideDelegationStop(params: JsonObject): JsonObject { const phase = params.phase as StopPhase; @@ -361,19 +369,29 @@ export function decideDelegationStop(params: JsonObject): JsonObject { requireThat(typeof params.acknowledged === "boolean", "delegation stop acknowledgement fact required"); requireThat(typeof params.operation_lock_free === "boolean" && typeof params.worker_lane_released === "boolean", "delegation stop release facts required"); + requireThat(hostProcessDrains.includes(params.host_process as HostProcessDrain), + "delegation stop host process drain fact required"); requireThat(params.timed_out === undefined || typeof params.timed_out === "boolean", "delegation stop timeout fact must be boolean"); requireThat(phase !== "acknowledged" || params.acknowledged === true, "an acknowledged stop cannot lose its acknowledgement"); const operationFree = params.operation_lock_free === true; - const released = operationFree && params.worker_lane_released === true; + const workerReleased = operationFree && params.worker_lane_released === true; + const host = params.host_process as HostProcessDrain; + const hostDrained = host === "drained" || host === "not_launched"; + const pending = !operationFree ? "operation_lock_still_held" + : params.worker_lane_released !== true ? "worker_lane_release_unproven" + : host === "draining" ? "host_process_still_running" : "host_process_drain_unproven"; if (params.acknowledged === true) { - if (released) return {phase: "settled", terminal: true, reason: "acknowledged_and_worker_released"}; - return {phase: "acknowledged", terminal: false, - reason: operationFree ? "worker_lane_release_unproven" : "operation_lock_still_held"}; + if (workerReleased && hostDrained) { + return {phase: "settled", terminal: true, reason: "acknowledged_worker_and_host_released"}; + } + return {phase: "acknowledged", terminal: false, reason: pending}; + } + if (workerReleased && host !== "draining") { + return {phase: "unknown", terminal: true, reason: "holder_gone_without_acknowledgement"}; } - if (released) return {phase: "unknown", terminal: true, reason: "holder_gone_without_acknowledgement"}; - return {phase: "requested", terminal: false, reason: operationFree ? "worker_lane_release_unproven" + return {phase: "requested", terminal: false, reason: operationFree ? pending : params.timed_out === true ? "holder_still_running_after_grace" : "awaiting_acknowledgement"}; } diff --git a/loopx/control_plane/collaboration/delegation_inventory.py b/loopx/control_plane/collaboration/delegation_inventory.py index f070107f80..4fb2fc3a84 100644 --- a/loopx/control_plane/collaboration/delegation_inventory.py +++ b/loopx/control_plane/collaboration/delegation_inventory.py @@ -12,6 +12,12 @@ if TYPE_CHECKING: from ...collaboration_mcp import Delegations +# Files kept beside an execution record and never merged into it: the stop +# receipt, and the record naming the native Host its Turn launched. +DELEGATION_STOP_RECEIPT_SUFFIX = ".stop.json" +DELEGATION_HOST_PROCESS_SUFFIX = ".host.json" +DELEGATION_RECORD_SIDECAR_SUFFIXES = (DELEGATION_STOP_RECEIPT_SUFFIX, DELEGATION_HOST_PROCESS_SUFFIX) + def read_delegation_inventory(service: Delegations, *, limit: int = 20, cursor: str | None = None) -> dict: @@ -24,8 +30,8 @@ def addresses(): try: entries = directory.iterdir() for path in entries: - if path.suffix != ".json" or path.name.endswith(".stop.json"): - continue # stop receipts sit beside their execution record + if path.suffix != ".json" or path.name.endswith(DELEGATION_RECORD_SIDECAR_SUFFIXES): + continue if not BARE_SHA256_PATTERN.fullmatch(path.stem): raise ValueError("unexpected delegation record address; reconcile inventory storage") if query["cursor"] is None or path.stem > query["cursor"]: diff --git a/loopx/control_plane/turn_driver/host_process.ts b/loopx/control_plane/turn_driver/host_process.ts index 685a5c39fa..0a752a32bf 100644 --- a/loopx/control_plane/turn_driver/host_process.ts +++ b/loopx/control_plane/turn_driver/host_process.ts @@ -22,6 +22,9 @@ export interface HostProcessResult { group_signal_sent: boolean; } export type HostProcessOutput = {kind: "stdout" | "stderr"; text: string}; +/** The group this owner will clean up, reported once the Host is spawned. + * ``process_group`` is null where cleanup is tree best effort (Windows). */ +export type HostProcessSpawned = {kind: "spawned"; pid: number; process_group: number | null}; export const HOST_PROCESS_TERMINATE_GRACE_MS = 300; /** Restrict transport size separately from the caller's public result budget. */ @@ -47,7 +50,8 @@ function signalGroup(child: ChildProcessWithoutNullStreams, signal: NodeJS.Signa } export async function runHostProcess(request: HostProcessRequest, - output: (item: HostProcessOutput) => Promise, signal?: AbortSignal): Promise { + output: (item: HostProcessOutput) => Promise, signal?: AbortSignal, + spawned?: (item: HostProcessSpawned) => Promise): Promise { const base: HostProcessResult = {kind: "result", outcome: "spawn_failed", returncode: null, signal: null, output_complete: true, cleanup_scope: process.platform === "win32" ? "process_tree_best_effort" : "process_group", group_signal_sent: false}; @@ -116,6 +120,12 @@ export async function runHostProcess(request: HostProcessRequest, }; const reads = Promise.all([read("stdout"), read("stderr")]); child.stdin.on("error", () => {}); // A Host may close stdin before consuming it. + if (child.pid && spawned) { + // A caller that cannot record the owned group cancels rather than run unaccounted. + try { await spawned({kind: "spawned", pid: child.pid, + process_group: process.platform === "win32" ? null : child.pid}); } + catch { complete = false; stop("cancelled"); } + } child.stdin.end(request.input); try { await exited; diff --git a/loopx/control_plane/turn_driver/host_process_bridge.ts b/loopx/control_plane/turn_driver/host_process_bridge.ts index b1c261c8bf..d34038cfa5 100644 --- a/loopx/control_plane/turn_driver/host_process_bridge.ts +++ b/loopx/control_plane/turn_driver/host_process_bridge.ts @@ -20,7 +20,7 @@ process.stdin.on("data", (chunk: Buffer) => { accepted = true; const line = pending.subarray(0, newline).toString("utf8"); pending = Buffer.alloc(0); void (async () => { - try { await emit(await runHostProcess(decodeHostProcessRequest(JSON.parse(line)), emit, owner.signal)); } + try { await emit(await runHostProcess(decodeHostProcessRequest(JSON.parse(line)), emit, owner.signal, emit)); } catch { process.exitCode = 1; } finally { process.stdin.destroy(); } })(); diff --git a/loopx/control_plane/turn_driver/host_process_transport.py b/loopx/control_plane/turn_driver/host_process_transport.py index 4f51148fa6..5a6e250534 100644 --- a/loopx/control_plane/turn_driver/host_process_transport.py +++ b/loopx/control_plane/turn_driver/host_process_transport.py @@ -3,12 +3,15 @@ from __future__ import annotations import json +import os import subprocess import sys +import tempfile from collections.abc import Callable, Sequence from pathlib import Path from typing import Any +from ...file_lock import lock_holder_host_label from ..effect_runtime import _node_executable # Keep Python's Windows executable/batch launcher compatibility. No timeout, @@ -16,6 +19,72 @@ _WINDOWS_COMMAND_RELAY = "import subprocess,sys;sys.exit(subprocess.call(sys.argv[1:]))" +# A launching owner that must later prove its Host drained names a record path +# here. The transport consumes it: the Host never inherits it, so a nested +# LoopX run inside the Host cannot overwrite its parent's record. +HOST_PROCESS_RECORD_ENV = "LOOPX_HOST_PROCESS_RECORD" +HOST_PROCESS_RECORD_SCHEMA_VERSION = "loopx_host_process_record_v0" +# Drain facts read back from a record; the caller's typed decision interprets them. +HOST_PROCESS_NOT_LAUNCHED = "not_launched" +HOST_PROCESS_DRAINED = "drained" +HOST_PROCESS_DRAINING = "draining" +HOST_PROCESS_UNATTRIBUTABLE = "unattributable" + + +def _write_host_process_record(path: Path, record: dict[str, Any]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + descriptor, temporary = tempfile.mkstemp(prefix=f".{path.name}.", suffix=".tmp", dir=path.parent) + try: + with os.fdopen(descriptor, "w", encoding="utf-8") as handle: + json.dump(record, handle, separators=(",", ":")) + os.replace(temporary, path) + finally: + Path(temporary).unlink(missing_ok=True) + + +def _process_group_present(pgid: int) -> bool: + try: + os.killpg(pgid, 0) + except ProcessLookupError: + return False + except PermissionError: + return True # it exists; it is simply not ours to signal + return True + + +def host_process_drain(record_path: Path) -> str: + """Read whether the Host a record names, and its process group, have exited. + + Read-only: nothing is signalled, and the TS-owned supervisor keeps cleanup. + No record means no Host was launched under it. ``draining`` while the + supervising bridge or the Host's group still has a member. A record from + another machine, one without a reported group whose supervisor is gone, + or a platform without process groups proves nothing: ``unattributable``. + """ + + try: + record = json.loads(record_path.read_text(encoding="utf-8")) + except FileNotFoundError: + return HOST_PROCESS_NOT_LAUNCHED + except (OSError, ValueError): + return HOST_PROCESS_UNATTRIBUTABLE + if (not isinstance(record, dict) or record.get("schema_version") != HOST_PROCESS_RECORD_SCHEMA_VERSION + or record.get("host") != lock_holder_host_label() or not hasattr(os, "killpg")): + return HOST_PROCESS_UNATTRIBUTABLE + bridge, group = record.get("bridge_pid"), record.get("process_group") + if record.get("phase") != "finished": + if not isinstance(bridge, int) or bridge <= 1: + return HOST_PROCESS_UNATTRIBUTABLE + if _process_group_present(bridge): # the bridge leads its own session + return HOST_PROCESS_DRAINING + if group is None: + # The supervisor left before reporting a group it may already have spawned. + return HOST_PROCESS_UNATTRIBUTABLE if record.get("phase") == "launching" else HOST_PROCESS_DRAINED + if not isinstance(group, int) or group <= 1: + return HOST_PROCESS_UNATTRIBUTABLE + return HOST_PROCESS_DRAINING if _process_group_present(group) else HOST_PROCESS_DRAINED + + class HostOutputLines: """Frame LF records without retaining raw trajectories or an unbounded line.""" @@ -76,6 +145,9 @@ def run_host_process( "stdout_limit_bytes": stdout_limit_bytes, } bridge = Path(__file__).with_name("host_process_bridge.ts") + environment = os.environ.copy() + record_value = environment.pop(HOST_PROCESS_RECORD_ENV, "") + record_path = Path(record_value) if record_value else None with subprocess.Popen( [ _node_executable(), @@ -90,10 +162,19 @@ def run_host_process( encoding="utf-8", errors="strict", start_new_session=True, + env=environment, ) as proc: assert proc.stdin is not None and proc.stdout is not None result = None + record = None try: + if record_path is not None: + # Written before the request, so no Host exists that the record + # does not name; a record that cannot be written launches nothing. + record = {"schema_version": HOST_PROCESS_RECORD_SCHEMA_VERSION, + "host": lock_holder_host_label(), "owner_pid": os.getpid(), + "bridge_pid": proc.pid, "phase": "launching", "process_group": None} + _write_host_process_record(record_path, record) proc.stdin.write( json.dumps(request, ensure_ascii=False, separators=(",", ":")) + "\n" ) @@ -101,7 +182,13 @@ def run_host_process( for line in proc.stdout: event = json.loads(line) kind = event.get("kind") - if kind in {"stdout", "stderr"} and isinstance(event.get("text"), str): + if kind == "spawned" and isinstance(event.get("pid"), int): + if record is not None: + group = event.get("process_group") + record.update(phase="spawned", host_pid=event["pid"], + process_group=group if isinstance(group, int) else None) + _write_host_process_record(record_path, record) + elif kind in {"stdout", "stderr"} and isinstance(event.get("text"), str): consume = on_stdout if kind == "stdout" else on_stderr if consume is not None: consume(event["text"]) @@ -124,6 +211,10 @@ def run_host_process( except subprocess.TimeoutExpired: proc.kill() proc.wait() + if record is not None and result is not None and proc.returncode == 0: + # The supervisor returned only after cleaning its group; the group is still re-read. + record["phase"] = "finished" + _write_host_process_record(record_path, record) if proc.returncode != 0 or result is None: raise RuntimeError("Managed Host process supervision returned no result") return result From fd49df60ef302b6e025c47ce2bf289e12c9fa83c Mon Sep 17 00:00:00 2001 From: song Date: Wed, 30 Sep 2026 13:22:28 +0800 Subject: [PATCH 10/54] test(delegation): prove a stop settles only after the owned Host exits A real File/SQLite fixture whose host and same-group child ignore SIGTERM asserts both are gone at the instant settled returns, with a platform-valid process check in place of the /proc read that treated a live macOS process as exited. Interrupted cleanup, repeated reads, an unacknowledged holder and an unrecorded group get their own negative coverage. Signed-off-by: song --- tests/control_plane/test_host_process.py | 63 +++++++++++++ tests/control_plane_ts/delegation.test.ts | 33 ++++++- tests/control_plane_ts/host_process.test.ts | 24 +++++ tests/test_delegation_inventory.py | 2 + tests/test_local_delegation.py | 99 +++++++++++++++++++-- 5 files changed, 211 insertions(+), 10 deletions(-) diff --git a/tests/control_plane/test_host_process.py b/tests/control_plane/test_host_process.py index 7da98714ef..558725565d 100644 --- a/tests/control_plane/test_host_process.py +++ b/tests/control_plane/test_host_process.py @@ -2,6 +2,7 @@ from __future__ import annotations +import json import os import signal import subprocess @@ -174,3 +175,65 @@ def test_windows_transport_relay_preserves_argv_and_stdin(tmp_path: Path) -> Non ) assert result["ok"] is True assert result["value"] == {"args": values, "input": {}} + + +@pytest.mark.skipif(os.name == "nt", reason="POSIX process-group drain readback") +def test_host_process_record_names_the_owned_group_and_is_not_inherited(tmp_path: Path, monkeypatch) -> None: + from loopx.control_plane.turn_driver.host_process_transport import ( + HOST_PROCESS_RECORD_ENV, host_process_drain, + ) + + record_path = tmp_path / "op.host.json" + assert host_process_drain(record_path) == "not_launched" + monkeypatch.setenv(HOST_PROCESS_RECORD_ENV, str(record_path)) + host = ("import json,os,sys;print(json.dumps({'env': os.environ.get(%r), 'pid': os.getpid()," + " 'pgid': os.getpgid(0)}))" % HOST_PROCESS_RECORD_ENV) + result = _run_host({}, argv=[sys.executable, "-c", host], project=tmp_path, timeout_seconds=5) + assert result["ok"] is True + # A nested LoopX run inside the Host cannot overwrite its parent's record. + assert result["value"]["env"] is None + record = json.loads(record_path.read_text()) + assert record["phase"] == "finished" + assert record["host_pid"] == result["value"]["pid"] == record["process_group"] == result["value"]["pgid"] + assert host_process_drain(record_path) == "drained" + + +@pytest.mark.skipif(os.name == "nt", reason="POSIX process-group drain readback") +def test_host_process_drain_reads_live_groups_and_refuses_unattributable_records(tmp_path: Path) -> None: + from loopx.control_plane.turn_driver.host_process_transport import ( + HOST_PROCESS_RECORD_SCHEMA_VERSION, host_process_drain, + ) + from loopx.file_lock import lock_holder_host_label + + path = tmp_path / "op.host.json" + live = subprocess.Popen([sys.executable, "-c", "import time;time.sleep(60)"], start_new_session=True) + gone = subprocess.Popen([sys.executable, "-c", "pass"], start_new_session=True) + gone.wait(timeout=10) + + def drain(**fields): + path.write_text(json.dumps({"schema_version": HOST_PROCESS_RECORD_SCHEMA_VERSION, + "host": lock_holder_host_label(), "owner_pid": 1, **fields})) + return host_process_drain(path) + + try: + # A live supervisor or a live Host group is still draining. + assert drain(phase="launching", bridge_pid=live.pid, process_group=None) == "draining" + assert drain(phase="spawned", bridge_pid=gone.pid, process_group=live.pid) == "draining" + assert drain(phase="finished", bridge_pid=gone.pid, process_group=live.pid) == "draining" + assert drain(phase="spawned", bridge_pid=gone.pid, process_group=gone.pid) == "drained" + # A supervisor gone before it reported a group may have spawned one anyway. + assert drain(phase="launching", bridge_pid=gone.pid, process_group=None) == "unattributable" + assert drain(phase="finished", bridge_pid=gone.pid, process_group=None) == "drained" + for fields in ({"phase": "spawned", "bridge_pid": None, "process_group": gone.pid}, + {"phase": "spawned", "bridge_pid": gone.pid, "process_group": "1"}, + {"phase": "spawned", "bridge_pid": gone.pid, "process_group": 1}, + {"phase": "spawned", "bridge_pid": gone.pid, "process_group": gone.pid, + "host": "another-machine"}, + {"phase": "spawned", "bridge_pid": gone.pid, "process_group": gone.pid, + "schema_version": "other"}): + assert drain(**fields) == "unattributable", fields + path.write_text("{not json") + assert host_process_drain(path) == "unattributable" + finally: + live.kill() + live.wait(timeout=10) diff --git a/tests/control_plane_ts/delegation.test.ts b/tests/control_plane_ts/delegation.test.ts index 2352c26d8c..f97a99c957 100644 --- a/tests/control_plane_ts/delegation.test.ts +++ b/tests/control_plane_ts/delegation.test.ts @@ -117,7 +117,8 @@ test("stopped is terminal and reachable only from open observations", () => { }); test("a stop settles only on an acknowledgement plus released holders; time alone proves nothing", () => { - const open = {phase: "requested", acknowledged: false, operation_lock_free: false, worker_lane_released: false}; + const open = {phase: "requested", acknowledged: false, operation_lock_free: false, worker_lane_released: false, + host_process: "drained"}; assert.deepEqual(decideDelegationStop(open), {phase: "requested", terminal: false, reason: "awaiting_acknowledgement"}); assert.deepEqual(decideDelegationStop({...open, timed_out: true}), {phase: "requested", terminal: false, reason: "holder_still_running_after_grace"}); @@ -129,22 +130,46 @@ test("a stop settles only on an acknowledgement plus released holders; time alon {phase: "requested", terminal: false, reason: "worker_lane_release_unproven"}); assert.deepEqual(decideDelegationStop({...open, operation_lock_free: true, worker_lane_released: true}), {phase: "unknown", terminal: true, reason: "holder_gone_without_acknowledgement"}); - const acked = {phase: "acknowledged", acknowledged: true, operation_lock_free: false, worker_lane_released: false}; + const acked = {phase: "acknowledged", acknowledged: true, operation_lock_free: false, worker_lane_released: false, + host_process: "drained"}; assert.deepEqual(decideDelegationStop(acked), {phase: "acknowledged", terminal: false, reason: "operation_lock_still_held"}); assert.deepEqual(decideDelegationStop({...acked, worker_lane_released: true}), {phase: "acknowledged", terminal: false, reason: "operation_lock_still_held"}); assert.deepEqual(decideDelegationStop({...acked, operation_lock_free: true}), {phase: "acknowledged", terminal: false, reason: "worker_lane_release_unproven"}); assert.deepEqual(decideDelegationStop({...acked, phase: "requested", operation_lock_free: true, worker_lane_released: true}), - {phase: "settled", terminal: true, reason: "acknowledged_and_worker_released"}); + {phase: "settled", terminal: true, reason: "acknowledged_worker_and_host_released"}); assert.deepEqual(decideDelegationStop({...acked, operation_lock_free: true, worker_lane_released: true, timed_out: true}), - {phase: "settled", terminal: true, reason: "acknowledged_and_worker_released"}); + {phase: "settled", terminal: true, reason: "acknowledged_worker_and_host_released"}); for (const patch of [{phase: "settled"}, {phase: "unknown"}, {phase: "noop"}, {acknowledged: "yes"}, + {host_process: undefined}, {host_process: "exited"}, {host_process: true}, {operation_lock_free: 1}, {worker_lane_released: undefined}, {lane_lock_free: true, worker_lane_released: undefined}, {timed_out: "later"}, {phase: "acknowledged", acknowledged: false}]) assert.throws(() => decideDelegationStop({...open, ...patch})); }); +test("a released worker and lane never settle a stop while the native Host still drains", () => { + const released = {phase: "acknowledged", acknowledged: true, operation_lock_free: true, worker_lane_released: true}; + assert.deepEqual(decideDelegationStop({...released, host_process: "draining", timed_out: true}), + {phase: "acknowledged", terminal: false, reason: "host_process_still_running"}); + // Without an attributable drain the stop stays open for a later same-identity read. + assert.deepEqual(decideDelegationStop({...released, host_process: "unattributable"}), + {phase: "acknowledged", terminal: false, reason: "host_process_drain_unproven"}); + for (const host_process of ["drained", "not_launched"]) + assert.deepEqual(decideDelegationStop({...released, host_process}), + {phase: "settled", terminal: true, reason: "acknowledged_worker_and_host_released"}); + // A held lock still dominates a drained Host. + assert.deepEqual(decideDelegationStop({...released, operation_lock_free: false, host_process: "drained"}), + {phase: "acknowledged", terminal: false, reason: "operation_lock_still_held"}); + // A vanished holder is unknown only once its Host is no longer seen running. + const vanished = {...released, phase: "requested", acknowledged: false}; + assert.deepEqual(decideDelegationStop({...vanished, host_process: "draining"}), + {phase: "requested", terminal: false, reason: "host_process_still_running"}); + for (const host_process of ["drained", "not_launched", "unattributable"]) + assert.deepEqual(decideDelegationStop({...vanished, host_process}), + {phase: "unknown", terminal: true, reason: "holder_gone_without_acknowledgement"}); +}); + test("a false rejection can reopen only for exact validated settlement recovery", () => { const evidence = { from: "rejected", diff --git a/tests/control_plane_ts/host_process.test.ts b/tests/control_plane_ts/host_process.test.ts index cbea7db0d4..215af4348c 100644 --- a/tests/control_plane_ts/host_process.test.ts +++ b/tests/control_plane_ts/host_process.test.ts @@ -1,4 +1,5 @@ import assert from "node:assert/strict"; +import {existsSync} from "node:fs"; import {mkdtemp, readFile, rm} from "node:fs/promises"; import {join} from "node:path"; import {tmpdir} from "node:os"; @@ -76,3 +77,26 @@ test("output consumer failure cancels execution rather than leaving an orphan", async () => { throw new Error("consumer left"); }); assert.equal(result.outcome, "cancelled"); assert.equal(result.output_complete, false); }); + +test("the spawned Host group is reported once before input, and an unrecorded group never runs", {skip: process.platform === "win32"}, async t => { + const root = await mkdtemp(join(tmpdir(), "loopx-host-spawned-")); + t.after(() => rm(root, {recursive: true, force: true})); + const seen: unknown[] = []; + let stdout = ""; + const result = await runHostProcess(request(`process.stdout.write(String(process.pid)+' '+String(require('child_process').execSync('ps -o pgid= -p '+process.pid)).trim())`), + async item => { stdout += item.text; }, undefined, async item => { seen.push(item); }); + const [pid, pgid] = stdout.split(" ").map(Number); + assert.equal(result.outcome, "exited"); + assert.deepEqual(seen, [{kind: "spawned", pid, process_group: pid}]); assert.equal(pgid, pid); + // A caller that cannot record the owned group runs nothing unaccounted for, and + // the armed host proves it really started, so this is not an unspawned process. + const script = (marker: string) => `require('fs').writeFileSync(${JSON.stringify(marker)},'') + process.stdin.on('data',()=>{});setInterval(()=>{},1000)`; + const recorded = join(root, "recorded"); + const refused = await runHostProcess(request(script(recorded)), async () => {}, undefined, async () => { + const until = Date.now() + 2000; + while (!existsSync(recorded) && Date.now() < until) await delay(5); + assert.ok(existsSync(recorded), "the Host never started"); + throw new Error("record unavailable"); }); + assert.equal(refused.outcome, "cancelled"); assert.equal(refused.output_complete, false); +}); diff --git a/tests/test_delegation_inventory.py b/tests/test_delegation_inventory.py index b75dec65c9..9ac402e4c4 100644 --- a/tests/test_delegation_inventory.py +++ b/tests/test_delegation_inventory.py @@ -59,6 +59,8 @@ def test_corruption_and_stopped_worker_do_not_hide_healthy_sibling(service, monk # A stop receipt beside its record is not another record and reads back as stopped. assert runner.stop("healthy", execute=True)["phase"] == "settled" assert runner._stop_path(runner.path("healthy")).exists() + # Nor is the record naming the native Host an operation's Turn launched. + runner._host_process_record(runner.path("healthy")).write_text("{}") page = runner.operations() by_id = {row["operation_id"]: row for row in page["items"] if row["operation_id"]} assert len(page["items"]) == 4 and by_id["healthy"]["status"] == "stopped" diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index 90fa1c2ee3..871b2517a8 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -36,6 +36,11 @@ actor = envelope['agent_id'] counter = workspace / 'host-invocations' counter.write_text(str(int(counter.read_text()) + 1 if counter.exists() else 1)) +if (root / 'ignore-term').exists(): + import signal, subprocess + signal.signal(signal.SIGTERM, signal.SIG_IGN) + child = subprocess.Popen([sys.executable, '-c', 'import signal, time; signal.signal(signal.SIGTERM, signal.SIG_IGN); time.sleep(600)']) + (root / 'host-child-pid').write_text(str(child.pid)) if (root / 'hold').exists(): (root / 'host-pid').write_text(str(os.getpid())) (root / 'host-started').touch() @@ -288,14 +293,16 @@ def until(predicate, timeout=45): def process_gone(pid): + """Platform-valid: a zombie has exited; a process ``ps`` cannot see is gone.""" try: os.kill(pid, 0) except ProcessLookupError: return True - try: - return "State:\tZ" in Path(f"/proc/{pid}/status").read_text() - except OSError: - return True + except PermissionError: + return False + state = subprocess.run(["ps", "-o", "stat=", "-p", str(pid)], + capture_output=True, text=True).stdout.strip() + return not state or state.startswith("Z") def start_held_worker(service, operation="analysis-stop"): @@ -320,7 +327,9 @@ def test_stop_while_executing_is_acknowledged_by_the_worker_and_settles(service, assert lane["state"] == "live" and lane["holder"]["pid"] != before["worker"]["pid"] assert os.getpgid(lane["holder"]["pid"]) == before["worker"]["pgid"] receipt = runner.stop("analysis-stop", execute=True) + host_gone = process_gone(host_pid) # the instant settled returns, not after a wait assert receipt["phase"] == "settled" and receipt["status"] == "stopped", receipt + assert host_gone stop = receipt["stop"] assert stop["requested_by"] == "lead" and stop["requested_status"] == "running" assert stop["worker"]["pid"] == before["worker"]["pid"] @@ -328,12 +337,12 @@ def test_stop_while_executing_is_acknowledged_by_the_worker_and_settles(service, assert stop["ack"]["observed_status"] == "running" and stop["ack"]["turn_key"] assert stop["settled"]["operation_lock_free"] and stop["settled"]["worker_lane_released"] assert stop["settled"]["lane_state"] in {"dead", "released"} + assert stop["settled"]["host_process"] == "drained" assert stop["settled"]["turn_journal_status"] == "in_progress" assert stop["lease"] == {"required": False, "released": None} # The acknowledged record is final: nobody writes it again, the Todo stays open, - # the host process group is gone and the member's Turn lane can be taken. + # the worker exits and the member's Turn lane can be taken. frozen = path.read_bytes() - assert until(lambda: process_gone(host_pid), timeout=20) assert until(lambda: process_gone(before["worker"]["pid"]), timeout=20) with try_exclusive_file_lock(runner._lane_target(binding)) as held: assert held is not None @@ -364,6 +373,7 @@ def test_stop_without_a_holder_is_acknowledged_by_the_requester(service, monkeyp assert receipt["stop"]["ack"]["pid"] == os.getpid() and receipt["stop"]["ack"]["source"] == "requester" assert receipt["stop"]["settled"]["turn_journal_status"] is None assert receipt["stop"]["settled"]["lane_state"] in {"absent", "released"} + assert receipt["stop"]["settled"]["host_process"] == "not_launched" frozen = runner.path("analysis-idle").read_bytes() with pytest.raises(ValueError, match="start a new operation id"): runner.resume("analysis-idle") @@ -534,3 +544,80 @@ def test_fenced_write_after_another_process_stop_writes_nothing(service, monkeyp acknowledged = _read(runner._stop_path(entry)) assert acknowledged["phase"] == "acknowledged" and acknowledged["ack"]["source"] == "worker_entry" assert runner.stop("analysis-entry", execute=True)["phase"] == "settled" + + +def test_stop_settles_only_after_the_owned_host_and_its_descendants_exit(service): + """A host and its same-group child that ignore SIGTERM keep the stop open until they exit. + + The postcondition is checked at the instant ``settled`` returns, not after a wait. + """ + root, runner = service + (root / "ignore-term").touch() + host_pid = start_held_worker(service) + assert until(lambda: (root / "host-child-pid").exists()) + child_pid = int((root / "host-child-pid").read_text()) + assert not process_gone(host_pid) and not process_gone(child_pid) + receipt = runner.stop("analysis-stop", execute=True) + host_gone, child_gone = process_gone(host_pid), process_gone(child_pid) + assert receipt["phase"] == "settled", receipt + assert host_gone and child_gone, (receipt, host_gone, child_gone) + settled = receipt["stop"]["settled"] + assert settled["host_process"] == "drained", settled + # Rereading a settled stop neither reopens it nor admits or completes anything. + frozen = runner.path("analysis-stop").read_bytes() + assert runner.stop("analysis-stop", execute=True) == receipt + assert runner.read("analysis-stop")["stop"]["phase"] == "settled" + assert runner.path("analysis-stop").read_bytes() == frozen + assert int((root / "analyst" / "initial" / "host-invocations").read_text()) == 1 + assert not demo.canonical_tasks(root)["todo_analyst-initial"]["done"] + assert returns(runner.root, runner.goal_id, "lead")["items"] == [] + + +def test_interrupted_host_cleanup_keeps_the_stop_open_until_a_reread_sees_it_drained(service, monkeypatch): + """A Host supervisor that never finishes cleaning up leaves no settlement to claim. + + Rereads with the same identity stay acknowledged without admitting or completing + anything, and settle once the Host's group is observed gone. + """ + import signal + from loopx import collaboration_mcp + + root, runner = service + monkeypatch.setattr(collaboration_mcp, "DELEGATION_STOP_GRACE_SECONDS", 2.0) + host_pid = start_held_worker(service) + path = runner.path("analysis-stop") + record = json.loads(runner._host_process_record(path).read_text()) + assert record["phase"] == "spawned" and record["host_pid"] == host_pid == record["process_group"] + bridge = record["bridge_pid"] + os.kill(bridge, signal.SIGSTOP) # the supervisor cannot run its cleanup + try: + first = runner.stop("analysis-stop", execute=True) + assert first["phase"] == "acknowledged" and first["status"] == "stopped", first + assert first["reason"] == "host_process_still_running" and not process_gone(host_pid) + os.kill(bridge, signal.SIGKILL) # and now never will: the Host is orphaned + assert until(lambda: process_gone(bridge), timeout=20) + frozen = path.read_bytes() + for _ in range(3): + again = runner.stop("analysis-stop", execute=True) + assert again["phase"] == "acknowledged" and again["reason"] == "host_process_still_running", again + assert again["stop"]["stop_id"] == first["stop"]["stop_id"] + assert again["stop"]["ack"] == first["stop"]["ack"] and again["stop"]["settled"] is None + assert runner.read("analysis-stop")["stop"]["phase"] == "acknowledged" + assert not process_gone(host_pid) and path.read_bytes() == frozen + with pytest.raises(ValueError, match="start a new operation id"): + runner.resume("analysis-stop") + finally: + for target, sig in ((bridge, signal.SIGCONT), (host_pid, signal.SIGKILL)): + try: + os.killpg(target, sig) if target == host_pid else os.kill(target, sig) + except ProcessLookupError: + pass + assert until(lambda: process_gone(host_pid), timeout=20) + receipt = runner.stop("analysis-stop", execute=True) + assert receipt["phase"] == "settled" and receipt["stop"]["settled"]["host_process"] == "drained", receipt + assert receipt["stop"]["stop_id"] == first["stop"]["stop_id"] + assert runner.stop("analysis-stop", execute=True) == receipt + assert path.read_bytes() == frozen + assert int((root / "analyst" / "initial" / "host-invocations").read_text()) == 1 + assert not demo.canonical_tasks(root)["todo_analyst-initial"]["done"] + assert returns(runner.root, runner.goal_id, "lead")["items"] == [] From 226a664d114bb77122c34a32773cb43530fd9480 Mon Sep 17 00:00:00 2001 From: song Date: Wed, 30 Sep 2026 13:22:28 +0800 Subject: [PATCH 11/54] docs(delegation): state what a stop receipt proves about the native Host Signed-off-by: song --- docs/reference/goal-chat-continuation.md | 5 ++-- docs/reference/local-delegation.md | 30 +++++++++++++++--------- 2 files changed, 22 insertions(+), 13 deletions(-) diff --git a/docs/reference/goal-chat-continuation.md b/docs/reference/goal-chat-continuation.md index 7c367d68d3..7f9ab29ce1 100644 --- a/docs/reference/goal-chat-continuation.md +++ b/docs/reference/goal-chat-continuation.md @@ -102,7 +102,8 @@ messages; images use the ordinary conversation after pausing. Disabling mode or deleting a binding does not cancel already admitted children; stop one with `loopx delegation stop --execute` (or `stop_delegation`), read its receipt, and retain its evidence. Only `settled` proves the worker - acknowledged and released its locks; stopped work needs a new operation id. + acknowledged and released its locks and its native host exited; stopped work + needs a new operation id. The ordinary native command path also remains available without delegation: @@ -147,6 +148,6 @@ For a disposable mixed-team setup, use the 或整个 Goal。额度是含历史用量的总量,正在执行的请求可能超额,成员另行计量。 回滚旧版本前先暂停或关闭 Chat 服务;退出或撤销绑定不自动取消已启动的成员, 用 `loopx delegation stop --execute`(或 `stop_delegation`)停止单个成员并阅读回执: -只有 `settled` 证明 worker 已确认并释放锁;已停止的工作需要新的 operation id。 +只有 `settled` 证明 worker 已确认并释放锁且原生 host 已退出;已停止的工作需要新的 operation id。 此按钮目前限本机 managed Codex Goal 对话,不宣称 Lark、挂接会话或其他主力 驱动等价。可用下方示例准备一次隔离的本地 DSH+云端 Ark 协作。 diff --git a/docs/reference/local-delegation.md b/docs/reference/local-delegation.md index b5e3e39fa2..8f61a1dac4 100644 --- a/docs/reference/local-delegation.md +++ b/docs/reference/local-delegation.md @@ -226,14 +226,19 @@ than maintaining separate rules. states what was proven. The request is written beside the execution record (`.stop.json`), never into it, so a worker that is still holding the operation cannot overwrite it. A worker on this machine receives `SIGTERM` for -its whole process group, which ends its Turn child and its host; it -acknowledges from under its own lock, marks the record `stopped` and releases -its hard task lease. When nobody holds the operation, the requester +its whole process group, which ends its Turn child; the native host runs in +its own process group, and its supervisor terminates that group once the Turn +child is gone. The worker acknowledges from under its own lock, marks the +record `stopped` and releases its hard task lease. When nobody holds the operation, the requester acknowledges itself. A worker on another machine is never signalled; it finds the request at its next checkpoint or at its next record write, which is -refused. The receipt `phase` is `settled` only when an acknowledgement exists -and both the operation lock and the member's Turn lane lock are free; -`unknown` means the holder vanished before acknowledging, and `noop` means the +refused. The receipt `phase` is `settled` only when an acknowledgement exists, +the operation lock is free, the member's Turn lane holder record shows it +released by the stopped worker (the lane is read, never taken), and the native +host the Turn launched has exited together with every process in its group. +If that host cannot be attributed, or its supervisor never finished cleaning +up, the stop stays `acknowledged` and a later `stop` rereads it. `unknown` +means the holder vanished before acknowledging, and `noop` means the work was already accepted, rejected or stopped. `requested` or `acknowledged` means it is still winding down: call `stop` again. A grace timeout never turns into a receipt. Stopped work is not resumed; `resume` refuses it and a new @@ -244,11 +249,14 @@ member's Todo stays open, so the coordinator decides what happens next. 中文:`stop --execute` 结束一个成员的有界工作,并返回一份只陈述已证明事实的 回执。停止请求写在执行记录旁边的 `.stop.json`,从不写进记录本身, 因此仍持有该 operation 的 worker 无法覆盖它。本机 worker 会收到整个进程组的 -`SIGTERM`,其 Turn 子进程和 host 一并结束;worker 在自己的锁下确认,把记录标为 +`SIGTERM`,其 Turn 子进程随之结束;原生 host 在自己的进程组中运行,Turn 子进程 +退出后由其 supervisor 终止整个 host 进程组。worker 在自己的锁下确认,把记录标为 `stopped` 并释放硬任务租约。没有持有者时由请求方自行确认。另一台机器上的 worker 不会被发信号,它在下一个检查点或下一次写记录时发现请求,写入被拒绝。 -只有存在确认且 operation 锁与成员 Turn lane 锁都已释放时,`phase` 才是 -`settled`;`unknown` 表示持有者在确认前消失;`noop` 表示工作已 accepted、 +只有存在确认、operation 锁已释放、成员 Turn lane 的持有者记录显示已被停止的 +worker 释放(只读 lane,从不获取),且该 Turn 启动的原生 host 及其进程组内所有进程 +都已退出时,`phase` 才是 `settled`。host 无法归属或其 supervisor 未完成清理时, +停止保持 `acknowledged`,之后再次调用 `stop` 会重新读取。`unknown` 表示持有者在确认前消失;`noop` 表示工作已 accepted、 rejected 或 stopped;`requested`/`acknowledged` 表示仍在收尾,再次调用 `stop`。 宽限期超时永远不会变成回执。已停止的工作不能 `resume`,新范围需要新的 operation id。Turn journal 保留 `in_progress` 条目供检查,记录不会被改写成完成; @@ -651,7 +659,7 @@ unchanged and cannot launch workers. With it, the Agent can: This cannot retarget the work or silently create a replacement Turn. 5. Call `stop_delegation(operation_id)` to end one member. Read its `phase`: `settled` is the only receipt that the worker acknowledged and released its - locks; `unknown` means the holder vanished first; `noop` means the work had + locks and that the native host and its process group exited; `unknown` means the holder vanished first; `noop` means the work had already ended. Stopped work cannot be resumed; use a new operation id. Configure the member's host to expose its own identity-bound collaboration @@ -693,7 +701,7 @@ concurrent executions still use the same kernel lock and original Turn journal. | Requesting MCP conversation closes | The detached bounded worker continues; another connection reads the original operation. | | Duplicate start/resume while work runs | Operation identity, task lock and Turn journal prevent another concurrent execution. | | Worker process or machine stops | Reconnect with the same operator configuration and credentials, then resume the original Turn. | -| Member stopped on request | The worker acknowledges under its lock, its host and Turn child are ended, its lease is released; `settled` needs that acknowledgement plus free locks, `unknown` means the holder vanished first. The record is `stopped`; resume refuses it. | +| Member stopped on request | The worker acknowledges under its lock, its Turn child is ended and the host supervisor terminates the host group, its lease is released; `settled` needs that acknowledgement, free locks and an exited host group, `unknown` means the holder vanished first. The record is `stopped`; resume refuses it. | | Ark is computing without local tools | The already-started cloud turn can continue. It is not dependent on the local conversation. | | Ark requests a local tool while the host is absent | It waits for the local tool result. Recovery observes the original input/session and executes only previously unstarted tool calls. | | Tool execution or send acknowledgement is uncertain | Do not repeat the effect. Preserve the receipt/session for explicit reconciliation. | From 01be96c65c7c1c226ab62f99f1536e9e3979072a Mon Sep 17 00:00:00 2001 From: song Date: Wed, 30 Sep 2026 10:44:40 -0400 Subject: [PATCH 12/54] fix(delegation): give a stop a boundary against its effects, its lease and the platform MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three blockers, all reproduced on the reviewed head before fixing. **A stop could settle after the member's effects committed.** The worker ran one pre-check and then committed Todo completion and reply publication outside any lock the stop takes, so a stop written first still reported `settled` for work that had landed. Both effects now commit inside the dispatch lock a stop also takes, so the two sides linearize: a stop written first means neither effect runs, and a stop written after leaves their acceptance intact. `_observe` gained a locked entry point because the kernel file lock is not reentrant and the widened critical section must not nest. **A failed required lease release was terminal and never retried.** `_settle_stop` did not pass the release to the typed decision, so an authority failure produced `settled` with `lease_released=false` on both backends and every later read returned the same receipt — leaving the member's Todo blocked until the lease TTL with resume already refused. `decideDelegationStop` now treats a required release as part of what `settled` promises, and the settle read retries the release under the stop's own lock. An operation that held no required lease omits the fact rather than claiming a release. **A launched Host on Windows could never settle.** `host_process_drain` returned `unattributable` whenever `killpg` was absent, so a finished Host left the stop pending forever with no converging or actionable path. That case is now `unsupported_platform`, distinct from an attribution failure, and `stop --execute` fails fast with an error naming the platform boundary instead of returning a receipt no read can settle. The public reference documents the lease obligation, the effect ordering and that boundary in both languages. Coverage: a stop written before the effects (both backends) asserts neither effect commits and the receipt still settles; a failed release asserts `acknowledged`/`required_lease_release_unproven` and that the next read retries and settles; a Host on a platform without process groups asserts the actionable failure. The TS decision pins the lease obligation on both the acknowledged and vanished-holder paths, mutation-checked by dropping it. Signed-off-by: song --- docs/reference/local-delegation.md | 36 +++-- loopx/collaboration_mcp.py | 109 ++++++++++----- .../control_plane/collaboration/delegation.ts | 16 ++- .../turn_driver/host_process_transport.py | 19 ++- tests/control_plane_ts/delegation.test.ts | 18 +++ tests/test_local_delegation.py | 129 +++++++++++++++++- 6 files changed, 281 insertions(+), 46 deletions(-) diff --git a/docs/reference/local-delegation.md b/docs/reference/local-delegation.md index 8f61a1dac4..a32ff944e2 100644 --- a/docs/reference/local-delegation.md +++ b/docs/reference/local-delegation.md @@ -234,17 +234,28 @@ acknowledges itself. A worker on another machine is never signalled; it finds the request at its next checkpoint or at its next record write, which is refused. The receipt `phase` is `settled` only when an acknowledgement exists, the operation lock is free, the member's Turn lane holder record shows it -released by the stopped worker (the lane is read, never taken), and the native -host the Turn launched has exited together with every process in its group. -If that host cannot be attributed, or its supervisor never finished cleaning -up, the stop stays `acknowledged` and a later `stop` rereads it. `unknown` +released by the stopped worker (the lane is read, never taken), the native +host the Turn launched has exited together with every process in its group, +and a required hard task lease was actually released. A release that failed is +retried under the stop's own lock on the next read, so it never becomes a +`settled` receipt that leaves the member's Todo blocked until the lease TTL; +while it is unproven the stop stays `acknowledged` with +`required_lease_release_unproven`. If that host cannot be attributed, or its +supervisor never finished cleaning up, the stop stays `acknowledged` and a +later `stop` rereads it. On a platform without process groups the launched +host cannot be proven drained at all, so `stop --execute` fails with an +actionable error naming that boundary rather than leaving a receipt no read +can settle. `unknown` means the holder vanished before acknowledging, and `noop` means the work was already accepted, rejected or stopped. `requested` or `acknowledged` means it is still winding down: call `stop` again. A grace timeout never turns into a receipt. Stopped work is not resumed; `resume` refuses it and a new scope needs a new operation id. The Turn journal keeps its `in_progress` entry for inspection, and the record is never rewritten as a completion. A stopped -member's Todo stays open, so the coordinator decides what happens next. +member's Todo stays open, so the coordinator decides what happens next. The +member's Todo completion and reply publication commit under the same lock a +stop takes, so a stop written first means neither effect lands, and a stop +written after both leaves their acceptance intact. 中文:`stop --execute` 结束一个成员的有界工作,并返回一份只陈述已证明事实的 回执。停止请求写在执行记录旁边的 `.stop.json`,从不写进记录本身, @@ -254,13 +265,20 @@ member's Todo stays open, so the coordinator decides what happens next. `stopped` 并释放硬任务租约。没有持有者时由请求方自行确认。另一台机器上的 worker 不会被发信号,它在下一个检查点或下一次写记录时发现请求,写入被拒绝。 只有存在确认、operation 锁已释放、成员 Turn lane 的持有者记录显示已被停止的 -worker 释放(只读 lane,从不获取),且该 Turn 启动的原生 host 及其进程组内所有进程 -都已退出时,`phase` 才是 `settled`。host 无法归属或其 supervisor 未完成清理时, -停止保持 `acknowledged`,之后再次调用 `stop` 会重新读取。`unknown` 表示持有者在确认前消失;`noop` 表示工作已 accepted、 +worker 释放(只读 lane,从不获取)、该 Turn 启动的原生 host 及其进程组内所有进程 +都已退出,且必需的硬任务租约确实释放成功时,`phase` 才是 `settled`。释放失败会在 +下一次读取时于 stop 自己的锁下重试,因此不会产生一份「已结算」却让成员 Todo 被 +租约阻塞到 TTL 的回执;在释放得到证明前,停止保持 `acknowledged`,原因为 +`required_lease_release_unproven`。host 无法归属或其 supervisor 未完成清理时, +停止保持 `acknowledged`,之后再次调用 `stop` 会重新读取。在没有进程组的平台上, +启动过的 host 根本无法被证明已收尾,因此 `stop --execute` 会以指明该平台边界的 +可操作错误失败,而不是留下一份任何读取都无法结算的回执。`unknown` 表示持有者在确认前消失;`noop` 表示工作已 accepted、 rejected 或 stopped;`requested`/`acknowledged` 表示仍在收尾,再次调用 `stop`。 宽限期超时永远不会变成回执。已停止的工作不能 `resume`,新范围需要新的 operation id。Turn journal 保留 `in_progress` 条目供检查,记录不会被改写成完成; -成员的 Todo 仍然打开,由协调者决定下一步。 +成员的 Todo 仍然打开,由协调者决定下一步。成员的 Todo 完成与回执发布在 stop +所取的同一把锁下提交,因此先写入停止则两个效果都不会落地,后写入停止则其验收结果 +保持不变。 This entrypoint does not create Agents, grant bindings or wake an idle Codex conversation. The existing host/LoopX continuation policy owns the next lead diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index 8158261e58..82d1906a70 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -43,7 +43,8 @@ ) from .control_plane.turn_driver.host_binding import turn_host_arg_option from .control_plane.turn_driver.host_process_transport import ( - HOST_PROCESS_DRAINING, HOST_PROCESS_RECORD_ENV, host_process_drain, + HOST_PROCESS_DRAINING, HOST_PROCESS_RECORD_ENV, HOST_PROCESS_UNSUPPORTED_PLATFORM, + host_process_drain, ) from .control_plane.turn_driver.lane_fence import ( TURN_LANE_ABSENT, TURN_LANE_DEAD, TURN_LANE_LIVE, TURN_LANE_RELEASED, @@ -776,12 +777,15 @@ def _read_current(self, operation_id: str) -> dict: result["error"] = row["error"] return result - def _observe(self, path: Path, row: dict, status: str, **facts) -> None: + def _observe(self, path: Path, row: dict, status: str, *, already_locked: bool = False, **facts) -> None: decision = effect_runtime_result("collaboration.delegation.observe", { "from": row["status"], "to": status, **facts, }) row.update(status=decision["status"]) - self._fenced_write(path, row) + if already_locked: + self._fenced_write_locked(path, row) + else: + self._fenced_write(path, row) def _fenced_write(self, path: Path, row: dict) -> None: """Write the execution record only while no unacknowledged stop fences this process. @@ -791,10 +795,20 @@ def _fenced_write(self, path: Path, row: dict) -> None: """ with exclusive_file_lock(self._dispatch_lock(path)): - stop = self._read_stop(path) - if stop is not None and not self._acknowledged_here(stop): - raise DelegationFenced() - _write(path, row) + self._fenced_write_locked(path, row) + + def _fenced_write_locked(self, path: Path, row: dict) -> None: + """The same write for a caller that already holds the dispatch lock. + + The lock is a kernel file lock, so it is not reentrant: a caller that + widened its critical section to cover a whole effect group must use this + entry point rather than nesting ``_fenced_write``. + """ + + stop = self._read_stop(path) + if stop is not None and not self._acknowledged_here(stop): + raise DelegationFenced() + _write(path, row) @staticmethod def _acknowledged_here(stop: dict) -> bool: @@ -1094,10 +1108,35 @@ def _settle_stop(self, path: Path) -> dict: return self._stop_receipt(row, binding, None) if stop["phase"] not in DELEGATION_STOP_OPEN_PHASES: return self._stop_receipt(row, binding, stop) + # A launched Host on a platform that cannot prove its group exited + # has no converging stop: say so plainly rather than leaving the + # caller with an acknowledged receipt it can never settle. + unsupported = host_process_drain(self._host_process_record(path)) == HOST_PROCESS_UNSUPPORTED_PLATFORM + if unsupported: + raise ValueError( + "delegation stop cannot prove the launched Host drained on this platform: " + "process groups are unavailable, so the Host supervisor is best-effort. " + "Stop the member's Host through its own supervisor and re-read the receipt." + ) facts = {"operation_lock_free": self._operation_lock_free(path)} facts["worker_lane_released"], lane_state = self._worker_lane_released(row, stop, binding) # Read last: a Host seen drained after its worker and lane let go stays drained. facts["host_process"] = host_process_drain(self._host_process_record(path)) + # A required lease the acknowledgement could not release is retried + # here, under the same lock that guards the record: every later + # read of this stop is another attempt, instead of one failure + # turning into a permanent `settled` with the lease still held. + lease = stop.get("lease") if isinstance(stop.get("lease"), dict) else {} + if lease.get("required") is True and lease.get("released") is not True: + retried = self._release_delegation_lease(row, binding) + if retried != lease: + stop["lease"] = retried + _write(self._stop_path(path), stop) + lease = retried + if lease.get("required") is True: + # Absent means the operation held no required lease, which is not + # the same claim as a lease that was released. + facts["lease_released"] = lease.get("released") is True decision = effect_runtime_result("collaboration.delegation.stop", { "phase": stop["phase"], "acknowledged": stop.get("ack") is not None, "timed_out": time.time() - stop["requested_at"] > DELEGATION_STOP_GRACE_SECONDS, @@ -1106,10 +1145,8 @@ def _settle_stop(self, path: Path) -> dict: if decision["phase"] != stop["phase"] or decision.get("reason") != stop.get("reason"): stop.update(phase=decision["phase"], reason=decision.get("reason")) if decision["phase"] in DELEGATION_STOP_TERMINAL_PHASES: - lease = stop.get("lease") if isinstance(stop.get("lease"), dict) else {} stop["settled"] = { "at": time.time(), **facts, "lane_state": lane_state, - "lease_released": lease.get("released"), "turn_journal_status": self._turn_journal_status(row, binding), } _write(self._stop_path(path), stop) @@ -1553,30 +1590,38 @@ def _execute(self, path: Path, row: dict, binding: dict) -> None: return self._bound(row, require_active=True) # revocation or rebinding while the model ran delegation_results.require_dependencies(self, binding, delegation_results.operation_brief(self, row)) - if not todo_completed_for_settlement: + # Both effects commit inside the dispatch lock that a stop also + # takes, so the two sides linearize: either a stop is written first + # and neither effect runs, or both effects commit first and the stop + # that follows reports a record that already reached its terminal + # observation. Committing them outside the lock let a stop settle + # for a member whose Todo and reply had already landed. + with exclusive_file_lock(self._dispatch_lock(path)): self._raise_if_stop_requested(path) - self._complete_delegated_todo(row, binding) - row["artifacts"] = self._accepted(binding) - if not (_root(self.root) / "replies" / request_id / "conclusion.json").exists(): - return_result( - self.root, - self.goal_id, - binding["agent_id"], - request_id, - json.dumps( - { - "todo_id": binding["todo_id"], - "status": "accepted", - "artifacts": [ - {k: v for k, v in item.items() if k != "text"} - for item in row["artifacts"] - ], - } - ), - registry=self.registry, - caller_goal_ref=self._caller_goal_ref(), - ) - self._observe(path, row, "accepted", canonical_done=True, acceptance_ready=True, artifacts_current=True) + if not todo_completed_for_settlement: + self._complete_delegated_todo(row, binding) + row["artifacts"] = self._accepted(binding) + if not (_root(self.root) / "replies" / request_id / "conclusion.json").exists(): + return_result( + self.root, + self.goal_id, + binding["agent_id"], + request_id, + json.dumps( + { + "todo_id": binding["todo_id"], + "status": "accepted", + "artifacts": [ + {k: v for k, v in item.items() if k != "text"} + for item in row["artifacts"] + ], + } + ), + registry=self.registry, + caller_goal_ref=self._caller_goal_ref(), + ) + self._observe(path, row, "accepted", already_locked=True, + canonical_done=True, acceptance_ready=True, artifacts_current=True) except (ValueError, KeyError, subprocess.TimeoutExpired, EffectRuntimeRemoteError) as exc: # Retain uncertain execution for explicit same-operation recovery. # No fresh Turn is ever created because its client timed out. diff --git a/loopx/control_plane/collaboration/delegation.ts b/loopx/control_plane/collaboration/delegation.ts index bfed6e3ba7..1346045266 100644 --- a/loopx/control_plane/collaboration/delegation.ts +++ b/loopx/control_plane/collaboration/delegation.ts @@ -372,6 +372,13 @@ export function decideDelegationStop(params: JsonObject): JsonObject { "delegation stop release facts required"); requireThat(hostProcessDrains.includes(params.host_process as HostProcessDrain), "delegation stop host process drain fact required"); + // A required hard lease is released by the stop itself. Its release is part + // of what the receipt promises: a member whose lease is still held can block + // its Todo until the lease TTL, which is not a safe stop and is not something + // the owner can act on. `undefined` means the operation held no required + // lease. + requireThat(params.lease_released === undefined || typeof params.lease_released === "boolean", + "delegation stop lease release fact must be boolean"); requireThat(params.timed_out === undefined || typeof params.timed_out === "boolean", "delegation stop timeout fact must be boolean"); requireThat(phase !== "acknowledged" || params.acknowledged === true, @@ -380,16 +387,19 @@ export function decideDelegationStop(params: JsonObject): JsonObject { const workerReleased = operationFree && params.worker_lane_released === true; const host = params.host_process as HostProcessDrain; const hostDrained = host === "drained" || host === "not_launched"; + const leaseReleased = params.lease_released !== false; const pending = !operationFree ? "operation_lock_still_held" : params.worker_lane_released !== true ? "worker_lane_release_unproven" - : host === "draining" ? "host_process_still_running" : "host_process_drain_unproven"; + : host === "draining" ? "host_process_still_running" + : host !== "drained" && host !== "not_launched" ? "host_process_drain_unproven" + : "required_lease_release_unproven"; if (params.acknowledged === true) { - if (workerReleased && hostDrained) { + if (workerReleased && hostDrained && leaseReleased) { return {phase: "settled", terminal: true, reason: "acknowledged_worker_and_host_released"}; } return {phase: "acknowledged", terminal: false, reason: pending}; } - if (workerReleased && host !== "draining") { + if (workerReleased && host !== "draining" && leaseReleased) { return {phase: "unknown", terminal: true, reason: "holder_gone_without_acknowledgement"}; } return {phase: "requested", terminal: false, reason: operationFree ? pending diff --git a/loopx/control_plane/turn_driver/host_process_transport.py b/loopx/control_plane/turn_driver/host_process_transport.py index 5a6e250534..95405308c4 100644 --- a/loopx/control_plane/turn_driver/host_process_transport.py +++ b/loopx/control_plane/turn_driver/host_process_transport.py @@ -29,6 +29,19 @@ HOST_PROCESS_DRAINED = "drained" HOST_PROCESS_DRAINING = "draining" HOST_PROCESS_UNATTRIBUTABLE = "unattributable" +HOST_PROCESS_UNSUPPORTED_PLATFORM = "unsupported_platform" + + +def host_process_drain_supported() -> bool: + """Whether this platform can prove an owned Host group has exited. + + The TS supervisor cleans up a process group on POSIX and a process tree + best-effort on Windows. Only the group gives the stop a fact it can prove, + so the caller must say so rather than reporting a settlement it cannot + support. + """ + + return hasattr(os, "killpg") def _write_host_process_record(path: Path, record: dict[str, Any]) -> None: @@ -69,8 +82,12 @@ def host_process_drain(record_path: Path) -> str: except (OSError, ValueError): return HOST_PROCESS_UNATTRIBUTABLE if (not isinstance(record, dict) or record.get("schema_version") != HOST_PROCESS_RECORD_SCHEMA_VERSION - or record.get("host") != lock_holder_host_label() or not hasattr(os, "killpg")): + or record.get("host") != lock_holder_host_label()): return HOST_PROCESS_UNATTRIBUTABLE + if not hasattr(os, "killpg"): + # A launched Host on a platform without process groups is never proven + # drained. This is a platform boundary, not an attribution failure. + return HOST_PROCESS_UNSUPPORTED_PLATFORM bridge, group = record.get("bridge_pid"), record.get("process_group") if record.get("phase") != "finished": if not isinstance(bridge, int) or bridge <= 1: diff --git a/tests/control_plane_ts/delegation.test.ts b/tests/control_plane_ts/delegation.test.ts index f97a99c957..9533586679 100644 --- a/tests/control_plane_ts/delegation.test.ts +++ b/tests/control_plane_ts/delegation.test.ts @@ -158,6 +158,24 @@ test("a released worker and lane never settle a stop while the native Host still for (const host_process of ["drained", "not_launched"]) assert.deepEqual(decideDelegationStop({...released, host_process}), {phase: "settled", terminal: true, reason: "acknowledged_worker_and_host_released"}); + // A required lease the stop could not release is not a settlement: the + // member's Todo can stay blocked by it until the lease TTL. + for (const host_process of ["drained", "not_launched"]) { + assert.deepEqual(decideDelegationStop({...released, host_process, lease_released: false}), + {phase: "acknowledged", terminal: false, reason: "required_lease_release_unproven"}); + // An operation that held no required lease omits the fact, which settles. + assert.deepEqual(decideDelegationStop({...released, host_process}), + {phase: "settled", terminal: true, reason: "acknowledged_worker_and_host_released"}); + } + assert.throws(() => decideDelegationStop({...released, host_process: "drained", lease_released: "yes"}), + /lease release fact/); + // A vanished holder whose required lease is still held is not terminal either. + const unacked = {phase: "requested", acknowledged: false, operation_lock_free: true, + worker_lane_released: true, host_process: "drained"}; + assert.deepEqual(decideDelegationStop({...unacked, lease_released: false}), + {phase: "requested", terminal: false, reason: "required_lease_release_unproven"}); + assert.deepEqual(decideDelegationStop(unacked), + {phase: "unknown", terminal: true, reason: "holder_gone_without_acknowledgement"}); // A held lock still dominates a drained Host. assert.deepEqual(decideDelegationStop({...released, operation_lock_free: false, host_process: "drained"}), {phase: "acknowledged", terminal: false, reason: "operation_lock_still_held"}); diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index 871b2517a8..1448e13b56 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -17,11 +17,12 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "examples" / "managed-research-team")) import research_team as demo # noqa: E402 from test_managed_research_scenario import fixture # noqa: E402 -from loopx.collaboration_mcp import DelegationFenced, Delegations # noqa: E402 +from loopx.collaboration_mcp import DelegationFenced, DelegationStopRequested, Delegations # noqa: E402 from loopx.control_plane.collaboration.peers import returns # noqa: E402 from loopx.control_plane.collaboration.inbox import _read # noqa: E402 from loopx.control_plane.turn_driver.lane_fence import turn_lane_liveness, turn_lane_singleflight # noqa: E402 from loopx.file_lock import exclusive_file_lock, try_exclusive_file_lock # noqa: E402 +from loopx.file_lock import lock_holder_host_label # noqa: E402 HOST = '''import json, os, sys, time @@ -621,3 +622,129 @@ def test_interrupted_host_cleanup_keeps_the_stop_open_until_a_reread_sees_it_dra assert int((root / "analyst" / "initial" / "host-invocations").read_text()) == 1 assert not demo.canonical_tasks(root)["todo_analyst-initial"]["done"] assert returns(runner.root, runner.goal_id, "lead")["items"] == [] + + +def test_a_stop_written_before_the_effects_linearizes_against_them(service, monkeypatch): + """Todo completion and reply publication commit under the stop's own lock. + + The pre-check alone was not a boundary: the worker could pass it, a stop + could be persisted, and both external effects still committed before the + fenced record write, leaving `settled`/`stopped` for a member whose work had + landed. Holding the dispatch lock across both effects makes the two sides + linearize in either order. + """ + from loopx.control_plane.collaboration.inbox import _write as write_inbox + + root, runner = service + runner.start("analysis", "analysis-race", brief()) + path = runner.path("analysis-race") + row = _read(path) + # Reproduce the interleaving: the stop exists before the worker reaches its + # effects. A same-process request is what the worker then acknowledges. + write_inbox(runner._stop_path(path), runner._new_stop_record( + row, requested_by=runner.agent_id, worker=None)) + + # The checkpoint after the model returns refuses to run the effects. + with pytest.raises(DelegationStopRequested): + with exclusive_file_lock(runner._dispatch_lock(path)): + runner._raise_if_stop_requested(path) + + runner.execute("analysis-race") + assert _read(path)["status"] == "stopped" + # Neither effect committed for a stop that was written first. + assert not demo.canonical_tasks(root)["todo_analyst-initial"]["done"] + assert not (root / "runtime" / "replies" / "analysis-race" / "conclusion.json").exists() + assert not (root / "host-started").exists() + receipt = runner.stop("analysis-race", execute=True) + assert receipt["phase"] == "settled" and receipt["status"] == "stopped" + + +def test_a_failed_required_lease_release_keeps_the_stop_open_and_retries(service, monkeypatch): + """A required lease the stop could not release is not a settlement. + + The member's Todo can stay blocked by that lease until its TTL, so reporting + `settled` would be a terminal claim the owner cannot act on. The release is + retried under the stop's own lock on the next read instead of being attempted + once and forgotten. + """ + from loopx import collaboration_mcp as delegation + from loopx.control_plane.collaboration.inbox import _write as write_inbox + + root, runner = service + monkeypatch.setattr(runner, "_spawn", lambda _: None) + runner.start("analysis", "analysis-lease", brief()) + path = runner.path("analysis-lease") + row = _read(path) + # A bounded member task holds a required lease; make the record say so. + row["task_lease"] = {"required": True, "idempotency_key": "lease-1", "version": 1} + row["status"] = "stopped" + runner._fenced_write(path, row) + write_inbox(runner._stop_path(path), { + **runner._new_stop_record(row, requested_by=runner.agent_id, worker=None), + "phase": "acknowledged", + "ack": {"pid": os.getpid(), "host": lock_holder_host_label(), "at": time.time(), + "source": "requester", "observed_status": "stopped", "turn_key": None}, + # The acknowledgement could not release it, which is what the receipt records. + "lease": {"required": True, "released": False, "error": "authority unavailable"}, + }) + + attempts = [] + outcomes = [RuntimeError("authority temporarily unavailable"), {"released": True}] + + def flaky_release(**kwargs): + attempts.append(kwargs) + outcome = outcomes[min(len(attempts) - 1, len(outcomes) - 1)] + if isinstance(outcome, Exception): + raise outcome + return outcome + + monkeypatch.setattr(delegation, "release_task_lease", flaky_release) + + # The first settle read tries the release, fails, and must not settle. + receipt = runner.stop("analysis-lease", execute=True) + assert attempts, "the stop never attempted the required release" + assert receipt["phase"] == "acknowledged", receipt + assert receipt["stop"]["reason"] == "required_lease_release_unproven" + assert receipt["stop"]["lease"]["released"] is not True + + # The next read retries the release and only then settles. + settled = runner.stop("analysis-lease", execute=True) + assert len(attempts) >= 2, "the failed release was never retried" + assert settled["phase"] == "settled", settled + assert settled["stop"]["lease"]["released"] is True + assert settled["stop"]["settled"]["lease_released"] is True + + # Once settled the receipt is stable, and a released lease is not re-attempted. + before = len(attempts) + assert runner.stop("analysis-lease", execute=True) == settled + assert len(attempts) == before + + +def test_a_launched_host_on_a_platform_without_process_groups_fails_fast(service, monkeypatch): + """A stop that cannot prove its Host drained says so instead of never settling. + + Windows cleanup is process-tree best effort, so no fact proves the Host's + descendants exited. An acknowledged receipt the caller can never settle is + worse than an actionable refusal that names the platform boundary. + """ + from loopx.control_plane.turn_driver import host_process_transport + + root, runner = service + monkeypatch.setattr(runner, "_spawn", lambda _: None) + runner.start("analysis", "analysis-platform", brief()) + path = runner.path("analysis-platform") + record = runner._host_process_record(path) + record.parent.mkdir(parents=True, exist_ok=True) + record.write_text(json.dumps({ + "schema_version": host_process_transport.HOST_PROCESS_RECORD_SCHEMA_VERSION, + "host": lock_holder_host_label(), "phase": "finished", + "bridge_pid": os.getpid(), "process_group": os.getpid(), + })) + + # The launched Host is real, but this platform cannot prove it drained. + assert host_process_transport.host_process_drain(record) == host_process_transport.HOST_PROCESS_DRAINED + monkeypatch.delattr(host_process_transport.os, "killpg", raising=False) + assert host_process_transport.host_process_drain(record) == host_process_transport.HOST_PROCESS_UNSUPPORTED_PLATFORM + + with pytest.raises(ValueError, match="cannot prove the launched Host drained"): + runner.stop("analysis-platform", execute=True) From 00442681f48ce729aa71218991720a939ad1db65 Mon Sep 17 00:00:00 2001 From: song Date: Wed, 30 Sep 2026 14:36:34 -0400 Subject: [PATCH 13/54] fix(delegation): take the lease obligation from the operation record MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `_acknowledge_stop` persists the ACK before it releases the hard lease, so a process loss between those two writes leaves the sidecar with no `lease` field at all. `_settle_stop` read the obligation from that field, took the absence as "no required lease", and settled: the caller was told the member is safely stopped while the native authority still reported its lease active, with resume already refused and the Todo blocked until the TTL. Reproduced on both File and SQLite by acquiring a real required lease and losing the process at the release entry. The obligation now comes from the operation record — the canonical `row.task_lease.required`, which the crash cannot lose. Absent sidecar state means "the release result has not been written yet", not "nothing was owed", so the release is attempted and only an actual release lets the decision settle. That is one source of truth rather than a second boolean kept in sync with it; the released fact is still persisted beside the stop so a later read does not release twice. Coverage pins both ends of the window on both authorities: - a crash after the ACK with the lease still owed, and a release that succeeds on the next read, settles and reports `lease_released: true`; - the same window with the release still failing stays `acknowledged` with `required_lease_release_unproven` instead of reporting `settled`. Mutation-checked: deriving the obligation from the sidecar again fails all four. Signed-off-by: song --- docs/reference/local-delegation.md | 12 +++-- loopx/collaboration_mcp.py | 22 +++++---- tests/test_local_delegation.py | 76 ++++++++++++++++++++++++++++++ 3 files changed, 98 insertions(+), 12 deletions(-) diff --git a/docs/reference/local-delegation.md b/docs/reference/local-delegation.md index a32ff944e2..bb715f31c9 100644 --- a/docs/reference/local-delegation.md +++ b/docs/reference/local-delegation.md @@ -236,7 +236,10 @@ refused. The receipt `phase` is `settled` only when an acknowledgement exists, the operation lock is free, the member's Turn lane holder record shows it released by the stopped worker (the lane is read, never taken), the native host the Turn launched has exited together with every process in its group, -and a required hard task lease was actually released. A release that failed is +and a required hard task lease was actually released. The obligation is read +from the operation record, not from the stop sidecar: the acknowledgement is +persisted before the lease is released, so a crash in between must not turn +"not yet written" into "nothing was owed". A release that failed is retried under the stop's own lock on the next read, so it never becomes a `settled` receipt that leaves the member's Todo blocked until the lease TTL; while it is unproven the stop stays `acknowledged` with @@ -266,9 +269,10 @@ written after both leaves their acceptance intact. worker 不会被发信号,它在下一个检查点或下一次写记录时发现请求,写入被拒绝。 只有存在确认、operation 锁已释放、成员 Turn lane 的持有者记录显示已被停止的 worker 释放(只读 lane,从不获取)、该 Turn 启动的原生 host 及其进程组内所有进程 -都已退出,且必需的硬任务租约确实释放成功时,`phase` 才是 `settled`。释放失败会在 -下一次读取时于 stop 自己的锁下重试,因此不会产生一份「已结算」却让成员 Todo 被 -租约阻塞到 TTL 的回执;在释放得到证明前,停止保持 `acknowledged`,原因为 +都已退出,且必需的硬任务租约确实释放成功时,`phase` 才是 `settled`。该义务取自 +操作记录而非 stop sidecar:确认会先于释放落盘,因此两者之间发生崩溃时,不能把 +「尚未写入」当成「本就不需要释放」。释放失败会在下一次读取时于 stop 自己的锁下重试, +因此不会产生一份「已结算」却让成员 Todo 被租约阻塞到 TTL 的回执;在释放得到证明前,停止保持 `acknowledged`,原因为 `required_lease_release_unproven`。host 无法归属或其 supervisor 未完成清理时, 停止保持 `acknowledged`,之后再次调用 `stop` 会重新读取。在没有进程组的平台上, 启动过的 host 根本无法被证明已收尾,因此 `stop --execute` 会以指明该平台边界的 diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index 82d1906a70..fe308760c8 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -1122,20 +1122,26 @@ def _settle_stop(self, path: Path) -> dict: facts["worker_lane_released"], lane_state = self._worker_lane_released(row, stop, binding) # Read last: a Host seen drained after its worker and lane let go stays drained. facts["host_process"] = host_process_drain(self._host_process_record(path)) - # A required lease the acknowledgement could not release is retried - # here, under the same lock that guards the record: every later - # read of this stop is another attempt, instead of one failure - # turning into a permanent `settled` with the lease still held. + # The obligation comes from the operation record, not from the stop + # sidecar. The acknowledgement writes its ACK before it releases the + # lease, so a process loss in between leaves the sidecar with no + # `lease` field at all — and reading that as "nothing was owed" + # settles a stop whose member still holds an active hard lease, with + # resume already refused and the Todo blocked until the TTL. + # + # The canonical `row.task_lease.required` is the source of truth, so + # the obligation survives the crash. A release is retried here, under + # the same lock that guards the record: every later read is another + # attempt rather than one failure becoming permanent. + owed = isinstance(row.get("task_lease"), dict) and row["task_lease"].get("required") is True lease = stop.get("lease") if isinstance(stop.get("lease"), dict) else {} - if lease.get("required") is True and lease.get("released") is not True: + if owed and lease.get("released") is not True: retried = self._release_delegation_lease(row, binding) if retried != lease: stop["lease"] = retried _write(self._stop_path(path), stop) lease = retried - if lease.get("required") is True: - # Absent means the operation held no required lease, which is not - # the same claim as a lease that was released. + if owed: facts["lease_released"] = lease.get("released") is True decision = effect_runtime_result("collaboration.delegation.stop", { "phase": stop["phase"], "acknowledged": stop.get("ack") is not None, diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index 1448e13b56..51733ca484 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -17,6 +17,7 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "examples" / "managed-research-team")) import research_team as demo # noqa: E402 from test_managed_research_scenario import fixture # noqa: E402 +from loopx import collaboration_mcp as delegation_module # noqa: E402 from loopx.collaboration_mcp import DelegationFenced, DelegationStopRequested, Delegations # noqa: E402 from loopx.control_plane.collaboration.peers import returns # noqa: E402 from loopx.control_plane.collaboration.inbox import _read # noqa: E402 @@ -748,3 +749,78 @@ def test_a_launched_host_on_a_platform_without_process_groups_fails_fast(service with pytest.raises(ValueError, match="cannot prove the launched Host drained"): runner.stop("analysis-platform", execute=True) + + +def test_a_crash_between_the_ack_and_the_lease_result_keeps_the_stop_open(service, monkeypatch): + """The lease obligation survives process loss after the acknowledgement. + + `_acknowledge_stop` writes the ACK before it releases the lease, so a crash + in between leaves the sidecar with no `lease` field. Reading that as "nothing + was owed" settles a stop whose member still holds an active hard lease, with + resume already refused and the Todo blocked until the TTL. The obligation + comes from the operation record, which the crash cannot lose. + """ + from loopx.control_plane.collaboration.inbox import _write as write_inbox + + root, runner = service + monkeypatch.setattr(runner, "_spawn", lambda _: None) + runner.start("analysis", "analysis-crash", brief()) + path = runner.path("analysis-crash") + row = _read(path) + # The member holds a real required lease, as a bounded task does. + row["task_lease"] = {"required": True, "idempotency_key": "lease-crash", "version": 1} + right_after_ack = {**row, "status": "stopped"} + runner._fenced_write(path, right_after_ack) + + # The crash window: ACK persisted, no lease result written yet. + write_inbox(runner._stop_path(path), { + **runner._new_stop_record(right_after_ack, requested_by=runner.agent_id, worker=None), + "phase": "acknowledged", + "ack": {"pid": os.getpid(), "host": lock_holder_host_label(), "at": time.time(), + "source": "requester", "observed_status": "stopped", "turn_key": None}, + }) + sidecar = _read(runner._stop_path(path)) + assert "lease" not in sidecar or sidecar.get("lease") is None + assert _read(path)["task_lease"]["required"] is True + + # A release that succeeds on this read lets the stop settle, and the receipt + # says the lease really is gone. + monkeypatch.setattr(delegation_module, "release_task_lease", + lambda **kw: {"released": True}) + settled = runner.stop("analysis-crash", execute=True) + assert settled["phase"] == "settled", settled + assert settled["stop"]["lease"]["released"] is True + assert settled["stop"]["settled"]["lease_released"] is True + + +def test_a_crash_between_the_ack_and_the_lease_result_never_settles_unreleased(service, monkeypatch): + """The same window, with the release still failing, must not report `settled`. + + `settled` tells the owner the member is safely stopped. Claiming it while a + required lease is provably still active is exactly the terminal distortion + the crash window used to produce. + """ + from loopx.control_plane.collaboration.inbox import _write as write_inbox + + root, runner = service + monkeypatch.setattr(runner, "_spawn", lambda _: None) + runner.start("analysis", "analysis-crash-open", brief()) + path = runner.path("analysis-crash-open") + row = _read(path) + row["task_lease"] = {"required": True, "idempotency_key": "lease-open", "version": 1} + right_after_ack = {**row, "status": "stopped"} + runner._fenced_write(path, right_after_ack) + write_inbox(runner._stop_path(path), { + **runner._new_stop_record(right_after_ack, requested_by=runner.agent_id, worker=None), + "phase": "acknowledged", + "ack": {"pid": os.getpid(), "host": lock_holder_host_label(), "at": time.time(), + "source": "requester", "observed_status": "stopped", "turn_key": None}, + }) + + def still_held(**kwargs): + raise RuntimeError("authority unavailable") + + monkeypatch.setattr(delegation_module, "release_task_lease", still_held) + receipt = runner.stop("analysis-crash-open", execute=True) + assert receipt["phase"] == "acknowledged", receipt + assert receipt["stop"]["reason"] == "required_lease_release_unproven" From 7f90c629f03ce4e4f5d7529c467a6b4a16a692c0 Mon Sep 17 00:00:00 2001 From: song Date: Thu, 1 Oct 2026 05:42:47 -0400 Subject: [PATCH 14/54] fix(delegation): fence recovered completion against concurrent stop Signed-off-by: song --- docs/reference/local-delegation.md | 7 ++ loopx/collaboration_mcp.py | 80 ++++++++------- tests/test_delegation_stop_recovery.py | 129 +++++++++++++++++++++++++ 3 files changed, 174 insertions(+), 42 deletions(-) create mode 100644 tests/test_delegation_stop_recovery.py diff --git a/docs/reference/local-delegation.md b/docs/reference/local-delegation.md index 5462a62edd..c714cea027 100644 --- a/docs/reference/local-delegation.md +++ b/docs/reference/local-delegation.md @@ -259,6 +259,10 @@ member's Todo stays open, so the coordinator decides what happens next. The member's Todo completion and reply publication commit under the same lock a stop takes, so a stop written first means neither effect lands, and a stop written after both leaves their acceptance intact. +Recovery of an already validated Turn uses the same fenced completion entry: +it retains the lock through the original Turn's settlement and result publication, +without rerunning the Host. A competing stop waits for that acceptance or wins +before completion starts; a lock-acquisition timeout requires retrying `stop`. 中文:`stop --execute` 结束一个成员的有界工作,并返回一份只陈述已证明事实的 回执。停止请求写在执行记录旁边的 `.stop.json`,从不写进记录本身, @@ -283,6 +287,9 @@ operation id。Turn journal 保留 `in_progress` 条目供检查,记录不会 成员的 Todo 仍然打开,由协调者决定下一步。成员的 Todo 完成与回执发布在 stop 所取的同一把锁下提交,因此先写入停止则两个效果都不会落地,后写入停止则其验收结果 保持不变。 +恢复已通过验证的 Turn 也使用同一个带锁的完成入口,直到原 Turn 结算与结果发布 +结束才释放锁,不会重新运行 Host。并发 stop 等待该验收结果,或在完成开始前先取得 +停止边界;获取锁超时则需要重试 `stop`。 This entrypoint does not create Agents, grant bindings or wake an idle Codex conversation. The existing host/LoopX continuation policy owns the next lead diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index bcbce71d50..2cd701e3a5 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -1346,7 +1346,8 @@ def _execution_arguments(self, binding: dict, operation_id: str) -> list[str]: *binding["host_args"]] def _record_turn_result( - self, path: Path, row: dict, result: dict, *, publish: bool = True + self, path: Path, row: dict, result: dict, *, publish: bool = True, + already_locked: bool = False, ) -> None: turn_key = result.get("resume_turn_key") if turn_key: @@ -1366,7 +1367,9 @@ def _record_turn_result( ) } if publish: - self._observe(path, row, "turn_returned") + self._observe(path, row, "turn_returned", already_locked=already_locked) + elif already_locked: + self._fenced_write_locked(path, row) else: self._fenced_write(path, row) @@ -1601,9 +1604,9 @@ def _execute(self, path: Path, row: dict, binding: dict) -> None: # an otherwise clean Git worktree fail canonical validation. self._clear_delegation_bootstrap(row, binding) try: - todo_completed_for_settlement = False result = row["turn_result"] - if result.get("status") != "committed" or result.get("result_kind") != "validated_progress": + needs_settlement = result.get("status") != "committed" or result.get("result_kind") != "validated_progress" + if needs_settlement: journal = self._validated_turn_journal(row, binding) if journal is None: row["error"] = str( @@ -1613,57 +1616,50 @@ def _execute(self, path: Path, row: dict, binding: dict) -> None: )[:180] self._observe(path, row, "rejected") return - if not self._receiver_adopted(row, binding): - row["error"] = "delegation receiver did not adopt the request" - self._observe(path, row, "rejected") - return - self._bound(row, require_active=True) - delegation_results.require_dependencies( - self, binding, delegation_results.operation_brief(self, row) - ) - if not isinstance(row.get("task_lease"), dict): - self._acquire_delegation_lease(path, row, binding) - self._raise_if_stop_requested(path) - self._complete_delegated_todo(row, binding) - todo_completed_for_settlement = True - self._raise_if_stop_requested(path) - result = self._cli( - binding, - "turn", - "run-once", - *common, - "--resume-turn-key", - row["turn_key"], - *execution, - "--execute", - timeout=binding["timeout_seconds"] + 60, - host_record=self._host_process_record(path), - ) - self._record_turn_result(path, row, result, publish=False) - if result.get("status") != "committed" or result.get("result_kind") != "validated_progress": - raise ValueError( - str( - result.get("error") - or result.get("reason") - or "validated delegation settlement remains incomplete" - ) - ) if not self._receiver_adopted(row, binding): row["error"] = "delegation receiver did not adopt the request" self._observe(path, row, "rejected") return self._bound(row, require_active=True) # revocation or rebinding while the model ran delegation_results.require_dependencies(self, binding, delegation_results.operation_brief(self, row)) + if needs_settlement and not isinstance(row.get("task_lease"), dict): + self._acquire_delegation_lease(path, row, binding) # Both effects commit inside the dispatch lock that a stop also # takes, so the two sides linearize: either a stop is written first # and neither effect runs, or both effects commit first and the stop # that follows reports a record that already reached its terminal # observation. Committing them outside the lock let a stop settle - # for a member whose Todo and reply had already landed. + # for a member whose Todo and reply had already landed. Recovery of + # a validated journal uses this same completion entry, retaining the + # fence through settlement, result publication and acceptance. with exclusive_file_lock(self._dispatch_lock(path)): self._raise_if_stop_requested(path) - if not todo_completed_for_settlement: - self._complete_delegated_todo(row, binding) + self._complete_delegated_todo(row, binding) + if needs_settlement: + result = self._cli( + binding, + "turn", + "run-once", + *common, + "--resume-turn-key", + row["turn_key"], + *execution, + "--execute", + timeout=binding["timeout_seconds"] + 60, + host_record=self._host_process_record(path), + ) + self._record_turn_result(path, row, result, publish=False, already_locked=True) + if result.get("status") != "committed" or result.get("result_kind") != "validated_progress": + raise ValueError(str(result.get("error") or result.get("reason") + or "validated delegation settlement remains incomplete")) + if not self._receiver_adopted(row, binding): + row["error"] = "delegation receiver did not adopt the request" + self._observe(path, row, "rejected", already_locked=True) + return + self._bound(row, require_active=True) + delegation_results.require_dependencies( + self, binding, delegation_results.operation_brief(self, row) + ) row["artifacts"] = self._accepted(binding) if not (_root(self.root) / "replies" / request_id / "conclusion.json").exists(): return_result( diff --git a/tests/test_delegation_stop_recovery.py b/tests/test_delegation_stop_recovery.py new file mode 100644 index 0000000000..37c21ea033 --- /dev/null +++ b/tests/test_delegation_stop_recovery.py @@ -0,0 +1,129 @@ +"""Stop ordering against native completion of a validated, unsettled Turn.""" + +import sys +from concurrent.futures import ThreadPoolExecutor, TimeoutError as FutureTimeout +from contextlib import contextmanager +from threading import Event, get_ident + +import pytest + +from test_local_delegation import brief, demo, service as service +from loopx import collaboration_mcp as delegation +from loopx.control_plane.collaboration.inbox import _read +from loopx.control_plane.collaboration.peers import returns +from loopx.file_lock import exclusive_file_lock + + +def recoverable_boundary(service, monkeypatch): + root, runner = service + # The Host adopts and supplies an artifact, but only delegation publishes + # the result. Keep that effect independently observable in this fixture. + host = root / "fixture-host.py" + host.write_text("\n".join(line for line in host.read_text().splitlines() + if not line.startswith(" return_result("))) + monkeypatch.setattr(runner, "_spawn", lambda _: None) + runner.start("analysis", "analysis-recovery", brief()) + module_command = delegation._python_module_command + # Exit the real native CLI after its validated checkpoint is persisted, + # before any settlement callback runs. Host output and task validation are + # real; completion, settlement and authority are never replaced. + interruption = """ +import os, runpy +from loopx.control_plane.turn_driver import executor +persist = executor._write_journal +def checkpoint(path, snapshot, **kwargs): + persist(path, snapshot, **kwargs) + if snapshot.get('completed_phases') == ['host_execute', 'typed_result', 'validation']: + os._exit(86) +executor._write_journal = checkpoint +runpy.run_module('loopx.cli', run_name='__main__') +""" + with monkeypatch.context() as setup: + setup.setattr(delegation, "_python_module_command", lambda module: + [sys.executable, "-c", interruption] if module == "loopx.cli" + else module_command(module)) + runner.execute("analysis-recovery") + path = runner.path("analysis-recovery") + row = _read(path) + assert row["status"] == "running" and row.get("error") + # The interrupted CLI did not return an accepted result. Record that + # observation through the owner, then use public same-operation resume; + # the typed recovery transition accepts only a rejected observation. + runner._observe(path, row, "rejected") + assert runner.resume("analysis-recovery")["status"] == "turn_returned" + row = _read(path) + journal = runner._validated_turn_journal(row, runner.binding("analysis")) + assert journal["task_validation"]["ok"] is True + assert journal["host_result"]["result_kind"] == "validated_progress" + assert not demo.canonical_tasks(root)["todo_analyst-initial"]["done"] + assert returns(runner.root, runner.goal_id, "lead")["items"] == [] + # The test process owns the operation; it must never signal its own group. + monkeypatch.setattr(runner, "_signal_worker", lambda *_: None) + return root, runner + + +def test_stop_wins_before_recovered_completion(service, monkeypatch): + root, runner = recoverable_boundary(service, monkeypatch) + validated = runner._validated_turn_journal + receipts = [] + + def stop_after_validation(row, binding): + journal = validated(row, binding) + assert journal is not None + receipts.append(runner.stop("analysis-recovery", execute=True)) + return journal + + monkeypatch.setattr(runner, "_validated_turn_journal", stop_after_validation) + runner.execute("analysis-recovery") + assert len(receipts) == 1 and receipts[0]["phase"] == "requested" + receipt = runner.stop("analysis-recovery", execute=True) + assert receipt["phase"] == "settled" and receipt["status"] == "stopped" + assert not demo.canonical_tasks(root)["todo_analyst-initial"]["done"] + assert returns(runner.root, runner.goal_id, "lead")["items"] == [] + assert (root / "analyst" / "initial" / "host-invocations").read_text() == "1" + with pytest.raises(ValueError, match="start a new operation id"): + runner.resume("analysis-recovery") + + +def test_recovered_completion_wins_over_a_concurrent_stop(service, monkeypatch): + root, runner = recoverable_boundary(service, monkeypatch) + complete = runner._complete_delegated_todo + main_thread = get_ident() + attempted = Event() + dispatch = runner._dispatch_lock(runner.path("analysis-recovery")) + futures = [] + + @contextmanager + def observed_lock(target, **kwargs): + if target == dispatch and get_ident() != main_thread: + attempted.set() + kwargs["timeout_seconds"] = 30 + with exclusive_file_lock(target, **kwargs) as held: + yield held + + monkeypatch.setattr(delegation, "exclusive_file_lock", observed_lock) + with ThreadPoolExecutor(max_workers=1) as pool: + def concurrent_stop(row, binding): + future = pool.submit(runner.stop, "analysis-recovery", execute=True) + futures.append(future) + assert attempted.wait(10) + # Give stop the chance to persist if completion has no fence. + # Do not assert the lock implementation: run native completion and + # judge the final canonical state and public receipt instead. + try: + future.result(timeout=0.3) + except FutureTimeout: + pass + complete(row, binding) + + monkeypatch.setattr(runner, "_complete_delegated_todo", concurrent_stop) + runner.execute("analysis-recovery") + assert len(futures) == 1 + receipt = futures[0].result(timeout=30) + receipt = runner.stop("analysis-recovery", execute=True) + assert demo.canonical_tasks(root)["todo_analyst-initial"]["done"] + assert receipt["phase"] == "noop" and receipt["status"] == "accepted", receipt + assert runner.read("analysis-recovery")["status"] == "accepted" + assert len(returns(runner.root, runner.goal_id, "lead")["items"]) == 1 + assert not runner._stop_path(runner.path("analysis-recovery")).exists() + assert (root / "analyst" / "initial" / "host-invocations").read_text() == "1" From 9c64053cbfb39a234b6a8faec4931da5cb52e000 Mon Sep 17 00:00:00 2001 From: song Date: Thu, 1 Oct 2026 07:39:32 -0400 Subject: [PATCH 15/54] fix(workspace): report stopped delegation records without claiming drain Signed-off-by: song --- apps/presentation/dashboard/src/data/chat.ts | 3 +++ .../personal-workspace-browser/team-evidence.mjs | 12 ++++++++++-- 2 files changed, 13 insertions(+), 2 deletions(-) diff --git a/apps/presentation/dashboard/src/data/chat.ts b/apps/presentation/dashboard/src/data/chat.ts index c7b57a1519..ae566dc507 100644 --- a/apps/presentation/dashboard/src/data/chat.ts +++ b/apps/presentation/dashboard/src/data/chat.ts @@ -1111,6 +1111,9 @@ export function delegationStateLabel(row: {status: string; worker_active?: boole if (row.status === "unavailable") return zh ? "无法核验" : "Unavailable"; if (row.status === "accepted") return zh ? "已通过当前验收" : "Currently accepted"; if (row.status === "rejected") return zh ? "未通过验收" : "Rejected"; + // The operation may be marked stopped before its Host group fully drains. + // Only the separate stop receipt proves that execution has been released. + if (row.status === "stopped") return zh ? "停止已登记" : "Stop recorded"; if (row.recovery_required) return zh ? "需要恢复原执行" : "Original execution needs recovery"; if (row.status === "running" && row.worker_active) return zh ? "执行中" : "Executing"; if (row.status === "turn_returned" && row.worker_active) return zh ? "正在验收" : "Validating"; diff --git a/examples/personal-workspace-browser/team-evidence.mjs b/examples/personal-workspace-browser/team-evidence.mjs index 4b4b18ba6b..74442a2b5b 100644 --- a/examples/personal-workspace-browser/team-evidence.mjs +++ b/examples/personal-workspace-browser/team-evidence.mjs @@ -26,8 +26,10 @@ export const teamEvidenceScenario = { ? {items: [{record_id: "a".repeat(64), operation_id: "accepted-analysis", agent_id: "local-analyst", status: "accepted", recovery_required: false, artifacts: [{ref: "report.md", sha256: "9".repeat(64)}]}], has_more: false, next_cursor: null, page_readback_complete: true} - : {items: [{record_id: "0".repeat(64), operation_id: null, status: "unavailable", recovery_required: null}], - has_more: true, next_cursor: "0".repeat(64), page_readback_complete: false}}); + : {items: [{record_id: "0".repeat(64), operation_id: null, status: "unavailable", recovery_required: null}, + {record_id: "1".repeat(64), operation_id: "stopped-analysis", agent_id: "local-analyst", + status: "stopped", worker_active: false, recovery_required: false}], + has_more: true, next_cursor: "1".repeat(64), page_readback_complete: false}}); }; await page.route("**/api/chat/sessions/*/loopx", laterAcceptedPage); Object.assign(mode, {enabled: true, paused: false, active_turn_id: "fixture-loopx-turn", native: {status: "active", tokenBudget: 100000}}); @@ -39,6 +41,12 @@ export const teamEvidenceScenario = { assert.equal(await results.getByLabel("当前报告").evaluate(el => el === document.activeElement), false, "Automatic readback must not steal focus"); assert.equal(inspectedRequests, 2, "Accepted work after an unreadable first page should still be discovered without another click"); + await page.getByRole("button", {name: "团队执行情况", exact: true}).click(); + const executionDetails = page.getByRole("dialog", {name: "团队执行情况", exact: true}); + await executionDetails.getByText("local-analyst · 停止已登记", {exact: true}).waitFor(); + assert(!(await executionDetails.innerText()).includes("执行已释放"), + "A stopped operation row alone must not claim its Host group is drained"); + await executionDetails.getByRole("button", {name: "关闭", exact: true}).click(); const goalNav = page.getByRole("navigation", {name: "Goal 视图"}); await goalNav.getByRole("button", {name: "成果", exact: true}).click(); const fileResults = page.getByRole("region", {name: "团队成果", exact: true}); From 0ec7c16205e99c6969bce1e3acbbd1f80cfe0d19 Mon Sep 17 00:00:00 2001 From: song Date: Fri, 2 Oct 2026 01:48:01 -0400 Subject: [PATCH 16/54] fix(dashboard): give the recorded stop its own pulse label The stop label is published verbatim in the execution dialog, so the pulse bucket uses distinct wording instead of repeating the same public string. Signed-off-by: song --- .../src/features/personal-workspace/goal-team-work.tsx | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/apps/presentation/dashboard/src/features/personal-workspace/goal-team-work.tsx b/apps/presentation/dashboard/src/features/personal-workspace/goal-team-work.tsx index 92fcda617c..f9373651bc 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/goal-team-work.tsx +++ b/apps/presentation/dashboard/src/features/personal-workspace/goal-team-work.tsx @@ -21,7 +21,7 @@ const PULSE_LABELS: Record = { validating: {zh: "正在验收", en: "Validating"}, accepted: {zh: "已通过", en: "Accepted"}, attention: {zh: "需要处理", en: "Needs attention"}, - stopped: {zh: "停止已登记", en: "Stop recorded"}, + stopped: {zh: "已登记停止", en: "Stop on record"}, dispatched: {zh: "等待回读", en: "Awaiting readback"}, unknown: {zh: "状态未知", en: "Unknown"}, }; From 80f5490a4450d12d4bf0d0cebab183bca94f43da Mon Sep 17 00:00:00 2001 From: song Date: Fri, 2 Oct 2026 19:22:25 +0800 Subject: [PATCH 17/54] fix(delegation): restore the dashboard build and the spawned-callback test Drop the stray closing brace a merge left in chat.ts (TS1128 at build), and pass the spawned callback in runHostProcess's fifth slot instead of the grace. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: song --- apps/presentation/dashboard/src/data/chat.ts | 1 - tests/control_plane_ts/host_process.test.ts | 4 ++-- 2 files changed, 2 insertions(+), 3 deletions(-) diff --git a/apps/presentation/dashboard/src/data/chat.ts b/apps/presentation/dashboard/src/data/chat.ts index 54a9556eba..5a48cf9cf2 100644 --- a/apps/presentation/dashboard/src/data/chat.ts +++ b/apps/presentation/dashboard/src/data/chat.ts @@ -1142,7 +1142,6 @@ export function delegationStateLabel(row: DelegationStateFacts, zh: boolean) { const label = DELEGATION_STATE_LABELS[delegationState(row)]; return zh ? label.zh : label.en; } -} export function fetchLoopXTeamWork(sessionId: string, cursor?: string) { return requestJson(`/api/chat/sessions/${sessionId}/loopx`, { method: "POST", body: JSON.stringify({operation: "operations", limit: 10, ...(cursor ? {cursor} : {})}), diff --git a/tests/control_plane_ts/host_process.test.ts b/tests/control_plane_ts/host_process.test.ts index 6a569ffbde..3c90a25b40 100644 --- a/tests/control_plane_ts/host_process.test.ts +++ b/tests/control_plane_ts/host_process.test.ts @@ -122,7 +122,7 @@ test("the spawned Host group is reported once before input, and an unrecorded gr const seen: unknown[] = []; let stdout = ""; const result = await runHostProcess(request(`process.stdout.write(String(process.pid)+' '+String(require('child_process').execSync('ps -o pgid= -p '+process.pid)).trim())`), - async item => { stdout += item.text; }, undefined, async item => { seen.push(item); }); + async item => { stdout += item.text; }, undefined, undefined, async item => { seen.push(item); }); const [pid, pgid] = stdout.split(" ").map(Number); assert.equal(result.outcome, "exited"); assert.deepEqual(seen, [{kind: "spawned", pid, process_group: pid}]); assert.equal(pgid, pid); @@ -131,7 +131,7 @@ test("the spawned Host group is reported once before input, and an unrecorded gr const script = (marker: string) => `require('fs').writeFileSync(${JSON.stringify(marker)},'') process.stdin.on('data',()=>{});setInterval(()=>{},1000)`; const recorded = join(root, "recorded"); - const refused = await runHostProcess(request(script(recorded)), async () => {}, undefined, async () => { + const refused = await runHostProcess(request(script(recorded)), async () => {}, undefined, undefined, async () => { const until = Date.now() + 2000; while (!existsSync(recorded) && Date.now() < until) await delay(5); assert.ok(existsSync(recorded), "the Host never started"); From 8c10912032bb7c2746e32d54cb4172caf714a49a Mon Sep 17 00:00:00 2001 From: song Date: Fri, 2 Oct 2026 19:22:25 +0800 Subject: [PATCH 18/54] fix(delegation): release a stopped member's lease by its execution identity The stop read idempotency_key/version from the outer task_lease, but the real acquisition nests the canonical lease under task_lease.lease, so every annotated stop raised KeyError and left the lease active. Name the lease by the execution key _acquire_delegation_lease used, read the canonical lease, and release it at its current version only while the same owner, key and recorded epoch still hold it; renewal no longer breaks the CAS and another generation is left alone. A missing annotation under a promoted hard-lease authority is still an obligation. Native File and SQLite tests cover annotated, renewed, unannotated, ACK-then-loss and foreign-generation stops. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: song --- loopx/collaboration_mcp.py | 215 ++++++++++-------------- tests/test_delegation_lease_lifetime.py | 87 +++++++++- tests/test_delegation_stop_recovery.py | 10 +- tests/test_local_delegation.py | 15 +- 4 files changed, 188 insertions(+), 139 deletions(-) diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index 48fa2a6007..3a0d9a19b7 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -50,7 +50,7 @@ TURN_LANE_ABSENT, TURN_LANE_DEAD, TURN_LANE_LIVE, TURN_LANE_RELEASED, turn_lane_liveness, turn_lane_target, ) -from .control_plane.work_items.task_lease import release_task_lease +from .control_plane.work_items.task_lease import inspect_task_lease, release_task_lease from .control_plane.collaboration.inbox import _hash, _read, _write, _root, _receipt from .control_plane.collaboration.delegation_inventory import ( DELEGATION_HOST_PROCESS_SUFFIX, DELEGATION_STOP_RECEIPT_SUFFIX, @@ -1197,141 +1197,108 @@ def _acknowledge_stop(self, path: Path, row: dict, binding: dict, *, source: str _write(self._stop_path(path), stop) def _release_delegation_lease(self, row: dict, binding: dict) -> dict: - lease = row.get("task_lease") - if not isinstance(lease, dict) or lease.get("required") is not True: + try: + owed = self._lease_obligation(row) + except (OSError, RuntimeError, ValueError, EffectRuntimeRemoteError) as exc: + return {"required": True, "released": False, + "error": ("lease obligation unreadable: " + str(exc))[:180]} + if owed is None: return {"required": False, "released": None} - return self._release_owned_task_lease( - lease, binding, - idempotency_key=str(lease["idempotency_key"]), - expected_version=lease.get("version"), - ) + return self._release_owned_task_lease(owed, binding) + + def _lease_obligation(self, row: dict) -> dict | None: + """The hard lease this operation may hold, named by its own execution identity. + + `_acquire_delegation_lease` only accepts a lease whose key is + `_turn_instance_id(row)`, and that key survives in the operation record, + so the identity never depends on the annotation's shape or on the + in-memory row that acquired it. A native claim commits before the + annotation is saved, so a missing annotation is an absent fact, not + proof that nothing is owed: under a promoted hard-lease authority the + obligation stands until the canonical lease shows this execution no + longer holds it. The recorded epoch, when there is one, fences the + release against another generation under the same key. + """ - def _release_owned_task_lease( - self, lease: dict, binding: dict, *, idempotency_key: str, expected_version: object, - ) -> dict: - """Release one lease this operation is proven to hold, by its own identity.""" + recorded = row.get("task_lease") + if isinstance(recorded, dict) and recorded.get("required") is not True: + return None + if not (isinstance(recorded, dict) and recorded.get("required") is True): + if not local_authority_is_promoted(runtime_root=self.root, goal_id=self.goal_id): + return None + if show_goal_handoff_mode( + registry_path=self.registry, runtime_root_arg=str(self.root), goal_id=self.goal_id, + )["handoff_mode"] != "hard_lease": + return None + recorded = {} + acquired = recorded.get("lease") if isinstance(recorded.get("lease"), dict) else {} + return {"idempotency_key": self._turn_instance_id(row), + "lease_epoch": acquired.get("lease_epoch")} + + def _release_owned_task_lease(self, owed: dict, binding: dict) -> dict: + """Release the lease only while this execution still holds it, at its current version. + + Renewal advances the version, so the acquisition's version is no CAS for + a later release. The canonical lease is read first: the exact owner, key + and (when recorded) epoch must match before its current version is + released. A lease another execution holds, or none at all, is not this + stop's to release and blocks nothing on its behalf. + """ + key = owed["idempotency_key"] + try: + inspection = inspect_task_lease( + registry_path=self.registry, runtime_root=self.root, + goal_id=self.goal_id, todo_id=binding["todo_id"], + ) + if inspection.get("ok") is not True: + raise RuntimeError(str(inspection.get("error") or "lease inspection unavailable")) + except (ValueError, OSError, RuntimeError, EffectRuntimeRemoteError) as exc: + return {"required": True, "released": False, "idempotency_key": key, + "error": ("lease obligation unreadable: " + str(exc))[:180]} + held = inspection.get("lease") + if (not isinstance(held, dict) + or held.get("owner") != binding["agent_id"] + or held.get("idempotency_key") != key + or (owed.get("lease_epoch") is not None + and held.get("lease_epoch") != owed["lease_epoch"])): + return {"required": True, "released": None, "held": False, "idempotency_key": key} + if held.get("status") == "released": + return {"required": True, "released": True, "idempotency_key": key, + "version": held.get("version")} try: result = release_task_lease( runtime_root=self.root, goal_id=self.goal_id, todo_id=binding["todo_id"], - owner=binding["agent_id"], idempotency_key=idempotency_key, - expected_version=expected_version, registry_path=self.registry, + owner=binding["agent_id"], idempotency_key=key, + expected_version=held.get("version"), registry_path=self.registry, ) except (ValueError, OSError, RuntimeError, EffectRuntimeRemoteError) as exc: - return {"required": True, "released": False, "error": str(exc)[:180]} + return {"required": True, "released": False, "idempotency_key": key, + "error": str(exc)[:180]} return {"required": True, "released": result.get("released") is True, + "idempotency_key": key, "version": held.get("version"), "missing": result.get("missing") is True} - def _acquired_lease_identity(self, row: dict) -> str: - """The lease key this operation acquires with, rebuilt from its own identity. - - `_acquire_delegation_lease` passes `_turn_instance_id(row)`, which is - either the recorded `turn_instance_id` or a value derived from the - operation's request id. Both survive in the operation record, so a later - process can name the lease it holds without the in-memory row that - acquired it. - """ - - return self._turn_instance_id(row) - - def _settle_lease_obligation( - self, path: Path, row: dict, binding: dict, stop: dict, - ) -> tuple[dict, bool | None]: + def _settle_lease_obligation(self, path: Path, row: dict, binding: dict, stop: dict) -> bool | None: """Try the required release once and record what it proved. - Returns the row a caller should keep reading from and the - `lease_released` fact, or `None` when the operation owed no required - lease at all — `None` keeps the typed planner's `undefined` meaning - instead of claiming a release that was never owed. + Returns the `lease_released` fact, or `None` when the operation owed no + required lease at all — `None` keeps the typed planner's `undefined` + meaning instead of claiming a release that was never owed, and also + when this execution no longer holds the lease it acquired. A release + already proven on the receipt is not attempted again. """ - owed = row.get("task_lease") - if not (isinstance(owed, dict) and owed.get("required") is True): - owed = self._canonical_lease_obligation(row, binding) - if owed is not None and owed.get("discovered") is True: - # Persist the reconciled identity beside the stopped record: a - # later reader names the same lease instead of re-deriving it, - # and the obligation is no longer only a projection. - recorded = _read(path) - if not isinstance(recorded.get("task_lease"), dict): - recorded["task_lease"] = {key: value for key, value in owed.items() - if key != "discovered"} - _write(path, recorded) - row = recorded - if owed is None: - return row, None - if owed.get("released") is False and not owed.get("idempotency_key"): - # Reconciliation could not name the lease at all. That is an - # unproven obligation, never a release: keep the stop open and let - # the next read try to read the authority again. - if owed is not stop.get("lease"): - stop["lease"] = owed - _write(self._stop_path(path), stop) - return row, False recorded = stop.get("lease") if isinstance(stop.get("lease"), dict) else {} - released = recorded if recorded.get("released") is True else None - if released is None: - released = self._release_owned_task_lease( - owed, binding, idempotency_key=str(owed["idempotency_key"]), - expected_version=owed.get("version"), - ) - if released is not stop.get("lease"): + if recorded.get("released") is True: + return True + released = self._release_delegation_lease(row, binding) + if released != stop.get("lease"): stop["lease"] = released _write(self._stop_path(path), stop) - return row, released.get("released") is True - - def _canonical_lease_obligation(self, row: dict, binding: dict) -> dict | None: - """Reconcile a required lease the operation record never annotated. - - A native claim commits before `_acquire_delegation_lease` saves the - record, so process loss in that window leaves a real active lease the - record cannot prove. A missing annotation is an absent fact, not proof - that nothing is owed. Ask the canonical authority whether this - operation's own lease key still holds an active lease, with the exact - owner and execution key it acquired under. Anything less — another - owner, another key, another epoch, or an inspection that cannot be - answered — is not this operation's obligation to release, and never a - settlement. In legacy handoff modes there is no hard lease to owe. - """ - - try: - if not local_authority_is_promoted(runtime_root=self.root, goal_id=self.goal_id): - return None - handoff_mode = show_goal_handoff_mode( - registry_path=self.registry, runtime_root_arg=str(self.root), goal_id=self.goal_id, - )["handoff_mode"] - except (OSError, RuntimeError, ValueError) as exc: - return {"required": True, "released": False, - "error": ("lease obligation unreadable: " + str(exc))[:180]} - if handoff_mode != "hard_lease": - return None - lease_key = self._acquired_lease_identity(row) - try: - from .control_plane.work_items.task_lease import inspect_task_lease - - inspection = inspect_task_lease( - registry_path=self.registry, runtime_root=self.root, - goal_id=self.goal_id, todo_id=binding["todo_id"], - ) - except (ValueError, OSError, RuntimeError, EffectRuntimeRemoteError) as exc: - return {"required": True, "released": False, - "error": ("lease obligation unreadable: " + str(exc))[:180]} - if inspection.get("ok") is not True or inspection.get("active") is not True: + if released.get("required") is not True or released.get("held") is False: return None - held = inspection.get("lease") - if (not isinstance(held, dict) - or held.get("owner") != binding["agent_id"] - or held.get("idempotency_key") != lease_key - or held.get("status") != "active"): - # Another execution's lease on the same Todo is not this stop's to release. - return None - version = held.get("version") - return { - "required": True, "handoff_mode": "hard_lease", - "idempotency_key": lease_key, - "version": version if isinstance(version, int) and not isinstance(version, bool) else None, - "discovered": True, - } + return released.get("released") is True def _settle_stop(self, path: Path) -> dict: with exclusive_file_lock(self._dispatch_lock(path)): @@ -1363,13 +1330,13 @@ def _settle_stop(self, path: Path) -> dict: # settles a stop whose member still holds an active hard lease, with # resume already refused and the Todo blocked until the TTL. # - # The record is checked first, then reconciled against the canonical - # authority: a native claim commits before the record that annotates - # it is saved, so the operation can hold a real lease it never wrote - # down. A release is retried here, under the same lock that guards the - # record: every later read is another attempt rather than one failure - # becoming permanent. - row, lease_released = self._settle_lease_obligation(path, row, binding, stop) + # The record names the execution key; the canonical lease decides + # whether that execution still holds it and at which version: a + # native claim commits before its annotation is saved, and renewal + # moves the version after it. A release is retried here, under the + # same lock that guards the record: every later read is another + # attempt rather than one failure becoming permanent. + lease_released = self._settle_lease_obligation(path, row, binding, stop) if lease_released is not None: facts["lease_released"] = lease_released decision = effect_runtime_result("collaboration.delegation.stop", { diff --git a/tests/test_delegation_lease_lifetime.py b/tests/test_delegation_lease_lifetime.py index 76b5dd8ee0..e4a095a0d4 100644 --- a/tests/test_delegation_lease_lifetime.py +++ b/tests/test_delegation_lease_lifetime.py @@ -19,10 +19,10 @@ from tests.control_plane.host_process_fixture import COUNTER_PROCESS_SOURCE -def prepare_lease(root, runner, monkeypatch, *, ttl=20): +def prepare_lease(root, runner, monkeypatch, *, ttl=20, operation_id="lease-lifetime", renew=True): monkeypatch.setattr(runner, "_spawn", lambda _: None) - runner.start("analysis", "lease-lifetime", brief()) - row = _read(runner.path("lease-lifetime")) + runner.start("analysis", operation_id, brief()) + row = _read(runner.path(operation_id)) # Only a brand-new disposable fixture. Do not migrate an active Goal or # weaken the public quiescent mode-change contract to set up a test. program = ''' @@ -40,8 +40,10 @@ def prepare_lease(root, runner, monkeypatch, *, ttl=20): capture_output=True, text=True, timeout=30) assert prepared.returncode == 0, prepared.stderr binding = runner.binding("analysis") - runner._acquire_delegation_lease(runner.path("lease-lifetime"), row, binding) + runner._acquire_delegation_lease(runner.path(operation_id), row, binding) lease = row["task_lease"]["lease"] + if not renew: + return lease renewed = runner._cli(binding, "task-lease", "renew", "--goal-id", runner.goal_id, "--todo-id", binding["todo_id"], "--owner", binding["agent_id"], "--idempotency-key", lease["idempotency_key"], "--expected-version", str(lease["version"]), @@ -151,3 +153,80 @@ def test_real_revocation_or_new_execution_stops_nested_host_without_acceptance(s runner.execute("lease-lifetime") assert _read(runner.path("lease-lifetime"))["status"] != "accepted" assert (root / "analyst" / "initial" / "host-invocations").read_text() == "1" + + +def settled_stop(runner, operation_id): + receipt = runner.stop(operation_id, execute=True) + assert receipt["phase"] == "settled" and receipt["status"] == "stopped", receipt + return receipt + + +@pytest.mark.parametrize("window", ["annotated", "renewed", "unannotated"]) +def test_real_stop_releases_the_lease_its_execution_acquired(service, monkeypatch, window): + """Native acquire -> public stop -> independent inspect, on both providers. + + `annotated` is the ordinary record `_acquire_delegation_lease` writes, + `renewed` moves the canonical version past the recorded one, and + `unannotated` drops the annotation as if the process were lost between the + native claim and the record write. Each must release the exact lease and + settle; a retry is the same receipt. + """ + root, runner = service + original = prepare_lease(root, runner, monkeypatch, operation_id="lease-stop", + renew=window == "renewed") + path = runner.path("lease-stop") + if window == "unannotated": + row = _read(path) + del row["task_lease"] + runner._fenced_write(path, row) + if window == "renewed": + assert inspect(runner)["lease"]["version"] > _read(path)["task_lease"]["lease"]["version"] + receipt = settled_stop(runner, "lease-stop") + assert receipt["stop"]["lease"]["released"] is True + assert receipt["stop"]["settled"]["lease_released"] is True + released = inspect(runner)["lease"] + assert released["status"] == "released" + assert released["idempotency_key"] == original["idempotency_key"] + assert runner.stop("lease-stop", execute=True) == receipt + with pytest.raises(ValueError, match="start a new operation id"): + runner.resume("lease-stop") + + +def test_real_stop_survives_loss_after_its_acknowledgement(service, monkeypatch): + """ACK durable, release never committed, then a fresh instance reads the stop.""" + from loopx import collaboration_mcp as delegation + + root, runner = service + prepare_lease(root, runner, monkeypatch, operation_id="lease-ack-loss") + + def lost(**kwargs): + raise RuntimeError("fixture lost the process before release") + + with monkeypatch.context() as loss: + loss.setattr(delegation, "release_task_lease", lost) + receipt = runner.stop("lease-ack-loss", execute=True) + assert receipt["phase"] == "acknowledged", receipt + assert receipt["stop"]["reason"] == "required_lease_release_unproven" + assert inspect(runner)["lease"]["status"] == "active" + fresh = delegation.Delegations(runner.root, runner.registry, runner.goal_id, runner.agent_id, runner.config) + settled_stop(fresh, "lease-ack-loss") + assert inspect(runner)["lease"]["status"] == "released" + + +def test_real_stop_leaves_a_newer_generation_alone(service, monkeypatch): + """Another execution's lease on the same Todo is not this stop's to release.""" + root, runner = service + original = prepare_lease(root, runner, monkeypatch, operation_id="lease-foreign") + binding = runner.binding("analysis") + current = inspect(runner)["lease"] + runner._cli(binding, "task-lease", "release", "--goal-id", runner.goal_id, + "--todo-id", "todo_analyst-initial", "--owner", "analyst", + "--idempotency-key", original["idempotency_key"], "--expected-version", str(current["version"])) + acquired = runner._cli(binding, "task-lease", "acquire", "--goal-id", runner.goal_id, + "--todo-id", "todo_analyst-initial", "--owner", "analyst", + "--idempotency-key", "new-execution", "--expected-version", str(current["version"])) + assert acquired["lease"]["lease_epoch"] > original["lease_epoch"] + receipt = settled_stop(runner, "lease-foreign") + assert "lease_released" not in receipt["stop"]["settled"] + held = inspect(runner)["lease"] + assert held["status"] == "active" and held["idempotency_key"] == "new-execution" diff --git a/tests/test_delegation_stop_recovery.py b/tests/test_delegation_stop_recovery.py index ec1ec79ab9..b824140741 100644 --- a/tests/test_delegation_stop_recovery.py +++ b/tests/test_delegation_stop_recovery.py @@ -143,7 +143,6 @@ def canonical_lease_at_the_native_edge(service, monkeypatch, operation_id, *, Python reconciliation runs for real instead of being stubbed out. """ from loopx.control_plane.collaboration.inbox import _write as write_inbox - from loopx.control_plane.work_items import task_lease as lease_module root, runner = service monkeypatch.setattr(runner, "_spawn", lambda _: None) @@ -165,7 +164,7 @@ def canonical_lease_at_the_native_edge(service, monkeypatch, operation_id, *, monkeypatch.setattr(delegation, "local_authority_is_promoted", lambda **kwargs: True) monkeypatch.setattr(delegation, "show_goal_handoff_mode", lambda **kwargs: {"handoff_mode": "hard_lease"}) - monkeypatch.setattr(lease_module, "inspect_task_lease", lambda **kwargs: { + monkeypatch.setattr(delegation, "inspect_task_lease", lambda **kwargs: { "ok": True, "action": "inspect", "active": active, "legacy_fallback_used": False, "lease": {"owner": owner, "idempotency_key": lease_key + key_suffix, "status": "active", "version": version}, @@ -222,9 +221,6 @@ def test_a_recovered_lease_obligation_settles_only_after_its_exact_release(servi "todo_id": "todo_analyst-initial", "owner": "analyst", "idempotency_key": lease_key, "expected_version": 4, "registry_path": runner.registry}] - # The recovered identity is persisted, so a later read names the same lease. - recovered = _read(runner.path("analysis-lease-recover")) - assert recovered["task_lease"]["idempotency_key"] == lease_key assert runner.stop("analysis-lease-recover", execute=True) == receipt assert len(releases) == 1 @@ -256,15 +252,13 @@ def test_an_unreadable_lease_obligation_keeps_the_stop_open(service, monkeypatch settled stop would trade a wrong terminal receipt for a crash, so the stop stays open and the next read retries the authority instead. """ - from loopx.control_plane.work_items import task_lease as lease_module - root, runner, lease_key = canonical_lease_at_the_native_edge( service, monkeypatch, "analysis-lease-unreadable") def unreadable(**kwargs): raise RuntimeError("native authority store unavailable") - monkeypatch.setattr(lease_module, "inspect_task_lease", unreadable) + monkeypatch.setattr(delegation, "inspect_task_lease", unreadable) releases = [] monkeypatch.setattr(delegation, "release_task_lease", lambda **kwargs: releases.append(kwargs) or {"released": True}) diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index f537b006f5..7845c797ef 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -787,6 +787,15 @@ def test_a_stop_written_before_the_effects_linearizes_against_them(service, monk assert receipt["phase"] == "settled" and receipt["status"] == "stopped" +def held_lease(monkeypatch, row, key): + """The annotation `_acquire_delegation_lease` writes, and a canonical lease that still holds it.""" + row["turn_instance_id"] = key + lease = {"owner": "analyst", "idempotency_key": key, "status": "active", "version": 1, "lease_epoch": 1} + monkeypatch.setattr(delegation_module, "inspect_task_lease", + lambda **kwargs: {"ok": True, "action": "inspect", "active": True, "lease": lease}) + return {"required": True, "handoff_mode": "hard_lease", "lease": dict(lease)} + + def test_a_failed_required_lease_release_keeps_the_stop_open_and_retries(service, monkeypatch): """A required lease the stop could not release is not a settlement. @@ -804,7 +813,7 @@ def test_a_failed_required_lease_release_keeps_the_stop_open_and_retries(service path = runner.path("analysis-lease") row = _read(path) # A bounded member task holds a required lease; make the record say so. - row["task_lease"] = {"required": True, "idempotency_key": "lease-1", "version": 1} + row["task_lease"] = held_lease(monkeypatch, row, "lease-1") row["status"] = "stopped" runner._fenced_write(path, row) write_inbox(runner._stop_path(path), { @@ -895,7 +904,7 @@ def test_a_crash_between_the_ack_and_the_lease_result_keeps_the_stop_open(servic path = runner.path("analysis-crash") row = _read(path) # The member holds a real required lease, as a bounded task does. - row["task_lease"] = {"required": True, "idempotency_key": "lease-crash", "version": 1} + row["task_lease"] = held_lease(monkeypatch, row, "lease-crash") right_after_ack = {**row, "status": "stopped"} runner._fenced_write(path, right_after_ack) @@ -934,7 +943,7 @@ def test_a_crash_between_the_ack_and_the_lease_result_never_settles_unreleased(s runner.start("analysis", "analysis-crash-open", brief()) path = runner.path("analysis-crash-open") row = _read(path) - row["task_lease"] = {"required": True, "idempotency_key": "lease-open", "version": 1} + row["task_lease"] = held_lease(monkeypatch, row, "lease-open") right_after_ack = {**row, "status": "stopped"} runner._fenced_write(path, right_after_ack) write_inbox(runner._stop_path(path), { From 95f55c128e8144a4bd822a2aceed02159caa8e1c Mon Sep 17 00:00:00 2001 From: song Date: Fri, 2 Oct 2026 19:41:15 +0800 Subject: [PATCH 19/54] refactor(delegation): move stop lease release into its own module collaboration_mcp.py grew past the 2000-line module ceiling. Move the stop's lease obligation, exact release and receipt settlement into control_plane/collaboration/delegation_stop_lease.py beside the other delegation helpers instead of raising the ceiling. A stop that owed no lease again leaves the receipt's lease field unset. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: song --- loopx/collaboration_mcp.py | 111 +--------------- .../collaboration/delegation_stop_lease.py | 120 ++++++++++++++++++ tests/test_delegation_lease_lifetime.py | 3 +- tests/test_delegation_stop_recovery.py | 17 +-- tests/test_local_delegation.py | 11 +- 5 files changed, 139 insertions(+), 123 deletions(-) create mode 100644 loopx/control_plane/collaboration/delegation_stop_lease.py diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index 3a0d9a19b7..1a629fb85d 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -50,7 +50,6 @@ TURN_LANE_ABSENT, TURN_LANE_DEAD, TURN_LANE_LIVE, TURN_LANE_RELEASED, turn_lane_liveness, turn_lane_target, ) -from .control_plane.work_items.task_lease import inspect_task_lease, release_task_lease from .control_plane.collaboration.inbox import _hash, _read, _write, _root, _receipt from .control_plane.collaboration.delegation_inventory import ( DELEGATION_HOST_PROCESS_SUFFIX, DELEGATION_STOP_RECEIPT_SUFFIX, @@ -62,7 +61,7 @@ collaboration_goal_scope, decide_collaboration_lifecycle, ) -from .control_plane.collaboration import delegation_results, delegation_validation +from .control_plane.collaboration import delegation_results, delegation_stop_lease, delegation_validation from .control_plane.collaboration.peers import ( _goal, consume_return, @@ -1190,116 +1189,12 @@ def _acknowledge_stop(self, path: Path, row: dict, binding: dict, *, source: str self._clear_delegation_bootstrap(row, binding) except (OSError, ValueError): pass # the bootstrap is host input; its state never blocks the receipt - lease = self._release_delegation_lease(row, binding) + lease = delegation_stop_lease.release(self, row, binding) with exclusive_file_lock(self._dispatch_lock(path)): stop = self._read_stop(path) or stop stop["lease"] = lease _write(self._stop_path(path), stop) - def _release_delegation_lease(self, row: dict, binding: dict) -> dict: - try: - owed = self._lease_obligation(row) - except (OSError, RuntimeError, ValueError, EffectRuntimeRemoteError) as exc: - return {"required": True, "released": False, - "error": ("lease obligation unreadable: " + str(exc))[:180]} - if owed is None: - return {"required": False, "released": None} - return self._release_owned_task_lease(owed, binding) - - def _lease_obligation(self, row: dict) -> dict | None: - """The hard lease this operation may hold, named by its own execution identity. - - `_acquire_delegation_lease` only accepts a lease whose key is - `_turn_instance_id(row)`, and that key survives in the operation record, - so the identity never depends on the annotation's shape or on the - in-memory row that acquired it. A native claim commits before the - annotation is saved, so a missing annotation is an absent fact, not - proof that nothing is owed: under a promoted hard-lease authority the - obligation stands until the canonical lease shows this execution no - longer holds it. The recorded epoch, when there is one, fences the - release against another generation under the same key. - """ - - recorded = row.get("task_lease") - if isinstance(recorded, dict) and recorded.get("required") is not True: - return None - if not (isinstance(recorded, dict) and recorded.get("required") is True): - if not local_authority_is_promoted(runtime_root=self.root, goal_id=self.goal_id): - return None - if show_goal_handoff_mode( - registry_path=self.registry, runtime_root_arg=str(self.root), goal_id=self.goal_id, - )["handoff_mode"] != "hard_lease": - return None - recorded = {} - acquired = recorded.get("lease") if isinstance(recorded.get("lease"), dict) else {} - return {"idempotency_key": self._turn_instance_id(row), - "lease_epoch": acquired.get("lease_epoch")} - - def _release_owned_task_lease(self, owed: dict, binding: dict) -> dict: - """Release the lease only while this execution still holds it, at its current version. - - Renewal advances the version, so the acquisition's version is no CAS for - a later release. The canonical lease is read first: the exact owner, key - and (when recorded) epoch must match before its current version is - released. A lease another execution holds, or none at all, is not this - stop's to release and blocks nothing on its behalf. - """ - - key = owed["idempotency_key"] - try: - inspection = inspect_task_lease( - registry_path=self.registry, runtime_root=self.root, - goal_id=self.goal_id, todo_id=binding["todo_id"], - ) - if inspection.get("ok") is not True: - raise RuntimeError(str(inspection.get("error") or "lease inspection unavailable")) - except (ValueError, OSError, RuntimeError, EffectRuntimeRemoteError) as exc: - return {"required": True, "released": False, "idempotency_key": key, - "error": ("lease obligation unreadable: " + str(exc))[:180]} - held = inspection.get("lease") - if (not isinstance(held, dict) - or held.get("owner") != binding["agent_id"] - or held.get("idempotency_key") != key - or (owed.get("lease_epoch") is not None - and held.get("lease_epoch") != owed["lease_epoch"])): - return {"required": True, "released": None, "held": False, "idempotency_key": key} - if held.get("status") == "released": - return {"required": True, "released": True, "idempotency_key": key, - "version": held.get("version")} - try: - result = release_task_lease( - runtime_root=self.root, goal_id=self.goal_id, todo_id=binding["todo_id"], - owner=binding["agent_id"], idempotency_key=key, - expected_version=held.get("version"), registry_path=self.registry, - ) - except (ValueError, OSError, RuntimeError, EffectRuntimeRemoteError) as exc: - return {"required": True, "released": False, "idempotency_key": key, - "error": str(exc)[:180]} - return {"required": True, "released": result.get("released") is True, - "idempotency_key": key, "version": held.get("version"), - "missing": result.get("missing") is True} - - def _settle_lease_obligation(self, path: Path, row: dict, binding: dict, stop: dict) -> bool | None: - """Try the required release once and record what it proved. - - Returns the `lease_released` fact, or `None` when the operation owed no - required lease at all — `None` keeps the typed planner's `undefined` - meaning instead of claiming a release that was never owed, and also - when this execution no longer holds the lease it acquired. A release - already proven on the receipt is not attempted again. - """ - - recorded = stop.get("lease") if isinstance(stop.get("lease"), dict) else {} - if recorded.get("released") is True: - return True - released = self._release_delegation_lease(row, binding) - if released != stop.get("lease"): - stop["lease"] = released - _write(self._stop_path(path), stop) - if released.get("required") is not True or released.get("held") is False: - return None - return released.get("released") is True - def _settle_stop(self, path: Path) -> dict: with exclusive_file_lock(self._dispatch_lock(path)): row = _read(path) @@ -1336,7 +1231,7 @@ def _settle_stop(self, path: Path) -> dict: # moves the version after it. A release is retried here, under the # same lock that guards the record: every later read is another # attempt rather than one failure becoming permanent. - lease_released = self._settle_lease_obligation(path, row, binding, stop) + lease_released = delegation_stop_lease.settle(self, path, row, binding, stop) if lease_released is not None: facts["lease_released"] = lease_released decision = effect_runtime_result("collaboration.delegation.stop", { diff --git a/loopx/control_plane/collaboration/delegation_stop_lease.py b/loopx/control_plane/collaboration/delegation_stop_lease.py new file mode 100644 index 0000000000..3a57c59c4c --- /dev/null +++ b/loopx/control_plane/collaboration/delegation_stop_lease.py @@ -0,0 +1,120 @@ +"""Release the hard lease a stopped delegation acquired, by its own execution identity. + +A stop owes the release of the lease its execution acquired; it never owns a +lease another execution holds. The identity comes from the operation record +and the version from the canonical lease, so a renewed or unannotated lease is +still released exactly, and the native lifecycle remains the only writer. +""" +from __future__ import annotations + +from .inbox import _write +from ..coordination.local_authority import local_authority_is_promoted +from ..effect_runtime import EffectRuntimeRemoteError +from ..todos.handoff_mode import show_goal_handoff_mode +from ..work_items.task_lease import inspect_task_lease, release_task_lease + +_AUTHORITY_ERRORS = (ValueError, OSError, RuntimeError, EffectRuntimeRemoteError) + + +def obligation(service, row): + """The hard lease this operation may hold, named by its own execution identity. + + `_acquire_delegation_lease` only accepts a lease whose key is + `_turn_instance_id(row)`, and that key survives in the operation record, so + the identity never depends on the annotation's shape or on the in-memory + row that acquired it. A native claim commits before the annotation is + saved, so a missing annotation is an absent fact, not proof that nothing is + owed: under a promoted hard-lease authority the obligation stands until the + canonical lease shows this execution no longer holds it. The recorded + epoch, when there is one, fences the release against another generation + under the same key. + """ + + recorded = row.get("task_lease") + if isinstance(recorded, dict) and recorded.get("required") is not True: + return None + if not (isinstance(recorded, dict) and recorded.get("required") is True): + if not local_authority_is_promoted(runtime_root=service.root, goal_id=service.goal_id): + return None + if show_goal_handoff_mode( + registry_path=service.registry, runtime_root_arg=str(service.root), goal_id=service.goal_id, + )["handoff_mode"] != "hard_lease": + return None + recorded = {} + acquired = recorded.get("lease") if isinstance(recorded.get("lease"), dict) else {} + return {"idempotency_key": service._turn_instance_id(row), + "lease_epoch": acquired.get("lease_epoch")} + + +def release(service, row, binding): + """Release the lease only while this execution still holds it, at its current version. + + Renewal advances the version, so the acquisition's version is no CAS for a + later release. The canonical lease is read first: the exact owner, key and + (when recorded) epoch must match before its current version is released. + A lease another execution holds, or none at all, is reported `held: false`: + it is not this stop's to release and blocks nothing on its behalf. + """ + + try: + owed = obligation(service, row) + except _AUTHORITY_ERRORS as exc: + return {"required": True, "released": False, + "error": ("lease obligation unreadable: " + str(exc))[:180]} + if owed is None: + return {"required": False, "released": None} + key = owed["idempotency_key"] + try: + inspection = inspect_task_lease( + registry_path=service.registry, runtime_root=service.root, + goal_id=service.goal_id, todo_id=binding["todo_id"], + ) + if inspection.get("ok") is not True: + raise RuntimeError(str(inspection.get("error") or "lease inspection unavailable")) + except _AUTHORITY_ERRORS as exc: + return {"required": True, "released": False, "idempotency_key": key, + "error": ("lease obligation unreadable: " + str(exc))[:180]} + held = inspection.get("lease") + if (not isinstance(held, dict) + or held.get("owner") != binding["agent_id"] + or held.get("idempotency_key") != key + or (owed.get("lease_epoch") is not None + and held.get("lease_epoch") != owed["lease_epoch"])): + return {"required": True, "released": None, "held": False, "idempotency_key": key} + if held.get("status") == "released": + return {"required": True, "released": True, "idempotency_key": key, + "version": held.get("version")} + try: + result = release_task_lease( + runtime_root=service.root, goal_id=service.goal_id, todo_id=binding["todo_id"], + owner=binding["agent_id"], idempotency_key=key, + expected_version=held.get("version"), registry_path=service.registry, + ) + except _AUTHORITY_ERRORS as exc: + return {"required": True, "released": False, "idempotency_key": key, + "error": str(exc)[:180]} + return {"required": True, "released": result.get("released") is True, + "idempotency_key": key, "version": held.get("version"), + "missing": result.get("missing") is True} + + +def settle(service, path, row, binding, stop): + """Try the required release once and record what it proved on the stop receipt. + + Returns the `lease_released` fact, or `None` when the operation owed no + required lease at all, or this execution no longer holds the lease it + acquired. `None` keeps the typed planner's `undefined` meaning instead of + claiming a release that was never owed. A release already proven on the + receipt is not attempted again. Callers hold the operation's dispatch lock. + """ + + recorded = stop.get("lease") if isinstance(stop.get("lease"), dict) else {} + if recorded.get("released") is True: + return True + released = release(service, row, binding) + if released.get("required") is not True: + return None # nothing owed: the receipt keeps no lease fact + if released != stop.get("lease"): + stop["lease"] = released + _write(service._stop_path(path), stop) + return None if released.get("held") is False else released.get("released") is True diff --git a/tests/test_delegation_lease_lifetime.py b/tests/test_delegation_lease_lifetime.py index e4a095a0d4..620eced9b7 100644 --- a/tests/test_delegation_lease_lifetime.py +++ b/tests/test_delegation_lease_lifetime.py @@ -14,6 +14,7 @@ import pytest from test_local_delegation import HOST, brief, service # noqa: F401 +from loopx.control_plane.collaboration import delegation_stop_lease as stop_lease from loopx.control_plane.collaboration.inbox import _read from loopx.control_plane.coordination.local_authority import read_canonical_todos_if_promoted from tests.control_plane.host_process_fixture import COUNTER_PROCESS_SOURCE @@ -203,7 +204,7 @@ def lost(**kwargs): raise RuntimeError("fixture lost the process before release") with monkeypatch.context() as loss: - loss.setattr(delegation, "release_task_lease", lost) + loss.setattr(stop_lease, "release_task_lease", lost) receipt = runner.stop("lease-ack-loss", execute=True) assert receipt["phase"] == "acknowledged", receipt assert receipt["stop"]["reason"] == "required_lease_release_unproven" diff --git a/tests/test_delegation_stop_recovery.py b/tests/test_delegation_stop_recovery.py index b824140741..c66b8bd366 100644 --- a/tests/test_delegation_stop_recovery.py +++ b/tests/test_delegation_stop_recovery.py @@ -11,6 +11,7 @@ from test_local_delegation import brief, demo, service as service from loopx import collaboration_mcp as delegation +from loopx.control_plane.collaboration import delegation_stop_lease as stop_lease from loopx.control_plane.collaboration.inbox import _read from loopx.control_plane.collaboration.peers import returns from loopx.file_lock import exclusive_file_lock, lock_holder_host_label @@ -161,10 +162,10 @@ def canonical_lease_at_the_native_edge(service, monkeypatch, operation_id, *, "ack": {"pid": os.getpid(), "host": lock_holder_host_label(), "at": time.time(), "source": "requester", "observed_status": "stopped", "turn_key": None}, }) - monkeypatch.setattr(delegation, "local_authority_is_promoted", lambda **kwargs: True) - monkeypatch.setattr(delegation, "show_goal_handoff_mode", + monkeypatch.setattr(stop_lease, "local_authority_is_promoted", lambda **kwargs: True) + monkeypatch.setattr(stop_lease, "show_goal_handoff_mode", lambda **kwargs: {"handoff_mode": "hard_lease"}) - monkeypatch.setattr(delegation, "inspect_task_lease", lambda **kwargs: { + monkeypatch.setattr(stop_lease, "inspect_task_lease", lambda **kwargs: { "ok": True, "action": "inspect", "active": active, "legacy_fallback_used": False, "lease": {"owner": owner, "idempotency_key": lease_key + key_suffix, "status": "active", "version": version}, @@ -191,7 +192,7 @@ def test_a_lease_the_record_never_annotated_still_blocks_settlement(service, mon def unavailable(**kwargs): raise RuntimeError("authority unavailable") - monkeypatch.setattr(delegation, "release_task_lease", unavailable) + monkeypatch.setattr(stop_lease, "release_task_lease", unavailable) receipt = runner.stop("analysis-lease-window", execute=True) assert receipt["phase"] == "acknowledged", receipt assert receipt["stop"]["reason"] == "required_lease_release_unproven" @@ -210,7 +211,7 @@ def test_a_recovered_lease_obligation_settles_only_after_its_exact_release(servi root, runner, lease_key = canonical_lease_at_the_native_edge( service, monkeypatch, "analysis-lease-recover") releases = [] - monkeypatch.setattr(delegation, "release_task_lease", + monkeypatch.setattr(stop_lease, "release_task_lease", lambda **kwargs: releases.append(kwargs) or {"released": True}) receipt = runner.stop("analysis-lease-recover", execute=True) @@ -236,7 +237,7 @@ def test_a_foreign_lease_generation_is_not_this_stops_obligation(service, monkey root, runner, lease_key = canonical_lease_at_the_native_edge( service, monkeypatch, "analysis-lease-foreign", key_suffix="-older") releases = [] - monkeypatch.setattr(delegation, "release_task_lease", + monkeypatch.setattr(stop_lease, "release_task_lease", lambda **kwargs: releases.append(kwargs) or {"released": True}) receipt = runner.stop("analysis-lease-foreign", execute=True) @@ -258,9 +259,9 @@ def test_an_unreadable_lease_obligation_keeps_the_stop_open(service, monkeypatch def unreadable(**kwargs): raise RuntimeError("native authority store unavailable") - monkeypatch.setattr(delegation, "inspect_task_lease", unreadable) + monkeypatch.setattr(stop_lease, "inspect_task_lease", unreadable) releases = [] - monkeypatch.setattr(delegation, "release_task_lease", + monkeypatch.setattr(stop_lease, "release_task_lease", lambda **kwargs: releases.append(kwargs) or {"released": True}) receipt = runner.stop("analysis-lease-unreadable", execute=True) diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index 7845c797ef..94720ee247 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -17,7 +17,7 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "examples" / "managed-research-team")) import research_team as demo # noqa: E402 from test_managed_research_scenario import fixture # noqa: E402 -from loopx import collaboration_mcp as delegation_module # noqa: E402 +from loopx.control_plane.collaboration import delegation_stop_lease as stop_lease # noqa: E402 from loopx.collaboration_mcp import DelegationFenced, DelegationStopRequested, Delegations # noqa: E402 from loopx.control_plane.collaboration.peers import returns # noqa: E402 from loopx.control_plane.collaboration.inbox import _read # noqa: E402 @@ -791,7 +791,7 @@ def held_lease(monkeypatch, row, key): """The annotation `_acquire_delegation_lease` writes, and a canonical lease that still holds it.""" row["turn_instance_id"] = key lease = {"owner": "analyst", "idempotency_key": key, "status": "active", "version": 1, "lease_epoch": 1} - monkeypatch.setattr(delegation_module, "inspect_task_lease", + monkeypatch.setattr(stop_lease, "inspect_task_lease", lambda **kwargs: {"ok": True, "action": "inspect", "active": True, "lease": lease}) return {"required": True, "handoff_mode": "hard_lease", "lease": dict(lease)} @@ -804,7 +804,6 @@ def test_a_failed_required_lease_release_keeps_the_stop_open_and_retries(service retried under the stop's own lock on the next read instead of being attempted once and forgotten. """ - from loopx import collaboration_mcp as delegation from loopx.control_plane.collaboration.inbox import _write as write_inbox root, runner = service @@ -835,7 +834,7 @@ def flaky_release(**kwargs): raise outcome return outcome - monkeypatch.setattr(delegation, "release_task_lease", flaky_release) + monkeypatch.setattr(stop_lease, "release_task_lease", flaky_release) # The first settle read tries the release, fails, and must not settle. receipt = runner.stop("analysis-lease", execute=True) @@ -921,7 +920,7 @@ def test_a_crash_between_the_ack_and_the_lease_result_keeps_the_stop_open(servic # A release that succeeds on this read lets the stop settle, and the receipt # says the lease really is gone. - monkeypatch.setattr(delegation_module, "release_task_lease", + monkeypatch.setattr(stop_lease, "release_task_lease", lambda **kw: {"released": True}) settled = runner.stop("analysis-crash", execute=True) assert settled["phase"] == "settled", settled @@ -956,7 +955,7 @@ def test_a_crash_between_the_ack_and_the_lease_result_never_settles_unreleased(s def still_held(**kwargs): raise RuntimeError("authority unavailable") - monkeypatch.setattr(delegation_module, "release_task_lease", still_held) + monkeypatch.setattr(stop_lease, "release_task_lease", still_held) receipt = runner.stop("analysis-crash-open", execute=True) assert receipt["phase"] == "acknowledged", receipt assert receipt["stop"]["reason"] == "required_lease_release_unproven" From cd1d3b13cf8480a2e100ad8ce71e3c98b65f0901 Mon Sep 17 00:00:00 2001 From: song Date: Fri, 2 Oct 2026 21:20:45 +0800 Subject: [PATCH 20/54] fix(host): run the bridge on the caller's environment, not the ambient one run_host_process accepted an environment and then replaced it with os.environ.copy(), so an explicit mapping was ignored: a caller that excluded an ambient variable had it reintroduced, and the pinned release and Host record a delegated _cli sets were dropped, leaving the Host unattributed. Copy the caller's mapping (or the ambient one when absent), consume the record marker from that copy only, and never mutate the caller's dict. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: song --- .../turn_driver/host_process_transport.py | 11 ++- tests/control_plane/test_host_process.py | 71 +++++++++++++++++++ 2 files changed, 79 insertions(+), 3 deletions(-) diff --git a/loopx/control_plane/turn_driver/host_process_transport.py b/loopx/control_plane/turn_driver/host_process_transport.py index 8535990d1e..121865595b 100644 --- a/loopx/control_plane/turn_driver/host_process_transport.py +++ b/loopx/control_plane/turn_driver/host_process_transport.py @@ -166,8 +166,13 @@ def run_host_process( if delegated_lease is not None: request["delegated_lease"] = delegated_lease bridge = Path(__file__).with_name("host_process_bridge.ts") - environment = os.environ.copy() - record_value = environment.pop(HOST_PROCESS_RECORD_ENV, "") + # The bridge runs on the caller's environment, which may pin the release the + # Host must run and name the record its supervisor owns. Copy it instead of + # replacing it with the ambient one, and consume the record env from the + # copy only: the Host below must not inherit a marker for its supervisor's + # record, and the caller's mapping is never mutated. + bridge_environment = os.environ.copy() if environment is None else dict(environment) + record_value = bridge_environment.pop(HOST_PROCESS_RECORD_ENV, "") record_path = Path(record_value) if record_value else None with subprocess.Popen( [ @@ -183,7 +188,7 @@ def run_host_process( encoding="utf-8", errors="strict", start_new_session=True, - env=environment, + env=bridge_environment, ) as proc: assert proc.stdin is not None and proc.stdout is not None result = None diff --git a/tests/control_plane/test_host_process.py b/tests/control_plane/test_host_process.py index 11aef04f7a..d81f05582f 100644 --- a/tests/control_plane/test_host_process.py +++ b/tests/control_plane/test_host_process.py @@ -12,8 +12,10 @@ import pytest +from loopx import collaboration_mcp as delegation_module from loopx.control_plane.turn_driver.executor import _run_host from loopx.control_plane.turn_driver.host_process_transport import ( + HOST_PROCESS_RECORD_ENV, HostOutputLines, run_host_process, ) @@ -267,3 +269,72 @@ def drain(**fields): finally: live.kill() live.wait(timeout=10) + + +@pytest.mark.skipif(os.name == "nt", reason="POSIX process-group transport parity") +def test_explicit_environment_reaches_the_host_and_stays_out_of_the_record(tmp_path: Path) -> None: + """A caller-supplied environment is used, and the record marker is not inherited. + + The bridge runs on the caller's mapping: it may pin the release the Host must + run, and it names the record the supervisor owns. Replacing that mapping with + the ambient one would silently change which code the Host runs and lose the + record; passing it through unchanged would leak the record marker into a + nested run that must not overwrite its parent's record. + """ + from loopx.control_plane.turn_driver.host_process_transport import ( + HOST_PROCESS_RECORD_ENV, host_process_drain, run_host_process, + ) + + record_path = tmp_path / "op.host.json" + excluded = "LOOPX_TRANSPORT_PARITY_EXCLUDED" + selected = "LOOPX_TRANSPORT_PARITY_SELECTED" + host = ("import json,os,sys;print(json.dumps({'selected': os.environ.get(%r)," + " 'excluded': os.environ.get(%r), 'record': os.environ.get(%r)," + " 'pgid': os.getpgid(0), 'pid': os.getpid()}))" % (selected, excluded, HOST_PROCESS_RECORD_ENV)) + environment = {**os.environ, selected: "caller-selected", HOST_PROCESS_RECORD_ENV: str(record_path)} + environment.pop(excluded, None) + chunks: list[str] = [] + observation = run_host_process([sys.executable, "-c", host], project=tmp_path, input_text="", + timeout_seconds=15, environment=environment, + on_stdout=chunks.append) + assert observation["outcome"] == "exited", observation + value = json.loads("".join(chunks)) + assert value["selected"] == "caller-selected" + assert value["excluded"] is None + # The record is the supervisor's, and it is not a Host input. + assert value["record"] is None + record = json.loads(record_path.read_text()) + assert record["phase"] == "finished" + assert record["host_pid"] == value["pid"] == record["process_group"] == value["pgid"] + assert host_process_drain(record_path) == "drained" + # The caller's mapping is the caller's. + assert environment[HOST_PROCESS_RECORD_ENV] == str(record_path) + assert environment[selected] == "caller-selected" + + +@pytest.mark.skipif(os.name == "nt", reason="POSIX process-group transport parity") +def test_hard_leased_cli_pins_the_release_and_the_record_through_the_transport(tmp_path, monkeypatch) -> None: + """`_cli`'s pinned release and Host record only reach the CLI through the transport. + + The managed CLI runs under the bridge, so the environment `_cli` builds is + only effective if the transport consumes it. Capture what `_cli` hands the + transport instead of restating the native propagation the parity test covers. + """ + from loopx.collaboration_mcp import Delegations + + runner = Delegations(tmp_path, tmp_path / "registry.json", "goal", "lead", tmp_path / "config.json") + monkeypatch.setenv("PYTHONPATH", "/ambient") + handed = {} + + def transport(*args, **kwargs): + handed.update(kwargs) + kwargs["on_stdout"]("{}") + return {"outcome": "exited", "output_complete": True, "returncode": 0} + + monkeypatch.setattr("loopx.control_plane.turn_driver.host_process_transport.run_host_process", transport) + record = tmp_path / "op.host.json" + runner._cli({"agent_id": "analyst", "todo_id": "todo", "workspace": str(tmp_path)}, "todo", "claim", + host_record=record, delegated_lease={"lease": {}, "ttl_seconds": None, + "renew_argv": [], "read_argv": []}) + assert handed["environment"]["PYTHONPATH"].split(os.pathsep)[0] == str(delegation_module._release_root()) + assert handed["environment"][HOST_PROCESS_RECORD_ENV] == str(record) From afb76160fbcfe3fe3d872a19c3bfaf7f27b35a2b Mon Sep 17 00:00:00 2001 From: song Date: Sat, 3 Oct 2026 01:27:52 +0800 Subject: [PATCH 21/54] fix(delegation): prove nested Host drain before releasing its lease Signed-off-by: song --- loopx/collaboration_mcp.py | 91 ++++++++++++------- .../collaboration/delegation_stop_lease.py | 4 +- .../turn_driver/delegated_cli.py | 9 ++ 3 files changed, 67 insertions(+), 37 deletions(-) diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index 1a629fb85d..270a0f6835 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -43,7 +43,8 @@ ) from .control_plane.turn_driver.host_binding import turn_host_arg_option from .control_plane.turn_driver.host_process_transport import ( - HOST_PROCESS_DRAINING, HOST_PROCESS_RECORD_ENV, HOST_PROCESS_UNSUPPORTED_PLATFORM, + HOST_PROCESS_DRAINED, HOST_PROCESS_DRAINING, HOST_PROCESS_NOT_LAUNCHED, + HOST_PROCESS_RECORD_ENV, HOST_PROCESS_UNATTRIBUTABLE, HOST_PROCESS_UNSUPPORTED_PLATFORM, host_process_drain, ) from .control_plane.turn_driver.lane_fence import ( @@ -1141,19 +1142,42 @@ def _signal_worker(self, path: Path, stop: dict) -> None: # The Host supervisor sits outside the worker's group and cleans up on its own. self._await_host_drain(path, time.monotonic() + DELEGATION_STOP_GRACE_SECONDS) + def _delegation_host_drain(self, path: Path) -> str: + """Read both groups of a hard-leased Turn; never signal from readback. + + The leased CLI and actual Host use separate sessions. The CLI cannot + prove its nested Host exited merely by exiting itself. Old leased + records that only named that outer group lack the required proof. + """ + host_record = self._host_process_record(path) + host = host_process_drain(host_record) + outer = host_process_drain(host_record.with_suffix(".cli.host.json")) + if HOST_PROCESS_UNSUPPORTED_PLATFORM in (outer, host): + return HOST_PROCESS_UNSUPPORTED_PLATFORM + if outer == HOST_PROCESS_NOT_LAUNCHED: + row = _read(path) + lease = row.get("task_lease") + if host != HOST_PROCESS_NOT_LAUNCHED and isinstance(lease, dict) and lease.get("required") is True: + return HOST_PROCESS_UNATTRIBUTABLE + return host + for fact in (HOST_PROCESS_UNATTRIBUTABLE, HOST_PROCESS_DRAINING): + if fact in (outer, host): + return fact + return HOST_PROCESS_DRAINED + def _await_host_drain(self, path: Path, deadline: float) -> None: """Wait, never kill: the TS Host supervisor owns terminating its process group.""" - while (host_process_drain(self._host_process_record(path)) == HOST_PROCESS_DRAINING + while (self._delegation_host_drain(path) == HOST_PROCESS_DRAINING and time.monotonic() < deadline): time.sleep(0.1) def _acknowledge_stop(self, path: Path, row: dict, binding: dict, *, source: str) -> None: - """Acknowledge from under the operation lock: mark stopped, then release the hard lease. + """Acknowledge under the operation lock; resource readback releases the lease later. Only the operation-lock holder calls this. The record is transitioned as - it is on disk, so state that a fenced write refused stays unwritten; the - lease is released from what this process acquired, which may be newer - than the record. A missing stop, one already acknowledged or finished, + it is on disk, so state that a fenced write refused stays unwritten. An + ACK cannot prove the nested Host drained and must not release its + lease. A missing stop, one already acknowledged or finished, and a record that already reached a terminal observation stay untouched. """ @@ -1189,11 +1213,6 @@ def _acknowledge_stop(self, path: Path, row: dict, binding: dict, *, source: str self._clear_delegation_bootstrap(row, binding) except (OSError, ValueError): pass # the bootstrap is host input; its state never blocks the receipt - lease = delegation_stop_lease.release(self, row, binding) - with exclusive_file_lock(self._dispatch_lock(path)): - stop = self._read_stop(path) or stop - stop["lease"] = lease - _write(self._stop_path(path), stop) def _settle_stop(self, path: Path) -> dict: with exclusive_file_lock(self._dispatch_lock(path)): @@ -1207,7 +1226,7 @@ def _settle_stop(self, path: Path) -> dict: # A launched Host on a platform that cannot prove its group exited # has no converging stop: say so plainly rather than leaving the # caller with an acknowledged receipt it can never settle. - unsupported = host_process_drain(self._host_process_record(path)) == HOST_PROCESS_UNSUPPORTED_PLATFORM + unsupported = self._delegation_host_drain(path) == HOST_PROCESS_UNSUPPORTED_PLATFORM if unsupported: raise ValueError( "delegation stop cannot prove the launched Host drained on this platform: " @@ -1217,28 +1236,23 @@ def _settle_stop(self, path: Path) -> dict: facts = {"operation_lock_free": self._operation_lock_free(path)} facts["worker_lane_released"], lane_state = self._worker_lane_released(row, stop, binding) # Read last: a Host seen drained after its worker and lane let go stays drained. - facts["host_process"] = host_process_drain(self._host_process_record(path)) - # The obligation comes from the operation record, not from the stop - # sidecar. The acknowledgement writes its ACK before it releases the - # lease, so a process loss in between leaves the sidecar with no - # `lease` field at all — and reading that as "nothing was owed" - # settles a stop whose member still holds an active hard lease, with - # resume already refused and the Todo blocked until the TTL. - # - # The record names the execution key; the canonical lease decides - # whether that execution still holds it and at which version: a - # native claim commits before its annotation is saved, and renewal - # moves the version after it. A release is retried here, under the - # same lock that guards the record: every later read is another - # attempt rather than one failure becoming permanent. - lease_released = delegation_stop_lease.settle(self, path, row, binding, stop) - if lease_released is not None: - facts["lease_released"] = lease_released - decision = effect_runtime_result("collaboration.delegation.stop", { + facts["host_process"] = self._delegation_host_drain(path) + inputs = { "phase": stop["phase"], "acknowledged": stop.get("ack") is not None, "timed_out": time.time() - stop["requested_at"] > DELEGATION_STOP_GRACE_SECONDS, **facts, - }) + } + # Ask the existing typed owner whether ACK/lock/lane/Host facts + # permit settlement before attempting a lease release. This first + # decision is only a preflight, never a persisted receipt. In + # particular, an unavailable nested supervisor cannot hand off a + # lease while its actual Host is still running. + decision = effect_runtime_result("collaboration.delegation.stop", inputs) + if decision["phase"] == "settled": + lease_released = delegation_stop_lease.settle(self, path, row, binding, stop) + if lease_released is not None: + facts["lease_released"] = lease_released + decision = effect_runtime_result("collaboration.delegation.stop", {**inputs, **facts}) if decision["phase"] != stop["phase"] or decision.get("reason") != stop.get("reason"): stop.update(phase=decision["phase"], reason=decision.get("reason")) if decision["phase"] in DELEGATION_STOP_TERMINAL_PHASES: @@ -1302,9 +1316,16 @@ def _cli(self, binding: dict, *args: str, timeout: int = 60, from .control_plane.turn_driver.host_process_transport import run_host_process chunks = [] + nested_record_args = [] + if host_record is not None: + # Only our private CLI re-arms the marker for the real Turn + # transport. Never pass it through an arbitrary user Host. + environment[HOST_PROCESS_RECORD_ENV] = str(host_record.with_suffix(".cli.host.json")) + nested_record_args = ["--host-process-record", str(host_record)] try: observation = run_host_process( - [*_python_module_command("loopx.control_plane.turn_driver.delegated_cli"), *arguments], + [*_python_module_command("loopx.control_plane.turn_driver.delegated_cli"), + *nested_record_args, *arguments], project=Path(binding["workspace"]), input_text="", timeout_seconds=timeout, environment=environment, delegated_lease=delegated_lease, on_stdout=chunks.append, @@ -1387,9 +1408,9 @@ def execute(self, operation_id: str) -> None: if row["status"] == "prepared": self._observe(path, row, "rejected") except DelegationStopRequested as stop: - # SIGTERM, a checkpoint or a fenced write: the host child is already - # gone (its run-once exits with this process's exception), the - # bootstrap was cleared, and only the acknowledgement remains. + # SIGTERM, a checkpoint or a fenced write acknowledges intent. + # Each Host supervisor may still be cleaning its own group; + # only subsequent resource readback may settle or release. self._acknowledge_stop(path, row, binding, source=stop.source) def _execution_arguments(self, binding: dict, operation_id: str) -> list[str]: diff --git a/loopx/control_plane/collaboration/delegation_stop_lease.py b/loopx/control_plane/collaboration/delegation_stop_lease.py index 3a57c59c4c..80c5a1b1f5 100644 --- a/loopx/control_plane/collaboration/delegation_stop_lease.py +++ b/loopx/control_plane/collaboration/delegation_stop_lease.py @@ -112,9 +112,9 @@ def settle(service, path, row, binding, stop): if recorded.get("released") is True: return True released = release(service, row, binding) - if released.get("required") is not True: - return None # nothing owed: the receipt keeps no lease fact if released != stop.get("lease"): stop["lease"] = released _write(service._stop_path(path), stop) + if released.get("required") is not True: + return None # preserve the no-obligation receipt, but add no release fact return None if released.get("held") is False else released.get("released") is True diff --git a/loopx/control_plane/turn_driver/delegated_cli.py b/loopx/control_plane/turn_driver/delegated_cli.py index 065f88d132..84eb828ff7 100644 --- a/loopx/control_plane/turn_driver/delegated_cli.py +++ b/loopx/control_plane/turn_driver/delegated_cli.py @@ -4,8 +4,12 @@ normal stack unwinding so each managed-process transport closes its control pipe and waits for its own Host cleanup before this CLI exits. """ +import os import runpy import signal +import sys + +from .host_process_transport import HOST_PROCESS_RECORD_ENV def _cancel(_signal, _frame): @@ -13,5 +17,10 @@ def _cancel(_signal, _frame): if __name__ == "__main__": + # This private launcher alone carries the nested record across the outer + # leased supervisor. The actual Host transport consumes it before launch. + if len(sys.argv) >= 3 and sys.argv[1] == "--host-process-record": + os.environ[HOST_PROCESS_RECORD_ENV] = sys.argv[2] + del sys.argv[1:3] signal.signal(signal.SIGTERM, _cancel) runpy.run_module("loopx.cli", run_name="__main__") From 0b78b3d4b0bf4527b083c691438a2cad985585e2 Mon Sep 17 00:00:00 2001 From: song Date: Sat, 3 Oct 2026 01:27:52 +0800 Subject: [PATCH 22/54] test(delegation): cover unavailable nested supervisors and drain attribution Signed-off-by: song --- tests/control_plane/test_host_process.py | 5 +- tests/test_delegation_stop_nested_host.py | 112 ++++++++++++++++++++++ 2 files changed, 116 insertions(+), 1 deletion(-) create mode 100644 tests/test_delegation_stop_nested_host.py diff --git a/tests/control_plane/test_host_process.py b/tests/control_plane/test_host_process.py index d81f05582f..eddbd1a134 100644 --- a/tests/control_plane/test_host_process.py +++ b/tests/control_plane/test_host_process.py @@ -328,6 +328,7 @@ def test_hard_leased_cli_pins_the_release_and_the_record_through_the_transport(t def transport(*args, **kwargs): handed.update(kwargs) + handed["argv"] = args[0] kwargs["on_stdout"]("{}") return {"outcome": "exited", "output_complete": True, "returncode": 0} @@ -337,4 +338,6 @@ def transport(*args, **kwargs): host_record=record, delegated_lease={"lease": {}, "ttl_seconds": None, "renew_argv": [], "read_argv": []}) assert handed["environment"]["PYTHONPATH"].split(os.pathsep)[0] == str(delegation_module._release_root()) - assert handed["environment"][HOST_PROCESS_RECORD_ENV] == str(record) + assert handed["environment"][HOST_PROCESS_RECORD_ENV] == str(record.with_suffix(".cli.host.json")) + index = handed["argv"].index("--host-process-record") + assert handed["argv"][index + 1] == str(record) diff --git a/tests/test_delegation_stop_nested_host.py b/tests/test_delegation_stop_nested_host.py new file mode 100644 index 0000000000..69a9b7ec64 --- /dev/null +++ b/tests/test_delegation_stop_nested_host.py @@ -0,0 +1,112 @@ +"""Hard-lease stop must account for the real nested Host before safe handoff.""" +from __future__ import annotations + +# Imported fixtures are parameterized over real File and SQLite authorities. +# ruff: noqa: F811 +import os +import signal + +import pytest + +from test_local_delegation import HOST, process_gone, service, until # noqa: F401 +from test_delegation_lease_lifetime import inspect, prepare_lease +from loopx.collaboration_mcp import Delegations +from loopx.control_plane.collaboration.inbox import _read +from loopx.control_plane.collaboration.peers import returns + + +@pytest.mark.skipif(os.name == "nt", reason="POSIX nested supervisor interruption") +@pytest.mark.parametrize("pause_supervisor", [False, True]) +def test_stop_waits_for_actual_nested_host_before_releasing_lease(service, monkeypatch, pause_supervisor): + root, runner = service + operation = "nested-stop" + with monkeypatch.context() as setup: + original = prepare_lease(root, runner, setup, operation_id=operation, renew=False) + # Identify the actual supervisor from its child, independently of the + # implementation's recorded process attribution. + host = HOST.replace("counter = workspace / 'host-invocations'", """ +(root / 'supervisor-pid').write_text(str(os.getppid())) +(root / 'host-record-env').write_text(str(os.environ.get('LOOPX_HOST_PROCESS_RECORD'))) +counter = workspace / 'host-invocations'""") + (root / "fixture-host.py").write_text(host) + (root / "hold").touch() + (root / "ignore-term").touch() + runner._spawn(operation) + host_pid = child_pid = supervisor = None + try: + assert until(lambda: (root / "host-started").exists()), _read(runner.path(operation)) + host_pid = int((root / "host-pid").read_text()) + child_pid = int((root / "host-child-pid").read_text()) + supervisor = int((root / "supervisor-pid").read_text()) + assert not process_gone(host_pid) and not process_gone(child_pid) + assert (root / "host-record-env").read_text() == "None" + if pause_supervisor: + os.kill(supervisor, signal.SIGSTOP) + receipt = runner.stop(operation, execute=True) + if pause_supervisor: + assert receipt["phase"] == "acknowledged", receipt + assert receipt["reason"] == "host_process_still_running", receipt + assert not process_gone(host_pid) and not process_gone(child_pid) + lease = inspect(runner) + assert lease["active"] and lease["lease"]["idempotency_key"] == original["idempotency_key"] + # A fresh requester must preserve the same receipt and lease while + # the unavailable supervisor leaves actual execution alive. + fresh = Delegations(runner.root, runner.registry, runner.goal_id, runner.agent_id, runner.config) + repeated = fresh.stop(operation, execute=True) + assert repeated["phase"] == "acknowledged" + assert repeated["stop"]["stop_id"] == receipt["stop"]["stop_id"] + assert inspect(runner)["active"] + # Cleanup only the independently identified fixture group. The + # product readback must observe this, never signal unrelated groups. + os.killpg(host_pid, signal.SIGKILL) + assert until(lambda: process_gone(host_pid) and process_gone(child_pid)) + receipt = fresh.stop(operation, execute=True) + assert receipt["phase"] == "settled", receipt + assert process_gone(host_pid) and process_gone(child_pid), receipt + assert inspect(runner)["lease"]["status"] == "released" + assert runner.stop(operation, execute=True) == receipt + assert runner.read(operation)["stop"]["phase"] == "settled" + with pytest.raises(ValueError, match="start a new operation id"): + runner.resume(operation) + assert (root / "analyst" / "initial" / "host-invocations").read_text() == "1" + assert returns(runner.root, runner.goal_id, "lead")["items"] == [] + assert runner.operations()["page_readback_complete"] + finally: + for pid, group in ((host_pid, True), (supervisor, False)): + if pid is not None: + try: + os.killpg(pid, signal.SIGKILL) if group else os.kill(pid, signal.SIGKILL) + except ProcessLookupError: + pass + + +@pytest.mark.skipif(os.name == "nt", reason="POSIX process-group attribution") +@pytest.mark.parametrize("legacy_record", [True, False]) +def test_missing_or_unreadable_nested_attribution_cannot_release_a_lease(service, monkeypatch, legacy_record): + """Outer-only historical evidence or a corrupt inner record is not drain.""" + import json + import subprocess + + from loopx.control_plane.turn_driver.host_process_transport import HOST_PROCESS_RECORD_SCHEMA_VERSION + from loopx.file_lock import lock_holder_host_label + + root, runner = service + operation = "unproven-nested-stop" + prepare_lease(root, runner, monkeypatch, operation_id=operation, renew=False) + dead = subprocess.Popen(["true"]) + dead.wait(timeout=5) + record = runner._host_process_record(runner.path(operation)) + outer = {"schema_version": HOST_PROCESS_RECORD_SCHEMA_VERSION, + "host": lock_holder_host_label(), "phase": "finished", + "bridge_pid": dead.pid, "process_group": dead.pid} + if legacy_record: + record.write_text(json.dumps(outer)) + else: + record.with_suffix(".cli.host.json").write_text(json.dumps(outer)) + record.write_text("{corrupt nested attribution") + evidence = record.read_bytes() + stopped = runner.stop(operation, execute=True) + assert stopped["phase"] == "acknowledged" + assert stopped["reason"] == "host_process_drain_unproven" + assert inspect(runner)["active"] + assert record.read_bytes() == evidence From 89dc8b969c2f5a3ccdedec5d1d9da24174606216 Mon Sep 17 00:00:00 2001 From: song Date: Sat, 3 Oct 2026 01:27:52 +0800 Subject: [PATCH 23/54] docs(delegation): explain nested drain and explicit stop recovery Signed-off-by: song --- docs/reference/local-delegation.md | 21 +++++++++++++++++---- 1 file changed, 17 insertions(+), 4 deletions(-) diff --git a/docs/reference/local-delegation.md b/docs/reference/local-delegation.md index ce8e667f31..14ed01bd07 100644 --- a/docs/reference/local-delegation.md +++ b/docs/reference/local-delegation.md @@ -229,7 +229,15 @@ operation cannot overwrite it. A worker on this machine receives `SIGTERM` for its whole process group, which ends its Turn child; the native host runs in its own process group, and its supervisor terminates that group once the Turn child is gone. The worker acknowledges from under its own lock, marks the -record `stopped` and releases its hard task lease. When nobody holds the operation, the requester +record `stopped`. Lease release waits for resource readback to prove that +the worker, Turn lane and every owned Host group have drained. In hard-lease +mode, the leased CLI supervisor and the actual nested Host have separate +records and process groups; neither one proves the other has exited. The +private CLI forwards a separate record address to the Turn transport, which +consumes it before launching user Host code. Old hard-lease records that only +cover the outer CLI cannot prove drain and remain `acknowledged`; reconcile +that original execution rather than deleting its evidence or reusing its key. +When nobody holds the operation, the requester acknowledges itself. A worker on another machine is never signalled; it finds the request at its next checkpoint or at its next record write, which is refused. The receipt `phase` is `settled` only when an acknowledgement exists, @@ -240,7 +248,7 @@ and a required hard task lease was actually released. The obligation is read from the operation record, not from the stop sidecar: the acknowledgement is persisted before the lease is released, so a crash in between must not turn "not yet written" into "nothing was owed". A release that failed is -retried under the stop's own lock on the next read, so it never becomes a +retried under the stop's own lock on the next explicit `stop`, so it never becomes a `settled` receipt that leaves the member's Todo blocked until the lease TTL; while it is unproven the stop stays `acknowledged` with `required_lease_release_unproven`. If that host cannot be attributed, or its @@ -269,13 +277,18 @@ before completion starts; a lock-acquisition timeout requires retrying `stop`. 因此仍持有该 operation 的 worker 无法覆盖它。本机 worker 会收到整个进程组的 `SIGTERM`,其 Turn 子进程随之结束;原生 host 在自己的进程组中运行,Turn 子进程 退出后由其 supervisor 终止整个 host 进程组。worker 在自己的锁下确认,把记录标为 -`stopped` 并释放硬任务租约。没有持有者时由请求方自行确认。另一台机器上的 +`stopped`。只有读回证明 worker、Turn lane 及所有归属的 Host 进程组都已退出, +才会释放硬任务租约。硬租约模式的外层 CLI 与内层真实 Host 分别记录、分别检查, +外层退出不能证明内层退出。私有 CLI 只把独立记录地址交给 Turn transport, +由它在启动用户 Host 前消费,用户 Host 不继承该标记。旧硬租约记录若只覆盖外层 +CLI,则无法证明排空,保持 `acknowledged`;应核对原执行,不能删除证据或复用其 key。 +没有持有者时由请求方自行确认。另一台机器上的 worker 不会被发信号,它在下一个检查点或下一次写记录时发现请求,写入被拒绝。 只有存在确认、operation 锁已释放、成员 Turn lane 的持有者记录显示已被停止的 worker 释放(只读 lane,从不获取)、该 Turn 启动的原生 host 及其进程组内所有进程 都已退出,且必需的硬任务租约确实释放成功时,`phase` 才是 `settled`。该义务取自 操作记录而非 stop sidecar:确认会先于释放落盘,因此两者之间发生崩溃时,不能把 -「尚未写入」当成「本就不需要释放」。释放失败会在下一次读取时于 stop 自己的锁下重试, +「尚未写入」当成「本就不需要释放」。释放失败会在下一次显式调用 `stop` 时于其锁下重试, 因此不会产生一份「已结算」却让成员 Todo 被租约阻塞到 TTL 的回执;在释放得到证明前,停止保持 `acknowledged`,原因为 `required_lease_release_unproven`。host 无法归属或其 supervisor 未完成清理时, 停止保持 `acknowledged`,之后再次调用 `stop` 会重新读取。在没有进程组的平台上, From 2add90002dacfb93c6cbd547b0af29cd9f1cf4b5 Mon Sep 17 00:00:00 2001 From: song Date: Sat, 3 Oct 2026 08:29:31 +0800 Subject: [PATCH 24/54] fix(delegation): reject unsupported stop before cancellation effects Signed-off-by: song --- loopx/collaboration_mcp.py | 27 +++++++++++++++++---------- tests/test_local_delegation.py | 32 +++++++++++++++++++++++++++++--- 2 files changed, 46 insertions(+), 13 deletions(-) diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index c109c164f9..d67d1b6bf3 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -1075,9 +1075,12 @@ def stop(self, operation_id: str, *, execute: bool) -> dict: if stop is None: if row["status"] in DELEGATION_TERMINAL_STATUSES: return self._stop_receipt(row, binding, None) + self._require_host_drain_support(path) stop = self._new_stop_record(row, requested_by=self.agent_id, worker=self._lock_holder_worker(path, row)) _write(self._stop_path(path), stop) + elif stop["phase"] in DELEGATION_STOP_OPEN_PHASES: + self._require_host_drain_support(path) if stop["phase"] in DELEGATION_STOP_OPEN_PHASES and stop.get("ack") is None: if stop.get("worker") is None: # No worker was named when the request was written, so whoever owns @@ -1221,6 +1224,19 @@ def _acknowledge_stop(self, path: Path, row: dict, binding: dict, *, source: str except (OSError, ValueError): pass # the bootstrap is host input; its state never blocks the receipt + def _require_host_drain_support(self, path: Path) -> None: + """Reject under the dispatch fence before cancellation writes or signals. + + Keep the same guard during settlement for an existing cancellation + intent restored on a platform that cannot prove Host drain. + """ + if self._delegation_host_drain(path) == HOST_PROCESS_UNSUPPORTED_PLATFORM: + raise ValueError( + "delegation stop cannot prove the launched Host drained on this platform: " + "process groups are unavailable, so the Host supervisor is best-effort. " + "Stop the member's Host through its own supervisor and re-read the receipt." + ) + def _settle_stop(self, path: Path) -> dict: with exclusive_file_lock(self._dispatch_lock(path)): row = _read(path) @@ -1230,16 +1246,7 @@ def _settle_stop(self, path: Path) -> dict: return self._stop_receipt(row, binding, None) if stop["phase"] not in DELEGATION_STOP_OPEN_PHASES: return self._stop_receipt(row, binding, stop) - # A launched Host on a platform that cannot prove its group exited - # has no converging stop: say so plainly rather than leaving the - # caller with an acknowledged receipt it can never settle. - unsupported = self._delegation_host_drain(path) == HOST_PROCESS_UNSUPPORTED_PLATFORM - if unsupported: - raise ValueError( - "delegation stop cannot prove the launched Host drained on this platform: " - "process groups are unavailable, so the Host supervisor is best-effort. " - "Stop the member's Host through its own supervisor and re-read the receipt." - ) + self._require_host_drain_support(path) facts = {"operation_lock_free": self._operation_lock_free(path)} facts["worker_lane_released"], lane_state = self._worker_lane_released(row, stop, binding) # Read last: a Host seen drained after its worker and lane let go stays drained. diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index 94720ee247..ec5b6c8213 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -877,13 +877,39 @@ def test_a_launched_host_on_a_platform_without_process_groups_fails_fast(service "bridge_pid": os.getpid(), "process_group": os.getpid(), })) - # The launched Host is real, but this platform cannot prove it drained. + # A persisted finished Host record still needs platform drain capability. assert host_process_transport.host_process_drain(record) == host_process_transport.HOST_PROCESS_DRAINED monkeypatch.delattr(host_process_transport.os, "killpg", raising=False) assert host_process_transport.host_process_drain(record) == host_process_transport.HOST_PROCESS_UNSUPPORTED_PLATFORM - with pytest.raises(ValueError, match="cannot prove the launched Host drained"): - runner.stop("analysis-platform", execute=True) + operation_before = path.read_bytes() + stop_path = runner._stop_path(path) + assert not stop_path.exists() + + def unexpected_effect(*args, **kwargs): + pytest.fail("unsupported stop must not acknowledge, signal, or release its lease") + + monkeypatch.setattr(runner, "_acknowledge_stop", unexpected_effect) + monkeypatch.setattr(runner, "_signal_worker", unexpected_effect) + from loopx.control_plane.collaboration import delegation_stop_lease + monkeypatch.setattr(delegation_stop_lease, "settle", unexpected_effect) + for _ in range(2): + with pytest.raises(ValueError, match="cannot prove the launched Host drained"): + runner.stop("analysis-platform", execute=True) + assert path.read_bytes() == operation_before + assert not stop_path.exists() + + # An existing request remains recoverable evidence, never a fabricated ACK. + stop = runner._new_stop_record( + json.loads(operation_before), requested_by=runner.agent_id, worker=None, + ) + stop_path.write_text(json.dumps(stop)) + stop_before = stop_path.read_bytes() + for _ in range(2): + with pytest.raises(ValueError, match="cannot prove the launched Host drained"): + runner.stop("analysis-platform", execute=True) + assert path.read_bytes() == operation_before + assert stop_path.read_bytes() == stop_before def test_a_crash_between_the_ack_and_the_lease_result_keeps_the_stop_open(service, monkeypatch): From 0c7037e878d6b22916f594d726bc02339a1be4bc Mon Sep 17 00:00:00 2001 From: song Date: Sat, 3 Oct 2026 08:29:31 +0800 Subject: [PATCH 25/54] docs(delegation): clarify side-effect-free platform refusal Signed-off-by: song --- docs/reference/local-delegation.md | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/docs/reference/local-delegation.md b/docs/reference/local-delegation.md index 14ed01bd07..358c65aa10 100644 --- a/docs/reference/local-delegation.md +++ b/docs/reference/local-delegation.md @@ -255,8 +255,9 @@ while it is unproven the stop stays `acknowledged` with supervisor never finished cleaning up, the stop stays `acknowledged` and a later `stop` rereads it. On a platform without process groups the launched host cannot be proven drained at all, so `stop --execute` fails with an -actionable error naming that boundary rather than leaving a receipt no read -can settle. `unknown` +actionable error naming that boundary before writing a cancellation intent, +acknowledging, signalling a worker, or releasing a lease. Repeating a refused +request preserves the operation and any existing stop receipt unchanged. `unknown` means the holder vanished before acknowledging, and `noop` means the work was already accepted, rejected or stopped. `requested` or `acknowledged` means it is still winding down: call `stop` again. A grace timeout never turns From ddd14113fc9a53a53d6435f04fe5b74ac6eb4895 Mon Sep 17 00:00:00 2001 From: song Date: Sat, 3 Oct 2026 08:50:17 +0800 Subject: [PATCH 26/54] fix(delegation): refuse unsupported stop across the Host launch window Signed-off-by: song --- loopx/collaboration_mcp.py | 16 +++++++++++-- tests/test_local_delegation.py | 41 +++++++++++++++++++++++++++++++++- 2 files changed, 54 insertions(+), 3 deletions(-) diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index d67d1b6bf3..c7151030bc 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -29,6 +29,7 @@ from .file_lock import ( exclusive_file_lock, lock_holder_host_label, lock_holder_liveness, + LOCK_HOLDER_ABSENT, LOCK_HOLDER_DEAD, LOCK_HOLDER_RELEASED, LOCK_HOLDER_FOREIGN_HOST, LOCK_HOLDER_LIVE, LockAcquisitionPolicy, LockAcquireTimeoutError, ) from .control_plane.effect_runtime import ( @@ -45,7 +46,7 @@ from .control_plane.turn_driver.host_process_transport import ( HOST_PROCESS_DRAINED, HOST_PROCESS_DRAINING, HOST_PROCESS_NOT_LAUNCHED, HOST_PROCESS_RECORD_ENV, HOST_PROCESS_UNATTRIBUTABLE, HOST_PROCESS_UNSUPPORTED_PLATFORM, - host_process_drain, + host_process_drain, host_process_drain_supported, ) from .control_plane.turn_driver.lane_fence import ( TURN_LANE_ABSENT, TURN_LANE_DEAD, TURN_LANE_LIVE, TURN_LANE_RELEASED, @@ -1230,7 +1231,18 @@ def _require_host_drain_support(self, path: Path) -> None: Keep the same guard during settlement for an existing cancellation intent restored on a platform that cannot prove Host drain. """ - if self._delegation_host_drain(path) == HOST_PROCESS_UNSUPPORTED_PLATFORM: + drain = self._delegation_host_drain(path) + unsupported = drain == HOST_PROCESS_UNSUPPORTED_PLATFORM + if not host_process_drain_supported(): + # An active worker can have passed its last checkpoint without yet + # creating the Host record. Reading absence alone races that launch. + # The dispatch fence prevents a new worker's fenced entry while we + # check the holder; do not probe/take its operation lock here. + holder, _ = lock_holder_liveness(path) + unsupported = drain != HOST_PROCESS_NOT_LAUNCHED or holder not in { + LOCK_HOLDER_ABSENT, LOCK_HOLDER_DEAD, LOCK_HOLDER_RELEASED, + } + if unsupported: raise ValueError( "delegation stop cannot prove the launched Host drained on this platform: " "process groups are unavailable, so the Host supervisor is best-effort. " diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index ec5b6c8213..10ba0d81ed 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -496,7 +496,9 @@ def test_stop_without_a_holder_is_acknowledged_by_the_requester(service, monkeyp root, runner = service monkeypatch.setattr(runner, "_spawn", lambda _: None) runner.start("analysis", "analysis-idle", brief()) - receipt = runner.stop("analysis-idle", execute=True) + with monkeypatch.context() as platform: + platform.delattr(os, "killpg", raising=False) + receipt = runner.stop("analysis-idle", execute=True) assert receipt["phase"] == "settled" and receipt["status"] == "stopped" assert receipt["stop"]["worker"] is None and receipt["stop"]["requested_status"] == "prepared" assert receipt["stop"]["ack"]["pid"] == os.getpid() and receipt["stop"]["ack"]["source"] == "requester" @@ -912,6 +914,43 @@ def unexpected_effect(*args, **kwargs): assert stop_path.read_bytes() == stop_before + +def test_unsupported_stop_during_host_launch_preserves_continuation(service, monkeypatch): + """No record yet does not prove an active worker cannot launch a Host.""" + from loopx.control_plane.turn_driver import host_process_transport + + root, runner = service + monkeypatch.setattr(runner, "_spawn", lambda _: None) + runner.start("analysis", "platform-launch", brief()) + path = runner.path("platform-launch") + entering, continue_launch = Event(), Event() + cli = runner._cli + + def pause_launch(binding, *args, **kwargs): + if args[:2] == ("turn", "run-once") and kwargs.get("host_record"): + entering.set() + assert continue_launch.wait(20) + return cli(binding, *args, **kwargs) + + monkeypatch.setattr(runner, "_cli", pause_launch) + with ThreadPoolExecutor(max_workers=1) as pool: + worker = pool.submit(runner.execute, "platform-launch") + try: + assert entering.wait(20) + assert not runner._host_process_record(path).exists() + before = path.read_bytes() + with monkeypatch.context() as platform: + platform.delattr(host_process_transport.os, "killpg", raising=False) + with pytest.raises(ValueError, match="cannot prove the launched Host drained"): + runner.stop("platform-launch", execute=True) + assert path.read_bytes() == before + assert not runner._stop_path(path).exists() + finally: + continue_launch.set() + worker.result(timeout=60) + assert runner.read("platform-launch")["status"] == "accepted" + assert (root / "analyst" / "initial" / "host-invocations").read_text() == "1" + def test_a_crash_between_the_ack_and_the_lease_result_keeps_the_stop_open(service, monkeypatch): """The lease obligation survives process loss after the acknowledgement. From 071d2abc23203d85422e26d53f3cbaf15ca77d70 Mon Sep 17 00:00:00 2001 From: song Date: Sat, 3 Oct 2026 08:50:17 +0800 Subject: [PATCH 27/54] docs(delegation): explain platform refusal before Host attribution Signed-off-by: song --- docs/reference/local-delegation.md | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/docs/reference/local-delegation.md b/docs/reference/local-delegation.md index 358c65aa10..bf2496c4a4 100644 --- a/docs/reference/local-delegation.md +++ b/docs/reference/local-delegation.md @@ -257,7 +257,9 @@ later `stop` rereads it. On a platform without process groups the launched host cannot be proven drained at all, so `stop --execute` fails with an actionable error naming that boundary before writing a cancellation intent, acknowledging, signalling a worker, or releasing a lease. Repeating a refused -request preserves the operation and any existing stop receipt unchanged. `unknown` +request preserves the operation and any existing stop receipt unchanged. An +active or unattributable worker is also refused before its Host record appears; +a not-yet-started operation with no holder can still be cancelled. `unknown` means the holder vanished before acknowledging, and `noop` means the work was already accepted, rejected or stopped. `requested` or `acknowledged` means it is still winding down: call `stop` again. A grace timeout never turns @@ -294,7 +296,9 @@ worker 释放(只读 lane,从不获取)、该 Turn 启动的原生 host `required_lease_release_unproven`。host 无法归属或其 supervisor 未完成清理时, 停止保持 `acknowledged`,之后再次调用 `stop` 会重新读取。在没有进程组的平台上, 启动过的 host 根本无法被证明已收尾,因此 `stop --execute` 会以指明该平台边界的 -可操作错误失败,而不是留下一份任何读取都无法结算的回执。`unknown` 表示持有者在确认前消失;`noop` 表示工作已 accepted、 +可操作错误在写入停止意图、确认、发送信号或释放租约之前失败,重复拒绝不修改原记录。 +Host 记录尚未出现但 worker 仍活跃或无法归属时也拒绝;没有持有者且尚未启动的 +operation 仍可安全取消。`unknown` 表示持有者在确认前消失;`noop` 表示工作已 accepted、 rejected 或 stopped;`requested`/`acknowledged` 表示仍在收尾,再次调用 `stop`。 宽限期超时永远不会变成回执。已停止的工作不能 `resume`,新范围需要新的 operation id。Turn journal 保留 `in_progress` 条目供检查,记录不会被改写成完成; From c5ef416be310a962c60e9aabf6fafb2ec8553329 Mon Sep 17 00:00:00 2001 From: song Date: Sat, 3 Oct 2026 08:56:25 +0800 Subject: [PATCH 28/54] refactor(delegation): isolate worker stop signal adaptation Signed-off-by: song --- loopx/collaboration_mcp.py | 73 +---------------- .../collaboration/delegation_stop_signal.py | 78 +++++++++++++++++++ tests/test_local_delegation.py | 5 +- 3 files changed, 86 insertions(+), 70 deletions(-) create mode 100644 loopx/control_plane/collaboration/delegation_stop_signal.py diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index c7151030bc..48287397ba 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -52,6 +52,9 @@ TURN_LANE_ABSENT, TURN_LANE_DEAD, TURN_LANE_LIVE, TURN_LANE_RELEASED, turn_lane_liveness, turn_lane_target, ) +from .control_plane.collaboration.delegation_stop_signal import ( + DelegationFenced, DelegationStopRequested, WorkerStopSignal, install_worker_stop_signal, +) from .control_plane.collaboration.inbox import _hash, _read, _write, _root, _receipt from .control_plane.collaboration.delegation_inventory import ( DELEGATION_HOST_PROCESS_SUFFIX, DELEGATION_STOP_RECEIPT_SUFFIX, @@ -359,74 +362,6 @@ def consume_peer_result(request_id: str, result_key: str = "conclusion") -> dict DELEGATION_STOPPED_MESSAGE = "delegation operation was stopped; start a new operation id" -class DelegationStopRequested(BaseException): - """A stop reached the worker that owns this operation; it must acknowledge, not finish. - - A ``BaseException`` like ``KeyboardInterrupt``: a termination request must - not be swallowed by an ``except Exception`` and turned into further work. - """ - - def __init__(self, source: str) -> None: - super().__init__(source) - self.source = source - - -class DelegationFenced(DelegationStopRequested): - """A stop this process never acknowledged fences its execution-record write. - - The write is refused before it happens: a late-returning or other-host - worker records no Turn result, completes no Todo and publishes nothing. - """ - - def __init__(self) -> None: - super().__init__("fenced") - - -class _WorkerStopSignal: - """Turn SIGTERM into a stop request only when a stop was written for this operation. - - Without a stop receipt the signal keeps its default meaning, so a shutdown - still leaves the operation recoverable by ``resume`` instead of stopping it. - Later signals are absorbed while the acknowledgement is written. - """ - - def __init__(self, stop_path: Path) -> None: - self.stop_path = stop_path - self.armed = True - - def __call__(self, signum: int, frame: object) -> None: - if not self.armed: - return - try: - requested = self.stop_path.exists() - except OSError: - requested = False - if not requested: - signal.signal(signum, signal.SIG_DFL) - os.kill(os.getpid(), signum) - return - self.armed = False - raise DelegationStopRequested("SIGTERM") - - def disarm(self) -> None: - self.armed = False - - -def install_worker_stop_signal(stop_path: Path) -> _WorkerStopSignal | None: - """Install the detached worker's SIGTERM handler; ``None`` where signals are unsupported.""" - - if not hasattr(signal, "SIGTERM"): - return None - handler = _WorkerStopSignal(stop_path) - try: - signal.signal(signal.SIGTERM, handler) - except (ValueError, OSError): - # Not the main thread, or a platform without handler support: the - # worker still honours stop files at every checkpoint and fenced write. - return None - return handler - - def execution_row_path(root: Path, goal_id: str, agent_id: str, operation_id: str) -> Path: """The requester-scoped durable operation record; readable without a service.""" return _root(root) / "executions" / _hash([goal_id, agent_id]) / (_hash(operation_id) + ".json") @@ -471,7 +406,7 @@ def __init__(self, root: Path, registry: Path, goal_id: str, agent_id: str, conf self.root, self.registry = root.resolve(), registry.resolve() self.goal_id, self.agent_id, self.config = goal_id, agent_id, config.resolve() self._goal_ref_lock = Lock() - self._stop_signal: _WorkerStopSignal | None = None + self._stop_signal: WorkerStopSignal | None = None try: self.goal_ref = capture_collaboration_goal_ref( self.registry, diff --git a/loopx/control_plane/collaboration/delegation_stop_signal.py b/loopx/control_plane/collaboration/delegation_stop_signal.py new file mode 100644 index 0000000000..2781053a3c --- /dev/null +++ b/loopx/control_plane/collaboration/delegation_stop_signal.py @@ -0,0 +1,78 @@ +"""Translate worker termination into an explicit delegation stop when requested. + +This host adapter owns signal unwinding only. The delegation service persists +ACKs and the typed control plane decides whether execution has drained. +""" +from __future__ import annotations + +import os +import signal +from pathlib import Path + + +class DelegationStopRequested(BaseException): + """A stop reached the worker that owns this operation; it must acknowledge, not finish. + + A ``BaseException`` like ``KeyboardInterrupt``: a termination request must + not be swallowed by an ``except Exception`` and turned into further work. + """ + + def __init__(self, source: str) -> None: + super().__init__(source) + self.source = source + + +class DelegationFenced(DelegationStopRequested): + """A stop this process never acknowledged fences its execution-record write. + + The write is refused before it happens: a late-returning or other-host + worker records no Turn result, completes no Todo and publishes nothing. + """ + + def __init__(self) -> None: + super().__init__("fenced") + + +class WorkerStopSignal: + """Turn SIGTERM into a stop request only when a stop was written for this operation. + + Without a stop receipt the signal keeps its default meaning, so a shutdown + still leaves the operation recoverable by ``resume`` instead of stopping it. + Later signals are absorbed while the acknowledgement is written. + """ + + def __init__(self, stop_path: Path) -> None: + self.stop_path = stop_path + self.armed = True + + def __call__(self, signum: int, frame: object) -> None: + if not self.armed: + return + try: + requested = self.stop_path.exists() + except OSError: + requested = False + if not requested: + signal.signal(signum, signal.SIG_DFL) + os.kill(os.getpid(), signum) + return + self.armed = False + raise DelegationStopRequested("SIGTERM") + + def disarm(self) -> None: + self.armed = False + + +def install_worker_stop_signal(stop_path: Path) -> WorkerStopSignal | None: + """Install the detached worker's SIGTERM handler; ``None`` where signals are unsupported.""" + + if not hasattr(signal, "SIGTERM"): + return None + handler = WorkerStopSignal(stop_path) + try: + signal.signal(signal.SIGTERM, handler) + except (ValueError, OSError): + # Not the main thread, or a platform without handler support: the + # worker still honours stop files at every checkpoint and fenced write. + return None + return handler diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index 10ba0d81ed..28e5aa44a8 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -18,7 +18,10 @@ import research_team as demo # noqa: E402 from test_managed_research_scenario import fixture # noqa: E402 from loopx.control_plane.collaboration import delegation_stop_lease as stop_lease # noqa: E402 -from loopx.collaboration_mcp import DelegationFenced, DelegationStopRequested, Delegations # noqa: E402 +from loopx.collaboration_mcp import Delegations # noqa: E402 +from loopx.control_plane.collaboration.delegation_stop_signal import ( # noqa: E402 + DelegationFenced, DelegationStopRequested, +) from loopx.control_plane.collaboration.peers import returns # noqa: E402 from loopx.control_plane.collaboration.inbox import _read # noqa: E402 from loopx.control_plane.turn_driver.lane_fence import turn_lane_liveness, turn_lane_singleflight # noqa: E402 From 13f4c9367d613a686628946f6c872f59c587ff73 Mon Sep 17 00:00:00 2001 From: song Date: Sat, 3 Oct 2026 10:17:43 +0800 Subject: [PATCH 29/54] test(delegation): isolate cached and child Effect runtime routes Signed-off-by: song --- tests/test_local_delegation.py | 26 ++++++++++++++++++++++++-- 1 file changed, 24 insertions(+), 2 deletions(-) diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index 28e5aa44a8..befb13a07a 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -5,12 +5,14 @@ from pathlib import Path import subprocess import sys +import tempfile import time from concurrent.futures import ThreadPoolExecutor, TimeoutError as FutureTimeout from contextlib import contextmanager from threading import Event, get_ident import pytest +from tests.control_plane.canonical_authority_fixture import isolate_sqlite_runtime from mcp import ClientSession, StdioServerParameters from mcp.client.stdio import stdio_client @@ -60,8 +62,7 @@ @pytest.fixture(params=["file", "sqlite"]) def service(tmp_path, request, monkeypatch): - for name in ("TMPDIR", "TEMP", "TMP"): - monkeypatch.setenv(name, str(tmp_path)) + isolate_sqlite_runtime(tmp_path, monkeypatch) root = tmp_path / "team" demo.prepare(root, provider=request.param) fixture(root) @@ -84,6 +85,27 @@ def brief(): "return_requirement": "Return the independently checked artifact"} +@pytest.mark.parametrize("provider", ["file", "sqlite"]) +def test_delegation_fixture_isolates_cached_and_child_runtime_routes(tmp_path, monkeypatch, provider): + """A warmed parent must use the same private Effect server as its CLI.""" + from types import SimpleNamespace + + from loopx.control_plane.effect_runtime import _runtime_dir + + cached, isolated = tmp_path / "cached", tmp_path / "isolated" + cached.mkdir() + isolated.mkdir() + monkeypatch.setattr(tempfile, "tempdir", str(cached)) + service.__wrapped__(isolated, SimpleNamespace(param=provider), monkeypatch) + + assert _runtime_dir().parent == isolated + child = subprocess.check_output([ + sys.executable, "-c", + "from loopx.control_plane.effect_runtime import _runtime_dir; print(_runtime_dir())", + ], text=True).strip() + assert Path(child) == _runtime_dir() + + @pytest.mark.parametrize("operation", ["--help", "x y", "x\ny", "x;echo", "x/../y"]) def test_worker_rejects_unbounded_operation_arguments(tmp_path, monkeypatch, operation): from loopx import collaboration_mcp as delegation From 5cb742472f9430a3d756b110431efdcd5c84e960 Mon Sep 17 00:00:00 2001 From: song Date: Sat, 3 Oct 2026 10:46:25 +0800 Subject: [PATCH 30/54] fix(delegation): preserve the first lease supervision failure Signed-off-by: song --- loopx/collaboration_mcp.py | 6 +- .../turn_driver/leased_host_process.ts | 56 ++++++++++++++----- .../control_plane/test_leased_host_process.py | 18 ++++++ 3 files changed, 64 insertions(+), 16 deletions(-) diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index 48287397ba..a376531910 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -1294,7 +1294,11 @@ def _cli(self, binding: dict, *args: str, timeout: int = 60, except RuntimeError as exc: raise ValueError("delegation managed CLI supervision unavailable; reconcile the original Turn") from exc if observation["outcome"] != "exited" or not observation["output_complete"]: - raise ValueError(f"delegation lease supervision stopped ({observation['outcome']}); reconcile the original Turn") + detail = observation["outcome"] + failure = observation.get("lease_failure") + if failure is not None: + detail += f"; lease:{failure['boundary']}/{failure['reason']}" + raise ValueError(f"delegation lease supervision stopped ({detail}); reconcile the original Turn") stdout, returncode = "".join(chunks), observation["returncode"] try: value = json.loads(stdout) diff --git a/loopx/control_plane/turn_driver/leased_host_process.ts b/loopx/control_plane/turn_driver/leased_host_process.ts index 46c746eb3b..15613abfd8 100644 --- a/loopx/control_plane/turn_driver/leased_host_process.ts +++ b/loopx/control_plane/turn_driver/leased_host_process.ts @@ -14,7 +14,23 @@ export interface DelegatedHostLease { ttl_seconds: number; } -class LeaseTransportUnavailable extends Error {} +type LeaseFailureReason = "owner_cancelled" | "execution_proof_rejected" | "lease_inactive" + | "execution_identity_changed" | "transport_unavailable" | "renewal_rejected" + | "renewal_not_advanced" | "proved_deadline_elapsed" | "lease_observation_failed"; +type LeaseFailureBoundary = "initial_proof" | "renewal" | "final_proof" | "deadline"; +interface LeaseFailure {reason: LeaseFailureReason; boundary: LeaseFailureBoundary} +export type LeasedHostProcessResult = HostProcessResult & {lease_failure?: LeaseFailure}; + +class LeaseSupervisionFailure extends Error { + readonly reason: LeaseFailureReason; + constructor(reason: LeaseFailureReason, message: string = reason) { + super(message); + this.reason = reason; + } +} +class LeaseTransportUnavailable extends LeaseSupervisionFailure { + constructor(message: string) { super("transport_unavailable", message); } +} export function decodeDelegatedHostLease(raw: unknown): DelegatedHostLease { const value = requireJsonObject(raw, "delegated Host lease"); @@ -33,12 +49,13 @@ export function decodeDelegatedHostLease(raw: unknown): DelegatedHostLease { export async function runLeasedHostProcess(request: HostProcessRequest, context: DelegatedHostLease, output: (item: HostProcessOutput) => Promise, owner: AbortSignal, - spawned?: (item: HostProcessSpawned) => Promise): Promise { + spawned?: (item: HostProcessSpawned) => Promise): Promise { const controller = new AbortController(); const abort = () => controller.abort(); owner.addEventListener("abort", abort, {once: true}); if (owner.aborted) abort(); - let lease = context.lease, lost = false, finished = false; + let lease = context.lease, finished = false; + let failure: LeaseFailure | undefined; let renewalTimer: ReturnType | undefined; let expiryTimer: ReturnType | undefined; let pending: Promise | undefined; @@ -46,11 +63,13 @@ export async function runLeasedHostProcess(request: HostProcessRequest, context: // execution is still current and returns its latest version, not a new key. const currentProof = (raw: unknown): LeaseRecord => { const result = requireJsonObject(raw, "delegated current proof"); + if (result.ok !== true) throw new LeaseSupervisionFailure("execution_proof_rejected"); const record = requireJsonObject(result.lease, "delegated current lease"); const observed = canonicalTaskLease(record, String(lease.goal_id), String(lease.todo_id)); - if (result.ok !== true || !leaseIsActive(observed, new Date()) || observed.owner !== lease.owner || + if (!leaseIsActive(observed, new Date())) throw new LeaseSupervisionFailure("lease_inactive"); + if (observed.owner !== lease.owner || observed.idempotency_key !== lease.idempotency_key || leaseEpoch(observed) !== leaseEpoch(lease) || - leaseVersion(observed) < leaseVersion(lease)) throw new Error("delegated execution proof lost"); + leaseVersion(observed) < leaseVersion(lease)) throw new LeaseSupervisionFailure("execution_identity_changed"); return observed; }; const cli = async (argv: string[]): Promise => { @@ -65,7 +84,14 @@ export async function runLeasedHostProcess(request: HostProcessRequest, context: try { return JSON.parse(stdout); } catch { throw new LeaseTransportUnavailable("delegated lease reply unavailable"); } }; - const lose = () => { lost = true; abort(); }; + const lose = (reason: LeaseFailureReason, boundary: LeaseFailureBoundary) => { + // Keep the first failure, before aborting its in-flight transport. A later + // cancellation or deadline must not replace the original causal boundary. + failure ??= {reason: owner.aborted ? "owner_cancelled" : reason, boundary}; + abort(); + }; + const reject = (error: unknown, boundary: LeaseFailureBoundary) => + lose(error instanceof LeaseSupervisionFailure ? error.reason : "lease_observation_failed", boundary); const clearTimers = () => { clearTimeout(renewalTimer); clearTimeout(expiryTimer); }; @@ -73,9 +99,9 @@ export async function runLeasedHostProcess(request: HostProcessRequest, context: clearTimers(); const expires = parseIsoTimestamp(String(lease.expires_at)); const remaining = expires === null ? 0 : expires.valueOf() - Date.now(); - if (remaining <= 0) { lose(); return; } + if (remaining <= 0) { lose("proved_deadline_elapsed", "deadline"); return; } // Stop at the last proved deadline even if renewal or its transport hangs. - expiryTimer = setTimeout(lose, remaining); + expiryTimer = setTimeout(() => lose("proved_deadline_elapsed", "deadline"), remaining); renewalTimer = setTimeout(() => { pending = (async () => { const argv = [...context.renew_argv, "--expected-version", String(leaseVersion(lease)), @@ -89,19 +115,19 @@ export async function runLeasedHostProcess(request: HostProcessRequest, context: // adopt a newer version by editing the rejected renewal request. renewed = await cli(argv); } - if (requireJsonObject(renewed, "delegated renewal").ok !== true) throw new Error("delegated renewal rejected"); + if (requireJsonObject(renewed, "delegated renewal").ok !== true) throw new LeaseSupervisionFailure("renewal_rejected"); const observed = currentProof(await cli(context.read_argv)); - if (leaseVersion(observed) <= leaseVersion(lease)) throw new Error("delegated lease did not renew"); + if (leaseVersion(observed) <= leaseVersion(lease)) throw new LeaseSupervisionFailure("renewal_not_advanced"); lease = observed; if (!finished) schedule(); - } catch { lose(); } + } catch (error) { reject(error, "renewal"); } })(); }, Math.max(1, Math.min(30_000, Math.floor(remaining / 2)))); }; try { // No Host input or process launch precedes current execution readback. try { lease = currentProof(await cli(context.read_argv)); } - catch { lose(); } + catch (error) { reject(error, "initial_proof"); } schedule(); // The CLI's TERM adapter unwinds the nested Host transport (bounded at // five seconds). Allow that acknowledgement before a forced group kill. @@ -112,13 +138,13 @@ export async function runLeasedHostProcess(request: HostProcessRequest, context: finished = true; clearTimeout(renewalTimer); await pending; - if (!lost) { + if (!failure) { try { lease = currentProof(await cli(context.read_argv)); } - catch { lose(); } + catch (error) { reject(error, "final_proof"); } } finished = true; clearTimers(); - return lost ? {...result, outcome: "cancelled", output_complete: false} : result; + return failure ? {...result, outcome: "cancelled", output_complete: false, lease_failure: failure} : result; } finally { finished = true; clearTimers(); diff --git a/tests/control_plane/test_leased_host_process.py b/tests/control_plane/test_leased_host_process.py index 5a0ad0dcb8..47d01d1d4e 100644 --- a/tests/control_plane/test_leased_host_process.py +++ b/tests/control_plane/test_leased_host_process.py @@ -88,9 +88,13 @@ def test_real_renewal_faults_keep_original_deadline_and_identity(canonical_execu if fault == "lost_reply": assert observed["outcome"] == "exited" and observed["output_complete"] is True assert current["lease"]["version"] > context["lease"]["version"] + assert "lease_failure" not in observed else: assert observed["outcome"] == "cancelled" and observed["output_complete"] is False assert current["lease"]["version"] == context["lease"]["version"] + assert observed["lease_failure"] == ( + {"reason": "renewal_rejected", "boundary": "renewal"} if fault == "rejected" else + {"reason": "proved_deadline_elapsed", "boundary": "deadline"}) @pytest.mark.skipif(os.name == "nt", reason="POSIX owned process-group qualification") @@ -109,3 +113,17 @@ def lost_reader(_text): before = marker.read_bytes() time.sleep(0.2) assert marker.read_bytes() == before, "control pipe loss must finish forced cleanup before returning" + + +@pytest.mark.skipif(os.name == "nt", reason="POSIX owned process-group qualification") +def test_released_initial_proof_explains_refusal_before_host_launch(canonical_execution, tmp_path): + context, command, selected = canonical_execution + command("task-lease", "release", *selected, "--owner", "worker", + "--idempotency-key", "original-execution", "--expected-version", "1") + marker = tmp_path / "must-not-launch" + observed = run_host_process([sys.executable, "-c", + "from pathlib import Path; import sys; Path(sys.argv[1]).touch()", str(marker)], + project=tmp_path, input_text="", timeout_seconds=10, delegated_lease=context) + assert observed["outcome"] == "cancelled" and observed["output_complete"] is False + assert not marker.exists() + assert observed["lease_failure"] == {"reason": "lease_inactive", "boundary": "initial_proof"} From 694866a806798b87fe8de86b09b9538080112ea5 Mon Sep 17 00:00:00 2001 From: song Date: Sat, 3 Oct 2026 10:46:25 +0800 Subject: [PATCH 31/54] docs(delegation): explain lease failure readback boundaries Signed-off-by: song --- docs/reference/local-delegation.md | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/docs/reference/local-delegation.md b/docs/reference/local-delegation.md index bf2496c4a4..cb9be0e241 100644 --- a/docs/reference/local-delegation.md +++ b/docs/reference/local-delegation.md @@ -756,6 +756,20 @@ concurrent executions still use the same kernel lock and original Turn journal. ## Disconnect and recovery +A managed delegation error retains the first typed lease failure as +`lease:/`: for example, `initial_proof/lease_inactive`, +`renewal/renewal_rejected`, or `deadline/proved_deadline_elapsed`. A `cancelled` +Host outcome alone does not mean a user requested stop or the lease was released. +Inspect the original execution and canonical lease before recovery. These +observations do not extend deadlines, grant authority, or change stop settlement. +Successful and ordinary unleased Host results retain their existing shape. + +受管委派错误通过 `lease:/` 保留首个类型化租约失败原因, +区分启动前证明失败、续期拒绝和已证明期限到达。仅有 Host 的 `cancelled` +结果不代表用户请求停止,也不证明租约已释放;恢复前需读回原执行与 canonical +租约。这些诊断不延长期限、不授予权限,也不改变停止结算条件;成功执行与普通 +无租约 Host 的结果结构保持不变。 + | Interruption | Behavior and recovery | | --- | --- | | Requesting MCP conversation closes | The detached bounded worker continues; another connection reads the original operation. | From 759349919ac39658173aabefd4a300edf4a584ba Mon Sep 17 00:00:00 2001 From: song Date: Sat, 3 Oct 2026 22:47:02 +0800 Subject: [PATCH 32/54] test(delegation): isolate and retire fixture-owned Effect runtimes The delegation service fixture changed TMPDIR/TEMP/TMP for its children but left Python's cached tempfile.tempdir untouched, so a warmed parent could route to a different Effect runtime than its native CLI children. It also left each case's private runtime alive until the five-minute idle shutdown, accumulating servers across the suite, including failed setups. Reuse the canonical isolate_sqlite_runtime helper for both routes and register a finalizer that shuts the fixture's own runtime down through the existing restart API. The extended isolation test fails on the old fixture and passes afterward; a neighboring runtime survives teardown. Ported from the closed #5308 (13f4c9367, bdbd44076) as a standalone test-infrastructure change. Production code is unchanged. Co-Authored-By: Claude Fable 5.1 Signed-off-by: song --- tests/test_local_delegation.py | 62 ++++++++++++++++++++++++++++++++-- 1 file changed, 60 insertions(+), 2 deletions(-) diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index 45c57a7948..1b8ab86373 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -1,15 +1,18 @@ """Production delegation/Turn/TS completion with an explicit fixture model host.""" import json +import os import asyncio from pathlib import Path import subprocess import sys +import tempfile import time from concurrent.futures import ThreadPoolExecutor, TimeoutError as FutureTimeout from contextlib import contextmanager from threading import Event, get_ident import pytest +from tests.control_plane.canonical_authority_fixture import isolate_sqlite_runtime from mcp import ClientSession, StdioServerParameters from mcp.client.stdio import stdio_client @@ -47,8 +50,22 @@ @pytest.fixture(params=["file", "sqlite"]) def service(tmp_path, request, monkeypatch): - for name in ("TMPDIR", "TEMP", "TMP"): - monkeypatch.setenv(name, str(tmp_path)) + isolate_sqlite_runtime(tmp_path, monkeypatch) + # Each case owns a private server. Do not accumulate five-minute idle + # runtimes across the delegation suite, including failed setup/test cases. + runtime_env = os.environ.copy() + def retire_runtime(): + subprocess.run([ + sys.executable, "-c", + "from pathlib import Path; import sys; " + "from loopx.control_plane.effect_runtime import _runtime_dir, restart_effect_runtime; " + "assert _runtime_dir().parent == Path(sys.argv[1]); " + "result = restart_effect_runtime(); " + "assert result['status'] in {'stopped', 'not_running'}, result", + str(tmp_path), + ], cwd=Path(__file__).resolve().parents[1], env=runtime_env, + capture_output=True, text=True, timeout=30, check=True) + request.addfinalizer(retire_runtime) root = tmp_path / "team" demo.prepare(root, provider=request.param) fixture(root) @@ -71,6 +88,47 @@ def brief(): "return_requirement": "Return the independently checked artifact"} +@pytest.mark.parametrize("provider", ["file", "sqlite"]) +def test_delegation_fixture_isolates_cached_and_child_runtime_routes(tmp_path, monkeypatch, provider): + """A warmed parent must use the same private Effect server as its CLI.""" + from types import SimpleNamespace + + from loopx.control_plane.effect_runtime import ( + _runtime_dir, _serving_runtime_identity, restart_effect_runtime, effect_runtime_result, + ) + + cached, isolated = tmp_path / "cached", tmp_path / "isolated" + cached.mkdir() + isolated.mkdir() + monkeypatch.setattr(tempfile, "tempdir", str(cached)) + finalizers = [] + service.__wrapped__(isolated, SimpleNamespace(param=provider, addfinalizer=finalizers.append), monkeypatch) + + assert _runtime_dir().parent == isolated + child = subprocess.check_output([ + sys.executable, "-c", + "from loopx.control_plane.effect_runtime import _runtime_dir; print(_runtime_dir())", + ], text=True).strip() + assert Path(child) == _runtime_dir() + effect_runtime_result("runtime.ping", {}) + assert _serving_runtime_identity() is not None + try: + # Teardown targets the captured route, even if another test changed + # the parent cache/environment. A neighboring runtime must survive. + with monkeypatch.context() as neighbor: + isolate_sqlite_runtime(cached, neighbor) + neighbor_pid = effect_runtime_result("runtime.ping", {})["pid"] + try: + for finalize in reversed(finalizers): + finalize() + assert effect_runtime_result("runtime.ping", {})["pid"] == neighbor_pid + finally: + restart_effect_runtime() + assert _serving_runtime_identity() is None, "fixture must retire its private runtime before the next case" + finally: + restart_effect_runtime() + + @pytest.mark.parametrize("operation", ["--help", "x y", "x\ny", "x;echo", "x/../y"]) def test_worker_rejects_unbounded_operation_arguments(tmp_path, monkeypatch, operation): from loopx import collaboration_mcp as delegation From f36991ea09931c6059ebf6c6bcadcac26aa82f80 Mon Sep 17 00:00:00 2001 From: song Date: Sat, 3 Oct 2026 12:11:09 +0800 Subject: [PATCH 33/54] fix(delegation): retain the latest proved expiry through final readback Signed-off-by: song --- .../turn_driver/leased_host_process.ts | 5 +- .../control_plane/test_leased_host_process.py | 95 +++++++++++++++++-- 2 files changed, 89 insertions(+), 11 deletions(-) diff --git a/loopx/control_plane/turn_driver/leased_host_process.ts b/loopx/control_plane/turn_driver/leased_host_process.ts index 15613abfd8..809ce78dae 100644 --- a/loopx/control_plane/turn_driver/leased_host_process.ts +++ b/loopx/control_plane/turn_driver/leased_host_process.ts @@ -102,6 +102,9 @@ export async function runLeasedHostProcess(request: HostProcessRequest, context: if (remaining <= 0) { lose("proved_deadline_elapsed", "deadline"); return; } // Stop at the last proved deadline even if renewal or its transport hangs. expiryTimer = setTimeout(() => lose("proved_deadline_elapsed", "deadline"), remaining); + // A returned Host still needs final proof under the latest proved expiry. + // Only periodic renewal ends at Host return; expiry supervision does not. + if (finished) return; renewalTimer = setTimeout(() => { pending = (async () => { const argv = [...context.renew_argv, "--expected-version", String(leaseVersion(lease)), @@ -119,7 +122,7 @@ export async function runLeasedHostProcess(request: HostProcessRequest, context: const observed = currentProof(await cli(context.read_argv)); if (leaseVersion(observed) <= leaseVersion(lease)) throw new LeaseSupervisionFailure("renewal_not_advanced"); lease = observed; - if (!finished) schedule(); + schedule(); } catch (error) { reject(error, "renewal"); } })(); }, Math.max(1, Math.min(30_000, Math.floor(remaining / 2)))); diff --git a/tests/control_plane/test_leased_host_process.py b/tests/control_plane/test_leased_host_process.py index 47d01d1d4e..0bb76809a1 100644 --- a/tests/control_plane/test_leased_host_process.py +++ b/tests/control_plane/test_leased_host_process.py @@ -63,8 +63,8 @@ def command(*arguments): @pytest.mark.skipif(os.name == "nt", reason="POSIX owned process-group qualification") -@pytest.mark.parametrize("fault", ["rejected", "hung", "lost_reply"]) -def test_real_renewal_faults_keep_original_deadline_and_identity(canonical_execution, tmp_path, fault): +@pytest.mark.parametrize("fault", ["rejected", "hung", "lost_reply", "late_reply"]) +def test_real_renewal_faults_keep_original_deadline_and_identity(canonical_execution, tmp_path, fault, record_property): context, command, selected = canonical_execution if fault == "rejected": context["renew_argv"] = [sys.executable, "-c", "import json;print(json.dumps({'ok':False}))"] @@ -72,26 +72,53 @@ def test_real_renewal_faults_keep_original_deadline_and_identity(canonical_execu context["renew_argv"] = [sys.executable, "-c", "import time;time.sleep(60)"] else: marker = tmp_path / "lost-reply" - relay = """import subprocess,sys + trace = tmp_path / "lease-commands.jsonl" + relay = """import json,subprocess,sys,time from pathlib import Path -marker=Path(sys.argv[1]) -result=subprocess.run(sys.argv[2:],capture_output=True,text=True,check=True) -if marker.exists(): print(result.stdout,end='') -else: marker.touch() +marker,trace,phase=Path(sys.argv[1]),Path(sys.argv[2]),sys.argv[3] +hold_until=float(sys.argv[4]) +def record(event,**values): + with trace.open('a') as stream: + stream.write(json.dumps({'event':event,'phase':phase,'at':time.time(),**values})+'\\n') +record('started',intent=sys.argv[5:]) +result=subprocess.run(sys.argv[5:],capture_output=True,text=True) +try: reply=json.loads(result.stdout) +except ValueError: reply={} +record('returned',returncode=result.returncode,reply=reply) +if phase=='renew' and hold_until: + time.sleep(max(0,hold_until-time.time())+2) +if phase=='read' or marker.exists(): print(result.stdout,end='') +else: marker.touch();record('reply_dropped') +sys.exit(result.returncode) """ - context["renew_argv"] = [sys.executable, "-c", relay, str(marker), *context["renew_argv"]] + held_deadline = (datetime.fromisoformat(context["lease"]["expires_at"].replace("Z", "+00:00")).timestamp() + if fault == "late_reply" else 0) + for field, phase in (("renew_argv", "renew"), ("read_argv", "read")): + context[field] = [sys.executable, "-c", relay, str(marker), str(trace), phase, + str(held_deadline), *context[field]] observed = run_host_process([sys.executable, "-c", "import time;time.sleep(35);print('finished')"], project=tmp_path, input_text="", timeout_seconds=45, delegated_lease=context) + evidence = {"original": context["lease"], "observed": observed, + "trace": trace.read_text() if fault in {"lost_reply", "late_reply"} and trace.exists() else ""} + # Keep the first observation in JUnit even if subsequent canonical readback + # fails or pytest removes its temporary directory. Only synthetic fixtures. + record_property("lease_supervision", json.dumps(evidence)) current = command("task-lease", "inspect", *selected) + evidence["current"] = current assert current["lease"]["lease_epoch"] == context["lease"]["lease_epoch"] assert current["lease"]["idempotency_key"] == context["lease"]["idempotency_key"] if fault == "lost_reply": - assert observed["outcome"] == "exited" and observed["output_complete"] is True + assert observed["outcome"] == "exited" and observed["output_complete"] is True, evidence assert current["lease"]["version"] > context["lease"]["version"] assert "lease_failure" not in observed else: assert observed["outcome"] == "cancelled" and observed["output_complete"] is False - assert current["lease"]["version"] == context["lease"]["version"] + if fault == "late_reply": + # A committed renewal is not timely execution proof. The old + # deadline still stops the Host while its reply is unavailable. + assert current["lease"]["version"] > context["lease"]["version"], evidence + else: + assert current["lease"]["version"] == context["lease"]["version"], evidence assert observed["lease_failure"] == ( {"reason": "renewal_rejected", "boundary": "renewal"} if fault == "rejected" else {"reason": "proved_deadline_elapsed", "boundary": "deadline"}) @@ -127,3 +154,51 @@ def test_released_initial_proof_explains_refusal_before_host_launch(canonical_ex assert observed["outcome"] == "cancelled" and observed["output_complete"] is False assert not marker.exists() assert observed["lease_failure"] == {"reason": "lease_inactive", "boundary": "initial_proof"} + + +@pytest.mark.skipif(os.name == "nt", reason="POSIX owned process-group qualification") +def test_returned_host_uses_renewed_deadline_during_final_proof(canonical_execution, tmp_path, record_property): + """An in-flight renewal can finish after Host exit, before final readback.""" + context, _, _ = canonical_execution + relay = """import json,os,subprocess,sys,time +from pathlib import Path +root,phase,old_expiry=Path(sys.argv[1]),sys.argv[2],float(sys.argv[3]) +if phase=='read': + count=root/'read-count' + number=int(count.read_text())+1 if count.exists() else 1 + count.write_text(str(number)) + if number==3: + (root/'final-proof-started').write_text(str(time.time())) + time.sleep(max(0,old_expiry+1-time.time())) +result=subprocess.run(sys.argv[4:],capture_output=True,text=True,check=True) +if phase=='renew': + (root/'renewed').write_text(result.stdout) + pid=int((root/'host-pid').read_text()) + while True: + try: os.kill(pid,0) + except ProcessLookupError: break + time.sleep(0.01) + # The parent has reaped the actual Host; let its pending exit callbacks + # finish before delivering the renewal response and fresh proof. + time.sleep(0.2) +print(result.stdout,end='') +""" + expiry = datetime.fromisoformat(context["lease"]["expires_at"].replace("Z", "+00:00")).timestamp() + for field, phase in (("renew_argv", "renew"), ("read_argv", "read")): + context[field] = [sys.executable, "-c", relay, str(tmp_path), phase, str(expiry), *context[field]] + host = """import os,sys,time +from pathlib import Path +root=Path(sys.argv[1]);(root/'host-pid').write_text(str(os.getpid())) +while not (root/'renewed').exists(): time.sleep(0.01) +print('finished') +""" + observed = run_host_process([sys.executable, "-c", host, str(tmp_path)], project=tmp_path, + input_text="", timeout_seconds=45, delegated_lease=context) + renewal = json.loads((tmp_path / "renewed").read_text()) + evidence = {"observed": observed, "renewal": renewal, "original": context["lease"]} + record_property("final_proof", json.dumps(evidence)) + assert float((tmp_path / "final-proof-started").read_text()) < expiry, evidence + assert renewal["lease"]["version"] > context["lease"]["version"], evidence + assert time.time() > expiry + assert observed["outcome"] == "exited" and observed["output_complete"] is True, evidence + assert "lease_failure" not in observed From 6800247e49cf87a2b13cd72cb21761c862429ef1 Mon Sep 17 00:00:00 2001 From: song Date: Sat, 3 Oct 2026 12:11:10 +0800 Subject: [PATCH 34/54] test(delegation): qualify deadlines at their actual lifecycle boundaries Signed-off-by: song --- tests/control_plane_ts/host_process.test.ts | 35 ++++++++++++++++++++- tests/test_delegation_lease_lifetime.py | 34 ++++++++++++++------ 2 files changed, 59 insertions(+), 10 deletions(-) diff --git a/tests/control_plane_ts/host_process.test.ts b/tests/control_plane_ts/host_process.test.ts index 3c90a25b40..f28ba91f93 100644 --- a/tests/control_plane_ts/host_process.test.ts +++ b/tests/control_plane_ts/host_process.test.ts @@ -63,9 +63,27 @@ for (const mode of ["timeout", "abort", "leader_exit", "closed_pipes"] as const) ${mode === "leader_exit" || mode === "closed_pipes" ? "process.exit(0)" : "setInterval(()=>{},1000)"} }},5)`; const controller = new AbortController(); + let expire: (() => void) | undefined; + let ready = false; + if (mode === "timeout") { + // Control only the supervisor deadline, not real process IO or cleanup. + // This case proves drain of a known-ready descendant. A separate real + // 500ms case below covers expiry before readiness, without assuming IO. + const timer = globalThis.setTimeout; + t.mock.method(globalThis, "setTimeout", (callback: () => void, ms: number) => { + if (ms !== 500) return timer(callback, ms); + expire = callback; + return timer(() => { controller.abort(); }, 3000); // fixture readiness watchdog + }); + } const result = await runHostProcess(request(script, {timeout_ms: mode === "timeout" ? 500 : 3000}), async item => { - if (mode === "abort" && item.text.includes("ready")) controller.abort(); + if (item.text.includes("ready")) { + ready = true; + if (mode === "abort") controller.abort(); + if (mode === "timeout") { assert.ok(expire); expire(); } + } }, controller.signal); + assert.equal(ready, true, "descendant readiness was not established"); assert.equal(result.outcome, mode === "abort" ? "cancelled" : mode === "timeout" ? "timeout" : "exited"); assert.equal(result.cleanup_scope, "process_group"); assert.equal(result.group_signal_sent, true); const counter = await readFile(marker, "utf8"); await delay(100); @@ -74,6 +92,21 @@ for (const mode of ["timeout", "abort", "leader_exit", "closed_pipes"] as const) }); } +test("real timeout before readiness prevents later Host effects", {skip: process.platform === "win32"}, async t => { + const root = await mkdtemp(join(tmpdir(), "loopx-host-pre-ready-")); + t.after(() => rm(root, {recursive: true, force: true})); + const marker = join(root, "late-effect"); + const result = await runHostProcess(request( + `setTimeout(()=>require('fs').writeFileSync(${JSON.stringify(marker)},'unexpected'),900)`, + {timeout_ms: 500}), async () => {}); + assert.equal(result.outcome, "timeout"); + assert.equal(result.cleanup_scope, "process_group"); + assert.equal(result.group_signal_sent, true); + assert.equal(existsSync(marker), false); + await delay(900); + assert.equal(existsSync(marker), false, "Host performed an effect after timeout returned"); +}); + test("output consumer failure cancels execution rather than leaving an orphan", async () => { const result = await runHostProcess(request(`setInterval(()=>process.stdout.write('tick\\n'),10)`), async () => { throw new Error("consumer left"); }); diff --git a/tests/test_delegation_lease_lifetime.py b/tests/test_delegation_lease_lifetime.py index 03d41ec566..eeadee7e0a 100644 --- a/tests/test_delegation_lease_lifetime.py +++ b/tests/test_delegation_lease_lifetime.py @@ -57,6 +57,15 @@ def inspect(runner): "--todo-id", "todo_analyst-initial") +def renew_current_lease(runner, cli, ttl): + binding = runner.binding("analysis") + current = inspect(runner)["lease"] + return cli(binding, "task-lease", "renew", "--goal-id", runner.goal_id, + "--todo-id", binding["todo_id"], "--owner", binding["agent_id"], + "--idempotency-key", current["idempotency_key"], + "--expected-version", str(current["version"]), "--ttl-seconds", str(ttl))["lease"] + + @pytest.fixture(params=["file", "sqlite"]) def completion_service(tmp_path, request, monkeypatch): """Pin a real slow final acceptance command before preparing authority.""" @@ -109,13 +118,8 @@ def test_completion_renews_before_validation_and_replays_each_intent(completion_ def enter_completion(row, binding): nonlocal shortened if shortened is None: - current = inspect(runner)["lease"] - # Fix the phase boundary, independent of whether the Host happened - # to cross its earlier renewal timer. Use the real canonical API. - shortened = cli(binding, "task-lease", "renew", "--goal-id", runner.goal_id, - "--todo-id", binding["todo_id"], "--owner", binding["agent_id"], - "--idempotency-key", current["idempotency_key"], - "--expected-version", str(current["version"]), "--ttl-seconds", "20")["lease"] + # Start the short lease at the boundary this test qualifies. + shortened = renew_current_lease(runner, cli, 20) # Cross the actual pre-renewal deadline, not an assumed amount of # CLI startup time. Allow cold claim/renew commands to reach the # boundary; the independent validator still outlives that lease. @@ -168,11 +172,22 @@ def observe_reply(binding, *args, **kwargs): @pytest.mark.parametrize("authority_loss", ["expiry", "replacement"]) def test_completion_renewal_receipt_cannot_revive_lost_execution(service, monkeypatch, authority_loss): root, runner = service - prepare_lease(root, runner, monkeypatch) - cli = runner._cli + # Setup and Host execution are not the completion-recovery deadline. + # The 20s lease begins at completion, and the later 1s expiry/new epoch + # still proves a historical renewal receipt cannot revive the execution. + prepare_lease(root, runner, monkeypatch, ttl=None) + cli, complete = runner._cli, runner._complete_delegated_todo dropped = False + shortened = False completions = [] + def enter_completion(row, binding): + nonlocal shortened + if not shortened: + renew_current_lease(runner, cli, 20) + shortened = True + return complete(row, binding) + def lose_renewal_reply(binding, *args, **kwargs): nonlocal dropped if args[:2] == ("todo", "complete"): @@ -183,6 +198,7 @@ def lose_renewal_reply(binding, *args, **kwargs): raise ValueError("fixture lost renewal response before terminal intent") return result + monkeypatch.setattr(runner, "_complete_delegated_todo", enter_completion) monkeypatch.setattr(runner, "_cli", lose_renewal_reply) runner.execute("lease-lifetime") row = _read(runner.path("lease-lifetime")) From fc00c63b272db2d1a6546d527173eeff65d5740d Mon Sep 17 00:00:00 2001 From: song Date: Sat, 3 Oct 2026 12:11:10 +0800 Subject: [PATCH 35/54] docs(delegation): clarify expiry supervision after Host return Signed-off-by: song --- docs/reference/local-delegation.md | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/docs/reference/local-delegation.md b/docs/reference/local-delegation.md index cb9be0e241..1081bb035f 100644 --- a/docs/reference/local-delegation.md +++ b/docs/reference/local-delegation.md @@ -763,12 +763,18 @@ Host outcome alone does not mean a user requested stop or the lease was released Inspect the original execution and canonical lease before recovery. These observations do not extend deadlines, grant authority, or change stop settlement. Successful and ordinary unleased Host results retain their existing shape. +After the Host returns, periodic renewal stops but final execution readback +remains bounded by the latest proved lease expiry. An in-flight renewal that +finishes with fresh canonical proof advances that deadline; a committed renewal +whose proof is unavailable does not. 受管委派错误通过 `lease:/` 保留首个类型化租约失败原因, 区分启动前证明失败、续期拒绝和已证明期限到达。仅有 Host 的 `cancelled` 结果不代表用户请求停止,也不证明租约已释放;恢复前需读回原执行与 canonical 租约。这些诊断不延长期限、不授予权限,也不改变停止结算条件;成功执行与普通 -无租约 Host 的结果结构保持不变。 +无租约 Host 的结果结构保持不变。Host 返回后不再安排周期续期,但最终执行读回 +仍受最新已证明的租约期限约束;进行中的续期取得新鲜 canonical 证明后更新该期限, +仅有续期提交而没有及时取得证明不能延长执行权限。 | Interruption | Behavior and recovery | | --- | --- | From e6bb8e44346600672fd38c5cefa575adbd8208a7 Mon Sep 17 00:00:00 2001 From: song Date: Sat, 3 Oct 2026 14:37:50 +0800 Subject: [PATCH 36/54] test(delegation): retire fixture-owned Effect runtimes Signed-off-by: song --- tests/test_local_delegation.py | 39 ++++++++++++++++++++++++++++++++-- 1 file changed, 37 insertions(+), 2 deletions(-) diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index befb13a07a..999f6fe8c7 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -63,6 +63,21 @@ @pytest.fixture(params=["file", "sqlite"]) def service(tmp_path, request, monkeypatch): isolate_sqlite_runtime(tmp_path, monkeypatch) + # Each case owns a private server. Do not accumulate five-minute idle + # runtimes across the delegation suite, including failed setup/test cases. + runtime_env = os.environ.copy() + def retire_runtime(): + subprocess.run([ + sys.executable, "-c", + "from pathlib import Path; import sys; " + "from loopx.control_plane.effect_runtime import _runtime_dir, restart_effect_runtime; " + "assert _runtime_dir().parent == Path(sys.argv[1]); " + "result = restart_effect_runtime(); " + "assert result['status'] in {'stopped', 'not_running'}, result", + str(tmp_path), + ], cwd=Path(__file__).resolve().parents[1], env=runtime_env, + capture_output=True, text=True, timeout=30, check=True) + request.addfinalizer(retire_runtime) root = tmp_path / "team" demo.prepare(root, provider=request.param) fixture(root) @@ -90,13 +105,16 @@ def test_delegation_fixture_isolates_cached_and_child_runtime_routes(tmp_path, m """A warmed parent must use the same private Effect server as its CLI.""" from types import SimpleNamespace - from loopx.control_plane.effect_runtime import _runtime_dir + from loopx.control_plane.effect_runtime import ( + _runtime_dir, _serving_runtime_identity, restart_effect_runtime, effect_runtime_result, + ) cached, isolated = tmp_path / "cached", tmp_path / "isolated" cached.mkdir() isolated.mkdir() monkeypatch.setattr(tempfile, "tempdir", str(cached)) - service.__wrapped__(isolated, SimpleNamespace(param=provider), monkeypatch) + finalizers = [] + service.__wrapped__(isolated, SimpleNamespace(param=provider, addfinalizer=finalizers.append), monkeypatch) assert _runtime_dir().parent == isolated child = subprocess.check_output([ @@ -104,6 +122,23 @@ def test_delegation_fixture_isolates_cached_and_child_runtime_routes(tmp_path, m "from loopx.control_plane.effect_runtime import _runtime_dir; print(_runtime_dir())", ], text=True).strip() assert Path(child) == _runtime_dir() + effect_runtime_result("runtime.ping", {}) + assert _serving_runtime_identity() is not None + try: + # Teardown targets the captured route, even if another test changed + # the parent cache/environment. A neighboring runtime must survive. + with monkeypatch.context() as neighbor: + isolate_sqlite_runtime(cached, neighbor) + neighbor_pid = effect_runtime_result("runtime.ping", {})["pid"] + try: + for finalize in reversed(finalizers): + finalize() + assert effect_runtime_result("runtime.ping", {})["pid"] == neighbor_pid + finally: + restart_effect_runtime() + assert _serving_runtime_identity() is None, "fixture must retire its private runtime before the next case" + finally: + restart_effect_runtime() @pytest.mark.parametrize("operation", ["--help", "x y", "x\ny", "x;echo", "x/../y"]) From cda3b07ca34dcf4aafbb435c2a04eb8f0f445710 Mon Sep 17 00:00:00 2001 From: song Date: Sun, 4 Oct 2026 07:46:13 +0800 Subject: [PATCH 37/54] test(delegation): preserve fixture cleanup across preview integration Signed-off-by: song --- tests/test_delegation_authority_recheck.py | 4 ++-- tests/test_delegation_preflight.py | 3 ++- tests/test_delegation_preview_reuse.py | 3 ++- tests/test_local_delegation.py | 3 ++- 4 files changed, 8 insertions(+), 5 deletions(-) diff --git a/tests/test_delegation_authority_recheck.py b/tests/test_delegation_authority_recheck.py index 879e5a4992..33d3716b2e 100644 --- a/tests/test_delegation_authority_recheck.py +++ b/tests/test_delegation_authority_recheck.py @@ -11,9 +11,9 @@ @pytest.fixture -def sqlite_service(tmp_path, monkeypatch): +def sqlite_service(tmp_path, request, monkeypatch): return delegation_service.__wrapped__( - tmp_path, SimpleNamespace(param="sqlite"), monkeypatch + tmp_path, SimpleNamespace(param="sqlite", addfinalizer=request.addfinalizer), monkeypatch ) diff --git a/tests/test_delegation_preflight.py b/tests/test_delegation_preflight.py index 3be137f435..06e15a9804 100644 --- a/tests/test_delegation_preflight.py +++ b/tests/test_delegation_preflight.py @@ -123,7 +123,8 @@ def test_structured_previews_isolate_concurrent_registry_and_workspace_facts( ): """Same public Goal name, different authorities: no cwd or decision leakage.""" first_root, first = service - provider_request = SimpleNamespace(param=request.node.callspec.params["service"]) + provider_request = SimpleNamespace(param=request.node.callspec.params["service"], + addfinalizer=request.addfinalizer) second_root, second = delegation_service.__wrapped__( tmp_path / "second", provider_request, monkeypatch ) diff --git a/tests/test_delegation_preview_reuse.py b/tests/test_delegation_preview_reuse.py index 6543ea67e8..98f61d7a61 100644 --- a/tests/test_delegation_preview_reuse.py +++ b/tests/test_delegation_preview_reuse.py @@ -282,7 +282,8 @@ def read(): def test_concurrent_services_preserve_registry_runtime_and_workspace_partition(service, tmp_path, request, monkeypatch): root, first = service second_root, second = delegation_service.__wrapped__( - tmp_path / "second", SimpleNamespace(param=request.node.callspec.params["delegation_service"]), monkeypatch + tmp_path / "second", SimpleNamespace(param=request.node.callspec.params["delegation_service"], + addfinalizer=request.addfinalizer), monkeypatch ) second = reusable_service(second) expected = [runner.inspect("analysis") for runner in (first, second)] diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index 999f6fe8c7..9607ade13b 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -66,8 +66,9 @@ def service(tmp_path, request, monkeypatch): # Each case owns a private server. Do not accumulate five-minute idle # runtimes across the delegation suite, including failed setup/test cases. runtime_env = os.environ.copy() + run_cleanup = subprocess.run # Tests may replace the shared subprocess module's run. def retire_runtime(): - subprocess.run([ + run_cleanup([ sys.executable, "-c", "from pathlib import Path; import sys; " "from loopx.control_plane.effect_runtime import _runtime_dir, restart_effect_runtime; " From 87fffec25dcbedd2e8c97ede93e4229e9c1e0aa5 Mon Sep 17 00:00:00 2001 From: song Date: Sun, 4 Oct 2026 07:46:13 +0800 Subject: [PATCH 38/54] fix(delegation): resolve canonical stop obligations through clear owners Signed-off-by: song --- loopx/collaboration_mcp.py | 117 ++++++------------ .../control_plane/collaboration/delegation.ts | 94 ++++++++------ .../collaboration/delegation_stop_lease.py | 107 ++++++++-------- .../turn_driver/host_process_transport.py | 72 +++++++++-- tests/control_plane/test_host_process.py | 71 ++++++++++- .../control_plane/test_leased_host_process.py | 22 +++- tests/control_plane_ts/delegation.test.ts | 113 ++++++++++------- tests/test_delegation_lease_lifetime.py | 22 ++-- tests/test_delegation_stop_recovery.py | 12 +- tests/test_local_delegation.py | 33 ++--- 10 files changed, 408 insertions(+), 255 deletions(-) diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index 6383afc7c5..eb1f071070 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -44,9 +44,8 @@ ) from .control_plane.turn_driver.host_binding import turn_host_arg_option from .control_plane.turn_driver.host_process_transport import ( - HOST_PROCESS_DRAINED, HOST_PROCESS_DRAINING, HOST_PROCESS_NOT_LAUNCHED, - HOST_PROCESS_RECORD_ENV, HOST_PROCESS_UNATTRIBUTABLE, HOST_PROCESS_UNSUPPORTED_PLATFORM, - host_process_drain, host_process_drain_supported, + HOST_PROCESS_DRAINING, HOST_PROCESS_RECORD_ENV, + execution_host_drain, require_execution_host_drain_supported, ) from .control_plane.turn_driver.lane_fence import ( TURN_LANE_ABSENT, TURN_LANE_DEAD, TURN_LANE_LIVE, TURN_LANE_RELEASED, @@ -1018,14 +1017,15 @@ def stop(self, operation_id: str, *, execute: bool) -> dict: """Stop one bounded member and return a receipt that says what was proven. ``requested`` is written beside the execution record, never into it. - When no worker holds the operation, this caller takes the lock, marks - the record stopped and releases the hard lease itself. A same-host - holder is signalled by process group and given a bounded grace to - acknowledge; another host's holder is left to find the request at its - next checkpoint or fenced write. ``settled`` and ``unknown`` come from - the typed decision over lock facts and the drain of the native Host the - Turn launched, whose TS supervisor alone terminates it; elapsed time - proves nothing. + When no worker holds the operation, this caller takes the lock and + marks the record stopped. A same-host holder is signalled by process + group and given a bounded grace to acknowledge; another host's holder + is left to find the request at its next checkpoint or fenced write. + ``settled`` and ``unknown`` come from the typed decision over lock + facts, the Host transport's drain of everything the Turn launched, + whose TS supervisor alone terminates it, and the hard lease, which is + resolved against canonical authority only once that execution is + proven gone; elapsed time proves nothing. """ require_operation_id(operation_id) @@ -1041,12 +1041,12 @@ def stop(self, operation_id: str, *, execute: bool) -> dict: if stop is None: if row["status"] in DELEGATION_TERMINAL_STATUSES: return self._stop_receipt(row, binding, None) - self._require_host_drain_support(path) + require_execution_host_drain_supported(self._host_drain(path)) stop = self._new_stop_record(row, requested_by=self.agent_id, worker=self._lock_holder_worker(path, row)) _write(self._stop_path(path), stop) elif stop["phase"] in DELEGATION_STOP_OPEN_PHASES: - self._require_host_drain_support(path) + require_execution_host_drain_supported(self._host_drain(path)) if stop["phase"] in DELEGATION_STOP_OPEN_PHASES and stop.get("ack") is None: if stop.get("worker") is None: # No worker was named when the request was written, so whoever owns @@ -1118,33 +1118,18 @@ def _signal_worker(self, path: Path, stop: dict) -> None: # The Host supervisor sits outside the worker's group and cleans up on its own. self._await_host_drain(path, time.monotonic() + DELEGATION_STOP_GRACE_SECONDS) - def _delegation_host_drain(self, path: Path) -> str: - """Read both groups of a hard-leased Turn; never signal from readback. - - The leased CLI and actual Host use separate sessions. The CLI cannot - prove its nested Host exited merely by exiting itself. Old leased - records that only named that outer group lack the required proof. - """ - host_record = self._host_process_record(path) - host = host_process_drain(host_record) - outer = host_process_drain(host_record.with_suffix(".cli.host.json")) - if HOST_PROCESS_UNSUPPORTED_PLATFORM in (outer, host): - return HOST_PROCESS_UNSUPPORTED_PLATFORM - if outer == HOST_PROCESS_NOT_LAUNCHED: - row = _read(path) - lease = row.get("task_lease") - if host != HOST_PROCESS_NOT_LAUNCHED and isinstance(lease, dict) and lease.get("required") is True: - return HOST_PROCESS_UNATTRIBUTABLE - return host - for fact in (HOST_PROCESS_UNATTRIBUTABLE, HOST_PROCESS_DRAINING): - if fact in (outer, host): - return fact - return HOST_PROCESS_DRAINED + def _host_drain(self, path: Path) -> str: + """The Host transport's read of everything this operation's Turn launched; never signals.""" + # Read only: before stop intent exists, taking the operation lock can + # refuse a legitimate launch. The Host boundary owns platform policy. + holder, _ = lock_holder_liveness(path) + return execution_host_drain(self._host_process_record(path), launch_possible=holder not in { + LOCK_HOLDER_ABSENT, LOCK_HOLDER_DEAD, LOCK_HOLDER_RELEASED, + }) def _await_host_drain(self, path: Path, deadline: float) -> None: """Wait, never kill: the TS Host supervisor owns terminating its process group.""" - while (self._delegation_host_drain(path) == HOST_PROCESS_DRAINING - and time.monotonic() < deadline): + while self._host_drain(path) == HOST_PROCESS_DRAINING and time.monotonic() < deadline: time.sleep(0.1) def _acknowledge_stop(self, path: Path, row: dict, binding: dict, *, source: str) -> None: @@ -1170,11 +1155,11 @@ def _acknowledge_stop(self, path: Path, row: dict, binding: dict, *, source: str transition = effect_runtime_result("collaboration.delegation.observe", { "from": observed, "to": "stopped", }) - # This process holds the operation lock; its lane and Host drain are read later. + # This process holds the operation lock; its lane, Host drain and lease are read later. phase = effect_runtime_result("collaboration.delegation.stop", { "phase": stop["phase"], "acknowledged": True, "operation_lock_free": False, "worker_lane_released": False, - "host_process": HOST_PROCESS_DRAINING, + "host_process": HOST_PROCESS_DRAINING, "lease": delegation_stop_lease.LEASE_UNCHECKED, }) current["status"] = transition["status"] _write(path, current) @@ -1190,30 +1175,6 @@ def _acknowledge_stop(self, path: Path, row: dict, binding: dict, *, source: str except (OSError, ValueError): pass # the bootstrap is host input; its state never blocks the receipt - def _require_host_drain_support(self, path: Path) -> None: - """Reject under the dispatch fence before cancellation writes or signals. - - Keep the same guard during settlement for an existing cancellation - intent restored on a platform that cannot prove Host drain. - """ - drain = self._delegation_host_drain(path) - unsupported = drain == HOST_PROCESS_UNSUPPORTED_PLATFORM - if not host_process_drain_supported(): - # An active worker can have passed its last checkpoint without yet - # creating the Host record. Reading absence alone races that launch. - # The dispatch fence prevents a new worker's fenced entry while we - # check the holder; do not probe/take its operation lock here. - holder, _ = lock_holder_liveness(path) - unsupported = drain != HOST_PROCESS_NOT_LAUNCHED or holder not in { - LOCK_HOLDER_ABSENT, LOCK_HOLDER_DEAD, LOCK_HOLDER_RELEASED, - } - if unsupported: - raise ValueError( - "delegation stop cannot prove the launched Host drained on this platform: " - "process groups are unavailable, so the Host supervisor is best-effort. " - "Stop the member's Host through its own supervisor and re-read the receipt." - ) - def _settle_stop(self, path: Path) -> dict: with exclusive_file_lock(self._dispatch_lock(path)): row = _read(path) @@ -1223,26 +1184,24 @@ def _settle_stop(self, path: Path) -> dict: return self._stop_receipt(row, binding, None) if stop["phase"] not in DELEGATION_STOP_OPEN_PHASES: return self._stop_receipt(row, binding, stop) - self._require_host_drain_support(path) + require_execution_host_drain_supported(self._host_drain(path)) facts = {"operation_lock_free": self._operation_lock_free(path)} facts["worker_lane_released"], lane_state = self._worker_lane_released(row, stop, binding) # Read last: a Host seen drained after its worker and lane let go stays drained. - facts["host_process"] = self._delegation_host_drain(path) + facts["host_process"] = self._host_drain(path) + facts["lease"] = delegation_stop_lease.LEASE_UNCHECKED inputs = { "phase": stop["phase"], "acknowledged": stop.get("ack") is not None, "timed_out": time.time() - stop["requested_at"] > DELEGATION_STOP_GRACE_SECONDS, - **facts, } - # Ask the existing typed owner whether ACK/lock/lane/Host facts - # permit settlement before attempting a lease release. This first - # decision is only a preflight, never a persisted receipt. In - # particular, an unavailable nested supervisor cannot hand off a - # lease while its actual Host is still running. - decision = effect_runtime_result("collaboration.delegation.stop", inputs) - if decision["phase"] == "settled": - lease_released = delegation_stop_lease.settle(self, path, row, binding, stop) - if lease_released is not None: - facts["lease_released"] = lease_released + decision = effect_runtime_result("collaboration.delegation.stop", {**inputs, **facts}) + if decision["action"] == "resolve_lease": + # The typed owner found the stopped execution gone: only now is + # its lease resolved, against canonical authority by the + # execution's own identity, and only the resolved fact can make + # the receipt terminal. Under the same lock as the record, every + # later read retries a release that has not been proven. + facts["lease"] = delegation_stop_lease.settle(self, path, row, binding, stop) decision = effect_runtime_result("collaboration.delegation.stop", {**inputs, **facts}) if decision["phase"] != stop["phase"] or decision.get("reason") != stop.get("reason"): stop.update(phase=decision["phase"], reason=decision.get("reason")) @@ -1324,9 +1283,9 @@ def _cli(self, binding: dict, *args: str, timeout: int = 60, chunks = [] nested_record_args = [] if host_record is not None: - # Only our private CLI re-arms the marker for the real Turn - # transport. Never pass it through an arbitrary user Host. - environment[HOST_PROCESS_RECORD_ENV] = str(host_record.with_suffix(".cli.host.json")) + # The transport records its leased supervisor beside this record; + # only our private CLI re-arms it for the actual Host it starts. + # Never pass it through an arbitrary user Host. nested_record_args = ["--host-process-record", str(host_record)] try: observation = run_host_process( diff --git a/loopx/control_plane/collaboration/delegation.ts b/loopx/control_plane/collaboration/delegation.ts index 4bb123a12a..257271bd7a 100644 --- a/loopx/control_plane/collaboration/delegation.ts +++ b/loopx/control_plane/collaboration/delegation.ts @@ -348,27 +348,50 @@ export function transitionDelegationObservation(params: JsonObject): JsonObject type StopPhase = "requested" | "acknowledged" | "settled" | "unknown"; const openStopPhases: readonly StopPhase[] = ["requested", "acknowledged"]; -/** What the host read back about the native Host process the operation launched. */ +/** What the Host transport read back about everything the operation's Turn launched. */ type HostProcessDrain = "not_launched" | "drained" | "draining" | "unattributable"; const hostProcessDrains: readonly HostProcessDrain[] = ["not_launched", "drained", "draining", "unattributable"]; +/** What canonical authority proved about the hard lease the stopped execution may hold. + * + * `unchecked` until the stop resolves it, which happens only once the execution + * is proven gone; it never means that no lease was owed. `not_owed` is proven + * by canonical authority for the execution's own identity, not by its record. + */ +type StopLease = "unchecked" | "not_owed" | "released" | "release_unproven" | "obligation_unproven"; +const stopLeases: readonly StopLease[] = ["unchecked", "not_owed", "released", "release_unproven", "obligation_unproven"]; +/** The next step for the host: resolve the lease from canonical authority, or record a receipt phase. */ +type DelegationStopStep = + | {action: "resolve_lease"} + | {action: "record"; phase: StopPhase; terminal: boolean; reason: string}; + +function stopRecord(phase: StopPhase, reason: string): DelegationStopStep { + return {action: "record", phase, terminal: !openStopPhases.includes(phase), reason}; +} /** Advance one stop request from host release facts; a receipt is never inferred from time. * - * ``settled`` needs the acknowledgement of a process that held the operation - * lock, that lock free again, the member's Turn lane released by the stopped - * worker's process group, and the native Host the operation launched drained - * together with its process group. A worker and its lane can let go while the - * Host supervisor is still terminating the Host, so their release proves - * nothing about the Host. The host reads the lane from its holder record and - * never takes it, so a legitimate Turn is not refused, and a holder it cannot - * attribute is not released. A Host drain that cannot be attributed keeps an - * acknowledged stop open so a later read with the same identity can still - * settle it. Everything released without an acknowledgement means the named - * holder vanished before recording what it observed, which is ``unknown`` - * rather than a fake settlement. A grace timeout on its own moves nothing: a - * worker still holding a lock still runs. + * The stopped execution is gone once the operation lock is free, the member's + * Turn lane was released by the stopped worker's process group, and the Host + * transport reads everything the Turn launched as drained or never launched. + * A worker and its lane can let go while the Host supervisor is still + * terminating the Host, so their release proves nothing about the Host. The + * host reads the lane from its holder record and never takes it, so a + * legitimate Turn is not refused, and a holder it cannot attribute is not + * released. + * + * Only then may the host resolve the execution's hard lease: its release hands + * the member's Todo on, so it waits for the execution, and a receipt never + * becomes terminal before that lease is resolved. ``settled`` needs the + * acknowledgement of a process that held the operation lock, the execution + * gone and the lease released or proven not owed. Everything released without + * an acknowledgement means the named holder vanished before recording what it + * observed, which is ``unknown`` rather than a fake settlement; with a Host + * that cannot be attributed nothing more can be learned, so it is ``unknown`` + * without touching the lease. An acknowledged stop whose Host cannot be + * attributed stays open for a later read with the same identity. A grace + * timeout on its own moves nothing: a worker still holding a lock still runs. */ -export function decideDelegationStop(params: JsonObject): JsonObject { +export function decideDelegationStop(params: JsonObject): DelegationStopStep { const phase = params.phase as StopPhase; requireThat(openStopPhases.includes(phase), "delegation stop decision requires an open stop phase"); requireThat(typeof params.acknowledged === "boolean", "delegation stop acknowledgement fact required"); @@ -376,38 +399,37 @@ export function decideDelegationStop(params: JsonObject): JsonObject { "delegation stop release facts required"); requireThat(hostProcessDrains.includes(params.host_process as HostProcessDrain), "delegation stop host process drain fact required"); - // A required hard lease is released by the stop itself. Its release is part - // of what the receipt promises: a member whose lease is still held can block - // its Todo until the lease TTL, which is not a safe stop and is not something - // the owner can act on. `undefined` means the operation held no required - // lease. - requireThat(params.lease_released === undefined || typeof params.lease_released === "boolean", - "delegation stop lease release fact must be boolean"); + requireThat(stopLeases.includes(params.lease as StopLease), "delegation stop lease fact required"); requireThat(params.timed_out === undefined || typeof params.timed_out === "boolean", "delegation stop timeout fact must be boolean"); requireThat(phase !== "acknowledged" || params.acknowledged === true, "an acknowledged stop cannot lose its acknowledgement"); const operationFree = params.operation_lock_free === true; - const workerReleased = operationFree && params.worker_lane_released === true; const host = params.host_process as HostProcessDrain; - const hostDrained = host === "drained" || host === "not_launched"; - const leaseReleased = params.lease_released !== false; - const pending = !operationFree ? "operation_lock_still_held" + const lease = params.lease as StopLease; + const executionPending = !operationFree ? "operation_lock_still_held" : params.worker_lane_released !== true ? "worker_lane_release_unproven" : host === "draining" ? "host_process_still_running" - : host !== "drained" && host !== "not_launched" ? "host_process_drain_unproven" - : "required_lease_release_unproven"; + : host === "unattributable" ? "host_process_drain_unproven" + : null; + requireThat(executionPending === null || lease === "unchecked", + "a delegation stop resolves its lease only after its execution is proven gone"); + const leaseStep = (open: StopPhase, terminal: StopPhase, reason: string): DelegationStopStep => + lease === "unchecked" ? {action: "resolve_lease"} + : lease === "release_unproven" ? stopRecord(open, "required_lease_release_unproven") + : lease === "obligation_unproven" ? stopRecord(open, "lease_obligation_unproven") + : stopRecord(terminal, reason); if (params.acknowledged === true) { - if (workerReleased && hostDrained && leaseReleased) { - return {phase: "settled", terminal: true, reason: "acknowledged_worker_and_host_released"}; - } - return {phase: "acknowledged", terminal: false, reason: pending}; + return executionPending === null + ? leaseStep("acknowledged", "settled", "acknowledged_worker_and_host_released") + : stopRecord("acknowledged", executionPending); } - if (workerReleased && host !== "draining" && leaseReleased) { - return {phase: "unknown", terminal: true, reason: "holder_gone_without_acknowledgement"}; + if (executionPending === null) return leaseStep("requested", "unknown", "holder_gone_without_acknowledgement"); + if (executionPending === "host_process_drain_unproven") { + return stopRecord("unknown", "holder_gone_without_acknowledgement"); } - return {phase: "requested", terminal: false, reason: operationFree ? pending - : params.timed_out === true ? "holder_still_running_after_grace" : "awaiting_acknowledgement"}; + return stopRecord("requested", operationFree ? executionPending + : params.timed_out === true ? "holder_still_running_after_grace" : "awaiting_acknowledgement"); } /** Whether an accepted result may produce a wake intent at all. diff --git a/loopx/control_plane/collaboration/delegation_stop_lease.py b/loopx/control_plane/collaboration/delegation_stop_lease.py index 80c5a1b1f5..d87e404969 100644 --- a/loopx/control_plane/collaboration/delegation_stop_lease.py +++ b/loopx/control_plane/collaboration/delegation_stop_lease.py @@ -1,47 +1,47 @@ -"""Release the hard lease a stopped delegation acquired, by its own execution identity. +"""Resolve the hard lease a stopped delegation's execution may hold, by its own identity. A stop owes the release of the lease its execution acquired; it never owns a -lease another execution holds. The identity comes from the operation record -and the version from the canonical lease, so a renewed or unannotated lease is -still released exactly, and the native lifecycle remains the only writer. +lease another execution holds. Canonical authority decides whether that lease +is still held: the identity comes from the operation record, the version from +the canonical lease, and the native lifecycle remains the only writer. The +operation's own `task_lease` annotation is a hint, never proof that nothing is +owed. The result is one of the typed stop owner's lease facts. """ from __future__ import annotations from .inbox import _write from ..coordination.local_authority import local_authority_is_promoted from ..effect_runtime import EffectRuntimeRemoteError -from ..todos.handoff_mode import show_goal_handoff_mode from ..work_items.task_lease import inspect_task_lease, release_task_lease _AUTHORITY_ERRORS = (ValueError, OSError, RuntimeError, EffectRuntimeRemoteError) +# The typed stop owner's lease facts (`StopLease` in delegation.ts). +LEASE_UNCHECKED = "unchecked" +LEASE_NOT_OWED = "not_owed" +LEASE_RELEASED = "released" +LEASE_RELEASE_UNPROVEN = "release_unproven" +LEASE_OBLIGATION_UNPROVEN = "obligation_unproven" def obligation(service, row): - """The hard lease this operation may hold, named by its own execution identity. + """The execution identity canonical authority must be asked about, or None. - `_acquire_delegation_lease` only accepts a lease whose key is - `_turn_instance_id(row)`, and that key survives in the operation record, so - the identity never depends on the annotation's shape or on the in-memory - row that acquired it. A native claim commits before the annotation is - saved, so a missing annotation is an absent fact, not proof that nothing is - owed: under a promoted hard-lease authority the obligation stands until the - canonical lease shows this execution no longer holds it. The recorded - epoch, when there is one, fences the release against another generation - under the same key. + `None` only when canonical authority proves that no delegation lease can + exist: the Goal's local authority is not promoted, which is also when + `_acquire_delegation_lease` acquires none. Otherwise the canonical lease + decides, whatever the operation recorded: a native claim commits before + its annotation is saved, and an empty, malformed or stale `required: false` + annotation is no more proof than a missing one. A recorded epoch still + fences the release against another generation under the same key. """ - recorded = row.get("task_lease") - if isinstance(recorded, dict) and recorded.get("required") is not True: + recorded = row.get("task_lease") if isinstance(row.get("task_lease"), dict) else {} + required = recorded.get("required") is True + if not local_authority_is_promoted(runtime_root=service.root, goal_id=service.goal_id): + if required: + raise ValueError("canonical authority disappeared under a required delegation lease") return None - if not (isinstance(recorded, dict) and recorded.get("required") is True): - if not local_authority_is_promoted(runtime_root=service.root, goal_id=service.goal_id): - return None - if show_goal_handoff_mode( - registry_path=service.registry, runtime_root_arg=str(service.root), goal_id=service.goal_id, - )["handoff_mode"] != "hard_lease": - return None - recorded = {} - acquired = recorded.get("lease") if isinstance(recorded.get("lease"), dict) else {} + acquired = recorded.get("lease") if required and isinstance(recorded.get("lease"), dict) else {} return {"idempotency_key": service._turn_instance_id(row), "lease_epoch": acquired.get("lease_epoch")} @@ -49,20 +49,22 @@ def obligation(service, row): def release(service, row, binding): """Release the lease only while this execution still holds it, at its current version. + Called only once the typed stop owner found the stopped execution gone. Renewal advances the version, so the acquisition's version is no CAS for a later release. The canonical lease is read first: the exact owner, key and (when recorded) epoch must match before its current version is released. - A lease another execution holds, or none at all, is reported `held: false`: - it is not this stop's to release and blocks nothing on its behalf. + A lease another execution holds, or none at all, is `not_owed`: it is not + this stop's to release and blocks nothing on its behalf. An authority that + cannot be read leaves the obligation unproven rather than assumed absent. """ try: owed = obligation(service, row) except _AUTHORITY_ERRORS as exc: - return {"required": True, "released": False, + return {"state": LEASE_OBLIGATION_UNPROVEN, "error": ("lease obligation unreadable: " + str(exc))[:180]} if owed is None: - return {"required": False, "released": None} + return {"state": LEASE_NOT_OWED} key = owed["idempotency_key"] try: inspection = inspect_task_lease( @@ -72,18 +74,16 @@ def release(service, row, binding): if inspection.get("ok") is not True: raise RuntimeError(str(inspection.get("error") or "lease inspection unavailable")) except _AUTHORITY_ERRORS as exc: - return {"required": True, "released": False, "idempotency_key": key, + return {"state": LEASE_OBLIGATION_UNPROVEN, "idempotency_key": key, "error": ("lease obligation unreadable: " + str(exc))[:180]} held = inspection.get("lease") if (not isinstance(held, dict) or held.get("owner") != binding["agent_id"] or held.get("idempotency_key") != key - or (owed.get("lease_epoch") is not None - and held.get("lease_epoch") != owed["lease_epoch"])): - return {"required": True, "released": None, "held": False, "idempotency_key": key} + or (owed["lease_epoch"] is not None and held.get("lease_epoch") != owed["lease_epoch"])): + return {"state": LEASE_NOT_OWED, "idempotency_key": key} if held.get("status") == "released": - return {"required": True, "released": True, "idempotency_key": key, - "version": held.get("version")} + return {"state": LEASE_RELEASED, "idempotency_key": key, "version": held.get("version")} try: result = release_task_lease( runtime_root=service.root, goal_id=service.goal_id, todo_id=binding["todo_id"], @@ -91,30 +91,27 @@ def release(service, row, binding): expected_version=held.get("version"), registry_path=service.registry, ) except _AUTHORITY_ERRORS as exc: - return {"required": True, "released": False, "idempotency_key": key, - "error": str(exc)[:180]} - return {"required": True, "released": result.get("released") is True, - "idempotency_key": key, "version": held.get("version"), - "missing": result.get("missing") is True} + return {"state": LEASE_RELEASE_UNPROVEN, "idempotency_key": key, + "version": held.get("version"), "error": str(exc)[:180]} + # A committed `missing` means the Todo has no lease at all, so none is held for us. + state = (LEASE_RELEASED if result.get("released") is True + else LEASE_NOT_OWED if result.get("missing") is True else LEASE_RELEASE_UNPROVEN) + return {"state": state, "idempotency_key": key, "version": held.get("version")} def settle(service, path, row, binding, stop): - """Try the required release once and record what it proved on the stop receipt. + """Resolve the lease once and record what it proved on the stop receipt. - Returns the `lease_released` fact, or `None` when the operation owed no - required lease at all, or this execution no longer holds the lease it - acquired. `None` keeps the typed planner's `undefined` meaning instead of - claiming a release that was never owed. A release already proven on the - receipt is not attempted again. Callers hold the operation's dispatch lock. + Returns the lease fact for the typed stop owner. A release already proven + on the receipt is not attempted again. Callers hold the operation's + dispatch lock. """ recorded = stop.get("lease") if isinstance(stop.get("lease"), dict) else {} - if recorded.get("released") is True: - return True - released = release(service, row, binding) - if released != stop.get("lease"): - stop["lease"] = released + if recorded.get("state") == LEASE_RELEASED: + return LEASE_RELEASED + resolved = release(service, row, binding) + if resolved != stop.get("lease"): + stop["lease"] = resolved _write(service._stop_path(path), stop) - if released.get("required") is not True: - return None # preserve the no-obligation receipt, but add no release fact - return None if released.get("held") is False else released.get("released") is True + return resolved["state"] diff --git a/loopx/control_plane/turn_driver/host_process_transport.py b/loopx/control_plane/turn_driver/host_process_transport.py index 121865595b..d6126c099a 100644 --- a/loopx/control_plane/turn_driver/host_process_transport.py +++ b/loopx/control_plane/turn_driver/host_process_transport.py @@ -30,6 +30,14 @@ HOST_PROCESS_DRAINING = "draining" HOST_PROCESS_UNATTRIBUTABLE = "unattributable" HOST_PROCESS_UNSUPPORTED_PLATFORM = "unsupported_platform" +# Each record says which group its supervisor owns: the actual Host, or a leased +# CLI that supervises a nested Host recorded under the owner's record path. +HOST_PROCESS_SUPERVISES_HOST = "host" +HOST_PROCESS_SUPERVISES_NESTED_HOST = "nested_host" +# When one execution's records disagree, the first fact present wins: a group +# still seen running outranks a missing proof, which outranks a proven exit. +_DRAIN_PRECEDENCE = (HOST_PROCESS_UNSUPPORTED_PLATFORM, HOST_PROCESS_DRAINING, + HOST_PROCESS_UNATTRIBUTABLE, HOST_PROCESS_DRAINED, HOST_PROCESS_NOT_LAUNCHED) def host_process_drain_supported() -> bool: @@ -65,14 +73,20 @@ def _process_group_present(pgid: int) -> bool: return True -def host_process_drain(record_path: Path) -> str: - """Read whether the Host a record names, and its process group, have exited. +def host_process_supervisor_record(record_path: Path) -> Path: + """Where a leased CLI's own supervisor records its group, beside the owner's record.""" + return record_path.with_suffix(".cli.host.json") + + +def host_process_drain(record_path: Path, *, supervises: str = HOST_PROCESS_SUPERVISES_HOST) -> str: + """Read whether the group one record names, and its supervisor, have exited. Read-only: nothing is signalled, and the TS-owned supervisor keeps cleanup. - No record means no Host was launched under it. ``draining`` while the - supervising bridge or the Host's group still has a member. A record from - another machine, one without a reported group whose supervisor is gone, - or a platform without process groups proves nothing: ``unattributable``. + No record means nothing was launched under it. ``draining`` while the + supervising bridge or the owned group still has a member. A record from + another machine, one that does not say it supervises ``supervises``, or + one without a reported group whose supervisor is gone proves nothing: + ``unattributable``. A platform without process groups never proves a drain. """ try: @@ -82,7 +96,7 @@ def host_process_drain(record_path: Path) -> str: except (OSError, ValueError): return HOST_PROCESS_UNATTRIBUTABLE if (not isinstance(record, dict) or record.get("schema_version") != HOST_PROCESS_RECORD_SCHEMA_VERSION - or record.get("host") != lock_holder_host_label()): + or record.get("host") != lock_holder_host_label() or record.get("supervises") != supervises): return HOST_PROCESS_UNATTRIBUTABLE if not hasattr(os, "killpg"): # A launched Host on a platform without process groups is never proven @@ -102,6 +116,40 @@ def host_process_drain(record_path: Path) -> str: return HOST_PROCESS_DRAINING if _process_group_present(group) else HOST_PROCESS_DRAINED +def execution_host_drain(record_path: Path, *, launch_possible: bool = False) -> str: + """Whether everything an owner launched under one record has exited. + + The owner names one record. A plain run's supervisor writes it for the + actual Host. A leased run's supervisor owns the private leased CLI, which + exits on its own clock and supervises the actual Host in another session: + that supervisor records its group beside the owner's record, and the Host + the CLI starts writes the owner's record. Both must have exited; an outer + exit proves nothing about the nested Host. A record that does not say what + it supervises, such as one written before records carried that fact, + attributes nothing. The caller supplies whether its launch owner may still + start a Host; on unsupported platforms an absent record cannot close that + pre-record launch window. + """ + + supervisor = host_process_drain(host_process_supervisor_record(record_path), + supervises=HOST_PROCESS_SUPERVISES_NESTED_HOST) + host = host_process_drain(record_path) + drain = next(state for state in _DRAIN_PRECEDENCE if state in {supervisor, host}) + if not host_process_drain_supported() and (launch_possible or drain != HOST_PROCESS_NOT_LAUNCHED): + return HOST_PROCESS_UNSUPPORTED_PLATFORM + return drain + + +def require_execution_host_drain_supported(drain: str) -> None: + """Reject cancellation when the Host boundary cannot prove execution drain.""" + if drain == HOST_PROCESS_UNSUPPORTED_PLATFORM: + raise ValueError( + "cannot prove the launched Host drained on this platform: " + "process groups are unavailable, so the Host supervisor is best-effort. " + "Stop the Host through its own supervisor and re-read its execution state." + ) + + class HostOutputLines: """Frame LF records without retaining raw trajectories or an unbounded line.""" @@ -174,6 +222,13 @@ def run_host_process( bridge_environment = os.environ.copy() if environment is None else dict(environment) record_value = bridge_environment.pop(HOST_PROCESS_RECORD_ENV, "") record_path = Path(record_value) if record_value else None + supervises = HOST_PROCESS_SUPERVISES_HOST + if record_path is not None and delegated_lease is not None: + # A leased run starts the private leased CLI, which records the actual + # Host it starts under the owner's record. This supervisor records its + # own group beside that record, so readback can prove both exited. + record_path = host_process_supervisor_record(record_path) + supervises = HOST_PROCESS_SUPERVISES_NESTED_HOST with subprocess.Popen( [ _node_executable(), @@ -199,7 +254,8 @@ def run_host_process( # does not name; a record that cannot be written launches nothing. record = {"schema_version": HOST_PROCESS_RECORD_SCHEMA_VERSION, "host": lock_holder_host_label(), "owner_pid": os.getpid(), - "bridge_pid": proc.pid, "phase": "launching", "process_group": None} + "supervises": supervises, "bridge_pid": proc.pid, + "phase": "launching", "process_group": None} _write_host_process_record(record_path, record) proc.stdin.write( json.dumps(request, ensure_ascii=False, separators=(",", ":")) + "\n" diff --git a/tests/control_plane/test_host_process.py b/tests/control_plane/test_host_process.py index eddbd1a134..91d0bf8c93 100644 --- a/tests/control_plane/test_host_process.py +++ b/tests/control_plane/test_host_process.py @@ -244,7 +244,8 @@ def test_host_process_drain_reads_live_groups_and_refuses_unattributable_records def drain(**fields): path.write_text(json.dumps({"schema_version": HOST_PROCESS_RECORD_SCHEMA_VERSION, - "host": lock_holder_host_label(), "owner_pid": 1, **fields})) + "host": lock_holder_host_label(), "owner_pid": 1, + "supervises": "host", **fields})) return host_process_drain(path) try: @@ -262,7 +263,12 @@ def drain(**fields): {"phase": "spawned", "bridge_pid": gone.pid, "process_group": gone.pid, "host": "another-machine"}, {"phase": "spawned", "bridge_pid": gone.pid, "process_group": gone.pid, - "schema_version": "other"}): + "schema_version": "other"}, + # A record must say which group it supervises. + {"phase": "spawned", "bridge_pid": gone.pid, "process_group": gone.pid, + "supervises": None}, + {"phase": "spawned", "bridge_pid": gone.pid, "process_group": gone.pid, + "supervises": "nested_host"}): assert drain(**fields) == "unattributable", fields path.write_text("{not json") assert host_process_drain(path) == "unattributable" @@ -338,6 +344,65 @@ def transport(*args, **kwargs): host_record=record, delegated_lease={"lease": {}, "ttl_seconds": None, "renew_argv": [], "read_argv": []}) assert handed["environment"]["PYTHONPATH"].split(os.pathsep)[0] == str(delegation_module._release_root()) - assert handed["environment"][HOST_PROCESS_RECORD_ENV] == str(record.with_suffix(".cli.host.json")) + # One record names the execution; the transport places its leased supervisor's record. + assert handed["environment"][HOST_PROCESS_RECORD_ENV] == str(record) index = handed["argv"].index("--host-process-record") assert handed["argv"][index + 1] == str(record) + + +@pytest.mark.skipif(os.name == "nt", reason="POSIX process-group drain readback") +def test_execution_drain_needs_every_group_a_leased_run_launched(tmp_path: Path) -> None: + """The Host transport alone reads one execution's records, whatever its topology. + + A leased run's supervisor records the private CLI beside the owner's record + and the actual Host writes that record, so an outer exit proves nothing + about the nested Host. A record that does not say what it supervises, or + sits where the other kind belongs, attributes nothing. + """ + from loopx.control_plane.turn_driver.host_process_transport import ( + HOST_PROCESS_RECORD_SCHEMA_VERSION, execution_host_drain, host_process_supervisor_record, + ) + from loopx.file_lock import lock_holder_host_label + + record = tmp_path / "op.host.json" + supervisor = host_process_supervisor_record(record) + live = subprocess.Popen([sys.executable, "-c", "import time;time.sleep(60)"], start_new_session=True) + gone = subprocess.Popen([sys.executable, "-c", "pass"], start_new_session=True) + gone.wait(timeout=10) + + def written(path, supervises, group): + path.write_text(json.dumps({"schema_version": HOST_PROCESS_RECORD_SCHEMA_VERSION, + "host": lock_holder_host_label(), "owner_pid": 1, "supervises": supervises, + "phase": "finished", "bridge_pid": gone.pid, "process_group": group})) + + cases = [ + # (supervisor record, owner's record, observation) + (None, None, "not_launched"), + (None, ("host", gone.pid), "drained"), + (None, ("host", live.pid), "draining"), + (("nested_host", gone.pid), None, "drained"), + (("nested_host", gone.pid), ("host", gone.pid), "drained"), + # The leased CLI exited while its nested Host still runs. + (("nested_host", gone.pid), ("host", live.pid), "draining"), + (("nested_host", live.pid), ("host", gone.pid), "draining"), + # A group still seen running outranks a missing proof. + (("nested_host", live.pid), "corrupt", "draining"), + (("nested_host", gone.pid), "corrupt", "unattributable"), + # Records that do not say what they supervise, or say the wrong thing. + (None, (None, gone.pid), "unattributable"), + (None, ("nested_host", gone.pid), "unattributable"), + (("host", gone.pid), ("host", gone.pid), "unattributable"), + ((None, gone.pid), None, "unattributable"), + ] + try: + for outer, owned, expected in cases: + for path, fact in ((supervisor, outer), (record, owned)): + path.unlink(missing_ok=True) + if fact == "corrupt": + path.write_text("{corrupt") + elif fact is not None: + written(path, *fact) + assert execution_host_drain(record) == expected, (outer, owned) + finally: + live.kill() + live.wait(timeout=10) diff --git a/tests/control_plane/test_leased_host_process.py b/tests/control_plane/test_leased_host_process.py index 0bb76809a1..f2c667b25a 100644 --- a/tests/control_plane/test_leased_host_process.py +++ b/tests/control_plane/test_leased_host_process.py @@ -17,7 +17,9 @@ ) from loopx.control_plane.coordination.local_authority_shadow_projection import canonical_bytes from loopx.control_plane.coordination.runtime_shadow import build_todo_runtime_shadow_projection -from loopx.control_plane.turn_driver.host_process_transport import run_host_process +from loopx.control_plane.turn_driver.host_process_transport import ( + HOST_PROCESS_RECORD_ENV, execution_host_drain, host_process_supervisor_record, run_host_process, +) from tests.control_plane.host_process_fixture import COUNTER_PROCESS_SOURCE @@ -142,6 +144,24 @@ def lost_reader(_text): assert marker.read_bytes() == before, "control pipe loss must finish forced cleanup before returning" +@pytest.mark.skipif(os.name == "nt", reason="POSIX owned process-group qualification") +def test_leased_supervisor_records_its_own_group_beside_the_owners_record(canonical_execution, tmp_path, monkeypatch): + """The owner names one record; a leased supervisor leaves it to the nested Host.""" + context, _, _ = canonical_execution + record = tmp_path / "op.host.json" + monkeypatch.setenv(HOST_PROCESS_RECORD_ENV, str(record)) + chunks = [] + observed = run_host_process([sys.executable, "-c", "import os,sys;print(os.environ.get(sys.argv[1]))", + HOST_PROCESS_RECORD_ENV], project=tmp_path, input_text="", + timeout_seconds=30, delegated_lease=context, on_stdout=chunks.append) + assert observed["outcome"] == "exited", observed + assert "".join(chunks).strip() == "None" # the leased child never inherits the marker + assert not record.exists() + supervisor = json.loads(host_process_supervisor_record(record).read_text()) + assert (supervisor["supervises"], supervisor["phase"]) == ("nested_host", "finished") + assert execution_host_drain(record) == "drained" + + @pytest.mark.skipif(os.name == "nt", reason="POSIX owned process-group qualification") def test_released_initial_proof_explains_refusal_before_host_launch(canonical_execution, tmp_path): context, command, selected = canonical_execution diff --git a/tests/control_plane_ts/delegation.test.ts b/tests/control_plane_ts/delegation.test.ts index dd501cd6bd..50441b90ab 100644 --- a/tests/control_plane_ts/delegation.test.ts +++ b/tests/control_plane_ts/delegation.test.ts @@ -181,31 +181,38 @@ test("stopped is terminal and reachable only from open observations", () => { observation}).status, "stopped"); }); +const stopRecord = (phase: string, reason: string) => + ({action: "record", phase, terminal: phase === "settled" || phase === "unknown", reason}); + test("a stop settles only on an acknowledgement plus released holders; time alone proves nothing", () => { const open = {phase: "requested", acknowledged: false, operation_lock_free: false, worker_lane_released: false, - host_process: "drained"}; - assert.deepEqual(decideDelegationStop(open), {phase: "requested", terminal: false, reason: "awaiting_acknowledgement"}); + host_process: "drained", lease: "unchecked"}; + assert.deepEqual(decideDelegationStop(open), stopRecord("requested", "awaiting_acknowledgement")); assert.deepEqual(decideDelegationStop({...open, timed_out: true}), - {phase: "requested", terminal: false, reason: "holder_still_running_after_grace"}); + stopRecord("requested", "holder_still_running_after_grace")); // A lane release without a free operation lock is not a vanished holder. assert.deepEqual(decideDelegationStop({...open, worker_lane_released: true, timed_out: true}), - {phase: "requested", terminal: false, reason: "holder_still_running_after_grace"}); + stopRecord("requested", "holder_still_running_after_grace")); // A free operation lock with an unattributed lane holder proves nothing yet. assert.deepEqual(decideDelegationStop({...open, operation_lock_free: true, timed_out: true}), - {phase: "requested", terminal: false, reason: "worker_lane_release_unproven"}); - assert.deepEqual(decideDelegationStop({...open, operation_lock_free: true, worker_lane_released: true}), - {phase: "unknown", terminal: true, reason: "holder_gone_without_acknowledgement"}); + stopRecord("requested", "worker_lane_release_unproven")); + const gone = {...open, operation_lock_free: true, worker_lane_released: true}; + assert.deepEqual(decideDelegationStop(gone), {action: "resolve_lease"}); + assert.deepEqual(decideDelegationStop({...gone, lease: "not_owed"}), + stopRecord("unknown", "holder_gone_without_acknowledgement")); const acked = {phase: "acknowledged", acknowledged: true, operation_lock_free: false, worker_lane_released: false, - host_process: "drained"}; - assert.deepEqual(decideDelegationStop(acked), {phase: "acknowledged", terminal: false, reason: "operation_lock_still_held"}); + host_process: "drained", lease: "unchecked"}; + assert.deepEqual(decideDelegationStop(acked), stopRecord("acknowledged", "operation_lock_still_held")); assert.deepEqual(decideDelegationStop({...acked, worker_lane_released: true}), - {phase: "acknowledged", terminal: false, reason: "operation_lock_still_held"}); + stopRecord("acknowledged", "operation_lock_still_held")); assert.deepEqual(decideDelegationStop({...acked, operation_lock_free: true}), - {phase: "acknowledged", terminal: false, reason: "worker_lane_release_unproven"}); - assert.deepEqual(decideDelegationStop({...acked, phase: "requested", operation_lock_free: true, worker_lane_released: true}), - {phase: "settled", terminal: true, reason: "acknowledged_worker_and_host_released"}); - assert.deepEqual(decideDelegationStop({...acked, operation_lock_free: true, worker_lane_released: true, timed_out: true}), - {phase: "settled", terminal: true, reason: "acknowledged_worker_and_host_released"}); + stopRecord("acknowledged", "worker_lane_release_unproven")); + const released = {...acked, operation_lock_free: true, worker_lane_released: true}; + for (const patch of [{phase: "requested"}, {timed_out: true}]) { + assert.deepEqual(decideDelegationStop({...released, ...patch}), {action: "resolve_lease"}); + assert.deepEqual(decideDelegationStop({...released, ...patch, lease: "not_owed"}), + stopRecord("settled", "acknowledged_worker_and_host_released")); + } for (const patch of [{phase: "settled"}, {phase: "unknown"}, {phase: "noop"}, {acknowledged: "yes"}, {host_process: undefined}, {host_process: "exited"}, {host_process: true}, {operation_lock_free: 1}, {worker_lane_released: undefined}, {lane_lock_free: true, worker_lane_released: undefined}, @@ -214,43 +221,61 @@ test("a stop settles only on an acknowledgement plus released holders; time alon }); test("a released worker and lane never settle a stop while the native Host still drains", () => { - const released = {phase: "acknowledged", acknowledged: true, operation_lock_free: true, worker_lane_released: true}; + const released = {phase: "acknowledged", acknowledged: true, operation_lock_free: true, worker_lane_released: true, + lease: "unchecked"}; assert.deepEqual(decideDelegationStop({...released, host_process: "draining", timed_out: true}), - {phase: "acknowledged", terminal: false, reason: "host_process_still_running"}); + stopRecord("acknowledged", "host_process_still_running")); // Without an attributable drain the stop stays open for a later same-identity read. assert.deepEqual(decideDelegationStop({...released, host_process: "unattributable"}), - {phase: "acknowledged", terminal: false, reason: "host_process_drain_unproven"}); + stopRecord("acknowledged", "host_process_drain_unproven")); for (const host_process of ["drained", "not_launched"]) - assert.deepEqual(decideDelegationStop({...released, host_process}), - {phase: "settled", terminal: true, reason: "acknowledged_worker_and_host_released"}); - // A required lease the stop could not release is not a settlement: the - // member's Todo can stay blocked by it until the lease TTL. - for (const host_process of ["drained", "not_launched"]) { - assert.deepEqual(decideDelegationStop({...released, host_process, lease_released: false}), - {phase: "acknowledged", terminal: false, reason: "required_lease_release_unproven"}); - // An operation that held no required lease omits the fact, which settles. - assert.deepEqual(decideDelegationStop({...released, host_process}), - {phase: "settled", terminal: true, reason: "acknowledged_worker_and_host_released"}); - } - assert.throws(() => decideDelegationStop({...released, host_process: "drained", lease_released: "yes"}), - /lease release fact/); - // A vanished holder whose required lease is still held is not terminal either. - const unacked = {phase: "requested", acknowledged: false, operation_lock_free: true, - worker_lane_released: true, host_process: "drained"}; - assert.deepEqual(decideDelegationStop({...unacked, lease_released: false}), - {phase: "requested", terminal: false, reason: "required_lease_release_unproven"}); - assert.deepEqual(decideDelegationStop(unacked), - {phase: "unknown", terminal: true, reason: "holder_gone_without_acknowledgement"}); + assert.deepEqual(decideDelegationStop({...released, host_process, lease: "released"}), + stopRecord("settled", "acknowledged_worker_and_host_released")); // A held lock still dominates a drained Host. assert.deepEqual(decideDelegationStop({...released, operation_lock_free: false, host_process: "drained"}), - {phase: "acknowledged", terminal: false, reason: "operation_lock_still_held"}); - // A vanished holder is unknown only once its Host is no longer seen running. + stopRecord("acknowledged", "operation_lock_still_held")); + // A vanished holder is unknown only once its Host is no longer seen running; + // with a Host that cannot be attributed nothing more can be learned, and its + // lease is left unresolved rather than handed on beside a possible Host. const vanished = {...released, phase: "requested", acknowledged: false}; assert.deepEqual(decideDelegationStop({...vanished, host_process: "draining"}), - {phase: "requested", terminal: false, reason: "host_process_still_running"}); - for (const host_process of ["drained", "not_launched", "unattributable"]) - assert.deepEqual(decideDelegationStop({...vanished, host_process}), - {phase: "unknown", terminal: true, reason: "holder_gone_without_acknowledgement"}); + stopRecord("requested", "host_process_still_running")); + assert.deepEqual(decideDelegationStop({...vanished, host_process: "unattributable"}), + stopRecord("unknown", "holder_gone_without_acknowledgement")); + for (const host_process of ["drained", "not_launched"]) + assert.deepEqual(decideDelegationStop({...vanished, host_process, lease: "not_owed"}), + stopRecord("unknown", "holder_gone_without_acknowledgement")); +}); + +test("the lease is resolved only once the execution is gone, and only a resolved lease ends a stop", () => { + const facts = {operation_lock_free: true, worker_lane_released: true, host_process: "drained"}; + const stops = [ + {stop: {phase: "acknowledged", acknowledged: true, ...facts}, open: "acknowledged", + terminal: stopRecord("settled", "acknowledged_worker_and_host_released")}, + {stop: {phase: "requested", acknowledged: false, ...facts}, open: "requested", + terminal: stopRecord("unknown", "holder_gone_without_acknowledgement")}, + ]; + for (const {stop, open, terminal} of stops) { + // Not checked yet is a next step, never a receipt, and never "nothing owed". + assert.deepEqual(decideDelegationStop({...stop, lease: "unchecked"}), {action: "resolve_lease"}); + for (const lease of ["not_owed", "released"]) assert.deepEqual(decideDelegationStop({...stop, lease}), terminal); + // A held lease whose release is unproven, and an obligation canonical + // authority could not confirm, both keep the stop open, distinguishably. + assert.deepEqual(decideDelegationStop({...stop, lease: "release_unproven"}), + stopRecord(open, "required_lease_release_unproven")); + assert.deepEqual(decideDelegationStop({...stop, lease: "obligation_unproven"}), + stopRecord(open, "lease_obligation_unproven")); + // Omission no longer reads as "no lease"; the old boolean is not a lease fact. + for (const patch of [{}, {lease: undefined}, {lease: "held"}, {lease: true}, {lease_released: true}]) + assert.throws(() => decideDelegationStop({...stop, ...patch}), /lease fact required/); + } + // A lease is never resolved beside an execution that may still run. + for (const pending of [{operation_lock_free: false}, {worker_lane_released: false}, + {host_process: "draining"}, {host_process: "unattributable"}]) { + for (const lease of ["not_owed", "released", "release_unproven", "obligation_unproven"]) + assert.throws(() => decideDelegationStop({...stops[0].stop, ...pending, lease}), /proven gone/); + assert.notEqual(decideDelegationStop({...stops[0].stop, ...pending, lease: "unchecked"}).action, "resolve_lease"); + } }); test("a false rejection can reopen only for exact validated settlement recovery", () => { diff --git a/tests/test_delegation_lease_lifetime.py b/tests/test_delegation_lease_lifetime.py index 65102ba4b0..73fc473ec6 100644 --- a/tests/test_delegation_lease_lifetime.py +++ b/tests/test_delegation_lease_lifetime.py @@ -357,29 +357,37 @@ def settled_stop(runner, operation_id): return receipt -@pytest.mark.parametrize("window", ["annotated", "renewed", "unannotated"]) +# The operation's annotation is a hint; canonical authority decides what is owed. +STALE_ANNOTATIONS = {"empty": {}, "stale_not_required": {"required": False, "handoff_mode": "legacy"}, + "malformed": {"required": "yes", "lease": [1]}} + + +@pytest.mark.parametrize("window", ["annotated", "renewed", "unannotated", *STALE_ANNOTATIONS]) def test_real_stop_releases_the_lease_its_execution_acquired(service, monkeypatch, window): """Native acquire -> public stop -> independent inspect, on both providers. `annotated` is the ordinary record `_acquire_delegation_lease` writes, `renewed` moves the canonical version past the recorded one, and `unannotated` drops the annotation as if the process were lost between the - native claim and the record write. Each must release the exact lease and - settle; a retry is the same receipt. + native claim and the record write. An empty, stale `required: false` or + malformed annotation is no proof that nothing is owed either. Each must + release the exact lease and settle with a receipt that matches the + canonical lease; a retry is the same receipt. """ root, runner = service original = prepare_lease(root, runner, monkeypatch, ttl=None, operation_id="lease-stop") path = runner.path("lease-stop") - if window == "unannotated": + if window == "unannotated" or window in STALE_ANNOTATIONS: row = _read(path) del row["task_lease"] + if window in STALE_ANNOTATIONS: + row["task_lease"] = STALE_ANNOTATIONS[window] runner._fenced_write(path, row) if window == "renewed": renew_current_lease(runner, runner._cli, 60) assert inspect(runner)["lease"]["version"] > _read(path)["task_lease"]["lease"]["version"] receipt = settled_stop(runner, "lease-stop") - assert receipt["stop"]["lease"]["released"] is True - assert receipt["stop"]["settled"]["lease_released"] is True + assert receipt["stop"]["lease"]["state"] == receipt["stop"]["settled"]["lease"] == "released" released = inspect(runner)["lease"] assert released["status"] == "released" assert released["idempotency_key"] == original["idempotency_key"] @@ -423,6 +431,6 @@ def test_real_stop_leaves_a_newer_generation_alone(service, monkeypatch): "--idempotency-key", "new-execution", "--expected-version", str(current["version"])) assert acquired["lease"]["lease_epoch"] > original["lease_epoch"] receipt = settled_stop(runner, "lease-foreign") - assert "lease_released" not in receipt["stop"]["settled"] + assert receipt["stop"]["settled"]["lease"] == "not_owed" held = inspect(runner)["lease"] assert held["status"] == "active" and held["idempotency_key"] == "new-execution" diff --git a/tests/test_delegation_stop_recovery.py b/tests/test_delegation_stop_recovery.py index c66b8bd366..59d886e384 100644 --- a/tests/test_delegation_stop_recovery.py +++ b/tests/test_delegation_stop_recovery.py @@ -163,8 +163,6 @@ def canonical_lease_at_the_native_edge(service, monkeypatch, operation_id, *, "source": "requester", "observed_status": "stopped", "turn_key": None}, }) monkeypatch.setattr(stop_lease, "local_authority_is_promoted", lambda **kwargs: True) - monkeypatch.setattr(stop_lease, "show_goal_handoff_mode", - lambda **kwargs: {"handoff_mode": "hard_lease"}) monkeypatch.setattr(stop_lease, "inspect_task_lease", lambda **kwargs: { "ok": True, "action": "inspect", "active": active, "legacy_fallback_used": False, "lease": {"owner": owner, "idempotency_key": lease_key + key_suffix, @@ -196,7 +194,7 @@ def unavailable(**kwargs): receipt = runner.stop("analysis-lease-window", execute=True) assert receipt["phase"] == "acknowledged", receipt assert receipt["stop"]["reason"] == "required_lease_release_unproven" - assert receipt["stop"]["lease"]["released"] is not True + assert receipt["stop"]["lease"]["state"] == "release_unproven" with pytest.raises(ValueError, match="start a new operation id"): runner.resume("analysis-lease-window") @@ -216,8 +214,7 @@ def test_a_recovered_lease_obligation_settles_only_after_its_exact_release(servi receipt = runner.stop("analysis-lease-recover", execute=True) assert receipt["phase"] == "settled", receipt - assert receipt["stop"]["lease"]["released"] is True - assert receipt["stop"]["settled"]["lease_released"] is True + assert receipt["stop"]["lease"]["state"] == receipt["stop"]["settled"]["lease"] == "released" assert releases == [{"runtime_root": runner.root, "goal_id": runner.goal_id, "todo_id": "todo_analyst-initial", "owner": "analyst", "idempotency_key": lease_key, @@ -243,7 +240,7 @@ def test_a_foreign_lease_generation_is_not_this_stops_obligation(service, monkey receipt = runner.stop("analysis-lease-foreign", execute=True) assert releases == [], "a foreign lease generation was released" assert receipt["phase"] == "settled", receipt - assert "lease_released" not in receipt["stop"]["settled"] + assert receipt["stop"]["settled"]["lease"] == "not_owed" def test_an_unreadable_lease_obligation_keeps_the_stop_open(service, monkeypatch): @@ -266,7 +263,8 @@ def unreadable(**kwargs): receipt = runner.stop("analysis-lease-unreadable", execute=True) assert receipt["phase"] == "acknowledged", receipt - assert receipt["stop"]["reason"] == "required_lease_release_unproven" + assert receipt["stop"]["reason"] == "lease_obligation_unproven" assert releases == [], "an unnamed obligation must not be released" # The reason survives on the receipt instead of becoming a crash. + assert receipt["stop"]["lease"]["state"] == "obligation_unproven" assert "unreadable" in receipt["stop"]["lease"]["error"] diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index 9607ade13b..1ee8723a8e 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -529,7 +529,8 @@ def test_stop_while_executing_is_acknowledged_by_the_worker_and_settles(service, assert stop["settled"]["lane_state"] in {"dead", "released"} assert stop["settled"]["host_process"] == "drained" assert stop["settled"]["turn_journal_status"] == "in_progress" - assert stop["lease"] == {"required": False, "released": None} + # Canonical authority, not the record's annotation, proved no lease was owed. + assert stop["lease"]["state"] == stop["settled"]["lease"] == "not_owed" # The acknowledged record is final: nobody writes it again, the Todo stays open, # the worker exits and the member's Turn lane can be taken. frozen = path.read_bytes() @@ -607,7 +608,10 @@ def kill_without_grace(target, stop): assert receipt["status"] == "running" and receipt["stop"]["ack"] is None, receipt assert receipt["stop"]["settled"]["operation_lock_free"] and receipt["stop"]["settled"]["worker_lane_released"] assert receipt["stop"]["settled"]["turn_journal_status"] == "in_progress" - assert receipt["stop"]["lease"] is None and len(killed) == 1 + # The execution is proven gone, so its lease is resolved before the terminal + # receipt; canonical authority proves this operation held none. + assert receipt["stop"]["lease"]["state"] == receipt["stop"]["settled"]["lease"] == "not_owed" + assert len(killed) == 1 assert until(lambda: process_gone(host_pid), timeout=20) assert runner.stop("analysis-stop", execute=True) == receipt with pytest.raises(ValueError, match="start a new operation id"): @@ -883,8 +887,8 @@ def test_a_failed_required_lease_release_keeps_the_stop_open_and_retries(service "phase": "acknowledged", "ack": {"pid": os.getpid(), "host": lock_holder_host_label(), "at": time.time(), "source": "requester", "observed_status": "stopped", "turn_key": None}, - # The acknowledgement could not release it, which is what the receipt records. - "lease": {"required": True, "released": False, "error": "authority unavailable"}, + # An earlier settlement could not release it, which is what the receipt records. + "lease": {"state": "release_unproven", "error": "authority unavailable"}, }) attempts = [] @@ -904,14 +908,13 @@ def flaky_release(**kwargs): assert attempts, "the stop never attempted the required release" assert receipt["phase"] == "acknowledged", receipt assert receipt["stop"]["reason"] == "required_lease_release_unproven" - assert receipt["stop"]["lease"]["released"] is not True + assert receipt["stop"]["lease"]["state"] == "release_unproven" # The next read retries the release and only then settles. settled = runner.stop("analysis-lease", execute=True) assert len(attempts) >= 2, "the failed release was never retried" assert settled["phase"] == "settled", settled - assert settled["stop"]["lease"]["released"] is True - assert settled["stop"]["settled"]["lease_released"] is True + assert settled["stop"]["lease"]["state"] == settled["stop"]["settled"]["lease"] == "released" # Once settled the receipt is stable, and a released lease is not re-attempted. before = len(attempts) @@ -936,7 +939,7 @@ def test_a_launched_host_on_a_platform_without_process_groups_fails_fast(service record.parent.mkdir(parents=True, exist_ok=True) record.write_text(json.dumps({ "schema_version": host_process_transport.HOST_PROCESS_RECORD_SCHEMA_VERSION, - "host": lock_holder_host_label(), "phase": "finished", + "host": lock_holder_host_label(), "supervises": "host", "phase": "finished", "bridge_pid": os.getpid(), "process_group": os.getpid(), })) @@ -1015,11 +1018,12 @@ def pause_launch(binding, *args, **kwargs): def test_a_crash_between_the_ack_and_the_lease_result_keeps_the_stop_open(service, monkeypatch): """The lease obligation survives process loss after the acknowledgement. - `_acknowledge_stop` writes the ACK before it releases the lease, so a crash - in between leaves the sidecar with no `lease` field. Reading that as "nothing - was owed" settles a stop whose member still holds an active hard lease, with - resume already refused and the Todo blocked until the TTL. The obligation - comes from the operation record, which the crash cannot lose. + The ACK is written long before the lease is resolved, so a crash in between + leaves the sidecar with no `lease` field. Reading that as "nothing was owed" + settles a stop whose member still holds an active hard lease, with resume + already refused and the Todo blocked until the TTL. The obligation comes + from canonical authority for the operation's own identity, which the crash + cannot lose. """ from loopx.control_plane.collaboration.inbox import _write as write_inbox @@ -1050,8 +1054,7 @@ def test_a_crash_between_the_ack_and_the_lease_result_keeps_the_stop_open(servic lambda **kw: {"released": True}) settled = runner.stop("analysis-crash", execute=True) assert settled["phase"] == "settled", settled - assert settled["stop"]["lease"]["released"] is True - assert settled["stop"]["settled"]["lease_released"] is True + assert settled["stop"]["lease"]["state"] == settled["stop"]["settled"]["lease"] == "released" def test_a_crash_between_the_ack_and_the_lease_result_never_settles_unreleased(service, monkeypatch): From 273363daa6d85846934f14e514a8a67a79fdc64f Mon Sep 17 00:00:00 2001 From: song Date: Sun, 4 Oct 2026 07:46:13 +0800 Subject: [PATCH 39/54] docs(delegation): document canonical lease and drain requirements Signed-off-by: song --- docs/reference/goal-chat-continuation.md | 6 +-- docs/reference/local-delegation.md | 69 +++++++++++++++--------- 2 files changed, 46 insertions(+), 29 deletions(-) diff --git a/docs/reference/goal-chat-continuation.md b/docs/reference/goal-chat-continuation.md index 94525871ce..d72c46f449 100644 --- a/docs/reference/goal-chat-continuation.md +++ b/docs/reference/goal-chat-continuation.md @@ -121,8 +121,8 @@ messages; images use the ordinary conversation after pausing. Disabling mode or deleting a binding does not cancel already admitted children; stop one with `loopx delegation stop --execute` (or `stop_delegation`), read its receipt, and retain its evidence. Only `settled` proves the worker - acknowledged and released its locks and its native host exited; stopped work - needs a new operation id. + acknowledged and released its locks, its native host exited and its hard + lease was released or proven not owed; stopped work needs a new operation id. The ordinary native command path also remains available without delegation: @@ -180,6 +180,6 @@ Turn 保持 `wake_dispatch_pending`;中断派发后仍处于 queued 的 Turn 或整个 Goal。额度是含历史用量的总量,正在执行的请求可能超额,成员另行计量。 回滚旧版本前先暂停或关闭 Chat 服务;退出或撤销绑定不自动取消已启动的成员, 用 `loopx delegation stop --execute`(或 `stop_delegation`)停止单个成员并阅读回执: -只有 `settled` 证明 worker 已确认并释放锁且原生 host 已退出;已停止的工作需要新的 operation id。 +只有 `settled` 证明 worker 已确认并释放锁、原生 host 已退出,且硬租约已释放或被证明无需释放;已停止的工作需要新的 operation id。 此按钮目前限本机 managed Codex Goal 对话,不宣称 Lark、挂接会话或其他主力 驱动等价。可用下方示例准备一次隔离的本地 DSH+云端 Ark 协作。 diff --git a/docs/reference/local-delegation.md b/docs/reference/local-delegation.md index bb349c94a6..51934bc2e3 100644 --- a/docs/reference/local-delegation.md +++ b/docs/reference/local-delegation.md @@ -426,14 +426,16 @@ operation cannot overwrite it. A worker on this machine receives `SIGTERM` for its whole process group, which ends its Turn child; the native host runs in its own process group, and its supervisor terminates that group once the Turn child is gone. The worker acknowledges from under its own lock, marks the -record `stopped`. Lease release waits for resource readback to prove that -the worker, Turn lane and every owned Host group have drained. In hard-lease -mode, the leased CLI supervisor and the actual nested Host have separate -records and process groups; neither one proves the other has exited. The -private CLI forwards a separate record address to the Turn transport, which -consumes it before launching user Host code. Old hard-lease records that only -cover the outer CLI cannot prove drain and remain `acknowledged`; reconcile -that original execution rather than deleting its evidence or reusing its key. +record `stopped`. The hard lease is resolved only once readback proves that +the worker, Turn lane and everything the Turn launched have exited. The Host +transport reads that from the one record the operation names: in hard-lease +mode the leased CLI's supervisor records its own group beside it and the +actual nested Host writes it, and neither exit proves the other. Each record +says which group it supervises; the private CLI forwards the record address to +the Turn transport, which consumes it before launching user Host code. A record +that does not say what it supervises, such as an older outer-only hard-lease +record, cannot prove drain and leaves the stop `acknowledged`; reconcile that +original execution rather than deleting its evidence or reusing its key. When nobody holds the operation, the requester acknowledges itself. A worker on another machine is never signalled; it finds the request at its next checkpoint or at its next record write, which is @@ -441,14 +443,19 @@ refused. The receipt `phase` is `settled` only when an acknowledgement exists, the operation lock is free, the member's Turn lane holder record shows it released by the stopped worker (the lane is read, never taken), the native host the Turn launched has exited together with every process in its group, -and a required hard task lease was actually released. The obligation is read -from the operation record, not from the stop sidecar: the acknowledgement is -persisted before the lease is released, so a crash in between must not turn -"not yet written" into "nothing was owed". A release that failed is +and the hard task lease that execution may hold is resolved: released, or +proven not owed. Canonical authority decides, for the operation's own +execution identity (owner, execution key, any recorded epoch and the current +version). The operation's `task_lease` annotation is only a hint: a missing, +empty, stale `required: false` or malformed annotation never proves that +nothing was owed, and a lease another execution holds is never released. The +receipt's `lease.state` is `released`, `not_owed`, `release_unproven` or +`obligation_unproven`. A release that failed is retried under the stop's own lock on the next explicit `stop`, so it never becomes a `settled` receipt that leaves the member's Todo blocked until the lease TTL; while it is unproven the stop stays `acknowledged` with -`required_lease_release_unproven`. If that host cannot be attributed, or its +`required_lease_release_unproven`, and an authority that cannot be read keeps +it open with `lease_obligation_unproven`. If that host cannot be attributed, or its supervisor never finished cleaning up, the stop stays `acknowledged` and a later `stop` rereads it. On a platform without process groups the launched host cannot be proven drained at all, so `stop --execute` fails with an @@ -457,7 +464,9 @@ acknowledging, signalling a worker, or releasing a lease. Repeating a refused request preserves the operation and any existing stop receipt unchanged. An active or unattributable worker is also refused before its Host record appears; a not-yet-started operation with no holder can still be cancelled. `unknown` -means the holder vanished before acknowledging, and `noop` means the +means the holder vanished before acknowledging; its lease is resolved the same +way once its Host is proven gone, and is left to its TTL rather than handed on +when that Host cannot be attributed. `noop` means the work was already accepted, rejected or stopped. `requested` or `acknowledged` means it is still winding down: call `stop` again. A grace timeout never turns into a receipt. Stopped work is not resumed; `resume` refuses it and a new @@ -477,25 +486,32 @@ before completion starts; a lock-acquisition timeout requires retrying `stop`. 因此仍持有该 operation 的 worker 无法覆盖它。本机 worker 会收到整个进程组的 `SIGTERM`,其 Turn 子进程随之结束;原生 host 在自己的进程组中运行,Turn 子进程 退出后由其 supervisor 终止整个 host 进程组。worker 在自己的锁下确认,把记录标为 -`stopped`。只有读回证明 worker、Turn lane 及所有归属的 Host 进程组都已退出, -才会释放硬任务租约。硬租约模式的外层 CLI 与内层真实 Host 分别记录、分别检查, -外层退出不能证明内层退出。私有 CLI 只把独立记录地址交给 Turn transport, -由它在启动用户 Host 前消费,用户 Host 不继承该标记。旧硬租约记录若只覆盖外层 -CLI,则无法证明排空,保持 `acknowledged`;应核对原执行,不能删除证据或复用其 key。 +`stopped`。只有读回证明 worker、Turn lane 及该 Turn 启动的全部进程都已退出, +才会处理硬任务租约。Host transport 从 operation 指定的唯一记录读回这一事实: +硬租约模式下,外层 CLI 的 supervisor 把自己的进程组记录在旁边,内层真实 Host +写入该记录,外层退出不能证明内层退出,反之亦然。每份记录都写明自己监管哪个进程组; +私有 CLI 只把记录地址交给 Turn transport,由它在启动用户 Host 前消费,用户 Host +不继承该标记。未写明监管对象的记录(例如只覆盖外层 CLI 的旧硬租约记录)无法证明 +排空,停止保持 `acknowledged`;应核对原执行,不能删除证据或复用其 key。 没有持有者时由请求方自行确认。另一台机器上的 worker 不会被发信号,它在下一个检查点或下一次写记录时发现请求,写入被拒绝。 只有存在确认、operation 锁已释放、成员 Turn lane 的持有者记录显示已被停止的 worker 释放(只读 lane,从不获取)、该 Turn 启动的原生 host 及其进程组内所有进程 -都已退出,且必需的硬任务租约确实释放成功时,`phase` 才是 `settled`。该义务取自 -操作记录而非 stop sidecar:确认会先于释放落盘,因此两者之间发生崩溃时,不能把 -「尚未写入」当成「本就不需要释放」。释放失败会在下一次显式调用 `stop` 时于其锁下重试, +都已退出,且该执行可能持有的硬任务租约已经处理(已释放,或被证明无需释放)时, +`phase` 才是 `settled`。是否需要释放由 canonical 权威按该 operation 自己的执行身份 +(owner、执行 key、已记录的 epoch 与当前版本)判定;operation 的 `task_lease` +注解只是线索,缺失、为空、过期的 `required: false` 或畸形注解都不能证明无需释放, +其他执行持有的租约也绝不会被释放。回执的 `lease.state` 为 `released`、`not_owed`、 +`release_unproven` 或 `obligation_unproven`。释放失败会在下一次显式调用 `stop` 时于其锁下重试, 因此不会产生一份「已结算」却让成员 Todo 被租约阻塞到 TTL 的回执;在释放得到证明前,停止保持 `acknowledged`,原因为 -`required_lease_release_unproven`。host 无法归属或其 supervisor 未完成清理时, +`required_lease_release_unproven`;权威无法读取时同样保持打开,原因为 +`lease_obligation_unproven`。host 无法归属或其 supervisor 未完成清理时, 停止保持 `acknowledged`,之后再次调用 `stop` 会重新读取。在没有进程组的平台上, 启动过的 host 根本无法被证明已收尾,因此 `stop --execute` 会以指明该平台边界的 可操作错误在写入停止意图、确认、发送信号或释放租约之前失败,重复拒绝不修改原记录。 Host 记录尚未出现但 worker 仍活跃或无法归属时也拒绝;没有持有者且尚未启动的 -operation 仍可安全取消。`unknown` 表示持有者在确认前消失;`noop` 表示工作已 accepted、 +operation 仍可安全取消。`unknown` 表示持有者在确认前消失;其 Host 被证明已退出后, +租约按同样方式处理,Host 无法归属时则留待 TTL,不会在 Host 可能仍运行时交出;`noop` 表示工作已 accepted、 rejected 或 stopped;`requested`/`acknowledged` 表示仍在收尾,再次调用 `stop`。 宽限期超时永远不会变成回执。已停止的工作不能 `resume`,新范围需要新的 operation id。Turn journal 保留 `in_progress` 条目供检查,记录不会被改写成完成; @@ -936,7 +952,8 @@ unchanged and cannot launch workers. With it, the Agent can: This cannot retarget the work or silently create a replacement Turn. 5. Call `stop_delegation(operation_id)` to end one member. Read its `phase`: `settled` is the only receipt that the worker acknowledged and released its - locks and that the native host and its process group exited; `unknown` means the holder vanished first; `noop` means the work had + locks, that the native host and its process group exited, and that its hard + lease was released or proven not owed; `unknown` means the holder vanished first; `noop` means the work had already ended. Stopped work cannot be resumed; use a new operation id. Configure the member's host to expose its own identity-bound collaboration @@ -998,7 +1015,7 @@ whose proof is unavailable does not. | Requesting MCP conversation closes | The detached bounded worker continues; another connection reads the original operation. | | Duplicate start/resume while work runs | Operation identity, task lock and Turn journal prevent another concurrent execution. | | Worker process or machine stops | Reconnect with the same operator configuration and credentials, then resume the original Turn. | -| Member stopped on request | The worker acknowledges under its lock, its Turn child is ended and the host supervisor terminates the host group, its lease is released; `settled` needs that acknowledgement, free locks and an exited host group, `unknown` means the holder vanished first. The record is `stopped`; resume refuses it. | +| Member stopped on request | The worker acknowledges under its lock, its Turn child is ended and the host supervisor terminates the host group; only then is its hard lease resolved against canonical authority. `settled` needs that acknowledgement, free locks, an exited host group and the lease released or proven not owed, `unknown` means the holder vanished first. The record is `stopped`; resume refuses it. | | Ark is computing without local tools | The already-started cloud turn can continue. It is not dependent on the local conversation. | | Ark requests a local tool while the host is absent | It waits for the local tool result. Recovery observes the original input/session and executes only previously unstarted tool calls. | | Tool execution or send acknowledgement is uncertain | Do not repeat the effect. Preserve the receipt/session for explicit reconciliation. | From 7c3410dd8acc03973925a00807e28a5fec5e850f Mon Sep 17 00:00:00 2001 From: song Date: Sun, 4 Oct 2026 10:06:15 +0800 Subject: [PATCH 40/54] test(delegation): preserve fixture cleanup through mocks and setup failure Signed-off-by: song --- tests/test_delegation_authority_recheck.py | 6 ++-- tests/test_delegation_preflight.py | 5 +++- tests/test_delegation_preview_reuse.py | 7 ++++- tests/test_local_delegation.py | 34 ++++++++++++++++++++-- 4 files changed, 45 insertions(+), 7 deletions(-) diff --git a/tests/test_delegation_authority_recheck.py b/tests/test_delegation_authority_recheck.py index 879e5a4992..2c49d24b0d 100644 --- a/tests/test_delegation_authority_recheck.py +++ b/tests/test_delegation_authority_recheck.py @@ -11,9 +11,11 @@ @pytest.fixture -def sqlite_service(tmp_path, monkeypatch): +def sqlite_service(tmp_path, request, monkeypatch): return delegation_service.__wrapped__( - tmp_path, SimpleNamespace(param="sqlite"), monkeypatch + tmp_path, + SimpleNamespace(param="sqlite", addfinalizer=request.addfinalizer), + monkeypatch ) diff --git a/tests/test_delegation_preflight.py b/tests/test_delegation_preflight.py index 3be137f435..c24286b183 100644 --- a/tests/test_delegation_preflight.py +++ b/tests/test_delegation_preflight.py @@ -123,7 +123,10 @@ def test_structured_previews_isolate_concurrent_registry_and_workspace_facts( ): """Same public Goal name, different authorities: no cwd or decision leakage.""" first_root, first = service - provider_request = SimpleNamespace(param=request.node.callspec.params["service"]) + provider_request = SimpleNamespace( + param=request.node.callspec.params["service"], + addfinalizer=request.addfinalizer, + ) second_root, second = delegation_service.__wrapped__( tmp_path / "second", provider_request, monkeypatch ) diff --git a/tests/test_delegation_preview_reuse.py b/tests/test_delegation_preview_reuse.py index 6543ea67e8..d83ec65152 100644 --- a/tests/test_delegation_preview_reuse.py +++ b/tests/test_delegation_preview_reuse.py @@ -282,7 +282,12 @@ def read(): def test_concurrent_services_preserve_registry_runtime_and_workspace_partition(service, tmp_path, request, monkeypatch): root, first = service second_root, second = delegation_service.__wrapped__( - tmp_path / "second", SimpleNamespace(param=request.node.callspec.params["delegation_service"]), monkeypatch + tmp_path / "second", + SimpleNamespace( + param=request.node.callspec.params["delegation_service"], + addfinalizer=request.addfinalizer, + ), + monkeypatch ) second = reusable_service(second) expected = [runner.inspect("analysis") for runner in (first, second)] diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index 1b8ab86373..957597b9c2 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -54,8 +54,9 @@ def service(tmp_path, request, monkeypatch): # Each case owns a private server. Do not accumulate five-minute idle # runtimes across the delegation suite, including failed setup/test cases. runtime_env = os.environ.copy() + run_cleanup = subprocess.run # Tests may replace the shared module before teardown. def retire_runtime(): - subprocess.run([ + run_cleanup([ sys.executable, "-c", "from pathlib import Path; import sys; " "from loopx.control_plane.effect_runtime import _runtime_dir, restart_effect_runtime; " @@ -119,8 +120,11 @@ def test_delegation_fixture_isolates_cached_and_child_runtime_routes(tmp_path, m isolate_sqlite_runtime(cached, neighbor) neighbor_pid = effect_runtime_result("runtime.ping", {})["pid"] try: - for finalize in reversed(finalizers): - finalize() + with monkeypatch.context() as mocked_calls: + mocked_calls.setattr(subprocess, "run", lambda *args, **kwargs: + pytest.fail("teardown must retain its original runner")) + for finalize in reversed(finalizers): + finalize() assert effect_runtime_result("runtime.ping", {})["pid"] == neighbor_pid finally: restart_effect_runtime() @@ -129,6 +133,30 @@ def test_delegation_fixture_isolates_cached_and_child_runtime_routes(tmp_path, m restart_effect_runtime() +def test_delegation_fixture_retires_runtime_after_setup_failure(tmp_path, monkeypatch): + from types import SimpleNamespace + from loopx.control_plane.effect_runtime import ( + _serving_runtime_identity, restart_effect_runtime, effect_runtime_result, + ) + + finalizers = [] + def fail_after_runtime_start(*args, **kwargs): + effect_runtime_result("runtime.ping", {}) + raise RuntimeError("fixture setup interrupted") + + monkeypatch.setattr(demo, "prepare", fail_after_runtime_start) + try: + with pytest.raises(RuntimeError, match="fixture setup interrupted"): + service.__wrapped__(tmp_path, SimpleNamespace(param="file", addfinalizer=finalizers.append), monkeypatch) + assert _serving_runtime_identity() is not None + assert len(finalizers) == 1 + for finalize in reversed(finalizers): + finalize() + assert _serving_runtime_identity() is None + finally: + restart_effect_runtime() + + @pytest.mark.parametrize("operation", ["--help", "x y", "x\ny", "x;echo", "x/../y"]) def test_worker_rejects_unbounded_operation_arguments(tmp_path, monkeypatch, operation): from loopx import collaboration_mcp as delegation From 31970748f6fafb4f3b326262c780b8dba3cefd3d Mon Sep 17 00:00:00 2001 From: song Date: Sun, 4 Oct 2026 10:47:54 +0800 Subject: [PATCH 41/54] fix(host): preserve missing nested execution drain evidence Signed-off-by: song --- .../turn_driver/host_process_transport.py | 26 ++++++++++++++++--- 1 file changed, 22 insertions(+), 4 deletions(-) diff --git a/loopx/control_plane/turn_driver/host_process_transport.py b/loopx/control_plane/turn_driver/host_process_transport.py index d6126c099a..d7c3485d9b 100644 --- a/loopx/control_plane/turn_driver/host_process_transport.py +++ b/loopx/control_plane/turn_driver/host_process_transport.py @@ -78,11 +78,14 @@ def host_process_supervisor_record(record_path: Path) -> Path: return record_path.with_suffix(".cli.host.json") -def host_process_drain(record_path: Path, *, supervises: str = HOST_PROCESS_SUPERVISES_HOST) -> str: +def host_process_drain(record_path: Path, *, supervises: str = HOST_PROCESS_SUPERVISES_HOST, + expected: bool = False) -> str: """Read whether the group one record names, and its supervisor, have exited. Read-only: nothing is signalled, and the TS-owned supervisor keeps cleanup. - No record means nothing was launched under it. ``draining`` while the + A missing expected record proves nothing. An explicit pre-launch record + proves the nested Host has not started; its outer supervisor must also be + gone before the execution can settle. ``draining`` while the supervising bridge or the owned group still has a member. A record from another machine, one that does not say it supervises ``supervises``, or one without a reported group whose supervisor is gone proves nothing: @@ -92,12 +95,17 @@ def host_process_drain(record_path: Path, *, supervises: str = HOST_PROCESS_SUPE try: record = json.loads(record_path.read_text(encoding="utf-8")) except FileNotFoundError: - return HOST_PROCESS_NOT_LAUNCHED + return HOST_PROCESS_UNATTRIBUTABLE if expected else HOST_PROCESS_NOT_LAUNCHED except (OSError, ValueError): return HOST_PROCESS_UNATTRIBUTABLE if (not isinstance(record, dict) or record.get("schema_version") != HOST_PROCESS_RECORD_SCHEMA_VERSION or record.get("host") != lock_holder_host_label() or record.get("supervises") != supervises): return HOST_PROCESS_UNATTRIBUTABLE + if record.get("phase") == HOST_PROCESS_NOT_LAUNCHED: + # Written before the leased CLI can start; the nested transport replaces + # it with a launching record before sending any Host request. + return (HOST_PROCESS_NOT_LAUNCHED if record.get("process_group") is None + and "bridge_pid" not in record else HOST_PROCESS_UNATTRIBUTABLE) if not hasattr(os, "killpg"): # A launched Host on a platform without process groups is never proven # drained. This is a platform boundary, not an attribution failure. @@ -133,7 +141,7 @@ def execution_host_drain(record_path: Path, *, launch_possible: bool = False) -> supervisor = host_process_drain(host_process_supervisor_record(record_path), supervises=HOST_PROCESS_SUPERVISES_NESTED_HOST) - host = host_process_drain(record_path) + host = host_process_drain(record_path, expected=supervisor != HOST_PROCESS_NOT_LAUNCHED) drain = next(state for state in _DRAIN_PRECEDENCE if state in {supervisor, host}) if not host_process_drain_supported() and (launch_possible or drain != HOST_PROCESS_NOT_LAUNCHED): return HOST_PROCESS_UNSUPPORTED_PLATFORM @@ -256,6 +264,16 @@ def run_host_process( "host": lock_holder_host_label(), "owner_pid": os.getpid(), "supervises": supervises, "bridge_pid": proc.pid, "phase": "launching", "process_group": None} + if supervises == HOST_PROCESS_SUPERVISES_NESTED_HOST: + # Positive evidence for a CLI that exits before starting its + # Host. Missing evidence must never mean "not launched". + # Both writes precede the request that can start the CLI. + _write_host_process_record(Path(record_value), { + "schema_version": HOST_PROCESS_RECORD_SCHEMA_VERSION, + "host": lock_holder_host_label(), "owner_pid": os.getpid(), + "supervises": HOST_PROCESS_SUPERVISES_HOST, + "phase": HOST_PROCESS_NOT_LAUNCHED, "process_group": None, + }) _write_host_process_record(record_path, record) proc.stdin.write( json.dumps(request, ensure_ascii=False, separators=(",", ":")) + "\n" From 180425d0fbdaa7b0c8ccefc85e25223d4841841a Mon Sep 17 00:00:00 2001 From: song Date: Sun, 4 Oct 2026 10:47:54 +0800 Subject: [PATCH 42/54] test(delegation): retain live leases when nested host records vanish Signed-off-by: song --- tests/control_plane/test_host_process.py | 2 +- .../control_plane/test_leased_host_process.py | 5 +++- tests/test_delegation_stop_nested_host.py | 25 ++++++++++++++++--- 3 files changed, 26 insertions(+), 6 deletions(-) diff --git a/tests/control_plane/test_host_process.py b/tests/control_plane/test_host_process.py index 91d0bf8c93..f9d20b55e9 100644 --- a/tests/control_plane/test_host_process.py +++ b/tests/control_plane/test_host_process.py @@ -380,7 +380,7 @@ def written(path, supervises, group): (None, None, "not_launched"), (None, ("host", gone.pid), "drained"), (None, ("host", live.pid), "draining"), - (("nested_host", gone.pid), None, "drained"), + (("nested_host", gone.pid), None, "unattributable"), (("nested_host", gone.pid), ("host", gone.pid), "drained"), # The leased CLI exited while its nested Host still runs. (("nested_host", gone.pid), ("host", live.pid), "draining"), diff --git a/tests/control_plane/test_leased_host_process.py b/tests/control_plane/test_leased_host_process.py index f2c667b25a..f2dc428fab 100644 --- a/tests/control_plane/test_leased_host_process.py +++ b/tests/control_plane/test_leased_host_process.py @@ -156,10 +156,13 @@ def test_leased_supervisor_records_its_own_group_beside_the_owners_record(canoni timeout_seconds=30, delegated_lease=context, on_stdout=chunks.append) assert observed["outcome"] == "exited", observed assert "".join(chunks).strip() == "None" # the leased child never inherits the marker - assert not record.exists() + assert json.loads(record.read_text())["phase"] == "not_launched" supervisor = json.loads(host_process_supervisor_record(record).read_text()) assert (supervisor["supervises"], supervisor["phase"]) == ("nested_host", "finished") assert execution_host_drain(record) == "drained" + # Losing the positive never-launched proof must not retain that conclusion. + record.unlink() + assert execution_host_drain(record) == "unattributable" @pytest.mark.skipif(os.name == "nt", reason="POSIX owned process-group qualification") diff --git a/tests/test_delegation_stop_nested_host.py b/tests/test_delegation_stop_nested_host.py index 19d0c0ae07..ee4c89d6eb 100644 --- a/tests/test_delegation_stop_nested_host.py +++ b/tests/test_delegation_stop_nested_host.py @@ -16,8 +16,8 @@ @pytest.mark.skipif(os.name == "nt", reason="POSIX nested supervisor interruption") -@pytest.mark.parametrize("pause_supervisor", [False, True]) -def test_stop_waits_for_actual_nested_host_before_releasing_lease(service, monkeypatch, pause_supervisor): +@pytest.mark.parametrize("interruption", ["none", "paused", "missing_record"]) +def test_stop_waits_for_actual_nested_host_before_releasing_lease(service, monkeypatch, interruption): root, runner = service operation = "nested-stop" with monkeypatch.context() as setup: @@ -40,10 +40,10 @@ def test_stop_waits_for_actual_nested_host_before_releasing_lease(service, monke supervisor = int((root / "supervisor-pid").read_text()) assert not process_gone(host_pid) and not process_gone(child_pid) assert (root / "host-record-env").read_text() == "None" - if pause_supervisor: + if interruption != "none": os.kill(supervisor, signal.SIGSTOP) receipt = runner.stop(operation, execute=True) - if pause_supervisor: + if interruption != "none": assert receipt["phase"] == "acknowledged", receipt assert receipt["reason"] == "host_process_still_running", receipt assert not process_gone(host_pid) and not process_gone(child_pid) @@ -56,6 +56,23 @@ def test_stop_waits_for_actual_nested_host_before_releasing_lease(service, monke assert repeated["phase"] == "acknowledged" assert repeated["stop"]["stop_id"] == receipt["stop"]["stop_id"] assert inspect(runner)["active"] + if interruption == "missing_record": + record = runner._host_process_record(runner.path(operation)) + evidence = record.read_bytes() + record.unlink() + try: + missing = fresh.stop(operation, execute=True) + assert missing["phase"] == "acknowledged", missing + assert missing["reason"] == "host_process_drain_unproven", missing + assert missing["stop"]["stop_id"] == receipt["stop"]["stop_id"] + assert not process_gone(host_pid) and not process_gone(child_pid) + lease = inspect(runner) + assert lease["active"] and lease["lease"]["idempotency_key"] == original["idempotency_key"] + finally: + record.write_bytes(evidence) + restored = fresh.stop(operation, execute=True) + assert restored["phase"] == "acknowledged" + assert restored["reason"] == "host_process_still_running" # Cleanup only the independently identified fixture group. The # product readback must observe this, never signal unrelated groups. os.killpg(host_pid, signal.SIGKILL) From d8b865fb03c8d6c522015503a895faccdca4540f Mon Sep 17 00:00:00 2001 From: song Date: Sun, 4 Oct 2026 12:59:57 +0800 Subject: [PATCH 43/54] fix(delegation): avoid rearming idle timer after preview stop Signed-off-by: song --- .../delegation_preview_bridge.ts | 5 ++- tests/test_delegation_preview_reuse.py | 33 +++++++++++++++++++ 2 files changed, 37 insertions(+), 1 deletion(-) diff --git a/loopx/control_plane/collaboration/delegation_preview_bridge.ts b/loopx/control_plane/collaboration/delegation_preview_bridge.ts index 8e59142cd8..466b78c2e2 100644 --- a/loopx/control_plane/collaboration/delegation_preview_bridge.ts +++ b/loopx/control_plane/collaboration/delegation_preview_bridge.ts @@ -29,7 +29,10 @@ const stop = (reason: StopReason = "cancelled") => { if (lifetime) clearTimeout(lifetime); if (pending) clearTimeout(pending.timer); }; -const armIdle = () => { idle = setTimeout(() => stop("idle"), IDLE_MS); }; +const armIdle = () => { + // A result delivery may resume after cancellation has cleared the timers. + if (!stopped) idle = setTimeout(() => stop("idle"), IDLE_MS); +}; async function accept(value: unknown) { if (!value || typeof value !== "object" || Array.isArray(value)) throw new Error("invalid preview frame"); diff --git a/tests/test_delegation_preview_reuse.py b/tests/test_delegation_preview_reuse.py index d83ec65152..a4d0a0c1ed 100644 --- a/tests/test_delegation_preview_reuse.py +++ b/tests/test_delegation_preview_reuse.py @@ -502,3 +502,36 @@ def read_invalid_fence(*args): argv=("inspect",), timeout=5) assert len(calls) == 2, "invalid fence must not start a second worker" assert transport._process is None + + +@pytest.mark.skipif(sys.platform == "win32", reason="POSIX supervisor cancellation") +def test_stop_during_preview_delivery_does_not_rearm_idle_retirement(tmp_path): + """A result callback resumed after stop must not keep the supervisor alive.""" + from loopx.control_plane.collaboration.delegation_preview_transport import DelegationPreviewTransport + + worker = ("import json,sys\nfor line in sys.stdin:\n" + " r=json.loads(line);print(json.dumps({'kind':'preview','id':r['id']," + "'returncode':0,'value':{'read_only':True}}),flush=True)") + # Invoke the real signal handler during emit(), before the result callback + # resumes and tries to arm idle retirement. Only the event ordering is forced; + # Host execution, cancellation, process cleanup and readback remain real. + preload = ("const write=process.stdout.write.bind(process.stdout);" + "process.stdout.write=(chunk,...args)=>{" + "if(String(chunk).includes('\"kind\":\"preview\"'))process.emit('SIGTERM');" + "return write(chunk,...args)}") + environment = {**_pinned_release_environment(), + "NODE_OPTIONS": "--import=data:text/javascript," + quote(preload, safe="")} + transport = DelegationPreviewTransport() + try: + assert transport.preview(command=[sys.executable, "-c", worker], workspace=tmp_path, + release=tmp_path, environment=environment, + registry=tmp_path / "registry.json", runtime_root=tmp_path / "runtime", + goal_id="fixture", agent_id="lead", todo_id="todo_fixture", + argv=("inspect",), timeout=5) == {"read_only": True} + process = transport._process + assert process is not None + transport.close() + assert process.poll() == 0 + assert transport._process is None + finally: + transport.close() From ccac33d56c631eb29d5592f7fa71b532dc12a381 Mon Sep 17 00:00:00 2001 From: song Date: Sun, 4 Oct 2026 13:28:00 +0800 Subject: [PATCH 44/54] fix(delegation): require both leased Host attribution records Signed-off-by: song --- .../turn_driver/delegated_cli.py | 3 +- .../turn_driver/host_process_transport.py | 59 +++++++++++-------- 2 files changed, 37 insertions(+), 25 deletions(-) diff --git a/loopx/control_plane/turn_driver/delegated_cli.py b/loopx/control_plane/turn_driver/delegated_cli.py index 84eb828ff7..a639e34d0c 100644 --- a/loopx/control_plane/turn_driver/delegated_cli.py +++ b/loopx/control_plane/turn_driver/delegated_cli.py @@ -9,7 +9,7 @@ import signal import sys -from .host_process_transport import HOST_PROCESS_RECORD_ENV +from .host_process_transport import HOST_PROCESS_PARENT_ENV, HOST_PROCESS_RECORD_ENV def _cancel(_signal, _frame): @@ -21,6 +21,7 @@ def _cancel(_signal, _frame): # leased supervisor. The actual Host transport consumes it before launch. if len(sys.argv) >= 3 and sys.argv[1] == "--host-process-record": os.environ[HOST_PROCESS_RECORD_ENV] = sys.argv[2] + os.environ[HOST_PROCESS_PARENT_ENV] = "leased" del sys.argv[1:3] signal.signal(signal.SIGTERM, _cancel) runpy.run_module("loopx.cli", run_name="__main__") diff --git a/loopx/control_plane/turn_driver/host_process_transport.py b/loopx/control_plane/turn_driver/host_process_transport.py index d7c3485d9b..1522aca0db 100644 --- a/loopx/control_plane/turn_driver/host_process_transport.py +++ b/loopx/control_plane/turn_driver/host_process_transport.py @@ -23,6 +23,7 @@ # here. The transport consumes it: the Host never inherits it, so a nested # LoopX run inside the Host cannot overwrite its parent's record. HOST_PROCESS_RECORD_ENV = "LOOPX_HOST_PROCESS_RECORD" +HOST_PROCESS_PARENT_ENV = "LOOPX_HOST_PROCESS_PARENT" HOST_PROCESS_RECORD_SCHEMA_VERSION = "loopx_host_process_record_v0" # Drain facts read back from a record; the caller's typed decision interprets them. HOST_PROCESS_NOT_LAUNCHED = "not_launched" @@ -78,33 +79,32 @@ def host_process_supervisor_record(record_path: Path) -> Path: return record_path.with_suffix(".cli.host.json") -def host_process_drain(record_path: Path, *, supervises: str = HOST_PROCESS_SUPERVISES_HOST, - expected: bool = False) -> str: - """Read whether the group one record names, and its supervisor, have exited. - - Read-only: nothing is signalled, and the TS-owned supervisor keeps cleanup. - A missing expected record proves nothing. An explicit pre-launch record - proves the nested Host has not started; its outer supervisor must also be - gone before the execution can settle. ``draining`` while the - supervising bridge or the owned group still has a member. A record from - another machine, one that does not say it supervises ``supervises``, or - one without a reported group whose supervisor is gone proves nothing: - ``unattributable``. A platform without process groups never proves a drain. - """ - +def _read_host_process_record(path: Path) -> dict[str, Any] | None: try: - record = json.loads(record_path.read_text(encoding="utf-8")) + record = json.loads(path.read_text(encoding="utf-8")) except FileNotFoundError: - return HOST_PROCESS_UNATTRIBUTABLE if expected else HOST_PROCESS_NOT_LAUNCHED + return None except (OSError, ValueError): - return HOST_PROCESS_UNATTRIBUTABLE + return {} # Unreadable evidence is different from an absent record. + return record if isinstance(record, dict) else {} + + +def _host_process_drain(record: dict[str, Any] | None, *, supervises: str, + expected: bool) -> str: + """Read one owner's facts without interpreting missing expected evidence as exit.""" + if record is None: + return HOST_PROCESS_UNATTRIBUTABLE if expected else HOST_PROCESS_NOT_LAUNCHED if (not isinstance(record, dict) or record.get("schema_version") != HOST_PROCESS_RECORD_SCHEMA_VERSION - or record.get("host") != lock_holder_host_label() or record.get("supervises") != supervises): + or record.get("host") != lock_holder_host_label() or record.get("supervises") != supervises + or record.get("supervision") not in {"direct", "leased"} + or (supervises == HOST_PROCESS_SUPERVISES_NESTED_HOST and record["supervision"] != "leased") + or record.get("phase") not in {HOST_PROCESS_NOT_LAUNCHED, "launching", "spawned", "finished"}): return HOST_PROCESS_UNATTRIBUTABLE if record.get("phase") == HOST_PROCESS_NOT_LAUNCHED: # Written before the leased CLI can start; the nested transport replaces # it with a launching record before sending any Host request. - return (HOST_PROCESS_NOT_LAUNCHED if record.get("process_group") is None + return (HOST_PROCESS_NOT_LAUNCHED if record["supervision"] == "leased" + and record.get("process_group") is None and "bridge_pid" not in record else HOST_PROCESS_UNATTRIBUTABLE) if not hasattr(os, "killpg"): # A launched Host on a platform without process groups is never proven @@ -139,9 +139,16 @@ def execution_host_drain(record_path: Path, *, launch_possible: bool = False) -> pre-record launch window. """ - supervisor = host_process_drain(host_process_supervisor_record(record_path), - supervises=HOST_PROCESS_SUPERVISES_NESTED_HOST) - host = host_process_drain(record_path, expected=supervisor != HOST_PROCESS_NOT_LAUNCHED) + # Each side identifies the same leased execution. Either surviving record + # therefore requires its peer; loss of the outer record cannot erase an + # independently running CLI while the nested Host has not started or exited. + owned_record = _read_host_process_record(record_path) + supervisor_record = _read_host_process_record(host_process_supervisor_record(record_path)) + leased = supervisor_record is not None or (owned_record or {}).get("supervision") == "leased" + supervisor = _host_process_drain(supervisor_record, supervises=HOST_PROCESS_SUPERVISES_NESTED_HOST, + expected=leased) + host = _host_process_drain(owned_record, supervises=HOST_PROCESS_SUPERVISES_HOST, + expected=leased) drain = next(state for state in _DRAIN_PRECEDENCE if state in {supervisor, host}) if not host_process_drain_supported() and (launch_possible or drain != HOST_PROCESS_NOT_LAUNCHED): return HOST_PROCESS_UNSUPPORTED_PLATFORM @@ -229,6 +236,9 @@ def run_host_process( # record, and the caller's mapping is never mutated. bridge_environment = os.environ.copy() if environment is None else dict(environment) record_value = bridge_environment.pop(HOST_PROCESS_RECORD_ENV, "") + supervision = bridge_environment.pop(HOST_PROCESS_PARENT_ENV, "direct") + if supervision not in {"direct", "leased"}: + raise ValueError("invalid managed Host supervision scope") record_path = Path(record_value) if record_value else None supervises = HOST_PROCESS_SUPERVISES_HOST if record_path is not None and delegated_lease is not None: @@ -237,6 +247,7 @@ def run_host_process( # own group beside that record, so readback can prove both exited. record_path = host_process_supervisor_record(record_path) supervises = HOST_PROCESS_SUPERVISES_NESTED_HOST + supervision = "leased" with subprocess.Popen( [ _node_executable(), @@ -262,7 +273,7 @@ def run_host_process( # does not name; a record that cannot be written launches nothing. record = {"schema_version": HOST_PROCESS_RECORD_SCHEMA_VERSION, "host": lock_holder_host_label(), "owner_pid": os.getpid(), - "supervises": supervises, "bridge_pid": proc.pid, + "supervises": supervises, "supervision": supervision, "bridge_pid": proc.pid, "phase": "launching", "process_group": None} if supervises == HOST_PROCESS_SUPERVISES_NESTED_HOST: # Positive evidence for a CLI that exits before starting its @@ -271,7 +282,7 @@ def run_host_process( _write_host_process_record(Path(record_value), { "schema_version": HOST_PROCESS_RECORD_SCHEMA_VERSION, "host": lock_holder_host_label(), "owner_pid": os.getpid(), - "supervises": HOST_PROCESS_SUPERVISES_HOST, + "supervises": HOST_PROCESS_SUPERVISES_HOST, "supervision": "leased", "phase": HOST_PROCESS_NOT_LAUNCHED, "process_group": None, }) _write_host_process_record(record_path, record) From b6a359dcb7323e6795f45289e9e61cb78a2ec673 Mon Sep 17 00:00:00 2001 From: song Date: Sun, 4 Oct 2026 13:28:00 +0800 Subject: [PATCH 45/54] docs(delegation): clarify missing peer supervision evidence Signed-off-by: song --- docs/reference/local-delegation.md | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/docs/reference/local-delegation.md b/docs/reference/local-delegation.md index 51934bc2e3..2bd54fae2b 100644 --- a/docs/reference/local-delegation.md +++ b/docs/reference/local-delegation.md @@ -431,7 +431,10 @@ the worker, Turn lane and everything the Turn launched have exited. The Host transport reads that from the one record the operation names: in hard-lease mode the leased CLI's supervisor records its own group beside it and the actual nested Host writes it, and neither exit proves the other. Each record -says which group it supervises; the private CLI forwards the record address to +says which group it supervises and whether it belongs to a leased execution. +Either surviving leased record requires the other: a missing outer record is +not proof that its CLI exited, even when the nested Host never started. +The private CLI forwards the record address and supervision scope to the Turn transport, which consumes it before launching user Host code. A record that does not say what it supervises, such as an older outer-only hard-lease record, cannot prove drain and leaves the stop `acknowledged`; reconcile that @@ -489,7 +492,9 @@ before completion starts; a lock-acquisition timeout requires retrying `stop`. `stopped`。只有读回证明 worker、Turn lane 及该 Turn 启动的全部进程都已退出, 才会处理硬任务租约。Host transport 从 operation 指定的唯一记录读回这一事实: 硬租约模式下,外层 CLI 的 supervisor 把自己的进程组记录在旁边,内层真实 Host -写入该记录,外层退出不能证明内层退出,反之亦然。每份记录都写明自己监管哪个进程组; +写入该记录,外层退出不能证明内层退出,反之亦然。每份记录都写明自己监管哪个进程组、 +是否属于租约监督的执行。任一侧的记录保留时,另一侧缺失都不能证明退出;即使内层 +Host 尚未启动,也不能据此判定外层 CLI 已退出。 私有 CLI 只把记录地址交给 Turn transport,由它在启动用户 Host 前消费,用户 Host 不继承该标记。未写明监管对象的记录(例如只覆盖外层 CLI 的旧硬租约记录)无法证明 排空,停止保持 `acknowledged`;应核对原执行,不能删除证据或复用其 key。 From bb9250af63fceeac7c84f8e4bc8d6b72dec9b6a8 Mon Sep 17 00:00:00 2001 From: song Date: Sun, 4 Oct 2026 13:28:00 +0800 Subject: [PATCH 46/54] test(delegation): cover lost outer records and safe recovery Signed-off-by: song --- tests/control_plane/test_host_process.py | 55 ++++++++++++--------- tests/test_delegation_stop_nested_host.py | 60 +++++++++++++++++++++++ tests/test_local_delegation.py | 6 +-- 3 files changed, 95 insertions(+), 26 deletions(-) diff --git a/tests/control_plane/test_host_process.py b/tests/control_plane/test_host_process.py index f9d20b55e9..6a57219bd8 100644 --- a/tests/control_plane/test_host_process.py +++ b/tests/control_plane/test_host_process.py @@ -212,11 +212,11 @@ def test_windows_transport_relay_preserves_argv_and_stdin(tmp_path: Path) -> Non @pytest.mark.skipif(os.name == "nt", reason="POSIX process-group drain readback") def test_host_process_record_names_the_owned_group_and_is_not_inherited(tmp_path: Path, monkeypatch) -> None: from loopx.control_plane.turn_driver.host_process_transport import ( - HOST_PROCESS_RECORD_ENV, host_process_drain, + HOST_PROCESS_RECORD_ENV, execution_host_drain, ) record_path = tmp_path / "op.host.json" - assert host_process_drain(record_path) == "not_launched" + assert execution_host_drain(record_path) == "not_launched" monkeypatch.setenv(HOST_PROCESS_RECORD_ENV, str(record_path)) host = ("import json,os,sys;print(json.dumps({'env': os.environ.get(%r), 'pid': os.getpid()," " 'pgid': os.getpgid(0)}))" % HOST_PROCESS_RECORD_ENV) @@ -227,13 +227,13 @@ def test_host_process_record_names_the_owned_group_and_is_not_inherited(tmp_path record = json.loads(record_path.read_text()) assert record["phase"] == "finished" assert record["host_pid"] == result["value"]["pid"] == record["process_group"] == result["value"]["pgid"] - assert host_process_drain(record_path) == "drained" + assert execution_host_drain(record_path) == "drained" @pytest.mark.skipif(os.name == "nt", reason="POSIX process-group drain readback") def test_host_process_drain_reads_live_groups_and_refuses_unattributable_records(tmp_path: Path) -> None: from loopx.control_plane.turn_driver.host_process_transport import ( - HOST_PROCESS_RECORD_SCHEMA_VERSION, host_process_drain, + HOST_PROCESS_RECORD_SCHEMA_VERSION, execution_host_drain, ) from loopx.file_lock import lock_holder_host_label @@ -245,8 +245,8 @@ def test_host_process_drain_reads_live_groups_and_refuses_unattributable_records def drain(**fields): path.write_text(json.dumps({"schema_version": HOST_PROCESS_RECORD_SCHEMA_VERSION, "host": lock_holder_host_label(), "owner_pid": 1, - "supervises": "host", **fields})) - return host_process_drain(path) + "supervises": "host", "supervision": "direct", **fields})) + return execution_host_drain(path) try: # A live supervisor or a live Host group is still draining. @@ -257,7 +257,10 @@ def drain(**fields): # A supervisor gone before it reported a group may have spawned one anyway. assert drain(phase="launching", bridge_pid=gone.pid, process_group=None) == "unattributable" assert drain(phase="finished", bridge_pid=gone.pid, process_group=None) == "drained" - for fields in ({"phase": "spawned", "bridge_pid": None, "process_group": gone.pid}, + for fields in ({"phase": "invalid", "bridge_pid": gone.pid, "process_group": None}, + {"phase": "finished", "bridge_pid": gone.pid, "process_group": gone.pid, + "supervision": None}, + {"phase": "spawned", "bridge_pid": None, "process_group": gone.pid}, {"phase": "spawned", "bridge_pid": gone.pid, "process_group": "1"}, {"phase": "spawned", "bridge_pid": gone.pid, "process_group": 1}, {"phase": "spawned", "bridge_pid": gone.pid, "process_group": gone.pid, @@ -271,7 +274,7 @@ def drain(**fields): "supervises": "nested_host"}): assert drain(**fields) == "unattributable", fields path.write_text("{not json") - assert host_process_drain(path) == "unattributable" + assert execution_host_drain(path) == "unattributable" finally: live.kill() live.wait(timeout=10) @@ -288,7 +291,7 @@ def test_explicit_environment_reaches_the_host_and_stays_out_of_the_record(tmp_p nested run that must not overwrite its parent's record. """ from loopx.control_plane.turn_driver.host_process_transport import ( - HOST_PROCESS_RECORD_ENV, host_process_drain, run_host_process, + HOST_PROCESS_RECORD_ENV, execution_host_drain, run_host_process, ) record_path = tmp_path / "op.host.json" @@ -312,7 +315,7 @@ def test_explicit_environment_reaches_the_host_and_stays_out_of_the_record(tmp_p record = json.loads(record_path.read_text()) assert record["phase"] == "finished" assert record["host_pid"] == value["pid"] == record["process_group"] == value["pgid"] - assert host_process_drain(record_path) == "drained" + assert execution_host_drain(record_path) == "drained" # The caller's mapping is the caller's. assert environment[HOST_PROCESS_RECORD_ENV] == str(record_path) assert environment[selected] == "caller-selected" @@ -370,29 +373,35 @@ def test_execution_drain_needs_every_group_a_leased_run_launched(tmp_path: Path) gone = subprocess.Popen([sys.executable, "-c", "pass"], start_new_session=True) gone.wait(timeout=10) - def written(path, supervises, group): + def written(path, supervises, group, supervision): path.write_text(json.dumps({"schema_version": HOST_PROCESS_RECORD_SCHEMA_VERSION, "host": lock_holder_host_label(), "owner_pid": 1, "supervises": supervises, + "supervision": supervision, "phase": "finished", "bridge_pid": gone.pid, "process_group": group})) cases = [ # (supervisor record, owner's record, observation) (None, None, "not_launched"), - (None, ("host", gone.pid), "drained"), - (None, ("host", live.pid), "draining"), - (("nested_host", gone.pid), None, "unattributable"), - (("nested_host", gone.pid), ("host", gone.pid), "drained"), + (None, ("host", gone.pid, "direct"), "drained"), + (None, ("host", live.pid, "direct"), "draining"), + (("nested_host", gone.pid, "leased"), None, "unattributable"), + (("nested_host", gone.pid, "leased"), ("host", gone.pid, "leased"), "drained"), + # A later direct recovery Host still reads the earlier outer group. + (("nested_host", gone.pid, "leased"), ("host", gone.pid, "direct"), "drained"), + # Losing either leased record leaves the surviving peer insufficient. + (None, ("host", gone.pid, "leased"), "unattributable"), + (None, ("host", live.pid, "leased"), "draining"), # The leased CLI exited while its nested Host still runs. - (("nested_host", gone.pid), ("host", live.pid), "draining"), - (("nested_host", live.pid), ("host", gone.pid), "draining"), + (("nested_host", gone.pid, "leased"), ("host", live.pid, "leased"), "draining"), + (("nested_host", live.pid, "leased"), ("host", gone.pid, "leased"), "draining"), # A group still seen running outranks a missing proof. - (("nested_host", live.pid), "corrupt", "draining"), - (("nested_host", gone.pid), "corrupt", "unattributable"), + (("nested_host", live.pid, "leased"), "corrupt", "draining"), + (("nested_host", gone.pid, "leased"), "corrupt", "unattributable"), # Records that do not say what they supervise, or say the wrong thing. - (None, (None, gone.pid), "unattributable"), - (None, ("nested_host", gone.pid), "unattributable"), - (("host", gone.pid), ("host", gone.pid), "unattributable"), - ((None, gone.pid), None, "unattributable"), + (None, (None, gone.pid, "leased"), "unattributable"), + (None, ("nested_host", gone.pid, "leased"), "unattributable"), + (("host", gone.pid, "leased"), ("host", gone.pid, "leased"), "unattributable"), + ((None, gone.pid, "leased"), None, "unattributable"), ] try: for outer, owned, expected in cases: diff --git a/tests/test_delegation_stop_nested_host.py b/tests/test_delegation_stop_nested_host.py index ee4c89d6eb..78976550c7 100644 --- a/tests/test_delegation_stop_nested_host.py +++ b/tests/test_delegation_stop_nested_host.py @@ -27,6 +27,7 @@ def test_stop_waits_for_actual_nested_host_before_releasing_lease(service, monke host = HOST.replace("counter = workspace / 'host-invocations'", """ (root / 'supervisor-pid').write_text(str(os.getppid())) (root / 'host-record-env').write_text(str(os.environ.get('LOOPX_HOST_PROCESS_RECORD'))) +(root / 'host-parent-env').write_text(str(os.environ.get('LOOPX_HOST_PROCESS_PARENT'))) counter = workspace / 'host-invocations'""") (root / "fixture-host.py").write_text(host) (root / "hold").touch() @@ -40,6 +41,7 @@ def test_stop_waits_for_actual_nested_host_before_releasing_lease(service, monke supervisor = int((root / "supervisor-pid").read_text()) assert not process_gone(host_pid) and not process_gone(child_pid) assert (root / "host-record-env").read_text() == "None" + assert (root / "host-parent-env").read_text() == "None" if interruption != "none": os.kill(supervisor, signal.SIGSTOP) receipt = runner.stop(operation, execute=True) @@ -127,3 +129,61 @@ def test_missing_or_unreadable_nested_attribution_cannot_release_a_lease(service assert stopped["reason"] == "host_process_drain_unproven" assert inspect(runner)["active"] assert record.read_bytes() == evidence + + +@pytest.mark.skipif(os.name == "nt", reason="POSIX owned fixture") +def test_missing_outer_record_does_not_settle_live_leased_cli(service, monkeypatch): + """The nested Host has not started, but its leased CLI is independently alive.""" + import json + import sys + from concurrent.futures import ThreadPoolExecutor + + from loopx.control_plane.turn_driver.host_process_transport import ( + HOST_PROCESS_RECORD_ENV, host_process_supervisor_record, run_host_process, + ) + + root, runner = service + operation = "outer-record-loss" + original = prepare_lease(root, runner, monkeypatch, ttl=None, operation_id=operation) + path = runner.path(operation) + record = runner._host_process_record(path) + context = runner._delegated_lease_context(_read(path), runner.binding("analysis")) + marker, finish = root / "outer-live", root / "finish-outer" + source = ("import os,sys,time;from pathlib import Path;" + "Path(sys.argv[1]).write_text(str(os.getpid()));\n" + "while not Path(sys.argv[2]).exists():time.sleep(.02)") + with ThreadPoolExecutor(max_workers=1) as pool: + # The normal transport writes both records and runs under the real + # canonical lease. Model only recovery after the operation holder left; + # do not mock process liveness, drain, settlement or lease readback. + running = pool.submit(run_host_process, [sys.executable, "-c", source, str(marker), str(finish)], + project=root, input_text="", timeout_seconds=30, delegated_lease=context, + environment={**os.environ, HOST_PROCESS_RECORD_ENV: str(record)}) + evidence = None + try: + assert until(marker.exists) + pid = int(marker.read_text()) + assert not process_gone(pid) + outer = host_process_supervisor_record(record) + assert until(lambda: json.loads(outer.read_text())["phase"] == "spawned") + evidence = outer.read_bytes() + outer.unlink() + receipt = runner.stop(operation, execute=True) + actual = inspect(runner) + assert not process_gone(pid) + assert receipt["phase"] == "acknowledged", receipt + assert receipt["reason"] == "host_process_drain_unproven" + assert actual["active"] and actual["lease"]["idempotency_key"] == original["idempotency_key"] + outer.write_bytes(evidence) + restored = runner.stop(operation, execute=True) + assert restored["phase"] == "acknowledged" + assert restored["stop"]["stop_id"] == receipt["stop"]["stop_id"] + finally: + if evidence is not None: + outer.write_bytes(evidence) + finish.touch() + running.result(timeout=30) + settled = runner.stop(operation, execute=True) + assert settled["phase"] == "settled" + assert settled["stop"]["stop_id"] == receipt["stop"]["stop_id"] + assert inspect(runner)["lease"]["status"] == "released" diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index 3b9c94735f..48ad2c98b8 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -966,14 +966,14 @@ def test_a_launched_host_on_a_platform_without_process_groups_fails_fast(service record.parent.mkdir(parents=True, exist_ok=True) record.write_text(json.dumps({ "schema_version": host_process_transport.HOST_PROCESS_RECORD_SCHEMA_VERSION, - "host": lock_holder_host_label(), "supervises": "host", "phase": "finished", + "host": lock_holder_host_label(), "supervises": "host", "supervision": "direct", "phase": "finished", "bridge_pid": os.getpid(), "process_group": os.getpid(), })) # A persisted finished Host record still needs platform drain capability. - assert host_process_transport.host_process_drain(record) == host_process_transport.HOST_PROCESS_DRAINED + assert host_process_transport.execution_host_drain(record) == host_process_transport.HOST_PROCESS_DRAINED monkeypatch.delattr(host_process_transport.os, "killpg", raising=False) - assert host_process_transport.host_process_drain(record) == host_process_transport.HOST_PROCESS_UNSUPPORTED_PLATFORM + assert host_process_transport.execution_host_drain(record) == host_process_transport.HOST_PROCESS_UNSUPPORTED_PLATFORM operation_before = path.read_bytes() stop_path = runner._stop_path(path) From 84a7480c9b96d7c70932f15e4dfe2aa788b4ced5 Mon Sep 17 00:00:00 2001 From: song Date: Sun, 4 Oct 2026 13:39:11 +0800 Subject: [PATCH 47/54] fix(delegation): require positive proof before declaring no Host execution Signed-off-by: song --- loopx/collaboration_mcp.py | 3 ++- .../turn_driver/host_process_transport.py | 26 ++++++++++++++----- 2 files changed, 22 insertions(+), 7 deletions(-) diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index eb1f071070..fabbf23a4f 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -45,7 +45,7 @@ from .control_plane.turn_driver.host_binding import turn_host_arg_option from .control_plane.turn_driver.host_process_transport import ( HOST_PROCESS_DRAINING, HOST_PROCESS_RECORD_ENV, - execution_host_drain, require_execution_host_drain_supported, + execution_host_drain, prepare_host_process_record, require_execution_host_drain_supported, ) from .control_plane.turn_driver.lane_fence import ( TURN_LANE_ABSENT, TURN_LANE_DEAD, TURN_LANE_LIVE, TURN_LANE_RELEASED, @@ -675,6 +675,7 @@ def start(self, binding_id: str, operation_id: str, brief: dict, else: origin = ({"session_id": str(conversation["session_id"]), "turn_id": str(conversation["turn_id"])} if conversation else None) + prepare_host_process_record(self._host_process_record(path)) _write(path, {"identity": identity, "status": "prepared", "created_at": time.time(), **({"conversation": origin} if origin else {})}) self._spawn(operation_id) diff --git a/loopx/control_plane/turn_driver/host_process_transport.py b/loopx/control_plane/turn_driver/host_process_transport.py index 1522aca0db..31b67917fa 100644 --- a/loopx/control_plane/turn_driver/host_process_transport.py +++ b/loopx/control_plane/turn_driver/host_process_transport.py @@ -64,6 +64,20 @@ def _write_host_process_record(path: Path, record: dict[str, Any]) -> None: Path(temporary).unlink(missing_ok=True) +def prepare_host_process_record(record_path: Path) -> None: + """Record non-execution before a new operation can launch, never on replay. + + The operation creator holds its dispatch lock. Retain any existing evidence; + initialization must not erase a prior launch, including interrupted setup. + """ + if not record_path.exists(): + _write_host_process_record(record_path, { + "schema_version": HOST_PROCESS_RECORD_SCHEMA_VERSION, + "host": lock_holder_host_label(), "supervises": HOST_PROCESS_SUPERVISES_HOST, + "supervision": "direct", "phase": HOST_PROCESS_NOT_LAUNCHED, "process_group": None, + }) + + def _process_group_present(pgid: int) -> bool: try: os.killpg(pgid, 0) @@ -101,10 +115,9 @@ def _host_process_drain(record: dict[str, Any] | None, *, supervises: str, or record.get("phase") not in {HOST_PROCESS_NOT_LAUNCHED, "launching", "spawned", "finished"}): return HOST_PROCESS_UNATTRIBUTABLE if record.get("phase") == HOST_PROCESS_NOT_LAUNCHED: - # Written before the leased CLI can start; the nested transport replaces - # it with a launching record before sending any Host request. - return (HOST_PROCESS_NOT_LAUNCHED if record["supervision"] == "leased" - and record.get("process_group") is None + # Written before execution can start; its transport replaces this with + # a launching record before sending any Host request. + return (HOST_PROCESS_NOT_LAUNCHED if record.get("process_group") is None and "bridge_pid" not in record else HOST_PROCESS_UNATTRIBUTABLE) if not hasattr(os, "killpg"): # A launched Host on a platform without process groups is never proven @@ -136,7 +149,8 @@ def execution_host_drain(record_path: Path, *, launch_possible: bool = False) -> it supervises, such as one written before records carried that fact, attributes nothing. The caller supplies whether its launch owner may still start a Host; on unsupported platforms an absent record cannot close that - pre-record launch window. + pre-record launch window. Absence of the primary record is never proof of + non-execution; the creator records that fact before its first launch. """ # Each side identifies the same leased execution. Either surviving record @@ -148,7 +162,7 @@ def execution_host_drain(record_path: Path, *, launch_possible: bool = False) -> supervisor = _host_process_drain(supervisor_record, supervises=HOST_PROCESS_SUPERVISES_NESTED_HOST, expected=leased) host = _host_process_drain(owned_record, supervises=HOST_PROCESS_SUPERVISES_HOST, - expected=leased) + expected=True) drain = next(state for state in _DRAIN_PRECEDENCE if state in {supervisor, host}) if not host_process_drain_supported() and (launch_possible or drain != HOST_PROCESS_NOT_LAUNCHED): return HOST_PROCESS_UNSUPPORTED_PLATFORM From 54ff38b2b445cad04851ff9fdda83165f1f5b0c9 Mon Sep 17 00:00:00 2001 From: song Date: Sun, 4 Oct 2026 13:39:11 +0800 Subject: [PATCH 48/54] docs(delegation): distinguish missing proof from non-execution Signed-off-by: song --- docs/reference/local-delegation.md | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/docs/reference/local-delegation.md b/docs/reference/local-delegation.md index 2bd54fae2b..4eaf97b08f 100644 --- a/docs/reference/local-delegation.md +++ b/docs/reference/local-delegation.md @@ -432,6 +432,9 @@ transport reads that from the one record the operation names: in hard-lease mode the leased CLI's supervisor records its own group beside it and the actual nested Host writes it, and neither exit proves the other. Each record says which group it supervises and whether it belongs to a leased execution. +A new operation gets an explicit non-execution record before it can launch; +replaying an existing operation never recreates lost proof. Missing primary +evidence is unproven, even when both records are absent. Either surviving leased record requires the other: a missing outer record is not proof that its CLI exited, even when the nested Host never started. The private CLI forwards the record address and supervision scope to @@ -493,9 +496,11 @@ before completion starts; a lock-acquisition timeout requires retrying `stop`. 才会处理硬任务租约。Host transport 从 operation 指定的唯一记录读回这一事实: 硬租约模式下,外层 CLI 的 supervisor 把自己的进程组记录在旁边,内层真实 Host 写入该记录,外层退出不能证明内层退出,反之亦然。每份记录都写明自己监管哪个进程组、 -是否属于租约监督的执行。任一侧的记录保留时,另一侧缺失都不能证明退出;即使内层 +是否属于租约监督的执行。新操作在启动前先登记明确的未执行证明;重放已有操作 +不能重新创建丢失的证明。即使两份记录都缺失,也只能判定为未证明。 +任一侧的记录保留时,另一侧缺失都不能证明退出;即使内层 Host 尚未启动,也不能据此判定外层 CLI 已退出。 -私有 CLI 只把记录地址交给 Turn transport,由它在启动用户 Host 前消费,用户 Host +私有 CLI 只把记录地址和监督范围交给 Turn transport,由它在启动用户 Host 前消费,用户 Host 不继承该标记。未写明监管对象的记录(例如只覆盖外层 CLI 的旧硬租约记录)无法证明 排空,停止保持 `acknowledged`;应核对原执行,不能删除证据或复用其 key。 没有持有者时由请求方自行确认。另一台机器上的 From f02de77630d669b292f1728643be1c72018558c5 Mon Sep 17 00:00:00 2001 From: song Date: Sun, 4 Oct 2026 13:39:11 +0800 Subject: [PATCH 49/54] test(delegation): preserve stop obligations when all Host records are lost Signed-off-by: song --- tests/control_plane/test_host_process.py | 9 +++++++-- tests/test_delegation_stop_nested_host.py | 15 +++++++++++++-- tests/test_local_delegation.py | 2 +- 3 files changed, 21 insertions(+), 5 deletions(-) diff --git a/tests/control_plane/test_host_process.py b/tests/control_plane/test_host_process.py index 6a57219bd8..aa977a5973 100644 --- a/tests/control_plane/test_host_process.py +++ b/tests/control_plane/test_host_process.py @@ -212,10 +212,12 @@ def test_windows_transport_relay_preserves_argv_and_stdin(tmp_path: Path) -> Non @pytest.mark.skipif(os.name == "nt", reason="POSIX process-group drain readback") def test_host_process_record_names_the_owned_group_and_is_not_inherited(tmp_path: Path, monkeypatch) -> None: from loopx.control_plane.turn_driver.host_process_transport import ( - HOST_PROCESS_RECORD_ENV, execution_host_drain, + HOST_PROCESS_RECORD_ENV, execution_host_drain, prepare_host_process_record, ) record_path = tmp_path / "op.host.json" + assert execution_host_drain(record_path) == "unattributable" + prepare_host_process_record(record_path) assert execution_host_drain(record_path) == "not_launched" monkeypatch.setenv(HOST_PROCESS_RECORD_ENV, str(record_path)) host = ("import json,os,sys;print(json.dumps({'env': os.environ.get(%r), 'pid': os.getpid()," @@ -228,6 +230,9 @@ def test_host_process_record_names_the_owned_group_and_is_not_inherited(tmp_path assert record["phase"] == "finished" assert record["host_pid"] == result["value"]["pid"] == record["process_group"] == result["value"]["pgid"] assert execution_host_drain(record_path) == "drained" + observed = record_path.read_bytes() + prepare_host_process_record(record_path) + assert record_path.read_bytes() == observed, "initialization must not overwrite execution evidence" @pytest.mark.skipif(os.name == "nt", reason="POSIX process-group drain readback") @@ -381,7 +386,7 @@ def written(path, supervises, group, supervision): cases = [ # (supervisor record, owner's record, observation) - (None, None, "not_launched"), + (None, None, "unattributable"), (None, ("host", gone.pid, "direct"), "drained"), (None, ("host", live.pid, "direct"), "draining"), (("nested_host", gone.pid, "leased"), None, "unattributable"), diff --git a/tests/test_delegation_stop_nested_host.py b/tests/test_delegation_stop_nested_host.py index 78976550c7..b03380038a 100644 --- a/tests/test_delegation_stop_nested_host.py +++ b/tests/test_delegation_stop_nested_host.py @@ -8,7 +8,7 @@ import pytest -from test_local_delegation import HOST, process_gone, service, until # noqa: F401 +from test_local_delegation import HOST, brief, process_gone, service, until # noqa: F401 from test_delegation_lease_lifetime import inspect, prepare_lease from loopx.collaboration_mcp import Delegations from loopx.control_plane.collaboration.inbox import _read @@ -132,7 +132,8 @@ def test_missing_or_unreadable_nested_attribution_cannot_release_a_lease(service @pytest.mark.skipif(os.name == "nt", reason="POSIX owned fixture") -def test_missing_outer_record_does_not_settle_live_leased_cli(service, monkeypatch): +@pytest.mark.parametrize("missing_records", ["outer", "both"]) +def test_missing_outer_record_does_not_settle_live_leased_cli(service, monkeypatch, missing_records): """The nested Host has not started, but its leased CLI is independently alive.""" import json import sys @@ -167,20 +168,30 @@ def test_missing_outer_record_does_not_settle_live_leased_cli(service, monkeypat outer = host_process_supervisor_record(record) assert until(lambda: json.loads(outer.read_text())["phase"] == "spawned") evidence = outer.read_bytes() + owned_evidence = record.read_bytes() outer.unlink() + if missing_records == "both": + record.unlink() receipt = runner.stop(operation, execute=True) actual = inspect(runner) assert not process_gone(pid) assert receipt["phase"] == "acknowledged", receipt assert receipt["reason"] == "host_process_drain_unproven" assert actual["active"] and actual["lease"]["idempotency_key"] == original["idempotency_key"] + # Replaying an existing operation cannot manufacture new proof. + replayed = runner.start("analysis", operation, brief()) + assert replayed["stop"]["phase"] == "acknowledged" + if missing_records == "both": + assert not record.exists() outer.write_bytes(evidence) + record.write_bytes(owned_evidence) restored = runner.stop(operation, execute=True) assert restored["phase"] == "acknowledged" assert restored["stop"]["stop_id"] == receipt["stop"]["stop_id"] finally: if evidence is not None: outer.write_bytes(evidence) + record.write_bytes(owned_evidence) finish.touch() running.result(timeout=30) settled = runner.stop(operation, execute=True) diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index 48ad2c98b8..81ba713e2a 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -1028,7 +1028,7 @@ def pause_launch(binding, *args, **kwargs): worker = pool.submit(runner.execute, "platform-launch") try: assert entering.wait(20) - assert not runner._host_process_record(path).exists() + assert json.loads(runner._host_process_record(path).read_text())["phase"] == "not_launched" before = path.read_bytes() with monkeypatch.context() as platform: platform.delattr(host_process_transport.os, "killpg", raising=False) From 77145777a4a9e8de8cca69275aa7ebcd3f586490 Mon Sep 17 00:00:00 2001 From: song Date: Sun, 4 Oct 2026 15:42:09 +0800 Subject: [PATCH 50/54] test: preserve unexpected notification failure causes Signed-off-by: song --- tests/control_plane/test_refresh_checkpoint_isolation.py | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/tests/control_plane/test_refresh_checkpoint_isolation.py b/tests/control_plane/test_refresh_checkpoint_isolation.py index c82944278b..a7ce27f4a2 100644 --- a/tests/control_plane/test_refresh_checkpoint_isolation.py +++ b/tests/control_plane/test_refresh_checkpoint_isolation.py @@ -13,7 +13,7 @@ build_explore_node_event, explore_result_log_path, ) -from loopx.extensions.lark import goal_channel_contracts, goal_channel_runtime +from loopx.extensions.lark import goal_channel_contracts, goal_channel_lifecycle, goal_channel_runtime from loopx.extensions.lark.presentation import explore_results from loopx.extensions.runtime import install_extension from loopx.global_registry import sync_project_registry_to_global @@ -104,6 +104,12 @@ def _configure_sinks(registry, runtime): def test_stdout_recovery_requires_confirmation_to_resume_external_delivery( tmp_path, monkeypatch, capsys, hint_source, baseline, isolated, ): + def unexpected_notification_failure(error): + # This successful synthetic delivery must not hide a failure behind the + # public error projection. Preserve its cause in the test traceback. + pytest.fail(f"Unexpected notification failure: {type(error).__name__}") + + monkeypatch.setattr(goal_channel_lifecycle, "_exception_reason", unexpected_notification_failure) project, runtime, registry = _write_fixture(tmp_path) shared = tmp_path / "shared-runtime" monkeypatch.setenv("LOOPX_RUNTIME_ROOT", str(shared)) From 33177b78c75d1dd34d4fd10f665102efe4a1cda0 Mon Sep 17 00:00:00 2001 From: song Date: Sun, 4 Oct 2026 16:02:58 +0800 Subject: [PATCH 51/54] fix: require canonical absence before clearing stop lease obligation Signed-off-by: song --- loopx/collaboration_mcp.py | 2 +- .../collaboration/delegation_stop_lease.py | 21 +++++++++++++++---- 2 files changed, 18 insertions(+), 5 deletions(-) diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index fabbf23a4f..8659773ded 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -1201,7 +1201,7 @@ def _settle_stop(self, path: Path) -> dict: # its lease resolved, against canonical authority by the # execution's own identity, and only the resolved fact can make # the receipt terminal. Under the same lock as the record, every - # later read retries a release that has not been proven. + # later explicit stop retries a release that has not been proven. facts["lease"] = delegation_stop_lease.settle(self, path, row, binding, stop) decision = effect_runtime_result("collaboration.delegation.stop", {**inputs, **facts}) if decision["phase"] != stop["phase"] or decision.get("reason") != stop.get("reason"): diff --git a/loopx/control_plane/collaboration/delegation_stop_lease.py b/loopx/control_plane/collaboration/delegation_stop_lease.py index d87e404969..8a0190c1e4 100644 --- a/loopx/control_plane/collaboration/delegation_stop_lease.py +++ b/loopx/control_plane/collaboration/delegation_stop_lease.py @@ -11,7 +11,8 @@ from .inbox import _write from ..coordination.local_authority import local_authority_is_promoted -from ..effect_runtime import EffectRuntimeRemoteError +from ..coordination.coordination_state_contract_generated import LOCAL_COORDINATION_TODO_READ_REQUEST_SCHEMA +from ..effect_runtime import CANONICAL_AUTHORITY_READ_TIMEOUT_SECONDS, EffectRuntimeRemoteError, effect_runtime_result from ..work_items.task_lease import inspect_task_lease, release_task_lease _AUTHORITY_ERRORS = (ValueError, OSError, RuntimeError, EffectRuntimeRemoteError) @@ -26,9 +27,9 @@ def obligation(service, row): """The execution identity canonical authority must be asked about, or None. - `None` only when canonical authority proves that no delegation lease can - exist: the Goal's local authority is not promoted, which is also when - `_acquire_delegation_lease` acquires none. Otherwise the canonical lease + `None` only when the Goal is unpromoted and the canonical reader proves + its store absent. A missing cutover marker alone is not that proof: it + may have been lost after acquisition. Otherwise the canonical lease decides, whatever the operation recorded: a native claim commits before its annotation is saved, and an empty, malformed or stale `required: false` annotation is no more proof than a missing one. A recorded epoch still @@ -40,6 +41,18 @@ def obligation(service, row): if not local_authority_is_promoted(runtime_root=service.root, goal_id=service.goal_id): if required: raise ValueError("canonical authority disappeared under a required delegation lease") + # Use the existing provider-first reader, not a second Python test of + # provider paths or the operation's optional lease annotation. A loaded + # snapshot (even with this Todo missing) cannot qualify as never promoted. + snapshot = effect_runtime_result("coordination.local_authority.todo_read", { + "schema_version": LOCAL_COORDINATION_TODO_READ_REQUEST_SCHEMA, + "runtime_root": str(service.root), "goal_id": service.goal_id, + "todo_id": row["identity"]["binding"]["todo_id"], + }, timeout=CANONICAL_AUTHORITY_READ_TIMEOUT_SECONDS) + if (snapshot.get("status") != "missing" or snapshot.get("provider_revision") is not None + or snapshot.get("decision_read_from_provider") is not True + or snapshot.get("legacy_fallback_used") is not False): + raise ValueError("canonical authority absence unproven; reconcile the original authority route") return None acquired = recorded.get("lease") if required and isinstance(recorded.get("lease"), dict) else {} return {"idempotency_key": service._turn_instance_id(row), From c65f9248c69aec60d0fd5086c8ce7e32ba8bb9f9 Mon Sep 17 00:00:00 2001 From: song Date: Sun, 4 Oct 2026 16:02:59 +0800 Subject: [PATCH 52/54] test: retain stop lease obligation after authority marker loss Signed-off-by: song --- tests/test_delegation_stop_recovery.py | 43 ++++++++++++++++++++++++++ 1 file changed, 43 insertions(+) diff --git a/tests/test_delegation_stop_recovery.py b/tests/test_delegation_stop_recovery.py index 59d886e384..b28825e8d4 100644 --- a/tests/test_delegation_stop_recovery.py +++ b/tests/test_delegation_stop_recovery.py @@ -17,6 +17,49 @@ from loopx.file_lock import exclusive_file_lock, lock_holder_host_label +def test_missing_cutover_marker_does_not_erase_an_unannotated_lease(service, monkeypatch): + from test_delegation_lease_lifetime import inspect, prepare_lease + from loopx.control_plane.collaboration.inbox import _write + from loopx.control_plane.coordination.legacy_writer_fence import legacy_coordination_writer_fence_path + + root, runner = service + operation = "authority-evidence-loss" + with monkeypatch.context() as setup: + original_lease = prepare_lease(root, runner, setup, ttl=None, operation_id=operation) + path = runner.path(operation) + row = _read(path) + row.pop("task_lease") # Acquisition committed before its annotation survived. + _write(path, row) + fence = legacy_coordination_writer_fence_path(runtime_root=runner.root, goal_id=runner.goal_id) + original_fence = fence.read_bytes() + fence.unlink() + try: + receipt = runner.stop(operation, execute=True) + assert receipt["phase"] == "acknowledged", receipt + assert receipt["stop"]["lease"]["state"] == "obligation_unproven", receipt + assert not fence.exists(), "stop must not recreate authority evidence" + finally: + fence.write_bytes(original_fence) + held = inspect(runner) + assert held["active"] and held["lease"]["idempotency_key"] == original_lease["idempotency_key"] + recovered = runner.stop(operation, execute=True) + assert recovered["phase"] == "settled", recovered + assert recovered["stop"]["stop_id"] == receipt["stop"]["stop_id"] + assert recovered["stop"]["lease"]["state"] == "released" + assert inspect(runner)["lease"]["status"] == "released" + + +def test_never_promoted_authority_needs_no_delegation_lease(service): + from types import SimpleNamespace + + root, runner = service + unused_runtime = root / "never-promoted" + observer = SimpleNamespace(root=unused_runtime, goal_id=runner.goal_id) + row = {"identity": {"binding": {"todo_id": "todo_analyst-initial"}}} + assert stop_lease.obligation(observer, row) is None + assert not unused_runtime.exists(), "absence inspection must not create an authority" + + def recoverable_boundary(service, monkeypatch): root, runner = service # The Host adopts and supplies an artifact, but only delegation publishes From a7a48d7a34be279ec34c2d20d85883accbbe2d70 Mon Sep 17 00:00:00 2001 From: song Date: Sun, 4 Oct 2026 16:02:59 +0800 Subject: [PATCH 53/54] docs: clarify stop recovery after authority evidence loss Signed-off-by: song --- docs/reference/local-delegation.md | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/docs/reference/local-delegation.md b/docs/reference/local-delegation.md index 4eaf97b08f..b6fc778db5 100644 --- a/docs/reference/local-delegation.md +++ b/docs/reference/local-delegation.md @@ -455,7 +455,10 @@ execution identity (owner, execution key, any recorded epoch and the current version). The operation's `task_lease` annotation is only a hint: a missing, empty, stale `required: false` or malformed annotation never proves that nothing was owed, and a lease another execution holds is never released. The -receipt's `lease.state` is `released`, `not_owed`, `release_unproven` or +absence of the cutover marker also needs a provider-first read proving the +canonical store absent. An existing or unreadable store keeps the obligation +unproven until the original authority route is restored; stop never rebuilds it. +The receipt's `lease.state` is `released`, `not_owed`, `release_unproven` or `obligation_unproven`. A release that failed is retried under the stop's own lock on the next explicit `stop`, so it never becomes a `settled` receipt that leaves the member's Todo blocked until the lease TTL; @@ -511,7 +514,9 @@ worker 释放(只读 lane,从不获取)、该 Turn 启动的原生 host `phase` 才是 `settled`。是否需要释放由 canonical 权威按该 operation 自己的执行身份 (owner、执行 key、已记录的 epoch 与当前版本)判定;operation 的 `task_lease` 注解只是线索,缺失、为空、过期的 `required: false` 或畸形注解都不能证明无需释放, -其他执行持有的租约也绝不会被释放。回执的 `lease.state` 为 `released`、`not_owed`、 +其他执行持有的租约也绝不会被释放。切换标记缺失时,还必须由既有 provider-first +读取证明 canonical store 不存在;已有或不可读的 store 使义务保持未证明,直到原 +authority 路径恢复,stop 不会重建它。回执的 `lease.state` 为 `released`、`not_owed`、 `release_unproven` 或 `obligation_unproven`。释放失败会在下一次显式调用 `stop` 时于其锁下重试, 因此不会产生一份「已结算」却让成员 Todo 被租约阻塞到 TTL 的回执;在释放得到证明前,停止保持 `acknowledged`,原因为 `required_lease_release_unproven`;权威无法读取时同样保持打开,原因为 From f50e65192a99d9a288aa5765a11002469b5867e9 Mon Sep 17 00:00:00 2001 From: song Date: Sun, 4 Oct 2026 16:18:49 +0800 Subject: [PATCH 54/54] fix: ignore malformed optional lease generation hints Signed-off-by: song --- docs/reference/local-delegation.md | 4 ++-- loopx/control_plane/collaboration/delegation_stop_lease.py | 7 +++++-- tests/test_delegation_lease_lifetime.py | 3 ++- 3 files changed, 9 insertions(+), 5 deletions(-) diff --git a/docs/reference/local-delegation.md b/docs/reference/local-delegation.md index b6fc778db5..dab60c3c88 100644 --- a/docs/reference/local-delegation.md +++ b/docs/reference/local-delegation.md @@ -451,7 +451,7 @@ released by the stopped worker (the lane is read, never taken), the native host the Turn launched has exited together with every process in its group, and the hard task lease that execution may hold is resolved: released, or proven not owed. Canonical authority decides, for the operation's own -execution identity (owner, execution key, any recorded epoch and the current +execution identity (owner, execution key, any valid recorded epoch and the current version). The operation's `task_lease` annotation is only a hint: a missing, empty, stale `required: false` or malformed annotation never proves that nothing was owed, and a lease another execution holds is never released. The @@ -512,7 +512,7 @@ worker 不会被发信号,它在下一个检查点或下一次写记录时发 worker 释放(只读 lane,从不获取)、该 Turn 启动的原生 host 及其进程组内所有进程 都已退出,且该执行可能持有的硬任务租约已经处理(已释放,或被证明无需释放)时, `phase` 才是 `settled`。是否需要释放由 canonical 权威按该 operation 自己的执行身份 -(owner、执行 key、已记录的 epoch 与当前版本)判定;operation 的 `task_lease` +(owner、执行 key、已记录的有效 epoch 与当前版本)判定;operation 的 `task_lease` 注解只是线索,缺失、为空、过期的 `required: false` 或畸形注解都不能证明无需释放, 其他执行持有的租约也绝不会被释放。切换标记缺失时,还必须由既有 provider-first 读取证明 canonical store 不存在;已有或不可读的 store 使义务保持未证明,直到原 diff --git a/loopx/control_plane/collaboration/delegation_stop_lease.py b/loopx/control_plane/collaboration/delegation_stop_lease.py index 8a0190c1e4..0189aca6bd 100644 --- a/loopx/control_plane/collaboration/delegation_stop_lease.py +++ b/loopx/control_plane/collaboration/delegation_stop_lease.py @@ -32,7 +32,7 @@ def obligation(service, row): may have been lost after acquisition. Otherwise the canonical lease decides, whatever the operation recorded: a native claim commits before its annotation is saved, and an empty, malformed or stale `required: false` - annotation is no more proof than a missing one. A recorded epoch still + annotation is no more proof than a missing one. A valid recorded epoch still fences the release against another generation under the same key. """ @@ -55,8 +55,11 @@ def obligation(service, row): raise ValueError("canonical authority absence unproven; reconcile the original authority route") return None acquired = recorded.get("lease") if required and isinstance(recorded.get("lease"), dict) else {} + epoch = acquired.get("lease_epoch") + # Only a valid generation can fence another generation. Malformed optional + # evidence is no stronger than its absence; canonical identity still decides. return {"idempotency_key": service._turn_instance_id(row), - "lease_epoch": acquired.get("lease_epoch")} + "lease_epoch": epoch if type(epoch) is int and epoch > 0 else None} def release(service, row, binding): diff --git a/tests/test_delegation_lease_lifetime.py b/tests/test_delegation_lease_lifetime.py index 73fc473ec6..850fc39950 100644 --- a/tests/test_delegation_lease_lifetime.py +++ b/tests/test_delegation_lease_lifetime.py @@ -359,7 +359,8 @@ def settled_stop(runner, operation_id): # The operation's annotation is a hint; canonical authority decides what is owed. STALE_ANNOTATIONS = {"empty": {}, "stale_not_required": {"required": False, "handoff_mode": "legacy"}, - "malformed": {"required": "yes", "lease": [1]}} + "malformed": {"required": "yes", "lease": [1]}, + "invalid_epoch": {"required": True, "lease": {"lease_epoch": "unavailable"}}} @pytest.mark.parametrize("window", ["annotated", "renewed", "unannotated", *STALE_ANNOTATIONS])