diff --git a/apps/presentation/dashboard/src/data/use-typed-action-readback.ts b/apps/presentation/dashboard/src/data/use-typed-action-readback.ts index 443251b808..e7bc6d1b3f 100644 --- a/apps/presentation/dashboard/src/data/use-typed-action-readback.ts +++ b/apps/presentation/dashboard/src/data/use-typed-action-readback.ts @@ -1,12 +1,14 @@ -import { useQuery } from "@tanstack/react-query"; -import { listTypedActions } from "./chat"; +import { useQuery, useQueryClient } from "@tanstack/react-query"; +import { listTypedActions, type TypedActionProposal } from "./chat"; /** Visible workspace readback only: no confirmation, dispatch or effect owner. * Query keys fence scope changes; React Query serializes same-key requests, * cancels superseded reads and suspends background interval polling. */ export function useTypedActionReadback(readOnly: boolean, goalId: string | null | undefined) { - return useQuery({ - queryKey: ["typed-action-readback", goalId ?? "manager"], + const client = useQueryClient(); + const queryKey = ["typed-action-readback", goalId ?? "manager"]; + const query = useQuery({ + queryKey, queryFn: ({ signal }) => listTypedActions(goalId ? { goalId } : {}, AbortSignal.any([signal, AbortSignal.timeout(10_000)])), enabled: !readOnly, @@ -15,4 +17,18 @@ export function useTypedActionReadback(readOnly: boolean, goalId: string | null refetchOnWindowFocus: "always", retry: false, }); + return { + ...query, + async acceptProposal(proposal: TypedActionProposal, replacedId?: string) { + if (readOnly) return; + // The validated mutation response is already canonical readback. Fence + // older reads before publishing it, so a cached preview or delayed poll + // cannot replace an acknowledged receipt during a Goal status refresh. + await client.cancelQueries({ queryKey, exact: true }); + client.setQueryData(queryKey, current => current + ? [proposal, ...current.filter(row => row.proposal_id !== proposal.proposal_id + && row.proposal_id !== replacedId)] + : current); + }, + }; } diff --git a/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-page.tsx b/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-page.tsx index 8f2bf933e5..af0f6245d2 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-page.tsx +++ b/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-page.tsx @@ -1481,6 +1481,7 @@ export function PersonalWorkspacePage({ ))) { throw new ChatApiError(t("actionReview.targetChanged"), { error_code: "action_response_mismatch" }); } + await actionReadback.acceptProposal(result.proposal); const applied = workspaceProposal(result.proposal, t); setProposals((current) => ({ ...current, [proposal.previewId]: applied })); if (showDrawer) setSelection({ item: applied, kind: "proposal" }); @@ -1596,14 +1597,18 @@ export function PersonalWorkspacePage({ }); try { callbacks.onCancelProposal?.(proposal); - if (!callbacks.onCancelProposal) await cancelTypedAction(proposal.previewId); + if (!callbacks.onCancelProposal) { + await actionReadback.acceptProposal(await cancelTypedAction(proposal.previewId)); + } } catch (error) { setProposals((current) => ({ ...current, [proposal.previewId]: proposal })); setActionFeedback(t("feedback.cancelFailed", { error: error instanceof Error ? error.message : String(error) })); } }, onTransitionProposal: async (proposal, transition) => { - const transitioned = workspaceProposal(await transitionTypedAction(proposal.previewId, transition), t); + const result = await transitionTypedAction(proposal.previewId, transition); + await actionReadback.acceptProposal(result, transition === "regenerate" ? proposal.previewId : undefined); + const transitioned = workspaceProposal(result, t); const managerOwned = managerSessionProposalIds.includes(proposal.previewId) || managerChannelProposalIds.includes(proposal.previewId); rememberSessionProposal(transitioned.previewId, managerOwned ? null : proposal.goalId ?? selectedGoalId); diff --git a/docs/reference/protocols/peer-agent-directory-and-observation-v0.md b/docs/reference/protocols/peer-agent-directory-and-observation-v0.md index 7ad3a211e1..0474f315d5 100644 --- a/docs/reference/protocols/peer-agent-directory-and-observation-v0.md +++ b/docs/reference/protocols/peer-agent-directory-and-observation-v0.md @@ -252,6 +252,45 @@ host after a lost submission response before repeating the host message. The receiver's `manager-inbox read`, decision and `report` receipts remain distinct from host submission and from each other. +### Real-host qualification + +`examples/peer-handoff-live-smoke.py` is an explicit opt-in qualification of the +existing Codex app-server adapter and the same request/return CLI. It creates +two synthetic host threads in the selected authenticated home, a disposable +Goal/registry/runtime, and a bounded artifact pinned to the checkout head and +SHA-256. It does not resume a user thread or create replacement child workers. + +From the source checkout: + +```bash +uv run --extra test python examples/peer-handoff-live-smoke.py --execute-real-host +``` + +On Windows select `--codex-bin codex.cmd` if the installed launcher needs it. +The command consumes model quota and leaves the host's own test-thread records +in that home; no authentication or session records are copied between homes. +The caller authorizes these test submissions separately from route resolution. +An unreachable historical binding keeps automatic selection ambiguous; an +explicit exact link pins the existing reviewer. The real receiver reads and +adopts the request, independently checks the artifact, and returns its exact +head/digest. A fresh requester process restores its own thread, reads the +result and acknowledges consumption. Retry recovers one request. Output contains +compact assertions, without thread links, local paths or raw conversations. + +The qualification exposed two Windows blockers in this journey: private request +hashes and lock/claim suffixes exceed `MAX_PATH`, and POSIX-only input flags +prevent artifact readback. Private store/lock I/O now addresses the same physical +files using Win32 extended paths; identities, lock exclusion and storage layout +remain unchanged. Regular input files use the platform's binary/nonblocking +flags. Focused regression checks cover mutual exclusion, release, artifact +readback and one request across repeated delivery/consumption. + +This qualifies this local owned-host request/adopt/return slice. It does not +qualify remote hosts, grant message permission to an arbitrary App task, transfer +a lease, or close the overall R2/R3 collaboration acceptance. Route previews +continue to report `host_delivery: not_attempted`; the smoke's explicit host +submission and receiver receipts are separate evidence. + ## Target Identity Pinning A bounded wait, or the readback that a delivery produced a turn, must be pinned diff --git a/examples/peer-handoff-live-smoke.py b/examples/peer-handoff-live-smoke.py new file mode 100644 index 0000000000..31c4d4a19e --- /dev/null +++ b/examples/peer-handoff-live-smoke.py @@ -0,0 +1,238 @@ +#!/usr/bin/env python3 +"""Opt-in real Codex peer handoff using only disposable LoopX state. + +Run from the source checkout with an authenticated Codex home. This creates two +synthetic host threads and consumes model quota; it never resumes a user thread. +Only compact assertions are printed. Host transcripts stay in the selected home. +""" +from __future__ import annotations + +import argparse +import hashlib +import json +import os +import subprocess +import sys +import tempfile +from pathlib import Path + +from loopx.chat_agent import CodexChatAgentSession +from loopx.control_plane.effect_runtime import restart_effect_runtime + + +def tool(name, description, properties=None): + fields = properties or {} + return {"name": name, "description": description, "inputSchema": { + "type": "object", "properties": fields, "required": list(fields), + "additionalProperties": False, + }} + + +def qualify(workspace: Path, *, codex_bin: str, codex_home: Path) -> dict: + """The host owns thread submission; LoopX owns request/read/return receipts.""" + runtime = workspace / "runtime" + registry = workspace / "registry.json" + head = subprocess.check_output( + ["git", "rev-parse", "HEAD"], text=True + ).strip() + artifact = workspace / "review-packet.json" + artifact.write_text(json.dumps({"head": head, "demand": 15, + "allocated": 13, "reserved": 2}), encoding="utf-8") + digest = hashlib.sha256(artifact.read_bytes()).hexdigest() + brief = workspace / "brief.json" + brief.write_text(json.dumps({ + "schema_version": "collaboration_brief_v0", + "purpose": "Independently review the synthetic allocation artifact", + "context": f"Review exactly checkout head {head}; do not substitute another peer.", + "constraints": ["No child Agents", "No shell commands", "No external writes"], + "inputs": [{"ref": artifact.name, "description": "Synthetic review packet", + "sha256": digest}], + "acceptance": ["Allocated plus reserved equals demand", "Head and digest match"], + "return_requirement": "Return the verdict, exact head and artifact digest to requester", + }), encoding="utf-8") + env = {**os.environ, "LOOPX_CODEX_HOMES": str(codex_home)} + + def cli(agent, action, *args, ok=True): + process = subprocess.run([sys.executable, "-m", "loopx.cli", + "--registry", str(registry), "--runtime-root", str(runtime), + "--format", "json", "manager-inbox", action, + "--goal-id", "peer-qualification", "--agent-id", agent, *args], + env=env, cwd=workspace, text=True, capture_output=True, timeout=45) + value = json.loads(process.stdout) + assert process.returncode == (0 if ok else 1), value + return value + + events, calls = [], [] + sessions = [] + + def observe(method, value): + if method == "item/completed": + item = value.get("item") or {} + events.append(item.get("type")) + + def start(agent, tools, *, resume=None): + session = CodexChatAgentSession.start( + codex_bin=codex_bin, codex_home=codex_home, work_dir=workspace, + goal_id="peer-qualification", objective=f"Synthetic {agent} qualification", + execution_mode=True, sandbox="read-only", resume_thread_id=resume, + dynamic_tools=tools, isolate_process_tree=True, hard_timeout_sec=180, + host_config={"features.multi_agent": False}, + ) + sessions.append(session) + return session + + review_tools = [ + tool("peer_review_read", "Read your scoped Inbox request and its exact artifact"), + tool("peer_review_adopt", "Adopt the request after reading it"), + tool("peer_review_report", "Return the independent review to its requester", { + "head": {"type": "string"}, "sha256": {"type": "string"}, + "verdict": {"type": "string", "enum": ["accept", "reject"]}, + }), + ] + requester_tools = [ + tool("peer_return_read", "Read the review returned to this requester"), + tool("peer_return_consume", "Acknowledge the returned result after reading it"), + ] + try: + reviewer = start("reviewer", review_tools) + requester = start("requester", requester_tools) + reviewer_id, requester_id = reviewer.thread_id, requester.thread_id + # These are new test sessions. Finishing a real Turn makes the production + # local-store observer readable without fabricating its SQLite records. + for session in (reviewer, requester): + session.send("Reply fixture-ready. Do not use tools or create child Agents.", + on_event=observe) + reviewer.close() + requester.close() + registry.write_text(json.dumps({"goals": [{"id": "peer-qualification", + "repo": str(workspace), "coordination": { + "registered_agents": ["requester", "reviewer"], + "thread_agent_bindings": [ + {"agent_id": "reviewer", "host_surface": "codex-app", + "thread_id": reviewer_id}, + {"agent_id": "reviewer", "host_surface": "codex-app", + "thread_id": "historical-unreachable-fixture"}, + {"agent_id": "requester", "host_surface": "codex-app", + "thread_id": requester_id}, + ], + }}]}), encoding="utf-8") + common = ("--peer-agent-id", "reviewer", "--operation-id", "exact-head-review", + "--brief-file", str(brief), "--require-host-route") + refused = cli("requester", "request", *common, ok=False) + assert "ambiguous" in refused["error"] + selected = (*common, "--peer-thread-link", f"codex://threads/{reviewer_id}") + sent = cli("requester", "request", *selected) + rid = sent["request_id"] + assert sent["host_delivery"]["status"] == "not_attempted" + assert sent["host_delivery"]["thread_id"] == reviewer_id + assert cli("requester", "request", *selected)["replayed"] + # Resume only the pinned test thread through the owning host. This is an + # explicitly authorized test submission, not authority inferred from a route. + reviewer = start("reviewer", review_tools, resume=reviewer_id) + + def review_handler(name, arguments, identity): + assert identity["thread_id"] == reviewer_id + calls.append(name) + if name == "peer_review_read": + inbox = cli("reviewer", "read") + row = next(item for item in inbox["items"] if item["request_id"] == rid) + assert row["brief"]["inputs"][0]["sha256"] == digest + assert hashlib.sha256(artifact.read_bytes()).hexdigest() == digest + return {"ok": True, "request": row, "artifact": json.loads(artifact.read_text()), + "sha256": digest} + if name == "peer_review_adopt": + assert "peer_review_read" in calls + return cli("reviewer", "acknowledge", "--request-id", rid, + "--decision", "adopt", "--reason", "Independent exact-head review") + if name == "peer_review_report": + assert "peer_review_adopt" in calls + assert arguments == {"head": head, "sha256": digest, "verdict": "accept"} + return cli("reviewer", "report", "--request-id", rid, + "--phase", "conclusion", "--reply-text", f"ACCEPT head={head} sha256={digest}") + raise ValueError("unsupported qualification tool") + + reviewer.bound_tool_handler = review_handler + reviewer.send(sent["host_delivery"]["message"] + + " Use peer_review_read, inspect the arithmetic independently, then " + "peer_review_adopt and peer_review_report. Use only these three tools.", + on_event=observe) + reviewer.close() + # A fresh process restores the requester's original identity for result + # consumption; neither retry nor restart creates another peer request. + replay = cli("requester", "request", *selected) + assert replay["replayed"] and replay["request_id"] == rid + requester = start("requester", requester_tools, resume=requester_id) + + def return_handler(name, arguments, identity): + assert identity["thread_id"] == requester_id + calls.append(name) + if name == "peer_return_read": + result = cli("requester", "read") + item = next(item for item in result["peer_returns"]["items"] + if item["request_id"] == rid) + assert item["text"] == f"ACCEPT head={head} sha256={digest}" + return {"ok": True, "result": item} + if name == "peer_return_consume": + assert "peer_return_read" in calls + return cli("requester", "acknowledge-return", "--request-id", rid) + raise ValueError("unsupported qualification tool") + + requester.bound_tool_handler = return_handler + requester.send("The existing peer has returned its review. Use peer_return_read " + "then peer_return_consume; summarize its exact head and digest. " + "Use only these two tools.", on_event=observe) + required = {entry["name"] for entry in review_tools + requester_tools} + assert required <= set(calls), "real peers did not complete the exchange" + assert not cli("requester", "read").get("peer_returns", {}).get("items", []) + assert not {"collabAgentToolCall", "commandExecution"} & set(events) + status = cli("reviewer", "status", "--request-id", rid) + return {"ok": True, "host": "codex_app_server", "peers": 2, + "exact_thread_resumed": True, "request_replayed": True, + "receiver_adopted": True, "result_returned_and_consumed": True, + "replacement_workers": 0, "host_delivery_preview": "not_attempted", + "tracked_requests": len(status["rows"])} + finally: + for session in sessions: + session.close() + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--execute-real-host", action="store_true", + help="Authorize two synthetic Codex sessions and model usage") + parser.add_argument("--codex-bin", default="codex") + parser.add_argument("--codex-home", type=Path, + default=Path(os.environ.get("CODEX_HOME") or "~/.codex").expanduser()) + args = parser.parse_args() + if not args.execute_real_host: + parser.error("--execute-real-host is required; this smoke uses a real authenticated host") + with tempfile.TemporaryDirectory(prefix="lxp-") as folder: + # CLI reads start a reusable typed runtime whose Windows cwd keeps the + # workspace open. Give this smoke its own locator, then stop that owner + # before deleting the workspace; never stop a shared user's runtime. + runtime_temp = Path(folder) / "tmp" + runtime_temp.mkdir() + previous_temp = tempfile.tempdir + previous_env = {key: os.environ.get(key) for key in ("TMPDIR", "TEMP", "TMP")} + try: + tempfile.tempdir = str(runtime_temp) + os.environ.update({key: str(runtime_temp) for key in previous_env}) + try: + result = qualify(Path(folder), codex_bin=args.codex_bin, + codex_home=args.codex_home.resolve()) + finally: + restart = restart_effect_runtime() + if restart["status"] == "shutdown_pending": + raise RuntimeError("isolated typed runtime shutdown did not complete") + finally: + tempfile.tempdir = previous_temp + for key, value in previous_env.items(): + if value is None: + os.environ.pop(key, None) + else: + os.environ[key] = value + print(json.dumps(result, sort_keys=True)) + + +if __name__ == "__main__": + main() diff --git a/examples/personal-workspace-browser/steward-journey.mjs b/examples/personal-workspace-browser/steward-journey.mjs index a95079dbfd..75acad4a96 100644 --- a/examples/personal-workspace-browser/steward-journey.mjs +++ b/examples/personal-workspace-browser/steward-journey.mjs @@ -254,6 +254,22 @@ export const stewardJourneyScenario = { animations: "disabled", }); + // Hold a read started before confirmation. A later status refresh must + // not replace the acknowledged result with the cached preview, and the + // delayed preview must not reopen confirmation when its response lands. + let releaseReadback; + const heldReadback = new Promise(resolveRead => { releaseReadback = resolveRead; }); + const pendingRead = page.waitForRequest(request => request.method() === "GET" + && new URL(request.url()).pathname === "/api/actions", { timeout: 10_000 }); + const actionListRoute = /\/api\/actions(?:\?.*)?$/; + const holdReadback = async route => { + await heldReadback; + await route.fulfill({ contentType: "application/json", status: 200, + json: { ok: true, schema_version: "loopx_chat_action_list_v1", proposals: [teamPlanProposal()] } }); + }; + await page.route(actionListRoute, holdReadback); + await pendingRead; + // Beat 3: confirm, and record what the workspace actually reports after // the canonical owner ran. const confirm = drawer.getByRole("button", { name: "确认分配", exact: true }); @@ -269,16 +285,28 @@ export const stewardJourneyScenario = { check(api.durableWriteCount === 1, "the confirmed apply performed exactly one durable write"); const applied = drawer.getByRole("heading", { name: "已分配 1 项,1 项待安排", exact: true }); await applied.waitFor({ state: "visible", timeout: 15_000 }); + const outcomeText = await applied.innerText(); const resultText = await drawer.locator(".personal-team-plan-result").innerText(); const assignmentVisible = resultText.includes("agent-backend") && resultText.includes(READY_TODO); const gapVisible = resultText.includes("agent-reviewer") && resultText.includes(GAP_TODO) && resultText.includes("待安排 · 尚未加入此目标"); check(assignmentVisible && gapVisible, "the result names assigned work and pending work with its reason"); check(await confirm.count() === 0, "the completed result removes its confirmation control"); + await page.getByRole("button", { name: "刷新状态", exact: true }).click(); + await page.getByText("刚刚更新", { exact: true }).waitFor({ state: "visible" }); + check(await applied.count() === 1 && await confirm.count() === 0, + "a Goal status refresh retains the acknowledged assignment result"); + releaseReadback(); + await page.unroute(actionListRoute, holdReadback); + await page.waitForTimeout(200); + check(await applied.count() === 1 && await confirm.count() === 0, + "a read started before apply cannot reopen assignment confirmation"); + check(api.actionApplies.length === 1 && api.durableWriteCount === 1, + "readback and status refresh never reapply the confirmed assignment"); record("3-confirm", { applies: api.actionApplies.length, durable_writes: api.durableWriteCount, - outcome_text: await applied.innerText(), + outcome_text: outcomeText, outcome_fidelity: "assigned task and pending task with reason; execution remains unverified", }); gaps.push({ diff --git a/loopx/control_plane/collaboration/inbox.py b/loopx/control_plane/collaboration/inbox.py index 81ea9f9d27..a3191b75b1 100644 --- a/loopx/control_plane/collaboration/inbox.py +++ b/loopx/control_plane/collaboration/inbox.py @@ -16,6 +16,7 @@ from pathlib import Path from typing import TYPE_CHECKING, Any from ...file_lock import exclusive_file_lock +from ..runtime.file_paths import windows_extended_path from ..content_digest import BARE_SHA256_PATTERN, ENVELOPED_SHA256_PATTERN from ..todos.contract import TODO_ID_PATTERN @@ -40,7 +41,9 @@ def _hash(value: Any) -> str: def _root(runtime_root: Path) -> Path: """Retain the shipped storage address; Agent topology is not encoded in it.""" - return runtime_root / ".local" / "manager-context" + # Full request hashes plus lock sidecars can exceed MAX_PATH even in an + # ordinary workspace. Keep extended syntax inside the private store. + return windows_extended_path(runtime_root / ".local" / "manager-context") def _write(path: Path, value: dict) -> None: diff --git a/loopx/control_plane/collaboration/peers.py b/loopx/control_plane/collaboration/peers.py index e0cebe3d48..5589c1a39d 100644 --- a/loopx/control_plane/collaboration/peers.py +++ b/loopx/control_plane/collaboration/peers.py @@ -531,7 +531,7 @@ def input_readiness( # Nonblocking open plus fstat prevents a FIFO/device reference # from hanging the worker's entire Inbox read. with os.fdopen( - os.open(path, os.O_RDONLY | os.O_NONBLOCK), "rb" + os.open(path, os.O_RDONLY | getattr(os, "O_NONBLOCK", 0) | getattr(os, "O_BINARY", 0)), "rb" ) as stream: if not stat.S_ISREG(os.fstat(stream.fileno()).st_mode): raise OSError("input is not a regular file") diff --git a/loopx/control_plane/runtime/file_paths.py b/loopx/control_plane/runtime/file_paths.py new file mode 100644 index 0000000000..2bdcbbe821 --- /dev/null +++ b/loopx/control_plane/runtime/file_paths.py @@ -0,0 +1,18 @@ +"""Native filesystem addresses without Goal discovery or routing dependencies.""" + +from __future__ import annotations + +import os +from pathlib import Path + + +def windows_extended_path(path: Path) -> Path: + """Address the same Windows file beyond MAX_PATH; leave other hosts alone.""" + if os.name != "nt": + return path + address = os.path.abspath(path) + if address.startswith("\\\\?\\"): + return Path(address) + if address.startswith("\\\\"): + return Path("\\\\?\\UNC\\" + address[2:]) + return Path("\\\\?\\" + address) diff --git a/loopx/file_lock.py b/loopx/file_lock.py index 7ffa396822..fd5838575e 100644 --- a/loopx/file_lock.py +++ b/loopx/file_lock.py @@ -18,6 +18,8 @@ from typing import Any, Iterator, TextIO from uuid import uuid4 +from .control_plane.runtime.file_paths import windows_extended_path + try: # pragma: no cover - exercised on POSIX hosts in integration smokes. fcntl: Any = importlib.import_module("fcntl") except ImportError: # pragma: no cover @@ -121,7 +123,7 @@ def _policy(value: LockAcquisitionPolicy | str) -> LockAcquisitionPolicy: def _lock_path(path: Path) -> Path: - return path.with_name(f"{path.name}.lock") + return windows_extended_path(path.with_name(f"{path.name}.lock")) def _open_lock_descriptor(path: Path, *, flags: int) -> int: @@ -159,8 +161,14 @@ def lock_incident_path(path: Path) -> Path: def _lock_id(path: Path) -> str: - resolved = str(path.expanduser().resolve(strict=False)).encode("utf-8") - return hashlib.sha256(resolved).hexdigest()[:16] + address = str(path.expanduser().resolve(strict=False)) + if os.name == "nt": + # The address syntax must not create a second diagnostic lock identity. + if address.startswith("\\\\?\\UNC\\"): + address = "\\\\" + address[8:] + elif address.startswith("\\\\?\\"): + address = address[4:] + return hashlib.sha256(address.encode("utf-8")).hexdigest()[:16] def _identity( @@ -602,7 +610,7 @@ def try_exclusive_file_lock( def _effect_mutation_lock_path(path: Path) -> Path: - return Path(f"{path}{EFFECT_MUTATION_LOCK_SUFFIX}") + return windows_extended_path(Path(f"{path}{EFFECT_MUTATION_LOCK_SUFFIX}")) def _effect_mutation_claim_path(path: Path, token: str) -> Path: diff --git a/skills/loopx-self-repair/references/repair-patterns.md b/skills/loopx-self-repair/references/repair-patterns.md index af523e9445..70a9deb29c 100644 --- a/skills/loopx-self-repair/references/repair-patterns.md +++ b/skills/loopx-self-repair/references/repair-patterns.md @@ -24,6 +24,7 @@ teaches a reusable control-plane lesson. | `delegation_runtime_discovery_split` | A coordinator knows a child-count or model preference but cannot discover requester-authorized managed routes, so a direct SDK experiment is mistaken for formal LoopX delegation or the task stays unnecessarily serial. | Goal orchestration readback, requester-scoped binding directory, Turn host/profile readiness, operation journal, signed `before_plan`/`before_delegate`/`after_delegate_result` context, and parent artifact validation. | Execution grants and runtime facts existed only behind the dispatch CLI/MCP while planning consumed a separate prompt or model-preference snapshot; adding provider names to one skill would create another scheduler/config owner. | Store only an ignored Goal-local pointer to the existing operator binding file, project bounded public-safe requester routes through the generic capability context, and keep runtime readiness separate from route selection and task adoption. Recheck runtime/model/budget at dispatch, forbid silent substitution, and reconcile original operation receipts after return. Do not require every heartbeat to use every route or copy credentials/host arguments into registry, frontend, Lark or prompts. | | `goal_runtime_shadows_machine_credential` | A configured machine appears credential-less in another Goal or interpreter. | Machine credential status, Goal runtime root, launching interpreter SDK probe, Turn plan and dispatch. | Goal state location or process environment was mistaken for the machine authentication owner. | Resolve the canonical machine credential at planning and dispatch; keep SDK readiness interpreter-scoped and assignments requester/Goal-scoped. Verify two Goal roots, conflicting Goal-local stores, invalid machine-store refusal and explicit host selection. Never copy a credential or another Goal's grants to make readiness green. | | `budget_metric_overfitting` | A budget failure triggers automatic expansion, or mechanical compaction that removes useful semantics or breaks consumers. | Owning limit, matched base/head measurements, consumer/caller contract, original failure and revised validation. | A regression metric became the objective; historical ceilings or green tests replaced semantic judgment. | Follow the [budget decision guide](../../../docs/development/testing-and-quality.md#budget-failure-decisions), compare true redundancy, compatibility cost and justified headroom, and repair the existing contract/tests and review evidence. Preserve hard limits and frozen qualification results. | +| `validation_fixture_owner_drift` | CI repeatedly fails after a rebase because fixtures retain an old catalog count, omit current caller context or dispatch without the owning claim. | The current source contract, exact failing assertion, base/head entrypoint and durable state readback. | Validation encoded a dated snapshot or skipped current admission instead of preparing the operation it intended to challenge. | Reuse the shipped catalog and package metadata, explicitly isolate runtime routes, and acquire the real queue claim before dispatch. Retain the original negative invariant and exit/readback assertions. Do not weaken routing, execution admission or production boundaries to make stale fixtures pass. | | `skill_import_recreation` | Duplicate LoopX skills return after successful cleanup; imported entries display a fallback brand casing. | Compare installed files and metadata with source-host skills; correlate file creation times with structured host import receipts. | A later external-host import recreates command facades in another discovered root, omits display metadata, and bypasses installer reconciliation. | Attribute the writer from import receipts without guessing the human initiator; exclude already-installed LoopX skills from later imports and rerun managed reconciliation. Ensure standalone workflow entry installation writes Codex metadata, previews missing metadata repair, and records the full installed tree. Preserve user metadata and exact-host invocation behavior. | | `skill_discovery_split_ownership` | Duplicate skill names, conflicting PR-review routes, or canonical and legacy aliases appear together. | Enumerate discovered roots, resolve directory symlinks, compare skill hashes, managed markers, install receipts, and generated metadata. | Workflow and command installers wrote independently to overlapping host roots; dedupe was optional, omitted the bare entry name, or retired copies without proving a replacement. | Repair the shared installer reconciliation and every active installation path; preserve user changes and rich workflows, retire managed aliases from the Codex picker, test repeated installs and custom profiles, then verify a fresh host catalog. Do not treat deleting one visible duplicate or changing invocation policy as a durable fix. | | `skill_discovery_scope_eligibility_conflation` | Project delivery rejects a reusable workflow, and adding a project marker unexpectedly removes connection, repair, or review instructions from global installation. | Canonical scope markers, default shell/CLI and packaged install output, project-copy readback, and doctor required workflows. | One marker was treated as both exclusive project eligibility and default discovery; content richness was mistaken for project authority. | Declare reusable workflows global and capability-local workflows project; accept both explicit declarations for project copies while rejecting missing/unknown markers. Preserve global command routes, existing activation gates, and packaged resource parity. Never repair project delivery by hiding bootstrap or repair instructions from unconnected projects. | @@ -215,7 +216,7 @@ teaches a reusable control-plane lesson. | `candidate_preflight_negative_evidence_gap` | Issue-fix candidate screening accepts empty PR evidence and starts implementation even though the caller used a capped aggregate index or did not prove a direct all-state search. | Candidate preflight input, issue-specific numeric and semantic query receipts, truncation/completeness metadata, current issue body and comments. | Capability admission treated key presence or a naked empty list as proof that prior work was absent. | Keep provider queries outside the LoopX core, but require issue-specific complete, non-truncated receipts before a negative result can yield `proceed`; aggregate indexes remain candidate generators only. Do not add capability fields to generic Todos. | | `commit_hygiene_drift` | Broad commit includes temporary smokes, raw logs, local state, or unrelated docs. | `git status`, `git diff --stat`, `git ls-files --others --exclude-standard`, AGENTS.md. | Worktree was staged by chronology rather than reviewer logic. | Use explicit pathspecs, split commits, keep only durable smokes, and update AGENTS/skill if the failure mode recurs. | | `migration_terminal_receipt_replay_gap` | A migrated transaction succeeds once, then an idempotent legacy retry is rejected as receipt corruption or times out on a lock that the first response lost. | Versioned operation receipt, persisted state, lock owner/token, exact caller retry identity, pre-migration replay behavior, and focused native plus compatibility tests. | The new owner modeled only held/committed receipts and treated a closed no-op receipt as incomplete, or generated retry identity inside the callee after the caller's retry boundary. | Model terminal no-op receipts separately from held authority proofs, replay them without retired private tokens, and generate one stable operation id outside any transport retry while minting a new id for independent calls. Cover direct native replay and the compatibility adapter's response-loss boundary. | -| `typed_kernel_import_surface_drift` | Focused tests and premerge canaries pass, but CI mypy starts reporting unrelated errors in a previously unchecked module after a small typed-kernel change. | Exact mypy entry files, the new import edge, base/head module diffs, and whether the imported module owns the new behavior or is only a type source. | A directly checked kernel module imported a broad runtime owner for one enum or provider-selection helper, silently expanding strict mypy traversal and placing single-caller policy in the wrong bounded context. | Keep provider selection in its nearest caller, leave provider-neutral builders in the typed kernel, and avoid importing a broad runtime owner only to expose one caller-local decision. Run the exact CI mypy command after changing imports in configured kernel files. | +| `typed_kernel_import_surface_drift` | Focused tests and premerge canaries pass, but CI mypy starts reporting unrelated errors in a previously unchecked module after a small typed-kernel change. | Exact mypy entry files, the new import edge, base/head module diffs, and whether the imported module owns the new behavior or is only a type source. | A checked kernel dependency imported a broad runtime or Goal-routing owner for one enum, provider decision or filesystem helper, expanding strict mypy traversal beyond the owning contract. | Keep caller policy with its caller and shared filesystem helpers in a dependency-light runtime module. Validate behavior parity and the exact CI mypy command after changing kernel dependencies, using the same source installation and import environment; an extra PYTHONPATH can hide the imported-module failure. | | `packaged_surface_completion_illusion` | A development route or API-mocked browser smoke passes while the packaged product still serves an older UI, loses local history after refresh, hides every healthy Agent, or records important contradictions only as non-failing observations. | packaged entry title and first viewport, emitted bundle identity, real loopback capabilities, service-restart history, Last-Event-ID replay, mobile navigation close behavior, and the acceptance report's untested/observation fields. | UI completion was inferred from component tests or a Vite route; the release asset, launcher environment, durable local session path, and hard browser assertions evolved separately. | Make packaged-route parity a blocking acceptance contract. Build and launch the real bundle with a controlled executable path, convert semantic observations into assertions, persist status-only and Agent-backed messages independently, test refresh plus restart recovery, and require every design criterion to be PASS or explicitly gated before claiming the surface complete. | | `dashboard_status_source_generation_gap` | Switching between local and SSH sources leaves the old Goal list visible, or the selector and route oscillate between two sources after a slow tunnel ensure or status response. | User selection order, tunnel-ensure start/finish order, status fetch start/finish order, requested and loaded URLs, projection revision, route commits, and the rendered Goal identity. | Request freshness starts only after SSH preparation, a foreground request captures its revision before advancing it, or an older route commit clears a newer pending selection. | Fence the complete selection transaction—from user intent through tunnel ensure, fetch, payload commit, and route commit—with one monotonic selection generation plus a projection revision. Let only the current generation clear pending state; reject background responses whose target is no longer the committed source. Cover delayed A/B fetches and delayed SSH-then-local selection with a focused browser smoke that asserts payload, selector, and route remain aligned. | | `docs_surface_sprawl` | Root docs become hard for contributors to navigate; research/drafts/history compete with stable contracts. | `docs/README.md`, root docs count, docs governance smoke. | Documentation lacks audience and lifecycle ownership. | Move archive/outreach/research/reference material into indexed subdirs and keep new docs linked from an index. | @@ -306,3 +307,4 @@ raw logs and private traces stay in ignored local paths. ## Review ignores a Goal's CI waiting configuration Symptom: a managed review ignores `pull_request_review.wait_for_ci=false` and waits on remote CI after local evidence is complete. Read the named Goal's configuration and pass `--goal-id` through both review and readiness. Repair the capability transport, packet, and readiness owner together; retain required local checks, exact-head binding, valid approval, unresolved-thread rejection, and merge authority. Never change the global default to repair one Goal. Validate both default-on and explicit-off paths through the production CLI and the shared configuration editor. +| `windows_peer_handoff_path_and_input_gap` | A real Windows peer request fails while creating a lock/holder/claim, or Inbox input hashing raises an unsupported flag error. | Exact request identity, private path length, real CLI read/adopt/report/consume and lock-release readback in disposable state. | Hash-based request directories plus lock suffixes exceed MAX_PATH; POSIX-only input flags are assumed on Windows. | Use the same physical private-store and lock addresses through Win32 extended syntax, preserve full hashes and lock identity, and select supported binary/nonblocking flags. Prove one request across restart, mutual exclusion and release while the owner process remains alive. Keep raw host records and local paths out of public evidence. | diff --git a/tests/control_plane/test_native_child_replan_guard_cli.py b/tests/control_plane/test_native_child_replan_guard_cli.py index 465e910cad..6105df8a8f 100644 --- a/tests/control_plane/test_native_child_replan_guard_cli.py +++ b/tests/control_plane/test_native_child_replan_guard_cli.py @@ -34,7 +34,9 @@ def _fixture(tmp_path: Path, monkeypatch: pytest.MonkeyPatch, provider: str, tod state.write_text("---\nstatus: active\n---\n\n# Synthetic Goal\n\n## Agent Todo\n" + ( "\n- [ ] [P1] Validate the original source.\n" f" \n" + f"claimed_by={AGENT} action_kind=validate validation_command=pytest " + "continuation_policy=same_agent_non_delivery " + "required_capabilities=shell%2Cfilesystem_read -->\n" if todo_bound else "" )) index = runtime / "goals" / GOAL / "runs" / "index.jsonl" diff --git a/tests/test_chat_goal_configuration_api.py b/tests/test_chat_goal_configuration_api.py index a14964e1a4..d4b3ba7d33 100644 --- a/tests/test_chat_goal_configuration_api.py +++ b/tests/test_chat_goal_configuration_api.py @@ -509,7 +509,10 @@ def test_goal_configuration_service_rechecks_revision_before_write( monkeypatch: pytest.MonkeyPatch, ) -> None: registry_path = tmp_path / "registry.json" - registry_path.write_text("{}\n", encoding="utf-8") + import json + initial_registry = json.dumps({"common_runtime_root": str(tmp_path / "runtime"), + "goals": [{"id": "goal-example", "repo": str(tmp_path)}]}) + "\n" + registry_path.write_text(initial_registry, encoding="utf-8") calls: list[dict[str, Any]] = [] def configure_goal_stub(**kwargs: Any) -> dict[str, Any]: @@ -534,4 +537,4 @@ def configure_goal_stub(**kwargs: Any) -> dict[str, Any]: ) assert len(calls) == 1 - assert registry_path.read_text(encoding="utf-8") == "{}\n" + assert registry_path.read_text(encoding="utf-8") == initial_registry diff --git a/tests/test_file_lock.py b/tests/test_file_lock.py index 802e08d312..e3abe715d5 100644 --- a/tests/test_file_lock.py +++ b/tests/test_file_lock.py @@ -413,3 +413,56 @@ def fake_release(*args: object, **kwargs: object) -> bool: pass assert calls == [True] + + +@pytest.mark.skipif(os.name != "nt", reason="Win32 extended path regression") +def test_long_lock_paths_release_and_keep_one_identity(tmp_path): + from loopx.control_plane.runtime.file_paths import windows_extended_path + + target = tmp_path / ("nested-" * 12) / ("a" * 64 + ".json") + extended = windows_extended_path(target) + assert file_lock._lock_id(target) == file_lock._lock_id(extended) + with exclusive_cross_runtime_file_lock(target): + assert file_lock.cross_runtime_lock_witness(target)["token"] + # Long claim-file cleanup must succeed while this process is still alive. + with exclusive_cross_runtime_file_lock(extended, timeout_seconds=0): + with pytest.raises(LockAcquireTimeoutError): + with exclusive_cross_runtime_file_lock(target, timeout_seconds=0): + pytest.fail("normal and extended addresses acquired two locks") + assert not file_lock._effect_mutation_lock_path(target).exists() + + +@pytest.mark.skipif(os.name != "nt", reason="Win32 extended path regression") +def test_long_lock_paths_exclude_other_process_and_release_while_holder_alive(tmp_path): + from loopx.control_plane.runtime.file_paths import windows_extended_path + + target = tmp_path / ("nested-" * 24) / ("a" * 64 + ".json") + assert len(str(target)) > 260 + extended = windows_extended_path(target) + script = """ +import sys +from pathlib import Path +from loopx.file_lock import exclusive_cross_runtime_file_lock +with exclusive_cross_runtime_file_lock(Path(sys.argv[1])): + print("ready", flush=True) + sys.stdin.readline() +print("released", flush=True) +sys.stdin.readline() +""" + holder = subprocess.Popen([sys.executable, "-c", script, str(target)], + stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True) + try: + assert holder.stdout is not None and holder.stdin is not None + assert holder.stdout.readline().strip() == "ready" + with pytest.raises(LockAcquireTimeoutError): + with exclusive_cross_runtime_file_lock(extended, timeout_seconds=0): + pytest.fail("extended address bypassed another process's lock") + holder.stdin.write("release\n") + holder.stdin.flush() + assert holder.stdout.readline().strip() == "released" + assert holder.poll() is None + with exclusive_cross_runtime_file_lock(extended, timeout_seconds=0): + assert file_lock._lock_id(target) == file_lock._lock_id(extended) + assert not file_lock._effect_mutation_lock_path(target).exists() + finally: + _stop(holder) diff --git a/tests/test_packaged_skill_metadata.py b/tests/test_packaged_skill_metadata.py index fda1c66f73..26750eeec3 100644 --- a/tests/test_packaged_skill_metadata.py +++ b/tests/test_packaged_skill_metadata.py @@ -30,3 +30,14 @@ def test_packaged_scope_markers_ship_with_workflow_sources(): assert f"skills/{skill_id}/.loopx-skill-scope" in data_files[ f"share/loopx/skills/{skill_id}" ] + + +def test_packaged_skill_display_metadata_is_in_distribution(): + import tomllib + from loopx.skill_install_readback import PACKAGED_HOST_SKILL_IDS + + package = tomllib.loads((REPO_ROOT / "pyproject.toml").read_text(encoding="utf-8")) + data_files = package["tool"]["setuptools"]["data-files"] + for skill_id in PACKAGED_HOST_SKILL_IDS: + assert f"skills/{skill_id}/agents/openai.yaml" in data_files[ + f"share/loopx/skills/{skill_id}/agents"] diff --git a/tests/test_peer_collaboration.py b/tests/test_peer_collaboration.py index b00105c532..2ca475a600 100644 --- a/tests/test_peer_collaboration.py +++ b/tests/test_peer_collaboration.py @@ -123,6 +123,7 @@ def cli(root, registry, agent, action, *args, ok=True): capture_output=True, text=True, timeout=30, + cwd=registry.parent, ) result = json.loads(proc.stdout) assert proc.returncode == (0 if ok else 1), (proc.stdout, proc.stderr) @@ -450,7 +451,12 @@ def test_changed_missing_and_escaping_artifacts_are_explicit(scenario, tmp_path) assert input_readiness(registry, "delivery", brief)[0]["status"] == "changed" (root / "inputs/demand.csv").unlink() assert input_readiness(registry, "delivery", brief)[0]["status"] == "unavailable" - (root / "inputs/demand.csv").symlink_to(root.parent / "outside.csv") + try: + (root / "inputs/demand.csv").symlink_to(root.parent / "outside.csv") + except OSError as exc: + if sys.platform == "win32" and exc.winerror == 1314: + pytest.skip("Windows symlink fixture requires privileges") + raise assert ( input_readiness(registry, "delivery", brief)[0]["status"] == "outside_workspace" ) @@ -597,6 +603,7 @@ def test_consumption_rejects_corrupt_reply_and_receipt(scenario): consume_return(root, "delivery", "builder", rid) +@pytest.mark.skipif(sys.platform == "win32", reason="POSIX FIFO fixture") def test_special_file_read_is_bounded_and_stopped_goal_remains_readable(scenario): import os from loopx.control_plane.collaboration.peers import read_inbox @@ -733,3 +740,47 @@ def test_peer_update_rejects_another_results_read_and_consumption_receipts(scena consumed.write_bytes((folder / "conclusion.consumed.json").read_bytes()) with pytest.raises(ValueError, match="receipt scope"): returns(root, "delivery", "builder") +@pytest.mark.skipif(sys.platform != "win32", reason="Win32 extended path regression") +def test_peer_exchange_survives_long_private_store_paths(scenario): + root, registry, brief, *_ = scenario + from loopx.control_plane.collaboration.inbox import _root + + long_root = root / ("nested-runtime-" * 7) + packet = root / "long-path-brief.json" + packet.write_text(json.dumps(brief)) + args = ("--peer-agent-id", "reviewer", "--operation-id", "long-path-review", + "--brief-file", str(packet)) + sent = cli(long_root, registry, "builder", "request", *args) + rid = sent["request_id"] + assert cli(long_root, registry, "reviewer", "read")["items"][0]["request_id"] == rid + cli(long_root, registry, "reviewer", "acknowledge", "--request-id", rid, + "--decision", "adopt", "--reason", "Reviewing the pinned artifact") + cli(long_root, registry, "reviewer", "report", "--request-id", rid, + "--phase", "conclusion", "--reply-text", "Independent review complete") + assert cli(long_root, registry, "builder", "read")["peer_returns"]["items"][0]["request_id"] == rid + cli(long_root, registry, "builder", "acknowledge-return", "--request-id", rid) + replay = cli(long_root, registry, "builder", "request", *args) + assert replay["replayed"] and replay["request_id"] == rid + requests = [path for path in (_root(long_root) / "entries").glob("*/*.json") if path.stem == rid] + assert len(requests) == 1 and len(str(requests[0])) > 260 + + +@pytest.mark.skipif(sys.platform != "win32", reason="Win32 binary input regression") +def test_peer_binary_artifact_preserves_crlf_and_ctrl_z_digest(scenario): + root, registry, brief, *_ = scenario + content = b"before\r\n\x1aafter\r\n\x00\xff" + artifact = root / "inputs" / "packet.bin" + artifact.write_bytes(content) + digest = hashlib.sha256(content).hexdigest() + brief = {**brief, "inputs": [{"ref": "inputs/packet.bin", + "description": "Binary fixture", "sha256": digest}]} + packet = root / "binary-brief.json" + packet.write_text(json.dumps(brief), encoding="utf-8") + sent = cli(root, registry, "builder", "request", "--peer-agent-id", "reviewer", + "--operation-id", "binary-review", "--brief-file", str(packet)) + item = cli(root, registry, "reviewer", "read")["items"][0] + assert item["request_id"] == sent["request_id"] + [readiness] = item["input_readiness"] + assert readiness["status"] == "available" + assert readiness["observed_sha256"] == readiness["expected_sha256"] == digest + assert readiness["content_supplied"] is False