Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 8 additions & 6 deletions docs/architecture/rfcs/harness-selection-dsh-pi-v0.md
Original file line number Diff line number Diff line change
Expand Up @@ -531,7 +531,8 @@ Shipped enforcement, in delivery order:
the first bounded Todo of each *ready* lane through the canonical Todo owner,
creates nothing for a gap lane, and refuses an unknown Goal before any write.
The receipt records the proposal digest, so a replayed settlement reuses the
same lane Todo instead of adding a second row.
same lane Todo instead of adding a second row, and the receipt names every
lane Todo the settlement ensured.

The intake is still inert in production, and this section does not claim
otherwise. Nothing yet supplies `team_plan_context`, so a model-authored preview
Expand All @@ -540,11 +541,12 @@ that supplies the admission facts and the settlement that re-derives them must
stay one contract rather than two; and the apply entry point today is a governed
capability execution journal, so a confirmed Chat preview needs that bridge
before an owner confirmation can materialize lanes. Two further gaps belong with
this work: the published receipt carries the first lane Todo's identity rather
than the identity of every lane it created (the apply result computes the full
`lane_todo_ids` set, and the receipt field set is closed and persisted, so
publishing it is a bounded compatibility change), and a multi-lane preview has
no frontend confirmation surface yet.
this work: a confirmed plan has no Chat-side apply path yet, and a multi-lane
preview has no frontend confirmation surface. The readback is no longer one of
those gaps: the apply publishes every lane Todo it ensured under a bounded
`lane_todo_ids` field, that field is the one additive exception to the closed,
persisted receipt field set so a receipt written before it still validates, and
a team-plan receipt carries no monitor key because a plan is not a monitor.

### Relationship to the multi-agent contracts

Expand Down
9 changes: 5 additions & 4 deletions docs/architecture/rfcs/harness-selection-dsh-pi-v0.zh-CN.md
Original file line number Diff line number Diff line change
Expand Up @@ -414,15 +414,16 @@ Todo 创建、quota 或 goal policy——复用预览点名的身份,不得扩
落地时重新按本 Goal 已注册 Agent 与本机 shipment 的 advancement action kind 校验,经
canonical Todo owner 为每条 **ready** lane 创建首个有界 Todo,gap lane 不创建任何东西,
未知 Goal 在任何写入前就被拒绝,回执记录 proposal digest,因此重放结算复用同一条 lane
Todo 而不会新增第二行。
Todo 而不会新增第二行,并且回执点名这次确保的每一条 lane Todo。

这条入端口径目前在线上仍是**惰性**的,本节不作相反声明:还没有任何生产调用方传入
`team_plan_context`,因此模型产出的预览会在准入处被丢弃,而不会浮现给业主确认;提供准入事实
的适配器与重新推导这些事实的结算必须保持同一份契约而不是两份;而今天的落地入口是受治理能力
执行 journal,所以被确认的 Chat 预览还需要那座桥,业主确认才能真正建成 lane。另有两处缺口
属于这条工作线:已发布回执只带第一条 lane Todo 的身份,而不是它创建的全部 lane 身份(apply
结果里算了完整的 `lane_todo_ids`,但回执字段集是封闭且持久化的,发布它是一次有界的兼容性
变更);以及多 lane 预览还没有前端确认面。
属于这条工作线:被确认的计划还没有 Chat 侧的落地路径,以及多 lane 预览还没有前端确认面。
回读本身已经不再是缺口:落地会把这次确保的每一条 lane Todo 以有界字段 `lane_todo_ids`
发布出去;该字段是那个封闭且持久化的回执字段集的**唯一**加性例外,因此早前写下的回执仍然
通过校验,而团队计划回执不带 monitor key——计划不是 monitor。

### 与 multi-agent 契约的关系

Expand Down
52 changes: 45 additions & 7 deletions loopx/control_plane/work_items/governed_transition_proposal.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@

import hashlib
import json
import re
from collections.abc import Callable, Mapping, Sequence
from copy import deepcopy
from enum import StrEnum
Expand Down Expand Up @@ -39,6 +40,13 @@
"status",
"target_key",
}
# Receipts are persisted in the settlement journal, so the field set stays
# closed and a new field is admitted only as an explicitly bounded addition
# that an older receipt may still omit. `lane_todo_ids` is the readback of a
# team plan: every lane Todo the settlement ensured, not just the first one.
_OPTIONAL_RECEIPT_FIELDS = {"lane_todo_ids"}
_LANE_TODO_ID_LIMIT = 8
_LANE_TODO_ID = re.compile(r"^todo_[A-Za-z0-9]{1,40}$")


TransitionCheckpoint = Callable[[list[dict[str, Any]]], None]
Expand Down Expand Up @@ -87,7 +95,11 @@ def validate_governed_transition_receipts(
proposal_ids: set[str] = set()
for index, raw in enumerate(value):
receipt = _mapping(raw, f"governed transition receipt[{index}]")
if set(receipt) != _RECEIPT_FIELDS:
# An older receipt may omit the bounded readback field; nothing else may
# be added, so a receipt can never carry a field it did not mean to.
if not _RECEIPT_FIELDS <= set(receipt) <= (
_RECEIPT_FIELDS | _OPTIONAL_RECEIPT_FIELDS
):
raise ValueError("governed transition proposal receipt fields are invalid")
if receipt.get("schema_version") != GOVERNED_TRANSITION_RECEIPT_SCHEMA_VERSION:
raise ValueError("governed transition proposal receipt schema is invalid")
Expand All @@ -103,22 +115,43 @@ def validate_governed_transition_receipts(
raise ValueError("governed transition proposal receipt kind is invalid")
if receipt.get("status") != "committed":
raise ValueError("governed transition proposal receipt status is invalid")
for field in (
"proposal_digest",
"monitor_key",
"action",
"todo_id",
):
for field in ("proposal_digest", "action", "todo_id"):
if not isinstance(receipt.get(field), str) or not receipt[field]:
raise ValueError(
f"governed transition proposal receipt {field} is invalid"
)
# A monitor transition is identified by its monitor key, so that key is
# required there. A team plan is not a monitor and must not invent one,
# so its key is explicitly absent rather than an empty string.
monitor_key = receipt.get("monitor_key")
if receipt.get("kind") == STEWARD_TEAM_PLAN_PREVIEW_KIND:
if monitor_key is not None:
raise ValueError(
"governed transition proposal receipt monitor_key is invalid"
)
elif not isinstance(monitor_key, str) or not monitor_key:
raise ValueError(
"governed transition proposal receipt monitor_key is invalid"
)
if receipt.get("target_key") is not None and not isinstance(
receipt.get("target_key"), str
):
raise ValueError(
"governed transition proposal receipt target_key is invalid"
)
lane_todo_ids = receipt.get("lane_todo_ids")
if lane_todo_ids is not None and (
not isinstance(lane_todo_ids, list)
or not 1 <= len(lane_todo_ids) <= _LANE_TODO_ID_LIMIT
or len(set(lane_todo_ids)) != len(lane_todo_ids)
or any(
not isinstance(item, str) or not _LANE_TODO_ID.fullmatch(item)
for item in lane_todo_ids
)
):
raise ValueError(
"governed transition proposal receipt lane_todo_ids is invalid"
)
validate_public_safe_value(receipt, path=f"transition_receipts[{index}]")
receipts.append(receipt)
return receipts
Expand Down Expand Up @@ -413,6 +446,11 @@ def settle_governed_transition_proposals(
"status": "committed",
"target_key": result.get("target_key"),
}
lane_todo_ids = result.get("lane_todo_ids")
if lane_todo_ids:
# The apply ensured every ready lane's first Todo; a receipt that
# named only the first one could not be read as "what exists now".
receipt["lane_todo_ids"] = [str(item) for item in lane_todo_ids]
validate_public_safe_value(receipt, path="transition_receipt")
receipts.append(receipt)
by_proposal_id[proposal_id] = receipt
Expand Down
104 changes: 102 additions & 2 deletions tests/test_steward_team_plan_apply.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,13 +10,16 @@
from loopx.control_plane.work_items.governed_transition_proposal import (
GovernedTransitionSettlementPhase,
settle_governed_transition_proposals,
validate_governed_transition_receipts,
)

GOAL_ID = "team-plan-apply-fixture"
AGENT_ID = "agent-alpha"


def _fixture(tmp_path: Path) -> tuple[Path, Path]:
def _fixture(
tmp_path: Path, *, agents: tuple[str, ...] = (AGENT_ID,)
) -> tuple[Path, Path]:
project = tmp_path / "project"
runtime = tmp_path / "runtime"
state_file = f".codex/goals/{GOAL_ID}/ACTIVE_GOAL_STATE.md"
Expand Down Expand Up @@ -55,7 +58,7 @@ def _fixture(tmp_path: Path) -> tuple[Path, Path]:
"status": "connected-read-only",
},
"coordination": {
"registered_agents": [AGENT_ID],
"registered_agents": list(agents),
"agent_model": "peer_v1",
},
}
Expand Down Expand Up @@ -180,3 +183,100 @@ def test_an_unknown_goal_is_refused_before_any_todo(tmp_path: Path) -> None:
)

assert "loopx:todo " not in _todos(project)


def _second_lane() -> dict:
return {
"lane_id": "lane-beta",
"agent_id": "agent-beta",
"acceptance": "The second lane's first Todo is delivered with evidence",
"first_todo": {
"text": "Read back the second lane's bounded first turn",
"priority": "P2",
"task_class": "advancement_task",
"action_kind": "implement",
},
}


def test_the_receipt_names_every_lane_todo_it_created(tmp_path: Path) -> None:
"""One readback has to say what exists now, not only where it started."""

project, registry_path = _fixture(tmp_path, agents=(AGENT_ID, "agent-beta"))

receipts = _settle(registry_path, _proposal(extra_lane=_second_lane()))

receipt = receipts[0]
lane_todo_ids = receipt["lane_todo_ids"]
assert len(lane_todo_ids) == 2 and len(set(lane_todo_ids)) == 2
# The first lane Todo is still the receipt's own identity, so a reader that
# only knows the older field keeps working.
assert receipt["todo_id"] == lane_todo_ids[0]
assert _todos(project).count("loopx:todo ") == 2
# Both the plan readback and the older receipt shape validate, which is what
# a settlement journal does with its stored receipts.
assert len(validate_governed_transition_receipts(receipts)) == 1

# A replayed settlement reports the same lanes instead of an empty readback.
replay = _settle(registry_path, _proposal(extra_lane=_second_lane()))
assert replay[0]["action"] == "reused"
assert replay[0]["lane_todo_ids"] == lane_todo_ids
assert _todos(project).count("loopx:todo ") == 2


def _receipt(**overrides) -> dict:
receipt = {
"schema_version": "loopx_governed_transition_proposal_receipt_v0",
"proposal_id": "proposal-receipt-fixture",
"proposal_digest": "sha256:" + "a" * 64,
"kind": "continuous_monitor_upsert",
"monitor_key": "monitor-key-1",
"action": "updated",
"todo_id": "todo_1",
"status": "committed",
"target_key": None,
}
receipt.update(overrides)
return receipt


def test_the_lane_readback_is_optional_bounded_and_additive() -> None:
"""An older receipt stays valid; the new field is the only addition."""

assert len(validate_governed_transition_receipts([_receipt()])) == 1
assert len(
validate_governed_transition_receipts(
[_receipt(lane_todo_ids=["todo_1", "todo_2"])]
)
) == 1

for invalid_readback in (
[],
["todo_1", "todo_1"],
["todo_1", "not-a-todo-id"],
["todo_1", "todo_" + "a" * 41],
["todo_1"] * 9,
"todo_1",
):
with pytest.raises(ValueError, match="lane_todo_ids is invalid"):
validate_governed_transition_receipts(
[_receipt(lane_todo_ids=invalid_readback)]
)
# The field set stays closed: the readback is the only thing that may be
# added, and a monitor receipt still has to name its own key.
with pytest.raises(ValueError, match="receipt fields are invalid"):
validate_governed_transition_receipts([_receipt(unexpected_field=1)])
with pytest.raises(ValueError, match="monitor_key is invalid"):
validate_governed_transition_receipts([_receipt(monitor_key=None)])


def test_a_team_plan_receipt_must_not_invent_a_monitor_key() -> None:
"""A plan is not a monitor, so its receipt carries no monitor identity."""

team_plan = _receipt(kind="steward_team_plan_preview", monitor_key=None)

assert len(validate_governed_transition_receipts([team_plan])) == 1
with pytest.raises(ValueError, match="monitor_key is invalid"):
validate_governed_transition_receipts(
[{**team_plan, "monitor_key": "monitor-key-1"}]
)
Loading