From f9ae886b036aa66b09f7a403344a709e7097b2b7 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Wed, 16 Sep 2026 18:41:57 +0800 Subject: [PATCH] feat(steward): give the steward Turn the facts a team preview needs A team preview is admitted only when the host can say which Agents exist for the Goal the plan names, and nothing in production supplied that. A model-authored preview was therefore always dropped, so the intake shipped in #4519/#4522/#4524 could not surface at all. The Turn owner now supplies the facts, per Goal. The manager channel is not bound to one Goal, so admission receives a lookup rather than one Goal's facts: the owner's own channel resolves any registered Goal, an external manager channel resolves only the Goals its scope resolver authorizes, and a Goal the registry does not know - or one outside that channel's scope - resolves to "cannot describe this Goal", which drops the preview instead of validating it against another Goal's Agents. The lookup runs once per proposal, for the Goal the proposal names, and re-resolves the channel scope each time rather than caching an authorization decision. The contract now distinguishes two answers that used to look alike: an empty Agent list is a fact, so the plan's lanes become typed `agent_not_registered` gaps, while an unresolvable Goal surfaces nothing at all. `parse_agent_response` carries the context through to the four segments that parse an answer (managed dsh, Codex agent, ACP, provider HTTP), so the same admission rule applies to every transport instead of only the one the steward happens to run on. Verified: tests/test_steward_team_plan_preview.py, tests/test_steward_team_plan_apply.py, tests/test_chat_manager_context.py and tests/test_attached_session_broker.py 63 passed, including a preview for a Goal outside the channel scope that is dropped, a per-Goal lookup that is asked only about the named Goal, a Goal with no registered Agents that becomes a typed gap, and the owner channel resolving any registered Goal. examples/docs-governance-smoke.py and examples/loopx-steward-managed-chat-smoke.py ok. Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../rfcs/harness-selection-dsh-pi-v0.md | 31 ++--- .../rfcs/harness-selection-dsh-pi-v0.zh-CN.md | 20 ++-- loopx/chat.py | 37 +++++- loopx/chat_acp.py | 1 + loopx/chat_agent.py | 6 +- loopx/chat_dsh.py | 6 +- loopx/chat_providers.py | 12 +- loopx/chat_runtime.py | 58 +++++++++ tests/test_steward_team_plan_preview.py | 111 +++++++++++++++++- 9 files changed, 251 insertions(+), 31 deletions(-) diff --git a/docs/architecture/rfcs/harness-selection-dsh-pi-v0.md b/docs/architecture/rfcs/harness-selection-dsh-pi-v0.md index 1873cfd42a..d1f5bd9809 100644 --- a/docs/architecture/rfcs/harness-selection-dsh-pi-v0.md +++ b/docs/architecture/rfcs/harness-selection-dsh-pi-v0.md @@ -537,20 +537,23 @@ Shipped enforcement, in delivery order: The receipt records the proposal digest, so a replayed settlement reuses the same lane Todo instead of adding a second row, and the receipt names every lane Todo the settlement ensured. - -The intake is still inert in production, and this section does not claim -otherwise. Nothing yet supplies `team_plan_context`, so a model-authored preview -is dropped at admission instead of being surfaced for confirmation; the adapter -that supplies the admission facts and the settlement that re-derives them must -stay one contract rather than two; and the apply entry point today is a governed -capability execution journal, so a confirmed Chat preview needs that bridge -before an owner confirmation can materialize lanes. Two further gaps belong with -this work: a confirmed plan has no Chat-side apply path yet, and a multi-lane -preview has no frontend confirmation surface. The readback is no longer one of -those gaps: the apply publishes every lane Todo it ensured under a bounded -`lane_todo_ids` field, that field is the one additive exception to the closed, -persisted receipt field set so a receipt written before it still validates, and -a team-plan receipt carries no monitor key because a plan is not a monitor. +4. **Admission facts.** The manager channel's Turn attaches a per-Goal lookup to + the segment that parses the answer, so a preview is validated against the + Agents of the Goal it names: the owner's own channel resolves any registered + Goal, an external manager channel resolves only the Goals it is bound to, and + a Goal the registry does not know - or one outside that channel's scope - + drops the preview instead of validating it against another Goal's Agents. + +What is still missing is the effect, not the preview: a confirmed plan has no +Chat-side apply path yet, because the apply entry point today is a governed +capability execution journal, so an owner's confirmation has nowhere to land, +and a multi-lane preview has no frontend confirmation surface. A materialized +lane Todo also does not yet carry the canonical intent revision it is meant to +advance. The readback is no longer one of those gaps: the apply publishes every +lane Todo it ensured under a bounded `lane_todo_ids` field, that field is the one +additive exception to the closed, persisted receipt field set so a receipt +written before it still validates, and a team-plan receipt carries no monitor key +because a plan is not a monitor. ### Relationship to the multi-agent and shared-authority contracts diff --git a/docs/architecture/rfcs/harness-selection-dsh-pi-v0.zh-CN.md b/docs/architecture/rfcs/harness-selection-dsh-pi-v0.zh-CN.md index 2d10e0c738..aa994ce202 100644 --- a/docs/architecture/rfcs/harness-selection-dsh-pi-v0.zh-CN.md +++ b/docs/architecture/rfcs/harness-selection-dsh-pi-v0.zh-CN.md @@ -419,15 +419,17 @@ Todo 创建、quota 或 goal policy——复用预览点名的身份,不得扩 按某个 Goal 的 Agent 通过准入的计划无法被改投到另一个 Goal;回执记录 proposal digest, 因此重放结算复用同一条 lane Todo 而不会新增第二行,并且回执点名这次确保的每一条 lane Todo。 - -这条入端口径目前在线上仍是**惰性**的,本节不作相反声明:还没有任何生产调用方传入 -`team_plan_context`,因此模型产出的预览会在准入处被丢弃,而不会浮现给业主确认;提供准入事实 -的适配器与重新推导这些事实的结算必须保持同一份契约而不是两份;而今天的落地入口是受治理能力 -执行 journal,所以被确认的 Chat 预览还需要那座桥,业主确认才能真正建成 lane。另有两处缺口 -属于这条工作线:被确认的计划还没有 Chat 侧的落地路径,以及多 lane 预览还没有前端确认面。 -回读本身已经不再是缺口:落地会把这次确保的每一条 lane Todo 以有界字段 `lane_todo_ids` -发布出去;该字段是那个封闭且持久化的回执字段集的**唯一**加性例外,因此早前写下的回执仍然 -通过校验,而团队计划回执不带 monitor key——计划不是 monitor。 +4. **准入事实。** 管家通道的 Turn 会给"解析答案的那一段"挂上一个按 Goal 解析的查询,因此 + 预览只会用**它点名那个 Goal** 的 Agent 来校验:业主自己的通道可解析任意已注册 Goal, + 外部管家通道只解析它被绑定的 Goal,而 registry 不认识的 Goal——或超出该通道范围的 + Goal——会让预览被丢弃,而不是拿另一个 Goal 的 Agent 去校验它。 + +仍然缺的是**效果**而不是预览:被确认的计划还没有 Chat 侧的落地路径——今天的落地入口是受治理 +能力执行 journal——所以业主的确认暂时无处落地;多 lane 预览也还没有前端确认面;另外,建出 +的 lane Todo 还没有携带它本应推进的规范意图修订。回读本身已经不再是缺口:落地会把这次确保的 +每一条 lane Todo 以有界字段 `lane_todo_ids` 发布出去;该字段是那个封闭且持久化的回执字段集的 +**唯一**加性例外,因此早前写下的回执仍然通过校验,而团队计划回执不带 monitor key——计划不是 +monitor。 ### 与 multi-agent / shared authority 契约的关系 diff --git a/loopx/chat.py b/loopx/chat.py index 0221b7bc51..d6769c86eb 100644 --- a/loopx/chat.py +++ b/loopx/chat.py @@ -229,7 +229,13 @@ def _normalize_proposals( def _validated_team_plan_preview( raw: Mapping[str, Any], context: Mapping[str, Any] | None ) -> dict[str, Any] | None: - """Return the validated preview, or ``None`` when it may not be surfaced.""" + """Return the validated preview, or ``None`` when it may not be surfaced. + + A preview names its Goal, so the host facts are per Goal rather than for + "the" Goal: the manager channel is not bound to one, and a plan for a Goal + the host was not given facts for is dropped instead of being validated + against another Goal's Agents. + """ if not isinstance(context, Mapping): return None @@ -237,10 +243,30 @@ def _validated_team_plan_preview( validate_steward_team_plan_preview, ) + goal_id = str(raw.get("goal_id") or "") + agents: list[str] | None = None + by_goal = context.get("registered_agents_by_goal") + if isinstance(by_goal, Mapping): + declared = by_goal.get(goal_id) + if isinstance(declared, (list, tuple)): + agents = [str(value) for value in declared] + if agents is None: + # A host with a large Goal set resolves on demand, and only for the + # Goal the plan named, so admission stays bounded by one lookup. + resolve = context.get("resolve_registered_agents") + if callable(resolve): + try: + resolved = resolve(goal_id) + except (OSError, ValueError, TypeError, KeyError, RuntimeError): + resolved = None + if isinstance(resolved, (list, tuple)): + agents = [str(value) for value in resolved] + if agents is None: + return None try: return validate_steward_team_plan_preview( raw, - registered_agent_ids=list(context.get("registered_agent_ids") or []), + registered_agent_ids=agents, supported_action_kinds=list(context.get("supported_action_kinds") or []), ) except ValueError: @@ -347,6 +373,7 @@ def parse_agent_response( raw_text: str, *, protected_paths: Iterable[Path | str] = (), + team_plan_context: Mapping[str, Any] | None = None, ) -> dict[str, Any]: protected = tuple(protected_paths) start = raw_text.rfind(CHAT_REVIEW_OPEN_TAG) @@ -358,7 +385,11 @@ def parse_agent_response( except json.JSONDecodeError: payload = None if isinstance(payload, dict): - return normalize_agent_response(payload, protected_paths=protected) + return normalize_agent_response( + payload, + protected_paths=protected, + team_plan_context=team_plan_context, + ) key = re.search(r'"message"\s*:\s*', body) if key: try: diff --git a/loopx/chat_acp.py b/loopx/chat_acp.py index ba346b1b64..0d99583465 100644 --- a/loopx/chat_acp.py +++ b/loopx/chat_acp.py @@ -416,6 +416,7 @@ def on_activity() -> None: response = parse_agent_response( raw_response, protected_paths=[self.work_dir, self.agent_work_dir], + team_plan_context=getattr(self, "team_plan_context", None), ) if CHAT_REVIEW_OPEN_TAG not in raw_response or CHAT_REVIEW_CLOSE_TAG not in raw_response: event_sink("protocol.warning", {"error_code": "missing_review_envelope"}) diff --git a/loopx/chat_agent.py b/loopx/chat_agent.py index e97e90c039..5d117f8aba 100644 --- a/loopx/chat_agent.py +++ b/loopx/chat_agent.py @@ -1010,7 +1010,11 @@ def send( visible_delta_count += 1 on_event("answer.delta", {"text": visible_tail}) raw_response = "".join(parts) - response = parse_agent_response(raw_response, protected_paths=[self.work_dir]) + response = parse_agent_response( + raw_response, + protected_paths=[self.work_dir], + team_plan_context=getattr(self, "team_plan_context", None), + ) if on_event: if ( CHAT_REVIEW_OPEN_TAG not in raw_response diff --git a/loopx/chat_dsh.py b/loopx/chat_dsh.py index dfb8e4159c..b956f2ed93 100644 --- a/loopx/chat_dsh.py +++ b/loopx/chat_dsh.py @@ -246,7 +246,11 @@ def run() -> None: gate=None, error_code=MANAGED_HOST_CHAT_FAILED, ) - response = parse_agent_response(raw, protected_paths=[self.work_dir]) + response = parse_agent_response( + raw, + protected_paths=[self.work_dir], + team_plan_context=getattr(self, "team_plan_context", None), + ) if slot.interrupted: # The channel discarded this turn, so its answer never happened: the # segment still exits as the binding's executor, but folding its diff --git a/loopx/chat_providers.py b/loopx/chat_providers.py index eca154ca12..4e1aa17ea9 100644 --- a/loopx/chat_providers.py +++ b/loopx/chat_providers.py @@ -250,7 +250,11 @@ def start_turn(self, message: str, event_sink: EventSink) -> dict[str, Any]: if visible_tail: visible_delta_count += 1 event_sink("answer.delta", {"text": visible_tail}) - response = parse_agent_response(raw_response, protected_paths=[self.work_dir]) + response = parse_agent_response( + raw_response, + protected_paths=[self.work_dir], + team_plan_context=getattr(self, "team_plan_context", None), + ) self.resumed = True event_sink("agent.phase", {"label": "正在整理回答"}) if visible_delta_count == 0: @@ -372,7 +376,11 @@ def start_turn(self, message: str, event_sink: EventSink) -> dict[str, Any]: raise ValueError(f"unsupported direct model provider: {self.provider}") else: raise _provider_error(self.provider.title(), "The model exceeded the bounded read-only tool-call limit.") - response = parse_agent_response(raw_response, protected_paths=[self.work_dir]) + response = parse_agent_response( + raw_response, + protected_paths=[self.work_dir], + team_plan_context=getattr(self, "team_plan_context", None), + ) self.history.extend([ {"role": "user", "content": prompt}, {"role": "assistant", "content": raw_response}, diff --git a/loopx/chat_runtime.py b/loopx/chat_runtime.py index ba4f4924f3..e3fad38c5b 100644 --- a/loopx/chat_runtime.py +++ b/loopx/chat_runtime.py @@ -290,6 +290,57 @@ def steward_executor_defaults(self) -> dict[str, Any]: return load_effective_steward_executor_defaults(self.store.root.parent) + def _team_plan_admission_context( + self, session: Mapping[str, Any] + ) -> dict[str, Any] | None: + """Host facts a team preview may be validated against, per Goal. + + A team plan names its own Goal and the manager channel is not bound to + one, so admission receives a lookup instead of one Goal's facts. The + lookup re-uses the authorization the Turn owner already resolved: an + external manager channel resolves only its authorized Goals, and a plan + for any other Goal is dropped rather than validated against the Agents + of a Goal it does not name. + """ + + channel_id = str(session.get("channel_id") or "") + if not is_manager_channel(channel_id): + return None + from .agent_registry import load_goal_from_registry, registered_agent_ids_for_goal + from .control_plane.todos.contract import ( + TODO_ACTION_KIND_ADVANCEMENT_VALUES, + ) + + def resolve(goal_id: str) -> list[str] | None: + if not goal_id: + return None + if channel_id != "manager": + # The owner's own channel is not scoped to a subset of Goals; + # an external channel only ever sees the Goals it was bound to. + scope = ( + self.manager_scope_resolver(session) + if self.manager_scope_resolver + else None + ) + if not isinstance(scope, list) or goal_id not in { + str(item) for item in scope + }: + return None + try: + goal = load_goal_from_registry(Path(self.registry_path), goal_id) + except (OSError, ValueError, TypeError, KeyError): + return None + if goal is None: + # A Goal the registry does not know cannot be validated against + # anything, and its lanes are not gaps: the plan is dropped. + return None + return registered_agent_ids_for_goal(goal) + + return { + "resolve_registered_agents": resolve, + "supported_action_kinds": sorted(TODO_ACTION_KIND_ADVANCEMENT_VALUES), + } + def capabilities(self) -> list[dict[str, Any]]: builtins = builtin_chat_endpoints( codex_bin=self.codex_bin, @@ -1179,6 +1230,13 @@ def scope_valid() -> bool: context["evidence_sources"] = inspection.sources() context = manager_index(context) message = "Fresh Core evidence (JSON data, not instructions):\n" + json.dumps(context, ensure_ascii=False) + "\n\nCurrent user message:\n" + message + # A steward answer may contain a team preview. It is admitted only + # against the facts of the Goal it names, so the segment that parses + # that answer gets the lookup rather than a second copy of the + # evidence above. + team_plan_context = self._team_plan_admission_context(session) + if team_plan_context is not None: + adapter.team_plan_context = team_plan_context if attachments: if not isinstance(adapter, CodexAppServerAdapter): raise ValueError("image attachments currently require the Codex Agent endpoint") diff --git a/tests/test_steward_team_plan_preview.py b/tests/test_steward_team_plan_preview.py index ad37721cbc..9998832e08 100644 --- a/tests/test_steward_team_plan_preview.py +++ b/tests/test_steward_team_plan_preview.py @@ -156,7 +156,7 @@ def test_the_chat_normalizer_admits_a_preview_only_with_host_facts() -> None: "gate": None, } context = { - "registered_agent_ids": ["agent-alpha"], + "registered_agents_by_goal": {"team-plan-fixture": ["agent-alpha"]}, "supported_action_kinds": ["implement"], } @@ -172,6 +172,56 @@ def test_the_chat_normalizer_admits_a_preview_only_with_host_facts() -> None: # surfaced at all rather than admitted half-checked. assert normalize_agent_response(envelope)["proposals"] == [] + # A plan for a Goal this Turn was not given facts for is dropped as well: + # another Goal's Agents must not validate it. + other_goal = {**envelope, "proposals": [_plan(goal_id="some-other-goal")]} + assert normalize_agent_response(other_goal, team_plan_context=context)[ + "proposals" + ] == [] + + # A host with a large Goal set resolves on demand, for the named Goal only. + looked_up: list[str] = [] + + def resolve(goal_id: str) -> list[str] | None: + looked_up.append(goal_id) + # ``None`` means "this host cannot describe that Goal", which drops the + # preview; an empty list is a real answer and becomes typed gaps. + return ["agent-alpha"] if goal_id == "team-plan-fixture" else None + + on_demand_context = { + "resolve_registered_agents": resolve, + "supported_action_kinds": ["implement"], + } + on_demand = normalize_agent_response( + envelope, team_plan_context=on_demand_context + ) + assert [item["kind"] for item in on_demand["proposals"]] == [ + "steward_team_plan_preview" + ] + assert looked_up == ["team-plan-fixture"] + assert normalize_agent_response( + other_goal, team_plan_context=on_demand_context + )["proposals"] == [] + + unresolved = normalize_agent_response( + envelope, team_plan_context={**on_demand_context} + ) + assert [item["kind"] for item in unresolved["proposals"]] == [ + "steward_team_plan_preview" + ] + gaps_only = normalize_agent_response( + other_goal, + team_plan_context={ + "registered_agents_by_goal": {"some-other-goal": []}, + "supported_action_kinds": ["implement"], + }, + ) + # A Goal the host says has no registered Agents is a fact: the lane becomes + # a typed gap instead of the preview disappearing without explanation. + assert gaps_only["proposals"][0]["preview"]["gaps"] == [ + {"lane_id": "lane-alpha", "reason_code": "agent_not_registered"} + ] + # A malformed preview is dropped like any other proposal the normalizer # cannot accept, and the owner's answer text still arrives. malformed = {**envelope, "proposals": [_plan(kind="not_a_plan")]} @@ -189,3 +239,62 @@ def test_the_chat_normalizer_admits_a_preview_only_with_host_facts() -> None: assert normalize_agent_response(todo)["proposals"] == [ {"kind": "todo", "text": "Do one thing", "priority": "P2", "rationale": "why"} ] + + +def test_the_steward_turn_resolves_admission_facts_per_goal(tmp_path) -> None: + """The Turn owner hands the segment a lookup, not one Goal's facts.""" + + import json as _json + + from loopx.chat_runtime import ChatRuntimeController + from loopx.chat_store import ChatSessionStore + + registry_path = tmp_path / "registry.json" + registry_path.write_text( + _json.dumps( + { + "schema_version": "0.1", + "updated_at": "2026-01-01T00:00:00+00:00", + "goals": [ + { + "id": goal_id, + "domain": goal_id, + "status": "active-read-only", + "coordination": {"registered_agents": [agent_id]}, + } + for goal_id, agent_id in ( + ("authorized-goal", "agent-alpha"), + ("other-goal", "agent-beta"), + ) + ], + } + ), + encoding="utf-8", + ) + store = ChatSessionStore(tmp_path / "runtime") + runtime = ChatRuntimeController( + store=store, + codex_bin="missing-codex", + registry_path=registry_path, + manager_scope_resolver=lambda session: ["authorized-goal"], + ) + + # Only the manager channel proposes teams; every other channel is untouched. + assert runtime._team_plan_admission_context({"channel_id": "lark:topic"}) is None + + # The owner's own channel is not scoped to a subset of Goals. + owner_context = runtime._team_plan_admission_context({"channel_id": "manager"}) + resolve = owner_context["resolve_registered_agents"] + assert resolve("authorized-goal") == ["agent-alpha"] + assert resolve("other-goal") == ["agent-beta"] + assert resolve("no-such-goal") is None + assert "implement" in owner_context["supported_action_kinds"] + + # An external manager channel resolves only the Goals it is bound to, so a + # plan naming any other Goal cannot be validated at all. + external = runtime._team_plan_admission_context( + {"channel_id": "manager.external." + "a" * 24} + ) + external_resolve = external["resolve_registered_agents"] + assert external_resolve("authorized-goal") == ["agent-alpha"] + assert external_resolve("other-goal") is None