From eb12b2684a3b23b8b1503e063b80f15567f6068c Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Tue, 22 Sep 2026 13:33:26 +0800 Subject: [PATCH] feat(examples): launch configurable Luna DSH and Ark research teams Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../rfcs/loopx-overall-roadmap-v0.md | 10 +++ .../rfcs/loopx-overall-roadmap-v0.zh-CN.md | 7 ++ examples/managed-research-team/README.md | 76 +++++++++++++++++-- examples/managed-research-team/execution.py | 5 ++ .../managed-research-team/research_team.py | 56 +++++++++----- examples/managed-research-team/scenario.py | 20 ++++- examples/managed-research-team/server.py | 6 +- tests/test_managed_research_scenario.py | 31 ++++++++ tests/test_managed_research_team.py | 47 ++++++++++++ 9 files changed, 228 insertions(+), 30 deletions(-) diff --git a/docs/architecture/rfcs/loopx-overall-roadmap-v0.md b/docs/architecture/rfcs/loopx-overall-roadmap-v0.md index ecc20e1ac6..8b4abb2ed0 100644 --- a/docs/architecture/rfcs/loopx-overall-roadmap-v0.md +++ b/docs/architecture/rfcs/loopx-overall-roadmap-v0.md @@ -424,6 +424,16 @@ sessions, generic Agent creation, dynamic governed work derivation, complete inbox/queue/steer, authenticated remote authority and packaged frontend/Lark companion work remain R2/R3/R4/R6 boundaries. Existing Goals are not promoted. +The disposable example now also accepts `--team-size LUNA DSH ARK`: independent +Luna max Turns consume the initial filing, DSH consumes accepted analysis and +the correction, and Ark consumes accepted corrected evidence. Preparation +reuses machine credentials and creates only Goal-owned bindings; the DSH lead +uses the same delegation service and must adopt every configured result. This +removes hand-edited roster setup for the synthetic qualification route. It does +not close the release frontier: actual public-source research, a visual launch +control, original-conversation result return and member stop/recovery must be +qualified together before advertising the one-action research showcase. + An existing shell-capable coordinator uses `delegation list/operations/start/read/wait/resume` without replacing its session. Requester-scoped `operations` recovers durable work after context loss, independently rechecks accepted results and preserves unavailable branches and pagination; enabled MCP and newly tool-equipped Goal Chat use the same read model. Existing native threads retain their tool schema on resume. It starts no work and does not infer overall readiness from a display list. The example's `prepare` still only provisions isolated operator bindings. A Codex binding can now pass an exact model/reasoning effort into an independent resumable Turn Session and expose the same profile through preflight and planning; this remains separate from a native temporary child profile inside the parent execution. Next, feed actual execution/acceptance facts into existing R2 readiness, extend registration/runtime configuration for approved identity provisioning, and qualify original-request return/lead continuation. Unattended wake, full cross-host inbox/queue/steer and Lark parity remain separate requirements; an exact profile parameter does not promote G1/G3. The local Goal conversation now exposes that inventory on demand, with per-binding diff --git a/docs/architecture/rfcs/loopx-overall-roadmap-v0.zh-CN.md b/docs/architecture/rfcs/loopx-overall-roadmap-v0.zh-CN.md index e1d27983a3..a3e26f5954 100644 --- a/docs/architecture/rfcs/loopx-overall-roadmap-v0.zh-CN.md +++ b/docs/architecture/rfcs/loopx-overall-roadmap-v0.zh-CN.md @@ -369,6 +369,13 @@ Todo 完成入口分别执行当前 pinned 检查,accepted 返回读 canonical 完成。长期 attached 会话、通用 Agent 创建、动态受治理工作派生、完整 inbox/queue/steer、 认证远端权威与 packaged frontend/Lark 配套仍归 R2/R3/R4/R6;不晋升已有 Goal。 +隔离示例现在支持 `--team-size LUNA DSH ARK`:独立 Luna max Turn 分析原始财报, +DSH 采用已验收分析并处理修订,Ark 采用已验收的修订证据。准备阶段复用机器凭证, +仅生成 Goal 范围的执行绑定;DSH 协调员沿用同一委派服务,最终必须采用每位成员的 +产物。这消除了合成验收场景中手工修改成员配置的步骤,尚未关闭 release 缺口: +真实公开资料投研、可视化启动、原对话返回和成员停止/恢复,需要一起验证后才能 +宣传一键投研 showcase。 + 有 shell 能力的原 coordinator 可通过 `delegation list/operations/start/read/wait/resume` 调用已有执行 owner,无需替换会话。`operations` 从自身持久记录找回上下文丢失前的工作,重新核验 accepted,保留不可用分支与分页;已启用的 MCP 和新挂载工具的 Goal Chat 共用该读模型;已有原生线程恢复时保留原工具 schema。读取不启动工作,也不把展示列表当整体 readiness。合成示例 `prepare` 仍只准备隔离绑定。Codex binding 现可将精确 model/reasoning effort 送入独立且可续接的 Turn Session,并由同一 preflight/规划投影读回;这与父执行内部的原生临时 child profile 分开。下一步先将实际执行/验收事实接入已有 R2 readiness,再沿现有注册及 runtime 配置扩展经授权的身份创建,并验证原请求返回与主力继续推进。无人值守唤醒、完整跨宿主 inbox/queue/steer 和 Lark 等价仍分别验收,不因 profile 参数接通而晋升 G1/G3。 本地 Goal 对话现可按需读取该持久目录,并按绑定调用实际 Turn dry-run 与所选 diff --git a/examples/managed-research-team/README.md b/examples/managed-research-team/README.md index c8f7fa7c32..dd75ce46b5 100644 --- a/examples/managed-research-team/README.md +++ b/examples/managed-research-team/README.md @@ -23,8 +23,9 @@ uv sync --extra test --extra deepseek-harness uv pip install --python .venv/bin/python -e packages/loopx-ark-turn ``` -Set `ARK_API_KEY`, `ARK_MODEL_ID`, `ARK_ENVIRONMENT_ID` and `DEEPSEEK_API_KEY` -in the environment. The Ark Environment must already belong to the operator; +Set `ARK_API_KEY`, `ARK_MODEL_ID` and `ARK_ENVIRONMENT_ID` in the process +environment. DSH reuses the machine operator credential; `DEEPSEEK_API_KEY` +is an explicit environment override. The Ark Environment must already belong to the operator; this launcher never creates or deletes it. The local model defaults to `deepseek-v4-flash@high`; select another profile with `--dsh-model`. @@ -59,6 +60,65 @@ uv run --no-sync --extra test python -m loopx.cli \ --format json goal-acceptance verify --goal-id synthetic-managed-research --execute ``` +## Configure a Luna / DSH / Ark team + +Use `--team-size LUNA DSH ARK` to select positive counts for all three member +runtimes. Counts exclude the coordinator. For example, this single command +prepares the isolated Goal and starts a DSH coordinator with one member per +runtime: + +```bash +uv run --no-sync --extra test python examples/managed-research-team/research_team.py \ + run "$DEMO_ROOT" --team-size 1 1 1 \ + --model "$ARK_MODEL_ID" --environment-id "$ARK_ENVIRONMENT_ID" +``` + +Codex members use independent `gpt-5.6-luna@max` Turns and the machine's existing +Codex login. Install `codex` on `PATH` first. DSH uses the machine operator +credential or an explicit process environment override; no credential is +copied into the Goal. Ark uses the selected model and existing Environment. +This does not advertise a different DSH model version than the one the provider +actually serves. A generated binding is configuration, not a successful login +or model-availability check. `run` makes real, billable provider calls. + +Luna members analyze the initial filing. DSH members consume an accepted Luna +artifact and analyze the correction; Ark members check an accepted DSH +artifact. Predecessors are assigned round-robin within this sample graph. The +coordinator chooses execution order and questions through the existing +collaboration tools. The report must adopt **every** configured member, +including an extra Luna member without a downstream consumer. Missing canonical +completion or a changed artifact prevents report acceptance. Repeating the +same analysis with more models does not create independent source families. + +`--team-size 2 1 1` creates four member tasks. The same option works with +`prepare` and `prepare-chat`, which make **no model calls**. It cannot be combined +with `--topology`; omitting it preserves the existing local-led four-member +example. Zero/negative counts are rejected before creating a directory. More +members increase spend and can exhaust the isolated Goal's quota or lead's +20-minute deadline; this option is not a capacity or cost guarantee. + +Read `project/team.json`, `delegation-config.json` and canonical Todo state to +inspect the generated roster. The `validate-report` and acceptance commands +above recheck every configured dependency. The Chat route below consumes the +same generated configuration; it still requires explicit owner settings and +enabling, rather than silently starting from preparation. + +This remains a **synthetic acceptance example**, not a financial-research +showcase or proof of autonomous source discovery. An interrupted run is not +successful: use the existing delegation inventory/read/wait/resume operations +for the original executions. Do not rerun `run` against the same directory or +start replacements while their status is unknown. Pausing a Chat coordinator +does not stop already running members; retain their receipts and use the +individual provider's execution controls. Removing bindings prevents new +admission but does not cancel accepted work. + +中文:`--team-size 1 1 1` 表示一名 Luna max、一名 DSH、一名 Ark 成员,协调员 +另计。`run` 准备隔离团队并发起真实模型执行;`prepare`/`prepare-chat` 只准备。 +可改为 `2 1 1` 等正整数;每名成员都必须通过独立验收并被最终报告采用,不能用 +人数或注册成功代替协作证据。凭证复用机器配置,Goal 只持有分工和授权。此处是 +明确标注的合成财报验收场景,真实投研、可视化一键启动、团队级停止和宣传影片 +仍需分别验证。暂停协调员不会取消成员;运行中断时先恢复原执行,勿重复拉起。 + ## Goal Chat coordinator To use the local Goal conversation as the lead, prepare a new disposable team @@ -82,7 +142,7 @@ advances business phases. Pause while a member works, refresh the page, then continue to observe that original member's accepted result. Queue a correction for the next native turn or explicitly select inbox/steer. -This route returns the report in the conversation; the four member tasks have +This route returns the report in the conversation; all configured member tasks have independent canonical acceptance. It does **not** write `lead/report.json` or complete `todo_lead-report`. The `validate-report` command above applies to the DSH/Ark lead route, which has an explicitly bound report-writing tool. Both @@ -92,7 +152,7 @@ command above; do not infer report acceptance from a native completion label. 中文:用 `prepare-chat` 准备隔离团队,再启动上述本地 Chat。进入该 Goal 的 对话,原地选择 `lead`、已生成的执行配置和协调员额度,开启后让模型组织协作。 可在成员执行时暂停、刷新、恢复,检查成员结果仍回到原对话。此入口把综合报告 -返回对话;四个成员任务分别验收,报告 Todo 和整体 Goal 保留给所有者处理。 +返回对话;各个成员任务分别验收,报告 Todo 和整体 Goal 保留给所有者处理。 详见 [Goal 对话运行模式](../../docs/reference/goal-chat-continuation.md)。 ## Collaboration path @@ -121,7 +181,7 @@ its own analysis while members run. The nested cloud analyst still requests its local reviewer through the same service. No business phase argument is introduced. -After reading all four canonical completions and exact artifact hashes, the +After reading all configured canonical completions and exact artifact hashes, the lead writes `lead/report.json` with the fields described by `scenario.py` and the acceptance table below. Run `validate-report`, then complete the report through ordinary `todo complete --todo-id todo_lead-report --agent-id lead @@ -162,7 +222,7 @@ an additional route. It does not substitute for the primary local-led path. | Repost of issuer material | Same source | Old figures retained | One current-period source family; corrected repost is stale | `bootstrap.ts` creates only a fresh disposable canonical runtime and invokes -the production owner configuration API once. It binds four member criteria and +the production owner configuration API once. It binds each configured member criterion and one report criterion. Task instructions, the roster and verifier files are pinned; a member cannot change its own acceptance. Existing Goals are never promoted or rewritten by this bootstrap. @@ -170,8 +230,8 @@ promoted or rewritten by this bootstrap. Core delegation asks the TS acceptance owner for the exact task's criteria and runs those checks as its Turn validator. It then uses ordinary `todo complete`, which re-executes validation and atomically commits through the same TS owner. -The report separately checks all four canonical completions, current binding -guards, adopted hashes and financial conclusions. All five Todos may be done +The report separately checks all configured canonical completions, current binding +guards, adopted hashes and financial conclusions. All configured Todos may be done while the overall Goal remains active for its owner. A member's own peer conclusion is preserved. The delegation result independently diff --git a/examples/managed-research-team/execution.py b/examples/managed-research-team/execution.py index ea7e85c806..017b21e9d3 100644 --- a/examples/managed-research-team/execution.py +++ b/examples/managed-research-team/execution.py @@ -13,6 +13,11 @@ def host_arguments(root: Path, actor: str, revision: str, *, host: str, attempt: settings = json.loads((root / "settings.json").read_text()) coordinator = actor == "lead" workspace = root / "lead" if coordinator else root / actor / revision + if host == "codex": + # Delegation injects its existing native MCP tools. Model/effort are an + # independent Turn binding; no user profile or credential is copied. + return ["--host", "codex-cli", "--codex-model", "gpt-5.6-luna", + "--codex-reasoning-effort", "max", "--codex-sandbox", "workspace-write"] if host == "dsh": args = ["--host", "dsh", "--dsh-model", settings["dsh_model"], "--dsh-reasoning-effort", "high", "--dsh-home", str(root / "homes" / (actor + "-" + revision + "-" + str(attempt)))] diff --git a/examples/managed-research-team/research_team.py b/examples/managed-research-team/research_team.py index e76cd5d2e7..79d921947b 100644 --- a/examples/managed-research-team/research_team.py +++ b/examples/managed-research-team/research_team.py @@ -43,12 +43,13 @@ def cli(root: Path, *args: str, workspace: Path | None = None, timeout: int = 60 return result -def prepare(root: Path, provider: str = "file", topology: str = "cloud-led") -> None: +def prepare(root: Path, provider: str = "file", topology: str = "cloud-led", + team_size: tuple[int, int, int] | None = None) -> None: if root.exists(): raise ValueError("use_a_new_disposable_directory") + members = roster(topology, team_size) project = root / "project" project.mkdir(parents=True) - members = roster(topology) write(project / "team.json", members) (project / ".gitignore").write_text(".local/\nACTIVE_GOAL_STATE.md\n") (project / "README.md").write_text("Disposable synthetic research team.\n") @@ -91,17 +92,19 @@ def git(*args: str) -> None: "Read the assignment with read_assignment, or inspect the synthetic team/input files. Organize the registered " "members with list_execution_bindings/start_delegation/wait_delegation to analyze their authorized revisions. " "Use stable operation ids and collaboration_brief_v0 (purpose, context, constraints, inputs, acceptance, return_requirement). " - "Complete local-analyst before requesting cloud-reviewer, who must adopt its exact artifact. " - "Cloud-analyst is responsible for delegating its local-reviewer prerequisite through the same tools. " + "Inspect team.json: an upstream must pass acceptance before its consumer finishes. " + "A member with requester other than lead must be delegated by that requester, using parent_request_id. " "You can start independent branches concurrently. A running operation is not failure; wait for its original result. " "Read all final artifacts with read_accepted_evidence or canonical CLI readback. Decide questions and order yourself. Review their " - "accepted results, resolve differences, then write_report or lead/report.json with all four evidence hashes. " + "accepted results, resolve differences, then write_report or lead/report.json with every member evidence hash. " "Only return validated_progress after write_report confirms independent checks." if actor == "lead" else "Read TASK.md and DELEGATION.json, or use read_input/write_output. Read context and assess_request before working. " "Use list_execution_bindings to find any authorized child. If present, start_delegation to the child " "with a collaboration_brief_v0 and parent_request_id from DELEGATION.json; wait_delegation until accepted. " - "Then read_input again to obtain and adopt its exact artifact. Produce independently checked output.json for " + revision + "." + "Then read_input again to obtain and adopt its exact artifact. A Codex member can read local input.json and " + "write output.json directly, using the native loopx_delegation context and assess_request tools. " + "Produce independently checked output.json for " + revision + "." ) if actor != "lead": (root / actor / revision / "TASK.md").write_text(task(revision, text)) @@ -112,7 +115,7 @@ def git(*args: str) -> None: "validation_timeout_seconds": 5, "validation_files": pins}) bindings.append({"todo_id": identity, "criterion_ids": [actor + "-" + revision]}) write(root / "bootstrap.json", {"tasks": tasks, "document": { - "objective": "Deliver a revision-aware synthetic research report with four completed dependencies", + "objective": "Deliver a revision-aware synthetic research report with every configured dependency completed", "non_goals": ["Trading", "External research", "Owner approval of the whole Goal"], "criteria": criteria, "bindings": bindings, }}) @@ -152,28 +155,33 @@ def accepted_entry(worker: str, revision: str, output: dict) -> dict: def prepare_execution(root: Path, model: str, environment_id: str, dsh_model: str, - topology: str = "local-led") -> dict: + topology: str = "local-led", team_size: tuple[int, int, int] | None = None) -> dict: """Prepare a fresh operator fixture without starting a replacement lead.""" - prepare(root, topology=topology) + prepare(root, topology=topology, team_size=team_size) write(root / "settings.json", {"dsh_model": dsh_model, "ark_model": model, "environment_id": environment_id}) config = configure_delegations(root) return {"goal_id": GOAL, "agent_id": "lead", "registry": str(root / "registry.json"), "runtime_root": str(root / "runtime"), "execution_config": str(config), "workspace": str(root / "lead"), "execution_started": False, + "member_count": len(roster(topology, team_size)), "next_action": "Use delegation list/start/read/wait from the existing Agent session. " "Supply LOOPX_RESEARCH_DEMO_ROOT and the configured credentials when starting work. " "Independent task acceptance remains bound; prepare does not complete any task."} -def launch(root: Path, model: str, environment_id: str, dsh_model: str, topology: str = "local-led") -> dict: +def launch(root: Path, model: str, environment_id: str, dsh_model: str, topology: str = "local-led", + team_size: tuple[int, int, int] | None = None) -> dict: + roster(topology, team_size) + if team_size and not shutil.which("codex"): + raise ValueError("codex_cli_required_for_luna_members") if importlib.util.find_spec("deepseek_harness") is None: raise ValueError("install_loopx_deepseek_harness_extra_in_this_interpreter") if not os.environ.get("ARK_API_KEY") or not operator_provider_environ().get("DEEPSEEK_API_KEY"): raise ValueError("ARK_API_KEY_and_machine_or_environment_DEEPSEEK_credential_required") - prepare_execution(root, model, environment_id, dsh_model, topology) + prepare_execution(root, model, environment_id, dsh_model, topology, team_size) os.environ["LOOPX_RESEARCH_DEMO_ROOT"] = str(root) result = turn(root, "lead", "report", root / "lead", [sys.executable, str(HERE / "research_team.py"), "validate-report", str(root)], - host_arguments(root, "lead", "report", host="dsh" if topology == "local-led" else "ark"), 1200) + host_arguments(root, "lead", "report", host="ark" if topology == "cloud-led" else "dsh"), 1200) summary = {key: result.get(key) for key in ("status", "result_kind", "validation", "resume_turn_key", "error", "host_failure")} write(root / "lead-turn.json", summary) if result.get("status") == "committed" and result.get("result_kind") == "validated_progress": @@ -193,30 +201,42 @@ def main() -> None: p.add_argument("--model", default=os.environ.get("ARK_MODEL_ID")) p.add_argument("--environment-id", default=os.environ.get("ARK_ENVIRONMENT_ID")) p.add_argument("--dsh-model", default="deepseek-v4-flash") - p.add_argument("--topology", choices=["local-led", "cloud-led"], default="local-led") + p.add_argument("--topology", choices=["local-led", "cloud-led"], default=None) + p.add_argument("--team-size", nargs=3, type=int, metavar=("LUNA", "DSH", "ARK"), + help="Positive member counts, excluding the coordinator; uses a mixed dependency graph.") args = p.parse_args() + if args.team_size and args.topology: + p.error("--team-size selects the mixed graph; do not combine it with --topology") + if args.team_size and args.command not in {"prepare", "prepare-chat", "run"}: + p.error("--team-size is only valid when preparing or running a new team") + team_size = tuple(args.team_size) if args.team_size else None + topology = "mixed" if team_size else (args.topology or "local-led") + try: + roster(topology, team_size) + except ValueError as exc: + p.error(str(exc)) if args.command in {"prepare", "prepare-chat", "run"}: if not args.model or not args.environment_id: p.error("explicit model and existing environment required") if args.command == "prepare": print(json.dumps(prepare_execution(args.root.resolve(), args.model, args.environment_id, - args.dsh_model, args.topology))) + args.dsh_model, topology, team_size))) return if args.command == "prepare-chat": root = args.root.resolve() - prepare_execution(root, args.model, args.environment_id, args.dsh_model, args.topology) + prepare_execution(root, args.model, args.environment_id, args.dsh_model, topology, team_size) target = root / "project" / ".loopx" / "config" / "delegations.json" target.parent.mkdir(parents=True, exist_ok=True) shutil.copyfile(root / "delegation-config.json", target) with (root / "project" / "ACTIVE_GOAL_STATE.md").open("a") as stream: stream.write("\n## Objective\n\nOrganize the authorized members with loopx_collaboration. " - "Local-analyst must finish before cloud-reviewer adopts its exact evidence; " - "cloud-analyst delegates local-reviewer itself. Wait for independently accepted results. " + "Follow the prepared team.json dependency and requester graph; consume every configured member. " + "Wait for independently accepted results and adopt their exact artifacts. " "Return the corrected cash-flow comparison, source and period caveats, and exact dependency " "hashes in this conversation. The canonical report task and whole Goal remain for owner review.\n") print("Prepared Goal Chat team. Select lead and .loopx/config/delegations.json in LoopX mode settings.") return - result = launch(args.root.resolve(), args.model, args.environment_id, args.dsh_model, args.topology) + result = launch(args.root.resolve(), args.model, args.environment_id, args.dsh_model, topology, team_size) print(json.dumps(result)) if result.get("status") != "committed" or result.get("result_kind") != "validated_progress": raise SystemExit(1) diff --git a/examples/managed-research-team/scenario.py b/examples/managed-research-team/scenario.py index 6464d424f2..ddf963534f 100644 --- a/examples/managed-research-team/scenario.py +++ b/examples/managed-research-team/scenario.py @@ -9,7 +9,25 @@ REVISIONS = ("initial", "corrected") -def roster(topology: str) -> list[dict]: +def roster(topology: str, team_size: tuple[int, int, int] | None = None) -> list[dict]: + if team_size is not None: + if topology != "mixed" or len(team_size) != 3 or any(type(n) is not int or n < 1 for n in team_size): + raise ValueError("mixed_team_requires_positive_luna_dsh_ark_counts") + # Each layer adopts an exact predecessor. The lead must also consume every + # member, including extra analysts with no downstream reviewer assigned. + members = [] + previous = [] + for (host, revision), count in zip((("codex", "initial"), ("dsh", "corrected"), ("ark", "corrected")), team_size): + current = [] + for index in range(count): + worker = f"{host}-{index + 1}" + row = {"worker": worker, "revision": revision, "host": host} + if previous: + row["upstream"] = previous[index % len(previous)] + members.append(row) + current.append(worker + "/" + revision) + previous = current + return members if topology == "local-led": members = [{"worker": worker, "revision": revision, "host": host} for worker, revision, host in ( ("local-analyst", "initial", "dsh"), ("cloud-reviewer", "initial", "ark"), diff --git a/examples/managed-research-team/server.py b/examples/managed-research-team/server.py index 99a5890189..885d18f68d 100644 --- a/examples/managed-research-team/server.py +++ b/examples/managed-research-team/server.py @@ -33,12 +33,12 @@ def read_assignment() -> dict: path = root() return {"execution": "Use list_execution_bindings, then start_delegation for your bindings. " "Choose stable operation ids; wait_delegation returns running until finished. " - "Cloud-analyst delegates local-reviewer itself; cloud-reviewer consumes completed local-analyst. " - "Read the final four artifacts with read_accepted_evidence before write_report.", + "Follow each assignment upstream and requester; only its requester can delegate a member. " + "Read every final artifact with read_accepted_evidence before write_report.", "assignments": assignments(path), "inputs": [evidence(revision) for revision in REVISIONS], "objective": "Compare the initial and corrected evidence. Obtain an independently accepted result " "for each authorized worker/revision assignment. You choose questions/order; revise rejected work. " - "Use all four accepted artifacts in the report. Count source families corroborating " + "Use every configured member artifact in the report. Count source families corroborating " "CURRENT-period figures only, excluding historical comparison material. No trades or external information.", "report_fields": {"initial_normalized_fcf": "integer", "corrected_normalized_fcf": "integer", "revision_delta": "corrected minus initial", "growth_supported": "boolean", diff --git a/tests/test_managed_research_scenario.py b/tests/test_managed_research_scenario.py index fadebd048e..f008ce89b7 100644 --- a/tests/test_managed_research_scenario.py +++ b/tests/test_managed_research_scenario.py @@ -84,3 +84,34 @@ def test_report_rejects_broken_dependencies(tmp_path, mutation): (tmp_path / "lead" / "report.json").write_bytes(encoded(report)) with pytest.raises((ValueError, FileNotFoundError)): validate_report(tmp_path) + + +@pytest.mark.parametrize("counts", [(1, 1, 1), (3, 1, 2), (1, 3, 1)]) +def test_mixed_team_consumes_every_configured_member(tmp_path, counts): + from scenario import roster + + team = roster("mixed", counts) + assert len({row["worker"] for row in team}) == sum(counts) + assert [sum(row["host"] == host for row in team) for host in ("codex", "dsh", "ark")] == list(counts) + (tmp_path / "project").mkdir() + (tmp_path / "project" / "team.json").write_bytes(encoded(team)) + report = fixture(tmp_path) + assert validate_report(tmp_path)["revision_delta"] == -15 + # Extra workers cannot become decorative: omitting ANY of their accepted + # artifacts invalidates the aggregate, even when all numbers still agree. + for member in team: + identity = member["worker"] + "/" + member["revision"] + omitted = {**report, "dependencies": {k: v for k, v in report["dependencies"].items() if k != identity}} + (tmp_path / "lead" / "report.json").write_bytes(encoded(omitted)) + with pytest.raises(ValueError, match="lead_did_not_adopt_dependency"): + validate_report(tmp_path) + + +@pytest.mark.parametrize("counts", [(0, 1, 1), (-1, 1, 1), (True, 1, 1), (1, 1)]) +def test_invalid_counts_do_not_prepare_state(tmp_path, counts): + import research_team as demo + + root = tmp_path / "invalid" + with pytest.raises(ValueError, match="positive_luna_dsh_ark"): + demo.prepare(root, topology="mixed", team_size=counts) + assert not root.exists() diff --git a/tests/test_managed_research_team.py b/tests/test_managed_research_team.py index 7bf027aad2..b843e42b45 100644 --- a/tests/test_managed_research_team.py +++ b/tests/test_managed_research_team.py @@ -153,3 +153,50 @@ async def call(environment): assert not asyncio.run(call(env)).isError assert asyncio.run(call({**env, "LOOPX_TURN_TODO_ID": "todo_someone_else"})).isError assert not canonical_tasks(root)["todo_cloud-reviewer-initial"]["done"] + + +@pytest.mark.parametrize("command", ["prepare", "prepare-chat"]) +def test_mixed_team_cli_binds_profiles_and_canonical_acceptance(tmp_path, command): + root = tmp_path / "mixed" + result = subprocess.run([sys.executable, str(demo.HERE / "research_team.py"), command, str(root), + "--team-size", "2", "1", "1", "--model", "example-model", + "--environment-id", "example-environment"], capture_output=True, text=True) + assert result.returncode == 0, result.stderr + config = json.loads((root / "delegation-config.json").read_text()) + assert len(config["bindings"]) == 4 + actors = {row["agent_id"] for row in config["bindings"]} + assert actors == {"codex-1", "codex-2", "dsh-1", "ark-1"} + for binding in config["bindings"]: + assert binding["requesters"] == ["lead"] + assert Path(binding["workspace"]).is_relative_to(root) + if binding["agent_id"].startswith("codex-"): + arguments = binding["host_args"] + assert arguments[arguments.index("--codex-model") + 1] == "gpt-5.6-luna" + assert arguments[arguments.index("--codex-reasoning-effort") + 1] == "max" + if command == "prepare-chat": + assert json.loads((root / "project" / ".loopx/config/delegations.json").read_text()) == config + fixture(root) + # A matching downstream artifact is still unaccepted while its exact + # upstream canonical task remains open. Production TS owns the rejection. + with pytest.raises(RuntimeError, match="goal_acceptance_validation_rejected"): + demo.complete(root, "dsh-1", "corrected") + for actor, revision in [("codex-1", "initial"), ("dsh-1", "corrected"), ("ark-1", "corrected")]: + demo.complete(root, actor, revision) + with pytest.raises(RuntimeError, match="goal_acceptance_validation_rejected"): + demo.complete(root, "lead", "report") + demo.complete(root, "codex-2", "initial") + demo.complete(root, "lead", "report") + assert all(row["done"] for row in canonical_tasks(root).values()) + + +@pytest.mark.parametrize("options", [ + ["--team-size", "1", "0", "1"], + ["--team-size", "1", "1", "1", "--topology", "local-led"], +]) +def test_cli_rejects_invalid_or_conflicting_composition_before_writes(tmp_path, options): + root = tmp_path / "not-created" + result = subprocess.run([sys.executable, str(demo.HERE / "research_team.py"), "prepare", str(root), + "--model", "example-model", "--environment-id", "example-environment", *options], + capture_output=True, text=True) + assert result.returncode == 2 + assert not root.exists()