From e169ed9ade1633b1defeed0c0683e831e674806f Mon Sep 17 00:00:00 2001 From: Vesper Date: Thu, 20 Aug 2026 06:24:11 +0000 Subject: [PATCH 1/2] feat(engine): engine-probed capability manifest + pre-launch gate (issue #5 proposal 3) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Actors may declare required_capabilities from a controlled vocabulary (config.KNOWN_CAPABILITIES; unknown entries fail definition load). Before any claim or runtime spawn, the engine probes each required capability from the CLI itself — never adapter self-report — and refuses the launch when the probed manifest lacks one. v1 vocabulary: cli-version ( --version exits 0) and hermes-cli one-shot-source-tagging (chat --help shows -q/--source/ --pass-session-id, verified against real Hermes v0.20). The manifest (probe argv, exit, truncated output, full-output hash per capability) is recorded in the turn evidence as capability_manifest. Canary: a stub whose chat --help omits --source gets the launch refused with no claim, no spawn, no ledger commit. --- docs/technical-reference.md | 12 +++ examples/fakes/bin/fake-hermes | 9 +++ schemas/protocol.schema.json | 6 ++ schemas/runtime-evidence.schema.json | 33 ++++++++ src/multi_agent_dialogue/adapters/base.py | 86 +++++++++++++++++++++ src/multi_agent_dialogue/adapters/hermes.py | 20 +++++ src/multi_agent_dialogue/config.py | 35 +++++++++ src/multi_agent_dialogue/runner.py | 18 +++++ tests/test_config.py | 34 ++++++++ tests/test_real_contracts.py | 68 ++++++++++++++++ 10 files changed, 321 insertions(+) diff --git a/docs/technical-reference.md b/docs/technical-reference.md index e0144f3..cea48ed 100644 --- a/docs/technical-reference.md +++ b/docs/technical-reference.md @@ -289,6 +289,18 @@ all-rows behavior (every row is a main-conversation call there). The column check matches SQLite semantics case-insensitively, so a schema declaring `"Task"` still triggers the filter. +**Capability gate**: an actor may declare `required_capabilities` from +the controlled vocabulary (`cli-version`; hermes-cli: +`one-shot-source-tagging`). Before any claim or runtime spawn, the +engine probes each required capability from the CLI itself — `--version` +must exit 0; `chat --help` must show the `-q`/`--source`/ +`--pass-session-id` one-shot surface (verified against Hermes v0.20) — +and refuses the launch when the probed manifest lacks one. The adapter +only names what to probe; it never reports its own support. The +manifest (probe argv, exit, truncated output, full-output SHA-256 per +capability) is recorded in the turn's evidence as +`capability_manifest`. + **Message boundary**: every row in the matched session must carry a timestamp inside the invocation window — a row outside it means the session saw activity this launch cannot account for (late finalization diff --git a/examples/fakes/bin/fake-hermes b/examples/fakes/bin/fake-hermes index bc7e6a3..ab3dd94 100755 --- a/examples/fakes/bin/fake-hermes +++ b/examples/fakes/bin/fake-hermes @@ -121,6 +121,15 @@ def main() -> int: if argv == ["--version"]: print("fake-hermes 1.1.0 (fixture)") return 0 + if argv == ["chat", "--help"] or argv == ["chat", "-h"]: + # Mirrors the real v0.20 one-shot surface (verified against + # `hermes chat --help`). + print( + "usage: hermes chat [-h] [-q QUERY | --query-file PATH] " + "[-m MODEL] [-Q] [--pass-session-id] [--source SOURCE] " + "[--resume SESSION_ID]" + ) + return 0 if not argv or argv[0] != "chat": print( "fake-hermes: error: this fixture only implements the real " diff --git a/schemas/protocol.schema.json b/schemas/protocol.schema.json index cc96da3..c6ff557 100644 --- a/schemas/protocol.schema.json +++ b/schemas/protocol.schema.json @@ -60,6 +60,12 @@ }, "expected_provider": {"type": "string", "minLength": 1}, "expected_model": {"type": "string", "minLength": 1}, + "required_capabilities": { + "type": "array", + "uniqueItems": true, + "items": {"type": "string", "enum": ["cli-version", "one-shot-source-tagging"]}, + "description": "Controlled capability vocabulary. Before any runtime spawn the engine probes each listed capability from the CLI itself (never adapter self-report) and refuses the launch when the probed manifest lacks one. cli-version: ' --version' exits 0. one-shot-source-tagging (hermes-cli): ' chat --help' shows -q/--source/--pass-session-id." + }, "settings": { "type": "object", "description": "Non-secret adapter settings. fable-session: command_name, project, registry, state_dir, tmux_prefix, plus tmux_command_name (default 'tmux') used ONLY on post-launch failure to kill-session/has-session the exact generated lane. hermes-cli: command_name, hermes_home. command: argv template plus a REQUIRED identity_verifier_argv (self-reported worker output is never identity proof). All: optional env, timeout_seconds." diff --git a/schemas/runtime-evidence.schema.json b/schemas/runtime-evidence.schema.json index c82b19b..411b78a 100644 --- a/schemas/runtime-evidence.schema.json +++ b/schemas/runtime-evidence.schema.json @@ -132,6 +132,39 @@ "type": "string" } } + }, + "capability_manifest": { + "type": "object", + "description": "Engine-probed adapter capability manifest, present when the actor declares required_capabilities. Each capability entry carries ok plus the probe record (argv, exit_status, truncated output, output_sha256 over the full output) or an error. Probed from the CLI by the engine, never adapter self-report.", + "additionalProperties": false, + "required": ["capabilities", "probed_at"], + "properties": { + "capabilities": { + "type": "object", + "propertyNames": {"enum": ["cli-version", "one-shot-source-tagging"]}, + "additionalProperties": { + "type": "object", + "additionalProperties": false, + "required": ["ok"], + "properties": { + "ok": {"type": "boolean"}, + "error": {"type": "string"}, + "probe": { + "type": "object", + "additionalProperties": false, + "required": ["argv"], + "properties": { + "argv": {"type": "array", "items": {"type": "string"}}, + "exit_status": {"type": "integer"}, + "output": {"type": "string"}, + "output_sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"} + } + } + } + } + }, + "probed_at": {"type": "string"} + } } }, "examples": [ diff --git a/src/multi_agent_dialogue/adapters/base.py b/src/multi_agent_dialogue/adapters/base.py index ebe0f6c..1a9f825 100644 --- a/src/multi_agent_dialogue/adapters/base.py +++ b/src/multi_agent_dialogue/adapters/base.py @@ -200,6 +200,87 @@ def version_probe_argv(self, context: PrepareContext) -> list[str] | None: """ return None + def capability_probes( + self, context: PrepareContext + ) -> dict[str, tuple[list[str], "object"]]: + """Extra CLI capability probes: name -> (argv, check). + + ``check`` receives (full_output, returncode) and returns bool. + Every probe is run by the engine against the real CLI — the + adapter only names WHAT to probe, never reports its own support. + """ + return {} + + def _run_capability_probe( + self, + argv: list[str], + check, + context: PrepareContext, + where: str, + ) -> dict: + """Run one bounded capability probe under the actor's env. + + cwd is the dialogue directory (always exists) so the pre-launch + gate can probe before the turn's work directory is created. + """ + try: + probe_env = substitute_env( + context.actor.settings.get("env"), context.placeholders(), where + ) + result = run_command( + argv, env=probe_env, cwd=context.dialogue_dir, + timeout=VERSION_PROBE_TIMEOUT_SECONDS, + what=f"{where}: capability probe {' '.join(argv[:2])}", + ) + except AdapterError as exc: + return {"ok": False, "error": str(exc), "probe": {"argv": list(argv)}} + raw_output = (result.stdout or result.stderr or "") + try: + ok = bool(check(raw_output, result.returncode)) + except Exception as exc: # a broken check must fail closed, not crash + return { + "ok": False, + "error": f"capability check failed: {exc}", + "probe": {"argv": list(argv)}, + } + return { + "ok": ok, + "probe": { + "argv": list(argv), + "exit_status": result.returncode, + "output": raw_output.strip()[:500], + "output_sha256": artifacts.sha256_bytes( + raw_output.encode("utf-8") + ), + }, + } + + def probe_capabilities(self, context: PrepareContext) -> dict: + """Build the capability manifest by probing the CLI itself. + + ``cli-version`` is always probed when the adapter names a version + argv; adapter-specific probes come from ``capability_probes``. + """ + where = f"actor {context.actor.actor_id!r} ({self.name})" + capabilities: dict[str, dict] = {} + try: + version_argv = self.version_probe_argv(context) + except AdapterError as exc: + capabilities["cli-version"] = {"ok": False, "error": str(exc)} + else: + if version_argv is not None: + capabilities["cli-version"] = self._run_capability_probe( + version_argv, + lambda output, rc: rc == 0, + context, + where, + ) + for name, (argv, check) in self.capability_probes(context).items(): + capabilities[name] = self._run_capability_probe( + argv, check, context, where + ) + return {"capabilities": capabilities, "probed_at": utc_now()} + def cli_version_evidence(self, context: PrepareContext) -> dict | None: """Probe the adapter CLI version for the evidence record. @@ -270,6 +351,11 @@ def base_evidence(self, context: PrepareContext, *, provider: str, model: str, cli_version = self.cli_version_evidence(context) if cli_version is not None: record["cli_version"] = cli_version + if context.actor.required_capabilities: + # The manifest is probed from the CLI at execution time and + # rides inside the hashed evidence record; the pre-launch + # gate probed the same surface before any spawn. + record["capability_manifest"] = self.probe_capabilities(context) return record diff --git a/src/multi_agent_dialogue/adapters/hermes.py b/src/multi_agent_dialogue/adapters/hermes.py index e3695c2..abf22f2 100644 --- a/src/multi_agent_dialogue/adapters/hermes.py +++ b/src/multi_agent_dialogue/adapters/hermes.py @@ -112,6 +112,26 @@ def version_probe_argv(self, context: PrepareContext) -> list[str]: command_name = require_str_setting(settings, "command_name", where) return [command_name, "--version"] + def capability_probes( + self, context: PrepareContext + ) -> dict[str, tuple[list[str], "object"]]: + settings = context.actor.settings + where = f"actor {context.actor.actor_id!r} (hermes)" + command_name = require_str_setting(settings, "command_name", where) + return { + # The one-shot contract surface, probed from the real CLI's + # own chat --help (verified against Hermes v0.20). + "one-shot-source-tagging": ( + [command_name, "chat", "--help"], + lambda output, rc: ( + rc == 0 + and "--source" in output + and "--pass-session-id" in output + and "-q" in output + ), + ) + } + def _argv(self, command_name: str, prompt: str, source: str) -> tuple[str, ...]: return ( command_name, diff --git a/src/multi_agent_dialogue/config.py b/src/multi_agent_dialogue/config.py index 8f5869e..ff99c95 100644 --- a/src/multi_agent_dialogue/config.py +++ b/src/multi_agent_dialogue/config.py @@ -22,6 +22,19 @@ # names it. The latest entry is the version adapters write per turn. SUPPORTED_EVIDENCE_VERSIONS = (1,) +# Controlled capability vocabulary for actor `required_capabilities`. +# Every entry names a capability the ENGINE probes from the CLI itself +# (never adapter self-report); the pre-launch gate refuses a turn whose +# probed manifest lacks a required capability. Add entries only with a +# probe verified against the real CLI. +KNOWN_CAPABILITIES = ( + # ` --version` exits 0. + "cli-version", + # hermes-cli: ` chat --help` shows the one-shot contract + # surface (-q, --source, --pass-session-id). + "one-shot-source-tagging", +) + DEFAULT_OWNER_DECISIONS = ("APPROVE", "REJECT", "NEED_MORE_EVIDENCE") _ACTOR_KEYS = { @@ -31,6 +44,7 @@ "expected_provider", "expected_model", "settings", + "required_capabilities", } _TURN_KEYS = {"round_id", "actor_id", "purpose", "artifact_kind", "word_limit"} @@ -47,6 +61,9 @@ class Actor: expected_provider: str expected_model: str settings: dict[str, Any] = field(default_factory=dict) + # Capabilities (controlled vocabulary, KNOWN_CAPABILITIES) the + # engine must probe from the CLI before any runtime spawn. + required_capabilities: tuple[str, ...] = () @dataclass(frozen=True) @@ -128,6 +145,23 @@ def _parse_actor(raw: Any, position: int, errors: list[str]) -> Actor | None: if not isinstance(settings, dict): errors.append(f"{where}: settings must be an object") settings = {} + capabilities_raw = raw.get("required_capabilities", []) + capabilities: tuple[str, ...] = () + if ( + not isinstance(capabilities_raw, list) + or not all(isinstance(item, str) for item in capabilities_raw) + ): + errors.append(f"{where}: required_capabilities must be a list of strings") + else: + unknown = [item for item in capabilities_raw if item not in KNOWN_CAPABILITIES] + if unknown: + errors.append( + f"{where}: unknown required_capabilities {unknown}; the " + f"controlled vocabulary is {list(KNOWN_CAPABILITIES)}" + ) + if len(set(capabilities_raw)) != len(capabilities_raw): + errors.append(f"{where}: required_capabilities contains duplicates") + capabilities = tuple(capabilities_raw) if actor_id and _looks_secret(canonical_json(settings)): errors.append(f"{where}: settings must not embed credential material") if errors: @@ -140,6 +174,7 @@ def _parse_actor(raw: Any, position: int, errors: list[str]) -> Actor | None: expected_provider=provider, expected_model=model, settings=settings, + required_capabilities=capabilities, ) diff --git a/src/multi_agent_dialogue/runner.py b/src/multi_agent_dialogue/runner.py index f28378b..16603f6 100644 --- a/src/multi_agent_dialogue/runner.py +++ b/src/multi_agent_dialogue/runner.py @@ -278,6 +278,24 @@ def launch(dialogue: engine.Dialogue, actor_id: str, timeout: int | None = None) except adapters.AdapterError as exc: raise engine.ProtocolError(str(exc)) from exc + # Capability gate: when the definition requires capabilities, the + # engine probes them from the CLI itself and refuses the launch + # before any claim or runtime spawn. + required = context.actor.required_capabilities + if required: + manifest = adapter.probe_capabilities(context) + missing = [ + name + for name in required + if not manifest["capabilities"].get(name, {}).get("ok") + ] + if missing: + raise engine.ProtocolError( + f"actor {actor_id!r} requires capabilities the probed CLI " + f"does not show: {missing}; launch refused before any " + "runtime spawn" + ) + dialogue.claim(actor_id) # Forensics are best-effort: a receipt that cannot be written must # never hold a claim hostage or block a turn the ledger can prove. diff --git a/tests/test_config.py b/tests/test_config.py index dbd1d8f..85550c8 100644 --- a/tests/test_config.py +++ b/tests/test_config.py @@ -238,5 +238,39 @@ def test_duplicates_rejected(self) -> None: config.parse_definition(raw) +class RequiredCapabilitiesDefinitionTests(unittest.TestCase): + """Actor required_capabilities use the controlled vocabulary only.""" + + def test_default_is_empty(self) -> None: + definition = config.parse_definition(support.two_actor_definition()) + self.assertEqual(definition.actor("worker-a").required_capabilities, ()) + + def test_known_capability_accepted(self) -> None: + raw = support.two_actor_definition() + raw["actors"][0]["required_capabilities"] = ["cli-version"] + definition = config.parse_definition(raw) + self.assertEqual( + definition.actor("worker-a").required_capabilities, ("cli-version",) + ) + + def test_unknown_capability_rejected(self) -> None: + raw = support.two_actor_definition() + raw["actors"][0]["required_capabilities"] = ["teleport"] + with self.assertRaisesRegex(config.ConfigError, "unknown required_capabilities"): + config.parse_definition(raw) + + def test_duplicate_capabilities_rejected(self) -> None: + raw = support.two_actor_definition() + raw["actors"][0]["required_capabilities"] = ["cli-version", "cli-version"] + with self.assertRaisesRegex(config.ConfigError, "duplicates"): + config.parse_definition(raw) + + def test_non_string_capability_rejected(self) -> None: + raw = support.two_actor_definition() + raw["actors"][0]["required_capabilities"] = [1] + with self.assertRaisesRegex(config.ConfigError, "list of strings"): + config.parse_definition(raw) + + if __name__ == "__main__": unittest.main() diff --git a/tests/test_real_contracts.py b/tests/test_real_contracts.py index 0a375f3..0bd931d 100644 --- a/tests/test_real_contracts.py +++ b/tests/test_real_contracts.py @@ -1437,5 +1437,73 @@ def test_report_is_read_only(self) -> None: self.assertEqual(status, "", "report leaves the worktree untouched") +class CapabilityGateTests(HermesTestCase): + """Issue #5 proposal 3: the engine probes required capabilities from + the CLI itself and refuses the launch before any runtime spawn.""" + + def _dialogue_with_capabilities(self, command_name: str, + capabilities: list[str], + directory: str) -> engine.Dialogue: + raw = self.definition_raw() + raw["actors"][0]["settings"]["command_name"] = command_name + raw["actors"][0]["required_capabilities"] = capabilities + definition = config.parse_definition(raw) + return engine.init_dialogue(definition, self.base / directory) + + def test_capable_cli_passes_gate_and_records_manifest(self) -> None: + dialogue = self._dialogue_with_capabilities( + HERMES, ["cli-version", "one-shot-source-tagging"], "dialogue-ok" + ) + runner.launch(dialogue, "hermes-north") + manifest = self.evidence_for(dialogue, 0)["capability_manifest"] + capabilities = manifest["capabilities"] + self.assertTrue(capabilities["cli-version"]["ok"]) + self.assertTrue(capabilities["one-shot-source-tagging"]["ok"]) + probe = capabilities["one-shot-source-tagging"]["probe"] + self.assertEqual(probe["argv"], [HERMES, "chat", "--help"]) + self.assertEqual(len(probe["output_sha256"]), 64) + + def test_missing_capability_refuses_launch_before_any_spawn(self) -> None: + # The proposal's canary: the probed CLI lacks a required field, + # so the launch is refused with NO claim and NO runtime spawn. + stub = self.base / "hermes-no-source" + stub.write_text( + "#!/bin/sh\n" + 'if [ "$1" = "--version" ]; then echo "stub-hermes 1.0"; exit 0; fi\n' + 'if [ "$1" = "chat" ] && [ "$2" = "--help" ]; then\n' + ' echo "usage: hermes chat [-h] [-q QUERY] [-Q]"\n' + " exit 0\n" + "fi\n" + f'exec "{HERMES}" "$@"\n', + encoding="utf-8", + ) + os.chmod(stub, 0o755) + dialogue = self._dialogue_with_capabilities( + str(stub), ["one-shot-source-tagging"], "dialogue-refused" + ) + with self.assertRaisesRegex(engine.ProtocolError, "capabilities"): + runner.launch(dialogue, "hermes-north") + self.assertFalse(self.marker.exists(), "no runtime process may spawn") + self.assertIsNone(dialogue.state()["claim"]) + self.assertEqual(dialogue.state()["turn_index"], 0) + + def test_failing_version_probe_fails_the_gate(self) -> None: + stub = self.base / "hermes-broken-version" + stub.write_text( + "#!/bin/sh\n" + 'if [ "$1" = "--version" ]; then exit 1; fi\n' + f'exec "{HERMES}" "$@"\n', + encoding="utf-8", + ) + os.chmod(stub, 0o755) + dialogue = self._dialogue_with_capabilities( + str(stub), ["cli-version"], "dialogue-broken-version" + ) + with self.assertRaisesRegex(engine.ProtocolError, "cli-version"): + runner.launch(dialogue, "hermes-north") + self.assertFalse(self.marker.exists()) + self.assertEqual(dialogue.state()["turn_index"], 0) + + if __name__ == "__main__": unittest.main() From 9d0004c7f363de2a64599ca342accf168a85adbc Mon Sep 17 00:00:00 2001 From: Vesper Date: Thu, 20 Aug 2026 06:36:13 +0000 Subject: [PATCH 2/2] fix(engine): review hardening for the capability gate - gate wraps probe failures into ProtocolError (no raw AdapterError); - the evidence records the SAME manifest that gated the launch (runner attaches it to the frozen context via dataclasses.replace), so no divergent re-probe and no probe failure after completion; - hook AdapterError inside probe_capabilities is caught and recorded as manifest hook_error instead of crashing the gate; - flag checks use exact-token regexes (-q no longer matches --query, --source no longer matches --source-map) with a lookalike-flags test; - probe records gain output_truncated for cli_version parity; - probe annotation typed Callable[[str, int], bool]; - schema capability enums are pinned to config.KNOWN_CAPABILITIES by a cross-check test so they cannot drift. --- schemas/runtime-evidence.schema.json | 6 ++- src/multi_agent_dialogue/adapters/base.py | 58 +++++++++++++++------ src/multi_agent_dialogue/adapters/hermes.py | 16 ++++-- src/multi_agent_dialogue/runner.py | 8 ++- tests/test_config.py | 23 ++++++++ tests/test_real_contracts.py | 24 +++++++++ 6 files changed, 111 insertions(+), 24 deletions(-) diff --git a/schemas/runtime-evidence.schema.json b/schemas/runtime-evidence.schema.json index 411b78a..2fe84b4 100644 --- a/schemas/runtime-evidence.schema.json +++ b/schemas/runtime-evidence.schema.json @@ -157,13 +157,15 @@ "argv": {"type": "array", "items": {"type": "string"}}, "exit_status": {"type": "integer"}, "output": {"type": "string"}, - "output_sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"} + "output_sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"}, + "output_truncated": {"type": "boolean"} } } } } }, - "probed_at": {"type": "string"} + "probed_at": {"type": "string"}, + "hook_error": {"type": "string"} } } }, diff --git a/src/multi_agent_dialogue/adapters/base.py b/src/multi_agent_dialogue/adapters/base.py index 1a9f825..d524524 100644 --- a/src/multi_agent_dialogue/adapters/base.py +++ b/src/multi_agent_dialogue/adapters/base.py @@ -18,6 +18,7 @@ import os import subprocess from abc import ABC, abstractmethod +from collections.abc import Callable from dataclasses import dataclass, field from datetime import datetime, timezone from pathlib import Path @@ -47,6 +48,10 @@ class PrepareContext: task_file: Path turn_file: Path evidence_file: Path + # The manifest that gated this launch, when the actor declares + # required_capabilities; runner attaches it via dataclasses.replace + # so the evidence records exactly what gated — never a re-probe. + capability_manifest: dict | None = None def placeholders(self) -> dict[str, str]: return { @@ -202,7 +207,7 @@ def version_probe_argv(self, context: PrepareContext) -> list[str] | None: def capability_probes( self, context: PrepareContext - ) -> dict[str, tuple[list[str], "object"]]: + ) -> dict[str, tuple[list[str], Callable[[str, int], bool]]]: """Extra CLI capability probes: name -> (argv, check). ``check`` receives (full_output, returncode) and returns bool. @@ -243,17 +248,18 @@ def _run_capability_probe( "error": f"capability check failed: {exc}", "probe": {"argv": list(argv)}, } - return { - "ok": ok, - "probe": { - "argv": list(argv), - "exit_status": result.returncode, - "output": raw_output.strip()[:500], - "output_sha256": artifacts.sha256_bytes( - raw_output.encode("utf-8") - ), - }, + stripped = raw_output.strip() + probe_record: dict = { + "argv": list(argv), + "exit_status": result.returncode, + "output": stripped[:500], + # The hash attests the FULL probe output, computed before + # the 500-char storage truncation (cli_version parity). + "output_sha256": artifacts.sha256_bytes(raw_output.encode("utf-8")), } + if len(stripped) > 500: + probe_record["output_truncated"] = True + return {"ok": ok, "probe": probe_record} def probe_capabilities(self, context: PrepareContext) -> dict: """Build the capability manifest by probing the CLI itself. @@ -275,11 +281,26 @@ def probe_capabilities(self, context: PrepareContext) -> dict: context, where, ) - for name, (argv, check) in self.capability_probes(context).items(): + try: + extra_probes = self.capability_probes(context) + except AdapterError as exc: + # A broken hook must not crash the gate: its capabilities + # simply never show up as ok, and the error is recorded. + extra_probes = {} + hook_error = str(exc) + else: + hook_error = None + for name, (argv, check) in extra_probes.items(): capabilities[name] = self._run_capability_probe( argv, check, context, where ) - return {"capabilities": capabilities, "probed_at": utc_now()} + manifest: dict = { + "capabilities": capabilities, + "probed_at": utc_now(), + } + if hook_error is not None: + manifest["hook_error"] = hook_error + return manifest def cli_version_evidence(self, context: PrepareContext) -> dict | None: """Probe the adapter CLI version for the evidence record. @@ -352,10 +373,13 @@ def base_evidence(self, context: PrepareContext, *, provider: str, model: str, if cli_version is not None: record["cli_version"] = cli_version if context.actor.required_capabilities: - # The manifest is probed from the CLI at execution time and - # rides inside the hashed evidence record; the pre-launch - # gate probed the same surface before any spawn. - record["capability_manifest"] = self.probe_capabilities(context) + # The evidence records the SAME manifest that gated the + # launch (attached to the context by the runner); a fresh + # probe only happens if execute ran without the gate. + manifest = context.capability_manifest + if manifest is None: + manifest = self.probe_capabilities(context) + record["capability_manifest"] = manifest return record diff --git a/src/multi_agent_dialogue/adapters/hermes.py b/src/multi_agent_dialogue/adapters/hermes.py index abf22f2..b29cfbc 100644 --- a/src/multi_agent_dialogue/adapters/hermes.py +++ b/src/multi_agent_dialogue/adapters/hermes.py @@ -40,9 +40,11 @@ from __future__ import annotations +import re import secrets import sqlite3 import time +from collections.abc import Callable from pathlib import Path from .. import artifacts @@ -114,10 +116,16 @@ def version_probe_argv(self, context: PrepareContext) -> list[str]: def capability_probes( self, context: PrepareContext - ) -> dict[str, tuple[list[str], "object"]]: + ) -> dict[str, tuple[list[str], Callable[[str, int], bool]]]: settings = context.actor.settings where = f"actor {context.actor.actor_id!r} (hermes)" command_name = require_str_setting(settings, "command_name", where) + + def flag(name: str) -> re.Pattern: + # Exact-flag match: -q must not match --query, --source must + # not match --source-map. + return re.compile(r"(? None: with self.assertRaisesRegex(config.ConfigError, "duplicates"): config.parse_definition(raw) + def test_schema_capability_enums_match_config_vocabulary(self) -> None: + # The controlled vocabulary lives in config.KNOWN_CAPABILITIES; + # both schema enums must derive from it, never drift. + proto = json.loads( + (support.REPO_ROOT / "schemas" / "protocol.schema.json").read_text( + encoding="utf-8" + ) + ) + actor_props = proto["properties"]["actors"]["items"]["properties"] + schema_caps = actor_props["required_capabilities"]["items"]["enum"] + self.assertEqual( + sorted(schema_caps), sorted(config.KNOWN_CAPABILITIES) + ) + ev_schema = json.loads( + (support.REPO_ROOT / "schemas" / "runtime-evidence.schema.json").read_text( + encoding="utf-8" + ) + ) + ev_caps = ev_schema["properties"]["capability_manifest"]["properties"][ + "capabilities" + ]["propertyNames"]["enum"] + self.assertEqual(sorted(ev_caps), sorted(config.KNOWN_CAPABILITIES)) + class RequiredCapabilitiesDefinitionTests(unittest.TestCase): """Actor required_capabilities use the controlled vocabulary only.""" diff --git a/tests/test_real_contracts.py b/tests/test_real_contracts.py index 0bd931d..af94fc6 100644 --- a/tests/test_real_contracts.py +++ b/tests/test_real_contracts.py @@ -1487,6 +1487,30 @@ def test_missing_capability_refuses_launch_before_any_spawn(self) -> None: self.assertIsNone(dialogue.state()["claim"]) self.assertEqual(dialogue.state()["turn_index"], 0) + def test_lookalike_flags_do_not_satisfy_the_probe(self) -> None: + # --source-map must not satisfy --source; -q must not match + # --query/--quiet — the probe asserts exact flags. + stub = self.base / "hermes-lookalike-flags" + stub.write_text( + "#!/bin/sh\n" + 'if [ "$1" = "--version" ]; then echo "stub-hermes 1.0"; exit 0; fi\n' + 'if [ "$1" = "chat" ] && [ "$2" = "--help" ]; then\n' + ' echo "usage: hermes chat [-h] [-q QUERY] [--source-map FILE] ' + '[--pass-session-id]"\n' + " exit 0\n" + "fi\n" + f'exec "{HERMES}" "$@"\n', + encoding="utf-8", + ) + os.chmod(stub, 0o755) + dialogue = self._dialogue_with_capabilities( + str(stub), ["one-shot-source-tagging"], "dialogue-lookalike" + ) + with self.assertRaisesRegex(engine.ProtocolError, "capabilities"): + runner.launch(dialogue, "hermes-north") + self.assertFalse(self.marker.exists()) + self.assertEqual(dialogue.state()["turn_index"], 0) + def test_failing_version_probe_fails_the_gate(self) -> None: stub = self.base / "hermes-broken-version" stub.write_text(