Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 12 additions & 0 deletions docs/technical-reference.md
Original file line number Diff line number Diff line change
Expand Up @@ -289,6 +289,18 @@ all-rows behavior (every row is a main-conversation call there). The
column check matches SQLite semantics case-insensitively, so a schema
declaring `"Task"` still triggers the filter.

**Capability gate**: an actor may declare `required_capabilities` from
the controlled vocabulary (`cli-version`; hermes-cli:
`one-shot-source-tagging`). Before any claim or runtime spawn, the
engine probes each required capability from the CLI itself — `--version`
must exit 0; `chat --help` must show the `-q`/`--source`/
`--pass-session-id` one-shot surface (verified against Hermes v0.20) —
and refuses the launch when the probed manifest lacks one. The adapter
only names what to probe; it never reports its own support. The
manifest (probe argv, exit, truncated output, full-output SHA-256 per
capability) is recorded in the turn's evidence as
`capability_manifest`.

**Message boundary**: every row in the matched session must carry a
timestamp inside the invocation window — a row outside it means the
session saw activity this launch cannot account for (late finalization
Expand Down
9 changes: 9 additions & 0 deletions examples/fakes/bin/fake-hermes
Original file line number Diff line number Diff line change
Expand Up @@ -121,6 +121,15 @@ def main() -> int:
if argv == ["--version"]:
print("fake-hermes 1.1.0 (fixture)")
return 0
if argv == ["chat", "--help"] or argv == ["chat", "-h"]:
# Mirrors the real v0.20 one-shot surface (verified against
# `hermes chat --help`).
print(
"usage: hermes chat [-h] [-q QUERY | --query-file PATH] "
"[-m MODEL] [-Q] [--pass-session-id] [--source SOURCE] "
"[--resume SESSION_ID]"
)
return 0
if not argv or argv[0] != "chat":
print(
"fake-hermes: error: this fixture only implements the real "
Expand Down
6 changes: 6 additions & 0 deletions schemas/protocol.schema.json
Original file line number Diff line number Diff line change
Expand Up @@ -60,6 +60,12 @@
},
"expected_provider": {"type": "string", "minLength": 1},
"expected_model": {"type": "string", "minLength": 1},
"required_capabilities": {
"type": "array",
"uniqueItems": true,
"items": {"type": "string", "enum": ["cli-version", "one-shot-source-tagging"]},
Comment thread
askclaw-vesper marked this conversation as resolved.
"description": "Controlled capability vocabulary. Before any runtime spawn the engine probes each listed capability from the CLI itself (never adapter self-report) and refuses the launch when the probed manifest lacks one. cli-version: '<cli> --version' exits 0. one-shot-source-tagging (hermes-cli): '<cli> chat --help' shows -q/--source/--pass-session-id."
},
"settings": {
"type": "object",
"description": "Non-secret adapter settings. fable-session: command_name, project, registry, state_dir, tmux_prefix, plus tmux_command_name (default 'tmux') used ONLY on post-launch failure to kill-session/has-session the exact generated lane. hermes-cli: command_name, hermes_home. command: argv template plus a REQUIRED identity_verifier_argv (self-reported worker output is never identity proof). All: optional env, timeout_seconds."
Expand Down
35 changes: 35 additions & 0 deletions schemas/runtime-evidence.schema.json
Original file line number Diff line number Diff line change
Expand Up @@ -132,6 +132,41 @@
"type": "string"
}
}
},
"capability_manifest": {
"type": "object",
"description": "Engine-probed adapter capability manifest, present when the actor declares required_capabilities. Each capability entry carries ok plus the probe record (argv, exit_status, truncated output, output_sha256 over the full output) or an error. Probed from the CLI by the engine, never adapter self-report.",
"additionalProperties": false,
"required": ["capabilities", "probed_at"],
"properties": {
"capabilities": {
"type": "object",
"propertyNames": {"enum": ["cli-version", "one-shot-source-tagging"]},
"additionalProperties": {
"type": "object",
"additionalProperties": false,
"required": ["ok"],
"properties": {
"ok": {"type": "boolean"},
"error": {"type": "string"},
"probe": {
"type": "object",
"additionalProperties": false,
"required": ["argv"],
"properties": {
"argv": {"type": "array", "items": {"type": "string"}},
"exit_status": {"type": "integer"},
"output": {"type": "string"},
"output_sha256": {"type": "string", "pattern": "^[0-9a-f]{64}$"},
"output_truncated": {"type": "boolean"}
}
}
}
}
},
"probed_at": {"type": "string"},
"hook_error": {"type": "string"}
}
}
},
"examples": [
Expand Down
110 changes: 110 additions & 0 deletions src/multi_agent_dialogue/adapters/base.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,7 @@
import os
import subprocess
from abc import ABC, abstractmethod
from collections.abc import Callable
from dataclasses import dataclass, field
from datetime import datetime, timezone
from pathlib import Path
Expand Down Expand Up @@ -47,6 +48,10 @@ class PrepareContext:
task_file: Path
turn_file: Path
evidence_file: Path
# The manifest that gated this launch, when the actor declares
# required_capabilities; runner attaches it via dataclasses.replace
# so the evidence records exactly what gated — never a re-probe.
capability_manifest: dict | None = None

def placeholders(self) -> dict[str, str]:
return {
Expand Down Expand Up @@ -200,6 +205,103 @@ def version_probe_argv(self, context: PrepareContext) -> list[str] | None:
"""
return None

def capability_probes(
self, context: PrepareContext
) -> dict[str, tuple[list[str], Callable[[str, int], bool]]]:
"""Extra CLI capability probes: name -> (argv, check).

``check`` receives (full_output, returncode) and returns bool.
Every probe is run by the engine against the real CLI — the
adapter only names WHAT to probe, never reports its own support.
"""
return {}

def _run_capability_probe(
self,
argv: list[str],
check,
context: PrepareContext,
where: str,
) -> dict:
"""Run one bounded capability probe under the actor's env.

cwd is the dialogue directory (always exists) so the pre-launch
gate can probe before the turn's work directory is created.
"""
try:
probe_env = substitute_env(
context.actor.settings.get("env"), context.placeholders(), where
)
result = run_command(
argv, env=probe_env, cwd=context.dialogue_dir,
timeout=VERSION_PROBE_TIMEOUT_SECONDS,
what=f"{where}: capability probe {' '.join(argv[:2])}",
)
except AdapterError as exc:
return {"ok": False, "error": str(exc), "probe": {"argv": list(argv)}}
raw_output = (result.stdout or result.stderr or "")
try:
ok = bool(check(raw_output, result.returncode))
except Exception as exc: # a broken check must fail closed, not crash
return {
"ok": False,
"error": f"capability check failed: {exc}",
"probe": {"argv": list(argv)},
}
stripped = raw_output.strip()
probe_record: dict = {
"argv": list(argv),
"exit_status": result.returncode,
"output": stripped[:500],
# The hash attests the FULL probe output, computed before
# the 500-char storage truncation (cli_version parity).
"output_sha256": artifacts.sha256_bytes(raw_output.encode("utf-8")),
}
if len(stripped) > 500:
probe_record["output_truncated"] = True
return {"ok": ok, "probe": probe_record}

def probe_capabilities(self, context: PrepareContext) -> dict:
"""Build the capability manifest by probing the CLI itself.

``cli-version`` is always probed when the adapter names a version
argv; adapter-specific probes come from ``capability_probes``.
"""
where = f"actor {context.actor.actor_id!r} ({self.name})"
capabilities: dict[str, dict] = {}
try:
version_argv = self.version_probe_argv(context)
except AdapterError as exc:
capabilities["cli-version"] = {"ok": False, "error": str(exc)}
else:
if version_argv is not None:
capabilities["cli-version"] = self._run_capability_probe(
version_argv,
lambda output, rc: rc == 0,
context,
where,
)
try:
extra_probes = self.capability_probes(context)
except AdapterError as exc:
# A broken hook must not crash the gate: its capabilities
# simply never show up as ok, and the error is recorded.
extra_probes = {}
hook_error = str(exc)
else:
hook_error = None
for name, (argv, check) in extra_probes.items():
capabilities[name] = self._run_capability_probe(
argv, check, context, where
)
manifest: dict = {
"capabilities": capabilities,
"probed_at": utc_now(),
}
if hook_error is not None:
manifest["hook_error"] = hook_error
return manifest

def cli_version_evidence(self, context: PrepareContext) -> dict | None:
"""Probe the adapter CLI version for the evidence record.

Expand Down Expand Up @@ -270,6 +372,14 @@ def base_evidence(self, context: PrepareContext, *, provider: str, model: str,
cli_version = self.cli_version_evidence(context)
if cli_version is not None:
record["cli_version"] = cli_version
if context.actor.required_capabilities:
# The evidence records the SAME manifest that gated the
# launch (attached to the context by the runner); a fresh
# probe only happens if execute ran without the gate.
manifest = context.capability_manifest
if manifest is None:
manifest = self.probe_capabilities(context)
record["capability_manifest"] = manifest
return record


Expand Down
28 changes: 28 additions & 0 deletions src/multi_agent_dialogue/adapters/hermes.py
Original file line number Diff line number Diff line change
Expand Up @@ -40,9 +40,11 @@

from __future__ import annotations

import re
import secrets
import sqlite3
import time
from collections.abc import Callable
from pathlib import Path

from .. import artifacts
Expand Down Expand Up @@ -112,6 +114,32 @@ def version_probe_argv(self, context: PrepareContext) -> list[str]:
command_name = require_str_setting(settings, "command_name", where)
return [command_name, "--version"]

def capability_probes(
self, context: PrepareContext
) -> dict[str, tuple[list[str], Callable[[str, int], bool]]]:
settings = context.actor.settings
where = f"actor {context.actor.actor_id!r} (hermes)"
command_name = require_str_setting(settings, "command_name", where)

def flag(name: str) -> re.Pattern:
# Exact-flag match: -q must not match --query, --source must
# not match --source-map.
return re.compile(r"(?<![\w-])" + re.escape(name) + r"(?![\w-])")

return {
# The one-shot contract surface, probed from the real CLI's
# own chat --help (verified against Hermes v0.20).
"one-shot-source-tagging": (
[command_name, "chat", "--help"],
lambda output, rc: (
rc == 0
and flag("-q").search(output) is not None
and flag("--source").search(output) is not None
and flag("--pass-session-id").search(output) is not None
),
)
}

def _argv(self, command_name: str, prompt: str, source: str) -> tuple[str, ...]:
return (
command_name,
Expand Down
35 changes: 35 additions & 0 deletions src/multi_agent_dialogue/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,19 @@
# names it. The latest entry is the version adapters write per turn.
SUPPORTED_EVIDENCE_VERSIONS = (1,)

# Controlled capability vocabulary for actor `required_capabilities`.
# Every entry names a capability the ENGINE probes from the CLI itself
# (never adapter self-report); the pre-launch gate refuses a turn whose
# probed manifest lacks a required capability. Add entries only with a
# probe verified against the real CLI.
KNOWN_CAPABILITIES = (
# `<cli> --version` exits 0.
"cli-version",
# hermes-cli: `<cli> chat --help` shows the one-shot contract
# surface (-q, --source, --pass-session-id).
"one-shot-source-tagging",
)

DEFAULT_OWNER_DECISIONS = ("APPROVE", "REJECT", "NEED_MORE_EVIDENCE")

_ACTOR_KEYS = {
Expand All @@ -31,6 +44,7 @@
"expected_provider",
"expected_model",
"settings",
"required_capabilities",
}
_TURN_KEYS = {"round_id", "actor_id", "purpose", "artifact_kind", "word_limit"}

Expand All @@ -47,6 +61,9 @@ class Actor:
expected_provider: str
expected_model: str
settings: dict[str, Any] = field(default_factory=dict)
# Capabilities (controlled vocabulary, KNOWN_CAPABILITIES) the
# engine must probe from the CLI before any runtime spawn.
required_capabilities: tuple[str, ...] = ()


@dataclass(frozen=True)
Expand Down Expand Up @@ -128,6 +145,23 @@ def _parse_actor(raw: Any, position: int, errors: list[str]) -> Actor | None:
if not isinstance(settings, dict):
errors.append(f"{where}: settings must be an object")
settings = {}
capabilities_raw = raw.get("required_capabilities", [])
capabilities: tuple[str, ...] = ()
if (
not isinstance(capabilities_raw, list)
or not all(isinstance(item, str) for item in capabilities_raw)
):
errors.append(f"{where}: required_capabilities must be a list of strings")
else:
unknown = [item for item in capabilities_raw if item not in KNOWN_CAPABILITIES]
if unknown:
errors.append(
f"{where}: unknown required_capabilities {unknown}; the "
f"controlled vocabulary is {list(KNOWN_CAPABILITIES)}"
)
if len(set(capabilities_raw)) != len(capabilities_raw):
errors.append(f"{where}: required_capabilities contains duplicates")
capabilities = tuple(capabilities_raw)
if actor_id and _looks_secret(canonical_json(settings)):
errors.append(f"{where}: settings must not embed credential material")
if errors:
Expand All @@ -140,6 +174,7 @@ def _parse_actor(raw: Any, position: int, errors: list[str]) -> Actor | None:
expected_provider=provider,
expected_model=model,
settings=settings,
required_capabilities=capabilities,
)


Expand Down
24 changes: 24 additions & 0 deletions src/multi_agent_dialogue/runner.py
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,7 @@

from __future__ import annotations

import dataclasses
import json
import os
import secrets
Expand Down Expand Up @@ -278,6 +279,29 @@ def launch(dialogue: engine.Dialogue, actor_id: str, timeout: int | None = None)
except adapters.AdapterError as exc:
raise engine.ProtocolError(str(exc)) from exc

# Capability gate: when the definition requires capabilities, the
# engine probes them from the CLI itself and refuses the launch
# before any claim or runtime spawn.
required = context.actor.required_capabilities
if required:
try:
manifest = adapter.probe_capabilities(context)
except adapters.AdapterError as exc:
raise engine.ProtocolError(str(exc)) from exc
missing = [
name
for name in required
if not manifest["capabilities"].get(name, {}).get("ok")
]
if missing:
raise engine.ProtocolError(
f"actor {actor_id!r} requires capabilities the probed CLI "
f"does not show: {missing}; launch refused before any "
"runtime spawn"
)
# The evidence must record exactly what gated the launch.
context = dataclasses.replace(context, capability_manifest=manifest)

dialogue.claim(actor_id)
# Forensics are best-effort: a receipt that cannot be written must
# never hold a claim hostage or block a turn the ledger can prove.
Expand Down
Loading