From c843dc5588c05d03b578a95f027aaf39f37ab6d7 Mon Sep 17 00:00:00 2001 From: song Date: Fri, 4 Sep 2026 19:55:53 +0800 Subject: [PATCH 01/13] feat(reliability-diagnostics): add L1 shadow-observer envelope, receipt, and projection contract Provider-neutral observer envelope (strict allowlist; control-shaped and raw-material-shaped fields rejected and classified), bounded crash-isolated intake with a stats record, append-only NDJSON ledger helpers, treatment integrity receipt with a total valid|degraded|quarantined|invalid status, a read-only stage/stall/repetition/recovery projection, and a deterministic DSH-shaped fixture. Tests assert the RFC contract semantics, not output. Signed-off-by: song --- .../reliability_diagnostics/__init__.py | 89 ++++ .../reliability_diagnostics/envelope.py | 391 ++++++++++++++ .../reliability_diagnostics/fixture.py | 126 +++++ .../reliability_diagnostics/intake.py | 263 ++++++++++ .../reliability_diagnostics/ledger.py | 86 ++++ .../reliability_diagnostics/projection.py | 171 +++++++ .../reliability_diagnostics/receipt.py | 210 ++++++++ .../test_reliability_diagnostics.py | 479 ++++++++++++++++++ 8 files changed, 1815 insertions(+) create mode 100644 loopx/capabilities/reliability_diagnostics/__init__.py create mode 100644 loopx/capabilities/reliability_diagnostics/envelope.py create mode 100644 loopx/capabilities/reliability_diagnostics/fixture.py create mode 100644 loopx/capabilities/reliability_diagnostics/intake.py create mode 100644 loopx/capabilities/reliability_diagnostics/ledger.py create mode 100644 loopx/capabilities/reliability_diagnostics/projection.py create mode 100644 loopx/capabilities/reliability_diagnostics/receipt.py create mode 100644 tests/capabilities/test_reliability_diagnostics.py diff --git a/loopx/capabilities/reliability_diagnostics/__init__.py b/loopx/capabilities/reliability_diagnostics/__init__.py new file mode 100644 index 0000000000..d93d900b81 --- /dev/null +++ b/loopx/capabilities/reliability_diagnostics/__init__.py @@ -0,0 +1,89 @@ +"""Reliability Diagnostics: the L1 shadow-observer contract.""" + +from .envelope import ( + CAPABILITY_ID, + CONTROL_FIELD_FAMILIES, + DSH_PROVIDER_ID, + ENVELOPE_FIELDS, + OBSERVER_ENVELOPE_SCHEMA_VERSION, + OBSERVER_STATS_SCHEMA_VERSION, + RAW_MATERIAL_FIELD_FAMILIES, + SOURCE_REF_FIELDS, + SUMMARY_FIELDS, + ClockSource, + EnvelopeRejection, + ObserverEnvelope, + ObserverEnvelopeError, + ObserverEventKind, + normalize_observer_envelope, +) +from .fixture import FIXTURE_GOAL_ID, dsh_fixture_records, run_dsh_fixture +from .intake import ( + DEFAULT_BUFFER_BOUND, + ObserverStats, + ShadowObserverIntake, + normalize_observer_stats, +) +from .ledger import ( + LEDGER_DIRNAME, + append_ledger_records, + ledger_path, + ledger_ref, + parse_ndjson_lines, + read_ledger_records, +) +from .projection import ( + DIAGNOSTIC_PROJECTION_SCHEMA_VERSION, + DiagnosticSignal, + DiagnosticStage, + build_diagnostic_projection, +) +from .receipt import ( + INTEGRITY_RECEIPT_SCHEMA_VERSION, + LedgerReading, + ReceiptReason, + ReceiptStatus, + build_integrity_receipt, + read_ledger, +) + +__all__ = [ + "CAPABILITY_ID", + "CONTROL_FIELD_FAMILIES", + "DEFAULT_BUFFER_BOUND", + "DIAGNOSTIC_PROJECTION_SCHEMA_VERSION", + "DSH_PROVIDER_ID", + "ENVELOPE_FIELDS", + "FIXTURE_GOAL_ID", + "INTEGRITY_RECEIPT_SCHEMA_VERSION", + "LEDGER_DIRNAME", + "OBSERVER_ENVELOPE_SCHEMA_VERSION", + "OBSERVER_STATS_SCHEMA_VERSION", + "RAW_MATERIAL_FIELD_FAMILIES", + "SOURCE_REF_FIELDS", + "SUMMARY_FIELDS", + "ClockSource", + "DiagnosticSignal", + "DiagnosticStage", + "EnvelopeRejection", + "LedgerReading", + "ObserverEnvelope", + "ObserverEnvelopeError", + "ObserverEventKind", + "ObserverStats", + "ReceiptReason", + "ReceiptStatus", + "ShadowObserverIntake", + "append_ledger_records", + "build_diagnostic_projection", + "build_integrity_receipt", + "dsh_fixture_records", + "ledger_path", + "ledger_ref", + "normalize_observer_envelope", + "normalize_observer_stats", + "parse_ndjson_lines", + "read_ledger", + "read_ledger_records", + "run_dsh_fixture", +] diff --git a/loopx/capabilities/reliability_diagnostics/envelope.py b/loopx/capabilities/reliability_diagnostics/envelope.py new file mode 100644 index 0000000000..b52829502c --- /dev/null +++ b/loopx/capabilities/reliability_diagnostics/envelope.py @@ -0,0 +1,391 @@ +"""Provider-neutral L1 shadow-observer envelope contract. + +An observer envelope is the only record a harness event source may hand to +LoopX reliability diagnostics. It is intentionally narrow: identity, a +monotonic sequence, a declared clock, one typed event kind, a compact summary +of tokens and counters, and id-only source references. Every other field is +rejected and classified, so control-shaped or raw-material-shaped records can +never enter the diagnostic ledger. +""" + +from __future__ import annotations + +import re +from collections.abc import Mapping +from dataclasses import dataclass, field +from datetime import datetime +from enum import StrEnum +from typing import Any + +from ...control_plane.runtime.public_safety import ( + normalize_public_safe_field_name, + validate_public_safe_value, +) +from ...session_runtime import SOURCE_ID_KEYS + +CAPABILITY_ID = "reliability-diagnostics" +DSH_PROVIDER_ID = "dsh-session-events" +OBSERVER_ENVELOPE_SCHEMA_VERSION = "reliability_observer_envelope_v0" +OBSERVER_STATS_SCHEMA_VERSION = "reliability_observer_stats_v0" + +IDENTITY_TOKEN_PATTERN = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.:-]{0,120}$") +SUMMARY_TOKEN_PATTERN = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_./:-]{0,79}$") +MAX_SEQUENCE = 2**53 + + +class ObserverEventKind(StrEnum): + """Compact, provider-neutral event kinds a shadow observer may report.""" + + SESSION_STARTED = "session_started" + TURN_STARTED = "turn_started" + TURN_ENDED = "turn_ended" + STEP_STARTED = "step_started" + STEP_ENDED = "step_ended" + USER_MESSAGE = "user_message" + TOOL_CALLED = "tool_called" + TOOL_COMPLETED = "tool_completed" + AGENT_STATUS = "agent_status" + AGENT_PRE_STEP = "agent_pre_step" + AGENT_ERROR = "agent_error" + SESSION_DISPOSED = "session_disposed" + UNSUPPORTED = "unsupported" + + +class ClockSource(StrEnum): + HARNESS_EVENT_TIME = "harness_event_time" + OBSERVER_WALL_CLOCK = "observer_wall_clock" + FIXTURE = "fixture" + + +class EnvelopeRejection(StrEnum): + """Typed reasons an envelope is refused; each maps to a receipt counter.""" + + SCHEMA_MISMATCH = "schema_mismatch" + CONTROL_FIELD_REJECTED = "control_field_rejected" + RAW_MATERIAL_FIELD_REJECTED = "raw_material_field_rejected" + UNSUPPORTED_FIELD_REJECTED = "unsupported_field_rejected" + IDENTITY_INVALID = "identity_invalid" + SEQUENCE_INVALID = "sequence_invalid" + CLOCK_INVALID = "clock_invalid" + EVENT_KIND_INVALID = "event_kind_invalid" + SUMMARY_INVALID = "summary_invalid" + SOURCE_REF_INVALID = "source_ref_invalid" + PUBLIC_SAFETY_VIOLATION = "public_safety_violation" + + +ENVELOPE_FIELDS = frozenset( + { + "schema_version", + "capability_id", + "provider_id", + "goal_id", + "session_id", + "agent_id", + "sequence", + "observed_at", + "clock", + "event_kind", + "summary", + "source_refs", + } +) +CLOCK_FIELDS = frozenset({"source", "uncertainty_ms"}) +SUMMARY_INTEGER_FIELDS = frozenset({"turn", "step"}) +SUMMARY_TOKEN_FIELDS = frozenset( + { + "reason", + "status", + "tool_name", + "error_class", + "source_event_type", + "message_source_kind", + } +) +SUMMARY_FIELDS = SUMMARY_INTEGER_FIELDS | SUMMARY_TOKEN_FIELDS +SOURCE_REF_FIELDS = (frozenset(SOURCE_ID_KEYS) - {"session_id"}) | {"event_seq", "message_id"} + +# Exact field families (after public-safe key normalization) that reveal an +# outbound control path. Their presence is a contract violation, never data. +CONTROL_FIELD_FAMILIES = frozenset( + { + "command", + "commands", + "send", + "prompt", + "inject", + "schedule", + "retry", + "stop", + "resume", + "pause", + "cancel", + "gate", + "gatedecision", + "toolcall", + "toolinvocation", + "workerstate", + "continuation", + "callback", + "endpoint", + "outboundendpoint", + } +) +# Exact field families that carry protected task content or local material. +RAW_MATERIAL_FIELD_FAMILIES = frozenset( + { + "transcript", + "messages", + "content", + "text", + "arguments", + "result", + "output", + "tooloutput", + "stdout", + "stderr", + "log", + "logs", + "trace", + "path", + "localpath", + "cwd", + "credential", + "credentials", + "token", + "secret", + } +) + + +class ObserverEnvelopeError(ValueError): + def __init__(self, reason: EnvelopeRejection, detail: str) -> None: + super().__init__(f"{reason.value}: {detail}") + self.reason = reason + self.detail = detail + + +@dataclass(frozen=True) +class ObserverClock: + source: ClockSource + uncertainty_ms: int + + def as_dict(self) -> dict[str, Any]: + return {"source": self.source.value, "uncertainty_ms": self.uncertainty_ms} + + +@dataclass(frozen=True) +class ObserverEnvelope: + provider_id: str + goal_id: str + session_id: str + sequence: int + observed_at: str + clock: ObserverClock + event_kind: ObserverEventKind + agent_id: str | None = None + summary: Mapping[str, int | str] = field(default_factory=dict) + source_refs: Mapping[str, str] = field(default_factory=dict) + + def as_dict(self) -> dict[str, Any]: + record: dict[str, Any] = { + "schema_version": OBSERVER_ENVELOPE_SCHEMA_VERSION, + "capability_id": CAPABILITY_ID, + "provider_id": self.provider_id, + "goal_id": self.goal_id, + "session_id": self.session_id, + "sequence": self.sequence, + "observed_at": self.observed_at, + "clock": self.clock.as_dict(), + "event_kind": self.event_kind.value, + "summary": dict(self.summary), + "source_refs": dict(self.source_refs), + } + if self.agent_id is not None: + record["agent_id"] = self.agent_id + return record + + +def classify_rejected_field(key: object) -> EnvelopeRejection: + normalized = normalize_public_safe_field_name(key).replace("_", "") + if normalized in CONTROL_FIELD_FAMILIES: + return EnvelopeRejection.CONTROL_FIELD_REJECTED + if normalized in RAW_MATERIAL_FIELD_FAMILIES: + return EnvelopeRejection.RAW_MATERIAL_FIELD_REJECTED + return EnvelopeRejection.UNSUPPORTED_FIELD_REJECTED + + +def _reject_unknown_fields( + value: Mapping[str, Any], *, allowed: frozenset[str], context: str +) -> None: + unknown = [str(key) for key in value if str(key) not in allowed] + if not unknown: + return + reasons = sorted( + (classify_rejected_field(key) for key in unknown), + key=lambda reason: ( + reason is not EnvelopeRejection.CONTROL_FIELD_REJECTED, + reason is not EnvelopeRejection.RAW_MATERIAL_FIELD_REJECTED, + ), + ) + raise ObserverEnvelopeError( + reasons[0], f"{context} carries {len(unknown)} unsupported field(s)" + ) + + +def _identity(value: Any, *, name: str, optional: bool = False) -> str | None: + if value is None and optional: + return None + if not isinstance(value, str) or not IDENTITY_TOKEN_PATTERN.match(value): + raise ObserverEnvelopeError( + EnvelopeRejection.IDENTITY_INVALID, f"{name} must be an identity token" + ) + return value + + +def _sequence(value: Any) -> int: + if isinstance(value, bool) or not isinstance(value, int) or not 0 <= value < MAX_SEQUENCE: + raise ObserverEnvelopeError( + EnvelopeRejection.SEQUENCE_INVALID, + "sequence must be a non-negative integer", + ) + return value + + +def _observed_at(value: Any) -> str: + if not isinstance(value, str): + raise ObserverEnvelopeError( + EnvelopeRejection.CLOCK_INVALID, "observed_at must be ISO-8601 text" + ) + try: + parsed = datetime.fromisoformat(value.replace("Z", "+00:00")) + except ValueError as exc: + raise ObserverEnvelopeError( + EnvelopeRejection.CLOCK_INVALID, "observed_at is not ISO-8601" + ) from exc + if parsed.tzinfo is None: + raise ObserverEnvelopeError( + EnvelopeRejection.CLOCK_INVALID, "observed_at must carry a timezone" + ) + return value + + +def parse_observed_at(value: str) -> datetime: + return datetime.fromisoformat(value.replace("Z", "+00:00")) + + +def _clock(value: Any) -> ObserverClock: + if not isinstance(value, Mapping): + raise ObserverEnvelopeError( + EnvelopeRejection.CLOCK_INVALID, "clock must declare source and uncertainty" + ) + _reject_unknown_fields(value, allowed=CLOCK_FIELDS, context="clock") + try: + source = ClockSource(str(value.get("source"))) + except ValueError as exc: + raise ObserverEnvelopeError( + EnvelopeRejection.CLOCK_INVALID, "clock source is not a declared source" + ) from exc + uncertainty = value.get("uncertainty_ms") + if isinstance(uncertainty, bool) or not isinstance(uncertainty, int) or uncertainty < 0: + raise ObserverEnvelopeError( + EnvelopeRejection.CLOCK_INVALID, + "clock uncertainty_ms must be a non-negative integer", + ) + return ObserverClock(source=source, uncertainty_ms=uncertainty) + + +def _event_kind(value: Any) -> ObserverEventKind: + try: + return ObserverEventKind(str(value)) + except ValueError as exc: + raise ObserverEnvelopeError( + EnvelopeRejection.EVENT_KIND_INVALID, "event_kind is not a typed kind" + ) from exc + + +def _summary(value: Any) -> dict[str, int | str]: + if value is None: + return {} + if not isinstance(value, Mapping): + raise ObserverEnvelopeError( + EnvelopeRejection.SUMMARY_INVALID, "summary must be an object" + ) + _reject_unknown_fields(value, allowed=SUMMARY_FIELDS, context="summary") + compact: dict[str, int | str] = {} + for key, item in value.items(): + name = str(key) + if name in SUMMARY_INTEGER_FIELDS: + if isinstance(item, bool) or not isinstance(item, int) or item < 0: + raise ObserverEnvelopeError( + EnvelopeRejection.SUMMARY_INVALID, + f"summary.{name} must be a non-negative integer", + ) + elif not isinstance(item, str) or not SUMMARY_TOKEN_PATTERN.match(item): + raise ObserverEnvelopeError( + EnvelopeRejection.SUMMARY_INVALID, + f"summary.{name} must be a compact token", + ) + compact[name] = item + return compact + + +def _source_refs(value: Any) -> dict[str, str]: + if value is None: + return {} + if not isinstance(value, Mapping): + raise ObserverEnvelopeError( + EnvelopeRejection.SOURCE_REF_INVALID, "source_refs must be an object" + ) + _reject_unknown_fields(value, allowed=SOURCE_REF_FIELDS, context="source_refs") + refs: dict[str, str] = {} + for key, item in value.items(): + if not isinstance(item, str) or not IDENTITY_TOKEN_PATTERN.match(item): + raise ObserverEnvelopeError( + EnvelopeRejection.SOURCE_REF_INVALID, + f"source_refs.{key} must be an identity token", + ) + refs[str(key)] = item + return refs + + +def normalize_observer_envelope(record: Mapping[str, Any]) -> ObserverEnvelope: + """Validate one observer record and return its typed envelope. + + Rejection order is deliberate: unknown top-level fields are classified + first so a control-shaped record is reported as a control violation even + when the rest of the record is malformed. + """ + + if not isinstance(record, Mapping): + raise ObserverEnvelopeError( + EnvelopeRejection.SCHEMA_MISMATCH, "envelope must be an object" + ) + _reject_unknown_fields(record, allowed=ENVELOPE_FIELDS, context="envelope") + if record.get("schema_version") != OBSERVER_ENVELOPE_SCHEMA_VERSION: + raise ObserverEnvelopeError( + EnvelopeRejection.SCHEMA_MISMATCH, + f"schema_version must be {OBSERVER_ENVELOPE_SCHEMA_VERSION}", + ) + if record.get("capability_id") != CAPABILITY_ID: + raise ObserverEnvelopeError( + EnvelopeRejection.SCHEMA_MISMATCH, f"capability_id must be {CAPABILITY_ID}" + ) + envelope = ObserverEnvelope( + provider_id=_identity(record.get("provider_id"), name="provider_id") or "", + goal_id=_identity(record.get("goal_id"), name="goal_id") or "", + session_id=_identity(record.get("session_id"), name="session_id") or "", + agent_id=_identity(record.get("agent_id"), name="agent_id", optional=True), + sequence=_sequence(record.get("sequence")), + observed_at=_observed_at(record.get("observed_at")), + clock=_clock(record.get("clock")), + event_kind=_event_kind(record.get("event_kind")), + summary=_summary(record.get("summary")), + source_refs=_source_refs(record.get("source_refs")), + ) + try: + validate_public_safe_value(envelope.as_dict(), path="observer_envelope") + except ValueError as exc: + raise ObserverEnvelopeError( + EnvelopeRejection.PUBLIC_SAFETY_VIOLATION, str(exc) + ) from exc + return envelope diff --git a/loopx/capabilities/reliability_diagnostics/fixture.py b/loopx/capabilities/reliability_diagnostics/fixture.py new file mode 100644 index 0000000000..5e107eaef0 --- /dev/null +++ b/loopx/capabilities/reliability_diagnostics/fixture.py @@ -0,0 +1,126 @@ +"""Deterministic DSH-shaped observer fixture and its conformance run. + +The fixture is a fixed envelope stream shaped like the `dsh-session-events` +provider output. It deliberately exercises: a sequence gap (event loss), an +event with declared clock uncertainty above the degraded threshold, a +raw-material-bearing record that must be rejected, and a burst that overflows +the bounded buffer so backpressure drops are counted. Outbound endpoints stay +empty because the intake cannot express any. +""" + +from __future__ import annotations + +from datetime import datetime, timedelta, timezone +from typing import Any + +from .envelope import ( + CAPABILITY_ID, + DSH_PROVIDER_ID, + OBSERVER_ENVELOPE_SCHEMA_VERSION, + ClockSource, + ObserverEventKind, +) +from .intake import ShadowObserverIntake +from .projection import build_diagnostic_projection +from .receipt import build_integrity_receipt, read_ledger + +FIXTURE_GOAL_ID = "goal-dsh-fixture" +FIXTURE_SESSION_ID = "dsh-session-fixture" +FIXTURE_AGENT_ID = "agent-dsh-fixture" +FIXTURE_OBSERVER_ID = "dsh-session-events-fixture" +FIXTURE_BUFFER_BOUND = 20 +FIXTURE_UNCERTAIN_CLOCK_MS = 1500 +FIXTURE_START = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc) + + +def _envelope( + sequence: int, + kind: ObserverEventKind, + *, + seconds: int, + summary: dict[str, Any] | None = None, + source_refs: dict[str, str] | None = None, + clock_source: ClockSource = ClockSource.HARNESS_EVENT_TIME, + uncertainty_ms: int = 0, +) -> dict[str, Any]: + return { + "schema_version": OBSERVER_ENVELOPE_SCHEMA_VERSION, + "capability_id": CAPABILITY_ID, + "provider_id": DSH_PROVIDER_ID, + "goal_id": FIXTURE_GOAL_ID, + "session_id": FIXTURE_SESSION_ID, + "agent_id": FIXTURE_AGENT_ID, + "sequence": sequence, + "observed_at": (FIXTURE_START + timedelta(seconds=seconds)).isoformat(), + "clock": {"source": clock_source.value, "uncertainty_ms": uncertainty_ms}, + "event_kind": kind.value, + "summary": summary or {}, + "source_refs": {"event_seq": str(sequence), **(source_refs or {})}, + } + + +def dsh_fixture_records() -> list[dict[str, Any]]: + """The fixed DSH-shaped stream as the provider would hand it to the intake.""" + + k = ObserverEventKind + records = [ + _envelope(0, k.SESSION_STARTED, seconds=0), + _envelope(1, k.AGENT_STATUS, seconds=1, summary={"status": "running"}, clock_source=ClockSource.OBSERVER_WALL_CLOCK, uncertainty_ms=50), + _envelope(2, k.TURN_STARTED, seconds=1, summary={"turn": 1}), + _envelope(3, k.USER_MESSAGE, seconds=1, summary={"turn": 1, "message_source_kind": "user"}, source_refs={"message_id": "msg-1"}), + _envelope(4, k.STEP_STARTED, seconds=2, summary={"turn": 1, "step": 1}), + _envelope(5, k.TOOL_CALLED, seconds=3, summary={"turn": 1, "step": 1, "tool_name": "bash"}, source_refs={"tool_call_id": "call-1"}), + _envelope(6, k.TOOL_COMPLETED, seconds=8, summary={"turn": 1, "step": 1, "status": "ok"}, source_refs={"tool_call_id": "call-1"}), + _envelope(7, k.STEP_ENDED, seconds=9, summary={"turn": 1, "step": 1}), + _envelope(8, k.TURN_ENDED, seconds=9, summary={"turn": 1, "reason": "completed"}), + _envelope(9, k.AGENT_STATUS, seconds=9, summary={"status": "idle"}, clock_source=ClockSource.OBSERVER_WALL_CLOCK, uncertainty_ms=50), + # sequence 10 is intentionally missing: event loss before the observer. + _envelope(11, k.TURN_STARTED, seconds=60, summary={"turn": 2}, clock_source=ClockSource.OBSERVER_WALL_CLOCK, uncertainty_ms=FIXTURE_UNCERTAIN_CLOCK_MS), + _envelope(12, k.STEP_STARTED, seconds=61, summary={"turn": 2, "step": 1}), + _envelope(13, k.TOOL_CALLED, seconds=62, summary={"turn": 2, "step": 1, "tool_name": "read"}, source_refs={"tool_call_id": "call-2"}), + _envelope(14, k.TOOL_CALLED, seconds=63, summary={"turn": 2, "step": 1, "tool_name": "read"}, source_refs={"tool_call_id": "call-3"}), + _envelope(15, k.TOOL_CALLED, seconds=64, summary={"turn": 2, "step": 1, "tool_name": "read"}, source_refs={"tool_call_id": "call-4"}), + _envelope(16, k.AGENT_ERROR, seconds=70, summary={"turn": 2, "error_class": "ToolTimeout"}), + _envelope(17, k.STEP_ENDED, seconds=71, summary={"turn": 2, "step": 1}), + _envelope(18, k.TURN_ENDED, seconds=72, summary={"turn": 2, "reason": "completed"}), + ] + raw_material_record = _envelope(19, k.USER_MESSAGE, seconds=73, summary={"turn": 3}) + raw_material_record["transcript"] = "protected task content that must never be persisted" + records.append(raw_material_record) + # Burst arriving while the bounded buffer fills: sequences 20..23. The + # later disposal is also dropped, so trailing loss is visible only through + # the stats record, not through sequence gaps. + records.extend( + _envelope(sequence, k.AGENT_PRE_STEP, seconds=74, summary={"turn": 3}) + for sequence in range(20, 24) + ) + records.append(_envelope(24, k.SESSION_DISPOSED, seconds=90)) + return records + + +def run_dsh_fixture() -> dict[str, Any]: + """Feed the fixture through the reference intake without flushing mid-burst. + + Returns the persisted ledger records, the intake stats, the receipt, and + the projection so callers can assert contract semantics. + """ + + intake = ShadowObserverIntake( + provider_id=DSH_PROVIDER_ID, + observer_id=FIXTURE_OBSERVER_ID, + goal_id=FIXTURE_GOAL_ID, + clock_source=ClockSource.FIXTURE, + buffer_bound=FIXTURE_BUFFER_BOUND, + ) + accepted = [intake.observe(record) for record in dsh_fixture_records()] + ledger: list[dict[str, Any]] = [] + emitted_at = (FIXTURE_START + timedelta(seconds=95)).isoformat() + intake.flush(ledger.extend, emitted_at=emitted_at) + reading = read_ledger(ledger, goal_id=FIXTURE_GOAL_ID) + return { + "accepted_flags": accepted, + "ledger_records": ledger, + "stats": intake.stats(emitted_at=emitted_at).as_dict(), + "receipt": build_integrity_receipt(reading), + "projection": build_diagnostic_projection(reading), + } diff --git a/loopx/capabilities/reliability_diagnostics/intake.py b/loopx/capabilities/reliability_diagnostics/intake.py new file mode 100644 index 0000000000..3694546017 --- /dev/null +++ b/loopx/capabilities/reliability_diagnostics/intake.py @@ -0,0 +1,263 @@ +"""Bounded, crash-isolated observer intake and its stats record. + +The intake is the provider-neutral reference for the L1 observer runtime +contract: a bounded buffer whose overflow is counted rather than blocking, an +``observe`` call that never raises into the caller, and a stats record that +carries the receipt inputs (buffer bound, drops, failures, outbound endpoints) +next to the envelopes it accepted. The DSH TypeScript observer implements the +same shape and writes the same stats record. +""" + +from __future__ import annotations + +import re +from collections import deque +from collections.abc import Callable, Mapping +from dataclasses import dataclass, field +from typing import Any + +from .envelope import ( + CAPABILITY_ID, + IDENTITY_TOKEN_PATTERN, + OBSERVER_STATS_SCHEMA_VERSION, + ClockSource, + EnvelopeRejection, + ObserverEnvelope, + ObserverEnvelopeError, + normalize_observer_envelope, +) + +DEFAULT_BUFFER_BOUND = 256 +MAX_BUFFER_BOUND = 65_536 +_ENDPOINT_PATTERN = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.:/-]{0,200}$") + +STATS_FIELDS = frozenset( + { + "schema_version", + "capability_id", + "provider_id", + "observer_id", + "goal_id", + "emitted_at", + "observed_event_count", + "accepted_event_count", + "rejected_event_count", + "rejected_by_reason", + "buffer_bound", + "backpressure_drop_count", + "observer_failure_count", + "outbound_endpoints", + "observation_entered_worker_context", + "clock_source", + } +) + + +@dataclass(frozen=True) +class ObserverStats: + provider_id: str + observer_id: str + goal_id: str + emitted_at: str + observed_event_count: int + accepted_event_count: int + rejected_event_count: int + rejected_by_reason: Mapping[str, int] + buffer_bound: int + backpressure_drop_count: int + observer_failure_count: int + outbound_endpoints: tuple[str, ...] + observation_entered_worker_context: bool + clock_source: ClockSource + + def as_dict(self) -> dict[str, Any]: + return { + "schema_version": OBSERVER_STATS_SCHEMA_VERSION, + "capability_id": CAPABILITY_ID, + "provider_id": self.provider_id, + "observer_id": self.observer_id, + "goal_id": self.goal_id, + "emitted_at": self.emitted_at, + "observed_event_count": self.observed_event_count, + "accepted_event_count": self.accepted_event_count, + "rejected_event_count": self.rejected_event_count, + "rejected_by_reason": dict(self.rejected_by_reason), + "buffer_bound": self.buffer_bound, + "backpressure_drop_count": self.backpressure_drop_count, + "observer_failure_count": self.observer_failure_count, + "outbound_endpoints": list(self.outbound_endpoints), + "observation_entered_worker_context": self.observation_entered_worker_context, + "clock_source": self.clock_source.value, + } + + +def _count(value: Any, *, name: str) -> int: + if isinstance(value, bool) or not isinstance(value, int) or value < 0: + raise ValueError(f"observer stats {name} must be a non-negative integer") + return value + + +def normalize_observer_stats(record: Mapping[str, Any]) -> ObserverStats: + """Validate one stats record written by any observer implementation.""" + + if not isinstance(record, Mapping): + raise ValueError("observer stats must be an object") + unknown = sorted(str(key) for key in record if str(key) not in STATS_FIELDS) + if unknown: + raise ValueError(f"observer stats carry unsupported fields: {unknown}") + if record.get("schema_version") != OBSERVER_STATS_SCHEMA_VERSION: + raise ValueError(f"observer stats schema must be {OBSERVER_STATS_SCHEMA_VERSION}") + if record.get("capability_id") != CAPABILITY_ID: + raise ValueError(f"observer stats capability must be {CAPABILITY_ID}") + for key in ("provider_id", "observer_id", "goal_id"): + value = record.get(key) + if not isinstance(value, str) or not IDENTITY_TOKEN_PATTERN.match(value): + raise ValueError(f"observer stats {key} must be an identity token") + emitted_at = record.get("emitted_at") + if not isinstance(emitted_at, str) or not emitted_at.strip(): + raise ValueError("observer stats emitted_at is required") + reasons = record.get("rejected_by_reason") or {} + if not isinstance(reasons, Mapping): + raise ValueError("observer stats rejected_by_reason must be an object") + normalized_reasons = { + EnvelopeRejection(str(key)).value: _count(value, name=f"rejected_by_reason.{key}") + for key, value in reasons.items() + } + endpoints = record.get("outbound_endpoints") + if not isinstance(endpoints, list) or any( + not isinstance(item, str) or not _ENDPOINT_PATTERN.match(item) + for item in endpoints + ): + raise ValueError("observer stats outbound_endpoints must be a list of endpoint ids") + entered = record.get("observation_entered_worker_context") + if not isinstance(entered, bool): + raise ValueError("observer stats observation_entered_worker_context must be boolean") + buffer_bound = _count(record.get("buffer_bound"), name="buffer_bound") + if not 1 <= buffer_bound <= MAX_BUFFER_BOUND: + raise ValueError(f"observer stats buffer_bound must be within 1..{MAX_BUFFER_BOUND}") + return ObserverStats( + provider_id=str(record["provider_id"]), + observer_id=str(record["observer_id"]), + goal_id=str(record["goal_id"]), + emitted_at=emitted_at, + observed_event_count=_count( + record.get("observed_event_count"), name="observed_event_count" + ), + accepted_event_count=_count( + record.get("accepted_event_count"), name="accepted_event_count" + ), + rejected_event_count=_count( + record.get("rejected_event_count"), name="rejected_event_count" + ), + rejected_by_reason=normalized_reasons, + buffer_bound=buffer_bound, + backpressure_drop_count=_count( + record.get("backpressure_drop_count"), name="backpressure_drop_count" + ), + observer_failure_count=_count( + record.get("observer_failure_count"), name="observer_failure_count" + ), + outbound_endpoints=tuple(endpoints), + observation_entered_worker_context=entered, + clock_source=ClockSource(str(record.get("clock_source"))), + ) + + +@dataclass +class ShadowObserverIntake: + """Reference intake: bounded buffer, counted drops, isolated failures.""" + + provider_id: str + observer_id: str + goal_id: str + clock_source: ClockSource + buffer_bound: int = DEFAULT_BUFFER_BOUND + observed_event_count: int = 0 + accepted_event_count: int = 0 + rejected_event_count: int = 0 + backpressure_drop_count: int = 0 + observer_failure_count: int = 0 + rejected_by_reason: dict[str, int] = field(default_factory=dict) + _buffer: deque[ObserverEnvelope] = field(default_factory=deque, repr=False) + + def __post_init__(self) -> None: + if not 1 <= self.buffer_bound <= MAX_BUFFER_BOUND: + raise ValueError(f"buffer_bound must be within 1..{MAX_BUFFER_BOUND}") + for key in ("provider_id", "observer_id", "goal_id"): + if not IDENTITY_TOKEN_PATTERN.match(getattr(self, key)): + raise ValueError(f"{key} must be an identity token") + + @property + def buffered_count(self) -> int: + return len(self._buffer) + + def observe(self, record: Any) -> bool: + """Accept or refuse one record; never raise into the event source.""" + + self.observed_event_count += 1 + try: + envelope = normalize_observer_envelope(record) + except ObserverEnvelopeError as exc: + self.rejected_event_count += 1 + reason = exc.reason.value + self.rejected_by_reason[reason] = self.rejected_by_reason.get(reason, 0) + 1 + return False + except Exception: # noqa: BLE001 - crash isolation is the contract + self.observer_failure_count += 1 + return False + if envelope.goal_id != self.goal_id: + self.rejected_event_count += 1 + reason = EnvelopeRejection.IDENTITY_INVALID.value + self.rejected_by_reason[reason] = self.rejected_by_reason.get(reason, 0) + 1 + return False + if len(self._buffer) >= self.buffer_bound: + self.backpressure_drop_count += 1 + return False + self._buffer.append(envelope) + self.accepted_event_count += 1 + return True + + def drain(self) -> list[ObserverEnvelope]: + drained = list(self._buffer) + self._buffer.clear() + return drained + + def stats(self, *, emitted_at: str) -> ObserverStats: + return ObserverStats( + provider_id=self.provider_id, + observer_id=self.observer_id, + goal_id=self.goal_id, + emitted_at=emitted_at, + observed_event_count=self.observed_event_count, + accepted_event_count=self.accepted_event_count, + rejected_event_count=self.rejected_event_count, + rejected_by_reason=dict(self.rejected_by_reason), + buffer_bound=self.buffer_bound, + backpressure_drop_count=self.backpressure_drop_count, + observer_failure_count=self.observer_failure_count, + outbound_endpoints=(), + observation_entered_worker_context=False, + clock_source=self.clock_source, + ) + + def flush( + self, + sink: Callable[[list[dict[str, Any]]], None], + *, + emitted_at: str, + ) -> list[dict[str, Any]]: + """Hand drained envelopes plus a stats record to ``sink``. + + A failing sink is counted as an observer failure and its envelopes are + counted as backpressure drops; the failure never propagates. + """ + + envelopes = self.drain() + records = [envelope.as_dict() for envelope in envelopes] + try: + sink([*records, self.stats(emitted_at=emitted_at).as_dict()]) + except Exception: # noqa: BLE001 - crash isolation is the contract + self.observer_failure_count += 1 + self.backpressure_drop_count += len(records) + return [] + return records diff --git a/loopx/capabilities/reliability_diagnostics/ledger.py b/loopx/capabilities/reliability_diagnostics/ledger.py new file mode 100644 index 0000000000..9516387df4 --- /dev/null +++ b/loopx/capabilities/reliability_diagnostics/ledger.py @@ -0,0 +1,86 @@ +"""Append-only NDJSON diagnostic ledger under the LoopX runtime root. + +The ledger is LoopX-owned diagnostic state, independent from goal, todo, +gate, and session-runtime authority. Callers resolve the runtime root through +``loopx.paths.resolve_runtime_root`` and never publish the absolute path; the +public reference is the relative ``ledger_ref``. +""" + +from __future__ import annotations + +import json +from collections.abc import Iterable, Mapping +from pathlib import Path +from typing import Any + +from .envelope import IDENTITY_TOKEN_PATTERN + +LEDGER_DIRNAME = "reliability_diagnostics" + + +def _goal_file_stem(goal_id: str) -> str: + if not isinstance(goal_id, str) or not IDENTITY_TOKEN_PATTERN.match(goal_id): + raise ValueError("goal_id must be an identity token") + return goal_id.replace(":", "_") + + +def ledger_ref(goal_id: str) -> str: + """Public-safe, runtime-root-relative ledger reference.""" + + return f"{LEDGER_DIRNAME}/{_goal_file_stem(goal_id)}.ndjson" + + +def ledger_path(runtime_root: Path, goal_id: str) -> Path: + return Path(runtime_root).expanduser() / ledger_ref(goal_id) + + +def append_ledger_records(path: Path, records: Iterable[Mapping[str, Any]]) -> int: + """Append records as one JSON object per line; returns the appended count.""" + + lines = [json.dumps(dict(record), sort_keys=True, ensure_ascii=True) for record in records] + if not lines: + return 0 + path.parent.mkdir(parents=True, exist_ok=True) + with path.open("a", encoding="utf-8") as handle: + handle.write("\n".join(lines) + "\n") + return len(lines) + + +def read_ledger_records(path: Path) -> tuple[list[dict[str, Any]], int]: + """Return ledger objects in file order plus the malformed-line count.""" + + if not path.is_file(): + return [], 0 + records: list[dict[str, Any]] = [] + malformed = 0 + with path.open("r", encoding="utf-8") as handle: + for line in handle: + text = line.strip() + if not text: + continue + try: + value = json.loads(text) + except json.JSONDecodeError: + malformed += 1 + continue + if not isinstance(value, dict): + malformed += 1 + continue + records.append(value) + return records, malformed + + +def parse_ndjson_lines(lines: Iterable[str]) -> tuple[list[Any], int]: + """Parse NDJSON text lines without interpreting them; count malformed ones.""" + + parsed: list[Any] = [] + malformed = 0 + for line in lines: + text = line.strip() + if not text: + continue + try: + parsed.append(json.loads(text)) + except json.JSONDecodeError: + malformed += 1 + return parsed, malformed diff --git a/loopx/capabilities/reliability_diagnostics/projection.py b/loopx/capabilities/reliability_diagnostics/projection.py new file mode 100644 index 0000000000..bb004340a7 --- /dev/null +++ b/loopx/capabilities/reliability_diagnostics/projection.py @@ -0,0 +1,171 @@ +"""Compact read-only diagnostic projection derived from event kinds and timing. + +This projection is a sibling of the session-runtime projection, never merged +into it. It carries explicit boundary fields (``mode``, ``authority``) so a +consumer can prove it holds no runtime authority, and every signal is derived +from typed event kinds and observed timestamps only. +""" + +from __future__ import annotations + +from enum import StrEnum +from typing import Any + +from .envelope import CAPABILITY_ID, ObserverEnvelope, ObserverEventKind, parse_observed_at +from .receipt import LedgerReading, build_integrity_receipt + +DIAGNOSTIC_PROJECTION_SCHEMA_VERSION = "reliability_diagnostic_projection_v0" +DEFAULT_STALL_THRESHOLD_MS = 300_000 +DEFAULT_REPETITION_THRESHOLD = 3 +_TERMINAL_ERROR_REASONS = frozenset({"error", "failed", "failure", "aborted", "cancelled", "canceled", "timeout"}) + + +class DiagnosticStage(StrEnum): + UNKNOWN = "unknown" + IDLE = "idle" + RUNNING = "running" + TOOL_RUNNING = "tool_running" + ERRORED = "errored" + DISPOSED = "disposed" + + +class DiagnosticSignal(StrEnum): + STALL_SUSPECTED = "stall_suspected" + REPETITION_SUSPECTED = "repetition_suspected" + UNRECOVERED_ERROR = "unrecovered_error" + EVENT_LOSS = "event_loss" + INTEGRITY_NOT_VALID = "integrity_not_valid" + + +_ACTIVE_STAGES = frozenset({DiagnosticStage.RUNNING, DiagnosticStage.TOOL_RUNNING}) + + +def _stage_after(envelope: ObserverEnvelope) -> DiagnosticStage: + kind = envelope.event_kind + if kind is ObserverEventKind.SESSION_DISPOSED: + return DiagnosticStage.DISPOSED + if kind is ObserverEventKind.AGENT_ERROR: + return DiagnosticStage.ERRORED + if kind is ObserverEventKind.TOOL_CALLED: + return DiagnosticStage.TOOL_RUNNING + if kind in {ObserverEventKind.TURN_ENDED, ObserverEventKind.SESSION_STARTED}: + return DiagnosticStage.IDLE + if kind is ObserverEventKind.AGENT_STATUS: + return DiagnosticStage.RUNNING if envelope.summary.get("status") == "running" else DiagnosticStage.IDLE + if kind is ObserverEventKind.UNSUPPORTED: + return DiagnosticStage.UNKNOWN + return DiagnosticStage.RUNNING + + +def _ms_between(earlier: str, later: str) -> int: + return int((parse_observed_at(later) - parse_observed_at(earlier)).total_seconds() * 1000) + + +def build_diagnostic_projection( + reading: LedgerReading, + *, + as_of: str | None = None, + stall_threshold_ms: int = DEFAULT_STALL_THRESHOLD_MS, + repetition_threshold: int = DEFAULT_REPETITION_THRESHOLD, +) -> dict[str, Any]: + receipt = build_integrity_receipt(reading) + envelopes = reading.ordered_envelopes + + counts = {"turns_started": 0, "turns_ended": 0, "steps": 0, "tool_calls": 0, "errors": 0} + stage = DiagnosticStage.UNKNOWN + max_gap_ms = 0 + longest_run = 0 + longest_run_tool: str | None = None + current_run = 0 + current_tool: str | None = None + unrecovered_errors = 0 + recovered_errors = 0 + previous: ObserverEnvelope | None = None + for envelope in envelopes: + kind = envelope.event_kind + if previous is not None: + max_gap_ms = max(max_gap_ms, _ms_between(previous.observed_at, envelope.observed_at)) + if kind is ObserverEventKind.TURN_STARTED: + counts["turns_started"] += 1 + elif kind is ObserverEventKind.TURN_ENDED: + counts["turns_ended"] += 1 + if unrecovered_errors and str(envelope.summary.get("reason", "")) not in _TERMINAL_ERROR_REASONS: + recovered_errors += unrecovered_errors + unrecovered_errors = 0 + elif kind is ObserverEventKind.STEP_ENDED: + counts["steps"] += 1 + if unrecovered_errors: + recovered_errors += unrecovered_errors + unrecovered_errors = 0 + elif kind is ObserverEventKind.AGENT_ERROR: + counts["errors"] += 1 + unrecovered_errors += 1 + if kind is ObserverEventKind.TOOL_CALLED: + counts["tool_calls"] += 1 + tool = str(envelope.summary.get("tool_name", "")) + current_run = current_run + 1 if tool == current_tool else 1 + current_tool = tool + if current_run > longest_run: + longest_run, longest_run_tool = current_run, tool or None + elif kind not in {ObserverEventKind.TOOL_COMPLETED, ObserverEventKind.AGENT_PRE_STEP}: + current_run, current_tool = 0, None + stage = _stage_after(envelope) + previous = envelope + + last_observed_at = previous.observed_at if previous else None + effective_as_of = as_of or last_observed_at + last_event_age_ms = _ms_between(last_observed_at, effective_as_of) if last_observed_at and effective_as_of else 0 + stall_detected = stage in _ACTIVE_STAGES and last_event_age_ms >= stall_threshold_ms + repetition_detected = longest_run >= repetition_threshold + + signals: list[str] = [] + if stall_detected: + signals.append(DiagnosticSignal.STALL_SUSPECTED.value) + if repetition_detected: + signals.append(DiagnosticSignal.REPETITION_SUSPECTED.value) + if unrecovered_errors: + signals.append(DiagnosticSignal.UNRECOVERED_ERROR.value) + if receipt["lost_event_count"] or receipt["backpressure_drop_count"]: + signals.append(DiagnosticSignal.EVENT_LOSS.value) + if receipt["status"] != "valid": + signals.append(DiagnosticSignal.INTEGRITY_NOT_VALID.value) + + return { + "schema_version": DIAGNOSTIC_PROJECTION_SCHEMA_VERSION, + "capability_id": CAPABILITY_ID, + "goal_id": reading.goal_id, + "mode": "read_only", + "authority": "none", + "write_scope": "diagnostic_ledger_only", + "worker_influence": "none", + "provider_ids": receipt["provider_ids"], + "stage": stage.value, + "counts": counts, + "stall": { + "detected": stall_detected, + "threshold_ms": stall_threshold_ms, + "last_event_age_ms": last_event_age_ms, + "max_inter_event_gap_ms": max_gap_ms, + }, + "repetition": { + "detected": repetition_detected, + "threshold": repetition_threshold, + "longest_tool_run": longest_run, + "tool_name": longest_run_tool, + }, + "recovery": { + "error_count": counts["errors"], + "recovered_error_count": recovered_errors, + "unrecovered_error_count": unrecovered_errors, + }, + "signals": signals, + "integrity": {"status": receipt["status"], "reason_codes": receipt["reason_codes"]}, + "evidence": { + "observed_event_count": receipt["observed_event_count"], + "lost_event_count": receipt["lost_event_count"], + "session_count": receipt["session_count"], + "observed_from": receipt["observed_from"], + "observed_until": receipt["observed_until"], + "as_of": effective_as_of, + }, + } diff --git a/loopx/capabilities/reliability_diagnostics/receipt.py b/loopx/capabilities/reliability_diagnostics/receipt.py new file mode 100644 index 0000000000..2b08846967 --- /dev/null +++ b/loopx/capabilities/reliability_diagnostics/receipt.py @@ -0,0 +1,210 @@ +"""Treatment-integrity receipt for one goal's shadow-observer ledger. + +The receipt answers whether the ledger is admissible passive evidence. It is +computed only from ledger records: accepted envelopes and observer stats. Its +status enum is total and ordered; every non-``valid`` status names typed reason +codes so an operator never has to infer why evidence was downgraded. +""" + +from __future__ import annotations + +from collections import defaultdict +from collections.abc import Iterable, Mapping +from dataclasses import dataclass, field +from enum import StrEnum +from typing import Any + +from .envelope import ( + CAPABILITY_ID, + OBSERVER_ENVELOPE_SCHEMA_VERSION, + OBSERVER_STATS_SCHEMA_VERSION, + EnvelopeRejection, + ObserverEnvelope, + ObserverEnvelopeError, + normalize_observer_envelope, +) +from .intake import ObserverStats, normalize_observer_stats + +INTEGRITY_RECEIPT_SCHEMA_VERSION = "reliability_integrity_receipt_v0" +DEFAULT_CLOCK_UNCERTAINTY_DEGRADED_MS = 1000 + + +class ReceiptStatus(StrEnum): + VALID = "valid" + DEGRADED = "degraded" + QUARANTINED = "quarantined" + INVALID = "invalid" + + +class ReceiptReason(StrEnum): + NO_OBSERVATIONS = "no_observations" + OUTBOUND_ENDPOINT_CONFIGURED = "outbound_endpoint_configured" + OBSERVATION_ENTERED_WORKER_CONTEXT = "observation_entered_worker_context" + OBSERVER_FAILURE = "observer_failure" + CONTROL_FIELD_REJECTED = "control_field_rejected" + LEDGER_RECORD_INVALID = "ledger_record_invalid" + OBSERVER_STATS_MISSING = "observer_stats_missing" + SEQUENCE_GAP = "sequence_gap" + SEQUENCE_DUPLICATE = "sequence_duplicate" + BACKPRESSURE_DROP = "backpressure_drop" + RAW_MATERIAL_REJECTED = "raw_material_rejected" + UNSUPPORTED_FIELD_REJECTED = "unsupported_field_rejected" + CLOCK_UNCERTAINTY_EXCEEDED = "clock_uncertainty_exceeded" + + +_INVALID_REASONS = frozenset( + { + ReceiptReason.NO_OBSERVATIONS, + ReceiptReason.OUTBOUND_ENDPOINT_CONFIGURED, + ReceiptReason.OBSERVATION_ENTERED_WORKER_CONTEXT, + } +) +_QUARANTINE_REASONS = frozenset( + { + ReceiptReason.OBSERVER_FAILURE, + ReceiptReason.CONTROL_FIELD_REJECTED, + ReceiptReason.LEDGER_RECORD_INVALID, + } +) + + +@dataclass +class LedgerReading: + """Typed, ordered view over one goal ledger's records.""" + + goal_id: str + envelopes: list[ObserverEnvelope] = field(default_factory=list) + stats: dict[str, ObserverStats] = field(default_factory=dict) + invalid_record_count: int = 0 + + @property + def ordered_envelopes(self) -> list[ObserverEnvelope]: + return sorted(self.envelopes, key=lambda item: (item.observed_at, item.session_id, item.sequence)) + + +def read_ledger(records: Iterable[Any], *, goal_id: str, malformed_line_count: int = 0) -> LedgerReading: + reading = LedgerReading(goal_id=goal_id, invalid_record_count=malformed_line_count) + for record in records: + if not isinstance(record, Mapping): + reading.invalid_record_count += 1 + continue + schema = record.get("schema_version") + try: + if schema == OBSERVER_ENVELOPE_SCHEMA_VERSION: + envelope = normalize_observer_envelope(record) + if envelope.goal_id != goal_id: + raise ObserverEnvelopeError( + EnvelopeRejection.IDENTITY_INVALID, "goal_id does not match ledger" + ) + reading.envelopes.append(envelope) + elif schema == OBSERVER_STATS_SCHEMA_VERSION: + stats = normalize_observer_stats(record) + if stats.goal_id != goal_id: + raise ValueError("goal_id does not match ledger") + # Stats are cumulative per observer instance; the latest wins. + reading.stats[stats.observer_id] = stats + else: + raise ValueError("unknown ledger record schema") + except ValueError: + reading.invalid_record_count += 1 + return reading + + +def _sequence_accounting(envelopes: Iterable[ObserverEnvelope]) -> tuple[int, int]: + by_session: dict[str, list[int]] = defaultdict(list) + for envelope in envelopes: + by_session[envelope.session_id].append(envelope.sequence) + lost = 0 + duplicates = 0 + for sequences in by_session.values(): + ordered = sorted(sequences) + for previous, current in zip(ordered, ordered[1:]): + if current == previous: + duplicates += 1 + else: + lost += current - previous - 1 + return lost, duplicates + + +def build_integrity_receipt( + reading: LedgerReading, + *, + clock_uncertainty_degraded_ms: int = DEFAULT_CLOCK_UNCERTAINTY_DEGRADED_MS, +) -> dict[str, Any]: + envelopes = reading.ordered_envelopes + stats = list(reading.stats.values()) + lost, duplicates = _sequence_accounting(envelopes) + rejected_by_reason: dict[str, int] = defaultdict(int) + for item in stats: + for reason, count in item.rejected_by_reason.items(): + rejected_by_reason[reason] += count + outbound_endpoints = sorted({endpoint for item in stats for endpoint in item.outbound_endpoints}) + entered_worker_context = any(item.observation_entered_worker_context for item in stats) + observer_failures = sum(item.observer_failure_count for item in stats) + backpressure_drops = sum(item.backpressure_drop_count for item in stats) + max_uncertainty = max((item.clock.uncertainty_ms for item in envelopes), default=0) + clock_sources = sorted({item.clock.source.value for item in envelopes} | {item.clock_source.value for item in stats}) + + reasons: set[ReceiptReason] = set() + if not envelopes: + reasons.add(ReceiptReason.NO_OBSERVATIONS) + if outbound_endpoints: + reasons.add(ReceiptReason.OUTBOUND_ENDPOINT_CONFIGURED) + if entered_worker_context: + reasons.add(ReceiptReason.OBSERVATION_ENTERED_WORKER_CONTEXT) + if observer_failures: + reasons.add(ReceiptReason.OBSERVER_FAILURE) + if rejected_by_reason.get(EnvelopeRejection.CONTROL_FIELD_REJECTED.value): + reasons.add(ReceiptReason.CONTROL_FIELD_REJECTED) + if reading.invalid_record_count: + reasons.add(ReceiptReason.LEDGER_RECORD_INVALID) + if envelopes and not stats: + reasons.add(ReceiptReason.OBSERVER_STATS_MISSING) + if lost: + reasons.add(ReceiptReason.SEQUENCE_GAP) + if duplicates: + reasons.add(ReceiptReason.SEQUENCE_DUPLICATE) + if backpressure_drops: + reasons.add(ReceiptReason.BACKPRESSURE_DROP) + if rejected_by_reason.get(EnvelopeRejection.RAW_MATERIAL_FIELD_REJECTED.value): + reasons.add(ReceiptReason.RAW_MATERIAL_REJECTED) + if rejected_by_reason.get(EnvelopeRejection.UNSUPPORTED_FIELD_REJECTED.value): + reasons.add(ReceiptReason.UNSUPPORTED_FIELD_REJECTED) + if max_uncertainty > clock_uncertainty_degraded_ms: + reasons.add(ReceiptReason.CLOCK_UNCERTAINTY_EXCEEDED) + + if reasons & _INVALID_REASONS: + status = ReceiptStatus.INVALID + elif reasons & _QUARANTINE_REASONS: + status = ReceiptStatus.QUARANTINED + elif reasons: + status = ReceiptStatus.DEGRADED + else: + status = ReceiptStatus.VALID + + return { + "schema_version": INTEGRITY_RECEIPT_SCHEMA_VERSION, + "capability_id": CAPABILITY_ID, + "goal_id": reading.goal_id, + "status": status.value, + "reason_codes": sorted(reason.value for reason in reasons), + "provider_ids": sorted({item.provider_id for item in envelopes} | {item.provider_id for item in stats}), + "observer_ids": sorted(reading.stats), + "session_count": len({item.session_id for item in envelopes}), + "observed_event_count": len(envelopes), + "lost_event_count": lost, + "duplicate_sequence_count": duplicates, + "ledger_invalid_record_count": reading.invalid_record_count, + "rejected_event_count": sum(rejected_by_reason.values()), + "rejected_by_reason": dict(sorted(rejected_by_reason.items())), + "buffer_bound": max((item.buffer_bound for item in stats), default=None), + "backpressure_drop_count": backpressure_drops, + "observer_failure_count": observer_failures, + "clock": {"sources": clock_sources, "max_uncertainty_ms": max_uncertainty}, + "outbound_endpoints": outbound_endpoints, + "observation_entered_worker_context": entered_worker_context, + "event_kinds_consumed": sorted({item.event_kind.value for item in envelopes}), + "summary_fields_consumed": sorted({key for item in envelopes for key in item.summary}), + "observed_from": envelopes[0].observed_at if envelopes else None, + "observed_until": envelopes[-1].observed_at if envelopes else None, + } diff --git a/tests/capabilities/test_reliability_diagnostics.py b/tests/capabilities/test_reliability_diagnostics.py new file mode 100644 index 0000000000..8916ebb0c1 --- /dev/null +++ b/tests/capabilities/test_reliability_diagnostics.py @@ -0,0 +1,479 @@ +"""Contract tests for the reliability-diagnostics L1 shadow-observer slice. + +Expected values come from the design contract (RFC §3.1 and §7.4), not from +observed implementation output: no control path, no raw material, bounded and +counted loss, visible clocks, and a total status enum. +""" + +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any + +import pytest + +from loopx.capabilities.reliability_diagnostics import ( + CAPABILITY_ID, + CONTROL_FIELD_FAMILIES, + DSH_PROVIDER_ID, + ENVELOPE_FIELDS, + FIXTURE_GOAL_ID, + OBSERVER_ENVELOPE_SCHEMA_VERSION, + OBSERVER_STATS_SCHEMA_VERSION, + RAW_MATERIAL_FIELD_FAMILIES, + SOURCE_REF_FIELDS, + SUMMARY_FIELDS, + ClockSource, + DiagnosticSignal, + DiagnosticStage, + EnvelopeRejection, + ObserverEnvelopeError, + ObserverEventKind, + ReceiptReason, + ReceiptStatus, + ShadowObserverIntake, + append_ledger_records, + build_diagnostic_projection, + build_integrity_receipt, + dsh_fixture_records, + ledger_path, + ledger_ref, + normalize_observer_envelope, + normalize_observer_stats, + read_ledger, + read_ledger_records, + run_dsh_fixture, +) +from loopx.capabilities.reliability_diagnostics.envelope import classify_rejected_field + +GOAL = "goal-observer" +SESSION = "session-observer" +T0 = "2026-09-01T10:00:00+00:00" + + +def envelope( + sequence: int, + kind: ObserverEventKind = ObserverEventKind.STEP_ENDED, + *, + observed_at: str = T0, + summary: dict[str, Any] | None = None, + **overrides: Any, +) -> dict[str, Any]: + record: dict[str, Any] = { + "schema_version": OBSERVER_ENVELOPE_SCHEMA_VERSION, + "capability_id": CAPABILITY_ID, + "provider_id": DSH_PROVIDER_ID, + "goal_id": GOAL, + "session_id": SESSION, + "sequence": sequence, + "observed_at": observed_at, + "clock": {"source": ClockSource.HARNESS_EVENT_TIME.value, "uncertainty_ms": 0}, + "event_kind": kind.value, + "summary": summary or {}, + "source_refs": {"event_seq": str(sequence)}, + } + record.update(overrides) + return record + + +def stats(**overrides: Any) -> dict[str, Any]: + record: dict[str, Any] = { + "schema_version": OBSERVER_STATS_SCHEMA_VERSION, + "capability_id": CAPABILITY_ID, + "provider_id": DSH_PROVIDER_ID, + "observer_id": "observer-1", + "goal_id": GOAL, + "emitted_at": T0, + "observed_event_count": 1, + "accepted_event_count": 1, + "rejected_event_count": 0, + "rejected_by_reason": {}, + "buffer_bound": 8, + "backpressure_drop_count": 0, + "observer_failure_count": 0, + "outbound_endpoints": [], + "observation_entered_worker_context": False, + "clock_source": ClockSource.HARNESS_EVENT_TIME.value, + } + record.update(overrides) + return record + + +def at(seconds: int) -> str: + return f"2026-09-01T10:{seconds // 60:02d}:{seconds % 60:02d}+00:00" + + +def receipt_for(*records: dict[str, Any], malformed: int = 0) -> dict[str, Any]: + return build_integrity_receipt(read_ledger(records, goal_id=GOAL, malformed_line_count=malformed)) + + +# --- envelope schema ------------------------------------------------------- + + +def test_envelope_schema_has_no_control_or_raw_material_fields() -> None: + flattened = {name.replace("_", "") for name in ENVELOPE_FIELDS | SUMMARY_FIELDS | SOURCE_REF_FIELDS} + assert not flattened & CONTROL_FIELD_FAMILIES + assert not flattened & RAW_MATERIAL_FIELD_FAMILIES + + +def test_valid_envelope_round_trips() -> None: + record = envelope(3, ObserverEventKind.TOOL_CALLED, summary={"turn": 1, "step": 2, "tool_name": "bash"}) + record["source_refs"]["tool_call_id"] = "call-9" + normalized = normalize_observer_envelope(record) + assert normalized.event_kind is ObserverEventKind.TOOL_CALLED + assert normalized.as_dict() == record + + +@pytest.mark.parametrize( + "field", + ["command", "send", "schedule", "retry", "stop", "resume", "gate", "toolCall", "worker_state", "callback"], +) +def test_control_shaped_fields_are_rejected_as_control(field: str) -> None: + record = envelope(1, **{field: {"kind": "continue"}}) + record["sequence"] = -1 # also malformed: control classification must still win + with pytest.raises(ObserverEnvelopeError) as excinfo: + normalize_observer_envelope(record) + assert excinfo.value.reason is EnvelopeRejection.CONTROL_FIELD_REJECTED + + +@pytest.mark.parametrize("field", ["transcript", "tool_output", "stdout", "cwd", "messages", "arguments"]) +def test_raw_material_fields_are_rejected_and_classified(field: str) -> None: + with pytest.raises(ObserverEnvelopeError) as excinfo: + normalize_observer_envelope(envelope(1, **{field: "protected"})) + assert excinfo.value.reason is EnvelopeRejection.RAW_MATERIAL_FIELD_REJECTED + + +def test_raw_material_inside_summary_is_rejected() -> None: + with pytest.raises(ObserverEnvelopeError) as excinfo: + normalize_observer_envelope(envelope(1, summary={"text": "hello"})) + assert excinfo.value.reason is EnvelopeRejection.RAW_MATERIAL_FIELD_REJECTED + + +def test_unknown_field_is_rejected_as_unsupported() -> None: + with pytest.raises(ObserverEnvelopeError) as excinfo: + normalize_observer_envelope(envelope(1, colour="blue")) + assert excinfo.value.reason is EnvelopeRejection.UNSUPPORTED_FIELD_REJECTED + assert classify_rejected_field("agentSend") is EnvelopeRejection.UNSUPPORTED_FIELD_REJECTED + + +@pytest.mark.parametrize( + ("mutation", "reason"), + [ + ({"schema_version": "other_v9"}, EnvelopeRejection.SCHEMA_MISMATCH), + ({"capability_id": "session-runtime"}, EnvelopeRejection.SCHEMA_MISMATCH), + ({"goal_id": "goal with spaces"}, EnvelopeRejection.IDENTITY_INVALID), + ({"sequence": True}, EnvelopeRejection.SEQUENCE_INVALID), + ({"sequence": -4}, EnvelopeRejection.SEQUENCE_INVALID), + ({"observed_at": "2026-09-01T10:00:00"}, EnvelopeRejection.CLOCK_INVALID), + ({"clock": {"source": "gps", "uncertainty_ms": 0}}, EnvelopeRejection.CLOCK_INVALID), + ({"clock": {"source": "fixture", "uncertainty_ms": -1}}, EnvelopeRejection.CLOCK_INVALID), + ({"event_kind": "prompt_injected"}, EnvelopeRejection.EVENT_KIND_INVALID), + ({"summary": {"tool_name": "bash -c 'rm -rf'"}}, EnvelopeRejection.SUMMARY_INVALID), + ({"summary": {"turn": "one"}}, EnvelopeRejection.SUMMARY_INVALID), + ({"source_refs": {"tool_call_id": "/Users/someone/call"}}, EnvelopeRejection.SOURCE_REF_INVALID), + ({"source_refs": {"tool_call_id": "sk-abcdefghijklmnop0123"}}, EnvelopeRejection.PUBLIC_SAFETY_VIOLATION), + ], +) +def test_malformed_envelopes_carry_typed_reasons(mutation: dict[str, Any], reason: EnvelopeRejection) -> None: + record = envelope(1) + record.update(mutation) + with pytest.raises(ObserverEnvelopeError) as excinfo: + normalize_observer_envelope(record) + assert excinfo.value.reason is reason + + +# --- intake ---------------------------------------------------------------- + + +def intake(bound: int = 8) -> ShadowObserverIntake: + return ShadowObserverIntake( + provider_id=DSH_PROVIDER_ID, + observer_id="observer-1", + goal_id=GOAL, + clock_source=ClockSource.FIXTURE, + buffer_bound=bound, + ) + + +def test_intake_exposes_no_control_surface() -> None: + names = {name.lower() for name in dir(ShadowObserverIntake)} + assert not names & {"send", "command", "schedule", "retry", "stop", "resume", "pause", "inject"} + + +def test_intake_bounds_buffer_and_counts_drops_without_raising() -> None: + sink = intake(bound=2) + accepted = [sink.observe(envelope(index)) for index in range(5)] + assert accepted == [True, True, False, False, False] + record = sink.stats(emitted_at=T0).as_dict() + assert record["buffer_bound"] == 2 + assert record["accepted_event_count"] == 2 + assert record["backpressure_drop_count"] == 3 + assert record["outbound_endpoints"] == [] + assert record["observation_entered_worker_context"] is False + assert normalize_observer_stats(record).backpressure_drop_count == 3 + + +def test_intake_isolates_observer_crashes() -> None: + class Exploding(dict): + def __iter__(self): # noqa: ANN204 + raise RuntimeError("observer bug") + + sink = intake() + assert sink.observe(Exploding()) is False + assert sink.observe(envelope(0)) is True + assert sink.stats(emitted_at=T0).observer_failure_count == 1 + + +def test_intake_flush_failure_is_counted_not_raised() -> None: + sink = intake() + sink.observe(envelope(0)) + sink.observe(envelope(1)) + + def broken(_: list[dict[str, Any]]) -> None: + raise OSError("disk full") + + assert sink.flush(broken, emitted_at=T0) == [] + record = sink.stats(emitted_at=T0) + assert record.observer_failure_count == 1 + assert record.backpressure_drop_count == 2 + assert sink.buffered_count == 0 + + +def test_intake_rejects_and_counts_by_reason() -> None: + sink = intake() + sink.observe(envelope(0, transcript="x")) + sink.observe(envelope(1, command="stop")) + sink.observe(envelope(2, goal_id="other-goal")) + assert sink.stats(emitted_at=T0).rejected_by_reason == { + EnvelopeRejection.RAW_MATERIAL_FIELD_REJECTED.value: 1, + EnvelopeRejection.CONTROL_FIELD_REJECTED.value: 1, + EnvelopeRejection.IDENTITY_INVALID.value: 1, + } + + +def test_stats_record_rejects_unknown_fields() -> None: + with pytest.raises(ValueError, match="unsupported fields"): + normalize_observer_stats(stats(send_path="agent.send")) + + +# --- receipt --------------------------------------------------------------- + + +def test_receipt_without_observations_is_invalid() -> None: + receipt = receipt_for() + assert receipt["status"] == ReceiptStatus.INVALID.value + assert receipt["reason_codes"] == [ReceiptReason.NO_OBSERVATIONS.value] + + +def test_receipt_is_valid_for_contiguous_stream_with_stats() -> None: + receipt = receipt_for(envelope(0), envelope(1, observed_at=at(1)), envelope(2, observed_at=at(2)), stats()) + assert receipt["status"] == ReceiptStatus.VALID.value + assert receipt["reason_codes"] == [] + assert receipt["outbound_endpoints"] == [] + assert receipt["lost_event_count"] == 0 + assert receipt["observed_event_count"] == 3 + assert receipt["provider_ids"] == [DSH_PROVIDER_ID] + + +def test_receipt_counts_sequence_gaps_as_lost_events() -> None: + receipt = receipt_for(envelope(0), envelope(4, observed_at=at(1)), envelope(5, observed_at=at(2)), stats()) + assert receipt["lost_event_count"] == 3 + assert receipt["status"] == ReceiptStatus.DEGRADED.value + assert receipt["reason_codes"] == [ReceiptReason.SEQUENCE_GAP.value] + + +def test_receipt_counts_duplicate_sequences_separately() -> None: + receipt = receipt_for(envelope(0), envelope(0, observed_at=at(1)), stats()) + assert receipt["duplicate_sequence_count"] == 1 + assert receipt["lost_event_count"] == 0 + assert ReceiptReason.SEQUENCE_DUPLICATE.value in receipt["reason_codes"] + + +@pytest.mark.parametrize( + ("stats_override", "reason"), + [ + ({"observer_failure_count": 1}, ReceiptReason.OBSERVER_FAILURE), + ({"rejected_event_count": 1, "rejected_by_reason": {"control_field_rejected": 1}}, ReceiptReason.CONTROL_FIELD_REJECTED), + ], +) +def test_receipt_quarantines_observer_failure_and_control_fields(stats_override: dict[str, Any], reason: ReceiptReason) -> None: + receipt = receipt_for(envelope(0), stats(**stats_override)) + assert receipt["status"] == ReceiptStatus.QUARANTINED.value + assert receipt["reason_codes"] == [reason.value] + + +def test_receipt_quarantines_malformed_ledger_records() -> None: + receipt = receipt_for(envelope(0), stats(), {"schema_version": "unknown"}, malformed=1) + assert receipt["ledger_invalid_record_count"] == 2 + assert receipt["status"] == ReceiptStatus.QUARANTINED.value + + +@pytest.mark.parametrize( + ("stats_override", "reason"), + [ + ({"outbound_endpoints": ["loopx-continuation"]}, ReceiptReason.OUTBOUND_ENDPOINT_CONFIGURED), + ({"observation_entered_worker_context": True}, ReceiptReason.OBSERVATION_ENTERED_WORKER_CONTEXT), + ], +) +def test_receipt_invalidates_any_outbound_or_worker_context_path(stats_override: dict[str, Any], reason: ReceiptReason) -> None: + receipt = receipt_for(envelope(0), stats(observer_failure_count=1, **stats_override)) + assert receipt["status"] == ReceiptStatus.INVALID.value # invalid outranks quarantined + assert reason.value in receipt["reason_codes"] + assert ReceiptReason.OBSERVER_FAILURE.value in receipt["reason_codes"] + + +@pytest.mark.parametrize( + ("records", "reason"), + [ + ([envelope(0, clock={"source": "observer_wall_clock", "uncertainty_ms": 1001}), stats()], ReceiptReason.CLOCK_UNCERTAINTY_EXCEEDED), + ([envelope(0), stats(backpressure_drop_count=2)], ReceiptReason.BACKPRESSURE_DROP), + ([envelope(0), stats(rejected_event_count=1, rejected_by_reason={"raw_material_field_rejected": 1})], ReceiptReason.RAW_MATERIAL_REJECTED), + ([envelope(0)], ReceiptReason.OBSERVER_STATS_MISSING), + ], +) +def test_receipt_degrades_but_keeps_evidence(records: list[dict[str, Any]], reason: ReceiptReason) -> None: + receipt = receipt_for(*records) + assert receipt["status"] == ReceiptStatus.DEGRADED.value + assert receipt["reason_codes"] == [reason.value] + + +def test_receipt_clock_uncertainty_at_threshold_is_visible_not_degraded() -> None: + receipt = receipt_for(envelope(0, clock={"source": "observer_wall_clock", "uncertainty_ms": 1000}), stats()) + assert receipt["clock"] == {"sources": ["harness_event_time", "observer_wall_clock"], "max_uncertainty_ms": 1000} + assert receipt["status"] == ReceiptStatus.VALID.value + + +def test_receipt_sums_latest_stats_per_observer_instance() -> None: + receipt = receipt_for( + envelope(0), + stats(observer_id="a", backpressure_drop_count=1), + stats(observer_id="a", backpressure_drop_count=4), + stats(observer_id="b", backpressure_drop_count=2), + ) + assert receipt["backpressure_drop_count"] == 6 + assert receipt["observer_ids"] == ["a", "b"] + + +# --- projection ------------------------------------------------------------ + + +def projection_for(*records: dict[str, Any], **kwargs: Any) -> dict[str, Any]: + return build_diagnostic_projection(read_ledger(records, goal_id=GOAL), **kwargs) + + +def test_projection_declares_read_only_boundary() -> None: + projection = projection_for(envelope(0), stats()) + assert projection["mode"] == "read_only" + assert projection["authority"] == "none" + assert projection["write_scope"] == "diagnostic_ledger_only" + assert projection["worker_influence"] == "none" + assert set(projection) & {"command", "next_action", "recommended_action", "gate"} == set() + + +def test_projection_stall_requires_active_stage_and_silence() -> None: + running = projection_for( + envelope(0, ObserverEventKind.TURN_STARTED), + envelope(1, ObserverEventKind.STEP_STARTED, observed_at=at(1)), + stats(), + as_of=at(6 * 60), + stall_threshold_ms=300_000, + ) + assert running["stage"] == DiagnosticStage.RUNNING.value + assert running["stall"]["detected"] is True + assert DiagnosticSignal.STALL_SUSPECTED.value in running["signals"] + + idle = projection_for( + envelope(0, ObserverEventKind.TURN_STARTED), + envelope(1, ObserverEventKind.TURN_ENDED, observed_at=at(1), summary={"reason": "completed"}), + stats(), + as_of=at(6 * 60), + ) + assert idle["stage"] == DiagnosticStage.IDLE.value + assert idle["stall"]["detected"] is False + + +def test_projection_repetition_counts_consecutive_identical_tool_runs() -> None: + tools = ["read", "read", "bash", "read", "read", "read"] + records = [ + envelope(index, ObserverEventKind.TOOL_CALLED, observed_at=at(index), summary={"tool_name": tool}) + for index, tool in enumerate(tools) + ] + projection = projection_for(*records, stats()) + assert projection["repetition"] == {"detected": True, "threshold": 3, "longest_tool_run": 3, "tool_name": "read"} + below = projection_for(*records[:2], stats()) + assert below["repetition"]["detected"] is False + + +def test_projection_recovery_and_stage_transitions() -> None: + unrecovered = projection_for( + envelope(0, ObserverEventKind.STEP_STARTED), + envelope(1, ObserverEventKind.AGENT_ERROR, observed_at=at(1), summary={"error_class": "Timeout"}), + stats(), + ) + assert unrecovered["stage"] == DiagnosticStage.ERRORED.value + assert unrecovered["recovery"] == {"error_count": 1, "recovered_error_count": 0, "unrecovered_error_count": 1} + assert DiagnosticSignal.UNRECOVERED_ERROR.value in unrecovered["signals"] + + recovered = projection_for( + envelope(0, ObserverEventKind.STEP_STARTED), + envelope(1, ObserverEventKind.AGENT_ERROR, observed_at=at(1)), + envelope(2, ObserverEventKind.STEP_ENDED, observed_at=at(2)), + envelope(3, ObserverEventKind.SESSION_DISPOSED, observed_at=at(3)), + stats(), + ) + assert recovered["stage"] == DiagnosticStage.DISPOSED.value + assert recovered["recovery"]["unrecovered_error_count"] == 0 + assert recovered["recovery"]["recovered_error_count"] == 1 + assert recovered["signals"] == [] + + +def test_projection_surfaces_event_loss_and_integrity() -> None: + projection = projection_for(envelope(0), envelope(3, observed_at=at(1)), stats()) + assert DiagnosticSignal.EVENT_LOSS.value in projection["signals"] + assert projection["integrity"]["status"] == ReceiptStatus.DEGRADED.value + assert projection["evidence"]["lost_event_count"] == 2 + + +# --- ledger ---------------------------------------------------------------- + + +def test_ledger_ref_is_relative_and_goal_scoped(tmp_path: Path) -> None: + assert ledger_ref("goal:alpha") == "reliability_diagnostics/goal_alpha.ndjson" + assert ledger_path(tmp_path, "goal-a") == tmp_path / "reliability_diagnostics" / "goal-a.ndjson" + with pytest.raises(ValueError): + ledger_ref("../escape") + + +def test_ledger_append_is_line_oriented_and_tolerates_malformed_lines(tmp_path: Path) -> None: + path = ledger_path(tmp_path, GOAL) + assert append_ledger_records(path, [envelope(0)]) == 1 + assert append_ledger_records(path, []) == 0 + path.open("a", encoding="utf-8").write("not json\n") + assert append_ledger_records(path, [stats()]) == 1 + records, malformed = read_ledger_records(path) + assert [json.dumps(item, sort_keys=True) for item in records] == [ + json.dumps(envelope(0), sort_keys=True), + json.dumps(stats(), sort_keys=True), + ] + assert malformed == 1 + + +# --- fixture --------------------------------------------------------------- + + +def test_dsh_fixture_exercises_every_contract_hazard_and_stays_degraded() -> None: + result = run_dsh_fixture() + receipt = result["receipt"] + assert receipt["goal_id"] == FIXTURE_GOAL_ID + assert receipt["status"] == ReceiptStatus.DEGRADED.value + assert set(receipt["reason_codes"]) == { + ReceiptReason.SEQUENCE_GAP.value, + ReceiptReason.BACKPRESSURE_DROP.value, + ReceiptReason.RAW_MATERIAL_REJECTED.value, + ReceiptReason.CLOCK_UNCERTAINTY_EXCEEDED.value, + } + assert receipt["outbound_endpoints"] == [] + assert receipt["observer_failure_count"] == 0 + assert "transcript" not in json.dumps(result["ledger_records"]) + assert len(dsh_fixture_records()) == receipt["observed_event_count"] + receipt["rejected_event_count"] + receipt["backpressure_drop_count"] From cad7dc17352b5ded9dd1b07e2f47d25601b27ac9 Mon Sep 17 00:00:00 2001 From: song Date: Fri, 4 Sep 2026 20:00:54 +0800 Subject: [PATCH 02/13] feat(reliability-diagnostics): add DSH fixture smoke and CLI readback Deterministic fixture smoke proves the no-outbound-control invariant, bounded backpressure, visible loss and clock uncertainty, raw-material rejection, and the read-only projection boundary. Owner-local CLI module registers `loopx reliability-diagnostics ingest|receipt|status` (manpage-only help surface; no default help growth) and resolves the ledger through the existing runtime-root helper. Signed-off-by: song --- .../dsh-shadow-observer-fixture-smoke.py | 156 ++++++++++++++++ loopx/cli.py | 16 ++ loopx/cli_commands/reliability_diagnostics.py | 174 ++++++++++++++++++ loopx/help_surface.py | 1 + 4 files changed, 347 insertions(+) create mode 100644 examples/reliability_diagnostics/dsh-shadow-observer-fixture-smoke.py create mode 100644 loopx/cli_commands/reliability_diagnostics.py diff --git a/examples/reliability_diagnostics/dsh-shadow-observer-fixture-smoke.py b/examples/reliability_diagnostics/dsh-shadow-observer-fixture-smoke.py new file mode 100644 index 0000000000..3e9fc00745 --- /dev/null +++ b/examples/reliability_diagnostics/dsh-shadow-observer-fixture-smoke.py @@ -0,0 +1,156 @@ +#!/usr/bin/env python3 +"""Prove the L1 shadow-observer contract on the deterministic DSH-shaped fixture. + +Assertions come from the reliability-diagnostics design contract (RFC §3.1 +non-interference, §7.4 integrity receipt): no outbound control path, bounded +and counted loss, visible clock uncertainty, raw material never persisted, and +a read-only projection with no authority. The smoke also proves the CLI +readback path round-trips the same ledger. +""" + +from __future__ import annotations + +import json +import os +import subprocess +import sys +import tempfile +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(REPO_ROOT)) + +from loopx.capabilities.reliability_diagnostics import ( # noqa: E402 + CONTROL_FIELD_FAMILIES, + ENVELOPE_FIELDS, + FIXTURE_GOAL_ID, + RAW_MATERIAL_FIELD_FAMILIES, + dsh_fixture_records, + run_dsh_fixture, +) +from loopx.capabilities.reliability_diagnostics.fixture import ( # noqa: E402 + FIXTURE_BUFFER_BOUND, + FIXTURE_UNCERTAIN_CLOCK_MS, +) + + +def run_cli(*args: str, runtime_root: Path, stdin: str | None = None) -> dict[str, object]: + completed = subprocess.run( + [ + sys.executable, + "-m", + "loopx.cli", + "--runtime-root", + str(runtime_root), + "--format", + "json", + "reliability-diagnostics", + *args, + ], + cwd=REPO_ROOT, + env={**os.environ, "PYTHONPATH": str(REPO_ROOT)}, + input=stdin, + capture_output=True, + text=True, + check=False, + ) + assert completed.returncode == 0, completed.stderr or completed.stdout + return json.loads(completed.stdout) + + +def assert_no_outbound_control(receipt: dict, projection: dict) -> None: + assert receipt["outbound_endpoints"] == [], receipt + assert receipt["observation_entered_worker_context"] is False, receipt + assert projection["mode"] == "read_only", projection + assert projection["authority"] == "none", projection + assert projection["worker_influence"] == "none", projection + flattened = {name.replace("_", "") for name in ENVELOPE_FIELDS} + assert not flattened & CONTROL_FIELD_FAMILIES + assert not flattened & RAW_MATERIAL_FIELD_FAMILIES + + +def assert_bounded_failure(result: dict) -> None: + receipt = result["receipt"] + stats = result["stats"] + fixture_records = dsh_fixture_records() + # Every fixture record is accounted for exactly once: persisted, rejected, or dropped. + assert len(fixture_records) == ( + receipt["observed_event_count"] + + receipt["rejected_event_count"] + + receipt["backpressure_drop_count"] + ), (len(fixture_records), receipt) + assert stats["buffer_bound"] == FIXTURE_BUFFER_BOUND + assert receipt["backpressure_drop_count"] == 3, receipt + # Sequence 10 and the rejected sequence 19 are visible as gaps; trailing + # drops are visible only through the stats record. + assert receipt["lost_event_count"] == 2, receipt + assert receipt["clock"]["max_uncertainty_ms"] == FIXTURE_UNCERTAIN_CLOCK_MS + assert receipt["rejected_by_reason"] == {"raw_material_field_rejected": 1}, receipt + assert receipt["observer_failure_count"] == 0 + assert receipt["status"] == "degraded", receipt + assert set(receipt["reason_codes"]) == { + "sequence_gap", + "backpressure_drop", + "raw_material_rejected", + "clock_uncertainty_exceeded", + }, receipt + assert "transcript" not in json.dumps(result["ledger_records"]) + assert "protected task content" not in json.dumps(result) + + +def assert_projection_signals(projection: dict) -> None: + assert projection["repetition"] == { + "detected": True, + "threshold": 3, + "longest_tool_run": 3, + "tool_name": "read", + }, projection + assert projection["recovery"] == { + "error_count": 1, + "recovered_error_count": 1, + "unrecovered_error_count": 0, + }, projection + assert projection["stall"]["detected"] is False, projection + assert set(projection["signals"]) == { + "repetition_suspected", + "event_loss", + "integrity_not_valid", + }, projection + assert projection["integrity"]["status"] == "degraded" + + +def assert_cli_readback(result: dict) -> None: + with tempfile.TemporaryDirectory() as tmp: + runtime_root = Path(tmp) + ndjson = "\n".join(json.dumps(record) for record in result["ledger_records"]) + "\n" + ingest = run_cli("ingest", "--goal-id", FIXTURE_GOAL_ID, "--input", "-", runtime_root=runtime_root, stdin=ndjson) + assert ingest["ok"] is True, ingest + assert ingest["appended_record_count"] == len(result["ledger_records"]), ingest + assert ingest["ledger_ref"] == f"reliability_diagnostics/{FIXTURE_GOAL_ID}.ndjson" + assert str(runtime_root) not in json.dumps(ingest) + + receipt = run_cli("receipt", "--goal-id", FIXTURE_GOAL_ID, runtime_root=runtime_root) + assert receipt["receipt"] == result["receipt"], receipt + status = run_cli("status", "--goal-id", FIXTURE_GOAL_ID, runtime_root=runtime_root) + assert status["projection"] == result["projection"], status + + # A control-shaped record is refused at the ledger door and quarantines the receipt. + poisoned = dict(result["ledger_records"][0]) + poisoned["command"] = {"kind": "stop"} + rejected = run_cli("ingest", "--goal-id", FIXTURE_GOAL_ID, "--input", "-", runtime_root=runtime_root, stdin=json.dumps(poisoned) + "\n") + assert rejected["rejected_by_reason"] == {"control_field_rejected": 1}, rejected + assert run_cli("receipt", "--goal-id", FIXTURE_GOAL_ID, runtime_root=runtime_root)["receipt"]["status"] == "quarantined" + + +def main() -> int: + result = run_dsh_fixture() + assert_no_outbound_control(result["receipt"], result["projection"]) + assert_bounded_failure(result) + assert_projection_signals(result["projection"]) + assert_cli_readback(result) + print("reliability-diagnostics dsh-shadow-observer-fixture-smoke: ok") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/loopx/cli.py b/loopx/cli.py index 8dea8f8510..4d590754bd 100644 --- a/loopx/cli.py +++ b/loopx/cli.py @@ -156,6 +156,10 @@ handle_shared_goal_alignment_command, register_shared_goal_alignment_command, ) +from .cli_commands.reliability_diagnostics import ( + handle_reliability_diagnostics_command, + register_reliability_diagnostics_commands, +) from .cli_rollout import append_cli_rollout_event from .capabilities.project_skill_delivery.cli import ( handle_project_skill_command, @@ -254,6 +258,8 @@ def build_parser() -> LoopXArgumentParser: register_capability_commands(sub, add_subcommand_format) + register_reliability_diagnostics_commands(sub, add_subcommand_format) + register_extension_commands(sub, add_subcommand_format) register_change_quality_commands(sub, add_subcommand_format) @@ -442,6 +448,16 @@ def main(argv: list[str] | None = None) -> int: if capability_result is not None: return capability_result + reliability_diagnostics_result = handle_reliability_diagnostics_command( + args, + registry_path=registry_path, + runtime_root_arg=args.runtime_root, + output_format=output_format, + print_payload=print_payload, + ) + if reliability_diagnostics_result is not None: + return reliability_diagnostics_result + extension_result = handle_extension_command( args, runtime_root_arg=args.runtime_root, diff --git a/loopx/cli_commands/reliability_diagnostics.py b/loopx/cli_commands/reliability_diagnostics.py new file mode 100644 index 0000000000..b446a6ee31 --- /dev/null +++ b/loopx/cli_commands/reliability_diagnostics.py @@ -0,0 +1,174 @@ +"""Owner-local CLI readback for the reliability-diagnostics ledger.""" + +from __future__ import annotations + +import argparse +import sys +from collections.abc import Callable +from datetime import datetime, timezone +from pathlib import Path +from typing import Any + +from ..capabilities.reliability_diagnostics import ( + CAPABILITY_ID, + OBSERVER_STATS_SCHEMA_VERSION, + ClockSource, + ShadowObserverIntake, + append_ledger_records, + build_diagnostic_projection, + build_integrity_receipt, + ledger_path, + ledger_ref, + normalize_observer_stats, + parse_ndjson_lines, + read_ledger, + read_ledger_records, +) +from ..history import load_registry +from ..paths import resolve_runtime_root + +PrintPayload = Callable[[dict[str, Any], str, Callable[[dict[str, Any]], str]], str | None] +FormatSelector = Callable[..., str] +AddFormat = Callable[[argparse.ArgumentParser], None] + +INGEST_OBSERVER_ID = "loopx-cli-ingest" +INGEST_BUFFER_BOUND = 4096 + + +def register_reliability_diagnostics_commands( + subparsers: argparse._SubParsersAction[argparse.ArgumentParser], + add_subcommand_format: AddFormat, +) -> None: + parser = subparsers.add_parser( + "reliability-diagnostics", + help="Read back the L1 shadow-observer ledger: ingest, receipt, status.", + ) + commands = parser.add_subparsers(dest="reliability_diagnostics_command", required=True) + ingest = commands.add_parser( + "ingest", + help="Validate observer envelopes and append accepted ones to the goal ledger.", + ) + add_subcommand_format(ingest) + ingest.add_argument("--goal-id", required=True) + ingest.add_argument("--input", required=True, help="NDJSON file path, or - for stdin.") + receipt = commands.add_parser("receipt", help="Render the treatment-integrity receipt.") + add_subcommand_format(receipt) + receipt.add_argument("--goal-id", required=True) + status = commands.add_parser("status", help="Render the read-only diagnostic projection.") + add_subcommand_format(status) + status.add_argument("--goal-id", required=True) + status.add_argument("--as-of", help="Timezone-aware ISO-8601 time used for stall age.") + + +def _render(payload: dict[str, Any]) -> str: + lines = [f"# Reliability Diagnostics ({payload.get('command')})", ""] + for key in ("goal_id", "ledger_ref", "appended_record_count", "rejected_by_reason"): + if key in payload: + lines.append(f"- {key}: `{payload[key]}`") + for section in ("receipt", "projection"): + body = payload.get(section) + if isinstance(body, dict): + lines.append(f"- {section}.status: `{body.get('status') or body.get('integrity', {}).get('status')}`") + for key in ("stage", "signals", "reason_codes", "lost_event_count", "backpressure_drop_count"): + if key in body: + lines.append(f"- {section}.{key}: `{body[key]}`") + return "\n".join(lines) + "\n" + + +def _ingest(path: Path, goal_id: str, source: str) -> dict[str, Any]: + if source == "-": + lines = sys.stdin.read().splitlines() + else: + lines = Path(source).expanduser().read_text(encoding="utf-8").splitlines() + parsed, malformed = parse_ndjson_lines(lines) + intake = ShadowObserverIntake( + provider_id="loopx-core", + observer_id=INGEST_OBSERVER_ID, + goal_id=goal_id, + clock_source=ClockSource.OBSERVER_WALL_CLOCK, + buffer_bound=INGEST_BUFFER_BOUND, + ) + appended = 0 + passthrough_stats = 0 + for record in parsed: + if isinstance(record, dict) and record.get("schema_version") == OBSERVER_STATS_SCHEMA_VERSION: + try: + stats = normalize_observer_stats(record) + except ValueError: + malformed += 1 + continue + if stats.goal_id != goal_id: + malformed += 1 + continue + appended += append_ledger_records(path, [stats.as_dict()]) + passthrough_stats += 1 + continue + intake.observe(record) + if intake.buffered_count >= INGEST_BUFFER_BOUND: + appended += len(intake.flush(lambda records: append_ledger_records(path, records[:-1]), emitted_at=_now())) + appended += len(intake.flush(lambda records: append_ledger_records(path, records[:-1]), emitted_at=_now())) + stats_record = intake.stats(emitted_at=_now()) + # A clean ingest is a transparent copy of the observer output. The ingest + # gate records itself only when it refused, dropped, or failed something, + # so that violation stays durable and visible in the receipt. + gate_recorded = bool( + stats_record.rejected_event_count + or stats_record.backpressure_drop_count + or stats_record.observer_failure_count + ) + if gate_recorded: + appended += append_ledger_records(path, [stats_record.as_dict()]) + return { + "ok": True, + "command": "ingest", + "goal_id": goal_id, + "ledger_ref": ledger_ref(goal_id), + "appended_record_count": appended, + "accepted_envelope_count": stats_record.accepted_event_count, + "passthrough_stats_count": passthrough_stats, + "rejected_event_count": stats_record.rejected_event_count, + "rejected_by_reason": dict(stats_record.rejected_by_reason), + "malformed_line_count": malformed, + "observer_failure_count": stats_record.observer_failure_count, + "ingest_gate_recorded": gate_recorded, + } + + +def _now() -> str: + return datetime.now(timezone.utc).isoformat(timespec="seconds") + + +def handle_reliability_diagnostics_command( + args: argparse.Namespace, + *, + registry_path: Path, + runtime_root_arg: str | None, + output_format: FormatSelector, + print_payload: PrintPayload, +) -> int | None: + if args.command != "reliability-diagnostics": + return None + goal_id = str(args.goal_id) + runtime_root = resolve_runtime_root( + load_registry(registry_path) if registry_path.is_file() else {}, + runtime_root_arg, + registry_path=registry_path, + ) + try: + path = ledger_path(runtime_root, goal_id) + command = args.reliability_diagnostics_command + if command == "ingest": + payload = _ingest(path, goal_id, str(args.input)) + else: + records, malformed = read_ledger_records(path) + reading = read_ledger(records, goal_id=goal_id, malformed_line_count=malformed) + payload = {"ok": True, "command": command, "goal_id": goal_id, "ledger_ref": ledger_ref(goal_id)} + if command == "receipt": + payload["receipt"] = build_integrity_receipt(reading) + else: + payload["projection"] = build_diagnostic_projection(reading, as_of=args.as_of) + except (OSError, ValueError) as exc: + print(f"error: {CAPABILITY_ID}: {exc}", file=sys.stderr) + return 2 + print_payload(payload, output_format(args), _render) + return 0 diff --git a/loopx/help_surface.py b/loopx/help_surface.py index cc6e142d60..f562a02540 100644 --- a/loopx/help_surface.py +++ b/loopx/help_surface.py @@ -329,6 +329,7 @@ "promotion-gate", "read-only-map", "refresh-state", + "reliability-diagnostics", "register-authority-source", "registry-boundary", "reward", From 8deb94f05f6324b6ba2f7091c7caee75b7a7c916 Mon Sep 17 00:00:00 2001 From: song Date: Fri, 4 Sep 2026 20:11:26 +0800 Subject: [PATCH 03/13] feat(dsh-loopx-plugin): add default-off L1 shadow observer and register the capability observer.ts is the dsh-session-events provider: a physically separate module from driver.ts that consumes read-only harness events, keeps a bounded buffer with counted drops, isolates every hook and flush failure, and appends envelopes plus a stats record (empty outbound_endpoints) to the LoopX reliability_diagnostics ledger. apply() calls applyObserver only when LOOPX_DSH_SHADOW_OBSERVER_GOAL_ID declares one goal; otherwise no hook is registered and no file is written. The reliability-diagnostics catalog entry registers the builtin capability and declares dsh-session-events as an extension provider that is declared but not installed, enabled, or ready. Package README records the placement rationale and field tables; a parity test pins the shared field names across Python and TypeScript and asserts the observer imports nothing from the driver. Signed-off-by: song --- loopx/capabilities/catalog.py | 42 ++ .../reliability_diagnostics/README.md | 171 +++++++ .../reliability_diagnostics/README.zh-CN.md | 144 ++++++ .../reliability_diagnostics/catalog_entry.py | 98 ++++ packages/dsh-loopx-plugin/src/driver.ts | 6 + packages/dsh-loopx-plugin/src/index.ts | 10 + packages/dsh-loopx-plugin/src/observer.ts | 448 ++++++++++++++++++ .../dsh-loopx-plugin/tests/observer.spec.ts | 235 +++++++++ .../test_capability_extension_registry.py | 11 +- ...st_reliability_diagnostics_dsh_provider.py | 82 ++++ 10 files changed, 1246 insertions(+), 1 deletion(-) create mode 100644 loopx/capabilities/reliability_diagnostics/README.md create mode 100644 loopx/capabilities/reliability_diagnostics/README.zh-CN.md create mode 100644 loopx/capabilities/reliability_diagnostics/catalog_entry.py create mode 100644 packages/dsh-loopx-plugin/src/observer.ts create mode 100644 packages/dsh-loopx-plugin/tests/observer.spec.ts create mode 100644 tests/capabilities/test_reliability_diagnostics_dsh_provider.py diff --git a/loopx/capabilities/catalog.py b/loopx/capabilities/catalog.py index c17e7e9f97..c5f8b641a8 100644 --- a/loopx/capabilities/catalog.py +++ b/loopx/capabilities/catalog.py @@ -26,6 +26,7 @@ from .deep_research.catalog_entry import DEEP_RESEARCH_CATALOG_ENTRY from .public_safe_outbound.catalog_entry import PUBLIC_SAFE_OUTBOUND_CATALOG_ENTRY from .connector_registry.catalog_entry import CONNECTOR_REGISTRY_CATALOG_ENTRY +from .reliability_diagnostics.catalog_entry import RELIABILITY_DIAGNOSTICS_CATALOG_ENTRY from .registry import CapabilityRegistry CAPABILITY_CATALOG_SCHEMA_VERSION = "loopx_capability_catalog_v0" @@ -52,6 +53,7 @@ DEEP_RESEARCH_CATALOG_ENTRY, PUBLIC_SAFE_OUTBOUND_CATALOG_ENTRY, CONNECTOR_REGISTRY_CATALOG_ENTRY, + RELIABILITY_DIAGNOSTICS_CATALOG_ENTRY, ) # Preserve the original import surface while routing all reads through the registry. CAPABILITIES = BUILTIN_CAPABILITIES @@ -78,6 +80,44 @@ def _summary(record: Mapping[str, Any]) -> dict[str, Any]: } +def _register_declared_extension_providers( + registry: CapabilityRegistry, + record: Mapping[str, Any], +) -> None: + """Declare extension-delivered implementations named by a builtin entry. + + A provider distributed outside the Python extension lifecycle (for example + an npm plugin) has no manifest or state file. The owning capability + declares it so the catalog reports `declared=true` and + `installed=enabled=ready=false` until a real lifecycle proves otherwise. + """ + + for implementation in record.get("implementation_providers") or []: + if not isinstance(implementation, Mapping) or implementation.get("origin") != "extension": + continue + provider_id = str(implementation["provider_id"]) + if not any(provider["id"] == provider_id for provider in registry.providers()): + registry.register_provider( + { + "id": provider_id, + "origin": "extension", + "declared": True, + "installed": False, + "enabled": False, + "ready": False, + } + ) + registry.register_implementation( + { + "capability_id": record["id"], + "provider_id": provider_id, + "protocol": implementation["protocol"], + "package": implementation.get("package"), + "status": implementation.get("status"), + } + ) + + def build_capability_registry( extension_manifest_paths: Iterable[str | Path] = (), *, @@ -96,6 +136,8 @@ def build_capability_registry( ) for record in BUILTIN_CAPABILITIES: registry.register_capability(record) + for record in BUILTIN_CAPABILITIES: + _register_declared_extension_providers(registry, record) for manifest in extension_catalog_entries( extension_manifest_paths, state_file=extension_state_file, diff --git a/loopx/capabilities/reliability_diagnostics/README.md b/loopx/capabilities/reliability_diagnostics/README.md new file mode 100644 index 0000000000..f767091328 --- /dev/null +++ b/loopx/capabilities/reliability_diagnostics/README.md @@ -0,0 +1,171 @@ +# Reliability Diagnostics + +[中文](README.zh-CN.md) | [RFC](../../../docs/architecture/rfcs/long-running-agent-reliability-diagnostics-governed-delivery-v0.md) + +Status: experimental, built in, default off, goal scoped. This package ships +the P0 slice of the reliability-diagnostics RFC: the **L1 shadow observer** +contract and its first real event source, DeepSeek Harness (DSH). + +An L1 observer sees a long-running agent session and writes an independent +diagnostic record about it. It may **never** influence that session. This +capability makes that promise a machine contract rather than a policy: the +envelope schema cannot express a command, the receipt records the empty set of +outbound endpoints, observer failure is counted and quarantines the evidence, +and the projection carries `mode: read_only` and `authority: none`. + +```mermaid +flowchart LR + H["DSH agent loop"] -->|"read-only events"| O["observer.ts (dsh-session-events)"] + O -->|"envelopes + stats, NDJSON"| L["reliability_diagnostics/.ndjson"] + L --> R["integrity receipt"] + L --> P["read-only projection"] + O -. "no send, schedule, gate, tool, or worker-state path" .-> H +``` + +The dashed edge is an asserted absence. Tests reject envelopes carrying +control-shaped fields, the TypeScript module imports nothing from the +continuation driver, and the receipt turns `invalid` if any outbound endpoint +ever appears. + +## Placement Rationale + +- **Capability id `reliability-diagnostics`** (built in, provider + `loopx-core`). The caller outcome is "is this run admissible passive + evidence, and what does it say about stage, stall, repetition, and + recovery?" No existing capability owns diagnostics without authority. + Session runtime is a runtime-authority projection, so the diagnostic ledger + and projection are **siblings** of it, never merged into it. Ids are + kebab-case like every other catalog id; the package directory is + `reliability_diagnostics`. +- **Provider id `dsh-session-events`** (origin `extension`). It is delivered by + the npm package `packages/dsh-loopx-plugin` as `src/observer.ts`, physically + separate from `driver.ts`. Because an npm plugin has no Python + `extension.toml` lifecycle, the capability declares the provider on its + catalog entry and the registry reports it `declared=true`, + `installed=enabled=ready=false`. The precedent is + `repository_change_window`, which declares its `git-hook` provider the same + way. +- **Helpers stay local.** Ledger, receipt, and projection reducers live inside + this package. The only shared imports are the public-safe value validator + and the `SOURCE_ID_KEYS` identity tuple; the session-runtime substring + classifier is deliberately not reused. + +## Contract + +### Observer envelope (`reliability_observer_envelope_v0`) + +| Field | Type | Rule | +| --- | --- | --- | +| `schema_version` | literal | `reliability_observer_envelope_v0` | +| `capability_id` | literal | `reliability-diagnostics` | +| `provider_id` | identity token | e.g. `dsh-session-events` | +| `goal_id`, `session_id` | identity token | `^[A-Za-z0-9][A-Za-z0-9_.:-]{0,120}$` | +| `agent_id` | identity token, optional | | +| `sequence` | integer >= 0 | observer-assigned, monotonic per session; gaps are counted as loss | +| `observed_at` | ISO-8601 with timezone | | +| `clock.source` | enum | `harness_event_time`, `observer_wall_clock`, `fixture` | +| `clock.uncertainty_ms` | integer >= 0 | declared, never inferred | +| `event_kind` | enum | `session_started`, `turn_started`, `turn_ended`, `step_started`, `step_ended`, `user_message`, `tool_called`, `tool_completed`, `agent_status`, `agent_pre_step`, `agent_error`, `session_disposed`, `unsupported` | +| `summary` | object | only `turn`, `step` (integers) and `reason`, `status`, `tool_name`, `error_class`, `source_event_type`, `message_source_kind` (compact tokens) | +| `source_refs` | object | only id keys: `event_id`, `event_seq`, `tool_call_id`, `message_id`, `outcome_id`, `gate_id`, `approval_id`, `artifact_id`, `run_id`, `ref_id` | + +Any other field is rejected with a typed reason: `control_field_rejected` +(`command`, `send`, `prompt`, `schedule`, `retry`, `stop`, `resume`, `gate`, +`tool_call`, `worker_state`, ...), `raw_material_field_rejected` +(`transcript`, `messages`, `content`, `text`, `arguments`, `output`, +`stdout`, `stderr`, `log`, `cwd`, `token`, ...), or +`unsupported_field_rejected`. Values also pass the shared public-safe check, +so absolute local paths and credential-like tokens fail closed. + +### Observer stats (`reliability_observer_stats_v0`) + +Written by every observer implementation next to its envelopes. Fields: +`observer_id`, `emitted_at`, `observed_event_count`, `accepted_event_count`, +`rejected_event_count`, `rejected_by_reason`, `buffer_bound`, +`backpressure_drop_count`, `observer_failure_count`, `outbound_endpoints` +(must be `[]`), `observation_entered_worker_context` (must be `false`), +`clock_source`. Stats are cumulative per observer instance; the receipt keeps +the latest record per `observer_id` and sums across instances. + +### Integrity receipt (`reliability_integrity_receipt_v0`) + +| Field | Meaning | +| --- | --- | +| `status` | `valid`, `degraded`, `quarantined`, `invalid` (total, ordered) | +| `reason_codes` | typed list; empty only when `valid` | +| `observed_event_count`, `session_count` | accepted envelopes in the ledger | +| `lost_event_count`, `duplicate_sequence_count` | per-session sequence gaps and repeats | +| `ledger_invalid_record_count` | malformed or foreign records found in the ledger | +| `rejected_event_count`, `rejected_by_reason` | refusals reported by observers | +| `buffer_bound`, `backpressure_drop_count`, `observer_failure_count` | bounded-failure evidence | +| `clock.sources`, `clock.max_uncertainty_ms` | declared clocks; > 1000 ms degrades | +| `outbound_endpoints`, `observation_entered_worker_context` | must be `[]` / `false` | +| `event_kinds_consumed`, `summary_fields_consumed` | the exact sources and fields consumed | + +Status rules: `invalid` when there are no observations, any outbound endpoint, +or any observation entered worker context; otherwise `quarantined` when the +observer failed, a control-shaped record was seen, or the ledger holds +malformed records; otherwise `degraded` when events were lost, dropped, +duplicated, raw material was rejected, stats are missing, or clock uncertainty +exceeded the threshold; otherwise `valid`. + +### Diagnostic projection (`reliability_diagnostic_projection_v0`) + +| Field | Meaning | +| --- | --- | +| `mode`, `authority`, `write_scope`, `worker_influence` | `read_only`, `none`, `diagnostic_ledger_only`, `none` | +| `stage` | `unknown`, `idle`, `running`, `tool_running`, `errored`, `disposed` from the last event kind | +| `counts` | turns started/ended, steps, tool calls, errors | +| `stall` | detected only while active and silent for `threshold_ms` (default 300000) relative to `--as-of` | +| `repetition` | longest run of consecutive identical `tool_name` calls; detected at 3 | +| `recovery` | errors followed by a later completed step or non-error turn end count as recovered | +| `signals` | `stall_suspected`, `repetition_suspected`, `unrecovered_error`, `event_loss`, `integrity_not_valid` | +| `integrity` | the receipt status and reason codes | + +## Use It + +```bash +# Enable the DSH provider for exactly one goal, then start DSH as usual. +export LOOPX_DSH_SHADOW_OBSERVER_GOAL_ID= +# Optional: LOOPX_DSH_SHADOW_OBSERVER_LEDGER_DIR, LOOPX_DSH_SHADOW_OBSERVER_BUFFER_BOUND + +loopx reliability-diagnostics receipt --goal-id --format json +loopx reliability-diagnostics status --goal-id --format json +loopx reliability-diagnostics ingest --goal-id --input observer.ndjson --format json +``` + +The ledger lives at `/reliability_diagnostics/.ndjson`; +the default runtime root is the same one the rest of LoopX uses and the CLI +prints only the relative `ledger_ref`. `ingest` re-validates every line; a +clean ingest is a transparent copy, and the ingest gate records a stats record +of its own only when it refused, dropped, or failed something. + +With the environment variable unset the observer registers no hooks and +writes no files (feature-off parity). When set, `observer.ts` observes +`agent/session-start`, `agent/status`, `agent/error`, `agent/pre-step` +(pass-through), `session/event`, and `session/disposed`; token-level +`assistant/chunk` events are not consumed, which the receipt shows through +`event_kinds_consumed`. + +## Validation + +```bash +python3 examples/reliability_diagnostics/dsh-shadow-observer-fixture-smoke.py +python3 -m pytest tests/capabilities/test_reliability_diagnostics.py tests/capabilities/test_reliability_diagnostics_dsh_provider.py -q +cd packages/dsh-loopx-plugin && pnpm typecheck && pnpm test -- observer +``` + +The fixture is a fixed DSH-shaped stream with one missing sequence, one event +stamped with 1500 ms clock uncertainty, one raw-material-bearing record, and a +burst that overflows a 20-record buffer. Its receipt is `degraded` with exactly +`sequence_gap`, `backpressure_drop`, `raw_material_rejected`, and +`clock_uncertainty_exceeded`; the projection reports repetition on `read`, one +recovered error, and no stall. + +## Non-Goals In This Slice + +No dashboard surface, no L2 recommendations, no automatic recovery, no +writeback into goals, todos, gates, or session runtime, and no change to the +`loopx status` first screen. The observer attributes every session in the DSH +process to the single declared goal; per-session binding discovery is a +follow-up that must not reuse the driver's LoopX CLI path. diff --git a/loopx/capabilities/reliability_diagnostics/README.zh-CN.md b/loopx/capabilities/reliability_diagnostics/README.zh-CN.md new file mode 100644 index 0000000000..fe0b8ad434 --- /dev/null +++ b/loopx/capabilities/reliability_diagnostics/README.zh-CN.md @@ -0,0 +1,144 @@ +# Reliability Diagnostics 能力介绍 + +[English](README.md) | [RFC](../../../docs/architecture/rfcs/long-running-agent-reliability-diagnostics-governed-delivery-v0.md) + +状态:实验能力、内置、默认关闭、goal-scoped。本包交付 reliability-diagnostics RFC +的 P0 切片:**L1 shadow observer** 合约,以及第一个真实事件源 DeepSeek Harness(DSH)。 + +L1 observer 观察一个长时运行的 Agent 会话,并写下独立的诊断记录;它**永远不能** +影响该会话。本能力把这条承诺做成机器合约而不是口头规范:envelope schema 无法表达 +命令,receipt 记录 outbound endpoint 为空集,observer 故障被计数并让证据进入 +quarantined,projection 携带 `mode: read_only` 与 `authority: none`。 + +```mermaid +flowchart LR + H["DSH agent loop"] -->|"只读事件"| O["observer.ts (dsh-session-events)"] + O -->|"envelope + stats,NDJSON"| L["reliability_diagnostics/.ndjson"] + L --> R["integrity receipt"] + L --> P["只读 projection"] + O -. "没有 send / schedule / gate / tool / worker-state 通路" .-> H +``` + +虚线边表示一条被断言不存在的路径:测试拒绝带控制字段的 envelope,TypeScript 模块 +不从 continuation driver 导入任何东西,一旦出现 outbound endpoint,receipt 即为 `invalid`。 + +## 放置理由 + +- **能力 id `reliability-diagnostics`**(内置,provider `loopx-core`)。调用方结果是 + "这次运行是否可作为被动证据,它对 stage / stall / repetition / recovery 说了什么"。 + 没有现有能力拥有"无权威的诊断"这一结果。session runtime 是运行时权威投影,因此诊断 + ledger 与 projection 是它的**同级**,绝不合并进去。id 与其它 catalog 条目一样使用 + kebab-case;包目录为 `reliability_diagnostics`。 +- **Provider id `dsh-session-events`**(origin `extension`)。由 npm 包 + `packages/dsh-loopx-plugin` 的 `src/observer.ts` 交付,与 `driver.ts` 物理分离。 + npm 插件没有 Python `extension.toml` 生命周期,因此由能力在 catalog entry 上声明该 + provider,registry 报告 `declared=true`、`installed=enabled=ready=false`。先例是 + `repository_change_window` 声明其 `git-hook` provider 的方式。 +- **辅助逻辑留在本包内。** ledger、receipt、projection reducer 都在本包。仅共享 + public-safe 值校验器与 `SOURCE_ID_KEYS` 身份键;刻意不复用 session-runtime 的子串分类器。 + +## 合约 + +### Observer envelope(`reliability_observer_envelope_v0`) + +| 字段 | 类型 | 规则 | +| --- | --- | --- | +| `schema_version` | 字面量 | `reliability_observer_envelope_v0` | +| `capability_id` | 字面量 | `reliability-diagnostics` | +| `provider_id` | identity token | 例如 `dsh-session-events` | +| `goal_id`、`session_id` | identity token | `^[A-Za-z0-9][A-Za-z0-9_.:-]{0,120}$` | +| `agent_id` | identity token,可选 | | +| `sequence` | 整数 >= 0 | observer 分配、每会话单调;缺口计为丢失 | +| `observed_at` | 带时区的 ISO-8601 | | +| `clock.source` | 枚举 | `harness_event_time`、`observer_wall_clock`、`fixture` | +| `clock.uncertainty_ms` | 整数 >= 0 | 显式声明,绝不推断 | +| `event_kind` | 枚举 | `session_started`、`turn_started`、`turn_ended`、`step_started`、`step_ended`、`user_message`、`tool_called`、`tool_completed`、`agent_status`、`agent_pre_step`、`agent_error`、`session_disposed`、`unsupported` | +| `summary` | 对象 | 仅允许 `turn`、`step`(整数)与 `reason`、`status`、`tool_name`、`error_class`、`source_event_type`、`message_source_kind`(紧凑 token) | +| `source_refs` | 对象 | 仅允许 id 键:`event_id`、`event_seq`、`tool_call_id`、`message_id`、`outcome_id`、`gate_id`、`approval_id`、`artifact_id`、`run_id`、`ref_id` | + +其它任何字段都会被拒绝并带上类型化原因:`control_field_rejected`(`command`、`send`、 +`prompt`、`schedule`、`retry`、`stop`、`resume`、`gate`、`tool_call`、`worker_state` 等)、 +`raw_material_field_rejected`(`transcript`、`messages`、`content`、`text`、`arguments`、 +`output`、`stdout`、`stderr`、`log`、`cwd`、`token` 等)或 `unsupported_field_rejected`。 +值还要通过共享的 public-safe 检查,绝对本地路径与凭据样式 token 会 fail closed。 + +### Observer stats(`reliability_observer_stats_v0`) + +每个 observer 实现都会把它写在 envelope 旁边。字段:`observer_id`、`emitted_at`、 +`observed_event_count`、`accepted_event_count`、`rejected_event_count`、 +`rejected_by_reason`、`buffer_bound`、`backpressure_drop_count`、`observer_failure_count`、 +`outbound_endpoints`(必须为 `[]`)、`observation_entered_worker_context`(必须为 `false`)、 +`clock_source`。stats 按 observer 实例累计;receipt 取每个 `observer_id` 的最新记录并跨实例求和。 + +### Integrity receipt(`reliability_integrity_receipt_v0`) + +| 字段 | 含义 | +| --- | --- | +| `status` | `valid`、`degraded`、`quarantined`、`invalid`(全覆盖、有序) | +| `reason_codes` | 类型化列表;仅 `valid` 时为空 | +| `observed_event_count`、`session_count` | ledger 中被接受的 envelope | +| `lost_event_count`、`duplicate_sequence_count` | 每会话 sequence 缺口与重复 | +| `ledger_invalid_record_count` | ledger 中损坏或异类记录 | +| `rejected_event_count`、`rejected_by_reason` | observer 报告的拒绝 | +| `buffer_bound`、`backpressure_drop_count`、`observer_failure_count` | 有界失败证据 | +| `clock.sources`、`clock.max_uncertainty_ms` | 声明的时钟;> 1000 ms 时降级 | +| `outbound_endpoints`、`observation_entered_worker_context` | 必须为 `[]` / `false` | +| `event_kinds_consumed`、`summary_fields_consumed` | 实际消费的事件源与字段 | + +状态规则:无观测、任一 outbound endpoint、或观测进入 worker context 时为 `invalid`; +否则 observer 故障、出现控制字段记录、或 ledger 含损坏记录时为 `quarantined`;否则 +事件丢失、被丢弃、重复、拒绝了原始材料、缺少 stats、或时钟不确定度超阈值时为 +`degraded`;否则为 `valid`。 + +### Diagnostic projection(`reliability_diagnostic_projection_v0`) + +| 字段 | 含义 | +| --- | --- | +| `mode`、`authority`、`write_scope`、`worker_influence` | `read_only`、`none`、`diagnostic_ledger_only`、`none` | +| `stage` | 由最后一个事件种类得出:`unknown`、`idle`、`running`、`tool_running`、`errored`、`disposed` | +| `counts` | turn 开始/结束、step、tool 调用、错误 | +| `stall` | 仅在活跃且相对 `--as-of` 静默达 `threshold_ms`(默认 300000)时判定 | +| `repetition` | 连续相同 `tool_name` 的最长 run;达 3 判定 | +| `recovery` | 错误之后出现完成的 step 或非错误的 turn end 计为已恢复 | +| `signals` | `stall_suspected`、`repetition_suspected`、`unrecovered_error`、`event_loss`、`integrity_not_valid` | +| `integrity` | receipt 的 status 与 reason codes | + +## 使用方式 + +```bash +# 只为一个 goal 启用 DSH provider,然后照常启动 DSH。 +export LOOPX_DSH_SHADOW_OBSERVER_GOAL_ID= +# 可选:LOOPX_DSH_SHADOW_OBSERVER_LEDGER_DIR、LOOPX_DSH_SHADOW_OBSERVER_BUFFER_BOUND + +loopx reliability-diagnostics receipt --goal-id --format json +loopx reliability-diagnostics status --goal-id --format json +loopx reliability-diagnostics ingest --goal-id --input observer.ndjson --format json +``` + +ledger 位于 `/reliability_diagnostics/.ndjson`;默认 runtime root +与 LoopX 其它部分一致,CLI 只打印相对的 `ledger_ref`。`ingest` 会重新校验每一行;干净的 +ingest 是透明拷贝,只有当 ingest 门拒绝、丢弃或失败了某些记录时才会写入自己的 stats 记录。 + +未设置环境变量时,observer 不注册任何 hook、不写任何文件(feature-off parity)。 +设置后,`observer.ts` 观察 `agent/session-start`、`agent/status`、`agent/error`、 +`agent/pre-step`(透传)、`session/event`、`session/disposed`;token 级的 +`assistant/chunk` 不被消费,receipt 通过 `event_kinds_consumed` 让这一点可见。 + +## 验证 + +```bash +python3 examples/reliability_diagnostics/dsh-shadow-observer-fixture-smoke.py +python3 -m pytest tests/capabilities/test_reliability_diagnostics.py tests/capabilities/test_reliability_diagnostics_dsh_provider.py -q +cd packages/dsh-loopx-plugin && pnpm typecheck && pnpm test -- observer +``` + +fixture 是一条固定的 DSH 形态事件流:缺一个 sequence、一个事件带 1500 ms 时钟不确定度、 +一条带原始材料的记录、以及一段撑爆 20 条缓冲的突发。其 receipt 为 `degraded`,原因恰为 +`sequence_gap`、`backpressure_drop`、`raw_material_rejected`、`clock_uncertainty_exceeded`; +projection 报告 `read` 上的重复、一次已恢复的错误、无 stall。 + +## 本切片的非目标 + +不做 dashboard、不做 L2 建议、不做自动恢复、不回写 goal / todo / gate / session runtime、 +不改 `loopx status` 首屏。observer 把 DSH 进程内所有会话归属到唯一声明的 goal;按会话的 +绑定发现是后续工作,且不得复用 driver 的 LoopX CLI 通路。 diff --git a/loopx/capabilities/reliability_diagnostics/catalog_entry.py b/loopx/capabilities/reliability_diagnostics/catalog_entry.py new file mode 100644 index 0000000000..150339744a --- /dev/null +++ b/loopx/capabilities/reliability_diagnostics/catalog_entry.py @@ -0,0 +1,98 @@ +"""Capability-owned catalog entry for Reliability Diagnostics (L1 shadow observer).""" + +from __future__ import annotations + +from typing import Any + +from .envelope import ( + CAPABILITY_ID, + DSH_PROVIDER_ID, + OBSERVER_ENVELOPE_SCHEMA_VERSION, + OBSERVER_STATS_SCHEMA_VERSION, +) +from .projection import DIAGNOSTIC_PROJECTION_SCHEMA_VERSION +from .receipt import INTEGRITY_RECEIPT_SCHEMA_VERSION + +_README = "loopx/capabilities/reliability_diagnostics/README.md" + +RELIABILITY_DIAGNOSTICS_CATALOG_ENTRY: dict[str, Any] = { + "id": CAPABILITY_ID, + "origin": "builtin", + "visibility": "public", + "provider_id": "loopx-core", + "documentation": { + "source_root": "loopx/capabilities/reliability_diagnostics", + "site_root": "capabilities/reliability-diagnostics", + "canonical": "README.md", + }, + "title": "Passive reliability diagnostics from one-way harness events", + "status": "experimental", + "default_enabled": False, + "real_world_anchor": ( + "long-running agent sessions whose stalls, repetition, and recovery must " + "be observed without changing the worker" + ), + "user_value": ( + "Turn read-only harness events into an independent diagnostic ledger, a " + "treatment-integrity receipt, and a compact stall/repetition/recovery " + "projection that carries no runtime authority." + ), + "entry_command": "loopx reliability-diagnostics status --goal-id --format json", + "commands": [ + { + "command": ( + "loopx reliability-diagnostics ingest --goal-id " + "--input --format json" + ), + "purpose": "Validate observer envelopes and append accepted ones to the goal's diagnostic ledger.", + "write_boundary": ( + "appends only to the LoopX-owned diagnostic ledger; rejects control-shaped " + "and raw-material-bearing records; never touches goal, todo, gate, or session state" + ), + }, + { + "command": "loopx reliability-diagnostics receipt --goal-id --format json", + "purpose": "Render the treatment-integrity receipt (loss, drops, clock, endpoints, status).", + "write_boundary": "read-only ledger reduction; no state write", + }, + { + "command": "loopx reliability-diagnostics status --goal-id --format json", + "purpose": "Render the compact read-only stage/stall/repetition/recovery projection.", + "write_boundary": "read-only ledger reduction; `mode=read_only`, `authority=none`", + }, + ], + "implemented_protocols": [ + {"schema_version": schema_version, "module": module, "doc": _README} + for schema_version, module in ( + (OBSERVER_ENVELOPE_SCHEMA_VERSION, "loopx.capabilities.reliability_diagnostics.envelope"), + (OBSERVER_STATS_SCHEMA_VERSION, "loopx.capabilities.reliability_diagnostics.intake"), + (INTEGRITY_RECEIPT_SCHEMA_VERSION, "loopx.capabilities.reliability_diagnostics.receipt"), + (DIAGNOSTIC_PROJECTION_SCHEMA_VERSION, "loopx.capabilities.reliability_diagnostics.projection"), + ) + ], + "implementation_providers": [ + { + "provider_id": DSH_PROVIDER_ID, + "origin": "extension", + "protocol": OBSERVER_ENVELOPE_SCHEMA_VERSION, + "package": "packages/dsh-loopx-plugin", + "status": "declared_default_off", + } + ], + "smokes": [ + "python3 examples/reliability_diagnostics/dsh-shadow-observer-fixture-smoke.py", + "python3 -m pytest tests/capabilities/test_reliability_diagnostics.py -q", + ], + "docs": [_README, "loopx/capabilities/reliability_diagnostics/README.zh-CN.md"], + "boundaries": [ + "L1 only: the observer consumes read-only harness events and owns no send, schedule, retry, stop, resume, gate, tool, or worker-state path.", + "Envelopes are a strict allowlist; control-shaped and raw-material-shaped fields are rejected and counted, never persisted.", + "The diagnostic ledger and projection are siblings of session-runtime state and are never merged into goal, todo, gate, or quota authority.", + "Observer failure is isolated and counted; it marks the receipt quarantined and can never pause or fail the worker.", + "The DSH provider is default off and declared only; enabling it requires an explicit per-goal opt-in in the plugin environment.", + ], + "next_real_step": ( + "Enable the DSH provider for one goal, ingest its ledger, and compare the " + "receipt against a native baseline run to report observer overhead." + ), +} diff --git a/packages/dsh-loopx-plugin/src/driver.ts b/packages/dsh-loopx-plugin/src/driver.ts index 4df4051341..44ed917379 100644 --- a/packages/dsh-loopx-plugin/src/driver.ts +++ b/packages/dsh-loopx-plugin/src/driver.ts @@ -13,6 +13,7 @@ import { } from './cli.ts' import type { FileRunner, LoopXCommand } from './cli.ts' import { resolvePluginLoopXCommand } from './managed-runtime.ts' +import { applyObserver, resolveShadowObserverConfig } from './observer.ts' import { goalBarCoordinator } from './goalbar/events.ts' import type { GoalBarDriverActionReceipt, @@ -1159,6 +1160,11 @@ export class LoopXContinuationDriver { } export function apply(ctx: Context): void { + // L1 shadow observer: separate module, separate effect, no shared send path. + // Feature-off parity: with no declared goal nothing below registers a hook. + const observerConfig = resolveShadowObserverConfig() + if (observerConfig !== undefined) applyObserver(ctx, observerConfig) + const driver = new LoopXContinuationDriver({ isLiveAgent: agent => ctx.agents.get(agent.id) === agent, resolveCommand: signal => resolvePluginLoopXCommand({ signal }), diff --git a/packages/dsh-loopx-plugin/src/index.ts b/packages/dsh-loopx-plugin/src/index.ts index 2378bc9426..9ee114480c 100644 --- a/packages/dsh-loopx-plugin/src/index.ts +++ b/packages/dsh-loopx-plugin/src/index.ts @@ -70,4 +70,14 @@ export type { LoopXInitSummary, } from './init-command.ts' export { resolvePluginLoopXCommand } from './managed-runtime.ts' +export { + applyObserver, + resolveShadowObserverConfig, + ShadowObserver, +} from './observer.ts' +export type { + ObserverEnvelope, + ObserverStats, + ShadowObserverConfig, +} from './observer.ts' export type { LoopXRuntimeOptions } from './managed-runtime.ts' diff --git a/packages/dsh-loopx-plugin/src/observer.ts b/packages/dsh-loopx-plugin/src/observer.ts new file mode 100644 index 0000000000..7285c776b4 --- /dev/null +++ b/packages/dsh-loopx-plugin/src/observer.ts @@ -0,0 +1,448 @@ +/** + * L1 shadow observer for DeepSeek Harness (`dsh-session-events` provider). + * + * One-way only: this module consumes read-only harness events and appends + * `reliability_observer_envelope_v0` records plus a + * `reliability_observer_stats_v0` record to the LoopX reliability-diagnostics + * ledger. It imports nothing from `driver.ts`, owns no `agent.send`, inbox, + * timer, LoopX CLI, or continuation path, and every hook body is isolated so a + * failure is counted instead of propagating into the harness. Field names + * mirror `loopx/capabilities/reliability_diagnostics/envelope.py` exactly. + */ + +import { randomUUID } from 'node:crypto' +import { appendFile, mkdir } from 'node:fs/promises' +import { homedir } from 'node:os' +import { dirname, join, resolve } from 'node:path' +import type { Context } from '@deepseek-ai/cordis' +import type { Agent, AgentStatus } from '@deepseek-ai/dsh-agent' +import type { Session, SessionEvent } from '@deepseek-ai/dsh-session' + +export const CAPABILITY_ID = 'reliability-diagnostics' +export const PROVIDER_ID = 'dsh-session-events' +export const OBSERVER_ENVELOPE_SCHEMA_VERSION = 'reliability_observer_envelope_v0' +export const OBSERVER_STATS_SCHEMA_VERSION = 'reliability_observer_stats_v0' +export const LEDGER_DIRNAME = 'reliability_diagnostics' +export const DEFAULT_BUFFER_BOUND = 256 +export const MAX_BUFFER_BOUND = 65_536 +/** Declared skew for events stamped by the observer instead of the harness log. */ +export const WALL_CLOCK_UNCERTAINTY_MS = 50 + +export const ENV_GOAL_ID = 'LOOPX_DSH_SHADOW_OBSERVER_GOAL_ID' +export const ENV_LEDGER_DIR = 'LOOPX_DSH_SHADOW_OBSERVER_LEDGER_DIR' +export const ENV_BUFFER_BOUND = 'LOOPX_DSH_SHADOW_OBSERVER_BUFFER_BOUND' + +const IDENTITY_TOKEN = /^[A-Za-z0-9][A-Za-z0-9_.:-]{0,120}$/u +const SUMMARY_TOKEN = /^[A-Za-z0-9][A-Za-z0-9_./:-]{0,79}$/u + +export type ObserverEventKind = + | 'session_started' + | 'turn_started' + | 'turn_ended' + | 'step_started' + | 'step_ended' + | 'user_message' + | 'tool_called' + | 'tool_completed' + | 'agent_status' + | 'agent_pre_step' + | 'agent_error' + | 'session_disposed' + | 'unsupported' + +export type ClockSource = 'harness_event_time' | 'observer_wall_clock' | 'fixture' + +export interface ObserverEnvelope { + readonly schema_version: typeof OBSERVER_ENVELOPE_SCHEMA_VERSION + readonly capability_id: typeof CAPABILITY_ID + readonly provider_id: typeof PROVIDER_ID + readonly goal_id: string + readonly session_id: string + readonly agent_id?: string + readonly sequence: number + readonly observed_at: string + readonly clock: { readonly source: ClockSource, readonly uncertainty_ms: number } + readonly event_kind: ObserverEventKind + readonly summary: Readonly> + readonly source_refs: Readonly> +} + +export interface ObserverStats { + readonly schema_version: typeof OBSERVER_STATS_SCHEMA_VERSION + readonly capability_id: typeof CAPABILITY_ID + readonly provider_id: typeof PROVIDER_ID + readonly observer_id: string + readonly goal_id: string + readonly emitted_at: string + readonly observed_event_count: number + readonly accepted_event_count: number + readonly rejected_event_count: number + readonly rejected_by_reason: Readonly> + readonly buffer_bound: number + readonly backpressure_drop_count: number + readonly observer_failure_count: number + /** Always empty: the observer has no outbound control path to declare. */ + readonly outbound_endpoints: readonly [] + readonly observation_entered_worker_context: false + readonly clock_source: ClockSource +} + +export interface ShadowObserverConfig { + readonly goalId: string + readonly ledgerDir: string + readonly bufferBound: number +} + +export type LedgerAppender = (path: string, lines: readonly string[]) => Promise + +export interface ShadowObserverOptions { + readonly config: ShadowObserverConfig + readonly now?: (() => number) | undefined + readonly appendLines?: LedgerAppender | undefined + readonly warn?: ((message: string) => void) | undefined + readonly observerId?: string | undefined +} + +export function defaultLedgerDir(env: NodeJS.ProcessEnv = process.env): string { + const configured = env[ENV_LEDGER_DIR] + return resolve(configured?.trim() + ? configured + : join(homedir(), '.codex', 'loopx', LEDGER_DIRNAME)) +} + +/** + * The observer is OFF unless one exact goal id is declared. Returning + * `undefined` is the feature-off path: no hooks, no files. + */ +export function resolveShadowObserverConfig( + env: NodeJS.ProcessEnv = process.env, +): ShadowObserverConfig | undefined { + const goalId = env[ENV_GOAL_ID]?.trim() + if (!goalId || !IDENTITY_TOKEN.test(goalId)) return undefined + const rawBound = Number.parseInt(env[ENV_BUFFER_BOUND] ?? '', 10) + const bufferBound = Number.isInteger(rawBound) && rawBound >= 1 && rawBound <= MAX_BUFFER_BOUND + ? rawBound + : DEFAULT_BUFFER_BOUND + return { goalId, ledgerDir: defaultLedgerDir(env), bufferBound } +} + +export function ledgerPath(config: ShadowObserverConfig): string { + return join(config.ledgerDir, `${config.goalId.replaceAll(':', '_')}.ndjson`) +} + +async function appendLedgerLines(path: string, lines: readonly string[]): Promise { + if (lines.length === 0) return + await mkdir(dirname(path), { recursive: true }) + await appendFile(path, `${lines.join('\n')}\n`, 'utf8') +} + +function token(value: unknown): string | undefined { + const text = String(value ?? '') + return SUMMARY_TOKEN.test(text) ? text : undefined +} + +function identity(value: unknown): string | undefined { + const text = String(value ?? '') + return IDENTITY_TOKEN.test(text) ? text : undefined +} + +function count(value: unknown): number | undefined { + return typeof value === 'number' && Number.isInteger(value) && value >= 0 ? value : undefined +} + +interface CompactEvent { + readonly kind: ObserverEventKind + readonly summary: Record + readonly sourceRefs: Record +} + +function compactSessionEvent(event: SessionEvent): CompactEvent | undefined { + const data = event.data as Record + const summary: Record = {} + const sourceRefs: Record = { event_seq: String(event.seq) } + const put = (key: string, value: number | string | undefined): void => { + if (value !== undefined) summary[key] = value + } + const ref = (key: string, value: string | undefined): void => { + if (value !== undefined) sourceRefs[key] = value + } + put('turn', count(data.turn)) + put('step', count(data.step)) + switch (event.type) { + case 'turn/start': + return { kind: 'turn_started', summary, sourceRefs } + case 'turn/end': + put('reason', token(data.reason)) + return { kind: 'turn_ended', summary, sourceRefs } + case 'step/start': + return { kind: 'step_started', summary, sourceRefs } + case 'step/end': + return { kind: 'step_ended', summary, sourceRefs } + case 'user/message': { + const source = data.source as Record | undefined + put('message_source_kind', token(source?.kind)) + ref('message_id', identity(data.id)) + return { kind: 'user_message', summary, sourceRefs } + } + case 'tool/call': + put('tool_name', token(data.name)) + ref('tool_call_id', identity(data.callId)) + return { kind: 'tool_called', summary, sourceRefs } + case 'tool/result': { + const error = data.error as Record | undefined + const message = data.message as Record | undefined + const source = message?.source as Record | undefined + put('status', error === undefined ? 'ok' : 'error') + put('error_class', error === undefined ? undefined : token(error.code)) + ref('tool_call_id', identity(source?.callId)) + return { kind: 'tool_completed', summary, sourceRefs } + } + case 'assistant/chunk': + // Token-level chunks are not consumed: they carry model text and add no + // stage signal. Their absence is visible through `event_kinds_consumed`. + return undefined + default: + put('source_event_type', token(event.type)) + return { kind: 'unsupported', summary, sourceRefs } + } +} + +/** + * Bounded, crash-isolated observer. Mirrors + * `ShadowObserverIntake` on the Python side: overflow is counted, never + * blocking; failures are counted, never thrown; the stats record travels with + * every flush. + */ +export class ShadowObserver { + private readonly config: ShadowObserverConfig + private readonly now: () => number + private readonly appendLines: LedgerAppender + private readonly warn: (message: string) => void + private readonly observerId: string + private readonly sequences = new Map() + private buffer: ObserverEnvelope[] = [] + private flushing: Promise | undefined + private flushRequested = false + private disposed = false + private observedEventCount = 0 + private acceptedEventCount = 0 + private backpressureDropCount = 0 + private observerFailureCount = 0 + + constructor(options: ShadowObserverOptions) { + if (!IDENTITY_TOKEN.test(options.config.goalId)) throw new Error('goal id must be an identity token') + if (!Number.isInteger(options.config.bufferBound) + || options.config.bufferBound < 1 + || options.config.bufferBound > MAX_BUFFER_BOUND) { + throw new Error(`buffer bound must be within 1..${MAX_BUFFER_BOUND}`) + } + this.config = options.config + this.now = options.now ?? Date.now + this.appendLines = options.appendLines ?? appendLedgerLines + this.warn = options.warn ?? (() => {}) + this.observerId = options.observerId ?? `${PROVIDER_ID}-${randomUUID()}` + } + + get path(): string { + return ledgerPath(this.config) + } + + observeSessionStart(agent: Agent): void { + this.isolated(() => this.record('session_started', agent.session, agent, undefined, {}, {})) + } + + observeAgentStatus(agent: Agent, status: AgentStatus): void { + this.isolated(() => { + const summary: Record = {} + const compact = token(status) + if (compact !== undefined) summary.status = compact + this.record('agent_status', agent.session, agent, undefined, summary, {}) + if (status === 'idle') this.requestFlush() + }) + } + + observeAgentError(agent: Agent, detail: { readonly turn?: number, readonly step?: number, readonly error?: unknown }): void { + this.isolated(() => { + const summary: Record = {} + const turn = count(detail.turn) + const step = count(detail.step) + if (turn !== undefined) summary.turn = turn + if (step !== undefined) summary.step = step + const error = detail.error + const errorClass = error instanceof Error ? token(error.name) : undefined + if (errorClass !== undefined) summary.error_class = errorClass + this.record('agent_error', agent.session, agent, undefined, summary, {}) + this.requestFlush() + }) + } + + observePreStep(agent: Agent, detail: { readonly turn?: number, readonly step?: number }): void { + this.isolated(() => { + const summary: Record = {} + const turn = count(detail.turn) + const step = count(detail.step) + if (turn !== undefined) summary.turn = turn + if (step !== undefined) summary.step = step + this.record('agent_pre_step', agent.session, agent, undefined, summary, {}) + }) + } + + observeSessionEvent(session: Session, event: SessionEvent): void { + this.isolated(() => { + const compact = compactSessionEvent(event) + if (compact === undefined) return + const time = typeof event.time === 'number' && Number.isFinite(event.time) ? event.time : undefined + this.record(compact.kind, session, undefined, time, compact.summary, compact.sourceRefs) + if (event.type === 'turn/end') this.requestFlush() + }) + } + + observeSessionDisposed(session: Session): void { + this.isolated(() => { + this.record('session_disposed', session, undefined, undefined, {}, {}) + this.sequences.delete(String(session.id)) + this.requestFlush() + }) + } + + stats(): ObserverStats { + return { + schema_version: OBSERVER_STATS_SCHEMA_VERSION, + capability_id: CAPABILITY_ID, + provider_id: PROVIDER_ID, + observer_id: this.observerId, + goal_id: this.config.goalId, + emitted_at: new Date(this.now()).toISOString(), + observed_event_count: this.observedEventCount, + accepted_event_count: this.acceptedEventCount, + rejected_event_count: 0, + rejected_by_reason: {}, + buffer_bound: this.config.bufferBound, + backpressure_drop_count: this.backpressureDropCount, + observer_failure_count: this.observerFailureCount, + outbound_endpoints: [], + observation_entered_worker_context: false, + clock_source: 'harness_event_time', + } + } + + /** Write buffered envelopes plus a stats record; never rejects. */ + async flush(): Promise { + if (this.flushing !== undefined) { + this.flushRequested = true + await this.flushing + return + } + const taken = this.buffer + this.buffer = [] + const lines = [...taken, this.stats()].map(record => JSON.stringify(record)) + this.flushing = this.appendLines(this.path, lines).then( + () => undefined, + (error: unknown) => { + this.observerFailureCount += 1 + this.backpressureDropCount += taken.length + this.warn(`dsh-loopx shadow observer flush failed: ${error instanceof Error ? error.name : 'unknown'}`) + }, + ) + try { + await this.flushing + } finally { + this.flushing = undefined + } + if (this.flushRequested) { + this.flushRequested = false + await this.flush() + } + } + + async dispose(): Promise { + if (this.disposed) return + this.disposed = true + await this.flush() + } + + private isolated(body: () => void): void { + if (this.disposed) return + try { + body() + } catch (error: unknown) { + this.observerFailureCount += 1 + this.warn(`dsh-loopx shadow observer hook failed: ${error instanceof Error ? error.name : 'unknown'}`) + } + } + + private record( + kind: ObserverEventKind, + session: Session, + agent: Agent | undefined, + harnessTimeMs: number | undefined, + summary: Record, + sourceRefs: Record, + ): void { + this.observedEventCount += 1 + const sessionId = identity(session.id) + if (sessionId === undefined) throw new Error('session id is not an identity token') + const sequence = this.sequences.get(sessionId) ?? 0 + this.sequences.set(sessionId, sequence + 1) + if (this.buffer.length >= this.config.bufferBound) { + this.backpressureDropCount += 1 + this.requestFlush() + return + } + const observedAtMs = harnessTimeMs ?? this.now() + const agentId = agent === undefined ? undefined : identity(agent.id) + const envelope: ObserverEnvelope = { + schema_version: OBSERVER_ENVELOPE_SCHEMA_VERSION, + capability_id: CAPABILITY_ID, + provider_id: PROVIDER_ID, + goal_id: this.config.goalId, + session_id: sessionId, + ...(agentId === undefined ? {} : { agent_id: agentId }), + sequence, + observed_at: new Date(observedAtMs).toISOString(), + clock: harnessTimeMs === undefined + ? { source: 'observer_wall_clock', uncertainty_ms: WALL_CLOCK_UNCERTAINTY_MS } + : { source: 'harness_event_time', uncertainty_ms: 0 }, + event_kind: kind, + summary, + source_refs: sourceRefs, + } + this.buffer.push(envelope) + this.acceptedEventCount += 1 + if (this.buffer.length >= this.config.bufferBound) this.requestFlush() + } + + private requestFlush(): void { + void this.flush() + } +} + +/** + * Register read-only hooks only. Called by the Driver row's `apply()` solely + * when `resolveShadowObserverConfig()` returns a config; when it returns + * `undefined`, nothing here runs and feature-off parity holds. + */ +export function applyObserver(ctx: Context, config: ShadowObserverConfig): ShadowObserver { + const observer = new ShadowObserver({ + config, + warn: message => { ctx.logger.warn(message) }, + }) + ctx.effect(function* () { + ctx.on('agent/session-start', ({ agent }) => { observer.observeSessionStart(agent) }) + ctx.on('agent/status', ({ agent, status }) => { observer.observeAgentStatus(agent, status) }) + ctx.on('agent/error', ({ agent, turn, step, error }) => { + observer.observeAgentError(agent, { turn, step, error }) + }) + // Waterfall hook: observe, then hand the unchanged decision through. + ctx.on('agent/pre-step', ({ agent, turn, step }, next) => { + observer.observePreStep(agent, { turn, step }) + return next() + }) + ctx.on('session/event', (session, event) => { observer.observeSessionEvent(session, event) }) + ctx.on('session/disposed', session => { observer.observeSessionDisposed(session) }) + yield async () => { + await observer.dispose() + } + }, 'dsh-loopx shadow observer lifecycle') + return observer +} diff --git a/packages/dsh-loopx-plugin/tests/observer.spec.ts b/packages/dsh-loopx-plugin/tests/observer.spec.ts new file mode 100644 index 0000000000..a26afdf427 --- /dev/null +++ b/packages/dsh-loopx-plugin/tests/observer.spec.ts @@ -0,0 +1,235 @@ +import { readFileSync } from 'node:fs' +import { dirname, join } from 'node:path' +import { fileURLToPath } from 'node:url' +import { describe, expect, it } from 'vitest' +import type { Context } from '@deepseek-ai/cordis' +import type { Agent } from '@deepseek-ai/dsh-agent' +import type { Session, SessionEvent } from '@deepseek-ai/dsh-session' +import { + applyObserver, + ENV_GOAL_ID, + OBSERVER_ENVELOPE_SCHEMA_VERSION, + OBSERVER_STATS_SCHEMA_VERSION, + resolveShadowObserverConfig, + ShadowObserver, +} from '../src/observer.ts' +import type { + ObserverEnvelope, + ObserverStats, + ShadowObserverConfig, +} from '../src/observer.ts' + +const goalId = 'goal-observer-fixture' +const config: ShadowObserverConfig = { goalId, ledgerDir: '/ledger', bufferBound: 4 } +const ENVELOPE_FIELDS = [ + 'schema_version', 'capability_id', 'provider_id', 'goal_id', 'session_id', + 'sequence', 'observed_at', 'clock', 'event_kind', 'summary', 'source_refs', +] +const STATS_FIELDS = [ + 'schema_version', 'capability_id', 'provider_id', 'observer_id', 'goal_id', 'emitted_at', + 'observed_event_count', 'accepted_event_count', 'rejected_event_count', 'rejected_by_reason', + 'buffer_bound', 'backpressure_drop_count', 'observer_failure_count', 'outbound_endpoints', + 'observation_entered_worker_context', 'clock_source', +] + +function fakeSession(id = 'session-fixture'): Session { + return { id } as unknown as Session +} + +function fakeAgent(session: Session, id = 'agent-fixture'): Agent { + return { id, session, status: 'idle' } as unknown as Agent +} + +function sessionEvent(type: string, seq: number, time: number, data: unknown): SessionEvent { + return { type, seq, time, data } as unknown as SessionEvent +} + +interface Captured { + readonly appended: string[][] + readonly paths: string[] + readonly observer: ShadowObserver +} + +function observerWithCapture(options: { + readonly bufferBound?: number + readonly failAppend?: boolean +} = {}): Captured { + const appended: string[][] = [] + const paths: string[] = [] + const observer = new ShadowObserver({ + config: { ...config, bufferBound: options.bufferBound ?? config.bufferBound }, + now: () => 1_756_728_000_000, + observerId: 'observer-fixture', + appendLines: async (path, lines) => { + if (options.failAppend) throw new Error('disk full') + paths.push(path) + appended.push([...lines]) + }, + }) + return { appended, paths, observer } +} + +function parsed(lines: string[][]): Array { + return lines.flat().map(line => JSON.parse(line) as ObserverEnvelope | ObserverStats) +} + +describe('shadow observer configuration', () => { + it('is off unless one exact goal id is declared', () => { + expect(resolveShadowObserverConfig({})).toBeUndefined() + expect(resolveShadowObserverConfig({ [ENV_GOAL_ID]: ' ' })).toBeUndefined() + expect(resolveShadowObserverConfig({ [ENV_GOAL_ID]: 'not an id' })).toBeUndefined() + const resolved = resolveShadowObserverConfig({ + [ENV_GOAL_ID]: goalId, + LOOPX_DSH_SHADOW_OBSERVER_LEDGER_DIR: '/tmp/ledger', + LOOPX_DSH_SHADOW_OBSERVER_BUFFER_BOUND: '9', + }) + expect(resolved).toEqual({ goalId, ledgerDir: '/tmp/ledger', bufferBound: 9 }) + }) + + it('imports nothing from the driver and owns no send path', () => { + const here = dirname(fileURLToPath(import.meta.url)) + const source = readFileSync(join(here, '../src/observer.ts'), 'utf8') + expect(source).not.toMatch(/from '\.\/driver/u) + expect(source).not.toMatch(/from '\.\/cli/u) + expect(source).not.toMatch(/from '\.\/managed-runtime/u) + expect(source).not.toMatch(/\.send\(/u) + expect(source).not.toMatch(/\.inbox\b/u) + expect(source).not.toMatch(/setTimeout|setInterval/u) + }) +}) + +describe('shadow observer envelopes', () => { + it('maps DSH session events into the shared envelope shape', async () => { + const { appended, paths, observer } = observerWithCapture({ bufferBound: 16 }) + const session = fakeSession() + const agent = fakeAgent(session) + observer.observeSessionStart(agent) + observer.observeSessionEvent(session, sessionEvent('turn/start', 1, 1_756_728_001_000, { turn: 1 })) + observer.observeSessionEvent(session, sessionEvent('tool/call', 2, 1_756_728_002_000, { + turn: 1, step: 1, callId: 'call-1', name: 'bash', arguments: '{"cmd":"rm -rf /"}', + })) + observer.observeSessionEvent(session, sessionEvent('tool/result', 3, 1_756_728_003_000, { + turn: 1, step: 1, message: { id: 'm', role: 'user', content: [], source: { kind: 'tool', callId: 'call-1' } }, + error: { name: 'ToolError', code: 'timeout' }, + })) + observer.observeSessionEvent(session, sessionEvent('assistant/chunk', 4, 1_756_728_003_500, { turn: 1, step: 1 })) + observer.observeSessionEvent(session, sessionEvent('todo/write', 5, 1_756_728_004_000, { todos: [] })) + observer.observeSessionEvent(session, sessionEvent('turn/end', 6, 1_756_728_005_000, { turn: 1, reason: 'completed' })) + await observer.flush() + + const records = parsed(appended) + const envelopes = records.filter( + (record): record is ObserverEnvelope => record.schema_version === OBSERVER_ENVELOPE_SCHEMA_VERSION, + ) + expect(new Set(paths)).toEqual(new Set(['/ledger/goal-observer-fixture.ndjson'])) + expect(envelopes.map(item => item.event_kind)).toEqual([ + 'session_started', 'turn_started', 'tool_called', 'tool_completed', 'unsupported', 'turn_ended', + ]) + expect(envelopes.map(item => item.sequence)).toEqual([0, 1, 2, 3, 4, 5]) + for (const envelope of envelopes) { + expect(Object.keys(envelope).filter(key => key !== 'agent_id').sort()).toEqual([...ENVELOPE_FIELDS].sort()) + expect(envelope.goal_id).toBe(goalId) + expect(envelope.session_id).toBe('session-fixture') + } + expect(envelopes[0]?.clock).toEqual({ source: 'observer_wall_clock', uncertainty_ms: 50 }) + expect(envelopes[0]?.agent_id).toBe('agent-fixture') + expect(envelopes[1]?.clock).toEqual({ source: 'harness_event_time', uncertainty_ms: 0 }) + expect(envelopes[1]?.observed_at).toBe('2025-09-01T12:00:01.000Z') + expect(envelopes[2]?.summary).toEqual({ turn: 1, step: 1, tool_name: 'bash' }) + expect(envelopes[2]?.source_refs).toEqual({ event_seq: '2', tool_call_id: 'call-1' }) + expect(envelopes[3]?.summary).toEqual({ turn: 1, step: 1, status: 'error', error_class: 'timeout' }) + expect(envelopes[4]?.summary).toEqual({ source_event_type: 'todo/write' }) + expect(JSON.stringify(records)).not.toContain('rm -rf') + + const stats = records.at(-1) as ObserverStats + expect(stats.schema_version).toBe(OBSERVER_STATS_SCHEMA_VERSION) + expect(Object.keys(stats).sort()).toEqual([...STATS_FIELDS].sort()) + expect(stats.outbound_endpoints).toEqual([]) + expect(stats.observation_entered_worker_context).toBe(false) + expect(stats.observed_event_count).toBe(6) + expect(stats.accepted_event_count).toBe(6) + }) + + it('drops with a count while a flush is in flight and the buffer is full', async () => { + const appended: string[][] = [] + let release: (() => void) | undefined + const observer = new ShadowObserver({ + config: { ...config, bufferBound: 2 }, + now: () => 1_756_728_000_000, + observerId: 'observer-fixture', + appendLines: async (_path, lines) => { + appended.push([...lines]) + if (release === undefined) await new Promise(resolve => { release = resolve }) + }, + }) + const agent = fakeAgent(fakeSession()) + // Two observations fill the buffer and start a flush that stays pending. + observer.observePreStep(agent, { turn: 1, step: 1 }) + observer.observePreStep(agent, { turn: 1, step: 2 }) + // Two more refill the bound; the fifth has nowhere to go and is dropped. + observer.observePreStep(agent, { turn: 1, step: 3 }) + observer.observePreStep(agent, { turn: 1, step: 4 }) + observer.observePreStep(agent, { turn: 1, step: 5 }) + release?.() + await observer.flush() + const records = parsed(appended) + const stats = records.at(-1) as ObserverStats + expect(stats.buffer_bound).toBe(2) + expect(stats.accepted_event_count).toBe(4) + expect(stats.backpressure_drop_count).toBe(1) + expect(stats.observed_event_count).toBe(5) + // The sequence still advances for the dropped event so the loss is visible. + const envelopes = records.filter( + (record): record is ObserverEnvelope => record.schema_version === OBSERVER_ENVELOPE_SCHEMA_VERSION, + ) + expect(envelopes.map(item => item.sequence)).toEqual([0, 1, 2, 3]) + }) + + it('counts hook and flush failures instead of throwing', async () => { + const { observer } = observerWithCapture({ failAppend: true }) + const session = fakeSession() + expect(() => observer.observeSessionEvent(session, undefined as unknown as SessionEvent)).not.toThrow() + observer.observeSessionStart(fakeAgent(session)) + await expect(observer.flush()).resolves.toBeUndefined() + const stats = observer.stats() + expect(stats.observer_failure_count).toBe(2) + expect(stats.backpressure_drop_count).toBe(1) + }) +}) + +describe('applyObserver', () => { + it('registers read-only hooks and passes pre-step decisions through unchanged', async () => { + type Handler = (...args: unknown[]) => unknown + const handlers = new Map() + let disposeEffect: (() => unknown) | undefined + const warnings: string[] = [] + const ctx = { + logger: { warn(message: string) { warnings.push(message) } }, + on(event: string, handler: Handler) { + handlers.set(event, [...(handlers.get(event) ?? []), handler]) + }, + effect(effect: () => Generator) { + const yielded = effect().next().value + if (typeof yielded === 'function') disposeEffect = yielded as () => unknown + }, + } as unknown as Context + const observer = applyObserver(ctx, config) + expect([...handlers.keys()].sort()).toEqual([ + 'agent/error', 'agent/pre-step', 'agent/session-start', 'agent/status', 'session/disposed', 'session/event', + ]) + const session = fakeSession() + const agent = fakeAgent(session) + const decision = { kind: 'enter', messages: [] } + let nextCalls = 0 + const preStep = handlers.get('agent/pre-step')?.[0] + const result = await preStep?.({ agent, messages: [], turn: 1, step: 1, signal: new AbortController().signal }, async () => { + nextCalls += 1 + return decision + }) + expect(nextCalls).toBe(1) + expect(result).toBe(decision) + expect(observer.stats().observed_event_count).toBe(1) + expect(warnings).toEqual([]) + expect(typeof disposeEffect).toBe('function') + }) +}) diff --git a/tests/capabilities/test_capability_extension_registry.py b/tests/capabilities/test_capability_extension_registry.py index bf00f8795d..61fd054666 100644 --- a/tests/capabilities/test_capability_extension_registry.py +++ b/tests/capabilities/test_capability_extension_registry.py @@ -42,6 +42,7 @@ "deep-research", "public-safe-outbound", "connector-registry", + "reliability-diagnostics", ] @@ -156,7 +157,15 @@ def test_builtin_catalog_preserves_order_and_marks_provider() -> None: "installed": True, "enabled": True, "ready": True, - } + }, + { + "id": "dsh-session-events", + "origin": "extension", + "declared": True, + "installed": False, + "enabled": False, + "ready": False, + }, ] diff --git a/tests/capabilities/test_reliability_diagnostics_dsh_provider.py b/tests/capabilities/test_reliability_diagnostics_dsh_provider.py new file mode 100644 index 0000000000..592e528f46 --- /dev/null +++ b/tests/capabilities/test_reliability_diagnostics_dsh_provider.py @@ -0,0 +1,82 @@ +"""Registration and cross-language parity checks for the DSH observer provider.""" + +from __future__ import annotations + +import re +from pathlib import Path + +from loopx.capabilities.catalog import ( + build_capability_catalog_packet, + build_capability_detail_packet, +) +from loopx.capabilities.reliability_diagnostics import ( + CAPABILITY_ID, + DSH_PROVIDER_ID, + ENVELOPE_FIELDS, + OBSERVER_ENVELOPE_SCHEMA_VERSION, + OBSERVER_STATS_SCHEMA_VERSION, + ObserverEventKind, +) +from loopx.capabilities.reliability_diagnostics.intake import STATS_FIELDS + +ROOT = Path(__file__).resolve().parents[2] +OBSERVER_TS = ROOT / "packages/dsh-loopx-plugin/src/observer.ts" +DRIVER_TS = ROOT / "packages/dsh-loopx-plugin/src/driver.ts" + + +def test_catalog_declares_dsh_provider_without_claiming_readiness() -> None: + packet = build_capability_catalog_packet() + summary = next(item for item in packet["capabilities"] if item["id"] == CAPABILITY_ID) + assert summary["provider_id"] == "loopx-core" + assert summary["implementation_provider_count"] == 1 + provider = next(item for item in packet["providers"] if item["id"] == DSH_PROVIDER_ID) + assert provider == { + "id": DSH_PROVIDER_ID, + "origin": "extension", + "declared": True, + "installed": False, + "enabled": False, + "ready": False, + } + + detail = build_capability_detail_packet(CAPABILITY_ID)["capability"] + assert detail["default_enabled"] is False + [implementation] = detail["implementation_providers"] + assert implementation["provider_id"] == DSH_PROVIDER_ID + assert implementation["protocol"] == OBSERVER_ENVELOPE_SCHEMA_VERSION + assert implementation["provider_state"] == { + "declared": True, + "installed": False, + "enabled": False, + "ready": False, + } + + +def test_typescript_observer_shares_field_names_and_has_no_control_path() -> None: + source = OBSERVER_TS.read_text(encoding="utf-8") + assert f"'{OBSERVER_ENVELOPE_SCHEMA_VERSION}'" in source + assert f"'{OBSERVER_STATS_SCHEMA_VERSION}'" in source + assert f"'{CAPABILITY_ID}'" in source + assert f"'{DSH_PROVIDER_ID}'" in source + for field in ENVELOPE_FIELDS | STATS_FIELDS: + assert re.search(rf"\b{field}\b", source), field + for kind in ObserverEventKind: + assert f"'{kind.value}'" in source, kind + + # Physically separate from the continuation driver and its send path. + assert "from './driver" not in source + assert "from './cli" not in source + assert "from './managed-runtime" not in source + assert ".send(" not in source + assert ".inbox" not in source + assert "outbound_endpoints: []" in source + assert "observation_entered_worker_context: false" in source + + +def test_driver_applies_observer_only_when_enabled() -> None: + source = DRIVER_TS.read_text(encoding="utf-8") + assert "resolveShadowObserverConfig()" in source + assert "if (observerConfig !== undefined) applyObserver(ctx, observerConfig)" in source + # The driver must never hand its instance or send path to the observer. + assert "applyObserver(ctx, observerConfig)" in source + assert "applyObserver(ctx, observerConfig, driver" not in source From 828ad8a0db5121105dd3a55ad1a6fb8362403065 Mon Sep 17 00:00:00 2001 From: song Date: Fri, 4 Sep 2026 20:12:06 +0800 Subject: [PATCH 04/13] docs(reliability-diagnostics): index the capability and document the DSH shadow observer Adds the Reliability Diagnostics row to the capability index and site nav and documents the default-off dsh-session-events observer, its opt-in environment, consumed events, and privacy boundary in the plugin README. Signed-off-by: song --- loopx/capabilities/README.md | 1 + mkdocs.yaml | 1 + packages/dsh-loopx-plugin/README.md | 31 +++++++++++++++++++++++++++++ 3 files changed, 33 insertions(+) diff --git a/loopx/capabilities/README.md b/loopx/capabilities/README.md index fcd469ed81..16cf04420b 100644 --- a/loopx/capabilities/README.md +++ b/loopx/capabilities/README.md @@ -74,6 +74,7 @@ availability and maturity in the installed release. | Turn public/private content signals into reviewable source, angle, draft, feedback, and publish-gate packets | [Content Operations](content_ops/README.md) | | Inventory, archive, migrate, and rerank a material store without losing raw source authority | [Material Lifecycle](material_lifecycle/README.md) ([中文](material_lifecycle/README.zh-CN.md)) | | Inspect compatibility routes for public-safe external-value intake while callers migrate to outcome-owned capabilities | [Value Connectors](value_connectors/README.md) | +| Observe a long-running harness session one-way and read back an integrity receipt and stall/repetition/recovery projection with no runtime authority | [Reliability Diagnostics](reliability_diagnostics/README.md) ([中文](reliability_diagnostics/README.zh-CN.md)) | ## Contributor Navigation And Ownership diff --git a/mkdocs.yaml b/mkdocs.yaml index 5b5527165a..7eb55b0e3e 100644 --- a/mkdocs.yaml +++ b/mkdocs.yaml @@ -125,6 +125,7 @@ nav: - Periodic Report: capabilities/periodic-report/README.md - Content Operations: capabilities/content-ops/README.md - Value Connectors: capabilities/value-connectors/README.md + - Reliability Diagnostics: capabilities/reliability-diagnostics/README.md - Integrations: - integrations/README.md - Integration Guide: integration.md diff --git a/packages/dsh-loopx-plugin/README.md b/packages/dsh-loopx-plugin/README.md index 0774b6e78b..77b7347df4 100644 --- a/packages/dsh-loopx-plugin/README.md +++ b/packages/dsh-loopx-plugin/README.md @@ -129,6 +129,37 @@ managed launcher, startup readiness, and first-session `loopx` skill discovery. It requires Docker, `uv`, and network access for base images and never opens a browser or configures a model provider. +## Shadow observer (default off) + +`src/observer.ts` is the `dsh-session-events` provider for the LoopX +[Reliability Diagnostics](../../loopx/capabilities/reliability_diagnostics/README.md) +capability: an L1 shadow observer that consumes read-only harness events and +appends compact, public-safe envelopes plus an observer stats record to +`/reliability_diagnostics/.ndjson`. It is a +separate module from the Driver with no shared send path: it never calls +`agent.send`, touches the inbox, invokes the LoopX CLI, schedules, retries, +stops, or resumes anything. Every hook body and every flush is isolated, so an +observer failure is counted into the receipt instead of reaching DSH. + +It is off unless one exact goal is declared before DSH starts: + +```bash +export LOOPX_DSH_SHADOW_OBSERVER_GOAL_ID= +# optional: LOOPX_DSH_SHADOW_OBSERVER_LEDGER_DIR, LOOPX_DSH_SHADOW_OBSERVER_BUFFER_BOUND (default 256) +loopx reliability-diagnostics receipt --goal-id --format json +loopx reliability-diagnostics status --goal-id --format json +``` + +With the variable unset the Driver row registers no observer hook and writes no +file. When set, the observer consumes `agent/session-start`, `agent/status`, +`agent/error`, `agent/pre-step` (pass-through), `session/event`, and +`session/disposed`; it skips `assistant/chunk` and records tool names, turn and +step numbers, end reasons, and ids only, never arguments, outputs, prompts, or +paths. Sequence gaps, bounded-buffer drops, declared clock uncertainty, and +the always-empty outbound endpoint list make the run's admissibility as passive +evidence auditable from the receipt. In this slice every session in the DSH +process is attributed to the single declared goal. + ## GoalBar authority and privacy boundary `/loopx` is registered with Connection authority `loopback`. Loopback is a From d66d2edd618b76332b68dcf68fdf86ecc9df723a Mon Sep 17 00:00:00 2001 From: song Date: Fri, 4 Sep 2026 20:41:35 +0800 Subject: [PATCH 05/13] fix(reliability-diagnostics): keep catalog Any ceiling and approve packed observer types Review follow-up for the L1 shadow-observer slice. - `loopx/capabilities/catalog.py`: the new extension-provider declaration helper pushed the module's `Any` annotation count from 8 to 9, above the ceiling recorded in `loopx/canary/module_metric_baseline.json`, so the `control-plane-maintainability-ratchet` smoke reported one unreviewed finding. The helper only reads catalog-entry mappings, so type the record as `Mapping[str, object]`; the ratchet is back to zero unreviewed debt. - `packages/dsh-loopx-plugin/smoke/dsh-goalbar-runtime-smoke.mjs`: the packed runtime allowlist did not include the new `lib/types/observer.d.ts` declaration file, so `pnpm smoke:runtime` rejected the tarball. Approve that single entry; no new JS chunk is emitted for the observer. Validation: - python3 examples/control_plane/control-plane-maintainability-ratchet-smoke.py -> unreviewed: 0 - python3 -m pytest tests/capabilities/test_reliability_diagnostics_dsh_provider.py tests/capabilities/test_capability_extension_registry.py -> 23 passed - pnpm smoke:runtime (outside the restricted sandbox) -> passed Signed-off-by: song --- loopx/capabilities/catalog.py | 2 +- packages/dsh-loopx-plugin/smoke/dsh-goalbar-runtime-smoke.mjs | 1 + 2 files changed, 2 insertions(+), 1 deletion(-) diff --git a/loopx/capabilities/catalog.py b/loopx/capabilities/catalog.py index c5f8b641a8..0a71aeba13 100644 --- a/loopx/capabilities/catalog.py +++ b/loopx/capabilities/catalog.py @@ -82,7 +82,7 @@ def _summary(record: Mapping[str, Any]) -> dict[str, Any]: def _register_declared_extension_providers( registry: CapabilityRegistry, - record: Mapping[str, Any], + record: Mapping[str, object], ) -> None: """Declare extension-delivered implementations named by a builtin entry. diff --git a/packages/dsh-loopx-plugin/smoke/dsh-goalbar-runtime-smoke.mjs b/packages/dsh-loopx-plugin/smoke/dsh-goalbar-runtime-smoke.mjs index c257e4ba2d..46ce47030b 100644 --- a/packages/dsh-loopx-plugin/smoke/dsh-goalbar-runtime-smoke.mjs +++ b/packages/dsh-loopx-plugin/smoke/dsh-goalbar-runtime-smoke.mjs @@ -61,6 +61,7 @@ const packedStaticEntries = new Set([ 'package/lib/types/index.d.ts', 'package/lib/types/init-command.d.ts', 'package/lib/types/managed-runtime.d.ts', + 'package/lib/types/observer.d.ts', 'package/package.json', ]) const packedHashedEntries = [ From c170003c772e63809e6c31e3139fead065390304 Mon Sep 17 00:00:00 2001 From: song Date: Sat, 5 Sep 2026 00:18:04 +0800 Subject: [PATCH 06/13] fix(reliability-diagnostics): enforce observer treatment identity Signed-off-by: song --- .../reliability_diagnostics/__init__.py | 6 + .../reliability_diagnostics/envelope.py | 5 + .../reliability_diagnostics/intake.py | 177 +++++++++++-- .../reliability_diagnostics/projection.py | 14 +- .../reliability_diagnostics/receipt.py | 78 +++++- loopx/cli_commands/reliability_diagnostics.py | 94 ++++--- packages/dsh-loopx-plugin/cordis.patch.yml | 5 + packages/dsh-loopx-plugin/package.json | 4 + packages/dsh-loopx-plugin/src/driver.ts | 6 - packages/dsh-loopx-plugin/src/index.ts | 10 - packages/dsh-loopx-plugin/src/observer.ts | 233 ++++++++++++------ packages/dsh-loopx-plugin/tsdown.config.ts | 1 + 12 files changed, 481 insertions(+), 152 deletions(-) diff --git a/loopx/capabilities/reliability_diagnostics/__init__.py b/loopx/capabilities/reliability_diagnostics/__init__.py index d93d900b81..25e0f59f68 100644 --- a/loopx/capabilities/reliability_diagnostics/__init__.py +++ b/loopx/capabilities/reliability_diagnostics/__init__.py @@ -20,8 +20,11 @@ from .fixture import FIXTURE_GOAL_ID, dsh_fixture_records, run_dsh_fixture from .intake import ( DEFAULT_BUFFER_BOUND, + RUN_IDENTITY_FIELDS, + ObserverRunIdentity, ObserverStats, ShadowObserverIntake, + normalize_observer_run_identity, normalize_observer_stats, ) from .ledger import ( @@ -60,6 +63,7 @@ "OBSERVER_ENVELOPE_SCHEMA_VERSION", "OBSERVER_STATS_SCHEMA_VERSION", "RAW_MATERIAL_FIELD_FAMILIES", + "RUN_IDENTITY_FIELDS", "SOURCE_REF_FIELDS", "SUMMARY_FIELDS", "ClockSource", @@ -70,6 +74,7 @@ "ObserverEnvelope", "ObserverEnvelopeError", "ObserverEventKind", + "ObserverRunIdentity", "ObserverStats", "ReceiptReason", "ReceiptStatus", @@ -81,6 +86,7 @@ "ledger_path", "ledger_ref", "normalize_observer_envelope", + "normalize_observer_run_identity", "normalize_observer_stats", "parse_ndjson_lines", "read_ledger", diff --git a/loopx/capabilities/reliability_diagnostics/envelope.py b/loopx/capabilities/reliability_diagnostics/envelope.py index b52829502c..15924cc2c3 100644 --- a/loopx/capabilities/reliability_diagnostics/envelope.py +++ b/loopx/capabilities/reliability_diagnostics/envelope.py @@ -71,6 +71,7 @@ class EnvelopeRejection(StrEnum): SUMMARY_INVALID = "summary_invalid" SOURCE_REF_INVALID = "source_ref_invalid" PUBLIC_SAFETY_VIOLATION = "public_safety_violation" + OBSERVER_INTERNAL_FAILURE = "observer_internal_failure" ENVELOPE_FIELDS = frozenset( @@ -78,6 +79,7 @@ class EnvelopeRejection(StrEnum): "schema_version", "capability_id", "provider_id", + "observer_id", "goal_id", "session_id", "agent_id", @@ -176,6 +178,7 @@ def as_dict(self) -> dict[str, Any]: @dataclass(frozen=True) class ObserverEnvelope: provider_id: str + observer_id: str goal_id: str session_id: str sequence: int @@ -191,6 +194,7 @@ def as_dict(self) -> dict[str, Any]: "schema_version": OBSERVER_ENVELOPE_SCHEMA_VERSION, "capability_id": CAPABILITY_ID, "provider_id": self.provider_id, + "observer_id": self.observer_id, "goal_id": self.goal_id, "session_id": self.session_id, "sequence": self.sequence, @@ -372,6 +376,7 @@ def normalize_observer_envelope(record: Mapping[str, Any]) -> ObserverEnvelope: ) envelope = ObserverEnvelope( provider_id=_identity(record.get("provider_id"), name="provider_id") or "", + observer_id=_identity(record.get("observer_id"), name="observer_id") or "", goal_id=_identity(record.get("goal_id"), name="goal_id") or "", session_id=_identity(record.get("session_id"), name="session_id") or "", agent_id=_identity(record.get("agent_id"), name="agent_id", optional=True), diff --git a/loopx/capabilities/reliability_diagnostics/intake.py b/loopx/capabilities/reliability_diagnostics/intake.py index 3694546017..45b9a1960f 100644 --- a/loopx/capabilities/reliability_diagnostics/intake.py +++ b/loopx/capabilities/reliability_diagnostics/intake.py @@ -14,6 +14,7 @@ from collections import deque from collections.abc import Callable, Mapping from dataclasses import dataclass, field +from datetime import datetime from typing import Any from .envelope import ( @@ -31,6 +32,19 @@ MAX_BUFFER_BOUND = 65_536 _ENDPOINT_PATTERN = re.compile(r"^[A-Za-z0-9][A-Za-z0-9_.:/-]{0,200}$") +RUN_IDENTITY_FIELDS = frozenset( + { + "worker_id", + "model_id", + "task_id", + "environment_id", + "tools_id", + "budget_id", + "adapter_revision", + "observer_revision", + } +) + STATS_FIELDS = frozenset( { "schema_version", @@ -38,6 +52,9 @@ "provider_id", "observer_id", "goal_id", + "run_identity", + "event_sources", + "source_fields_consumed", "emitted_at", "observed_event_count", "accepted_event_count", @@ -46,18 +63,42 @@ "buffer_bound", "backpressure_drop_count", "observer_failure_count", + "peak_buffered_event_count", + "flush_attempt_count", "outbound_endpoints", "observation_entered_worker_context", + "observation_entered_scheduler_inputs", "clock_source", } ) +@dataclass(frozen=True) +class ObserverRunIdentity: + worker_id: str + model_id: str + task_id: str + environment_id: str + tools_id: str + budget_id: str + adapter_revision: str + observer_revision: str + + def as_dict(self) -> dict[str, str]: + return { + field_name: str(getattr(self, field_name)) + for field_name in RUN_IDENTITY_FIELDS + } + + @dataclass(frozen=True) class ObserverStats: provider_id: str observer_id: str goal_id: str + run_identity: ObserverRunIdentity + event_sources: tuple[str, ...] + source_fields_consumed: tuple[str, ...] emitted_at: str observed_event_count: int accepted_event_count: int @@ -66,8 +107,11 @@ class ObserverStats: buffer_bound: int backpressure_drop_count: int observer_failure_count: int + peak_buffered_event_count: int + flush_attempt_count: int outbound_endpoints: tuple[str, ...] observation_entered_worker_context: bool + observation_entered_scheduler_inputs: bool clock_source: ClockSource def as_dict(self) -> dict[str, Any]: @@ -77,6 +121,9 @@ def as_dict(self) -> dict[str, Any]: "provider_id": self.provider_id, "observer_id": self.observer_id, "goal_id": self.goal_id, + "run_identity": self.run_identity.as_dict(), + "event_sources": list(self.event_sources), + "source_fields_consumed": list(self.source_fields_consumed), "emitted_at": self.emitted_at, "observed_event_count": self.observed_event_count, "accepted_event_count": self.accepted_event_count, @@ -85,8 +132,11 @@ def as_dict(self) -> dict[str, Any]: "buffer_bound": self.buffer_bound, "backpressure_drop_count": self.backpressure_drop_count, "observer_failure_count": self.observer_failure_count, + "peak_buffered_event_count": self.peak_buffered_event_count, + "flush_attempt_count": self.flush_attempt_count, "outbound_endpoints": list(self.outbound_endpoints), "observation_entered_worker_context": self.observation_entered_worker_context, + "observation_entered_scheduler_inputs": self.observation_entered_scheduler_inputs, "clock_source": self.clock_source.value, } @@ -97,6 +147,39 @@ def _count(value: Any, *, name: str) -> int: return value +def _identity_token(value: Any, *, name: str) -> str: + if not isinstance(value, str) or not IDENTITY_TOKEN_PATTERN.match(value): + raise ValueError(f"observer stats {name} must be an identity token") + return value + + +def _token_list(value: Any, *, name: str) -> tuple[str, ...]: + if ( + not isinstance(value, list) + or not value + or any(not isinstance(item, str) or not _ENDPOINT_PATTERN.match(item) for item in value) + ): + raise ValueError(f"observer stats {name} must be a non-empty list of compact tokens") + if len(set(value)) != len(value): + raise ValueError(f"observer stats {name} must not contain duplicates") + return tuple(value) + + +def normalize_observer_run_identity(value: Any) -> ObserverRunIdentity: + if not isinstance(value, Mapping) or set(value) != RUN_IDENTITY_FIELDS: + raise ValueError( + "observer stats run_identity must contain exactly the pinned identity fields" + ) + return ObserverRunIdentity( + **{ + field_name: _identity_token( + value.get(field_name), name=f"run_identity.{field_name}" + ) + for field_name in RUN_IDENTITY_FIELDS + } + ) + + def normalize_observer_stats(record: Mapping[str, Any]) -> ObserverStats: """Validate one stats record written by any observer implementation.""" @@ -110,12 +193,16 @@ def normalize_observer_stats(record: Mapping[str, Any]) -> ObserverStats: if record.get("capability_id") != CAPABILITY_ID: raise ValueError(f"observer stats capability must be {CAPABILITY_ID}") for key in ("provider_id", "observer_id", "goal_id"): - value = record.get(key) - if not isinstance(value, str) or not IDENTITY_TOKEN_PATTERN.match(value): - raise ValueError(f"observer stats {key} must be an identity token") + _identity_token(record.get(key), name=key) emitted_at = record.get("emitted_at") - if not isinstance(emitted_at, str) or not emitted_at.strip(): - raise ValueError("observer stats emitted_at is required") + if not isinstance(emitted_at, str): + raise ValueError("observer stats emitted_at must be timezone-aware ISO-8601 text") + try: + parsed_emitted_at = datetime.fromisoformat(emitted_at.replace("Z", "+00:00")) + except ValueError as exc: + raise ValueError("observer stats emitted_at must be timezone-aware ISO-8601 text") from exc + if parsed_emitted_at.tzinfo is None: + raise ValueError("observer stats emitted_at must carry a timezone") reasons = record.get("rejected_by_reason") or {} if not isinstance(reasons, Mapping): raise ValueError("observer stats rejected_by_reason must be an object") @@ -132,33 +219,57 @@ def normalize_observer_stats(record: Mapping[str, Any]) -> ObserverStats: entered = record.get("observation_entered_worker_context") if not isinstance(entered, bool): raise ValueError("observer stats observation_entered_worker_context must be boolean") + entered_scheduler = record.get("observation_entered_scheduler_inputs") + if not isinstance(entered_scheduler, bool): + raise ValueError( + "observer stats observation_entered_scheduler_inputs must be boolean" + ) buffer_bound = _count(record.get("buffer_bound"), name="buffer_bound") if not 1 <= buffer_bound <= MAX_BUFFER_BOUND: raise ValueError(f"observer stats buffer_bound must be within 1..{MAX_BUFFER_BOUND}") + observed = _count(record.get("observed_event_count"), name="observed_event_count") + accepted = _count(record.get("accepted_event_count"), name="accepted_event_count") + rejected = _count(record.get("rejected_event_count"), name="rejected_event_count") + dropped = _count(record.get("backpressure_drop_count"), name="backpressure_drop_count") + if rejected != sum(normalized_reasons.values()): + raise ValueError( + "observer stats rejected_event_count must equal rejected_by_reason" + ) + if observed != accepted + rejected + dropped: + raise ValueError( + "observer stats observed_event_count must equal accepted + rejected + dropped" + ) + peak_buffered = _count( + record.get("peak_buffered_event_count"), name="peak_buffered_event_count" + ) + if peak_buffered > buffer_bound: + raise ValueError("observer stats peak_buffered_event_count exceeds buffer_bound") return ObserverStats( provider_id=str(record["provider_id"]), observer_id=str(record["observer_id"]), goal_id=str(record["goal_id"]), - emitted_at=emitted_at, - observed_event_count=_count( - record.get("observed_event_count"), name="observed_event_count" - ), - accepted_event_count=_count( - record.get("accepted_event_count"), name="accepted_event_count" - ), - rejected_event_count=_count( - record.get("rejected_event_count"), name="rejected_event_count" + run_identity=normalize_observer_run_identity(record.get("run_identity")), + event_sources=_token_list(record.get("event_sources"), name="event_sources"), + source_fields_consumed=_token_list( + record.get("source_fields_consumed"), name="source_fields_consumed" ), + emitted_at=emitted_at, + observed_event_count=observed, + accepted_event_count=accepted, + rejected_event_count=rejected, rejected_by_reason=normalized_reasons, buffer_bound=buffer_bound, - backpressure_drop_count=_count( - record.get("backpressure_drop_count"), name="backpressure_drop_count" - ), + backpressure_drop_count=dropped, observer_failure_count=_count( record.get("observer_failure_count"), name="observer_failure_count" ), + peak_buffered_event_count=peak_buffered, + flush_attempt_count=_count( + record.get("flush_attempt_count"), name="flush_attempt_count" + ), outbound_endpoints=tuple(endpoints), observation_entered_worker_context=entered, + observation_entered_scheduler_inputs=entered_scheduler, clock_source=ClockSource(str(record.get("clock_source"))), ) @@ -171,12 +282,17 @@ class ShadowObserverIntake: observer_id: str goal_id: str clock_source: ClockSource + run_identity: ObserverRunIdentity + event_sources: tuple[str, ...] + source_fields_consumed: tuple[str, ...] buffer_bound: int = DEFAULT_BUFFER_BOUND observed_event_count: int = 0 accepted_event_count: int = 0 rejected_event_count: int = 0 backpressure_drop_count: int = 0 observer_failure_count: int = 0 + peak_buffered_event_count: int = 0 + flush_attempt_count: int = 0 rejected_by_reason: dict[str, int] = field(default_factory=dict) _buffer: deque[ObserverEnvelope] = field(default_factory=deque, repr=False) @@ -186,6 +302,13 @@ def __post_init__(self) -> None: for key in ("provider_id", "observer_id", "goal_id"): if not IDENTITY_TOKEN_PATTERN.match(getattr(self, key)): raise ValueError(f"{key} must be an identity token") + if not isinstance(self.run_identity, ObserverRunIdentity): + raise ValueError("run_identity must be an ObserverRunIdentity") + normalize_observer_run_identity(self.run_identity.as_dict()) + _token_list(list(self.event_sources), name="event_sources") + _token_list( + list(self.source_fields_consumed), name="source_fields_consumed" + ) @property def buffered_count(self) -> int: @@ -204,8 +327,15 @@ def observe(self, record: Any) -> bool: return False except Exception: # noqa: BLE001 - crash isolation is the contract self.observer_failure_count += 1 + self.rejected_event_count += 1 + reason = EnvelopeRejection.OBSERVER_INTERNAL_FAILURE.value + self.rejected_by_reason[reason] = self.rejected_by_reason.get(reason, 0) + 1 return False - if envelope.goal_id != self.goal_id: + if ( + envelope.provider_id != self.provider_id + or envelope.observer_id != self.observer_id + or envelope.goal_id != self.goal_id + ): self.rejected_event_count += 1 reason = EnvelopeRejection.IDENTITY_INVALID.value self.rejected_by_reason[reason] = self.rejected_by_reason.get(reason, 0) + 1 @@ -215,6 +345,9 @@ def observe(self, record: Any) -> bool: return False self._buffer.append(envelope) self.accepted_event_count += 1 + self.peak_buffered_event_count = max( + self.peak_buffered_event_count, len(self._buffer) + ) return True def drain(self) -> list[ObserverEnvelope]: @@ -227,6 +360,9 @@ def stats(self, *, emitted_at: str) -> ObserverStats: provider_id=self.provider_id, observer_id=self.observer_id, goal_id=self.goal_id, + run_identity=self.run_identity, + event_sources=self.event_sources, + source_fields_consumed=self.source_fields_consumed, emitted_at=emitted_at, observed_event_count=self.observed_event_count, accepted_event_count=self.accepted_event_count, @@ -235,8 +371,11 @@ def stats(self, *, emitted_at: str) -> ObserverStats: buffer_bound=self.buffer_bound, backpressure_drop_count=self.backpressure_drop_count, observer_failure_count=self.observer_failure_count, + peak_buffered_event_count=self.peak_buffered_event_count, + flush_attempt_count=self.flush_attempt_count, outbound_endpoints=(), observation_entered_worker_context=False, + observation_entered_scheduler_inputs=False, clock_source=self.clock_source, ) @@ -254,10 +393,12 @@ def flush( envelopes = self.drain() records = [envelope.as_dict() for envelope in envelopes] + self.flush_attempt_count += 1 try: sink([*records, self.stats(emitted_at=emitted_at).as_dict()]) except Exception: # noqa: BLE001 - crash isolation is the contract self.observer_failure_count += 1 + self.accepted_event_count -= len(records) self.backpressure_drop_count += len(records) return [] return records diff --git a/loopx/capabilities/reliability_diagnostics/projection.py b/loopx/capabilities/reliability_diagnostics/projection.py index bb004340a7..c6b569b0d3 100644 --- a/loopx/capabilities/reliability_diagnostics/projection.py +++ b/loopx/capabilities/reliability_diagnostics/projection.py @@ -48,7 +48,11 @@ def _stage_after(envelope: ObserverEnvelope) -> DiagnosticStage: return DiagnosticStage.ERRORED if kind is ObserverEventKind.TOOL_CALLED: return DiagnosticStage.TOOL_RUNNING - if kind in {ObserverEventKind.TURN_ENDED, ObserverEventKind.SESSION_STARTED}: + if kind is ObserverEventKind.TURN_ENDED: + if str(envelope.summary.get("reason", "")) in _TERMINAL_ERROR_REASONS: + return DiagnosticStage.ERRORED + return DiagnosticStage.IDLE + if kind is ObserverEventKind.SESSION_STARTED: return DiagnosticStage.IDLE if kind is ObserverEventKind.AGENT_STATUS: return DiagnosticStage.RUNNING if envelope.summary.get("status") == "running" else DiagnosticStage.IDLE @@ -89,7 +93,13 @@ def build_diagnostic_projection( counts["turns_started"] += 1 elif kind is ObserverEventKind.TURN_ENDED: counts["turns_ended"] += 1 - if unrecovered_errors and str(envelope.summary.get("reason", "")) not in _TERMINAL_ERROR_REASONS: + terminal_error = ( + str(envelope.summary.get("reason", "")) in _TERMINAL_ERROR_REASONS + ) + if terminal_error and not unrecovered_errors: + counts["errors"] += 1 + unrecovered_errors = 1 + elif unrecovered_errors and not terminal_error: recovered_errors += unrecovered_errors unrecovered_errors = 0 elif kind is ObserverEventKind.STEP_ENDED: diff --git a/loopx/capabilities/reliability_diagnostics/receipt.py b/loopx/capabilities/reliability_diagnostics/receipt.py index 2b08846967..49b45b146b 100644 --- a/loopx/capabilities/reliability_diagnostics/receipt.py +++ b/loopx/capabilities/reliability_diagnostics/receipt.py @@ -44,6 +44,9 @@ class ReceiptReason(StrEnum): CONTROL_FIELD_REJECTED = "control_field_rejected" LEDGER_RECORD_INVALID = "ledger_record_invalid" OBSERVER_STATS_MISSING = "observer_stats_missing" + OBSERVER_STATS_MISMATCH = "observer_stats_mismatch" + IDENTITY_REJECTED = "identity_rejected" + OBSERVATION_ENTERED_SCHEDULER_INPUTS = "observation_entered_scheduler_inputs" SEQUENCE_GAP = "sequence_gap" SEQUENCE_DUPLICATE = "sequence_duplicate" BACKPRESSURE_DROP = "backpressure_drop" @@ -57,13 +60,17 @@ class ReceiptReason(StrEnum): ReceiptReason.NO_OBSERVATIONS, ReceiptReason.OUTBOUND_ENDPOINT_CONFIGURED, ReceiptReason.OBSERVATION_ENTERED_WORKER_CONTEXT, + ReceiptReason.OBSERVATION_ENTERED_SCHEDULER_INPUTS, + ReceiptReason.LEDGER_RECORD_INVALID, + ReceiptReason.OBSERVER_STATS_MISSING, + ReceiptReason.OBSERVER_STATS_MISMATCH, + ReceiptReason.IDENTITY_REJECTED, } ) _QUARANTINE_REASONS = frozenset( { ReceiptReason.OBSERVER_FAILURE, ReceiptReason.CONTROL_FIELD_REJECTED, - ReceiptReason.LEDGER_RECORD_INVALID, } ) @@ -101,7 +108,14 @@ def read_ledger(records: Iterable[Any], *, goal_id: str, malformed_line_count: i stats = normalize_observer_stats(record) if stats.goal_id != goal_id: raise ValueError("goal_id does not match ledger") - # Stats are cumulative per observer instance; the latest wins. + # Stats are cumulative per observer instance; the latest wins, + # but one observer id may never change its identity mid-ledger. + previous = reading.stats.get(stats.observer_id) + if previous is not None and ( + previous.provider_id != stats.provider_id + or previous.run_identity != stats.run_identity + ): + raise ValueError("observer stats identity changed within one ledger") reading.stats[stats.observer_id] = stats else: raise ValueError("unknown ledger record schema") @@ -111,9 +125,11 @@ def read_ledger(records: Iterable[Any], *, goal_id: str, malformed_line_count: i def _sequence_accounting(envelopes: Iterable[ObserverEnvelope]) -> tuple[int, int]: - by_session: dict[str, list[int]] = defaultdict(list) + by_session: dict[tuple[str, str, str], list[int]] = defaultdict(list) for envelope in envelopes: - by_session[envelope.session_id].append(envelope.sequence) + by_session[ + (envelope.provider_id, envelope.observer_id, envelope.session_id) + ].append(envelope.sequence) lost = 0 duplicates = 0 for sequences in by_session.values(): @@ -140,11 +156,34 @@ def build_integrity_receipt( rejected_by_reason[reason] += count outbound_endpoints = sorted({endpoint for item in stats for endpoint in item.outbound_endpoints}) entered_worker_context = any(item.observation_entered_worker_context for item in stats) + entered_scheduler_inputs = any( + item.observation_entered_scheduler_inputs for item in stats + ) observer_failures = sum(item.observer_failure_count for item in stats) backpressure_drops = sum(item.backpressure_drop_count for item in stats) max_uncertainty = max((item.clock.uncertainty_ms for item in envelopes), default=0) clock_sources = sorted({item.clock.source.value for item in envelopes} | {item.clock_source.value for item in stats}) + envelopes_by_observer: dict[str, list[ObserverEnvelope]] = defaultdict(list) + for envelope in envelopes: + envelopes_by_observer[envelope.observer_id].append(envelope) + stats_mismatch = False + for observer_id, observer_envelopes in envelopes_by_observer.items(): + observer_stats = reading.stats.get(observer_id) + if observer_stats is None: + continue + if ( + {item.provider_id for item in observer_envelopes} + != {observer_stats.provider_id} + or observer_stats.accepted_event_count != len(observer_envelopes) + ): + stats_mismatch = True + if any( + item.accepted_event_count and observer_id not in envelopes_by_observer + for observer_id, item in reading.stats.items() + ): + stats_mismatch = True + reasons: set[ReceiptReason] = set() if not envelopes: reasons.add(ReceiptReason.NO_OBSERVATIONS) @@ -152,14 +191,20 @@ def build_integrity_receipt( reasons.add(ReceiptReason.OUTBOUND_ENDPOINT_CONFIGURED) if entered_worker_context: reasons.add(ReceiptReason.OBSERVATION_ENTERED_WORKER_CONTEXT) + if entered_scheduler_inputs: + reasons.add(ReceiptReason.OBSERVATION_ENTERED_SCHEDULER_INPUTS) if observer_failures: reasons.add(ReceiptReason.OBSERVER_FAILURE) if rejected_by_reason.get(EnvelopeRejection.CONTROL_FIELD_REJECTED.value): reasons.add(ReceiptReason.CONTROL_FIELD_REJECTED) if reading.invalid_record_count: reasons.add(ReceiptReason.LEDGER_RECORD_INVALID) - if envelopes and not stats: + if envelopes and any( + observer_id not in reading.stats for observer_id in envelopes_by_observer + ): reasons.add(ReceiptReason.OBSERVER_STATS_MISSING) + if stats_mismatch: + reasons.add(ReceiptReason.OBSERVER_STATS_MISMATCH) if lost: reasons.add(ReceiptReason.SEQUENCE_GAP) if duplicates: @@ -170,6 +215,8 @@ def build_integrity_receipt( reasons.add(ReceiptReason.RAW_MATERIAL_REJECTED) if rejected_by_reason.get(EnvelopeRejection.UNSUPPORTED_FIELD_REJECTED.value): reasons.add(ReceiptReason.UNSUPPORTED_FIELD_REJECTED) + if rejected_by_reason.get(EnvelopeRejection.IDENTITY_INVALID.value): + reasons.add(ReceiptReason.IDENTITY_REJECTED) if max_uncertainty > clock_uncertainty_degraded_ms: reasons.add(ReceiptReason.CLOCK_UNCERTAINTY_EXCEEDED) @@ -189,9 +236,11 @@ def build_integrity_receipt( "status": status.value, "reason_codes": sorted(reason.value for reason in reasons), "provider_ids": sorted({item.provider_id for item in envelopes} | {item.provider_id for item in stats}), - "observer_ids": sorted(reading.stats), + "observer_ids": sorted(set(reading.stats) | set(envelopes_by_observer)), "session_count": len({item.session_id for item in envelopes}), - "observed_event_count": len(envelopes), + "observed_event_count": sum(item.observed_event_count for item in stats), + "accepted_event_count": sum(item.accepted_event_count for item in stats), + "persisted_event_count": len(envelopes), "lost_event_count": lost, "duplicate_sequence_count": duplicates, "ledger_invalid_record_count": reading.invalid_record_count, @@ -200,9 +249,24 @@ def build_integrity_receipt( "buffer_bound": max((item.buffer_bound for item in stats), default=None), "backpressure_drop_count": backpressure_drops, "observer_failure_count": observer_failures, + "peak_buffered_event_count": max( + (item.peak_buffered_event_count for item in stats), default=0 + ), + "flush_attempt_count": sum(item.flush_attempt_count for item in stats), "clock": {"sources": clock_sources, "max_uncertainty_ms": max_uncertainty}, "outbound_endpoints": outbound_endpoints, "observation_entered_worker_context": entered_worker_context, + "observation_entered_scheduler_inputs": entered_scheduler_inputs, + "run_identities": [ + item.run_identity.as_dict() + for item in sorted(stats, key=lambda candidate: candidate.observer_id) + ], + "event_sources": sorted( + {source for item in stats for source in item.event_sources} + ), + "source_fields_consumed": sorted( + {field for item in stats for field in item.source_fields_consumed} + ), "event_kinds_consumed": sorted({item.event_kind.value for item in envelopes}), "summary_fields_consumed": sorted({key for item in envelopes for key in item.summary}), "observed_from": envelopes[0].observed_at if envelopes else None, diff --git a/loopx/cli_commands/reliability_diagnostics.py b/loopx/cli_commands/reliability_diagnostics.py index b446a6ee31..281077c702 100644 --- a/loopx/cli_commands/reliability_diagnostics.py +++ b/loopx/cli_commands/reliability_diagnostics.py @@ -11,14 +11,16 @@ from ..capabilities.reliability_diagnostics import ( CAPABILITY_ID, + OBSERVER_ENVELOPE_SCHEMA_VERSION, OBSERVER_STATS_SCHEMA_VERSION, - ClockSource, - ShadowObserverIntake, + EnvelopeRejection, + ObserverEnvelopeError, append_ledger_records, build_diagnostic_projection, build_integrity_receipt, ledger_path, ledger_ref, + normalize_observer_envelope, normalize_observer_stats, parse_ndjson_lines, read_ledger, @@ -31,8 +33,7 @@ FormatSelector = Callable[..., str] AddFormat = Callable[[argparse.ArgumentParser], None] -INGEST_OBSERVER_ID = "loopx-cli-ingest" -INGEST_BUFFER_BOUND = 4096 +INGEST_VIOLATION_SCHEMA_VERSION = "reliability_ingest_violation_v0" def register_reliability_diagnostics_commands( @@ -81,55 +82,80 @@ def _ingest(path: Path, goal_id: str, source: str) -> dict[str, Any]: else: lines = Path(source).expanduser().read_text(encoding="utf-8").splitlines() parsed, malformed = parse_ndjson_lines(lines) - intake = ShadowObserverIntake( - provider_id="loopx-core", - observer_id=INGEST_OBSERVER_ID, - goal_id=goal_id, - clock_source=ClockSource.OBSERVER_WALL_CLOCK, - buffer_bound=INGEST_BUFFER_BOUND, - ) - appended = 0 + accepted_records: list[dict[str, Any]] = [] + accepted_envelopes = 0 passthrough_stats = 0 + rejected_by_reason: dict[str, int] = {} + + def reject(reason: EnvelopeRejection) -> None: + rejected_by_reason[reason.value] = rejected_by_reason.get(reason.value, 0) + 1 + for record in parsed: - if isinstance(record, dict) and record.get("schema_version") == OBSERVER_STATS_SCHEMA_VERSION: + if not isinstance(record, dict): + reject(EnvelopeRejection.SCHEMA_MISMATCH) + continue + schema = record.get("schema_version") + if schema == OBSERVER_STATS_SCHEMA_VERSION: try: stats = normalize_observer_stats(record) except ValueError: - malformed += 1 + reject(EnvelopeRejection.SCHEMA_MISMATCH) continue if stats.goal_id != goal_id: - malformed += 1 + reject(EnvelopeRejection.IDENTITY_INVALID) continue - appended += append_ledger_records(path, [stats.as_dict()]) + accepted_records.append(stats.as_dict()) passthrough_stats += 1 continue - intake.observe(record) - if intake.buffered_count >= INGEST_BUFFER_BOUND: - appended += len(intake.flush(lambda records: append_ledger_records(path, records[:-1]), emitted_at=_now())) - appended += len(intake.flush(lambda records: append_ledger_records(path, records[:-1]), emitted_at=_now())) - stats_record = intake.stats(emitted_at=_now()) - # A clean ingest is a transparent copy of the observer output. The ingest - # gate records itself only when it refused, dropped, or failed something, - # so that violation stays durable and visible in the receipt. - gate_recorded = bool( - stats_record.rejected_event_count - or stats_record.backpressure_drop_count - or stats_record.observer_failure_count - ) + if schema != OBSERVER_ENVELOPE_SCHEMA_VERSION: + reject(EnvelopeRejection.SCHEMA_MISMATCH) + continue + try: + envelope = normalize_observer_envelope(record) + except ObserverEnvelopeError as exc: + reject(exc.reason) + continue + if envelope.goal_id != goal_id: + reject(EnvelopeRejection.IDENTITY_INVALID) + continue + accepted_records.append(envelope.as_dict()) + accepted_envelopes += 1 + + for _ in range(malformed): + reject(EnvelopeRejection.SCHEMA_MISMATCH) + appended = append_ledger_records(path, accepted_records) + rejected_event_count = sum(rejected_by_reason.values()) + gate_recorded = rejected_event_count > 0 if gate_recorded: - appended += append_ledger_records(path, [stats_record.as_dict()]) + # The gate record is intentionally not an observer schema. The ledger + # reader retains it as a durable invalid-record reason instead of + # silently forgetting input that the ingest boundary refused. + appended += append_ledger_records( + path, + [ + { + "schema_version": INGEST_VIOLATION_SCHEMA_VERSION, + "capability_id": CAPABILITY_ID, + "goal_id": goal_id, + "emitted_at": _now(), + "malformed_line_count": malformed, + "rejected_event_count": rejected_event_count, + "rejected_by_reason": dict(sorted(rejected_by_reason.items())), + } + ], + ) return { "ok": True, "command": "ingest", "goal_id": goal_id, "ledger_ref": ledger_ref(goal_id), "appended_record_count": appended, - "accepted_envelope_count": stats_record.accepted_event_count, + "accepted_envelope_count": accepted_envelopes, "passthrough_stats_count": passthrough_stats, - "rejected_event_count": stats_record.rejected_event_count, - "rejected_by_reason": dict(stats_record.rejected_by_reason), + "rejected_event_count": rejected_event_count, + "rejected_by_reason": dict(sorted(rejected_by_reason.items())), "malformed_line_count": malformed, - "observer_failure_count": stats_record.observer_failure_count, + "observer_failure_count": 0, "ingest_gate_recorded": gate_recorded, } diff --git a/packages/dsh-loopx-plugin/cordis.patch.yml b/packages/dsh-loopx-plugin/cordis.patch.yml index 3a6b10ea73..0d1b0089a7 100644 --- a/packages/dsh-loopx-plugin/cordis.patch.yml +++ b/packages/dsh-loopx-plugin/cordis.patch.yml @@ -10,6 +10,11 @@ - id: loopx-driver name: dsh-loopx-plugin/driver + # Loaded as its own default-off Cordis row. It has no Driver or Agent + # injection and registers only session publication hooks when fully pinned. + - id: loopx-shadow-observer + name: dsh-loopx-plugin/observer + # The browser must not advertise readiness while first-run skill bootstrap is # still in flight. A safe bootstrap failure also publishes the service, so DSH # remains usable and /loopx-init can repair it. diff --git a/packages/dsh-loopx-plugin/package.json b/packages/dsh-loopx-plugin/package.json index 3f84427a7a..32dd836c0d 100644 --- a/packages/dsh-loopx-plugin/package.json +++ b/packages/dsh-loopx-plugin/package.json @@ -18,6 +18,10 @@ "types": "./lib/types/driver.d.ts", "default": "./lib/driver.js" }, + "./observer": { + "types": "./lib/types/observer.d.ts", + "default": "./lib/observer.js" + }, "./client": { "types": "./lib/types/client/index.d.ts", "default": "./lib/client.js" diff --git a/packages/dsh-loopx-plugin/src/driver.ts b/packages/dsh-loopx-plugin/src/driver.ts index 44ed917379..4df4051341 100644 --- a/packages/dsh-loopx-plugin/src/driver.ts +++ b/packages/dsh-loopx-plugin/src/driver.ts @@ -13,7 +13,6 @@ import { } from './cli.ts' import type { FileRunner, LoopXCommand } from './cli.ts' import { resolvePluginLoopXCommand } from './managed-runtime.ts' -import { applyObserver, resolveShadowObserverConfig } from './observer.ts' import { goalBarCoordinator } from './goalbar/events.ts' import type { GoalBarDriverActionReceipt, @@ -1160,11 +1159,6 @@ export class LoopXContinuationDriver { } export function apply(ctx: Context): void { - // L1 shadow observer: separate module, separate effect, no shared send path. - // Feature-off parity: with no declared goal nothing below registers a hook. - const observerConfig = resolveShadowObserverConfig() - if (observerConfig !== undefined) applyObserver(ctx, observerConfig) - const driver = new LoopXContinuationDriver({ isLiveAgent: agent => ctx.agents.get(agent.id) === agent, resolveCommand: signal => resolvePluginLoopXCommand({ signal }), diff --git a/packages/dsh-loopx-plugin/src/index.ts b/packages/dsh-loopx-plugin/src/index.ts index 9ee114480c..2378bc9426 100644 --- a/packages/dsh-loopx-plugin/src/index.ts +++ b/packages/dsh-loopx-plugin/src/index.ts @@ -70,14 +70,4 @@ export type { LoopXInitSummary, } from './init-command.ts' export { resolvePluginLoopXCommand } from './managed-runtime.ts' -export { - applyObserver, - resolveShadowObserverConfig, - ShadowObserver, -} from './observer.ts' -export type { - ObserverEnvelope, - ObserverStats, - ShadowObserverConfig, -} from './observer.ts' export type { LoopXRuntimeOptions } from './managed-runtime.ts' diff --git a/packages/dsh-loopx-plugin/src/observer.ts b/packages/dsh-loopx-plugin/src/observer.ts index 7285c776b4..4d978129ad 100644 --- a/packages/dsh-loopx-plugin/src/observer.ts +++ b/packages/dsh-loopx-plugin/src/observer.ts @@ -15,9 +15,11 @@ import { appendFile, mkdir } from 'node:fs/promises' import { homedir } from 'node:os' import { dirname, join, resolve } from 'node:path' import type { Context } from '@deepseek-ai/cordis' -import type { Agent, AgentStatus } from '@deepseek-ai/dsh-agent' import type { Session, SessionEvent } from '@deepseek-ai/dsh-session' +export const name = 'dsh-loopx-shadow-observer' +export const inject: readonly string[] = [] + export const CAPABILITY_ID = 'reliability-diagnostics' export const PROVIDER_ID = 'dsh-session-events' export const OBSERVER_ENVELOPE_SCHEMA_VERSION = 'reliability_observer_envelope_v0' @@ -29,9 +31,42 @@ export const MAX_BUFFER_BOUND = 65_536 export const WALL_CLOCK_UNCERTAINTY_MS = 50 export const ENV_GOAL_ID = 'LOOPX_DSH_SHADOW_OBSERVER_GOAL_ID' +export const ENV_SESSION_ID = 'LOOPX_DSH_SHADOW_OBSERVER_SESSION_ID' +export const ENV_RUN_IDENTITY = 'LOOPX_DSH_SHADOW_OBSERVER_RUN_IDENTITY_JSON' export const ENV_LEDGER_DIR = 'LOOPX_DSH_SHADOW_OBSERVER_LEDGER_DIR' export const ENV_BUFFER_BOUND = 'LOOPX_DSH_SHADOW_OBSERVER_BUFFER_BOUND' +export const EVENT_SOURCES = [ + 'session/created', + 'session/disposed', + 'session/event', +] as const +export const SOURCE_FIELDS_CONSUMED = [ + 'event.data.callId', + 'event.data.error.code', + 'event.data.id', + 'event.data.message.source.callId', + 'event.data.name', + 'event.data.reason.kind', + 'event.data.source.kind', + 'event.data.step', + 'event.data.turn', + 'event.seq', + 'event.time', + 'event.type', + 'session.id', +] as const +export const RUN_IDENTITY_FIELDS = [ + 'worker_id', + 'model_id', + 'task_id', + 'environment_id', + 'tools_id', + 'budget_id', + 'adapter_revision', + 'observer_revision', +] as const + const IDENTITY_TOKEN = /^[A-Za-z0-9][A-Za-z0-9_.:-]{0,120}$/u const SUMMARY_TOKEN = /^[A-Za-z0-9][A-Za-z0-9_./:-]{0,79}$/u @@ -56,6 +91,7 @@ export interface ObserverEnvelope { readonly schema_version: typeof OBSERVER_ENVELOPE_SCHEMA_VERSION readonly capability_id: typeof CAPABILITY_ID readonly provider_id: typeof PROVIDER_ID + readonly observer_id: string readonly goal_id: string readonly session_id: string readonly agent_id?: string @@ -67,12 +103,26 @@ export interface ObserverEnvelope { readonly source_refs: Readonly> } +export interface ObserverRunIdentity { + readonly worker_id: string + readonly model_id: string + readonly task_id: string + readonly environment_id: string + readonly tools_id: string + readonly budget_id: string + readonly adapter_revision: string + readonly observer_revision: string +} + export interface ObserverStats { readonly schema_version: typeof OBSERVER_STATS_SCHEMA_VERSION readonly capability_id: typeof CAPABILITY_ID readonly provider_id: typeof PROVIDER_ID readonly observer_id: string readonly goal_id: string + readonly run_identity: ObserverRunIdentity + readonly event_sources: readonly string[] + readonly source_fields_consumed: readonly string[] readonly emitted_at: string readonly observed_event_count: number readonly accepted_event_count: number @@ -81,14 +131,19 @@ export interface ObserverStats { readonly buffer_bound: number readonly backpressure_drop_count: number readonly observer_failure_count: number + readonly peak_buffered_event_count: number + readonly flush_attempt_count: number /** Always empty: the observer has no outbound control path to declare. */ readonly outbound_endpoints: readonly [] readonly observation_entered_worker_context: false + readonly observation_entered_scheduler_inputs: false readonly clock_source: ClockSource } export interface ShadowObserverConfig { readonly goalId: string + readonly sessionId: string + readonly runIdentity: ObserverRunIdentity readonly ledgerDir: string readonly bufferBound: number } @@ -111,19 +166,47 @@ export function defaultLedgerDir(env: NodeJS.ProcessEnv = process.env): string { } /** - * The observer is OFF unless one exact goal id is declared. Returning - * `undefined` is the feature-off path: no hooks, no files. + * The observer is OFF unless one exact goal, session, and pinned run identity + * are declared. Returning `undefined` is the feature-off path: no hooks or + * files. Partial or malformed configuration never broadens observation. */ export function resolveShadowObserverConfig( env: NodeJS.ProcessEnv = process.env, ): ShadowObserverConfig | undefined { const goalId = env[ENV_GOAL_ID]?.trim() - if (!goalId || !IDENTITY_TOKEN.test(goalId)) return undefined + const sessionId = env[ENV_SESSION_ID]?.trim() + const runIdentity = parseRunIdentity(env[ENV_RUN_IDENTITY]) + if (!goalId || !IDENTITY_TOKEN.test(goalId) + || !sessionId || !IDENTITY_TOKEN.test(sessionId) + || runIdentity === undefined) return undefined const rawBound = Number.parseInt(env[ENV_BUFFER_BOUND] ?? '', 10) const bufferBound = Number.isInteger(rawBound) && rawBound >= 1 && rawBound <= MAX_BUFFER_BOUND ? rawBound : DEFAULT_BUFFER_BOUND - return { goalId, ledgerDir: defaultLedgerDir(env), bufferBound } + return { goalId, sessionId, runIdentity, ledgerDir: defaultLedgerDir(env), bufferBound } +} + +function validRunIdentity(value: unknown): value is ObserverRunIdentity { + if (typeof value !== 'object' || value === null || Array.isArray(value)) return false + const record = value as Record + const actual = Object.keys(record).sort() + const expected = [...RUN_IDENTITY_FIELDS].sort() + return actual.length === expected.length + && expected.every((field, index) => actual[index] === field) + && RUN_IDENTITY_FIELDS.every((field) => { + const item = record[field] + return typeof item === 'string' && IDENTITY_TOKEN.test(item) + }) +} + +function parseRunIdentity(raw: string | undefined): ObserverRunIdentity | undefined { + if (raw === undefined) return undefined + try { + const parsed = JSON.parse(raw) as unknown + return validRunIdentity(parsed) ? parsed : undefined + } catch { + return undefined + } } export function ledgerPath(config: ShadowObserverConfig): string { @@ -172,7 +255,7 @@ function compactSessionEvent(event: SessionEvent): CompactEvent | undefined { case 'turn/start': return { kind: 'turn_started', summary, sourceRefs } case 'turn/end': - put('reason', token(data.reason)) + put('reason', token((data.reason as Record | undefined)?.kind)) return { kind: 'turn_ended', summary, sourceRefs } case 'step/start': return { kind: 'step_started', summary, sourceRefs } @@ -219,88 +302,63 @@ export class ShadowObserver { private readonly appendLines: LedgerAppender private readonly warn: (message: string) => void private readonly observerId: string - private readonly sequences = new Map() + private nextSequence = 0 private buffer: ObserverEnvelope[] = [] private flushing: Promise | undefined private flushRequested = false private disposed = false private observedEventCount = 0 private acceptedEventCount = 0 + private rejectedEventCount = 0 + private readonly rejectedByReason = new Map() private backpressureDropCount = 0 private observerFailureCount = 0 + private peakBufferedEventCount = 0 + private flushAttemptCount = 0 constructor(options: ShadowObserverOptions) { if (!IDENTITY_TOKEN.test(options.config.goalId)) throw new Error('goal id must be an identity token') + if (!IDENTITY_TOKEN.test(options.config.sessionId)) throw new Error('session id must be an identity token') + if (!validRunIdentity(options.config.runIdentity)) throw new Error('run identity must be fully pinned') if (!Number.isInteger(options.config.bufferBound) || options.config.bufferBound < 1 || options.config.bufferBound > MAX_BUFFER_BOUND) { throw new Error(`buffer bound must be within 1..${MAX_BUFFER_BOUND}`) } - this.config = options.config + const observerId = options.observerId ?? `${PROVIDER_ID}-${randomUUID()}` + if (!IDENTITY_TOKEN.test(observerId)) throw new Error('observer id must be an identity token') + this.config = { + ...options.config, + runIdentity: { ...options.config.runIdentity }, + } this.now = options.now ?? Date.now this.appendLines = options.appendLines ?? appendLedgerLines this.warn = options.warn ?? (() => {}) - this.observerId = options.observerId ?? `${PROVIDER_ID}-${randomUUID()}` + this.observerId = observerId } get path(): string { return ledgerPath(this.config) } - observeSessionStart(agent: Agent): void { - this.isolated(() => this.record('session_started', agent.session, agent, undefined, {}, {})) - } - - observeAgentStatus(agent: Agent, status: AgentStatus): void { - this.isolated(() => { - const summary: Record = {} - const compact = token(status) - if (compact !== undefined) summary.status = compact - this.record('agent_status', agent.session, agent, undefined, summary, {}) - if (status === 'idle') this.requestFlush() - }) - } - - observeAgentError(agent: Agent, detail: { readonly turn?: number, readonly step?: number, readonly error?: unknown }): void { - this.isolated(() => { - const summary: Record = {} - const turn = count(detail.turn) - const step = count(detail.step) - if (turn !== undefined) summary.turn = turn - if (step !== undefined) summary.step = step - const error = detail.error - const errorClass = error instanceof Error ? token(error.name) : undefined - if (errorClass !== undefined) summary.error_class = errorClass - this.record('agent_error', agent.session, agent, undefined, summary, {}) - this.requestFlush() - }) - } - - observePreStep(agent: Agent, detail: { readonly turn?: number, readonly step?: number }): void { - this.isolated(() => { - const summary: Record = {} - const turn = count(detail.turn) - const step = count(detail.step) - if (turn !== undefined) summary.turn = turn - if (step !== undefined) summary.step = step - this.record('agent_pre_step', agent.session, agent, undefined, summary, {}) - }) + observeSessionCreated(session: Session): void { + this.isolated(() => this.record('session_started', session, undefined, {}, {})) } observeSessionEvent(session: Session, event: SessionEvent): void { this.isolated(() => { + if (this.rejectIfUnbound(session)) return const compact = compactSessionEvent(event) if (compact === undefined) return const time = typeof event.time === 'number' && Number.isFinite(event.time) ? event.time : undefined - this.record(compact.kind, session, undefined, time, compact.summary, compact.sourceRefs) + this.record(compact.kind, session, time, compact.summary, compact.sourceRefs) if (event.type === 'turn/end') this.requestFlush() }) } observeSessionDisposed(session: Session): void { this.isolated(() => { - this.record('session_disposed', session, undefined, undefined, {}, {}) - this.sequences.delete(String(session.id)) + this.record('session_disposed', session, undefined, {}, {}) this.requestFlush() }) } @@ -312,16 +370,22 @@ export class ShadowObserver { provider_id: PROVIDER_ID, observer_id: this.observerId, goal_id: this.config.goalId, + run_identity: { ...this.config.runIdentity }, + event_sources: EVENT_SOURCES, + source_fields_consumed: SOURCE_FIELDS_CONSUMED, emitted_at: new Date(this.now()).toISOString(), observed_event_count: this.observedEventCount, accepted_event_count: this.acceptedEventCount, - rejected_event_count: 0, - rejected_by_reason: {}, + rejected_event_count: this.rejectedEventCount, + rejected_by_reason: Object.fromEntries([...this.rejectedByReason].sort(([left], [right]) => left.localeCompare(right))), buffer_bound: this.config.bufferBound, backpressure_drop_count: this.backpressureDropCount, observer_failure_count: this.observerFailureCount, + peak_buffered_event_count: this.peakBufferedEventCount, + flush_attempt_count: this.flushAttemptCount, outbound_endpoints: [], observation_entered_worker_context: false, + observation_entered_scheduler_inputs: false, clock_source: 'harness_event_time', } } @@ -335,11 +399,13 @@ export class ShadowObserver { } const taken = this.buffer this.buffer = [] + this.flushAttemptCount += 1 const lines = [...taken, this.stats()].map(record => JSON.stringify(record)) this.flushing = this.appendLines(this.path, lines).then( () => undefined, (error: unknown) => { this.observerFailureCount += 1 + this.acceptedEventCount -= taken.length this.backpressureDropCount += taken.length this.warn(`dsh-loopx shadow observer flush failed: ${error instanceof Error ? error.name : 'unknown'}`) }, @@ -363,10 +429,13 @@ export class ShadowObserver { private isolated(body: () => void): void { if (this.disposed) return + const observedBefore = this.observedEventCount try { body() } catch (error: unknown) { this.observerFailureCount += 1 + if (this.observedEventCount === observedBefore) this.observedEventCount += 1 + this.reject('observer_internal_failure') this.warn(`dsh-loopx shadow observer hook failed: ${error instanceof Error ? error.name : 'unknown'}`) } } @@ -374,32 +443,38 @@ export class ShadowObserver { private record( kind: ObserverEventKind, session: Session, - agent: Agent | undefined, harnessTimeMs: number | undefined, summary: Record, sourceRefs: Record, ): void { this.observedEventCount += 1 const sessionId = identity(session.id) - if (sessionId === undefined) throw new Error('session id is not an identity token') - const sequence = this.sequences.get(sessionId) ?? 0 - this.sequences.set(sessionId, sequence + 1) + if (sessionId === undefined || sessionId !== this.config.sessionId) { + this.reject('identity_invalid') + return + } + const sequence = this.nextSequence + this.nextSequence += 1 if (this.buffer.length >= this.config.bufferBound) { this.backpressureDropCount += 1 this.requestFlush() return } const observedAtMs = harnessTimeMs ?? this.now() - const agentId = agent === undefined ? undefined : identity(agent.id) + const observedAt = new Date(observedAtMs) + if (!Number.isFinite(observedAt.getTime())) { + this.reject('clock_invalid') + return + } const envelope: ObserverEnvelope = { schema_version: OBSERVER_ENVELOPE_SCHEMA_VERSION, capability_id: CAPABILITY_ID, provider_id: PROVIDER_ID, + observer_id: this.observerId, goal_id: this.config.goalId, session_id: sessionId, - ...(agentId === undefined ? {} : { agent_id: agentId }), sequence, - observed_at: new Date(observedAtMs).toISOString(), + observed_at: observedAt.toISOString(), clock: harnessTimeMs === undefined ? { source: 'observer_wall_clock', uncertainty_ms: WALL_CLOCK_UNCERTAINTY_MS } : { source: 'harness_event_time', uncertainty_ms: 0 }, @@ -409,35 +484,36 @@ export class ShadowObserver { } this.buffer.push(envelope) this.acceptedEventCount += 1 + this.peakBufferedEventCount = Math.max(this.peakBufferedEventCount, this.buffer.length) if (this.buffer.length >= this.config.bufferBound) this.requestFlush() } + private reject(reason: string): void { + this.rejectedEventCount += 1 + this.rejectedByReason.set(reason, (this.rejectedByReason.get(reason) ?? 0) + 1) + } + + private rejectIfUnbound(session: Session): boolean { + const sessionId = identity(session.id) + if (sessionId === this.config.sessionId) return false + this.observedEventCount += 1 + this.reject('identity_invalid') + return true + } + private requestFlush(): void { void this.flush() } } -/** - * Register read-only hooks only. Called by the Driver row's `apply()` solely - * when `resolveShadowObserverConfig()` returns a config; when it returns - * `undefined`, nothing here runs and feature-off parity holds. - */ -export function applyObserver(ctx: Context, config: ShadowObserverConfig): ShadowObserver { +/** Register only the session log's read-only publication hooks. */ +export function registerShadowObserver(ctx: Context, config: ShadowObserverConfig): ShadowObserver { const observer = new ShadowObserver({ config, warn: message => { ctx.logger.warn(message) }, }) ctx.effect(function* () { - ctx.on('agent/session-start', ({ agent }) => { observer.observeSessionStart(agent) }) - ctx.on('agent/status', ({ agent, status }) => { observer.observeAgentStatus(agent, status) }) - ctx.on('agent/error', ({ agent, turn, step, error }) => { - observer.observeAgentError(agent, { turn, step, error }) - }) - // Waterfall hook: observe, then hand the unchanged decision through. - ctx.on('agent/pre-step', ({ agent, turn, step }, next) => { - observer.observePreStep(agent, { turn, step }) - return next() - }) + ctx.on('session/created', session => { observer.observeSessionCreated(session) }) ctx.on('session/event', (session, event) => { observer.observeSessionEvent(session, event) }) ctx.on('session/disposed', session => { observer.observeSessionDisposed(session) }) yield async () => { @@ -446,3 +522,10 @@ export function applyObserver(ctx: Context, config: ShadowObserverConfig): Shado }, 'dsh-loopx shadow observer lifecycle') return observer } + +/** Cordis entrypoint. Partial configuration is the exact feature-off path. */ +export function apply(ctx: Context): void { + const config = resolveShadowObserverConfig() + if (config === undefined) return + registerShadowObserver(ctx, config) +} diff --git a/packages/dsh-loopx-plugin/tsdown.config.ts b/packages/dsh-loopx-plugin/tsdown.config.ts index e1610d7c89..6f69af31c8 100644 --- a/packages/dsh-loopx-plugin/tsdown.config.ts +++ b/packages/dsh-loopx-plugin/tsdown.config.ts @@ -6,6 +6,7 @@ export default defineConfig({ index: 'build-temp/host/index.js', 'init-command': 'build-temp/host/init-command.js', driver: 'build-temp/host/driver.js', + observer: 'build-temp/host/observer.js', }, outDir: 'lib', format: ['esm'], From 060000b54ce65284b23cde63a115ed305b3d1301 Mon Sep 17 00:00:00 2001 From: song Date: Sat, 5 Sep 2026 00:18:54 +0800 Subject: [PATCH 07/13] test(reliability-diagnostics): prove passive observer isolation Signed-off-by: song --- .../dsh-shadow-observer-fixture-smoke.py | 23 ++- .../reliability_diagnostics/fixture.py | 30 ++- .../smoke/dsh-client-artifact-smoke.mjs | 28 ++- .../smoke/dsh-goalbar-runtime-smoke.mjs | 2 + .../smoke/dsh-profile-smoke.mjs | 3 + .../dsh-loopx-plugin/tests/observer.spec.ts | 142 ++++++++++---- .../test_reliability_diagnostics.py | 185 ++++++++++++++++-- ...st_reliability_diagnostics_dsh_provider.py | 33 +++- 8 files changed, 374 insertions(+), 72 deletions(-) diff --git a/examples/reliability_diagnostics/dsh-shadow-observer-fixture-smoke.py b/examples/reliability_diagnostics/dsh-shadow-observer-fixture-smoke.py index 3e9fc00745..5a51ed7aae 100644 --- a/examples/reliability_diagnostics/dsh-shadow-observer-fixture-smoke.py +++ b/examples/reliability_diagnostics/dsh-shadow-observer-fixture-smoke.py @@ -61,6 +61,7 @@ def run_cli(*args: str, runtime_root: Path, stdin: str | None = None) -> dict[st def assert_no_outbound_control(receipt: dict, projection: dict) -> None: assert receipt["outbound_endpoints"] == [], receipt assert receipt["observation_entered_worker_context"] is False, receipt + assert receipt["observation_entered_scheduler_inputs"] is False, receipt assert projection["mode"] == "read_only", projection assert projection["authority"] == "none", projection assert projection["worker_influence"] == "none", projection @@ -73,12 +74,16 @@ def assert_bounded_failure(result: dict) -> None: receipt = result["receipt"] stats = result["stats"] fixture_records = dsh_fixture_records() - # Every fixture record is accounted for exactly once: persisted, rejected, or dropped. - assert len(fixture_records) == ( - receipt["observed_event_count"] + # Every fixture record is observed once, then accepted, rejected, or dropped. + assert len(fixture_records) == receipt["observed_event_count"], ( + len(fixture_records), + receipt, + ) + assert receipt["observed_event_count"] == ( + receipt["accepted_event_count"] + receipt["rejected_event_count"] + receipt["backpressure_drop_count"] - ), (len(fixture_records), receipt) + ), receipt assert stats["buffer_bound"] == FIXTURE_BUFFER_BOUND assert receipt["backpressure_drop_count"] == 3, receipt # Sequence 10 and the rejected sequence 19 are visible as gaps; trailing @@ -87,6 +92,12 @@ def assert_bounded_failure(result: dict) -> None: assert receipt["clock"]["max_uncertainty_ms"] == FIXTURE_UNCERTAIN_CLOCK_MS assert receipt["rejected_by_reason"] == {"raw_material_field_rejected": 1}, receipt assert receipt["observer_failure_count"] == 0 + assert receipt["persisted_event_count"] == receipt["accepted_event_count"] + assert receipt["event_sources"] == [ + "session/created", + "session/disposed", + "session/event", + ] assert receipt["status"] == "degraded", receipt assert set(receipt["reason_codes"]) == { "sequence_gap", @@ -134,12 +145,12 @@ def assert_cli_readback(result: dict) -> None: status = run_cli("status", "--goal-id", FIXTURE_GOAL_ID, runtime_root=runtime_root) assert status["projection"] == result["projection"], status - # A control-shaped record is refused at the ledger door and quarantines the receipt. + # A refused input leaves a durable invalid gate record in the ledger. poisoned = dict(result["ledger_records"][0]) poisoned["command"] = {"kind": "stop"} rejected = run_cli("ingest", "--goal-id", FIXTURE_GOAL_ID, "--input", "-", runtime_root=runtime_root, stdin=json.dumps(poisoned) + "\n") assert rejected["rejected_by_reason"] == {"control_field_rejected": 1}, rejected - assert run_cli("receipt", "--goal-id", FIXTURE_GOAL_ID, runtime_root=runtime_root)["receipt"]["status"] == "quarantined" + assert run_cli("receipt", "--goal-id", FIXTURE_GOAL_ID, runtime_root=runtime_root)["receipt"]["status"] == "invalid" def main() -> int: diff --git a/loopx/capabilities/reliability_diagnostics/fixture.py b/loopx/capabilities/reliability_diagnostics/fixture.py index 5e107eaef0..598dd3e609 100644 --- a/loopx/capabilities/reliability_diagnostics/fixture.py +++ b/loopx/capabilities/reliability_diagnostics/fixture.py @@ -20,7 +20,7 @@ ClockSource, ObserverEventKind, ) -from .intake import ShadowObserverIntake +from .intake import ObserverRunIdentity, ShadowObserverIntake from .projection import build_diagnostic_projection from .receipt import build_integrity_receipt, read_ledger @@ -31,6 +31,16 @@ FIXTURE_BUFFER_BOUND = 20 FIXTURE_UNCERTAIN_CLOCK_MS = 1500 FIXTURE_START = datetime(2026, 9, 1, 12, 0, 0, tzinfo=timezone.utc) +FIXTURE_RUN_IDENTITY = ObserverRunIdentity( + worker_id="dsh-worker-fixture", + model_id="model-fixture", + task_id="task-fixture", + environment_id="environment-fixture", + tools_id="tools-fixture", + budget_id="budget-fixture", + adapter_revision="dsh-adapter-fixture", + observer_revision="observer-fixture", +) def _envelope( @@ -47,6 +57,7 @@ def _envelope( "schema_version": OBSERVER_ENVELOPE_SCHEMA_VERSION, "capability_id": CAPABILITY_ID, "provider_id": DSH_PROVIDER_ID, + "observer_id": FIXTURE_OBSERVER_ID, "goal_id": FIXTURE_GOAL_ID, "session_id": FIXTURE_SESSION_ID, "agent_id": FIXTURE_AGENT_ID, @@ -110,6 +121,23 @@ def run_dsh_fixture() -> dict[str, Any]: observer_id=FIXTURE_OBSERVER_ID, goal_id=FIXTURE_GOAL_ID, clock_source=ClockSource.FIXTURE, + run_identity=FIXTURE_RUN_IDENTITY, + event_sources=("session/created", "session/disposed", "session/event"), + source_fields_consumed=( + "event.data.callId", + "event.data.error.code", + "event.data.id", + "event.data.message.source.callId", + "event.data.name", + "event.data.reason.kind", + "event.data.source.kind", + "event.data.step", + "event.data.turn", + "event.seq", + "event.time", + "event.type", + "session.id", + ), buffer_bound=FIXTURE_BUFFER_BOUND, ) accepted = [intake.observe(record) for record in dsh_fixture_records()] diff --git a/packages/dsh-loopx-plugin/smoke/dsh-client-artifact-smoke.mjs b/packages/dsh-loopx-plugin/smoke/dsh-client-artifact-smoke.mjs index 8baa88cbb8..89b87c22ab 100644 --- a/packages/dsh-loopx-plugin/smoke/dsh-client-artifact-smoke.mjs +++ b/packages/dsh-loopx-plugin/smoke/dsh-client-artifact-smoke.mjs @@ -16,6 +16,7 @@ const rows = [ ['loopx-goalbar', packageId], ['loopx-init-command', `${packageId}/init-command`], ['loopx-driver', `${packageId}/driver`], + ['loopx-shadow-observer', `${packageId}/observer`], ] const clientInject = [ '@deepseek-ai/dsh-client-connection', @@ -42,6 +43,7 @@ const packedStaticEntries = new Set([ 'package/lib/driver.js', 'package/lib/index.js', 'package/lib/init-command.js', + 'package/lib/observer.js', 'package/lib/types/cli.d.ts', 'package/lib/types/client/LoopXGoalBar.d.ts', 'package/lib/types/client/index.d.ts', @@ -57,6 +59,7 @@ const packedStaticEntries = new Set([ 'package/lib/types/index.d.ts', 'package/lib/types/init-command.d.ts', 'package/lib/types/managed-runtime.d.ts', + 'package/lib/types/observer.d.ts', 'package/package.json', ]) const packedHashedEntries = [ @@ -135,6 +138,10 @@ async function assertManifest(root) { types: './lib/types/client/index.d.ts', default: './lib/client.js', }) + assert.deepEqual(manifest.exports?.['./observer'], { + types: './lib/types/observer.d.ts', + default: './lib/observer.js', + }) assert.deepEqual(manifest.dsh?.client, { inject: clientInject, platform: 'web', @@ -419,13 +426,15 @@ async function assertHostExports(root) { root: requireFromPlugin.resolve(packageId), init: requireFromPlugin.resolve(`${packageId}/init-command`), driver: requireFromPlugin.resolve(`${packageId}/driver`), + observer: requireFromPlugin.resolve(`${packageId}/observer`), client: requireFromPlugin.resolve(`${packageId}/client`), } assert.equal(resolutions.client, join(root, 'lib', 'client.js')) - const [host, init, driver] = await Promise.all([ + const [host, init, driver, observer] = await Promise.all([ import(pathToFileURL(resolutions.root).href), import(pathToFileURL(resolutions.init).href), import(pathToFileURL(resolutions.driver).href), + import(pathToFileURL(resolutions.observer).href), ]) assert.equal(typeof host.apply, 'function') assert.equal(host.name, packageId) @@ -474,6 +483,23 @@ async function assertHostExports(root) { assert.equal(typeof host.initializeLoopX, 'function') assert.equal(typeof init.apply, 'function') assert.equal(typeof driver.apply, 'function') + assert.equal(observer.name, 'dsh-loopx-shadow-observer') + assert.deepEqual(observer.inject, []) + assert.equal(typeof observer.apply, 'function') + assert.equal(host.ShadowObserver, undefined) + + const [rootSource, driverSource, observerSource] = await Promise.all([ + readFile(resolutions.root, 'utf8'), + readFile(resolutions.driver, 'utf8'), + readFile(resolutions.observer, 'utf8'), + ]) + for (const source of [rootSource, driverSource]) { + assert(!source.includes('LOOPX_DSH_SHADOW_OBSERVER_GOAL_ID')) + assert(!source.includes('reliability_observer_envelope_v0')) + } + assert(observerSource.includes('LOOPX_DSH_SHADOW_OBSERVER_GOAL_ID')) + assert(!observerSource.includes('.send(')) + assert(!observerSource.includes('agent/pre-step')) } async function exerciseInstalled(installed) { diff --git a/packages/dsh-loopx-plugin/smoke/dsh-goalbar-runtime-smoke.mjs b/packages/dsh-loopx-plugin/smoke/dsh-goalbar-runtime-smoke.mjs index 46ce47030b..f63e394ddf 100644 --- a/packages/dsh-loopx-plugin/smoke/dsh-goalbar-runtime-smoke.mjs +++ b/packages/dsh-loopx-plugin/smoke/dsh-goalbar-runtime-smoke.mjs @@ -46,6 +46,7 @@ const packedStaticEntries = new Set([ 'package/lib/driver.js', 'package/lib/index.js', 'package/lib/init-command.js', + 'package/lib/observer.js', 'package/lib/types/cli.d.ts', 'package/lib/types/client/LoopXGoalBar.d.ts', 'package/lib/types/client/index.d.ts', @@ -1007,6 +1008,7 @@ esac ['loopx-goalbar', packageId], ['loopx-init-command', `${packageId}/init-command`], ['loopx-driver', `${packageId}/driver`], + ['loopx-shadow-observer', `${packageId}/observer`], ]) { const row = installedDump.indexOf(`id: ${id}`) assert( diff --git a/packages/dsh-loopx-plugin/smoke/dsh-profile-smoke.mjs b/packages/dsh-loopx-plugin/smoke/dsh-profile-smoke.mjs index 3029873da3..3b51514f9c 100755 --- a/packages/dsh-loopx-plugin/smoke/dsh-profile-smoke.mjs +++ b/packages/dsh-loopx-plugin/smoke/dsh-profile-smoke.mjs @@ -15,6 +15,7 @@ const rows = [ ['loopx-goalbar', packageId], ['loopx-init-command', `${packageId}/init-command`], ['loopx-driver', `${packageId}/driver`], + ['loopx-shadow-observer', `${packageId}/observer`], ] const packedStaticEntries = new Set([ 'package/LICENSE', @@ -25,6 +26,7 @@ const packedStaticEntries = new Set([ 'package/lib/driver.js', 'package/lib/index.js', 'package/lib/init-command.js', + 'package/lib/observer.js', 'package/lib/types/cli.d.ts', 'package/lib/types/client/LoopXGoalBar.d.ts', 'package/lib/types/client/index.d.ts', @@ -40,6 +42,7 @@ const packedStaticEntries = new Set([ 'package/lib/types/index.d.ts', 'package/lib/types/init-command.d.ts', 'package/lib/types/managed-runtime.d.ts', + 'package/lib/types/observer.d.ts', 'package/package.json', ]) const packedHashedEntries = [ diff --git a/packages/dsh-loopx-plugin/tests/observer.spec.ts b/packages/dsh-loopx-plugin/tests/observer.spec.ts index a26afdf427..834d2a26c3 100644 --- a/packages/dsh-loopx-plugin/tests/observer.spec.ts +++ b/packages/dsh-loopx-plugin/tests/observer.spec.ts @@ -3,13 +3,14 @@ import { dirname, join } from 'node:path' import { fileURLToPath } from 'node:url' import { describe, expect, it } from 'vitest' import type { Context } from '@deepseek-ai/cordis' -import type { Agent } from '@deepseek-ai/dsh-agent' import type { Session, SessionEvent } from '@deepseek-ai/dsh-session' import { - applyObserver, ENV_GOAL_ID, + ENV_RUN_IDENTITY, + ENV_SESSION_ID, OBSERVER_ENVELOPE_SCHEMA_VERSION, OBSERVER_STATS_SCHEMA_VERSION, + registerShadowObserver, resolveShadowObserverConfig, ShadowObserver, } from '../src/observer.ts' @@ -20,26 +21,41 @@ import type { } from '../src/observer.ts' const goalId = 'goal-observer-fixture' -const config: ShadowObserverConfig = { goalId, ledgerDir: '/ledger', bufferBound: 4 } +const sessionId = 'session-fixture' +const runIdentity = { + worker_id: 'worker-fixture', + model_id: 'model-fixture', + task_id: 'task-fixture', + environment_id: 'environment-fixture', + tools_id: 'tools-fixture', + budget_id: 'budget-fixture', + adapter_revision: 'adapter-fixture', + observer_revision: 'observer-fixture', +} as const +const config: ShadowObserverConfig = { + goalId, + sessionId, + runIdentity, + ledgerDir: '/ledger', + bufferBound: 4, +} const ENVELOPE_FIELDS = [ - 'schema_version', 'capability_id', 'provider_id', 'goal_id', 'session_id', + 'schema_version', 'capability_id', 'provider_id', 'observer_id', 'goal_id', 'session_id', 'sequence', 'observed_at', 'clock', 'event_kind', 'summary', 'source_refs', ] const STATS_FIELDS = [ - 'schema_version', 'capability_id', 'provider_id', 'observer_id', 'goal_id', 'emitted_at', + 'schema_version', 'capability_id', 'provider_id', 'observer_id', 'goal_id', 'run_identity', + 'event_sources', 'source_fields_consumed', 'emitted_at', 'observed_event_count', 'accepted_event_count', 'rejected_event_count', 'rejected_by_reason', - 'buffer_bound', 'backpressure_drop_count', 'observer_failure_count', 'outbound_endpoints', - 'observation_entered_worker_context', 'clock_source', + 'buffer_bound', 'backpressure_drop_count', 'observer_failure_count', 'peak_buffered_event_count', + 'flush_attempt_count', 'outbound_endpoints', 'observation_entered_worker_context', + 'observation_entered_scheduler_inputs', 'clock_source', ] function fakeSession(id = 'session-fixture'): Session { return { id } as unknown as Session } -function fakeAgent(session: Session, id = 'agent-fixture'): Agent { - return { id, session, status: 'idle' } as unknown as Agent -} - function sessionEvent(type: string, seq: number, time: number, data: unknown): SessionEvent { return { type, seq, time, data } as unknown as SessionEvent } @@ -74,16 +90,33 @@ function parsed(lines: string[][]): Array { } describe('shadow observer configuration', () => { - it('is off unless one exact goal id is declared', () => { + it('is off unless one exact goal, session, and run identity are declared', () => { expect(resolveShadowObserverConfig({})).toBeUndefined() expect(resolveShadowObserverConfig({ [ENV_GOAL_ID]: ' ' })).toBeUndefined() expect(resolveShadowObserverConfig({ [ENV_GOAL_ID]: 'not an id' })).toBeUndefined() + expect(resolveShadowObserverConfig({ + [ENV_GOAL_ID]: goalId, + [ENV_SESSION_ID]: sessionId, + })).toBeUndefined() const resolved = resolveShadowObserverConfig({ [ENV_GOAL_ID]: goalId, + [ENV_SESSION_ID]: sessionId, + [ENV_RUN_IDENTITY]: JSON.stringify(runIdentity), LOOPX_DSH_SHADOW_OBSERVER_LEDGER_DIR: '/tmp/ledger', LOOPX_DSH_SHADOW_OBSERVER_BUFFER_BOUND: '9', }) - expect(resolved).toEqual({ goalId, ledgerDir: '/tmp/ledger', bufferBound: 9 }) + expect(resolved).toEqual({ + goalId, + sessionId, + runIdentity, + ledgerDir: '/tmp/ledger', + bufferBound: 9, + }) + expect(resolveShadowObserverConfig({ + [ENV_GOAL_ID]: goalId, + [ENV_SESSION_ID]: sessionId, + [ENV_RUN_IDENTITY]: JSON.stringify({ ...runIdentity, extra: 'not-pinned' }), + })).toBeUndefined() }) it('imports nothing from the driver and owns no send path', () => { @@ -102,8 +135,7 @@ describe('shadow observer envelopes', () => { it('maps DSH session events into the shared envelope shape', async () => { const { appended, paths, observer } = observerWithCapture({ bufferBound: 16 }) const session = fakeSession() - const agent = fakeAgent(session) - observer.observeSessionStart(agent) + observer.observeSessionCreated(session) observer.observeSessionEvent(session, sessionEvent('turn/start', 1, 1_756_728_001_000, { turn: 1 })) observer.observeSessionEvent(session, sessionEvent('tool/call', 2, 1_756_728_002_000, { turn: 1, step: 1, callId: 'call-1', name: 'bash', arguments: '{"cmd":"rm -rf /"}', @@ -114,7 +146,10 @@ describe('shadow observer envelopes', () => { })) observer.observeSessionEvent(session, sessionEvent('assistant/chunk', 4, 1_756_728_003_500, { turn: 1, step: 1 })) observer.observeSessionEvent(session, sessionEvent('todo/write', 5, 1_756_728_004_000, { todos: [] })) - observer.observeSessionEvent(session, sessionEvent('turn/end', 6, 1_756_728_005_000, { turn: 1, reason: 'completed' })) + observer.observeSessionEvent(session, sessionEvent('turn/end', 6, 1_756_728_005_000, { + turn: 1, + reason: { kind: 'completed' }, + })) await observer.flush() const records = parsed(appended) @@ -130,15 +165,16 @@ describe('shadow observer envelopes', () => { expect(Object.keys(envelope).filter(key => key !== 'agent_id').sort()).toEqual([...ENVELOPE_FIELDS].sort()) expect(envelope.goal_id).toBe(goalId) expect(envelope.session_id).toBe('session-fixture') + expect(envelope.observer_id).toBe('observer-fixture') } expect(envelopes[0]?.clock).toEqual({ source: 'observer_wall_clock', uncertainty_ms: 50 }) - expect(envelopes[0]?.agent_id).toBe('agent-fixture') expect(envelopes[1]?.clock).toEqual({ source: 'harness_event_time', uncertainty_ms: 0 }) expect(envelopes[1]?.observed_at).toBe('2025-09-01T12:00:01.000Z') expect(envelopes[2]?.summary).toEqual({ turn: 1, step: 1, tool_name: 'bash' }) expect(envelopes[2]?.source_refs).toEqual({ event_seq: '2', tool_call_id: 'call-1' }) expect(envelopes[3]?.summary).toEqual({ turn: 1, step: 1, status: 'error', error_class: 'timeout' }) expect(envelopes[4]?.summary).toEqual({ source_event_type: 'todo/write' }) + expect(envelopes[5]?.summary).toEqual({ turn: 1, reason: 'completed' }) expect(JSON.stringify(records)).not.toContain('rm -rf') const stats = records.at(-1) as ObserverStats @@ -146,8 +182,14 @@ describe('shadow observer envelopes', () => { expect(Object.keys(stats).sort()).toEqual([...STATS_FIELDS].sort()) expect(stats.outbound_endpoints).toEqual([]) expect(stats.observation_entered_worker_context).toBe(false) + expect(stats.observation_entered_scheduler_inputs).toBe(false) + expect(stats.run_identity).toEqual(runIdentity) + expect(stats.event_sources).toEqual(['session/created', 'session/disposed', 'session/event']) expect(stats.observed_event_count).toBe(6) expect(stats.accepted_event_count).toBe(6) + expect(stats.observed_event_count).toBe( + stats.accepted_event_count + stats.rejected_event_count + stats.backpressure_drop_count, + ) }) it('drops with a count while a flush is in flight and the buffer is full', async () => { @@ -162,14 +204,14 @@ describe('shadow observer envelopes', () => { if (release === undefined) await new Promise(resolve => { release = resolve }) }, }) - const agent = fakeAgent(fakeSession()) + const session = fakeSession() // Two observations fill the buffer and start a flush that stays pending. - observer.observePreStep(agent, { turn: 1, step: 1 }) - observer.observePreStep(agent, { turn: 1, step: 2 }) + observer.observeSessionEvent(session, sessionEvent('step/start', 1, 1_756_728_001_000, { turn: 1, step: 1 })) + observer.observeSessionEvent(session, sessionEvent('step/start', 2, 1_756_728_002_000, { turn: 1, step: 2 })) // Two more refill the bound; the fifth has nowhere to go and is dropped. - observer.observePreStep(agent, { turn: 1, step: 3 }) - observer.observePreStep(agent, { turn: 1, step: 4 }) - observer.observePreStep(agent, { turn: 1, step: 5 }) + observer.observeSessionEvent(session, sessionEvent('step/start', 3, 1_756_728_003_000, { turn: 1, step: 3 })) + observer.observeSessionEvent(session, sessionEvent('step/start', 4, 1_756_728_004_000, { turn: 1, step: 4 })) + observer.observeSessionEvent(session, sessionEvent('step/start', 5, 1_756_728_005_000, { turn: 1, step: 5 })) release?.() await observer.flush() const records = parsed(appended) @@ -178,6 +220,7 @@ describe('shadow observer envelopes', () => { expect(stats.accepted_event_count).toBe(4) expect(stats.backpressure_drop_count).toBe(1) expect(stats.observed_event_count).toBe(5) + expect(stats.peak_buffered_event_count).toBe(2) // The sequence still advances for the dropped event so the loss is visible. const envelopes = records.filter( (record): record is ObserverEnvelope => record.schema_version === OBSERVER_ENVELOPE_SCHEMA_VERSION, @@ -185,20 +228,52 @@ describe('shadow observer envelopes', () => { expect(envelopes.map(item => item.sequence)).toEqual([0, 1, 2, 3]) }) + it('preserves the typed DSH turn-end reason used by recovery diagnostics', async () => { + const { appended, observer } = observerWithCapture() + observer.observeSessionEvent( + fakeSession(), + sessionEvent('turn/end', 1, 1_756_728_001_000, { + turn: 2, + reason: { kind: 'error', error: { code: 'UPSTREAM_FAILURE' } }, + }), + ) + await observer.flush() + const [envelope] = parsed(appended).filter( + (record): record is ObserverEnvelope => record.schema_version === OBSERVER_ENVELOPE_SCHEMA_VERSION, + ) + expect(envelope?.event_kind).toBe('turn_ended') + expect(envelope?.summary).toEqual({ turn: 2, reason: 'error' }) + expect(JSON.stringify(envelope)).not.toContain('UPSTREAM_FAILURE') + }) + it('counts hook and flush failures instead of throwing', async () => { const { observer } = observerWithCapture({ failAppend: true }) const session = fakeSession() expect(() => observer.observeSessionEvent(session, undefined as unknown as SessionEvent)).not.toThrow() - observer.observeSessionStart(fakeAgent(session)) + observer.observeSessionCreated(session) await expect(observer.flush()).resolves.toBeUndefined() const stats = observer.stats() expect(stats.observer_failure_count).toBe(2) expect(stats.backpressure_drop_count).toBe(1) + expect(stats.rejected_by_reason).toEqual({ observer_internal_failure: 1 }) + expect(stats.observed_event_count).toBe( + stats.accepted_event_count + stats.rejected_event_count + stats.backpressure_drop_count, + ) + }) + + it('rejects rather than attributing another session to the configured goal', () => { + const { observer } = observerWithCapture() + observer.observeSessionCreated(fakeSession('different-session')) + observer.observeSessionCreated(fakeSession()) + const stats = observer.stats() + expect(stats.accepted_event_count).toBe(1) + expect(stats.rejected_event_count).toBe(1) + expect(stats.rejected_by_reason).toEqual({ identity_invalid: 1 }) }) }) -describe('applyObserver', () => { - it('registers read-only hooks and passes pre-step decisions through unchanged', async () => { +describe('registerShadowObserver', () => { + it('registers only read-only session publication hooks', () => { type Handler = (...args: unknown[]) => unknown const handlers = new Map() let disposeEffect: (() => unknown) | undefined @@ -213,21 +288,12 @@ describe('applyObserver', () => { if (typeof yielded === 'function') disposeEffect = yielded as () => unknown }, } as unknown as Context - const observer = applyObserver(ctx, config) + const observer = registerShadowObserver(ctx, config) expect([...handlers.keys()].sort()).toEqual([ - 'agent/error', 'agent/pre-step', 'agent/session-start', 'agent/status', 'session/disposed', 'session/event', + 'session/created', 'session/disposed', 'session/event', ]) const session = fakeSession() - const agent = fakeAgent(session) - const decision = { kind: 'enter', messages: [] } - let nextCalls = 0 - const preStep = handlers.get('agent/pre-step')?.[0] - const result = await preStep?.({ agent, messages: [], turn: 1, step: 1, signal: new AbortController().signal }, async () => { - nextCalls += 1 - return decision - }) - expect(nextCalls).toBe(1) - expect(result).toBe(decision) + handlers.get('session/created')?.[0]?.(session) expect(observer.stats().observed_event_count).toBe(1) expect(warnings).toEqual([]) expect(typeof disposeEffect).toBe('function') diff --git a/tests/capabilities/test_reliability_diagnostics.py b/tests/capabilities/test_reliability_diagnostics.py index 8916ebb0c1..369eec294b 100644 --- a/tests/capabilities/test_reliability_diagnostics.py +++ b/tests/capabilities/test_reliability_diagnostics.py @@ -30,6 +30,7 @@ EnvelopeRejection, ObserverEnvelopeError, ObserverEventKind, + ObserverRunIdentity, ReceiptReason, ReceiptStatus, ShadowObserverIntake, @@ -46,10 +47,24 @@ run_dsh_fixture, ) from loopx.capabilities.reliability_diagnostics.envelope import classify_rejected_field +from loopx.cli_commands.reliability_diagnostics import _ingest GOAL = "goal-observer" SESSION = "session-observer" T0 = "2026-09-01T10:00:00+00:00" +OBSERVER = "observer-1" +RUN_IDENTITY = ObserverRunIdentity( + worker_id="worker-1", + model_id="model-1", + task_id="task-1", + environment_id="environment-1", + tools_id="tools-1", + budget_id="budget-1", + adapter_revision="adapter-1", + observer_revision="observer-revision-1", +) +EVENT_SOURCES = ["dsh-agent-hooks", "dsh-session-events"] +SOURCE_FIELDS = ["agent.id", "event.data", "event.seq", "event.time", "event.type", "session.id"] def envelope( @@ -64,6 +79,7 @@ def envelope( "schema_version": OBSERVER_ENVELOPE_SCHEMA_VERSION, "capability_id": CAPABILITY_ID, "provider_id": DSH_PROVIDER_ID, + "observer_id": OBSERVER, "goal_id": GOAL, "session_id": SESSION, "sequence": sequence, @@ -78,22 +94,31 @@ def envelope( def stats(**overrides: Any) -> dict[str, Any]: + accepted = overrides.get("accepted_event_count", 1) + rejected = overrides.get("rejected_event_count", 0) + dropped = overrides.get("backpressure_drop_count", 0) record: dict[str, Any] = { "schema_version": OBSERVER_STATS_SCHEMA_VERSION, "capability_id": CAPABILITY_ID, "provider_id": DSH_PROVIDER_ID, - "observer_id": "observer-1", + "observer_id": OBSERVER, "goal_id": GOAL, + "run_identity": RUN_IDENTITY.as_dict(), + "event_sources": EVENT_SOURCES, + "source_fields_consumed": SOURCE_FIELDS, "emitted_at": T0, - "observed_event_count": 1, - "accepted_event_count": 1, - "rejected_event_count": 0, + "observed_event_count": accepted + rejected + dropped, + "accepted_event_count": accepted, + "rejected_event_count": rejected, "rejected_by_reason": {}, "buffer_bound": 8, "backpressure_drop_count": 0, "observer_failure_count": 0, + "peak_buffered_event_count": min(accepted, 8), + "flush_attempt_count": 1, "outbound_endpoints": [], "observation_entered_worker_context": False, + "observation_entered_scheduler_inputs": False, "clock_source": ClockSource.HARNESS_EVENT_TIME.value, } record.update(overrides) @@ -192,6 +217,9 @@ def intake(bound: int = 8) -> ShadowObserverIntake: observer_id="observer-1", goal_id=GOAL, clock_source=ClockSource.FIXTURE, + run_identity=RUN_IDENTITY, + event_sources=tuple(EVENT_SOURCES), + source_fields_consumed=tuple(SOURCE_FIELDS), buffer_bound=bound, ) @@ -211,6 +239,12 @@ def test_intake_bounds_buffer_and_counts_drops_without_raising() -> None: assert record["backpressure_drop_count"] == 3 assert record["outbound_endpoints"] == [] assert record["observation_entered_worker_context"] is False + assert record["observation_entered_scheduler_inputs"] is False + assert record["observed_event_count"] == ( + record["accepted_event_count"] + + record["rejected_event_count"] + + record["backpressure_drop_count"] + ) assert normalize_observer_stats(record).backpressure_drop_count == 3 @@ -257,6 +291,25 @@ def test_stats_record_rejects_unknown_fields() -> None: normalize_observer_stats(stats(send_path="agent.send")) +@pytest.mark.parametrize( + "mutation", + [ + {"emitted_at": "not-a-time"}, + {"emitted_at": "2026-09-01T10:00:00"}, + {"accepted_event_count": 999, "observed_event_count": 1}, + {"run_identity": {"worker_id": "worker-1"}}, + {"event_sources": []}, + {"source_fields_consumed": ["event.type", "event.type"]}, + {"peak_buffered_event_count": 9}, + ], +) +def test_stats_record_rejects_inconsistent_or_incomplete_evidence( + mutation: dict[str, Any], +) -> None: + with pytest.raises(ValueError): + normalize_observer_stats(stats(**mutation)) + + # --- receipt --------------------------------------------------------------- @@ -267,7 +320,12 @@ def test_receipt_without_observations_is_invalid() -> None: def test_receipt_is_valid_for_contiguous_stream_with_stats() -> None: - receipt = receipt_for(envelope(0), envelope(1, observed_at=at(1)), envelope(2, observed_at=at(2)), stats()) + receipt = receipt_for( + envelope(0), + envelope(1, observed_at=at(1)), + envelope(2, observed_at=at(2)), + stats(accepted_event_count=3), + ) assert receipt["status"] == ReceiptStatus.VALID.value assert receipt["reason_codes"] == [] assert receipt["outbound_endpoints"] == [] @@ -277,14 +335,23 @@ def test_receipt_is_valid_for_contiguous_stream_with_stats() -> None: def test_receipt_counts_sequence_gaps_as_lost_events() -> None: - receipt = receipt_for(envelope(0), envelope(4, observed_at=at(1)), envelope(5, observed_at=at(2)), stats()) + receipt = receipt_for( + envelope(0), + envelope(4, observed_at=at(1)), + envelope(5, observed_at=at(2)), + stats(accepted_event_count=3), + ) assert receipt["lost_event_count"] == 3 assert receipt["status"] == ReceiptStatus.DEGRADED.value assert receipt["reason_codes"] == [ReceiptReason.SEQUENCE_GAP.value] def test_receipt_counts_duplicate_sequences_separately() -> None: - receipt = receipt_for(envelope(0), envelope(0, observed_at=at(1)), stats()) + receipt = receipt_for( + envelope(0), + envelope(0, observed_at=at(1)), + stats(accepted_event_count=2), + ) assert receipt["duplicate_sequence_count"] == 1 assert receipt["lost_event_count"] == 0 assert ReceiptReason.SEQUENCE_DUPLICATE.value in receipt["reason_codes"] @@ -303,10 +370,10 @@ def test_receipt_quarantines_observer_failure_and_control_fields(stats_override: assert receipt["reason_codes"] == [reason.value] -def test_receipt_quarantines_malformed_ledger_records() -> None: +def test_receipt_invalidates_malformed_ledger_records() -> None: receipt = receipt_for(envelope(0), stats(), {"schema_version": "unknown"}, malformed=1) assert receipt["ledger_invalid_record_count"] == 2 - assert receipt["status"] == ReceiptStatus.QUARANTINED.value + assert receipt["status"] == ReceiptStatus.INVALID.value @pytest.mark.parametrize( @@ -329,7 +396,6 @@ def test_receipt_invalidates_any_outbound_or_worker_context_path(stats_override: ([envelope(0, clock={"source": "observer_wall_clock", "uncertainty_ms": 1001}), stats()], ReceiptReason.CLOCK_UNCERTAINTY_EXCEEDED), ([envelope(0), stats(backpressure_drop_count=2)], ReceiptReason.BACKPRESSURE_DROP), ([envelope(0), stats(rejected_event_count=1, rejected_by_reason={"raw_material_field_rejected": 1})], ReceiptReason.RAW_MATERIAL_REJECTED), - ([envelope(0)], ReceiptReason.OBSERVER_STATS_MISSING), ], ) def test_receipt_degrades_but_keeps_evidence(records: list[dict[str, Any]], reason: ReceiptReason) -> None: @@ -338,6 +404,28 @@ def test_receipt_degrades_but_keeps_evidence(records: list[dict[str, Any]], reas assert receipt["reason_codes"] == [reason.value] +def test_receipt_without_stats_is_invalid() -> None: + receipt = receipt_for(envelope(0)) + assert receipt["status"] == ReceiptStatus.INVALID.value + assert receipt["reason_codes"] == [ReceiptReason.OBSERVER_STATS_MISSING.value] + + +def test_receipt_invalidates_unlinked_stats_and_identity_rejection() -> None: + unlinked = receipt_for(envelope(0), stats(provider_id="other-provider")) + assert unlinked["status"] == ReceiptStatus.INVALID.value + assert ReceiptReason.OBSERVER_STATS_MISMATCH.value in unlinked["reason_codes"] + + rejected = receipt_for( + envelope(0), + stats( + rejected_event_count=1, + rejected_by_reason={EnvelopeRejection.IDENTITY_INVALID.value: 1}, + ), + ) + assert rejected["status"] == ReceiptStatus.INVALID.value + assert ReceiptReason.IDENTITY_REJECTED.value in rejected["reason_codes"] + + def test_receipt_clock_uncertainty_at_threshold_is_visible_not_degraded() -> None: receipt = receipt_for(envelope(0, clock={"source": "observer_wall_clock", "uncertainty_ms": 1000}), stats()) assert receipt["clock"] == {"sources": ["harness_event_time", "observer_wall_clock"], "max_uncertainty_ms": 1000} @@ -346,7 +434,8 @@ def test_receipt_clock_uncertainty_at_threshold_is_visible_not_degraded() -> Non def test_receipt_sums_latest_stats_per_observer_instance() -> None: receipt = receipt_for( - envelope(0), + envelope(0, observer_id="a"), + envelope(0, observer_id="b", session_id="session-b"), stats(observer_id="a", backpressure_drop_count=1), stats(observer_id="a", backpressure_drop_count=4), stats(observer_id="b", backpressure_drop_count=2), @@ -375,7 +464,7 @@ def test_projection_stall_requires_active_stage_and_silence() -> None: running = projection_for( envelope(0, ObserverEventKind.TURN_STARTED), envelope(1, ObserverEventKind.STEP_STARTED, observed_at=at(1)), - stats(), + stats(accepted_event_count=2), as_of=at(6 * 60), stall_threshold_ms=300_000, ) @@ -386,7 +475,7 @@ def test_projection_stall_requires_active_stage_and_silence() -> None: idle = projection_for( envelope(0, ObserverEventKind.TURN_STARTED), envelope(1, ObserverEventKind.TURN_ENDED, observed_at=at(1), summary={"reason": "completed"}), - stats(), + stats(accepted_event_count=2), as_of=at(6 * 60), ) assert idle["stage"] == DiagnosticStage.IDLE.value @@ -399,9 +488,9 @@ def test_projection_repetition_counts_consecutive_identical_tool_runs() -> None: envelope(index, ObserverEventKind.TOOL_CALLED, observed_at=at(index), summary={"tool_name": tool}) for index, tool in enumerate(tools) ] - projection = projection_for(*records, stats()) + projection = projection_for(*records, stats(accepted_event_count=len(records))) assert projection["repetition"] == {"detected": True, "threshold": 3, "longest_tool_run": 3, "tool_name": "read"} - below = projection_for(*records[:2], stats()) + below = projection_for(*records[:2], stats(accepted_event_count=2)) assert below["repetition"]["detected"] is False @@ -409,7 +498,7 @@ def test_projection_recovery_and_stage_transitions() -> None: unrecovered = projection_for( envelope(0, ObserverEventKind.STEP_STARTED), envelope(1, ObserverEventKind.AGENT_ERROR, observed_at=at(1), summary={"error_class": "Timeout"}), - stats(), + stats(accepted_event_count=2), ) assert unrecovered["stage"] == DiagnosticStage.ERRORED.value assert unrecovered["recovery"] == {"error_count": 1, "recovered_error_count": 0, "unrecovered_error_count": 1} @@ -420,16 +509,36 @@ def test_projection_recovery_and_stage_transitions() -> None: envelope(1, ObserverEventKind.AGENT_ERROR, observed_at=at(1)), envelope(2, ObserverEventKind.STEP_ENDED, observed_at=at(2)), envelope(3, ObserverEventKind.SESSION_DISPOSED, observed_at=at(3)), - stats(), + stats(accepted_event_count=4), ) assert recovered["stage"] == DiagnosticStage.DISPOSED.value assert recovered["recovery"]["unrecovered_error_count"] == 0 assert recovered["recovery"]["recovered_error_count"] == 1 assert recovered["signals"] == [] + terminal_error = projection_for( + envelope( + 0, + ObserverEventKind.TURN_ENDED, + summary={"turn": 1, "reason": "error"}, + ), + stats(), + ) + assert terminal_error["stage"] == DiagnosticStage.ERRORED.value + assert terminal_error["recovery"] == { + "error_count": 1, + "recovered_error_count": 0, + "unrecovered_error_count": 1, + } + assert DiagnosticSignal.UNRECOVERED_ERROR.value in terminal_error["signals"] + def test_projection_surfaces_event_loss_and_integrity() -> None: - projection = projection_for(envelope(0), envelope(3, observed_at=at(1)), stats()) + projection = projection_for( + envelope(0), + envelope(3, observed_at=at(1)), + stats(accepted_event_count=2), + ) assert DiagnosticSignal.EVENT_LOSS.value in projection["signals"] assert projection["integrity"]["status"] == ReceiptStatus.DEGRADED.value assert projection["evidence"]["lost_event_count"] == 2 @@ -459,6 +568,39 @@ def test_ledger_append_is_line_oriented_and_tolerates_malformed_lines(tmp_path: assert malformed == 1 +def test_ingest_persists_a_durable_gate_for_refused_input(tmp_path: Path) -> None: + source = tmp_path / "observer.ndjson" + source.write_text( + "\n".join( + [ + json.dumps(envelope(0)), + "not json", + json.dumps(envelope(1, goal_id="other-goal")), + json.dumps(stats()), + ] + ) + + "\n", + encoding="utf-8", + ) + path = ledger_path(tmp_path, GOAL) + result = _ingest(path, GOAL, str(source)) + + assert result["accepted_envelope_count"] == 1 + assert result["passthrough_stats_count"] == 1 + assert result["malformed_line_count"] == 1 + assert result["rejected_by_reason"] == { + EnvelopeRejection.IDENTITY_INVALID.value: 1, + EnvelopeRejection.SCHEMA_MISMATCH.value: 1, + } + assert result["ingest_gate_recorded"] is True + records, malformed = read_ledger_records(path) + assert malformed == 0 + assert records[-1]["schema_version"] == "reliability_ingest_violation_v0" + receipt = build_integrity_receipt(read_ledger(records, goal_id=GOAL)) + assert receipt["status"] == ReceiptStatus.INVALID.value + assert ReceiptReason.LEDGER_RECORD_INVALID.value in receipt["reason_codes"] + + # --- fixture --------------------------------------------------------------- @@ -476,4 +618,9 @@ def test_dsh_fixture_exercises_every_contract_hazard_and_stays_degraded() -> Non assert receipt["outbound_endpoints"] == [] assert receipt["observer_failure_count"] == 0 assert "transcript" not in json.dumps(result["ledger_records"]) - assert len(dsh_fixture_records()) == receipt["observed_event_count"] + receipt["rejected_event_count"] + receipt["backpressure_drop_count"] + assert len(dsh_fixture_records()) == receipt["observed_event_count"] + assert receipt["observed_event_count"] == ( + receipt["accepted_event_count"] + + receipt["rejected_event_count"] + + receipt["backpressure_drop_count"] + ) diff --git a/tests/capabilities/test_reliability_diagnostics_dsh_provider.py b/tests/capabilities/test_reliability_diagnostics_dsh_provider.py index 592e528f46..7d5f73af20 100644 --- a/tests/capabilities/test_reliability_diagnostics_dsh_provider.py +++ b/tests/capabilities/test_reliability_diagnostics_dsh_provider.py @@ -2,6 +2,7 @@ from __future__ import annotations +import json import re from pathlib import Path @@ -22,6 +23,9 @@ ROOT = Path(__file__).resolve().parents[2] OBSERVER_TS = ROOT / "packages/dsh-loopx-plugin/src/observer.ts" DRIVER_TS = ROOT / "packages/dsh-loopx-plugin/src/driver.ts" +CORDIS_PATCH = ROOT / "packages/dsh-loopx-plugin/cordis.patch.yml" +PACKAGE_JSON = ROOT / "packages/dsh-loopx-plugin/package.json" +TSDOWN_CONFIG = ROOT / "packages/dsh-loopx-plugin/tsdown.config.ts" def test_catalog_declares_dsh_provider_without_claiming_readiness() -> None: @@ -69,14 +73,29 @@ def test_typescript_observer_shares_field_names_and_has_no_control_path() -> Non assert "from './managed-runtime" not in source assert ".send(" not in source assert ".inbox" not in source + assert "@deepseek-ai/dsh-agent" not in source + assert "ctx.on('agent/" not in source assert "outbound_endpoints: []" in source assert "observation_entered_worker_context: false" in source + assert "observation_entered_scheduler_inputs: false" in source -def test_driver_applies_observer_only_when_enabled() -> None: - source = DRIVER_TS.read_text(encoding="utf-8") - assert "resolveShadowObserverConfig()" in source - assert "if (observerConfig !== undefined) applyObserver(ctx, observerConfig)" in source - # The driver must never hand its instance or send path to the observer. - assert "applyObserver(ctx, observerConfig)" in source - assert "applyObserver(ctx, observerConfig, driver" not in source +def test_observer_is_a_separate_default_off_plugin_row() -> None: + driver = DRIVER_TS.read_text(encoding="utf-8") + observer = OBSERVER_TS.read_text(encoding="utf-8") + patch = CORDIS_PATCH.read_text(encoding="utf-8") + build = TSDOWN_CONFIG.read_text(encoding="utf-8") + manifest = json.loads(PACKAGE_JSON.read_text(encoding="utf-8")) + + assert "observer" not in driver.lower() + assert "export const inject: readonly string[] = []" in observer + assert "ctx.on('session/created'" in observer + assert "ctx.on('session/event'" in observer + assert "ctx.on('session/disposed'" in observer + assert "loopx-shadow-observer" in patch + assert "name: dsh-loopx-plugin/observer" in patch + assert "observer: 'build-temp/host/observer.js'" in build + assert manifest["exports"]["./observer"] == { + "types": "./lib/types/observer.d.ts", + "default": "./lib/observer.js", + } From d2885d5e6a1e397797b7e2ec6e92cf686d67626c Mon Sep 17 00:00:00 2001 From: song Date: Sat, 5 Sep 2026 00:19:44 +0800 Subject: [PATCH 08/13] docs(reliability-diagnostics): state prototype evidence limits Signed-off-by: song --- .../reliability_diagnostics/README.md | 96 +++++++++++-------- .../reliability_diagnostics/README.zh-CN.md | 65 ++++++++----- packages/dsh-loopx-plugin/README.md | 42 +++++--- 3 files changed, 123 insertions(+), 80 deletions(-) diff --git a/loopx/capabilities/reliability_diagnostics/README.md b/loopx/capabilities/reliability_diagnostics/README.md index f767091328..5a99547262 100644 --- a/loopx/capabilities/reliability_diagnostics/README.md +++ b/loopx/capabilities/reliability_diagnostics/README.md @@ -2,9 +2,11 @@ [中文](README.zh-CN.md) | [RFC](../../../docs/architecture/rfcs/long-running-agent-reliability-diagnostics-governed-delivery-v0.md) -Status: experimental, built in, default off, goal scoped. This package ships -the P0 slice of the reliability-diagnostics RFC: the **L1 shadow observer** -contract and its first real event source, DeepSeek Harness (DSH). +Status: experimental, built in, default off, goal and session scoped. This +package implements the prototype components described by the RFC's **P0 +roadmap phase**: the L1 shadow-observer contract and its first DSH event-source +adapter. It does **not** claim the P0 exit gate; an eligible C1 observer run, +C0 adapter-fidelity evidence, and measured overhead are still required. An L1 observer sees a long-running agent session and writes an independent diagnostic record about it. It may **never** influence that session. This @@ -22,10 +24,12 @@ flowchart LR O -. "no send, schedule, gate, tool, or worker-state path" .-> H ``` -The dashed edge is an asserted absence. Tests reject envelopes carrying -control-shaped fields, the TypeScript module imports nothing from the -continuation driver, and the receipt turns `invalid` if any outbound endpoint -ever appears. +The dashed edge is an asserted absence. The observer ships as its own Cordis +plugin entry with no Driver or Agent injection, consumes only session-log +publication events, and is absent from the Driver and package-root bundles. +Tests reject control-shaped fields, and the receipt turns `invalid` if any +outbound endpoint or scheduler/worker-context path appears. This is module and +hook isolation, not an OS-process-isolation claim. ## Placement Rationale @@ -38,8 +42,9 @@ ever appears. kebab-case like every other catalog id; the package directory is `reliability_diagnostics`. - **Provider id `dsh-session-events`** (origin `extension`). It is delivered by - the npm package `packages/dsh-loopx-plugin` as `src/observer.ts`, physically - separate from `driver.ts`. Because an npm plugin has no Python + the npm package `packages/dsh-loopx-plugin` through the explicit + `dsh-loopx-plugin/observer` entry and its own `loopx-shadow-observer` Cordis + row, separate from `driver.ts`. Because an npm plugin has no Python `extension.toml` lifecycle, the capability declares the provider on its catalog entry and the registry reports it `declared=true`, `installed=enabled=ready=false`. The precedent is @@ -59,6 +64,7 @@ ever appears. | `schema_version` | literal | `reliability_observer_envelope_v0` | | `capability_id` | literal | `reliability-diagnostics` | | `provider_id` | identity token | e.g. `dsh-session-events` | +| `observer_id` | identity token | stable for one observer instance and linked to stats | | `goal_id`, `session_id` | identity token | `^[A-Za-z0-9][A-Za-z0-9_.:-]{0,120}$` | | `agent_id` | identity token, optional | | | `sequence` | integer >= 0 | observer-assigned, monotonic per session; gaps are counted as loss | @@ -79,13 +85,17 @@ so absolute local paths and credential-like tokens fail closed. ### Observer stats (`reliability_observer_stats_v0`) -Written by every observer implementation next to its envelopes. Fields: -`observer_id`, `emitted_at`, `observed_event_count`, `accepted_event_count`, -`rejected_event_count`, `rejected_by_reason`, `buffer_bound`, -`backpressure_drop_count`, `observer_failure_count`, `outbound_endpoints` -(must be `[]`), `observation_entered_worker_context` (must be `false`), -`clock_source`. Stats are cumulative per observer instance; the receipt keeps -the latest record per `observer_id` and sums across instances. +Written by every observer implementation next to its envelopes. It pins +`worker_id`, `model_id`, `task_id`, `environment_id`, `tools_id`, `budget_id`, +`adapter_revision`, and `observer_revision` under `run_identity`; declares +`event_sources` and `source_fields_consumed`; and records timestamps, accepted +and rejected counts, typed rejection totals, buffer bound, drops, failures, +peak buffered events, flush attempts, clock source, outbound endpoints, and +both worker-context and scheduler-input influence flags. Counts must satisfy +`observed = accepted + rejected + dropped`, timestamps must be timezone-aware, +and every accepted envelope must link to matching provider/observer stats. +Stats are cumulative per observer instance; the receipt keeps the latest +record per `observer_id` and sums across instances. ### Integrity receipt (`reliability_integrity_receipt_v0`) @@ -93,21 +103,23 @@ the latest record per `observer_id` and sums across instances. | --- | --- | | `status` | `valid`, `degraded`, `quarantined`, `invalid` (total, ordered) | | `reason_codes` | typed list; empty only when `valid` | -| `observed_event_count`, `session_count` | accepted envelopes in the ledger | +| `observed_event_count`, `accepted_event_count`, `persisted_event_count` | source attempts, accepted count from stats, and linked envelopes in the ledger | | `lost_event_count`, `duplicate_sequence_count` | per-session sequence gaps and repeats | | `ledger_invalid_record_count` | malformed or foreign records found in the ledger | | `rejected_event_count`, `rejected_by_reason` | refusals reported by observers | | `buffer_bound`, `backpressure_drop_count`, `observer_failure_count` | bounded-failure evidence | | `clock.sources`, `clock.max_uncertainty_ms` | declared clocks; > 1000 ms degrades | -| `outbound_endpoints`, `observation_entered_worker_context` | must be `[]` / `false` | -| `event_kinds_consumed`, `summary_fields_consumed` | the exact sources and fields consumed | - -Status rules: `invalid` when there are no observations, any outbound endpoint, -or any observation entered worker context; otherwise `quarantined` when the -observer failed, a control-shaped record was seen, or the ledger holds -malformed records; otherwise `degraded` when events were lost, dropped, -duplicated, raw material was rejected, stats are missing, or clock uncertainty -exceeded the threshold; otherwise `valid`. +| `outbound_endpoints`, worker/scheduler influence flags | must be `[]` / `false` / `false` | +| `run_identities`, `event_sources`, `source_fields_consumed` | pinned treatment identity and declared adapter coverage | +| `event_kinds_consumed`, `summary_fields_consumed` | the event kinds and compact summary fields actually persisted | + +Status rules: `invalid` when there are no observations, stats are absent or do +not link exactly to persisted envelopes, an identity was rejected, the ledger +contains invalid input, any outbound endpoint exists, or observation entered +worker context or scheduler inputs. Otherwise `quarantined` covers observer +failure or control-shaped input. Event gaps, drops, duplicates, raw or +unsupported fields, and excess clock uncertainty are `degraded`; otherwise the +receipt is `valid`. ### Diagnostic projection (`reliability_diagnostic_projection_v0`) @@ -125,8 +137,10 @@ exceeded the threshold; otherwise `valid`. ## Use It ```bash -# Enable the DSH provider for exactly one goal, then start DSH as usual. +# Enable the DSH provider for one predeclared goal and one exact DSH session. export LOOPX_DSH_SHADOW_OBSERVER_GOAL_ID= +export LOOPX_DSH_SHADOW_OBSERVER_SESSION_ID= +export LOOPX_DSH_SHADOW_OBSERVER_RUN_IDENTITY_JSON='{"worker_id":"","model_id":"","task_id":"","environment_id":"","tools_id":"","budget_id":"","adapter_revision":"","observer_revision":""}' # Optional: LOOPX_DSH_SHADOW_OBSERVER_LEDGER_DIR, LOOPX_DSH_SHADOW_OBSERVER_BUFFER_BOUND loopx reliability-diagnostics receipt --goal-id --format json @@ -136,16 +150,17 @@ loopx reliability-diagnostics ingest --goal-id --input observer.ndjso The ledger lives at `/reliability_diagnostics/.ndjson`; the default runtime root is the same one the rest of LoopX uses and the CLI -prints only the relative `ledger_ref`. `ingest` re-validates every line; a -clean ingest is a transparent copy, and the ingest gate records a stats record -of its own only when it refused, dropped, or failed something. - -With the environment variable unset the observer registers no hooks and -writes no files (feature-off parity). When set, `observer.ts` observes -`agent/session-start`, `agent/status`, `agent/error`, `agent/pre-step` -(pass-through), `session/event`, and `session/disposed`; token-level -`assistant/chunk` events are not consumed, which the receipt shows through -`event_kinds_consumed`. +prints only the relative `ledger_ref`. `ingest` re-validates every line. A +clean ingest is a transparent copy; any malformed or rejected input appends a +durable `reliability_ingest_violation_v0` marker, making subsequent receipts +`invalid` instead of losing the failed gate at process exit. + +Unless all three required variables are valid, the observer row registers no +hooks and writes no files (feature-off parity). When enabled, `observer.ts` +observes only `session/created`, `session/event`, and `session/disposed`. +Events from any other session are rejected as `identity_invalid`, so they can +never be silently attributed to the configured goal. Token-level +`assistant/chunk` events are not consumed. ## Validation @@ -166,6 +181,7 @@ recovered error, and no stall. No dashboard surface, no L2 recommendations, no automatic recovery, no writeback into goals, todos, gates, or session runtime, and no change to the -`loopx status` first screen. The observer attributes every session in the DSH -process to the single declared goal; per-session binding discovery is a -follow-up that must not reuse the driver's LoopX CLI path. +`loopx status` first screen. The adapter requires an externally pinned +goal/session/run identity; it does not discover bindings through the Driver or +LoopX CLI. This prototype also does not provide matched native/L1 execution, +observer CPU/I/O/latency/storage measurement, or an eligible C1 run. diff --git a/loopx/capabilities/reliability_diagnostics/README.zh-CN.md b/loopx/capabilities/reliability_diagnostics/README.zh-CN.md index fe0b8ad434..fe2311a092 100644 --- a/loopx/capabilities/reliability_diagnostics/README.zh-CN.md +++ b/loopx/capabilities/reliability_diagnostics/README.zh-CN.md @@ -2,8 +2,10 @@ [English](README.md) | [RFC](../../../docs/architecture/rfcs/long-running-agent-reliability-diagnostics-governed-delivery-v0.md) -状态:实验能力、内置、默认关闭、goal-scoped。本包交付 reliability-diagnostics RFC -的 P0 切片:**L1 shadow observer** 合约,以及第一个真实事件源 DeepSeek Harness(DSH)。 +状态:实验能力、内置、默认关闭、按 goal 与 session 限定。本包实现 RFC 路线图 +**P0 阶段所描述的原型组件**:L1 shadow-observer 合约与第一个 DSH 事件源适配器。 +这里不宣称已经通过 P0 退出门槛;仍需 C0 adapter fidelity、一次合格的 C1 observer +实跑,以及明确的开销报告。 L1 observer 观察一个长时运行的 Agent 会话,并写下独立的诊断记录;它**永远不能** 影响该会话。本能力把这条承诺做成机器合约而不是口头规范:envelope schema 无法表达 @@ -19,8 +21,10 @@ flowchart LR O -. "没有 send / schedule / gate / tool / worker-state 通路" .-> H ``` -虚线边表示一条被断言不存在的路径:测试拒绝带控制字段的 envelope,TypeScript 模块 -不从 continuation driver 导入任何东西,一旦出现 outbound endpoint,receipt 即为 `invalid`。 +虚线边表示一条被断言不存在的路径:observer 通过独立 Cordis 插件入口装载,不注入 +Driver 或 Agent,只消费 session log 的发布事件,并且不进入 Driver 与包根 bundle。测试会 +拒绝带控制字段的 envelope;一旦出现 outbound endpoint、worker context 或 scheduler input +通路,receipt 即为 `invalid`。这里证明的是模块与 hook 隔离,不宣称 OS 进程隔离。 ## 放置理由 @@ -30,7 +34,8 @@ flowchart LR ledger 与 projection 是它的**同级**,绝不合并进去。id 与其它 catalog 条目一样使用 kebab-case;包目录为 `reliability_diagnostics`。 - **Provider id `dsh-session-events`**(origin `extension`)。由 npm 包 - `packages/dsh-loopx-plugin` 的 `src/observer.ts` 交付,与 `driver.ts` 物理分离。 + `packages/dsh-loopx-plugin` 的显式 `dsh-loopx-plugin/observer` 入口和独立 + `loopx-shadow-observer` Cordis 行交付,与 `driver.ts` 分离。 npm 插件没有 Python `extension.toml` 生命周期,因此由能力在 catalog entry 上声明该 provider,registry 报告 `declared=true`、`installed=enabled=ready=false`。先例是 `repository_change_window` 声明其 `git-hook` provider 的方式。 @@ -46,6 +51,7 @@ flowchart LR | `schema_version` | 字面量 | `reliability_observer_envelope_v0` | | `capability_id` | 字面量 | `reliability-diagnostics` | | `provider_id` | identity token | 例如 `dsh-session-events` | +| `observer_id` | identity token | 单个 observer 实例内稳定,并与 stats 关联 | | `goal_id`、`session_id` | identity token | `^[A-Za-z0-9][A-Za-z0-9_.:-]{0,120}$` | | `agent_id` | identity token,可选 | | | `sequence` | 整数 >= 0 | observer 分配、每会话单调;缺口计为丢失 | @@ -64,11 +70,14 @@ flowchart LR ### Observer stats(`reliability_observer_stats_v0`) -每个 observer 实现都会把它写在 envelope 旁边。字段:`observer_id`、`emitted_at`、 -`observed_event_count`、`accepted_event_count`、`rejected_event_count`、 -`rejected_by_reason`、`buffer_bound`、`backpressure_drop_count`、`observer_failure_count`、 -`outbound_endpoints`(必须为 `[]`)、`observation_entered_worker_context`(必须为 `false`)、 -`clock_source`。stats 按 observer 实例累计;receipt 取每个 `observer_id` 的最新记录并跨实例求和。 +每个 observer 实现都会把它写在 envelope 旁边。`run_identity` 固定 `worker_id`、 +`model_id`、`task_id`、`environment_id`、`tools_id`、`budget_id`、`adapter_revision`、 +`observer_revision`;同时声明 `event_sources` 和 `source_fields_consumed`,并记录时间、 +接受/拒绝计数、类型化拒绝原因、buffer 上限、丢弃、故障、峰值 buffer、flush 次数、 +时钟源、outbound endpoints,以及是否进入 worker context / scheduler inputs。计数必须满足 +`observed = accepted + rejected + dropped`,时间必须带时区,每个接受的 envelope 必须与 +provider/observer stats 精确关联。stats 按 observer 实例累计;receipt 取每个 +`observer_id` 的最新记录并跨实例求和。 ### Integrity receipt(`reliability_integrity_receipt_v0`) @@ -76,19 +85,20 @@ flowchart LR | --- | --- | | `status` | `valid`、`degraded`、`quarantined`、`invalid`(全覆盖、有序) | | `reason_codes` | 类型化列表;仅 `valid` 时为空 | -| `observed_event_count`、`session_count` | ledger 中被接受的 envelope | +| `observed_event_count`、`accepted_event_count`、`persisted_event_count` | 事件尝试数、stats 接受数、ledger 中关联的 envelope 数 | | `lost_event_count`、`duplicate_sequence_count` | 每会话 sequence 缺口与重复 | | `ledger_invalid_record_count` | ledger 中损坏或异类记录 | | `rejected_event_count`、`rejected_by_reason` | observer 报告的拒绝 | | `buffer_bound`、`backpressure_drop_count`、`observer_failure_count` | 有界失败证据 | | `clock.sources`、`clock.max_uncertainty_ms` | 声明的时钟;> 1000 ms 时降级 | -| `outbound_endpoints`、`observation_entered_worker_context` | 必须为 `[]` / `false` | -| `event_kinds_consumed`、`summary_fields_consumed` | 实际消费的事件源与字段 | +| `outbound_endpoints`、worker/scheduler influence flags | 必须为 `[]` / `false` / `false` | +| `run_identities`、`event_sources`、`source_fields_consumed` | 固定的 treatment identity 与适配器声明覆盖 | +| `event_kinds_consumed`、`summary_fields_consumed` | 实际持久化的事件种类与紧凑 summary 字段 | -状态规则:无观测、任一 outbound endpoint、或观测进入 worker context 时为 `invalid`; -否则 observer 故障、出现控制字段记录、或 ledger 含损坏记录时为 `quarantined`;否则 -事件丢失、被丢弃、重复、拒绝了原始材料、缺少 stats、或时钟不确定度超阈值时为 -`degraded`;否则为 `valid`。 +状态规则:无观测、stats 缺失或不能与持久化 envelope 精确关联、身份被拒绝、ledger +存在无效输入、任一 outbound endpoint,或观测进入 worker context / scheduler inputs 时 +为 `invalid`。否则 observer 故障或控制形态输入为 `quarantined`;事件缺口、丢弃、重复、 +原始/不支持字段或时钟不确定度超阈值为 `degraded`;其它情况才是 `valid`。 ### Diagnostic projection(`reliability_diagnostic_projection_v0`) @@ -106,8 +116,10 @@ flowchart LR ## 使用方式 ```bash -# 只为一个 goal 启用 DSH provider,然后照常启动 DSH。 +# 只为一个预先声明的 goal 和一个准确的 DSH session 启用 provider。 export LOOPX_DSH_SHADOW_OBSERVER_GOAL_ID= +export LOOPX_DSH_SHADOW_OBSERVER_SESSION_ID= +export LOOPX_DSH_SHADOW_OBSERVER_RUN_IDENTITY_JSON='{"worker_id":"","model_id":"","task_id":"","environment_id":"","tools_id":"","budget_id":"","adapter_revision":"","observer_revision":""}' # 可选:LOOPX_DSH_SHADOW_OBSERVER_LEDGER_DIR、LOOPX_DSH_SHADOW_OBSERVER_BUFFER_BOUND loopx reliability-diagnostics receipt --goal-id --format json @@ -117,12 +129,14 @@ loopx reliability-diagnostics ingest --goal-id --input observer.ndjso ledger 位于 `/reliability_diagnostics/.ndjson`;默认 runtime root 与 LoopX 其它部分一致,CLI 只打印相对的 `ledger_ref`。`ingest` 会重新校验每一行;干净的 -ingest 是透明拷贝,只有当 ingest 门拒绝、丢弃或失败了某些记录时才会写入自己的 stats 记录。 +ingest 是透明拷贝。任何损坏或被拒绝的输入都会追加持久化 +`reliability_ingest_violation_v0` 标记,让后续 receipt 成为 `invalid`,而不是在进程退出后 +丢失门禁失败。 -未设置环境变量时,observer 不注册任何 hook、不写任何文件(feature-off parity)。 -设置后,`observer.ts` 观察 `agent/session-start`、`agent/status`、`agent/error`、 -`agent/pre-step`(透传)、`session/event`、`session/disposed`;token 级的 -`assistant/chunk` 不被消费,receipt 通过 `event_kinds_consumed` 让这一点可见。 +三个必需变量未全部有效时,observer 行不注册任何 hook、不写任何文件(feature-off +parity)。启用后,`observer.ts` 只观察 `session/created`、`session/event`、 +`session/disposed`。其它 session 的事件一律以 `identity_invalid` 拒绝,因此不会静默归属 +到配置的 goal。token 级 `assistant/chunk` 不被消费。 ## 验证 @@ -140,5 +154,6 @@ projection 报告 `read` 上的重复、一次已恢复的错误、无 stall。 ## 本切片的非目标 不做 dashboard、不做 L2 建议、不做自动恢复、不回写 goal / todo / gate / session runtime、 -不改 `loopx status` 首屏。observer 把 DSH 进程内所有会话归属到唯一声明的 goal;按会话的 -绑定发现是后续工作,且不得复用 driver 的 LoopX CLI 通路。 +不改 `loopx status` 首屏。适配器要求外部固定 goal/session/run identity;不会通过 Driver +或 LoopX CLI 发现绑定。本原型也不提供 matched native/L1 执行、observer CPU/I/O/延迟/ +存储开销测量,或一次合格 C1 实跑。 diff --git a/packages/dsh-loopx-plugin/README.md b/packages/dsh-loopx-plugin/README.md index 77b7347df4..5b327ed5b2 100644 --- a/packages/dsh-loopx-plugin/README.md +++ b/packages/dsh-loopx-plugin/README.md @@ -131,34 +131,46 @@ never opens a browser or configures a model provider. ## Shadow observer (default off) -`src/observer.ts` is the `dsh-session-events` provider for the LoopX +`src/observer.ts`, exported as `dsh-loopx-plugin/observer`, is the +`dsh-session-events` provider for the LoopX [Reliability Diagnostics](../../loopx/capabilities/reliability_diagnostics/README.md) capability: an L1 shadow observer that consumes read-only harness events and appends compact, public-safe envelopes plus an observer stats record to `/reliability_diagnostics/.ndjson`. It is a -separate module from the Driver with no shared send path: it never calls -`agent.send`, touches the inbox, invokes the LoopX CLI, schedules, retries, -stops, or resumes anything. Every hook body and every flush is isolated, so an -observer failure is counted into the receipt instead of reaching DSH. +separate Cordis row and bundle from the Driver, with no Driver or Agent +injection and no shared send path. It never calls `agent.send`, touches the +inbox, invokes the LoopX CLI, schedules, retries, stops, or resumes anything. +Every hook body and every flush is isolated, so an observer failure is counted +into the receipt instead of reaching DSH. This is module and hook isolation, +not an OS-process-isolation claim. -It is off unless one exact goal is declared before DSH starts: +It is off unless one exact goal, DSH session, and complete run identity are +declared before DSH starts: ```bash export LOOPX_DSH_SHADOW_OBSERVER_GOAL_ID= +export LOOPX_DSH_SHADOW_OBSERVER_SESSION_ID= +export LOOPX_DSH_SHADOW_OBSERVER_RUN_IDENTITY_JSON='{"worker_id":"","model_id":"","task_id":"","environment_id":"","tools_id":"","budget_id":"","adapter_revision":"","observer_revision":""}' # optional: LOOPX_DSH_SHADOW_OBSERVER_LEDGER_DIR, LOOPX_DSH_SHADOW_OBSERVER_BUFFER_BOUND (default 256) loopx reliability-diagnostics receipt --goal-id --format json loopx reliability-diagnostics status --goal-id --format json ``` -With the variable unset the Driver row registers no observer hook and writes no -file. When set, the observer consumes `agent/session-start`, `agent/status`, -`agent/error`, `agent/pre-step` (pass-through), `session/event`, and -`session/disposed`; it skips `assistant/chunk` and records tool names, turn and -step numbers, end reasons, and ids only, never arguments, outputs, prompts, or -paths. Sequence gaps, bounded-buffer drops, declared clock uncertainty, and -the always-empty outbound endpoint list make the run's admissibility as passive -evidence auditable from the receipt. In this slice every session in the DSH -process is attributed to the single declared goal. +Unless all required variables are valid, the independent observer row +registers no hook and writes no file. When enabled, it consumes only +`session/created`, `session/event`, and `session/disposed`; it skips +`assistant/chunk` and records tool names, turn and step numbers, typed end +reasons, and ids only, never arguments, outputs, prompts, or paths. Events for +any session other than the exact configured session are rejected as +`identity_invalid`. The stats record pins worker/model/task/environment/tools/ +budget plus adapter and observer revisions, declares source coverage, and +proves count conservation. Sequence gaps, bounded-buffer drops, flush attempts, +declared clock uncertainty, and the empty outbound and influence fields make +the run's admissibility auditable from the receipt. + +This is an experimental adapter implementing the RFC's P0 prototype +components. It does not establish the RFC's P0 exit: C0 fidelity, a qualifying +C1 run, and measured observer overhead remain separate evidence gates. ## GoalBar authority and privacy boundary From d41630c84b8a5052a6bb0f40c5f34424ce53a96b Mon Sep 17 00:00:00 2001 From: song Date: Sat, 5 Sep 2026 00:40:11 +0800 Subject: [PATCH 09/13] fix(dsh-loopx-plugin): serialize observer flush completion Signed-off-by: song --- packages/dsh-loopx-plugin/src/observer.ts | 64 ++++++++++++------- .../dsh-loopx-plugin/tests/observer.spec.ts | 53 +++++++++++++++ 2 files changed, 95 insertions(+), 22 deletions(-) diff --git a/packages/dsh-loopx-plugin/src/observer.ts b/packages/dsh-loopx-plugin/src/observer.ts index 4d978129ad..dd6c7f94b0 100644 --- a/packages/dsh-loopx-plugin/src/observer.ts +++ b/packages/dsh-loopx-plugin/src/observer.ts @@ -392,32 +392,42 @@ export class ShadowObserver { /** Write buffered envelopes plus a stats record; never rejects. */ async flush(): Promise { - if (this.flushing !== undefined) { - this.flushRequested = true - await this.flushing - return + this.flushRequested = true + let operation = this.flushing + if (operation === undefined) { + operation = this.flushRequestedBatches() + this.flushing = operation } - const taken = this.buffer - this.buffer = [] - this.flushAttemptCount += 1 - const lines = [...taken, this.stats()].map(record => JSON.stringify(record)) - this.flushing = this.appendLines(this.path, lines).then( - () => undefined, - (error: unknown) => { - this.observerFailureCount += 1 - this.acceptedEventCount -= taken.length - this.backpressureDropCount += taken.length - this.warn(`dsh-loopx shadow observer flush failed: ${error instanceof Error ? error.name : 'unknown'}`) - }, - ) + await operation + } + + private async flushRequestedBatches(): Promise { try { - await this.flushing + while (this.flushRequested) { + this.flushRequested = false + await this.flushBatch() + } } finally { + // Clear ownership before resolving so a later caller either joins this + // drain or starts the next one; it can never observe a resolved owner. this.flushing = undefined } - if (this.flushRequested) { - this.flushRequested = false - await this.flush() + } + + private async flushBatch(): Promise { + const taken = this.buffer + this.buffer = [] + this.flushAttemptCount += 1 + try { + const lines = [...taken, this.stats()].map(record => JSON.stringify(record)) + await this.appendLines(this.path, lines) + } catch (error: unknown) { + this.observerFailureCount += 1 + this.acceptedEventCount -= taken.length + this.backpressureDropCount += taken.length + this.safeWarn( + `dsh-loopx shadow observer flush failed: ${error instanceof Error ? error.name : 'unknown'}`, + ) } } @@ -436,7 +446,9 @@ export class ShadowObserver { this.observerFailureCount += 1 if (this.observedEventCount === observedBefore) this.observedEventCount += 1 this.reject('observer_internal_failure') - this.warn(`dsh-loopx shadow observer hook failed: ${error instanceof Error ? error.name : 'unknown'}`) + this.safeWarn( + `dsh-loopx shadow observer hook failed: ${error instanceof Error ? error.name : 'unknown'}`, + ) } } @@ -501,6 +513,14 @@ export class ShadowObserver { return true } + private safeWarn(message: string): void { + try { + this.warn(message) + } catch { + // Logging is observational too; it must never reach the worker. + } + } + private requestFlush(): void { void this.flush() } diff --git a/packages/dsh-loopx-plugin/tests/observer.spec.ts b/packages/dsh-loopx-plugin/tests/observer.spec.ts index 834d2a26c3..36b4610723 100644 --- a/packages/dsh-loopx-plugin/tests/observer.spec.ts +++ b/packages/dsh-loopx-plugin/tests/observer.spec.ts @@ -228,6 +228,46 @@ describe('shadow observer envelopes', () => { expect(envelopes.map(item => item.sequence)).toEqual([0, 1, 2, 3]) }) + it('waits for the catch-up batch requested during an in-flight flush', async () => { + let releaseFirst = () => {} + let releaseSecond = () => {} + let markSecondStarted = () => {} + const firstGate = new Promise(resolve => { releaseFirst = resolve }) + const secondGate = new Promise(resolve => { releaseSecond = resolve }) + const secondStarted = new Promise(resolve => { markSecondStarted = resolve }) + const appended: string[][] = [] + const observer = new ShadowObserver({ + config: { ...config, bufferBound: 1 }, + now: () => 1_756_728_000_000, + observerId: 'observer-fixture', + appendLines: async (_path, lines) => { + appended.push([...lines]) + if (appended.length === 1) await firstGate + if (appended.length === 2) { + markSecondStarted() + await secondGate + } + }, + }) + const session = fakeSession() + observer.observeSessionEvent(session, sessionEvent('step/start', 1, 1_756_728_001_000, { turn: 1, step: 1 })) + observer.observeSessionEvent(session, sessionEvent('step/start', 2, 1_756_728_002_000, { turn: 1, step: 2 })) + + let completed = false + const completion = observer.flush().then(() => { completed = true }) + releaseFirst() + await secondStarted + expect(completed).toBe(false) + releaseSecond() + await completion + + const envelopes = parsed(appended).filter( + (record): record is ObserverEnvelope => record.schema_version === OBSERVER_ENVELOPE_SCHEMA_VERSION, + ) + expect(envelopes.map(item => item.sequence)).toEqual([0, 1]) + expect(observer.stats().flush_attempt_count).toBe(2) + }) + it('preserves the typed DSH turn-end reason used by recovery diagnostics', async () => { const { appended, observer } = observerWithCapture() observer.observeSessionEvent( @@ -261,6 +301,19 @@ describe('shadow observer envelopes', () => { ) }) + it('contains logger failures while reporting an observer failure', async () => { + const observer = new ShadowObserver({ + config, + observerId: 'observer-fixture', + appendLines: async () => { throw new Error('disk full') }, + warn: () => { throw new Error('logger unavailable') }, + }) + observer.observeSessionCreated(fakeSession()) + await expect(observer.flush()).resolves.toBeUndefined() + expect(observer.stats().observer_failure_count).toBe(1) + expect(observer.stats().backpressure_drop_count).toBe(1) + }) + it('rejects rather than attributing another session to the configured goal', () => { const { observer } = observerWithCapture() observer.observeSessionCreated(fakeSession('different-session')) From 723da990e6393e9094f6be12a8c7ab05a26af4df Mon Sep 17 00:00:00 2001 From: song Date: Sat, 5 Sep 2026 01:15:51 +0800 Subject: [PATCH 10/13] refactor(reliability-diagnostics): satisfy static quality gates Signed-off-by: song --- .../reliability_diagnostics/intake.py | 178 +++++++--- .../reliability_diagnostics/projection.py | 229 ++++++++---- .../reliability_diagnostics/receipt.py | 332 ++++++++++++------ loopx/cli_commands/reliability_diagnostics.py | 145 +++++--- packages/dsh-loopx-plugin/src/observer.ts | 10 +- .../test_reliability_diagnostics.py | 227 +++++++++--- 6 files changed, 787 insertions(+), 334 deletions(-) diff --git a/loopx/capabilities/reliability_diagnostics/intake.py b/loopx/capabilities/reliability_diagnostics/intake.py index 45b9a1960f..062fc2cce6 100644 --- a/loopx/capabilities/reliability_diagnostics/intake.py +++ b/loopx/capabilities/reliability_diagnostics/intake.py @@ -157,113 +157,183 @@ def _token_list(value: Any, *, name: str) -> tuple[str, ...]: if ( not isinstance(value, list) or not value - or any(not isinstance(item, str) or not _ENDPOINT_PATTERN.match(item) for item in value) + or any( + not isinstance(item, str) or not _ENDPOINT_PATTERN.match(item) + for item in value + ) ): - raise ValueError(f"observer stats {name} must be a non-empty list of compact tokens") + raise ValueError( + f"observer stats {name} must be a non-empty list of compact tokens" + ) if len(set(value)) != len(value): raise ValueError(f"observer stats {name} must not contain duplicates") return tuple(value) -def normalize_observer_run_identity(value: Any) -> ObserverRunIdentity: - if not isinstance(value, Mapping) or set(value) != RUN_IDENTITY_FIELDS: - raise ValueError( - "observer stats run_identity must contain exactly the pinned identity fields" - ) - return ObserverRunIdentity( - **{ - field_name: _identity_token( - value.get(field_name), name=f"run_identity.{field_name}" - ) - for field_name in RUN_IDENTITY_FIELDS - } - ) - +@dataclass(frozen=True) +class _ValidatedStatsCounts: + observed: int + accepted: int + rejected: int + dropped: int + buffer_bound: int + peak_buffered: int -def normalize_observer_stats(record: Mapping[str, Any]) -> ObserverStats: - """Validate one stats record written by any observer implementation.""" +def _stats_identity(record: Mapping[str, Any]) -> tuple[str, str, str]: if not isinstance(record, Mapping): raise ValueError("observer stats must be an object") unknown = sorted(str(key) for key in record if str(key) not in STATS_FIELDS) if unknown: raise ValueError(f"observer stats carry unsupported fields: {unknown}") if record.get("schema_version") != OBSERVER_STATS_SCHEMA_VERSION: - raise ValueError(f"observer stats schema must be {OBSERVER_STATS_SCHEMA_VERSION}") + raise ValueError( + f"observer stats schema must be {OBSERVER_STATS_SCHEMA_VERSION}" + ) if record.get("capability_id") != CAPABILITY_ID: raise ValueError(f"observer stats capability must be {CAPABILITY_ID}") - for key in ("provider_id", "observer_id", "goal_id"): - _identity_token(record.get(key), name=key) + return ( + _identity_token(record.get("provider_id"), name="provider_id"), + _identity_token(record.get("observer_id"), name="observer_id"), + _identity_token(record.get("goal_id"), name="goal_id"), + ) + + +def _stats_emitted_at(record: Mapping[str, Any]) -> str: emitted_at = record.get("emitted_at") if not isinstance(emitted_at, str): - raise ValueError("observer stats emitted_at must be timezone-aware ISO-8601 text") + raise ValueError( + "observer stats emitted_at must be timezone-aware ISO-8601 text" + ) try: parsed_emitted_at = datetime.fromisoformat(emitted_at.replace("Z", "+00:00")) except ValueError as exc: - raise ValueError("observer stats emitted_at must be timezone-aware ISO-8601 text") from exc + raise ValueError( + "observer stats emitted_at must be timezone-aware ISO-8601 text" + ) from exc if parsed_emitted_at.tzinfo is None: raise ValueError("observer stats emitted_at must carry a timezone") + return emitted_at + + +def _stats_rejection_counts(record: Mapping[str, Any]) -> dict[str, int]: reasons = record.get("rejected_by_reason") or {} if not isinstance(reasons, Mapping): raise ValueError("observer stats rejected_by_reason must be an object") - normalized_reasons = { - EnvelopeRejection(str(key)).value: _count(value, name=f"rejected_by_reason.{key}") + return { + EnvelopeRejection(str(key)).value: _count( + value, name=f"rejected_by_reason.{key}" + ) for key, value in reasons.items() } + + +def _stats_endpoints(record: Mapping[str, Any]) -> tuple[str, ...]: endpoints = record.get("outbound_endpoints") if not isinstance(endpoints, list) or any( not isinstance(item, str) or not _ENDPOINT_PATTERN.match(item) for item in endpoints ): - raise ValueError("observer stats outbound_endpoints must be a list of endpoint ids") - entered = record.get("observation_entered_worker_context") - if not isinstance(entered, bool): - raise ValueError("observer stats observation_entered_worker_context must be boolean") - entered_scheduler = record.get("observation_entered_scheduler_inputs") - if not isinstance(entered_scheduler, bool): raise ValueError( - "observer stats observation_entered_scheduler_inputs must be boolean" + "observer stats outbound_endpoints must be a list of endpoint ids" ) + return tuple(endpoints) + + +def _stats_boolean(record: Mapping[str, Any], name: str) -> bool: + value = record.get(name) + if not isinstance(value, bool): + raise ValueError(f"observer stats {name} must be boolean") + return value + + +def _stats_counts( + record: Mapping[str, Any], + rejected_by_reason: Mapping[str, int], +) -> _ValidatedStatsCounts: buffer_bound = _count(record.get("buffer_bound"), name="buffer_bound") if not 1 <= buffer_bound <= MAX_BUFFER_BOUND: - raise ValueError(f"observer stats buffer_bound must be within 1..{MAX_BUFFER_BOUND}") - observed = _count(record.get("observed_event_count"), name="observed_event_count") - accepted = _count(record.get("accepted_event_count"), name="accepted_event_count") - rejected = _count(record.get("rejected_event_count"), name="rejected_event_count") - dropped = _count(record.get("backpressure_drop_count"), name="backpressure_drop_count") - if rejected != sum(normalized_reasons.values()): + raise ValueError( + f"observer stats buffer_bound must be within 1..{MAX_BUFFER_BOUND}" + ) + counts = _ValidatedStatsCounts( + observed=_count( + record.get("observed_event_count"), name="observed_event_count" + ), + accepted=_count( + record.get("accepted_event_count"), name="accepted_event_count" + ), + rejected=_count( + record.get("rejected_event_count"), name="rejected_event_count" + ), + dropped=_count( + record.get("backpressure_drop_count"), name="backpressure_drop_count" + ), + buffer_bound=buffer_bound, + peak_buffered=_count( + record.get("peak_buffered_event_count"), name="peak_buffered_event_count" + ), + ) + if counts.rejected != sum(rejected_by_reason.values()): raise ValueError( "observer stats rejected_event_count must equal rejected_by_reason" ) - if observed != accepted + rejected + dropped: + if counts.observed != counts.accepted + counts.rejected + counts.dropped: raise ValueError( "observer stats observed_event_count must equal accepted + rejected + dropped" ) - peak_buffered = _count( - record.get("peak_buffered_event_count"), name="peak_buffered_event_count" + if counts.peak_buffered > counts.buffer_bound: + raise ValueError( + "observer stats peak_buffered_event_count exceeds buffer_bound" + ) + return counts + + +def normalize_observer_run_identity(value: Any) -> ObserverRunIdentity: + if not isinstance(value, Mapping) or set(value) != RUN_IDENTITY_FIELDS: + raise ValueError( + "observer stats run_identity must contain exactly the pinned identity fields" + ) + return ObserverRunIdentity( + **{ + field_name: _identity_token( + value.get(field_name), name=f"run_identity.{field_name}" + ) + for field_name in RUN_IDENTITY_FIELDS + } ) - if peak_buffered > buffer_bound: - raise ValueError("observer stats peak_buffered_event_count exceeds buffer_bound") + + +def normalize_observer_stats(record: Mapping[str, Any]) -> ObserverStats: + """Validate one stats record written by any observer implementation.""" + + provider_id, observer_id, goal_id = _stats_identity(record) + emitted_at = _stats_emitted_at(record) + normalized_reasons = _stats_rejection_counts(record) + endpoints = _stats_endpoints(record) + entered = _stats_boolean(record, "observation_entered_worker_context") + entered_scheduler = _stats_boolean(record, "observation_entered_scheduler_inputs") + counts = _stats_counts(record, normalized_reasons) return ObserverStats( - provider_id=str(record["provider_id"]), - observer_id=str(record["observer_id"]), - goal_id=str(record["goal_id"]), + provider_id=provider_id, + observer_id=observer_id, + goal_id=goal_id, run_identity=normalize_observer_run_identity(record.get("run_identity")), event_sources=_token_list(record.get("event_sources"), name="event_sources"), source_fields_consumed=_token_list( record.get("source_fields_consumed"), name="source_fields_consumed" ), emitted_at=emitted_at, - observed_event_count=observed, - accepted_event_count=accepted, - rejected_event_count=rejected, + observed_event_count=counts.observed, + accepted_event_count=counts.accepted, + rejected_event_count=counts.rejected, rejected_by_reason=normalized_reasons, - buffer_bound=buffer_bound, - backpressure_drop_count=dropped, + buffer_bound=counts.buffer_bound, + backpressure_drop_count=counts.dropped, observer_failure_count=_count( record.get("observer_failure_count"), name="observer_failure_count" ), - peak_buffered_event_count=peak_buffered, + peak_buffered_event_count=counts.peak_buffered, flush_attempt_count=_count( record.get("flush_attempt_count"), name="flush_attempt_count" ), @@ -306,9 +376,7 @@ def __post_init__(self) -> None: raise ValueError("run_identity must be an ObserverRunIdentity") normalize_observer_run_identity(self.run_identity.as_dict()) _token_list(list(self.event_sources), name="event_sources") - _token_list( - list(self.source_fields_consumed), name="source_fields_consumed" - ) + _token_list(list(self.source_fields_consumed), name="source_fields_consumed") @property def buffered_count(self) -> int: diff --git a/loopx/capabilities/reliability_diagnostics/projection.py b/loopx/capabilities/reliability_diagnostics/projection.py index c6b569b0d3..311b5dcb7a 100644 --- a/loopx/capabilities/reliability_diagnostics/projection.py +++ b/loopx/capabilities/reliability_diagnostics/projection.py @@ -8,16 +8,24 @@ from __future__ import annotations +from dataclasses import dataclass, field from enum import StrEnum from typing import Any -from .envelope import CAPABILITY_ID, ObserverEnvelope, ObserverEventKind, parse_observed_at +from .envelope import ( + CAPABILITY_ID, + ObserverEnvelope, + ObserverEventKind, + parse_observed_at, +) from .receipt import LedgerReading, build_integrity_receipt DIAGNOSTIC_PROJECTION_SCHEMA_VERSION = "reliability_diagnostic_projection_v0" DEFAULT_STALL_THRESHOLD_MS = 300_000 DEFAULT_REPETITION_THRESHOLD = 3 -_TERMINAL_ERROR_REASONS = frozenset({"error", "failed", "failure", "aborted", "cancelled", "canceled", "timeout"}) +_TERMINAL_ERROR_REASONS = frozenset( + {"error", "failed", "failure", "aborted", "cancelled", "canceled", "timeout"} +) class DiagnosticStage(StrEnum): @@ -55,14 +63,125 @@ def _stage_after(envelope: ObserverEnvelope) -> DiagnosticStage: if kind is ObserverEventKind.SESSION_STARTED: return DiagnosticStage.IDLE if kind is ObserverEventKind.AGENT_STATUS: - return DiagnosticStage.RUNNING if envelope.summary.get("status") == "running" else DiagnosticStage.IDLE + return ( + DiagnosticStage.RUNNING + if envelope.summary.get("status") == "running" + else DiagnosticStage.IDLE + ) if kind is ObserverEventKind.UNSUPPORTED: return DiagnosticStage.UNKNOWN return DiagnosticStage.RUNNING def _ms_between(earlier: str, later: str) -> int: - return int((parse_observed_at(later) - parse_observed_at(earlier)).total_seconds() * 1000) + return int( + (parse_observed_at(later) - parse_observed_at(earlier)).total_seconds() * 1000 + ) + + +@dataclass +class _ProjectionAccumulator: + counts: dict[str, int] = field( + default_factory=lambda: { + "turns_started": 0, + "turns_ended": 0, + "steps": 0, + "tool_calls": 0, + "errors": 0, + } + ) + stage: DiagnosticStage = DiagnosticStage.UNKNOWN + max_gap_ms: int = 0 + longest_run: int = 0 + longest_run_tool: str | None = None + current_run: int = 0 + current_tool: str | None = None + unrecovered_errors: int = 0 + recovered_errors: int = 0 + previous: ObserverEnvelope | None = None + + def consume(self, envelope: ObserverEnvelope) -> None: + if self.previous is not None: + self.max_gap_ms = max( + self.max_gap_ms, + _ms_between(self.previous.observed_at, envelope.observed_at), + ) + self._track_progress(envelope) + self._track_repetition(envelope) + self.stage = _stage_after(envelope) + self.previous = envelope + + def _track_progress(self, envelope: ObserverEnvelope) -> None: + kind = envelope.event_kind + if kind is ObserverEventKind.TURN_STARTED: + self.counts["turns_started"] += 1 + return + if kind is ObserverEventKind.TURN_ENDED: + self.counts["turns_ended"] += 1 + self._track_turn_end(envelope) + return + if kind is ObserverEventKind.STEP_ENDED: + self.counts["steps"] += 1 + self._recover_errors() + return + if kind is ObserverEventKind.AGENT_ERROR: + self.counts["errors"] += 1 + self.unrecovered_errors += 1 + + def _track_turn_end(self, envelope: ObserverEnvelope) -> None: + terminal_error = ( + str(envelope.summary.get("reason", "")) in _TERMINAL_ERROR_REASONS + ) + if terminal_error: + if not self.unrecovered_errors: + self.counts["errors"] += 1 + self.unrecovered_errors = 1 + return + self._recover_errors() + + def _recover_errors(self) -> None: + if not self.unrecovered_errors: + return + self.recovered_errors += self.unrecovered_errors + self.unrecovered_errors = 0 + + def _track_repetition(self, envelope: ObserverEnvelope) -> None: + kind = envelope.event_kind + if kind is ObserverEventKind.TOOL_CALLED: + self.counts["tool_calls"] += 1 + tool = str(envelope.summary.get("tool_name", "")) + self.current_run = self.current_run + 1 if tool == self.current_tool else 1 + self.current_tool = tool + if self.current_run > self.longest_run: + self.longest_run = self.current_run + self.longest_run_tool = tool or None + return + if kind not in { + ObserverEventKind.TOOL_COMPLETED, + ObserverEventKind.AGENT_PRE_STEP, + }: + self.current_run = 0 + self.current_tool = None + + +def _diagnostic_signals( + *, + stall_detected: bool, + repetition_detected: bool, + unrecovered_errors: int, + receipt: dict[str, Any], +) -> list[str]: + candidates = ( + (stall_detected, DiagnosticSignal.STALL_SUSPECTED), + (repetition_detected, DiagnosticSignal.REPETITION_SUSPECTED), + (bool(unrecovered_errors), DiagnosticSignal.UNRECOVERED_ERROR), + ( + bool(receipt["lost_event_count"] or receipt["backpressure_drop_count"]), + DiagnosticSignal.EVENT_LOSS, + ), + (receipt["status"] != "valid", DiagnosticSignal.INTEGRITY_NOT_VALID), + ) + return [signal.value for applies, signal in candidates if applies] def build_diagnostic_projection( @@ -75,70 +194,27 @@ def build_diagnostic_projection( receipt = build_integrity_receipt(reading) envelopes = reading.ordered_envelopes - counts = {"turns_started": 0, "turns_ended": 0, "steps": 0, "tool_calls": 0, "errors": 0} - stage = DiagnosticStage.UNKNOWN - max_gap_ms = 0 - longest_run = 0 - longest_run_tool: str | None = None - current_run = 0 - current_tool: str | None = None - unrecovered_errors = 0 - recovered_errors = 0 - previous: ObserverEnvelope | None = None + state = _ProjectionAccumulator() for envelope in envelopes: - kind = envelope.event_kind - if previous is not None: - max_gap_ms = max(max_gap_ms, _ms_between(previous.observed_at, envelope.observed_at)) - if kind is ObserverEventKind.TURN_STARTED: - counts["turns_started"] += 1 - elif kind is ObserverEventKind.TURN_ENDED: - counts["turns_ended"] += 1 - terminal_error = ( - str(envelope.summary.get("reason", "")) in _TERMINAL_ERROR_REASONS - ) - if terminal_error and not unrecovered_errors: - counts["errors"] += 1 - unrecovered_errors = 1 - elif unrecovered_errors and not terminal_error: - recovered_errors += unrecovered_errors - unrecovered_errors = 0 - elif kind is ObserverEventKind.STEP_ENDED: - counts["steps"] += 1 - if unrecovered_errors: - recovered_errors += unrecovered_errors - unrecovered_errors = 0 - elif kind is ObserverEventKind.AGENT_ERROR: - counts["errors"] += 1 - unrecovered_errors += 1 - if kind is ObserverEventKind.TOOL_CALLED: - counts["tool_calls"] += 1 - tool = str(envelope.summary.get("tool_name", "")) - current_run = current_run + 1 if tool == current_tool else 1 - current_tool = tool - if current_run > longest_run: - longest_run, longest_run_tool = current_run, tool or None - elif kind not in {ObserverEventKind.TOOL_COMPLETED, ObserverEventKind.AGENT_PRE_STEP}: - current_run, current_tool = 0, None - stage = _stage_after(envelope) - previous = envelope - - last_observed_at = previous.observed_at if previous else None + state.consume(envelope) + + last_observed_at = state.previous.observed_at if state.previous else None effective_as_of = as_of or last_observed_at - last_event_age_ms = _ms_between(last_observed_at, effective_as_of) if last_observed_at and effective_as_of else 0 - stall_detected = stage in _ACTIVE_STAGES and last_event_age_ms >= stall_threshold_ms - repetition_detected = longest_run >= repetition_threshold - - signals: list[str] = [] - if stall_detected: - signals.append(DiagnosticSignal.STALL_SUSPECTED.value) - if repetition_detected: - signals.append(DiagnosticSignal.REPETITION_SUSPECTED.value) - if unrecovered_errors: - signals.append(DiagnosticSignal.UNRECOVERED_ERROR.value) - if receipt["lost_event_count"] or receipt["backpressure_drop_count"]: - signals.append(DiagnosticSignal.EVENT_LOSS.value) - if receipt["status"] != "valid": - signals.append(DiagnosticSignal.INTEGRITY_NOT_VALID.value) + last_event_age_ms = ( + _ms_between(last_observed_at, effective_as_of) + if last_observed_at and effective_as_of + else 0 + ) + stall_detected = ( + state.stage in _ACTIVE_STAGES and last_event_age_ms >= stall_threshold_ms + ) + repetition_detected = state.longest_run >= repetition_threshold + signals = _diagnostic_signals( + stall_detected=stall_detected, + repetition_detected=repetition_detected, + unrecovered_errors=state.unrecovered_errors, + receipt=receipt, + ) return { "schema_version": DIAGNOSTIC_PROJECTION_SCHEMA_VERSION, @@ -149,27 +225,30 @@ def build_diagnostic_projection( "write_scope": "diagnostic_ledger_only", "worker_influence": "none", "provider_ids": receipt["provider_ids"], - "stage": stage.value, - "counts": counts, + "stage": state.stage.value, + "counts": state.counts, "stall": { "detected": stall_detected, "threshold_ms": stall_threshold_ms, "last_event_age_ms": last_event_age_ms, - "max_inter_event_gap_ms": max_gap_ms, + "max_inter_event_gap_ms": state.max_gap_ms, }, "repetition": { "detected": repetition_detected, "threshold": repetition_threshold, - "longest_tool_run": longest_run, - "tool_name": longest_run_tool, + "longest_tool_run": state.longest_run, + "tool_name": state.longest_run_tool, }, "recovery": { - "error_count": counts["errors"], - "recovered_error_count": recovered_errors, - "unrecovered_error_count": unrecovered_errors, + "error_count": state.counts["errors"], + "recovered_error_count": state.recovered_errors, + "unrecovered_error_count": state.unrecovered_errors, }, "signals": signals, - "integrity": {"status": receipt["status"], "reason_codes": receipt["reason_codes"]}, + "integrity": { + "status": receipt["status"], + "reason_codes": receipt["reason_codes"], + }, "evidence": { "observed_event_count": receipt["observed_event_count"], "lost_event_count": receipt["lost_event_count"], diff --git a/loopx/capabilities/reliability_diagnostics/receipt.py b/loopx/capabilities/reliability_diagnostics/receipt.py index 49b45b146b..184449a492 100644 --- a/loopx/capabilities/reliability_diagnostics/receipt.py +++ b/loopx/capabilities/reliability_diagnostics/receipt.py @@ -86,39 +86,61 @@ class LedgerReading: @property def ordered_envelopes(self) -> list[ObserverEnvelope]: - return sorted(self.envelopes, key=lambda item: (item.observed_at, item.session_id, item.sequence)) + return sorted( + self.envelopes, + key=lambda item: (item.observed_at, item.session_id, item.sequence), + ) -def read_ledger(records: Iterable[Any], *, goal_id: str, malformed_line_count: int = 0) -> LedgerReading: +def _append_envelope_record( + reading: LedgerReading, + record: Mapping[str, Any], +) -> None: + envelope = normalize_observer_envelope(record) + if envelope.goal_id != reading.goal_id: + raise ObserverEnvelopeError( + EnvelopeRejection.IDENTITY_INVALID, "goal_id does not match ledger" + ) + reading.envelopes.append(envelope) + + +def _store_stats_record( + reading: LedgerReading, + record: Mapping[str, Any], +) -> None: + stats = normalize_observer_stats(record) + if stats.goal_id != reading.goal_id: + raise ValueError("goal_id does not match ledger") + previous = reading.stats.get(stats.observer_id) + if previous is not None and ( + previous.provider_id != stats.provider_id + or previous.run_identity != stats.run_identity + ): + raise ValueError("observer stats identity changed within one ledger") + reading.stats[stats.observer_id] = stats + + +def _read_ledger_record(reading: LedgerReading, record: Mapping[str, Any]) -> None: + schema = record.get("schema_version") + if schema == OBSERVER_ENVELOPE_SCHEMA_VERSION: + _append_envelope_record(reading, record) + return + if schema == OBSERVER_STATS_SCHEMA_VERSION: + _store_stats_record(reading, record) + return + raise ValueError("unknown ledger record schema") + + +def read_ledger( + records: Iterable[Any], *, goal_id: str, malformed_line_count: int = 0 +) -> LedgerReading: reading = LedgerReading(goal_id=goal_id, invalid_record_count=malformed_line_count) for record in records: if not isinstance(record, Mapping): reading.invalid_record_count += 1 continue - schema = record.get("schema_version") try: - if schema == OBSERVER_ENVELOPE_SCHEMA_VERSION: - envelope = normalize_observer_envelope(record) - if envelope.goal_id != goal_id: - raise ObserverEnvelopeError( - EnvelopeRejection.IDENTITY_INVALID, "goal_id does not match ledger" - ) - reading.envelopes.append(envelope) - elif schema == OBSERVER_STATS_SCHEMA_VERSION: - stats = normalize_observer_stats(record) - if stats.goal_id != goal_id: - raise ValueError("goal_id does not match ledger") - # Stats are cumulative per observer instance; the latest wins, - # but one observer id may never change its identity mid-ledger. - previous = reading.stats.get(stats.observer_id) - if previous is not None and ( - previous.provider_id != stats.provider_id - or previous.run_identity != stats.run_identity - ): - raise ValueError("observer stats identity changed within one ledger") - reading.stats[stats.observer_id] = stats - else: - raise ValueError("unknown ledger record schema") + _read_ledger_record(reading, record) except ValueError: reading.invalid_record_count += 1 return reading @@ -142,92 +164,166 @@ def _sequence_accounting(envelopes: Iterable[ObserverEnvelope]) -> tuple[int, in return lost, duplicates -def build_integrity_receipt( - reading: LedgerReading, - *, - clock_uncertainty_degraded_ms: int = DEFAULT_CLOCK_UNCERTAINTY_DEGRADED_MS, -) -> dict[str, Any]: - envelopes = reading.ordered_envelopes - stats = list(reading.stats.values()) - lost, duplicates = _sequence_accounting(envelopes) - rejected_by_reason: dict[str, int] = defaultdict(int) - for item in stats: - for reason, count in item.rejected_by_reason.items(): - rejected_by_reason[reason] += count - outbound_endpoints = sorted({endpoint for item in stats for endpoint in item.outbound_endpoints}) - entered_worker_context = any(item.observation_entered_worker_context for item in stats) - entered_scheduler_inputs = any( - item.observation_entered_scheduler_inputs for item in stats - ) - observer_failures = sum(item.observer_failure_count for item in stats) - backpressure_drops = sum(item.backpressure_drop_count for item in stats) - max_uncertainty = max((item.clock.uncertainty_ms for item in envelopes), default=0) - clock_sources = sorted({item.clock.source.value for item in envelopes} | {item.clock_source.value for item in stats}) +@dataclass(frozen=True) +class _ReceiptFacts: + lost_event_count: int + duplicate_sequence_count: int + rejected_by_reason: dict[str, int] + outbound_endpoints: list[str] + entered_worker_context: bool + entered_scheduler_inputs: bool + observer_failure_count: int + backpressure_drop_count: int + max_clock_uncertainty_ms: int + clock_sources: list[str] + envelopes_by_observer: dict[str, list[ObserverEnvelope]] + stats_mismatch: bool - envelopes_by_observer: dict[str, list[ObserverEnvelope]] = defaultdict(list) + +def _group_envelopes_by_observer( + envelopes: Iterable[ObserverEnvelope], +) -> dict[str, list[ObserverEnvelope]]: + grouped: dict[str, list[ObserverEnvelope]] = defaultdict(list) for envelope in envelopes: - envelopes_by_observer[envelope.observer_id].append(envelope) - stats_mismatch = False + grouped[envelope.observer_id].append(envelope) + return grouped + + +def _observer_stats_mismatch( + reading: LedgerReading, + envelopes_by_observer: Mapping[str, list[ObserverEnvelope]], +) -> bool: for observer_id, observer_envelopes in envelopes_by_observer.items(): observer_stats = reading.stats.get(observer_id) if observer_stats is None: continue - if ( - {item.provider_id for item in observer_envelopes} - != {observer_stats.provider_id} - or observer_stats.accepted_event_count != len(observer_envelopes) - ): - stats_mismatch = True - if any( + provider_ids = {item.provider_id for item in observer_envelopes} + if provider_ids != { + observer_stats.provider_id + } or observer_stats.accepted_event_count != len(observer_envelopes): + return True + return any( item.accepted_event_count and observer_id not in envelopes_by_observer for observer_id, item in reading.stats.items() - ): - stats_mismatch = True - - reasons: set[ReceiptReason] = set() - if not envelopes: - reasons.add(ReceiptReason.NO_OBSERVATIONS) - if outbound_endpoints: - reasons.add(ReceiptReason.OUTBOUND_ENDPOINT_CONFIGURED) - if entered_worker_context: - reasons.add(ReceiptReason.OBSERVATION_ENTERED_WORKER_CONTEXT) - if entered_scheduler_inputs: - reasons.add(ReceiptReason.OBSERVATION_ENTERED_SCHEDULER_INPUTS) - if observer_failures: - reasons.add(ReceiptReason.OBSERVER_FAILURE) - if rejected_by_reason.get(EnvelopeRejection.CONTROL_FIELD_REJECTED.value): - reasons.add(ReceiptReason.CONTROL_FIELD_REJECTED) - if reading.invalid_record_count: - reasons.add(ReceiptReason.LEDGER_RECORD_INVALID) - if envelopes and any( - observer_id not in reading.stats for observer_id in envelopes_by_observer - ): - reasons.add(ReceiptReason.OBSERVER_STATS_MISSING) - if stats_mismatch: - reasons.add(ReceiptReason.OBSERVER_STATS_MISMATCH) - if lost: - reasons.add(ReceiptReason.SEQUENCE_GAP) - if duplicates: - reasons.add(ReceiptReason.SEQUENCE_DUPLICATE) - if backpressure_drops: - reasons.add(ReceiptReason.BACKPRESSURE_DROP) - if rejected_by_reason.get(EnvelopeRejection.RAW_MATERIAL_FIELD_REJECTED.value): - reasons.add(ReceiptReason.RAW_MATERIAL_REJECTED) - if rejected_by_reason.get(EnvelopeRejection.UNSUPPORTED_FIELD_REJECTED.value): - reasons.add(ReceiptReason.UNSUPPORTED_FIELD_REJECTED) - if rejected_by_reason.get(EnvelopeRejection.IDENTITY_INVALID.value): - reasons.add(ReceiptReason.IDENTITY_REJECTED) - if max_uncertainty > clock_uncertainty_degraded_ms: - reasons.add(ReceiptReason.CLOCK_UNCERTAINTY_EXCEEDED) + ) + + +def _collect_receipt_facts( + reading: LedgerReading, + envelopes: list[ObserverEnvelope], + stats: list[ObserverStats], +) -> _ReceiptFacts: + lost, duplicates = _sequence_accounting(envelopes) + rejected_by_reason: dict[str, int] = defaultdict(int) + for item in stats: + for reason, count in item.rejected_by_reason.items(): + rejected_by_reason[reason] += count + envelopes_by_observer = _group_envelopes_by_observer(envelopes) + return _ReceiptFacts( + lost_event_count=lost, + duplicate_sequence_count=duplicates, + rejected_by_reason=dict(rejected_by_reason), + outbound_endpoints=sorted( + {endpoint for item in stats for endpoint in item.outbound_endpoints} + ), + entered_worker_context=any( + item.observation_entered_worker_context for item in stats + ), + entered_scheduler_inputs=any( + item.observation_entered_scheduler_inputs for item in stats + ), + observer_failure_count=sum(item.observer_failure_count for item in stats), + backpressure_drop_count=sum(item.backpressure_drop_count for item in stats), + max_clock_uncertainty_ms=max( + (item.clock.uncertainty_ms for item in envelopes), default=0 + ), + clock_sources=sorted( + {item.clock.source.value for item in envelopes} + | {item.clock_source.value for item in stats} + ), + envelopes_by_observer=envelopes_by_observer, + stats_mismatch=_observer_stats_mismatch(reading, envelopes_by_observer), + ) + +def _receipt_reasons( + reading: LedgerReading, + envelopes: list[ObserverEnvelope], + facts: _ReceiptFacts, + *, + clock_uncertainty_degraded_ms: int, +) -> set[ReceiptReason]: + missing_stats = bool(envelopes) and any( + observer_id not in reading.stats for observer_id in facts.envelopes_by_observer + ) + rejected = facts.rejected_by_reason + candidates = ( + (not envelopes, ReceiptReason.NO_OBSERVATIONS), + (bool(facts.outbound_endpoints), ReceiptReason.OUTBOUND_ENDPOINT_CONFIGURED), + ( + facts.entered_worker_context, + ReceiptReason.OBSERVATION_ENTERED_WORKER_CONTEXT, + ), + ( + facts.entered_scheduler_inputs, + ReceiptReason.OBSERVATION_ENTERED_SCHEDULER_INPUTS, + ), + (bool(facts.observer_failure_count), ReceiptReason.OBSERVER_FAILURE), + ( + bool(rejected.get(EnvelopeRejection.CONTROL_FIELD_REJECTED.value)), + ReceiptReason.CONTROL_FIELD_REJECTED, + ), + (bool(reading.invalid_record_count), ReceiptReason.LEDGER_RECORD_INVALID), + (missing_stats, ReceiptReason.OBSERVER_STATS_MISSING), + (facts.stats_mismatch, ReceiptReason.OBSERVER_STATS_MISMATCH), + (bool(facts.lost_event_count), ReceiptReason.SEQUENCE_GAP), + (bool(facts.duplicate_sequence_count), ReceiptReason.SEQUENCE_DUPLICATE), + (bool(facts.backpressure_drop_count), ReceiptReason.BACKPRESSURE_DROP), + ( + bool(rejected.get(EnvelopeRejection.RAW_MATERIAL_FIELD_REJECTED.value)), + ReceiptReason.RAW_MATERIAL_REJECTED, + ), + ( + bool(rejected.get(EnvelopeRejection.UNSUPPORTED_FIELD_REJECTED.value)), + ReceiptReason.UNSUPPORTED_FIELD_REJECTED, + ), + ( + bool(rejected.get(EnvelopeRejection.IDENTITY_INVALID.value)), + ReceiptReason.IDENTITY_REJECTED, + ), + ( + facts.max_clock_uncertainty_ms > clock_uncertainty_degraded_ms, + ReceiptReason.CLOCK_UNCERTAINTY_EXCEEDED, + ), + ) + return {reason for applies, reason in candidates if applies} + + +def _receipt_status(reasons: set[ReceiptReason]) -> ReceiptStatus: if reasons & _INVALID_REASONS: - status = ReceiptStatus.INVALID - elif reasons & _QUARANTINE_REASONS: - status = ReceiptStatus.QUARANTINED - elif reasons: - status = ReceiptStatus.DEGRADED - else: - status = ReceiptStatus.VALID + return ReceiptStatus.INVALID + if reasons & _QUARANTINE_REASONS: + return ReceiptStatus.QUARANTINED + if reasons: + return ReceiptStatus.DEGRADED + return ReceiptStatus.VALID + + +def build_integrity_receipt( + reading: LedgerReading, + *, + clock_uncertainty_degraded_ms: int = DEFAULT_CLOCK_UNCERTAINTY_DEGRADED_MS, +) -> dict[str, Any]: + envelopes = reading.ordered_envelopes + stats = list(reading.stats.values()) + facts = _collect_receipt_facts(reading, envelopes, stats) + reasons = _receipt_reasons( + reading, + envelopes, + facts, + clock_uncertainty_degraded_ms=clock_uncertainty_degraded_ms, + ) + status = _receipt_status(reasons) return { "schema_version": INTEGRITY_RECEIPT_SCHEMA_VERSION, @@ -235,28 +331,34 @@ def build_integrity_receipt( "goal_id": reading.goal_id, "status": status.value, "reason_codes": sorted(reason.value for reason in reasons), - "provider_ids": sorted({item.provider_id for item in envelopes} | {item.provider_id for item in stats}), - "observer_ids": sorted(set(reading.stats) | set(envelopes_by_observer)), + "provider_ids": sorted( + {item.provider_id for item in envelopes} + | {item.provider_id for item in stats} + ), + "observer_ids": sorted(set(reading.stats) | set(facts.envelopes_by_observer)), "session_count": len({item.session_id for item in envelopes}), "observed_event_count": sum(item.observed_event_count for item in stats), "accepted_event_count": sum(item.accepted_event_count for item in stats), "persisted_event_count": len(envelopes), - "lost_event_count": lost, - "duplicate_sequence_count": duplicates, + "lost_event_count": facts.lost_event_count, + "duplicate_sequence_count": facts.duplicate_sequence_count, "ledger_invalid_record_count": reading.invalid_record_count, - "rejected_event_count": sum(rejected_by_reason.values()), - "rejected_by_reason": dict(sorted(rejected_by_reason.items())), + "rejected_event_count": sum(facts.rejected_by_reason.values()), + "rejected_by_reason": dict(sorted(facts.rejected_by_reason.items())), "buffer_bound": max((item.buffer_bound for item in stats), default=None), - "backpressure_drop_count": backpressure_drops, - "observer_failure_count": observer_failures, + "backpressure_drop_count": facts.backpressure_drop_count, + "observer_failure_count": facts.observer_failure_count, "peak_buffered_event_count": max( (item.peak_buffered_event_count for item in stats), default=0 ), "flush_attempt_count": sum(item.flush_attempt_count for item in stats), - "clock": {"sources": clock_sources, "max_uncertainty_ms": max_uncertainty}, - "outbound_endpoints": outbound_endpoints, - "observation_entered_worker_context": entered_worker_context, - "observation_entered_scheduler_inputs": entered_scheduler_inputs, + "clock": { + "sources": facts.clock_sources, + "max_uncertainty_ms": facts.max_clock_uncertainty_ms, + }, + "outbound_endpoints": facts.outbound_endpoints, + "observation_entered_worker_context": facts.entered_worker_context, + "observation_entered_scheduler_inputs": facts.entered_scheduler_inputs, "run_identities": [ item.run_identity.as_dict() for item in sorted(stats, key=lambda candidate: candidate.observer_id) @@ -268,7 +370,9 @@ def build_integrity_receipt( {field for item in stats for field in item.source_fields_consumed} ), "event_kinds_consumed": sorted({item.event_kind.value for item in envelopes}), - "summary_fields_consumed": sorted({key for item in envelopes for key in item.summary}), + "summary_fields_consumed": sorted( + {key for item in envelopes for key in item.summary} + ), "observed_from": envelopes[0].observed_at if envelopes else None, "observed_until": envelopes[-1].observed_at if envelopes else None, } diff --git a/loopx/cli_commands/reliability_diagnostics.py b/loopx/cli_commands/reliability_diagnostics.py index 281077c702..0e30177bd5 100644 --- a/loopx/cli_commands/reliability_diagnostics.py +++ b/loopx/cli_commands/reliability_diagnostics.py @@ -5,6 +5,7 @@ import argparse import sys from collections.abc import Callable +from dataclasses import dataclass from datetime import datetime, timezone from pathlib import Path from typing import Any @@ -29,13 +30,27 @@ from ..history import load_registry from ..paths import resolve_runtime_root -PrintPayload = Callable[[dict[str, Any], str, Callable[[dict[str, Any]], str]], str | None] +PrintPayload = Callable[ + [dict[str, Any], str, Callable[[dict[str, Any]], str]], str | None +] FormatSelector = Callable[..., str] AddFormat = Callable[[argparse.ArgumentParser], None] INGEST_VIOLATION_SCHEMA_VERSION = "reliability_ingest_violation_v0" +@dataclass(frozen=True) +class _AcceptedIngestRecord: + value: dict[str, Any] + kind: str + + +class _IngestRecordRejected(ValueError): + def __init__(self, reason: EnvelopeRejection) -> None: + super().__init__(reason.value) + self.reason = reason + + def register_reliability_diagnostics_commands( subparsers: argparse._SubParsersAction[argparse.ArgumentParser], add_subcommand_format: AddFormat, @@ -44,21 +59,31 @@ def register_reliability_diagnostics_commands( "reliability-diagnostics", help="Read back the L1 shadow-observer ledger: ingest, receipt, status.", ) - commands = parser.add_subparsers(dest="reliability_diagnostics_command", required=True) + commands = parser.add_subparsers( + dest="reliability_diagnostics_command", required=True + ) ingest = commands.add_parser( "ingest", help="Validate observer envelopes and append accepted ones to the goal ledger.", ) add_subcommand_format(ingest) ingest.add_argument("--goal-id", required=True) - ingest.add_argument("--input", required=True, help="NDJSON file path, or - for stdin.") - receipt = commands.add_parser("receipt", help="Render the treatment-integrity receipt.") + ingest.add_argument( + "--input", required=True, help="NDJSON file path, or - for stdin." + ) + receipt = commands.add_parser( + "receipt", help="Render the treatment-integrity receipt." + ) add_subcommand_format(receipt) receipt.add_argument("--goal-id", required=True) - status = commands.add_parser("status", help="Render the read-only diagnostic projection.") + status = commands.add_parser( + "status", help="Render the read-only diagnostic projection." + ) add_subcommand_format(status) status.add_argument("--goal-id", required=True) - status.add_argument("--as-of", help="Timezone-aware ISO-8601 time used for stall age.") + status.add_argument( + "--as-of", help="Timezone-aware ISO-8601 time used for stall age." + ) def _render(payload: dict[str, Any]) -> str: @@ -69,13 +94,58 @@ def _render(payload: dict[str, Any]) -> str: for section in ("receipt", "projection"): body = payload.get(section) if isinstance(body, dict): - lines.append(f"- {section}.status: `{body.get('status') or body.get('integrity', {}).get('status')}`") - for key in ("stage", "signals", "reason_codes", "lost_event_count", "backpressure_drop_count"): + lines.append( + f"- {section}.status: `{body.get('status') or body.get('integrity', {}).get('status')}`" + ) + for key in ( + "stage", + "signals", + "reason_codes", + "lost_event_count", + "backpressure_drop_count", + ): if key in body: lines.append(f"- {section}.{key}: `{body[key]}`") return "\n".join(lines) + "\n" +def _normalize_stats_for_ingest( + record: dict[str, Any], + goal_id: str, +) -> _AcceptedIngestRecord: + try: + normalized = normalize_observer_stats(record) + except ValueError as exc: + raise _IngestRecordRejected(EnvelopeRejection.SCHEMA_MISMATCH) from exc + if normalized.goal_id != goal_id: + raise _IngestRecordRejected(EnvelopeRejection.IDENTITY_INVALID) + return _AcceptedIngestRecord(normalized.as_dict(), "stats") + + +def _normalize_envelope_for_ingest( + record: dict[str, Any], + goal_id: str, +) -> _AcceptedIngestRecord: + try: + normalized = normalize_observer_envelope(record) + except ObserverEnvelopeError as exc: + raise _IngestRecordRejected(exc.reason) from exc + if normalized.goal_id != goal_id: + raise _IngestRecordRejected(EnvelopeRejection.IDENTITY_INVALID) + return _AcceptedIngestRecord(normalized.as_dict(), "envelope") + + +def _normalize_ingest_record(record: Any, goal_id: str) -> _AcceptedIngestRecord: + if not isinstance(record, dict): + raise _IngestRecordRejected(EnvelopeRejection.SCHEMA_MISMATCH) + schema = record.get("schema_version") + if schema == OBSERVER_STATS_SCHEMA_VERSION: + return _normalize_stats_for_ingest(record, goal_id) + if schema == OBSERVER_ENVELOPE_SCHEMA_VERSION: + return _normalize_envelope_for_ingest(record, goal_id) + raise _IngestRecordRejected(EnvelopeRejection.SCHEMA_MISMATCH) + + def _ingest(path: Path, goal_id: str, source: str) -> dict[str, Any]: if source == "-": lines = sys.stdin.read().splitlines() @@ -83,46 +153,26 @@ def _ingest(path: Path, goal_id: str, source: str) -> dict[str, Any]: lines = Path(source).expanduser().read_text(encoding="utf-8").splitlines() parsed, malformed = parse_ndjson_lines(lines) accepted_records: list[dict[str, Any]] = [] - accepted_envelopes = 0 - passthrough_stats = 0 + accepted_by_kind = {"envelope": 0, "stats": 0} rejected_by_reason: dict[str, int] = {} def reject(reason: EnvelopeRejection) -> None: rejected_by_reason[reason.value] = rejected_by_reason.get(reason.value, 0) + 1 for record in parsed: - if not isinstance(record, dict): - reject(EnvelopeRejection.SCHEMA_MISMATCH) - continue - schema = record.get("schema_version") - if schema == OBSERVER_STATS_SCHEMA_VERSION: - try: - stats = normalize_observer_stats(record) - except ValueError: - reject(EnvelopeRejection.SCHEMA_MISMATCH) - continue - if stats.goal_id != goal_id: - reject(EnvelopeRejection.IDENTITY_INVALID) - continue - accepted_records.append(stats.as_dict()) - passthrough_stats += 1 - continue - if schema != OBSERVER_ENVELOPE_SCHEMA_VERSION: - reject(EnvelopeRejection.SCHEMA_MISMATCH) - continue try: - envelope = normalize_observer_envelope(record) - except ObserverEnvelopeError as exc: + accepted = _normalize_ingest_record(record, goal_id) + except _IngestRecordRejected as exc: reject(exc.reason) continue - if envelope.goal_id != goal_id: - reject(EnvelopeRejection.IDENTITY_INVALID) - continue - accepted_records.append(envelope.as_dict()) - accepted_envelopes += 1 + accepted_records.append(accepted.value) + accepted_by_kind[accepted.kind] += 1 - for _ in range(malformed): - reject(EnvelopeRejection.SCHEMA_MISMATCH) + if malformed: + rejected_by_reason[EnvelopeRejection.SCHEMA_MISMATCH.value] = ( + rejected_by_reason.get(EnvelopeRejection.SCHEMA_MISMATCH.value, 0) + + malformed + ) appended = append_ledger_records(path, accepted_records) rejected_event_count = sum(rejected_by_reason.values()) gate_recorded = rejected_event_count > 0 @@ -150,8 +200,8 @@ def reject(reason: EnvelopeRejection) -> None: "goal_id": goal_id, "ledger_ref": ledger_ref(goal_id), "appended_record_count": appended, - "accepted_envelope_count": accepted_envelopes, - "passthrough_stats_count": passthrough_stats, + "accepted_envelope_count": accepted_by_kind["envelope"], + "passthrough_stats_count": accepted_by_kind["stats"], "rejected_event_count": rejected_event_count, "rejected_by_reason": dict(sorted(rejected_by_reason.items())), "malformed_line_count": malformed, @@ -187,12 +237,21 @@ def handle_reliability_diagnostics_command( payload = _ingest(path, goal_id, str(args.input)) else: records, malformed = read_ledger_records(path) - reading = read_ledger(records, goal_id=goal_id, malformed_line_count=malformed) - payload = {"ok": True, "command": command, "goal_id": goal_id, "ledger_ref": ledger_ref(goal_id)} + reading = read_ledger( + records, goal_id=goal_id, malformed_line_count=malformed + ) + payload = { + "ok": True, + "command": command, + "goal_id": goal_id, + "ledger_ref": ledger_ref(goal_id), + } if command == "receipt": payload["receipt"] = build_integrity_receipt(reading) else: - payload["projection"] = build_diagnostic_projection(reading, as_of=args.as_of) + payload["projection"] = build_diagnostic_projection( + reading, as_of=args.as_of + ) except (OSError, ValueError) as exc: print(f"error: {CAPABILITY_ID}: {exc}", file=sys.stderr) return 2 diff --git a/packages/dsh-loopx-plugin/src/observer.ts b/packages/dsh-loopx-plugin/src/observer.ts index dd6c7f94b0..379caf8068 100644 --- a/packages/dsh-loopx-plugin/src/observer.ts +++ b/packages/dsh-loopx-plugin/src/observer.ts @@ -189,8 +189,8 @@ export function resolveShadowObserverConfig( function validRunIdentity(value: unknown): value is ObserverRunIdentity { if (typeof value !== 'object' || value === null || Array.isArray(value)) return false const record = value as Record - const actual = Object.keys(record).sort() - const expected = [...RUN_IDENTITY_FIELDS].sort() + const actual = Object.keys(record).sort((left, right) => left.localeCompare(right)) + const expected = [...RUN_IDENTITY_FIELDS].sort((left, right) => left.localeCompare(right)) return actual.length === expected.length && expected.every((field, index) => actual[index] === field) && RUN_IDENTITY_FIELDS.every((field) => { @@ -220,13 +220,11 @@ async function appendLedgerLines(path: string, lines: readonly string[]): Promis } function token(value: unknown): string | undefined { - const text = String(value ?? '') - return SUMMARY_TOKEN.test(text) ? text : undefined + return typeof value === 'string' && SUMMARY_TOKEN.test(value) ? value : undefined } function identity(value: unknown): string | undefined { - const text = String(value ?? '') - return IDENTITY_TOKEN.test(text) ? text : undefined + return typeof value === 'string' && IDENTITY_TOKEN.test(value) ? value : undefined } function count(value: unknown): number | undefined { diff --git a/tests/capabilities/test_reliability_diagnostics.py b/tests/capabilities/test_reliability_diagnostics.py index 369eec294b..1e4f25d4ce 100644 --- a/tests/capabilities/test_reliability_diagnostics.py +++ b/tests/capabilities/test_reliability_diagnostics.py @@ -64,7 +64,14 @@ observer_revision="observer-revision-1", ) EVENT_SOURCES = ["dsh-agent-hooks", "dsh-session-events"] -SOURCE_FIELDS = ["agent.id", "event.data", "event.seq", "event.time", "event.type", "session.id"] +SOURCE_FIELDS = [ + "agent.id", + "event.data", + "event.seq", + "event.time", + "event.type", + "session.id", +] def envelope( @@ -130,20 +137,29 @@ def at(seconds: int) -> str: def receipt_for(*records: dict[str, Any], malformed: int = 0) -> dict[str, Any]: - return build_integrity_receipt(read_ledger(records, goal_id=GOAL, malformed_line_count=malformed)) + return build_integrity_receipt( + read_ledger(records, goal_id=GOAL, malformed_line_count=malformed) + ) # --- envelope schema ------------------------------------------------------- def test_envelope_schema_has_no_control_or_raw_material_fields() -> None: - flattened = {name.replace("_", "") for name in ENVELOPE_FIELDS | SUMMARY_FIELDS | SOURCE_REF_FIELDS} + flattened = { + name.replace("_", "") + for name in ENVELOPE_FIELDS | SUMMARY_FIELDS | SOURCE_REF_FIELDS + } assert not flattened & CONTROL_FIELD_FAMILIES assert not flattened & RAW_MATERIAL_FIELD_FAMILIES def test_valid_envelope_round_trips() -> None: - record = envelope(3, ObserverEventKind.TOOL_CALLED, summary={"turn": 1, "step": 2, "tool_name": "bash"}) + record = envelope( + 3, + ObserverEventKind.TOOL_CALLED, + summary={"turn": 1, "step": 2, "tool_name": "bash"}, + ) record["source_refs"]["tool_call_id"] = "call-9" normalized = normalize_observer_envelope(record) assert normalized.event_kind is ObserverEventKind.TOOL_CALLED @@ -152,7 +168,18 @@ def test_valid_envelope_round_trips() -> None: @pytest.mark.parametrize( "field", - ["command", "send", "schedule", "retry", "stop", "resume", "gate", "toolCall", "worker_state", "callback"], + [ + "command", + "send", + "schedule", + "retry", + "stop", + "resume", + "gate", + "toolCall", + "worker_state", + "callback", + ], ) def test_control_shaped_fields_are_rejected_as_control(field: str) -> None: record = envelope(1, **{field: {"kind": "continue"}}) @@ -162,24 +189,32 @@ def test_control_shaped_fields_are_rejected_as_control(field: str) -> None: assert excinfo.value.reason is EnvelopeRejection.CONTROL_FIELD_REJECTED -@pytest.mark.parametrize("field", ["transcript", "tool_output", "stdout", "cwd", "messages", "arguments"]) +@pytest.mark.parametrize( + "field", ["transcript", "tool_output", "stdout", "cwd", "messages", "arguments"] +) def test_raw_material_fields_are_rejected_and_classified(field: str) -> None: + record = envelope(1, **{field: "protected"}) with pytest.raises(ObserverEnvelopeError) as excinfo: - normalize_observer_envelope(envelope(1, **{field: "protected"})) + normalize_observer_envelope(record) assert excinfo.value.reason is EnvelopeRejection.RAW_MATERIAL_FIELD_REJECTED def test_raw_material_inside_summary_is_rejected() -> None: + record = envelope(1, summary={"text": "hello"}) with pytest.raises(ObserverEnvelopeError) as excinfo: - normalize_observer_envelope(envelope(1, summary={"text": "hello"})) + normalize_observer_envelope(record) assert excinfo.value.reason is EnvelopeRejection.RAW_MATERIAL_FIELD_REJECTED def test_unknown_field_is_rejected_as_unsupported() -> None: + record = envelope(1, colour="blue") with pytest.raises(ObserverEnvelopeError) as excinfo: - normalize_observer_envelope(envelope(1, colour="blue")) + normalize_observer_envelope(record) assert excinfo.value.reason is EnvelopeRejection.UNSUPPORTED_FIELD_REJECTED - assert classify_rejected_field("agentSend") is EnvelopeRejection.UNSUPPORTED_FIELD_REJECTED + assert ( + classify_rejected_field("agentSend") + is EnvelopeRejection.UNSUPPORTED_FIELD_REJECTED + ) @pytest.mark.parametrize( @@ -191,16 +226,33 @@ def test_unknown_field_is_rejected_as_unsupported() -> None: ({"sequence": True}, EnvelopeRejection.SEQUENCE_INVALID), ({"sequence": -4}, EnvelopeRejection.SEQUENCE_INVALID), ({"observed_at": "2026-09-01T10:00:00"}, EnvelopeRejection.CLOCK_INVALID), - ({"clock": {"source": "gps", "uncertainty_ms": 0}}, EnvelopeRejection.CLOCK_INVALID), - ({"clock": {"source": "fixture", "uncertainty_ms": -1}}, EnvelopeRejection.CLOCK_INVALID), + ( + {"clock": {"source": "gps", "uncertainty_ms": 0}}, + EnvelopeRejection.CLOCK_INVALID, + ), + ( + {"clock": {"source": "fixture", "uncertainty_ms": -1}}, + EnvelopeRejection.CLOCK_INVALID, + ), ({"event_kind": "prompt_injected"}, EnvelopeRejection.EVENT_KIND_INVALID), - ({"summary": {"tool_name": "bash -c 'rm -rf'"}}, EnvelopeRejection.SUMMARY_INVALID), + ( + {"summary": {"tool_name": "bash -c 'rm -rf'"}}, + EnvelopeRejection.SUMMARY_INVALID, + ), ({"summary": {"turn": "one"}}, EnvelopeRejection.SUMMARY_INVALID), - ({"source_refs": {"tool_call_id": "/Users/someone/call"}}, EnvelopeRejection.SOURCE_REF_INVALID), - ({"source_refs": {"tool_call_id": "sk-abcdefghijklmnop0123"}}, EnvelopeRejection.PUBLIC_SAFETY_VIOLATION), + ( + {"source_refs": {"tool_call_id": "/Users/someone/call"}}, + EnvelopeRejection.SOURCE_REF_INVALID, + ), + ( + {"source_refs": {"tool_call_id": "sk-abcdefghijklmnop0123"}}, + EnvelopeRejection.PUBLIC_SAFETY_VIOLATION, + ), ], ) -def test_malformed_envelopes_carry_typed_reasons(mutation: dict[str, Any], reason: EnvelopeRejection) -> None: +def test_malformed_envelopes_carry_typed_reasons( + mutation: dict[str, Any], reason: EnvelopeRejection +) -> None: record = envelope(1) record.update(mutation) with pytest.raises(ObserverEnvelopeError) as excinfo: @@ -226,7 +278,16 @@ def intake(bound: int = 8) -> ShadowObserverIntake: def test_intake_exposes_no_control_surface() -> None: names = {name.lower() for name in dir(ShadowObserverIntake)} - assert not names & {"send", "command", "schedule", "retry", "stop", "resume", "pause", "inject"} + assert not names & { + "send", + "command", + "schedule", + "retry", + "stop", + "resume", + "pause", + "inject", + } def test_intake_bounds_buffer_and_counts_drops_without_raising() -> None: @@ -287,8 +348,9 @@ def test_intake_rejects_and_counts_by_reason() -> None: def test_stats_record_rejects_unknown_fields() -> None: + record = stats(send_path="agent.send") with pytest.raises(ValueError, match="unsupported fields"): - normalize_observer_stats(stats(send_path="agent.send")) + normalize_observer_stats(record) @pytest.mark.parametrize( @@ -306,8 +368,9 @@ def test_stats_record_rejects_unknown_fields() -> None: def test_stats_record_rejects_inconsistent_or_incomplete_evidence( mutation: dict[str, Any], ) -> None: + record = stats(**mutation) with pytest.raises(ValueError): - normalize_observer_stats(stats(**mutation)) + normalize_observer_stats(record) # --- receipt --------------------------------------------------------------- @@ -361,17 +424,27 @@ def test_receipt_counts_duplicate_sequences_separately() -> None: ("stats_override", "reason"), [ ({"observer_failure_count": 1}, ReceiptReason.OBSERVER_FAILURE), - ({"rejected_event_count": 1, "rejected_by_reason": {"control_field_rejected": 1}}, ReceiptReason.CONTROL_FIELD_REJECTED), + ( + { + "rejected_event_count": 1, + "rejected_by_reason": {"control_field_rejected": 1}, + }, + ReceiptReason.CONTROL_FIELD_REJECTED, + ), ], ) -def test_receipt_quarantines_observer_failure_and_control_fields(stats_override: dict[str, Any], reason: ReceiptReason) -> None: +def test_receipt_quarantines_observer_failure_and_control_fields( + stats_override: dict[str, Any], reason: ReceiptReason +) -> None: receipt = receipt_for(envelope(0), stats(**stats_override)) assert receipt["status"] == ReceiptStatus.QUARANTINED.value assert receipt["reason_codes"] == [reason.value] def test_receipt_invalidates_malformed_ledger_records() -> None: - receipt = receipt_for(envelope(0), stats(), {"schema_version": "unknown"}, malformed=1) + receipt = receipt_for( + envelope(0), stats(), {"schema_version": "unknown"}, malformed=1 + ) assert receipt["ledger_invalid_record_count"] == 2 assert receipt["status"] == ReceiptStatus.INVALID.value @@ -379,13 +452,25 @@ def test_receipt_invalidates_malformed_ledger_records() -> None: @pytest.mark.parametrize( ("stats_override", "reason"), [ - ({"outbound_endpoints": ["loopx-continuation"]}, ReceiptReason.OUTBOUND_ENDPOINT_CONFIGURED), - ({"observation_entered_worker_context": True}, ReceiptReason.OBSERVATION_ENTERED_WORKER_CONTEXT), + ( + {"outbound_endpoints": ["loopx-continuation"]}, + ReceiptReason.OUTBOUND_ENDPOINT_CONFIGURED, + ), + ( + {"observation_entered_worker_context": True}, + ReceiptReason.OBSERVATION_ENTERED_WORKER_CONTEXT, + ), ], ) -def test_receipt_invalidates_any_outbound_or_worker_context_path(stats_override: dict[str, Any], reason: ReceiptReason) -> None: - receipt = receipt_for(envelope(0), stats(observer_failure_count=1, **stats_override)) - assert receipt["status"] == ReceiptStatus.INVALID.value # invalid outranks quarantined +def test_receipt_invalidates_any_outbound_or_worker_context_path( + stats_override: dict[str, Any], reason: ReceiptReason +) -> None: + receipt = receipt_for( + envelope(0), stats(observer_failure_count=1, **stats_override) + ) + assert ( + receipt["status"] == ReceiptStatus.INVALID.value + ) # invalid outranks quarantined assert reason.value in receipt["reason_codes"] assert ReceiptReason.OBSERVER_FAILURE.value in receipt["reason_codes"] @@ -393,12 +478,34 @@ def test_receipt_invalidates_any_outbound_or_worker_context_path(stats_override: @pytest.mark.parametrize( ("records", "reason"), [ - ([envelope(0, clock={"source": "observer_wall_clock", "uncertainty_ms": 1001}), stats()], ReceiptReason.CLOCK_UNCERTAINTY_EXCEEDED), - ([envelope(0), stats(backpressure_drop_count=2)], ReceiptReason.BACKPRESSURE_DROP), - ([envelope(0), stats(rejected_event_count=1, rejected_by_reason={"raw_material_field_rejected": 1})], ReceiptReason.RAW_MATERIAL_REJECTED), + ( + [ + envelope( + 0, clock={"source": "observer_wall_clock", "uncertainty_ms": 1001} + ), + stats(), + ], + ReceiptReason.CLOCK_UNCERTAINTY_EXCEEDED, + ), + ( + [envelope(0), stats(backpressure_drop_count=2)], + ReceiptReason.BACKPRESSURE_DROP, + ), + ( + [ + envelope(0), + stats( + rejected_event_count=1, + rejected_by_reason={"raw_material_field_rejected": 1}, + ), + ], + ReceiptReason.RAW_MATERIAL_REJECTED, + ), ], ) -def test_receipt_degrades_but_keeps_evidence(records: list[dict[str, Any]], reason: ReceiptReason) -> None: +def test_receipt_degrades_but_keeps_evidence( + records: list[dict[str, Any]], reason: ReceiptReason +) -> None: receipt = receipt_for(*records) assert receipt["status"] == ReceiptStatus.DEGRADED.value assert receipt["reason_codes"] == [reason.value] @@ -427,8 +534,14 @@ def test_receipt_invalidates_unlinked_stats_and_identity_rejection() -> None: def test_receipt_clock_uncertainty_at_threshold_is_visible_not_degraded() -> None: - receipt = receipt_for(envelope(0, clock={"source": "observer_wall_clock", "uncertainty_ms": 1000}), stats()) - assert receipt["clock"] == {"sources": ["harness_event_time", "observer_wall_clock"], "max_uncertainty_ms": 1000} + receipt = receipt_for( + envelope(0, clock={"source": "observer_wall_clock", "uncertainty_ms": 1000}), + stats(), + ) + assert receipt["clock"] == { + "sources": ["harness_event_time", "observer_wall_clock"], + "max_uncertainty_ms": 1000, + } assert receipt["status"] == ReceiptStatus.VALID.value @@ -457,7 +570,10 @@ def test_projection_declares_read_only_boundary() -> None: assert projection["authority"] == "none" assert projection["write_scope"] == "diagnostic_ledger_only" assert projection["worker_influence"] == "none" - assert set(projection) & {"command", "next_action", "recommended_action", "gate"} == set() + assert ( + set(projection) & {"command", "next_action", "recommended_action", "gate"} + == set() + ) def test_projection_stall_requires_active_stage_and_silence() -> None: @@ -474,7 +590,12 @@ def test_projection_stall_requires_active_stage_and_silence() -> None: idle = projection_for( envelope(0, ObserverEventKind.TURN_STARTED), - envelope(1, ObserverEventKind.TURN_ENDED, observed_at=at(1), summary={"reason": "completed"}), + envelope( + 1, + ObserverEventKind.TURN_ENDED, + observed_at=at(1), + summary={"reason": "completed"}, + ), stats(accepted_event_count=2), as_of=at(6 * 60), ) @@ -485,11 +606,21 @@ def test_projection_stall_requires_active_stage_and_silence() -> None: def test_projection_repetition_counts_consecutive_identical_tool_runs() -> None: tools = ["read", "read", "bash", "read", "read", "read"] records = [ - envelope(index, ObserverEventKind.TOOL_CALLED, observed_at=at(index), summary={"tool_name": tool}) + envelope( + index, + ObserverEventKind.TOOL_CALLED, + observed_at=at(index), + summary={"tool_name": tool}, + ) for index, tool in enumerate(tools) ] projection = projection_for(*records, stats(accepted_event_count=len(records))) - assert projection["repetition"] == {"detected": True, "threshold": 3, "longest_tool_run": 3, "tool_name": "read"} + assert projection["repetition"] == { + "detected": True, + "threshold": 3, + "longest_tool_run": 3, + "tool_name": "read", + } below = projection_for(*records[:2], stats(accepted_event_count=2)) assert below["repetition"]["detected"] is False @@ -497,11 +628,20 @@ def test_projection_repetition_counts_consecutive_identical_tool_runs() -> None: def test_projection_recovery_and_stage_transitions() -> None: unrecovered = projection_for( envelope(0, ObserverEventKind.STEP_STARTED), - envelope(1, ObserverEventKind.AGENT_ERROR, observed_at=at(1), summary={"error_class": "Timeout"}), + envelope( + 1, + ObserverEventKind.AGENT_ERROR, + observed_at=at(1), + summary={"error_class": "Timeout"}, + ), stats(accepted_event_count=2), ) assert unrecovered["stage"] == DiagnosticStage.ERRORED.value - assert unrecovered["recovery"] == {"error_count": 1, "recovered_error_count": 0, "unrecovered_error_count": 1} + assert unrecovered["recovery"] == { + "error_count": 1, + "recovered_error_count": 0, + "unrecovered_error_count": 1, + } assert DiagnosticSignal.UNRECOVERED_ERROR.value in unrecovered["signals"] recovered = projection_for( @@ -549,12 +689,17 @@ def test_projection_surfaces_event_loss_and_integrity() -> None: def test_ledger_ref_is_relative_and_goal_scoped(tmp_path: Path) -> None: assert ledger_ref("goal:alpha") == "reliability_diagnostics/goal_alpha.ndjson" - assert ledger_path(tmp_path, "goal-a") == tmp_path / "reliability_diagnostics" / "goal-a.ndjson" + assert ( + ledger_path(tmp_path, "goal-a") + == tmp_path / "reliability_diagnostics" / "goal-a.ndjson" + ) with pytest.raises(ValueError): ledger_ref("../escape") -def test_ledger_append_is_line_oriented_and_tolerates_malformed_lines(tmp_path: Path) -> None: +def test_ledger_append_is_line_oriented_and_tolerates_malformed_lines( + tmp_path: Path, +) -> None: path = ledger_path(tmp_path, GOAL) assert append_ledger_records(path, [envelope(0)]) == 1 assert append_ledger_records(path, []) == 0 From f0b58f648b92e3d0fcc51e6301c1ccb99a37916e Mon Sep 17 00:00:00 2001 From: song Date: Sat, 5 Sep 2026 02:00:01 +0800 Subject: [PATCH 11/13] fix(dsh-loopx-plugin): reject unsafe observer values before append Signed-off-by: song --- .../reliability_diagnostics/README.md | 12 +- .../reliability_diagnostics/README.zh-CN.md | 9 +- .../reliability_diagnostics/receipt.py | 6 + packages/dsh-loopx-plugin/README.md | 5 + packages/dsh-loopx-plugin/src/observer.ts | 74 +++++++++++- .../dsh-loopx-plugin/tests/observer.spec.ts | 112 +++++++++++++++++- .../test_reliability_diagnostics.py | 12 ++ ...st_reliability_diagnostics_dsh_provider.py | 84 ++++++++++++- ...iability_diagnostics_public_safety_v0.json | 28 +++++ 9 files changed, 324 insertions(+), 18 deletions(-) create mode 100644 tests/fixtures/control_plane/reliability_diagnostics_public_safety_v0.json diff --git a/loopx/capabilities/reliability_diagnostics/README.md b/loopx/capabilities/reliability_diagnostics/README.md index 5a99547262..a5afbc3b79 100644 --- a/loopx/capabilities/reliability_diagnostics/README.md +++ b/loopx/capabilities/reliability_diagnostics/README.md @@ -80,8 +80,10 @@ Any other field is rejected with a typed reason: `control_field_rejected` `tool_call`, `worker_state`, ...), `raw_material_field_rejected` (`transcript`, `messages`, `content`, `text`, `arguments`, `output`, `stdout`, `stderr`, `log`, `cwd`, `token`, ...), or -`unsupported_field_rejected`. Values also pass the shared public-safe check, -so absolute local paths and credential-like tokens fail closed. +`unsupported_field_rejected`. Every provider applies the equivalent recursive +public-safe value contract **before its first append**, and LoopX ingest applies +it again. Absolute local paths and credential-like tokens fail closed without +reaching ledger bytes; the producer counts them as `public_safety_violation`. ### Observer stats (`reliability_observer_stats_v0`) @@ -117,9 +119,9 @@ Status rules: `invalid` when there are no observations, stats are absent or do not link exactly to persisted envelopes, an identity was rejected, the ledger contains invalid input, any outbound endpoint exists, or observation entered worker context or scheduler inputs. Otherwise `quarantined` covers observer -failure or control-shaped input. Event gaps, drops, duplicates, raw or -unsupported fields, and excess clock uncertainty are `degraded`; otherwise the -receipt is `valid`. +failure, control-shaped input, or a producer-side `public_safety_violation`. +Event gaps, drops, duplicates, raw or unsupported fields, and excess clock +uncertainty are `degraded`; otherwise the receipt is `valid`. ### Diagnostic projection (`reliability_diagnostic_projection_v0`) diff --git a/loopx/capabilities/reliability_diagnostics/README.zh-CN.md b/loopx/capabilities/reliability_diagnostics/README.zh-CN.md index fe2311a092..9e8f8efc0c 100644 --- a/loopx/capabilities/reliability_diagnostics/README.zh-CN.md +++ b/loopx/capabilities/reliability_diagnostics/README.zh-CN.md @@ -66,7 +66,9 @@ Driver 或 Agent,只消费 session log 的发布事件,并且不进入 Drive `prompt`、`schedule`、`retry`、`stop`、`resume`、`gate`、`tool_call`、`worker_state` 等)、 `raw_material_field_rejected`(`transcript`、`messages`、`content`、`text`、`arguments`、 `output`、`stdout`、`stderr`、`log`、`cwd`、`token` 等)或 `unsupported_field_rejected`。 -值还要通过共享的 public-safe 检查,绝对本地路径与凭据样式 token 会 fail closed。 +每个 provider 都必须在**首次 append 前**执行等价的递归 public-safe 值契约,LoopX ingest +还会再次校验。绝对本地路径与凭据样式 token 会 fail closed,不会进入 ledger bytes; +producer 会把它们计为 `public_safety_violation`。 ### Observer stats(`reliability_observer_stats_v0`) @@ -97,8 +99,9 @@ provider/observer stats 精确关联。stats 按 observer 实例累计;receipt 状态规则:无观测、stats 缺失或不能与持久化 envelope 精确关联、身份被拒绝、ledger 存在无效输入、任一 outbound endpoint,或观测进入 worker context / scheduler inputs 时 -为 `invalid`。否则 observer 故障或控制形态输入为 `quarantined`;事件缺口、丢弃、重复、 -原始/不支持字段或时钟不确定度超阈值为 `degraded`;其它情况才是 `valid`。 +为 `invalid`。否则 observer 故障、控制形态输入或 producer 侧 +`public_safety_violation` 为 `quarantined`;事件缺口、丢弃、重复、原始/不支持字段或时钟 +不确定度超阈值为 `degraded`;其它情况才是 `valid`。 ### Diagnostic projection(`reliability_diagnostic_projection_v0`) diff --git a/loopx/capabilities/reliability_diagnostics/receipt.py b/loopx/capabilities/reliability_diagnostics/receipt.py index 184449a492..724a3142e0 100644 --- a/loopx/capabilities/reliability_diagnostics/receipt.py +++ b/loopx/capabilities/reliability_diagnostics/receipt.py @@ -42,6 +42,7 @@ class ReceiptReason(StrEnum): OBSERVATION_ENTERED_WORKER_CONTEXT = "observation_entered_worker_context" OBSERVER_FAILURE = "observer_failure" CONTROL_FIELD_REJECTED = "control_field_rejected" + PUBLIC_SAFETY_VIOLATION = "public_safety_violation" LEDGER_RECORD_INVALID = "ledger_record_invalid" OBSERVER_STATS_MISSING = "observer_stats_missing" OBSERVER_STATS_MISMATCH = "observer_stats_mismatch" @@ -71,6 +72,7 @@ class ReceiptReason(StrEnum): { ReceiptReason.OBSERVER_FAILURE, ReceiptReason.CONTROL_FIELD_REJECTED, + ReceiptReason.PUBLIC_SAFETY_VIOLATION, } ) @@ -273,6 +275,10 @@ def _receipt_reasons( bool(rejected.get(EnvelopeRejection.CONTROL_FIELD_REJECTED.value)), ReceiptReason.CONTROL_FIELD_REJECTED, ), + ( + bool(rejected.get(EnvelopeRejection.PUBLIC_SAFETY_VIOLATION.value)), + ReceiptReason.PUBLIC_SAFETY_VIOLATION, + ), (bool(reading.invalid_record_count), ReceiptReason.LEDGER_RECORD_INVALID), (missing_stats, ReceiptReason.OBSERVER_STATS_MISSING), (facts.stats_mismatch, ReceiptReason.OBSERVER_STATS_MISMATCH), diff --git a/packages/dsh-loopx-plugin/README.md b/packages/dsh-loopx-plugin/README.md index 5b327ed5b2..e98b7ab049 100644 --- a/packages/dsh-loopx-plugin/README.md +++ b/packages/dsh-loopx-plugin/README.md @@ -144,6 +144,11 @@ Every hook body and every flush is isolated, so an observer failure is counted into the receipt instead of reaching DSH. This is module and hook isolation, not an OS-process-isolation claim. +Before its first append, the producer applies the same recursive local-path, +credential-like value, and credential-field guard as the Python contract. +Unsafe event tokens or ids are counted as `public_safety_violation` and never +reach ledger bytes; CLI ingest independently re-validates persisted records. + It is off unless one exact goal, DSH session, and complete run identity are declared before DSH starts: diff --git a/packages/dsh-loopx-plugin/src/observer.ts b/packages/dsh-loopx-plugin/src/observer.ts index 379caf8068..3e6b33dcf5 100644 --- a/packages/dsh-loopx-plugin/src/observer.ts +++ b/packages/dsh-loopx-plugin/src/observer.ts @@ -69,6 +69,17 @@ export const RUN_IDENTITY_FIELDS = [ const IDENTITY_TOKEN = /^[A-Za-z0-9][A-Za-z0-9_.:-]{0,120}$/u const SUMMARY_TOKEN = /^[A-Za-z0-9][A-Za-z0-9_./:-]{0,79}$/u +const LOCAL_PATH_SURFACE = /(?]+|[A-Za-z]:[\\/](?:Users|Documents and Settings)[\\/][^\s`'"<>]+)/iu +const SECRET_LIKE_SURFACE = /(?:\bbearer\s+[a-z0-9._~+\/=-]{16,}|\b(?:access|secret)[_-]?key\s*[=:]\s*[^\s`'"<>]+|\b(?:ak|sk)\s*[=:]\s*[^\s`'"<>]+|(?]{12,})/iu +const CREDENTIAL_FIELD_FAMILIES = new Set([ + 'accesskey', 'accesstoken', 'apikey', 'authtoken', 'authorization', + 'clientsecret', 'cookie', 'credential', 'credentials', 'password', + 'privatekey', 'refreshtoken', 'secret', 'sessiontoken', 'token', +]) +const CREDENTIAL_FIELD_SUFFIXES = [ + '_token', '_secret', '_password', '_credential', '_credentials', +] as const +const UNBOUNDED_PAYLOAD_FIELD_FAMILIES = new Set(['requestbody', 'responsebody']) export type ObserverEventKind = | 'session_started' @@ -87,6 +98,12 @@ export type ObserverEventKind = export type ClockSource = 'harness_event_time' | 'observer_wall_clock' | 'fixture' +export type ObserverRejectionReason = + | 'identity_invalid' + | 'clock_invalid' + | 'public_safety_violation' + | 'observer_internal_failure' + export interface ObserverEnvelope { readonly schema_version: typeof OBSERVER_ENVELOPE_SCHEMA_VERSION readonly capability_id: typeof CAPABILITY_ID @@ -158,6 +175,35 @@ export interface ShadowObserverOptions { readonly observerId?: string | undefined } +function normalizePublicSafeFieldName(value: string): string { + return value + .replace(/([a-z0-9])([A-Z])/gu, '$1_$2') + .replace(/[^A-Za-z0-9]+/gu, '_') + .replace(/^_+|_+$/gu, '') + .toLowerCase() +} + +function isPublicSafeText(value: string): boolean { + return !LOCAL_PATH_SURFACE.test(value) && !SECRET_LIKE_SURFACE.test(value) +} + +/** Keep the producer boundary equivalent to LoopX's recursive Python guard. */ +function isPublicSafeValue(value: unknown): boolean { + if (typeof value === 'string') return isPublicSafeText(value) + if (Array.isArray(value)) return value.every(item => isPublicSafeValue(item)) + if (typeof value !== 'object' || value === null) return true + return Object.entries(value).every(([key, item]) => { + if (!isPublicSafeText(key)) return false + const normalized = normalizePublicSafeFieldName(key) + const flattened = normalized.replaceAll('_', '') + if (CREDENTIAL_FIELD_FAMILIES.has(flattened) + || CREDENTIAL_FIELD_SUFFIXES.some(suffix => normalized.endsWith(suffix))) return false + if (UNBOUNDED_PAYLOAD_FIELD_FAMILIES.has(flattened)) return false + if (normalized === 'raw' || normalized.startsWith('raw_')) return false + return isPublicSafeValue(item) + }) +} + export function defaultLedgerDir(env: NodeJS.ProcessEnv = process.env): string { const configured = env[ENV_LEDGER_DIR] return resolve(configured?.trim() @@ -177,7 +223,9 @@ export function resolveShadowObserverConfig( const sessionId = env[ENV_SESSION_ID]?.trim() const runIdentity = parseRunIdentity(env[ENV_RUN_IDENTITY]) if (!goalId || !IDENTITY_TOKEN.test(goalId) + || !isPublicSafeText(goalId) || !sessionId || !IDENTITY_TOKEN.test(sessionId) + || !isPublicSafeText(sessionId) || runIdentity === undefined) return undefined const rawBound = Number.parseInt(env[ENV_BUFFER_BOUND] ?? '', 10) const bufferBound = Number.isInteger(rawBound) && rawBound >= 1 && rawBound <= MAX_BUFFER_BOUND @@ -195,7 +243,7 @@ function validRunIdentity(value: unknown): value is ObserverRunIdentity { && expected.every((field, index) => actual[index] === field) && RUN_IDENTITY_FIELDS.every((field) => { const item = record[field] - return typeof item === 'string' && IDENTITY_TOKEN.test(item) + return typeof item === 'string' && IDENTITY_TOKEN.test(item) && isPublicSafeText(item) }) } @@ -315,8 +363,12 @@ export class ShadowObserver { private flushAttemptCount = 0 constructor(options: ShadowObserverOptions) { - if (!IDENTITY_TOKEN.test(options.config.goalId)) throw new Error('goal id must be an identity token') - if (!IDENTITY_TOKEN.test(options.config.sessionId)) throw new Error('session id must be an identity token') + if (!IDENTITY_TOKEN.test(options.config.goalId) || !isPublicSafeText(options.config.goalId)) { + throw new Error('goal id must be a public-safe identity token') + } + if (!IDENTITY_TOKEN.test(options.config.sessionId) || !isPublicSafeText(options.config.sessionId)) { + throw new Error('session id must be a public-safe identity token') + } if (!validRunIdentity(options.config.runIdentity)) throw new Error('run identity must be fully pinned') if (!Number.isInteger(options.config.bufferBound) || options.config.bufferBound < 1 @@ -324,7 +376,9 @@ export class ShadowObserver { throw new Error(`buffer bound must be within 1..${MAX_BUFFER_BOUND}`) } const observerId = options.observerId ?? `${PROVIDER_ID}-${randomUUID()}` - if (!IDENTITY_TOKEN.test(observerId)) throw new Error('observer id must be an identity token') + if (!IDENTITY_TOKEN.test(observerId) || !isPublicSafeText(observerId)) { + throw new Error('observer id must be a public-safe identity token') + } this.config = { ...options.config, runIdentity: { ...options.config.runIdentity }, @@ -417,7 +471,11 @@ export class ShadowObserver { this.buffer = [] this.flushAttemptCount += 1 try { - const lines = [...taken, this.stats()].map(record => JSON.stringify(record)) + const records = [...taken, this.stats()] + if (!records.every(record => isPublicSafeValue(record))) { + throw new Error('observer public-safety invariant failed before append') + } + const lines = records.map(record => JSON.stringify(record)) await this.appendLines(this.path, lines) } catch (error: unknown) { this.observerFailureCount += 1 @@ -492,13 +550,17 @@ export class ShadowObserver { summary, source_refs: sourceRefs, } + if (!isPublicSafeValue(envelope)) { + this.reject('public_safety_violation') + return + } this.buffer.push(envelope) this.acceptedEventCount += 1 this.peakBufferedEventCount = Math.max(this.peakBufferedEventCount, this.buffer.length) if (this.buffer.length >= this.config.bufferBound) this.requestFlush() } - private reject(reason: string): void { + private reject(reason: ObserverRejectionReason): void { this.rejectedEventCount += 1 this.rejectedByReason.set(reason, (this.rejectedByReason.get(reason) ?? 0) + 1) } diff --git a/packages/dsh-loopx-plugin/tests/observer.spec.ts b/packages/dsh-loopx-plugin/tests/observer.spec.ts index 36b4610723..f120bbfac9 100644 --- a/packages/dsh-loopx-plugin/tests/observer.spec.ts +++ b/packages/dsh-loopx-plugin/tests/observer.spec.ts @@ -39,6 +39,19 @@ const config: ShadowObserverConfig = { ledgerDir: '/ledger', bufferBound: 4, } +interface PublicSafetyFixture { + readonly schema_version: string + readonly unsafe_summary_tokens: readonly string[] + readonly unsafe_identity_tokens: readonly string[] + readonly invalid_source_ref_tokens: readonly string[] + readonly safe_summary_tokens: readonly string[] + readonly safe_identity_tokens: readonly string[] +} +const TEST_DIR = dirname(fileURLToPath(import.meta.url)) +const publicSafetyFixture = JSON.parse(readFileSync(join( + TEST_DIR, + '../../../tests/fixtures/control_plane/reliability_diagnostics_public_safety_v0.json', +), 'utf8')) as PublicSafetyFixture const ENVELOPE_FIELDS = [ 'schema_version', 'capability_id', 'provider_id', 'observer_id', 'goal_id', 'session_id', 'sequence', 'observed_at', 'clock', 'event_kind', 'summary', 'source_refs', @@ -117,11 +130,18 @@ describe('shadow observer configuration', () => { [ENV_SESSION_ID]: sessionId, [ENV_RUN_IDENTITY]: JSON.stringify({ ...runIdentity, extra: 'not-pinned' }), })).toBeUndefined() + expect(resolveShadowObserverConfig({ + [ENV_GOAL_ID]: goalId, + [ENV_SESSION_ID]: sessionId, + [ENV_RUN_IDENTITY]: JSON.stringify({ + ...runIdentity, + worker_id: publicSafetyFixture.unsafe_identity_tokens[0], + }), + })).toBeUndefined() }) it('imports nothing from the driver and owns no send path', () => { - const here = dirname(fileURLToPath(import.meta.url)) - const source = readFileSync(join(here, '../src/observer.ts'), 'utf8') + const source = readFileSync(join(TEST_DIR, '../src/observer.ts'), 'utf8') expect(source).not.toMatch(/from '\.\/driver/u) expect(source).not.toMatch(/from '\.\/cli/u) expect(source).not.toMatch(/from '\.\/managed-runtime/u) @@ -192,6 +212,94 @@ describe('shadow observer envelopes', () => { ) }) + it('enforces shared public-safety counterfactuals before the first append', async () => { + expect(publicSafetyFixture.schema_version).toBe( + 'reliability_diagnostics_public_safety_counterfactuals_v0', + ) + const { appended, observer } = observerWithCapture({ bufferBound: 32 }) + const session = fakeSession() + let sequence = 1 + for (const value of publicSafetyFixture.unsafe_summary_tokens) { + observer.observeSessionEvent( + session, + sessionEvent('tool/call', sequence, 1_756_728_001_000 + sequence, { + turn: 1, step: sequence, callId: `call-${sequence}`, name: value, + }), + ) + sequence += 1 + } + for (const value of publicSafetyFixture.unsafe_summary_tokens) { + observer.observeSessionEvent( + session, + sessionEvent('tool/result', sequence, 1_756_728_001_000 + sequence, { + turn: 1, + step: sequence, + error: { code: value }, + message: { source: { callId: `call-${sequence}` } }, + }), + ) + sequence += 1 + } + for (const value of publicSafetyFixture.unsafe_identity_tokens) { + observer.observeSessionEvent( + session, + sessionEvent('tool/call', sequence, 1_756_728_001_000 + sequence, { + turn: 1, step: sequence, callId: value, name: 'bash', + }), + ) + sequence += 1 + } + for (const value of publicSafetyFixture.invalid_source_ref_tokens) { + observer.observeSessionEvent( + session, + sessionEvent('tool/call', sequence, 1_756_728_001_000 + sequence, { + turn: 1, step: sequence, callId: value, name: 'bash', + }), + ) + sequence += 1 + } + for (const value of publicSafetyFixture.safe_summary_tokens) { + observer.observeSessionEvent( + session, + sessionEvent('tool/call', sequence, 1_756_728_001_000 + sequence, { + turn: 1, step: sequence, callId: `call-${sequence}`, name: value, + }), + ) + sequence += 1 + } + for (const value of publicSafetyFixture.safe_identity_tokens) { + observer.observeSessionEvent( + session, + sessionEvent('tool/call', sequence, 1_756_728_001_000 + sequence, { + turn: 1, step: sequence, callId: value, name: 'bash', + }), + ) + sequence += 1 + } + await observer.flush() + + const ledgerBytes = appended.flat().join('\n') + for (const value of [ + ...publicSafetyFixture.unsafe_summary_tokens, + ...publicSafetyFixture.unsafe_identity_tokens, + ...publicSafetyFixture.invalid_source_ref_tokens, + ]) expect(ledgerBytes).not.toContain(value) + const records = parsed(appended) + const envelopes = records.filter( + (record): record is ObserverEnvelope => record.schema_version === OBSERVER_ENVELOPE_SCHEMA_VERSION, + ) + expect(envelopes).toHaveLength( + publicSafetyFixture.invalid_source_ref_tokens.length + + publicSafetyFixture.safe_summary_tokens.length + + publicSafetyFixture.safe_identity_tokens.length, + ) + const stats = records.at(-1) as ObserverStats + const unsafeCount = (publicSafetyFixture.unsafe_summary_tokens.length * 2) + + publicSafetyFixture.unsafe_identity_tokens.length + expect(stats.rejected_by_reason).toEqual({ public_safety_violation: unsafeCount }) + expect(stats.observed_event_count).toBe(envelopes.length + unsafeCount) + }) + it('drops with a count while a flush is in flight and the buffer is full', async () => { const appended: string[][] = [] let release: (() => void) | undefined diff --git a/tests/capabilities/test_reliability_diagnostics.py b/tests/capabilities/test_reliability_diagnostics.py index 1e4f25d4ce..03f46463a8 100644 --- a/tests/capabilities/test_reliability_diagnostics.py +++ b/tests/capabilities/test_reliability_diagnostics.py @@ -475,6 +475,18 @@ def test_receipt_invalidates_any_outbound_or_worker_context_path( assert ReceiptReason.OBSERVER_FAILURE.value in receipt["reason_codes"] +def test_receipt_quarantines_a_trailing_public_safety_rejection() -> None: + receipt = receipt_for( + envelope(0), + stats( + rejected_event_count=1, + rejected_by_reason={EnvelopeRejection.PUBLIC_SAFETY_VIOLATION.value: 1}, + ), + ) + assert receipt["status"] == ReceiptStatus.QUARANTINED.value + assert receipt["reason_codes"] == [ReceiptReason.PUBLIC_SAFETY_VIOLATION.value] + + @pytest.mark.parametrize( ("records", "reason"), [ diff --git a/tests/capabilities/test_reliability_diagnostics_dsh_provider.py b/tests/capabilities/test_reliability_diagnostics_dsh_provider.py index 7d5f73af20..b413c114e8 100644 --- a/tests/capabilities/test_reliability_diagnostics_dsh_provider.py +++ b/tests/capabilities/test_reliability_diagnostics_dsh_provider.py @@ -6,6 +6,8 @@ import re from pathlib import Path +import pytest + from loopx.capabilities.catalog import ( build_capability_catalog_packet, build_capability_detail_packet, @@ -16,7 +18,11 @@ ENVELOPE_FIELDS, OBSERVER_ENVELOPE_SCHEMA_VERSION, OBSERVER_STATS_SCHEMA_VERSION, + EnvelopeRejection, + ObserverEnvelopeError, ObserverEventKind, + dsh_fixture_records, + normalize_observer_envelope, ) from loopx.capabilities.reliability_diagnostics.intake import STATS_FIELDS @@ -26,14 +32,22 @@ CORDIS_PATCH = ROOT / "packages/dsh-loopx-plugin/cordis.patch.yml" PACKAGE_JSON = ROOT / "packages/dsh-loopx-plugin/package.json" TSDOWN_CONFIG = ROOT / "packages/dsh-loopx-plugin/tsdown.config.ts" +PUBLIC_SAFETY_FIXTURE = ROOT / ( + "tests/fixtures/control_plane/reliability_diagnostics_public_safety_v0.json" +) +PUBLIC_SAFETY_CASES = json.loads(PUBLIC_SAFETY_FIXTURE.read_text(encoding="utf-8")) def test_catalog_declares_dsh_provider_without_claiming_readiness() -> None: packet = build_capability_catalog_packet() - summary = next(item for item in packet["capabilities"] if item["id"] == CAPABILITY_ID) + summary = next( + item for item in packet["capabilities"] if item["id"] == CAPABILITY_ID + ) assert summary["provider_id"] == "loopx-core" assert summary["implementation_provider_count"] == 1 - provider = next(item for item in packet["providers"] if item["id"] == DSH_PROVIDER_ID) + provider = next( + item for item in packet["providers"] if item["id"] == DSH_PROVIDER_ID + ) assert provider == { "id": DSH_PROVIDER_ID, "origin": "extension", @@ -80,6 +94,72 @@ def test_typescript_observer_shares_field_names_and_has_no_control_path() -> Non assert "observation_entered_scheduler_inputs: false" in source +@pytest.mark.parametrize( + ("field", "value"), + [ + (field, value) + for field in ("tool_name", "error_class") + for value in PUBLIC_SAFETY_CASES["unsafe_summary_tokens"] + ], +) +def test_shared_counterfactual_rejects_unsafe_summary_token( + field: str, + value: str, +) -> None: + record = { + **dsh_fixture_records()[5], + "summary": {"turn": 1, "step": 1, field: value}, + "source_refs": {"event_seq": "5", "tool_call_id": "call-safe"}, + } + with pytest.raises(ObserverEnvelopeError) as excinfo: + normalize_observer_envelope(record) + assert excinfo.value.reason is EnvelopeRejection.PUBLIC_SAFETY_VIOLATION + + +@pytest.mark.parametrize("value", PUBLIC_SAFETY_CASES["unsafe_identity_tokens"]) +def test_shared_counterfactual_rejects_unsafe_source_ref(value: str) -> None: + record = { + **dsh_fixture_records()[5], + "summary": {"turn": 1, "step": 1, "tool_name": "bash"}, + "source_refs": {"event_seq": "5", "tool_call_id": value}, + } + with pytest.raises(ObserverEnvelopeError) as excinfo: + normalize_observer_envelope(record) + assert excinfo.value.reason is EnvelopeRejection.PUBLIC_SAFETY_VIOLATION + + +@pytest.mark.parametrize("value", PUBLIC_SAFETY_CASES["invalid_source_ref_tokens"]) +def test_shared_counterfactual_rejects_local_path_source_ref_shape(value: str) -> None: + record = { + **dsh_fixture_records()[5], + "summary": {"turn": 1, "step": 1, "tool_name": "bash"}, + "source_refs": {"event_seq": "5", "tool_call_id": value}, + } + with pytest.raises(ObserverEnvelopeError) as excinfo: + normalize_observer_envelope(record) + assert excinfo.value.reason is EnvelopeRejection.SOURCE_REF_INVALID + + +@pytest.mark.parametrize("value", PUBLIC_SAFETY_CASES["safe_summary_tokens"]) +def test_shared_counterfactual_allows_safe_summary_token(value: str) -> None: + record = { + **dsh_fixture_records()[5], + "summary": {"turn": 1, "step": 1, "tool_name": value}, + "source_refs": {"event_seq": "5", "tool_call_id": "call-safe"}, + } + assert normalize_observer_envelope(record).summary["tool_name"] == value + + +@pytest.mark.parametrize("value", PUBLIC_SAFETY_CASES["safe_identity_tokens"]) +def test_shared_counterfactual_allows_safe_source_ref(value: str) -> None: + record = { + **dsh_fixture_records()[5], + "summary": {"turn": 1, "step": 1, "tool_name": "bash"}, + "source_refs": {"event_seq": "5", "tool_call_id": value}, + } + assert normalize_observer_envelope(record).source_refs["tool_call_id"] == value + + def test_observer_is_a_separate_default_off_plugin_row() -> None: driver = DRIVER_TS.read_text(encoding="utf-8") observer = OBSERVER_TS.read_text(encoding="utf-8") diff --git a/tests/fixtures/control_plane/reliability_diagnostics_public_safety_v0.json b/tests/fixtures/control_plane/reliability_diagnostics_public_safety_v0.json new file mode 100644 index 0000000000..e70b9fc6a6 --- /dev/null +++ b/tests/fixtures/control_plane/reliability_diagnostics_public_safety_v0.json @@ -0,0 +1,28 @@ +{ + "schema_version": "reliability_diagnostics_public_safety_counterfactuals_v0", + "unsafe_summary_tokens": [ + "sk-abcdefghijklmnop0123", + "ghp_abcdefghijklmnopqrstuvwx", + "eyJabcdefghijk.abcdefghijk.abcdefghijk", + "C:/Users/alice/private.txt" + ], + "unsafe_identity_tokens": [ + "sk-abcdefghijklmnop0123", + "ghp_abcdefghijklmnopqrstuvwx", + "eyJabcdefghijk.abcdefghijk.abcdefghijk" + ], + "invalid_source_ref_tokens": [ + "C:/Users/alice/private.txt" + ], + "safe_summary_tokens": [ + "skill-python", + "token_count", + "tool/read-file", + "error-timeout" + ], + "safe_identity_tokens": [ + "skill-python", + "token_count", + "call-123" + ] +} From d3fb76b13ccfdbb4aa71d70cf423a4cc70f59665 Mon Sep 17 00:00:00 2001 From: song Date: Sat, 5 Sep 2026 03:11:22 +0800 Subject: [PATCH 12/13] refactor(dsh-loopx-plugin): simplify observer safety patterns Signed-off-by: song --- packages/dsh-loopx-plugin/src/observer.ts | 25 +++++++++++++++++------ 1 file changed, 19 insertions(+), 6 deletions(-) diff --git a/packages/dsh-loopx-plugin/src/observer.ts b/packages/dsh-loopx-plugin/src/observer.ts index 3e6b33dcf5..40533332fc 100644 --- a/packages/dsh-loopx-plugin/src/observer.ts +++ b/packages/dsh-loopx-plugin/src/observer.ts @@ -69,8 +69,19 @@ export const RUN_IDENTITY_FIELDS = [ const IDENTITY_TOKEN = /^[A-Za-z0-9][A-Za-z0-9_.:-]{0,120}$/u const SUMMARY_TOKEN = /^[A-Za-z0-9][A-Za-z0-9_./:-]{0,79}$/u -const LOCAL_PATH_SURFACE = /(?]+|[A-Za-z]:[\\/](?:Users|Documents and Settings)[\\/][^\s`'"<>]+)/iu -const SECRET_LIKE_SURFACE = /(?:\bbearer\s+[a-z0-9._~+\/=-]{16,}|\b(?:access|secret)[_-]?key\s*[=:]\s*[^\s`'"<>]+|\b(?:ak|sk)\s*[=:]\s*[^\s`'"<>]+|(?]{12,})/iu +const LOCAL_PATH_SURFACES = [ + /(?]+/iu, + /(?]+/iu, +] as const +const SECRET_LIKE_SURFACES = [ + /\bbearer\s+[a-z0-9._~+/=-]{16,}/iu, + /\b(?:access|secret)[_-]?key\s*[=:]\s*[^\s`'"<>]+/iu, + /\b(?:ak|sk)\s*[=:]\s*[^\s`'"<>]+/iu, + /(?]{12,}/iu, +] as const const CREDENTIAL_FIELD_FAMILIES = new Set([ 'accesskey', 'accesstoken', 'apikey', 'authtoken', 'authorization', 'clientsecret', 'cookie', 'credential', 'credentials', 'password', @@ -176,15 +187,17 @@ export interface ShadowObserverOptions { } function normalizePublicSafeFieldName(value: string): string { - return value + const normalized = value .replace(/([a-z0-9])([A-Z])/gu, '$1_$2') - .replace(/[^A-Za-z0-9]+/gu, '_') - .replace(/^_+|_+$/gu, '') + .replace(/[^a-z0-9]+/giu, '_') .toLowerCase() + const withoutLeading = normalized.startsWith('_') ? normalized.slice(1) : normalized + return withoutLeading.endsWith('_') ? withoutLeading.slice(0, -1) : withoutLeading } function isPublicSafeText(value: string): boolean { - return !LOCAL_PATH_SURFACE.test(value) && !SECRET_LIKE_SURFACE.test(value) + return !LOCAL_PATH_SURFACES.some(pattern => pattern.test(value)) + && !SECRET_LIKE_SURFACES.some(pattern => pattern.test(value)) } /** Keep the producer boundary equivalent to LoopX's recursive Python guard. */ From f9dc0d999aff9a7e3f0218e382c3e1f3731b7004 Mon Sep 17 00:00:00 2001 From: song Date: Sat, 5 Sep 2026 23:23:57 +0800 Subject: [PATCH 13/13] docs(rfc): record the P0 shadow-observer checkpoint and DSH event-source decision MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Update the reliability-diagnostics RFC alongside the implementation, as requested in review, in both language mirrors: - §12 P0: add a dated checkpoint stating which P0 deliverables now exist as the default-off `reliability-diagnostics` capability and the `dsh-session-events` provider, and which remain open before P0 exit (an eligible C1 run on a real `dsh` session, the overhead report, and the retention/deletion profile from decision 4). - §14 decision 2: record DeepSeek Harness session events as the P0 event source, with the reason (typed `dsh` Turn host plus a same-session plugin with read-only hooks inside an existing packaged boundary). Pi stays the comparison candidate; the harness-selection evaluation shared with the Desktop Execution Frontends RFC is named as a follow-up deliverable. - §15: relate the observer to Desktop Execution Frontends Mode B as the passive diagnostic layer under the Desktop-owned runtime supervisor, with no supervisor authority. The capability README (English and Chinese) gains a matching "Relationship To The RFCs" section. No code, schema, CLI, or default behavior changes. Validation: - loopx canary premerge --from-git-diff --git-diff-base : 18 checks; the single failure (cli-output-budget-regression-smoke) is a local git 2.15 on PATH lacking `git worktree remove` and passes with a current git - git diff --check clean; DCO trailer present Signed-off-by: song --- ...bility-diagnostics-governed-delivery-v0.md | 23 +++++++++++++++++++ ...-diagnostics-governed-delivery-v0.zh-CN.md | 15 ++++++++++++ .../reliability_diagnostics/README.md | 12 ++++++++++ .../reliability_diagnostics/README.zh-CN.md | 9 ++++++++ 4 files changed, 59 insertions(+) diff --git a/docs/architecture/rfcs/long-running-agent-reliability-diagnostics-governed-delivery-v0.md b/docs/architecture/rfcs/long-running-agent-reliability-diagnostics-governed-delivery-v0.md index 1b89217dff..2de76224da 100644 --- a/docs/architecture/rfcs/long-running-agent-reliability-diagnostics-governed-delivery-v0.md +++ b/docs/architecture/rfcs/long-running-agent-reliability-diagnostics-governed-delivery-v0.md @@ -524,6 +524,16 @@ source. Prove the no-outbound-control invariant and bounded failure behavior. **Exit:** C0 adapter fidelity plus an eligible C1 observer run; public/private boundary and overhead are reported; no production authority exists. +**Checkpoint (2026-09):** the contract half of P0 exists as the default-off +built-in capability `reliability-diagnostics` with the extension provider +`dsh-session-events` in `packages/dsh-loopx-plugin`: provider-neutral +envelope and stats records, integrity receipt, read-only diagnostic +projection, a deterministic DSH-shaped fixture, and producer-side +public-safety rejection before the first ledger append. Still open before P0 +exit: an eligible C1 observer run on a real `dsh` session, the reported +overhead measurement, and the ledger retention and deletion profile from +decision 4 below. + ### P1 — Benchmark-qualified diagnostic pilot Run matched native and L1 arms on at least one suitable benchmark family and @@ -594,6 +604,13 @@ required before making a stronger commercial claim. pilot: software delivery, security/SRE, or research/AI4S? 2. Which event source and harness should define the P0 shadow-observer conformance fixture? + **Decided (2026-09): DeepSeek Harness (`dsh`) session events.** LoopX + already ships a typed `dsh` Turn host and a same-session plugin whose + read-only `session/event`, `agent/status`, `agent/error`, and + `session/disposed` hooks let the observer be proven non-interfering inside + an existing packaged boundary. Pi remains the comparison candidate; the + harness-selection evaluation shared with the Desktop Execution Frontends + RFC is a follow-up deliverable and will be recorded here. 3. Should the first two-to-four-week offer stop at L1 diagnostics by default, or include an optional L2 advisory week before any L3 seam? 4. Which data-retention, deletion, and support profiles belong in the first @@ -613,6 +630,12 @@ required before making a stronger commercial claim. owns benchmark truth, matched arms, C0–C4 evidence, and research integrity. - [Agent Management Observability MVP](../../product/surfaces/agent-management-observability-mvp.md) defines the read-only projection posture reused by L1/L2 operator surfaces. +- [Desktop Execution Frontends](./desktop-execution-frontends-v0.md) defines + Mode B, the Managed Agent Runtime in which LoopX Desktop launches and + supervises Pi or `dsh`. The L1 shadow observer is the passive diagnostic + layer under that mode's Desktop-owned runtime supervisor: its integrity + receipt and read-only projection are inputs the supervisor may project, and + the observer acquires none of the supervisor's authority. - [Shared Goal Authority and State Provider](./shared-goal-authority-state-provider-v0.md) defines the authority/provider boundary required only when L3/L4 uses shared coordination. diff --git a/docs/architecture/rfcs/long-running-agent-reliability-diagnostics-governed-delivery-v0.zh-CN.md b/docs/architecture/rfcs/long-running-agent-reliability-diagnostics-governed-delivery-v0.zh-CN.md index c83a916d81..1df5f918f7 100644 --- a/docs/architecture/rfcs/long-running-agent-reliability-diagnostics-governed-delivery-v0.zh-CN.md +++ b/docs/architecture/rfcs/long-running-agent-reliability-diagnostics-governed-delivery-v0.zh-CN.md @@ -447,6 +447,13 @@ to pay 的证明。 **Exit:** C0 adapter fidelity 加一条 eligible C1 observer run;报告 public/private boundary 与 overhead;不存在 production authority。 +**Checkpoint(2026-09):** P0 的 contract 部分已以默认关闭的 built-in capability +`reliability-diagnostics` 与 extension provider `dsh-session-events`(位于 +`packages/dsh-loopx-plugin`)落地:provider-neutral envelope 与 stats record、integrity receipt、 +read-only diagnostic projection、deterministic DSH-shaped fixture,以及首次写入 ledger 之前的 +producer 侧 public-safety 拒绝。P0 exit 之前仍未完成:在真实 `dsh` session 上的 eligible C1 +observer run、overhead 测量报告,以及下文 decision 4 的 ledger retention 与 deletion profile。 + ### P1 — Benchmark-qualified diagnostic pilot 在至少一个合适 benchmark family 和一个 non-benchmark rehearsal 上运行 matched native/L1 arm。 @@ -506,6 +513,10 @@ advantage 与 sustainable delivery evidence。 1. 首个产品 pilot 应选择哪个 initial ICP 与 reference workflow:software delivery、security/SRE, 还是 research/AI4S? 2. 哪个 event source 与 harness 应定义 P0 shadow-observer conformance fixture? + **已决定(2026-09):DeepSeek Harness(`dsh`)session events。** LoopX 已有 typed `dsh` + Turn host 与 same-session plugin,其只读 `session/event`、`agent/status`、`agent/error`、 + `session/disposed` hook 让 observer 能在既有打包边界内被证明 non-interfering。Pi 仍是 + 对比候选;与 Desktop Execution Frontends RFC 共享的 harness 选型评估是后续交付物,结论将记录在此。 3. 第一份两到四周 offer 默认应停在 L1 diagnostic,还是在进入任何 L3 seam 前增加可选 L2 advisory week? 4. 第一份 local/private/BYOC deployment pack 应包含哪些 data-retention、deletion 与 support profile? 5. 第一份 promotion packet 必须使用哪个 benchmark family 与 non-benchmark canary? @@ -520,6 +531,10 @@ advantage 与 sustainable delivery evidence。 拥有 benchmark truth、matched arm、C0–C4 evidence 与 research integrity。 - [Agent Management Observability MVP](../../product/surfaces/agent-management-observability-mvp.md) 定义 L1/L2 operator surface 复用的 read-only projection posture。 +- [Desktop Execution Frontends](./desktop-execution-frontends-v0.zh-CN.md) 定义 Mode B,即由 LoopX + Desktop 启动并监督 Pi 或 `dsh` 的 Managed Agent Runtime。L1 shadow observer 是该模式下 + Desktop-owned runtime supervisor 之下的被动诊断层:其 integrity receipt 与 read-only projection + 是 supervisor 可以投影的输入,observer 本身不获得 supervisor 的任何 authority。 - [Shared Goal Authority 与 State Provider](./shared-goal-authority-state-provider-v0.zh-CN.md) 定义只有在 L3/L4 使用 shared coordination 时才需要的 authority/provider boundary。 - [TypeScript Control-Plane Migration](./typescript-control-plane-migration-v0.zh-CN.md) diff --git a/loopx/capabilities/reliability_diagnostics/README.md b/loopx/capabilities/reliability_diagnostics/README.md index a5afbc3b79..588d1cc99c 100644 --- a/loopx/capabilities/reliability_diagnostics/README.md +++ b/loopx/capabilities/reliability_diagnostics/README.md @@ -55,6 +55,18 @@ hook isolation, not an OS-process-isolation claim. and the `SOURCE_ID_KEYS` identity tuple; the session-runtime substring classifier is deliberately not reused. +## Relationship To The RFCs + +- [Long-Running Agent Reliability Diagnostics](../../../docs/architecture/rfcs/long-running-agent-reliability-diagnostics-governed-delivery-v0.md) + owns this capability. This slice is the P0 contract checkpoint recorded in + its roadmap; the `dsh` event source is the recorded answer to owner + decision 2, and the C1 run, overhead report, and retention profile remain + open before P0 exit. +- [Desktop Execution Frontends](../../../docs/architecture/rfcs/desktop-execution-frontends-v0.md) + Mode B is the managed runtime this observer is built for: the Desktop-owned + runtime supervisor may consume the receipt and projection as diagnostic + inputs, while the observer keeps no supervisor authority. + ## Contract ### Observer envelope (`reliability_observer_envelope_v0`) diff --git a/loopx/capabilities/reliability_diagnostics/README.zh-CN.md b/loopx/capabilities/reliability_diagnostics/README.zh-CN.md index 9e8f8efc0c..d934977518 100644 --- a/loopx/capabilities/reliability_diagnostics/README.zh-CN.md +++ b/loopx/capabilities/reliability_diagnostics/README.zh-CN.md @@ -42,6 +42,15 @@ Driver 或 Agent,只消费 session log 的发布事件,并且不进入 Drive - **辅助逻辑留在本包内。** ledger、receipt、projection reducer 都在本包。仅共享 public-safe 值校验器与 `SOURCE_ID_KEYS` 身份键;刻意不复用 session-runtime 的子串分类器。 +## 与 RFC 的关系 + +- [Long-Running Agent Reliability Diagnostics](../../../docs/architecture/rfcs/long-running-agent-reliability-diagnostics-governed-delivery-v0.zh-CN.md) + 拥有这个 capability。本切片是其路线图中记录的 P0 contract checkpoint;`dsh` event source 是 + owner decision 2 的记录答案,C1 run、overhead 报告与 retention profile 在 P0 exit 前仍未完成。 +- [Desktop Execution Frontends](../../../docs/architecture/rfcs/desktop-execution-frontends-v0.zh-CN.md) + 的 Mode B 是这个 observer 面向的 managed runtime:Desktop-owned runtime supervisor 可以把 + receipt 与 projection 作为诊断输入消费,observer 不持有任何 supervisor authority。 + ## 合约 ### Observer envelope(`reliability_observer_envelope_v0`)