Skip to content

Commit 92fe93a

Browse files
authored
fix(evaluation): reject empty intake answers and protocol fallback (#5434)
Signed-off-by: huangruiteng <huangrt01@163.com>
1 parent a7c21ba commit 92fe93a

3 files changed

Lines changed: 138 additions & 8 deletions

File tree

‎docs/architecture/rfcs/app-conversation-and-async-inbox-v0.md‎

Lines changed: 8 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -544,9 +544,14 @@ skips rather than presenting an unqualified profile as passing. The suite uses
544544
the production prompt/parser and fixed public contexts. Its frozen outcomes
545545
cover new work, existing owners, ambiguity, stopped/ungranted recipients, scope
546546
correction, current-Goal follow-ups, quotes, negation and ordinary questions in
547-
Chinese/English. Output conflicts, omitted envelopes and truncated generations
548-
fail rather than being counted as successful intent recognition. Report model,
549-
prompt/case hashes, request settings, token usage and repeat count. This layer
547+
Chinese/English. An empty answer fails for either provider. The API lane also
548+
checks raw envelope conflicts, omitted tags and truncated generations; the
549+
Codex lane consumes adapter protocol warnings, including missing tags, rather
550+
than certifying its readable fallback. Codex raw envelope integrity beyond the
551+
adapter's warnings remains unqualified. Both receive the same fixture context
552+
and normalized request. Report model, template and per-case effective prompt
553+
hashes, case hash, request settings, available token usage and repeat count.
554+
Offline provider fixtures exercise the CLI and report without paid calls. This layer
550555
qualifies model interpretation of supplied evidence, not live discovery, actual
551556
dispatch, stop enforcement or full GQ01/GQ02 completion. Packaged browser and
552557
real collaboration transport tests qualify those separate boundaries.

‎examples/evaluations/chat-intake.py‎

Lines changed: 24 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -23,6 +23,9 @@
2323
def score(case, response):
2424
observed = "handoff" if response.get("context_handoff") else "draft" if response.get("goal_draft") else "answer"
2525
errors = []
26+
message = response.get("message")
27+
if not isinstance(message, str) or not message.strip():
28+
errors.append("missing_answer")
2629
if observed != case["expected"]:
2730
errors.append(f"expected_{case['expected']}_got_{observed}")
2831
if response.get("proposals") or response.get("protected_action"):
@@ -70,19 +73,36 @@ def main():
7073

7174
def run(case):
7275
started = time.monotonic()
73-
prompt = _turn_prompt(case["request"], context_summary=json.dumps(case["context"], ensure_ascii=False))
76+
# Match the real Codex adapter's input normalization for both providers.
77+
message = " ".join(case["request"].split())
78+
context = json.dumps(case["context"], ensure_ascii=False)
79+
prompt = _turn_prompt(message, context_summary=context)
80+
prompt_sha256 = hashlib.sha256(prompt.encode()).hexdigest()
7481
if args.provider == "codex":
82+
protocol_warning = False
83+
84+
def observe(event, _payload):
85+
nonlocal protocol_warning
86+
# Keep only the fact; provider payloads can contain secrets.
87+
if event == "protocol.warning":
88+
protocol_warning = True
89+
7590
try:
7691
with tempfile.TemporaryDirectory(prefix="loopx-public-intake-") as work:
7792
with CodexChatAgentSession.start(
7893
codex_bin="codex", work_dir=Path(work), goal_id="intake-evaluation",
7994
objective=json.dumps(case["context"], ensure_ascii=False), model=args.model,
8095
reasoning_effort="high", hard_timeout_sec=180,
8196
) as session:
82-
row = score(case, session.send(case["request"]))
97+
session.context_summary = context
98+
row = score(case, session.send(message, on_event=observe))
99+
if protocol_warning:
100+
row["passed"] = False
101+
row["errors"].append("protocol_warning")
83102
except Exception as error:
84103
row = {"id": case["id"], "passed": False, "errors": [type(error).__name__]}
85104
row["seconds"] = round(time.monotonic() - started, 2)
105+
row["prompt_sha256"] = prompt_sha256
86106
print(json.dumps(row), flush=True)
87107
return row
88108
body = {"model": args.model, "messages": [{"role": "user", "content": prompt}],
@@ -111,6 +131,7 @@ def run(case):
111131
# Provider errors can contain secrets, URLs or echoed prompts. Store only type.
112132
row = {"id": case["id"], "passed": False, "errors": [type(error).__name__]}
113133
row["seconds"] = round(time.monotonic() - started, 2)
134+
row["prompt_sha256"] = prompt_sha256
114135
print(json.dumps({k: row[k] for k in ("id", "passed", "errors")}), flush=True)
115136
return row
116137

@@ -120,7 +141,7 @@ def run(case):
120141
"prompt_sha256": hashlib.sha256(_turn_prompt("").encode()).hexdigest(),
121142
"request_settings": {"reasoning_effort": "high", "runtime_profile": "restricted"} if args.provider == "codex" else {"temperature": 0, "max_tokens": 8192, "thinking": "provider_default"},
122143
"passed": sum(row["passed"] for row in rows), "total": len(rows), "results": rows,
123-
"boundary": "Fixed public context with production prompt/parser; synthetic authoritative observations test intake, not real fact lookup. Scoring checks effects, recipients and evidence pointers; review factual conclusions separately. Codex uses the real restricted Chat adapter; operator-api also checks raw envelope integrity. No dynamic discovery, dispatch or work completion qualification."}
144+
"boundary": "Fixed public context with production prompt/parser; synthetic authoritative observations test intake, not real fact lookup. Top-level prompt_sha256 identifies the template; each result hashes its effective prompt including context and normalized request. Scoring requires a nonempty answer and checks effects, recipients and evidence pointers; review factual conclusions separately. Codex uses the real restricted Chat adapter and fails on its protocol warnings; operator-api also checks raw envelope integrity. No dynamic discovery, dispatch or work completion qualification."}
124145
args.output.parent.mkdir(parents=True, exist_ok=True)
125146
args.output.write_text(json.dumps(report, ensure_ascii=False, indent=2) + "\n")
126147
return 0 if report["passed"] == report["total"] else 1

‎tests/test_chat_intake_evaluation.py‎

Lines changed: 106 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,10 @@
11
"""Offline oracles for the opt-in model evaluation; no credentials or paid calls."""
22
import importlib.util
3+
import json
34
from pathlib import Path
5+
import sys
6+
7+
import pytest
48

59
ROOT = Path(__file__).resolve().parents[1]
610
spec = importlib.util.spec_from_file_location("chat_intake_eval", ROOT / "examples/evaluations/chat-intake.py")
@@ -25,8 +29,8 @@ def test_existing_work_requires_the_right_recipient_without_a_second_gate():
2529
def test_complete_new_request_does_not_need_another_question():
2630
case = {"id": "new", "expected": "draft", "ready": True}
2731
draft = {"completion_criteria": "Cited report", "question": ""}
28-
assert evaluation.score(case, {"goal_draft": draft})["passed"]
29-
assert not evaluation.score(case, {"goal_draft": {**draft, "question": "Ready to begin?"}})["passed"]
32+
assert evaluation.score(case, {"message": "Here is the draft.", "goal_draft": draft})["passed"]
33+
assert not evaluation.score(case, {"message": "Here is the draft.", "goal_draft": {**draft, "question": "Ready to begin?"}})["passed"]
3034

3135

3236
def test_ordinary_question_cannot_silently_become_work():
@@ -61,3 +65,103 @@ def test_shared_intake_guidance_precedes_handoff_without_widening_runtime():
6165
for profile in ["restricted", "trusted_owner"]:
6266
assert CONVERSATION_INTENT_RESOLUTION_INSTRUCTION in manager_agent_objective(profile)
6367
assert CONVERSATION_INTENT_RESOLUTION_INSTRUCTION not in _turn_prompt("Execute.", execution_mode=True)
68+
69+
70+
@pytest.mark.parametrize("message", [None, "", " \n\t", 7, ["Answer"]])
71+
def test_an_ordinary_answer_needs_visible_text(message):
72+
row = evaluation.score({"id": "ordinary", "expected": "answer"}, {"message": message})
73+
assert not row["passed"] and "missing_answer" in row["errors"]
74+
75+
76+
@pytest.mark.parametrize("provider", ["codex", "operator-api"])
77+
@pytest.mark.parametrize("message", ["A sourced explanation.", ""])
78+
def test_release_cli_scores_provider_output_and_hashes_effective_prompt(
79+
tmp_path, monkeypatch, provider, message
80+
):
81+
"""Exercise the real CLI/report flow; only provider inference is simulated."""
82+
from contextlib import contextmanager
83+
import hashlib
84+
from io import BytesIO
85+
86+
suite = {"contexts": {"ordinary": {"scope": "public", "goals": []}},
87+
"cases": [{"id": "ordinary", "context_ref": "ordinary", "expected": "answer",
88+
"request": "Explain\n what a Goal means."}]}
89+
cases = tmp_path / "cases.json"
90+
cases.write_text(json.dumps(suite))
91+
output = tmp_path / "report.json"
92+
monkeypatch.setattr(sys, "argv", ["chat-intake", "--live", "--provider", provider,
93+
"--model", "fixture-model", "--cases", str(cases), "--output", str(output)])
94+
monkeypatch.setattr(evaluation, "operator_provider_environ", lambda _: {"DEEPSEEK_API_KEY": "synthetic-test-only"})
95+
observed = []
96+
response = {"message": message, "proposals": [], "protected_action": None, "gate": None}
97+
98+
class CodexFixture:
99+
context_summary = "wrong startup context"
100+
101+
def send(self, request, *, on_event):
102+
observed.append(evaluation._turn_prompt(request, context_summary=self.context_summary))
103+
on_event("answer.final", {"response": response})
104+
return response
105+
106+
@contextmanager
107+
def start(**kwargs):
108+
assert kwargs["model"] == "fixture-model"
109+
yield CodexFixture()
110+
111+
def urlopen(request, **_):
112+
observed.append(json.loads(request.data)["messages"][0]["content"])
113+
envelope = evaluation.CHAT_REVIEW_OPEN_TAG + json.dumps(response) + evaluation.CHAT_REVIEW_CLOSE_TAG
114+
return BytesIO(json.dumps({"choices": [{"finish_reason": "stop", "message": {"content": envelope}}]}).encode())
115+
116+
monkeypatch.setattr(evaluation.CodexChatAgentSession, "start", start)
117+
monkeypatch.setattr(evaluation.urllib.request, "urlopen", urlopen)
118+
assert evaluation.main() == (0 if message else 1)
119+
report = json.loads(output.read_text())
120+
assert report["total"] == len(observed) == 1
121+
expected = evaluation._turn_prompt("Explain what a Goal means.", context_summary=json.dumps(suite["contexts"]["ordinary"], ensure_ascii=False))
122+
assert observed == [expected], "Both providers must receive the same case context and request"
123+
assert report["results"][0]["prompt_sha256"] == hashlib.sha256(expected.encode()).hexdigest()
124+
assert report["passed"] == bool(message)
125+
assert "synthetic-test-only" not in output.read_text()
126+
127+
128+
def test_codex_protocol_warning_fails_even_with_a_readable_fallback(tmp_path, monkeypatch, capsys):
129+
from contextlib import contextmanager
130+
131+
cases = tmp_path / "cases.json"
132+
cases.write_text(json.dumps({"contexts": {"ordinary": {}}, "cases": [
133+
{"id": "ordinary", "context_ref": "ordinary", "expected": "answer", "request": "Explain Goals."}]}))
134+
output = tmp_path / "report.json"
135+
monkeypatch.setattr(sys, "argv", ["chat-intake", "--live", "--provider", "codex", "--model", "fixture-model",
136+
"--cases", str(cases), "--output", str(output)])
137+
monkeypatch.setattr(evaluation, "operator_provider_environ", lambda _: {})
138+
139+
class WarningFixture:
140+
def send(self, request, *, on_event):
141+
on_event("protocol.warning", {"error_code": "missing_review_envelope", "detail": "do-not-persist-provider-payload"})
142+
return {"message": "A readable fallback that is not protocol-qualified."}
143+
144+
@contextmanager
145+
def start(**_):
146+
yield WarningFixture()
147+
148+
monkeypatch.setattr(evaluation.CodexChatAgentSession, "start", start)
149+
assert evaluation.main() == 1
150+
row = json.loads(output.read_text())["results"][0]
151+
assert row["errors"] == ["protocol_warning"] and not row["passed"]
152+
assert "do-not-persist-provider-payload" not in output.read_text() + capsys.readouterr().out
153+
154+
155+
def test_release_cli_without_live_never_reads_credentials_or_calls_provider(tmp_path, monkeypatch):
156+
monkeypatch.setattr(sys, "argv", ["chat-intake", "--model", "fixture-model", "--output", str(tmp_path / "report.json")])
157+
158+
def forbidden(*_, **__):
159+
pytest.fail("No provider or credential access without explicit live opt-in")
160+
161+
monkeypatch.setattr(evaluation, "operator_provider_environ", forbidden)
162+
monkeypatch.setattr(evaluation.CodexChatAgentSession, "start", forbidden)
163+
monkeypatch.setattr(evaluation.urllib.request, "urlopen", forbidden)
164+
with pytest.raises(SystemExit) as error:
165+
evaluation.main()
166+
assert error.value.code == 2
167+
assert not (tmp_path / "report.json").exists()

0 commit comments

Comments
 (0)