From 880774de4cab4c5a488d44ab08eb1f8ebf04eaf6 Mon Sep 17 00:00:00 2001 From: Stanley Phoong Date: Sat, 8 Aug 2026 17:39:43 -0700 Subject: [PATCH 1/4] feat(swe-bench): all-or-nothing merge gate scoped to exactly one run id merge_run(wq, run_id) refuses to emit an accuracy number unless every planned unit has a terminal result, none is abandoned, every unit accounts for exactly its planned instance IDS (a set comparison, never a count), the union equals the plan with no cross-shard duplicates, every plan_digest matches, and no unit carries an infra error. Refusal is a structured MergeRefusal naming the offending units and ids; there is no force flag and no partial-credit path. There is deliberately no --all: merge_run takes a required run id and treats a foreign run id or digest as a hard error, not a skip. verify_inventory() cross-checks claims, results and the id-union as independent producers, so a blind spot shared by one instrument cannot certify itself. --- .../swe_bench_distributed/__init__.py | 5 + .../evaluation/swe_bench_distributed/merge.py | 230 ++++++++++++++++++ .../swe_bench_distributed/test_merge.py | 189 ++++++++++++++ 3 files changed, 424 insertions(+) create mode 100644 src/inference_endpoint/evaluation/swe_bench_distributed/merge.py create mode 100644 tests/unit/evaluation/swe_bench_distributed/test_merge.py diff --git a/src/inference_endpoint/evaluation/swe_bench_distributed/__init__.py b/src/inference_endpoint/evaluation/swe_bench_distributed/__init__.py index 2042c3d55..3507289bb 100644 --- a/src/inference_endpoint/evaluation/swe_bench_distributed/__init__.py +++ b/src/inference_endpoint/evaluation/swe_bench_distributed/__init__.py @@ -11,6 +11,7 @@ exactly once. """ +from .merge import MergeRefusal, MergeResult, merge_run, verify_inventory from .queue import ( ClaimError, UnitOutcome, @@ -21,10 +22,14 @@ __all__ = [ "ClaimError", + "MergeRefusal", + "MergeResult", "Unit", "UnitOutcome", "UnitPlan", "UnitResult", "WorkQueue", + "merge_run", "plan_units", + "verify_inventory", ] diff --git a/src/inference_endpoint/evaluation/swe_bench_distributed/merge.py b/src/inference_endpoint/evaluation/swe_bench_distributed/merge.py new file mode 100644 index 000000000..a8dbee7cf --- /dev/null +++ b/src/inference_endpoint/evaluation/swe_bench_distributed/merge.py @@ -0,0 +1,230 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""The merge gate: refuse to emit an accuracy unless every id is accounted for. + +The single most important property of a sharded accuracy run is that it never +divides the results of 190 instances by 200. The gate is all-or-nothing by +design: there is no force flag and no partial-credit path, because a partial +number is indistinguishable from a real one once it leaves this module. + +The gate is also scoped to exactly one run. There is no ``merge_all``. Merging +"every run that looks finished" once re-merged hundreds of banked results +belonging to unrelated configurations into one number. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Any + +from .queue import UnitOutcome, UnitResult, WorkQueue +from .units import UnitPlan + + +class MergeRefusal(RuntimeError): + """The gate refused to produce an accuracy number. + + ``reasons`` lists every independent failure, so one merge attempt reports + everything wrong rather than the first thing wrong. + """ + + def __init__(self, run_id: str, reasons: list[str]) -> None: + self.run_id = run_id + self.reasons = reasons + super().__init__( + f"refusing to score run {run_id!r}: " + + "; ".join(reasons[:10]) + + (f" (+{len(reasons) - 10} more)" if len(reasons) > 10 else "") + ) + + +@dataclass(slots=True) +class MergeResult: + run_id: str + plan_digest: str + total_instances: int + resolved_instances: int + unit_count: int + + @property + def resolved_rate(self) -> float: + return self.resolved_instances / self.total_instances + + def to_dict(self) -> dict[str, Any]: + return { + "run_id": self.run_id, + "plan_digest": self.plan_digest, + "total_instances": self.total_instances, + "resolved_instances": self.resolved_instances, + "resolved_rate": self.resolved_rate, + "unit_count": self.unit_count, + } + + +@dataclass(slots=True) +class InventoryReport: + """Cross-check of three independently produced views of the same run. + + The units the plan asked for, the units the queue recorded results for, and + the instance ids those results claim to have covered are produced by + different code paths. Checking one against itself is how a verification pass + can agree with a broken system: the instrument shares the blind spot. These + must agree with each other. + """ + + missing_units: list[str] = field(default_factory=list) + foreign_units: list[str] = field(default_factory=list) + unreadable_units: list[str] = field(default_factory=list) + ownerless_claims: list[str] = field(default_factory=list) + claims_without_results: list[str] = field(default_factory=list) + + @property + def consistent(self) -> bool: + return not ( + self.missing_units + or self.foreign_units + or self.unreadable_units + or self.ownerless_claims + ) + + +def verify_inventory(queue: WorkQueue) -> InventoryReport: + """Compare the plan, the claim directory and the result directory.""" + report = InventoryReport() + plan_units = set(queue.plan.unit_ids) + + result_files = {path.stem for path in queue.results_dir.glob("*.json")} + report.foreign_units = sorted(result_files - plan_units) + report.missing_units = sorted(plan_units - result_files) + for unit_id in sorted(result_files & plan_units): + if queue.result(unit_id) is None: + report.unreadable_units.append(unit_id) + + for unit_id in sorted(queue.claimed_unit_ids()): + if queue.owner(unit_id) is None: + report.ownerless_claims.append(unit_id) + if unit_id not in result_files: + report.claims_without_results.append(unit_id) + return report + + +def merge_run(queue: WorkQueue, run_id: str) -> MergeResult: + """Score one run, or refuse. + + ``run_id`` is required and must match the queue's plan. Passing another + run's id is an error, not a filter. + """ + plan: UnitPlan = queue.plan + if run_id != plan.run_id: + raise MergeRefusal( + run_id, + [ + f"queue at {queue.root} holds run {plan.run_id!r}, not {run_id!r}; " + "a merge is always scoped to exactly one run" + ], + ) + + reasons: list[str] = [] + inventory = verify_inventory(queue) + if inventory.foreign_units: + reasons.append( + "results present for units outside the plan: " + + ", ".join(inventory.foreign_units[:5]) + ) + if inventory.unreadable_units: + reasons.append( + "unreadable result records: " + ", ".join(inventory.unreadable_units[:5]) + ) + if inventory.ownerless_claims: + reasons.append( + "claims with no readable owner: " + + ", ".join(inventory.ownerless_claims[:5]) + ) + if inventory.missing_units: + reasons.append( + f"{len(inventory.missing_units)} of {len(plan.units)} units have no " + "result: " + ", ".join(inventory.missing_units[:5]) + ) + + results: dict[str, UnitResult] = queue.results() + seen_ids: dict[str, str] = {} + resolved: set[str] = set() + + for unit in plan.units: + result = results.get(unit.unit_id) + if result is None: + continue + if result.plan_digest != plan.digest: + reasons.append( + f"{unit.unit_id}: result belongs to plan {result.plan_digest[:12]}, " + f"not {plan.digest[:12]}" + ) + continue + if result.abandoned: + reasons.append(f"{unit.unit_id}: abandoned after {result.attempt} attempts") + continue + if result.outcome is not UnitOutcome.SUCCEEDED: + reasons.append(f"{unit.unit_id}: outcome {result.outcome.value}") + continue + if result.infra_error_count > 0: + reasons.append( + f"{unit.unit_id}: {result.infra_error_count} instance(s) lost to " + "infrastructure" + ) + continue + + expected = set(unit.instance_ids) + accounted = set(result.accounted_instance_ids) + if len(result.accounted_instance_ids) != len(accounted): + reasons.append(f"{unit.unit_id}: duplicate instance ids in its own result") + continue + # Compare ids, never counts. A shard with one duplicate and one missing + # id has the right count and the wrong content. + if accounted != expected: + missing = sorted(expected - accounted) + extra = sorted(accounted - expected) + detail = [] + if missing: + detail.append(f"missing {', '.join(missing[:5])}") + if extra: + detail.append(f"unplanned {', '.join(extra[:5])}") + reasons.append(f"{unit.unit_id}: " + "; ".join(detail)) + continue + + for instance_id in result.accounted_instance_ids: + previous = seen_ids.get(instance_id) + if previous is not None: + reasons.append( + f"instance {instance_id} accounted for by both {previous} and " + f"{unit.unit_id}" + ) + continue + seen_ids[instance_id] = unit.unit_id + unplanned_resolved = set(result.resolved_instance_ids) - expected + if unplanned_resolved: + reasons.append( + f"{unit.unit_id}: resolved ids outside its shard: " + + ", ".join(sorted(unplanned_resolved)[:5]) + ) + continue + resolved.update(result.resolved_instance_ids) + + planned_ids = set(plan.instance_ids) + if not reasons and set(seen_ids) != planned_ids: + unaccounted = sorted(planned_ids - set(seen_ids)) + reasons.append( + f"{len(unaccounted)} planned instance(s) unaccounted for: " + + ", ".join(unaccounted[:5]) + ) + + if reasons: + raise MergeRefusal(run_id, reasons) + + return MergeResult( + run_id=run_id, + plan_digest=plan.digest, + total_instances=len(planned_ids), + resolved_instances=len(resolved), + unit_count=len(plan.units), + ) diff --git a/tests/unit/evaluation/swe_bench_distributed/test_merge.py b/tests/unit/evaluation/swe_bench_distributed/test_merge.py new file mode 100644 index 000000000..528ab9335 --- /dev/null +++ b/tests/unit/evaluation/swe_bench_distributed/test_merge.py @@ -0,0 +1,189 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""The merge gate: all-or-nothing, id-based, scoped to one run.""" + +from __future__ import annotations + +import inspect + +import pytest + +from inference_endpoint.evaluation.swe_bench_distributed.merge import ( + MergeRefusal, + merge_run, + verify_inventory, +) +from inference_endpoint.evaluation.swe_bench_distributed.queue import ( + UnitOutcome, + UnitResult, + WorkQueue, +) +from inference_endpoint.evaluation.swe_bench_distributed.units import plan_units + +pytestmark = pytest.mark.unit + +IDS = [f"repo__proj-{i:02d}" for i in range(20)] + + +@pytest.fixture +def queue(tmp_path): + return WorkQueue(tmp_path / "wq", plan_units("run-a", IDS, shard_size=10)) + + +def publish(queue: WorkQueue, unit_id: str, **overrides) -> None: + unit = queue.plan.unit(unit_id) + payload = { + "unit_id": unit_id, + "run_id": unit.run_id, + "plan_digest": queue.plan.digest, + "outcome": UnitOutcome.SUCCEEDED, + "accounted_instance_ids": unit.instance_ids, + "resolved_instance_ids": unit.instance_ids[:3], + } + payload.update(overrides) + queue.publish(UnitResult(**payload)) + + +def publish_all(queue: WorkQueue) -> None: + for unit_id in queue.plan.unit_ids: + publish(queue, unit_id) + + +class TestHappyPath: + def test_full_accounting_scores(self, queue): + publish_all(queue) + result = merge_run(queue, "run-a") + assert result.total_instances == 20 + assert result.resolved_instances == 6 + assert result.resolved_rate == pytest.approx(0.3) + assert result.unit_count == 2 + + +class TestRefusals: + def test_a_missing_unit_refuses(self, queue): + publish(queue, "run-a.s00") + # 10 results must never be divided by 20. + with pytest.raises(MergeRefusal, match="have no result"): + merge_run(queue, "run-a") + + def test_an_abandoned_unit_refuses(self, queue): + publish(queue, "run-a.s00") + publish(queue, "run-a.s01", abandoned=True, attempt=3) + with pytest.raises(MergeRefusal, match="abandoned"): + merge_run(queue, "run-a") + + def test_a_non_success_outcome_refuses(self, queue): + publish(queue, "run-a.s00") + publish(queue, "run-a.s01", outcome=UnitOutcome.FAILED) + with pytest.raises(MergeRefusal, match="outcome failed"): + merge_run(queue, "run-a") + + def test_infrastructure_damage_refuses(self, queue): + publish(queue, "run-a.s00") + publish(queue, "run-a.s01", infra_error_count=2) + with pytest.raises(MergeRefusal, match="lost to infrastructure"): + merge_run(queue, "run-a") + + def test_a_missing_id_refuses_even_though_the_count_is_wrong_by_one(self, queue): + publish(queue, "run-a.s00") + unit = queue.plan.unit("run-a.s01") + publish(queue, "run-a.s01", accounted_instance_ids=unit.instance_ids[:-1]) + with pytest.raises(MergeRefusal, match="missing"): + merge_run(queue, "run-a") + + def test_a_swapped_id_refuses_although_the_count_matches(self, queue): + # The whole point of comparing ids rather than counts: this shard has + # exactly ten entries and the wrong content. + publish(queue, "run-a.s00") + unit = queue.plan.unit("run-a.s01") + swapped = unit.instance_ids[:-1] + ("some__other-99",) + publish(queue, "run-a.s01", accounted_instance_ids=swapped) + with pytest.raises(MergeRefusal, match="unplanned"): + merge_run(queue, "run-a") + + def test_a_duplicated_id_within_one_unit_refuses(self, queue): + publish(queue, "run-a.s00") + unit = queue.plan.unit("run-a.s01") + duped = unit.instance_ids[:-1] + (unit.instance_ids[0],) + publish(queue, "run-a.s01", accounted_instance_ids=duped) + with pytest.raises(MergeRefusal, match="duplicate"): + merge_run(queue, "run-a") + + def test_resolved_ids_outside_the_shard_refuse(self, queue): + publish(queue, "run-a.s00") + unit = queue.plan.unit("run-a.s01") + publish( + queue, + "run-a.s01", + resolved_instance_ids=(*unit.instance_ids[:2], IDS[0]), + ) + with pytest.raises(MergeRefusal, match="outside its shard"): + merge_run(queue, "run-a") + + def test_a_foreign_plan_digest_refuses(self, queue): + publish(queue, "run-a.s00") + publish(queue, "run-a.s01") + path = queue.results_dir / "run-a.s01.json" + path.write_text(path.read_text().replace(queue.plan.digest, "f" * 64)) + with pytest.raises(MergeRefusal, match="belongs to plan"): + merge_run(queue, "run-a") + + def test_a_result_outside_the_plan_refuses(self, queue): + publish_all(queue) + (queue.results_dir / "other-run.s00.json").write_text("{}") + with pytest.raises(MergeRefusal, match="outside the plan"): + merge_run(queue, "run-a") + + def test_an_unreadable_result_refuses(self, queue): + publish_all(queue) + (queue.results_dir / "run-a.s00.json").write_text("not json") + with pytest.raises(MergeRefusal, match="unreadable"): + merge_run(queue, "run-a") + + def test_every_reason_is_reported_at_once(self, queue): + publish(queue, "run-a.s00", infra_error_count=1) + with pytest.raises(MergeRefusal) as excinfo: + merge_run(queue, "run-a") + assert len(excinfo.value.reasons) >= 2 + + +class TestScoping: + def test_a_merge_is_always_scoped_to_one_run(self, queue): + publish_all(queue) + with pytest.raises(MergeRefusal, match="scoped to exactly one run"): + merge_run(queue, "some-other-run") + + def test_there_is_no_merge_all(self): + # "Merge everything that looks finished" once combined hundreds of + # banked results from unrelated configurations into one number. + signature = inspect.signature(merge_run) + assert "run_id" in signature.parameters + assert signature.parameters["run_id"].default is inspect.Parameter.empty + assert not hasattr( + __import__( + "inference_endpoint.evaluation.swe_bench_distributed.merge", + fromlist=["merge"], + ), + "merge_all", + ) + + +class TestInventory: + def test_a_complete_run_is_consistent(self, queue): + publish_all(queue) + assert verify_inventory(queue).consistent + + def test_an_ownerless_claim_is_an_inventory_error(self, queue): + publish_all(queue) + claim_dir = queue.claims_dir / "run-a.s00" + claim_dir.mkdir(parents=True) + # Checking `owner` files with one tool and claim directories with + # another is how a verification pass agrees with a broken system. + report = verify_inventory(queue) + assert report.ownerless_claims == ["run-a.s00"] + assert not report.consistent + + def test_claims_without_results_are_reported(self, queue): + queue.claim("run-a.s00") + assert verify_inventory(queue).claims_without_results == ["run-a.s00"] From 58bd05474f683253eacf6192ad9ca03184fef961 Mon Sep 17 00:00:00 2001 From: Stanley Phoong Date: Wed, 26 Aug 2026 13:17:21 -0700 Subject: [PATCH 2/4] feat(swe-bench): withhold the headline accuracy, publish the honest ones The merge gate refuses a bad run, and that refusal is the most important property here. But a refusal that carries no numbers is not the end of the story: somebody still has to report *something*, and with the gate silent they compute it by hand from the artifacts -- which is exactly how a run that lost 106 of 200 instances to infrastructure came to be reported as 47.0% and compared against a complete-run reference of 70.67%. It was read as a model regression. It was attrition. `assess_run()` performs the whole of the gate's arithmetic without deciding anything, and returns a `CompletenessReport`. `merge_run()` becomes the strict all-or-nothing wrapper over it and attaches the report to both `MergeResult` and `MergeRefusal`, so a caller never has to choose between "a number" and "no information". `resolved_rate` is published only when the run is *structurally complete* -- every planned instance id accounted for exactly once -- **and** zero instances were lost to infrastructure. These are two different questions and conflating them gets both wrong. An instance the model attempted and failed is a legitimate score; one our own harness dropped never had the chance. A run short of instances has the wrong denominator; a complete run that leaned on the infrastructure has the wrong provenance. Two numbers are published either way: * `conditional_resolved_rate` -- resolved over the instances that actually completed. Honest about what it measures and not comparable to a complete-run reference. * `resolved_rate_lower_bound` -- resolved over everything planned. Infrastructure losses can only ever *add* resolutions, so this bounds the truth from below even on a badly degraded run. alongside `incomplete_instance_ids`, `infra_lost_instances`, `infra_lost_unit_ids` and a `resolved_rate_withheld_reason` that says which of the two conditions failed and by how much. Ported from the banked campaign's `wq_merge.sh:7-9`: "shard_merge.py refuses to print an accuracy unless all 20 shards account for exactly their own 10 ids, and that refusal is the single most important property in this campaign." What is added here is that the refusal now shows its working. Tests: `TestCompletenessGate` covers both decision boundaries -- complete vs incomplete, and infra-lost vs genuinely-empty (a model that resolved nothing is a score, not a casualty) -- plus an abandoned unit counting as infrastructure loss, the numbers surviving a refusal, and `assess_run` not raising on the run it is describing. Against the parent commit a refusal carries no `report` and no conditional or lower-bound figure at all. --- .../swe_bench_distributed/__init__.py | 11 +- .../evaluation/swe_bench_distributed/merge.py | 201 ++++++++++++++++-- .../swe_bench_distributed/test_merge.py | 135 ++++++++++++ 3 files changed, 334 insertions(+), 13 deletions(-) diff --git a/src/inference_endpoint/evaluation/swe_bench_distributed/__init__.py b/src/inference_endpoint/evaluation/swe_bench_distributed/__init__.py index 3507289bb..c97b98f55 100644 --- a/src/inference_endpoint/evaluation/swe_bench_distributed/__init__.py +++ b/src/inference_endpoint/evaluation/swe_bench_distributed/__init__.py @@ -11,7 +11,14 @@ exactly once. """ -from .merge import MergeRefusal, MergeResult, merge_run, verify_inventory +from .merge import ( + CompletenessReport, + MergeRefusal, + MergeResult, + assess_run, + merge_run, + verify_inventory, +) from .queue import ( ClaimError, UnitOutcome, @@ -22,6 +29,7 @@ __all__ = [ "ClaimError", + "CompletenessReport", "MergeRefusal", "MergeResult", "Unit", @@ -29,6 +37,7 @@ "UnitPlan", "UnitResult", "WorkQueue", + "assess_run", "merge_run", "plan_units", "verify_inventory", diff --git a/src/inference_endpoint/evaluation/swe_bench_distributed/merge.py b/src/inference_endpoint/evaluation/swe_bench_distributed/merge.py index a8dbee7cf..5ef6379fc 100644 --- a/src/inference_endpoint/evaluation/swe_bench_distributed/merge.py +++ b/src/inference_endpoint/evaluation/swe_bench_distributed/merge.py @@ -27,11 +27,23 @@ class MergeRefusal(RuntimeError): ``reasons`` lists every independent failure, so one merge attempt reports everything wrong rather than the first thing wrong. + + ``report`` carries the :class:`CompletenessReport` for the same run. A + refusal is not an absence of information: the conditional rate, the lower + bound and the ids that went missing are exactly what the operator needs in + order to act, and withholding them alongside the headline is what makes + people go and compute the wrong number by hand instead. """ - def __init__(self, run_id: str, reasons: list[str]) -> None: + def __init__( + self, + run_id: str, + reasons: list[str], + report: CompletenessReport | None = None, + ) -> None: self.run_id = run_id self.reasons = reasons + self.report = report super().__init__( f"refusing to score run {run_id!r}: " + "; ".join(reasons[:10]) @@ -39,6 +51,134 @@ def __init__(self, run_id: str, reasons: list[str]) -> None: ) +@dataclass(slots=True) +class CompletenessReport: + """What a run accounted for, and whether that permits an accuracy number. + + ``resolved_rate`` is published only when the run is *structurally complete* + -- every planned instance id reached a terminal state exactly once -- **and** + zero instances were lost to infrastructure. Those are two different + questions and conflating them gets both wrong: + + * An instance the model attempted and failed is a legitimate score. An + instance our own harness dropped is not: it never had the chance. + * A run that is short of instances has the wrong denominator; a run that is + complete but leaned on the infrastructure has the wrong provenance. + + Withholding the headline is the point. A run that lost 106 of 200 instances + to infrastructure and printed ``resolved / planned`` reported 47.0% as + though it were accuracy, and that number was then compared against a + complete-run reference of 70.67% and read as a model regression. It was + attrition. + + Two numbers are therefore *always* published, refusal or not: + + ``conditional_resolved_rate`` + Resolved over the instances that actually completed. Honest about what + it measures, and not comparable to a complete-run reference. + ``resolved_rate_lower_bound`` + Resolved over everything planned. Infrastructure losses can only ever + *add* resolutions, so this bounds the true rate from below even on a + badly degraded run. + """ + + run_id: str + plan_digest: str + total_instances: int + accounted_instance_ids: tuple[str, ...] = () + resolved_instance_ids: tuple[str, ...] = () + incomplete_instance_ids: tuple[str, ...] = () + infra_lost_instances: int = 0 + infra_lost_unit_ids: tuple[str, ...] = () + unit_count: int = 0 + reasons: list[str] = field(default_factory=list) + + @property + def accounted_instances(self) -> int: + return len(self.accounted_instance_ids) + + @property + def resolved_instances(self) -> int: + return len(self.resolved_instance_ids) + + @property + def complete(self) -> bool: + """Every planned instance id reached a terminal state exactly once.""" + return not self.incomplete_instance_ids + + @property + def publishable(self) -> bool: + return self.complete and self.infra_lost_instances == 0 and not self.reasons + + @property + def conditional_resolved_rate(self) -> float | None: + if not self.accounted_instances: + return None + return self.resolved_instances / self.accounted_instances + + @property + def resolved_rate_lower_bound(self) -> float | None: + if not self.total_instances: + return None + return self.resolved_instances / self.total_instances + + @property + def resolved_rate(self) -> float | None: + """The headline number, or ``None`` when it must be withheld.""" + if not self.publishable: + return None + return self.resolved_rate_lower_bound + + @property + def withheld_reason(self) -> str | None: + if self.publishable: + return None + why: list[str] = [] + if self.incomplete_instance_ids: + why.append( + f"{len(self.incomplete_instance_ids)} of {self.total_instances} " + "instances never reached a terminal state" + ) + if self.infra_lost_instances: + why.append( + f"{self.infra_lost_instances} instance(s) were lost to " + "infrastructure, not to the model" + ) + if self.reasons and not why: + why.append("; ".join(self.reasons[:5])) + conditional = self.conditional_resolved_rate + lower = self.resolved_rate_lower_bound + return ( + "NO PUBLISHABLE ACCURACY: " + + "; ".join(why) + + ". Lower bound " + + ("n/a" if lower is None else f"{lower:.4f}") + + " (infrastructure losses can only add resolutions); conditional " + + ("n/a" if conditional is None else f"{conditional:.4f}") + + f" over the {self.accounted_instances} instance(s) that completed. " + "Neither is comparable to a complete-run reference." + ) + + def to_dict(self) -> dict[str, Any]: + return { + "run_id": self.run_id, + "plan_digest": self.plan_digest, + "complete": self.complete, + "total_instances": self.total_instances, + "accounted_instances": self.accounted_instances, + "resolved_instances": self.resolved_instances, + "unit_count": self.unit_count, + "incomplete_instance_ids": list(self.incomplete_instance_ids), + "infra_lost_instances": self.infra_lost_instances, + "infra_lost_unit_ids": list(self.infra_lost_unit_ids), + "resolved_rate": self.resolved_rate, + "conditional_resolved_rate": self.conditional_resolved_rate, + "resolved_rate_lower_bound": self.resolved_rate_lower_bound, + "resolved_rate_withheld_reason": self.withheld_reason, + "reasons": list(self.reasons), + } + + @dataclass(slots=True) class MergeResult: run_id: str @@ -46,6 +186,9 @@ class MergeResult: total_instances: int resolved_instances: int unit_count: int + #: The accounting this result was published from. Present on the success + #: path too, so a caller never has to decide which of two shapes it holds. + report: CompletenessReport | None = None @property def resolved_rate(self) -> float: @@ -59,6 +202,7 @@ def to_dict(self) -> dict[str, Any]: "resolved_instances": self.resolved_instances, "resolved_rate": self.resolved_rate, "unit_count": self.unit_count, + **(self.report.to_dict() if self.report is not None else {}), } @@ -109,11 +253,13 @@ def verify_inventory(queue: WorkQueue) -> InventoryReport: return report -def merge_run(queue: WorkQueue, run_id: str) -> MergeResult: - """Score one run, or refuse. +def assess_run(queue: WorkQueue, run_id: str) -> CompletenessReport: + """Account for every planned instance id. Never raises on a bad run. - ``run_id`` is required and must match the queue's plan. Passing another - run's id is an error, not a filter. + This is the whole of the gate's arithmetic, separated from the decision to + refuse, so that a refused run still yields the conditional rate, the lower + bound and the ids that went missing. :func:`merge_run` is the strict + all-or-nothing wrapper over it. """ plan: UnitPlan = queue.plan if run_id != plan.run_id: @@ -150,6 +296,8 @@ def merge_run(queue: WorkQueue, run_id: str) -> MergeResult: results: dict[str, UnitResult] = queue.results() seen_ids: dict[str, str] = {} resolved: set[str] = set() + infra_lost = 0 + infra_lost_units: set[str] = set() for unit in plan.units: result = results.get(unit.unit_id) @@ -163,15 +311,21 @@ def merge_run(queue: WorkQueue, run_id: str) -> MergeResult: continue if result.abandoned: reasons.append(f"{unit.unit_id}: abandoned after {result.attempt} attempts") + infra_lost += len(unit.instance_ids) + infra_lost_units.add(unit.unit_id) continue if result.outcome is not UnitOutcome.SUCCEEDED: reasons.append(f"{unit.unit_id}: outcome {result.outcome.value}") continue if result.infra_error_count > 0: + # Recorded, not merely refused: how many instances the harness lost + # is the difference between an accuracy and an attrition figure. reasons.append( f"{unit.unit_id}: {result.infra_error_count} instance(s) lost to " "infrastructure" ) + infra_lost += result.infra_error_count + infra_lost_units.add(unit.unit_id) continue expected = set(unit.instance_ids) @@ -211,20 +365,43 @@ def merge_run(queue: WorkQueue, run_id: str) -> MergeResult: resolved.update(result.resolved_instance_ids) planned_ids = set(plan.instance_ids) - if not reasons and set(seen_ids) != planned_ids: - unaccounted = sorted(planned_ids - set(seen_ids)) + unaccounted = sorted(planned_ids - set(seen_ids)) + if unaccounted: reasons.append( f"{len(unaccounted)} planned instance(s) unaccounted for: " + ", ".join(unaccounted[:5]) ) - if reasons: - raise MergeRefusal(run_id, reasons) - - return MergeResult( + return CompletenessReport( run_id=run_id, plan_digest=plan.digest, total_instances=len(planned_ids), - resolved_instances=len(resolved), + accounted_instance_ids=tuple(sorted(seen_ids)), + resolved_instance_ids=tuple(sorted(resolved & planned_ids)), + incomplete_instance_ids=tuple(sorted(planned_ids - set(seen_ids))), + infra_lost_instances=infra_lost, + infra_lost_unit_ids=tuple(sorted(infra_lost_units)), unit_count=len(plan.units), + reasons=reasons, + ) + + +def merge_run(queue: WorkQueue, run_id: str) -> MergeResult: + """Score one run, or refuse. + + ``run_id`` is required and must match the queue's plan. Passing another + run's id is an error, not a filter. + """ + report = assess_run(queue, run_id) + if not report.publishable: + raise MergeRefusal( + run_id, report.reasons or [report.withheld_reason or ""], report + ) + return MergeResult( + run_id=run_id, + plan_digest=report.plan_digest, + total_instances=report.total_instances, + resolved_instances=report.resolved_instances, + unit_count=report.unit_count, + report=report, ) diff --git a/tests/unit/evaluation/swe_bench_distributed/test_merge.py b/tests/unit/evaluation/swe_bench_distributed/test_merge.py index 528ab9335..900833797 100644 --- a/tests/unit/evaluation/swe_bench_distributed/test_merge.py +++ b/tests/unit/evaluation/swe_bench_distributed/test_merge.py @@ -11,6 +11,7 @@ from inference_endpoint.evaluation.swe_bench_distributed.merge import ( MergeRefusal, + assess_run, merge_run, verify_inventory, ) @@ -187,3 +188,137 @@ def test_an_ownerless_claim_is_an_inventory_error(self, queue): def test_claims_without_results_are_reported(self, queue): queue.claim("run-a.s00") assert verify_inventory(queue).claims_without_results == ["run-a.s00"] + + +class TestCompletenessGate: + """`resolved_rate` is published only for a complete, uncontaminated run. + + A run that lost 106 of its 200 instances to infrastructure once printed + 47.0% as though it were accuracy, and that was then compared against a + complete-run reference of 70.67% and read as a model regression. It was + attrition. The gate refuses the headline; the conditional rate and the + lower bound are published either way so nobody has to recompute them by + hand from the artifacts. + """ + + def test_a_complete_run_publishes_the_headline(self, queue): + publish_all(queue) + + report = assess_run(queue, "run-a") + + assert report.complete + assert report.publishable + assert report.resolved_rate == pytest.approx(0.3) + assert report.withheld_reason is None + + def test_an_incomplete_run_withholds_the_headline(self, queue): + publish(queue, "run-a.s00") + + report = assess_run(queue, "run-a") + + assert not report.complete + assert report.resolved_rate is None + assert "NO PUBLISHABLE ACCURACY" in report.withheld_reason + + def test_an_incomplete_run_names_the_instances_it_lost(self, queue): + publish(queue, "run-a.s00") + + report = assess_run(queue, "run-a") + + assert list(report.incomplete_instance_ids) == sorted(IDS[10:]) + assert report.accounted_instances == 10 + + def test_the_conditional_rate_is_published_even_when_refused(self, queue): + publish(queue, "run-a.s00") + + report = assess_run(queue, "run-a") + + # 3 resolved of the 10 that actually ran. + assert report.conditional_resolved_rate == pytest.approx(0.3) + # 3 resolved of the 20 that were planned: infrastructure losses can + # only ever add resolutions, so this bounds the truth from below. + assert report.resolved_rate_lower_bound == pytest.approx(0.15) + + def test_infra_loss_withholds_the_headline_on_a_structurally_complete_run( + self, queue + ): + """Complete is not enough. The losses have to be the model's.""" + publish(queue, "run-a.s00") + publish(queue, "run-a.s01", infra_error_count=2) + + report = assess_run(queue, "run-a") + + assert report.infra_lost_instances == 2 + assert list(report.infra_lost_unit_ids) == ["run-a.s01"] + assert report.resolved_rate is None + assert "lost to" in report.withheld_reason + + def test_a_genuinely_empty_result_is_not_infra_loss(self, queue): + """A model that resolved nothing is a score, not a casualty.""" + publish(queue, "run-a.s00", resolved_instance_ids=()) + publish(queue, "run-a.s01", resolved_instance_ids=()) + + report = assess_run(queue, "run-a") + + assert report.complete + assert report.infra_lost_instances == 0 + assert report.publishable + assert report.resolved_rate == pytest.approx(0.0) + + def test_an_abandoned_unit_is_counted_as_infrastructure_loss(self, queue): + publish(queue, "run-a.s00") + publish(queue, "run-a.s01", abandoned=True, attempt=3) + + report = assess_run(queue, "run-a") + + assert report.infra_lost_instances == 10 + assert report.resolved_rate is None + + def test_a_refusal_still_carries_the_report(self, queue): + publish(queue, "run-a.s00") + + with pytest.raises(MergeRefusal) as excinfo: + merge_run(queue, "run-a") + + report = excinfo.value.report + assert report is not None + assert report.resolved_rate is None + assert report.conditional_resolved_rate == pytest.approx(0.3) + assert report.incomplete_instance_ids + + def test_a_published_result_carries_the_report_too(self, queue): + publish_all(queue) + + result = merge_run(queue, "run-a") + + assert result.report is not None + assert result.report.publishable + assert result.to_dict()["resolved_rate_withheld_reason"] is None + + def test_the_serialized_report_names_every_published_field(self, queue): + publish(queue, "run-a.s00") + + payload = assess_run(queue, "run-a").to_dict() + + assert payload["resolved_rate"] is None + assert payload["conditional_resolved_rate"] == pytest.approx(0.3) + assert payload["resolved_rate_lower_bound"] == pytest.approx(0.15) + assert payload["incomplete_instance_ids"] + assert payload["resolved_rate_withheld_reason"] + assert payload["complete"] is False + + def test_a_run_with_nothing_published_has_no_rate_at_all(self, queue): + report = assess_run(queue, "run-a") + + assert report.conditional_resolved_rate is None + assert report.resolved_rate is None + assert report.resolved_rate_lower_bound == pytest.approx(0.0) + + def test_assess_run_does_not_raise_on_a_broken_run(self, queue): + """The arithmetic must survive the run it is describing.""" + publish(queue, "run-a.s00", accounted_instance_ids=(IDS[0], IDS[0])) + + report = assess_run(queue, "run-a") + + assert not report.publishable + assert report.reasons From b25ce1e88bab0e7c2c7b4b73c8a6d834896b5396 Mon Sep 17 00:00:00 2001 From: Stanley Phoong Date: Thu, 27 Aug 2026 08:38:30 -0700 Subject: [PATCH 3/4] test(swe-bench): type merge result payload --- tests/unit/evaluation/swe_bench_distributed/test_merge.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tests/unit/evaluation/swe_bench_distributed/test_merge.py b/tests/unit/evaluation/swe_bench_distributed/test_merge.py index 900833797..30d2d5b62 100644 --- a/tests/unit/evaluation/swe_bench_distributed/test_merge.py +++ b/tests/unit/evaluation/swe_bench_distributed/test_merge.py @@ -6,6 +6,7 @@ from __future__ import annotations import inspect +from typing import Any import pytest @@ -34,7 +35,7 @@ def queue(tmp_path): def publish(queue: WorkQueue, unit_id: str, **overrides) -> None: unit = queue.plan.unit(unit_id) - payload = { + payload: dict[str, Any] = { "unit_id": unit_id, "run_id": unit.run_id, "plan_digest": queue.plan.digest, From 1c98068f690c5a022e1f67f3620a06f888337751 Mon Sep 17 00:00:00 2001 From: Stanley Phoong Date: Thu, 27 Aug 2026 08:48:19 -0700 Subject: [PATCH 4/4] style(swe-bench): apply pinned ruff grouping --- tests/unit/evaluation/swe_bench_distributed/test_merge.py | 1 - 1 file changed, 1 deletion(-) diff --git a/tests/unit/evaluation/swe_bench_distributed/test_merge.py b/tests/unit/evaluation/swe_bench_distributed/test_merge.py index 30d2d5b62..6de933a87 100644 --- a/tests/unit/evaluation/swe_bench_distributed/test_merge.py +++ b/tests/unit/evaluation/swe_bench_distributed/test_merge.py @@ -9,7 +9,6 @@ from typing import Any import pytest - from inference_endpoint.evaluation.swe_bench_distributed.merge import ( MergeRefusal, assess_run,