Skip to content

Commit 8007719

Browse files
authored
Merge pull request #5255 from karenchuu/karenchuu/benchmark-study-runtime-observation-coverage
test(benchmark): pin the runtime-observation path into the study campaign block
2 parents 00d07cb + 8ff0b3e commit 8007719

1 file changed

Lines changed: 79 additions & 0 deletions

File tree

‎tests/capabilities/test_benchmark_study_projection.py‎

Lines changed: 79 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -13,6 +13,7 @@
1313
BENCHMARK_STUDY_MANIFEST_SCHEMA_VERSION,
1414
build_benchmark_four_arm_contract,
1515
build_benchmark_study_dashboard,
16+
build_benchmark_runtime_observation,
1617
build_benchmark_upload_envelope,
1718
compact_benchmark_four_arm_contract,
1819
normalize_benchmark_case_insight_projection,
@@ -881,3 +882,81 @@ def test_behavior_findings_do_not_change_study_score_projection():
881882
before = build_benchmark_study_dashboard(manifest, [run])
882883
after = build_benchmark_study_dashboard(manifest, [run, finding])
883884
assert before == after
885+
886+
887+
def test_dashboard_projects_runtime_observations_into_the_campaign_block() -> None:
888+
"""RFC §7: the runtime-observation record kind must reach the campaign block.
889+
890+
The projection reads these counts straight off observations, so a payload
891+
naming a classification outside ``BenchmarkRuntimeClassification``, or a
892+
factorial count that disagrees with the list beside it, cannot come from this
893+
reducer. Both invariants are pinned here rather than asserted in prose.
894+
"""
895+
896+
from loopx.capabilities.benchmark_toolkit.runtime_observation import (
897+
BenchmarkRuntimeClassification,
898+
)
899+
900+
baseline = _row(arm_id="goal_plain", arm_role="baseline", feature=6, reward=0)
901+
treatment = _row(
902+
arm_id="loopx_plain",
903+
arm_role="treatment",
904+
feature=9,
905+
reward=1,
906+
anchor="goal_plain-case-1",
907+
)
908+
observations = [
909+
build_benchmark_runtime_observation(
910+
admission_active=True,
911+
job_receipt_state="resolved",
912+
runner_owner_state="alive",
913+
),
914+
build_benchmark_runtime_observation(
915+
admission_active=True,
916+
job_receipt_state="resolved",
917+
runner_owner_state="absent_after_grace",
918+
terminal_result_present=True,
919+
),
920+
build_benchmark_runtime_observation(
921+
admission_active=False,
922+
job_receipt_state="ambiguous",
923+
runner_owner_state="unknown",
924+
),
925+
]
926+
dashboard = build_benchmark_study_dashboard(
927+
_manifest(),
928+
[
929+
_envelope(baseline, record_kind="experiment_board_row", key="baseline"),
930+
_envelope(treatment, record_kind="experiment_board_row", key="treatment"),
931+
_envelope(_insight(), record_kind="case_insight_projection", key="insight"),
932+
]
933+
+ [
934+
_envelope(
935+
observation,
936+
record_kind="runtime_observation",
937+
key=f"observation-{index}",
938+
)
939+
for index, observation in enumerate(observations)
940+
],
941+
)
942+
campaign = dashboard["campaign"]
943+
944+
assert campaign["runtime_observation_count"] == 3
945+
counts = campaign["runtime_classification_counts"]
946+
assert counts == {
947+
"not_admitted": 1,
948+
"running_qualified": 1,
949+
"terminal_pending_reconcile": 1,
950+
}
951+
assert set(counts) <= {item.value for item in BenchmarkRuntimeClassification}
952+
assert sum(counts.values()) == campaign["runtime_observation_count"]
953+
assert campaign["factorial_contrast_count"] == len(dashboard["factorial_contrasts"])
954+
assert (
955+
campaign["factorial_contrast_countable_count"]
956+
<= campaign["factorial_contrast_count"]
957+
)
958+
assert dashboard["authority"]["factorial_comparison_source"] == (
959+
dashboard["factorial_contrasts"][0]["schema_version"]
960+
if dashboard["factorial_contrasts"]
961+
else None
962+
)

0 commit comments

Comments
 (0)