|
13 | 13 | BENCHMARK_STUDY_MANIFEST_SCHEMA_VERSION, |
14 | 14 | build_benchmark_four_arm_contract, |
15 | 15 | build_benchmark_study_dashboard, |
| 16 | + build_benchmark_runtime_observation, |
16 | 17 | build_benchmark_upload_envelope, |
17 | 18 | compact_benchmark_four_arm_contract, |
18 | 19 | normalize_benchmark_case_insight_projection, |
@@ -881,3 +882,81 @@ def test_behavior_findings_do_not_change_study_score_projection(): |
881 | 882 | before = build_benchmark_study_dashboard(manifest, [run]) |
882 | 883 | after = build_benchmark_study_dashboard(manifest, [run, finding]) |
883 | 884 | assert before == after |
| 885 | + |
| 886 | + |
| 887 | +def test_dashboard_projects_runtime_observations_into_the_campaign_block() -> None: |
| 888 | + """RFC §7: the runtime-observation record kind must reach the campaign block. |
| 889 | +
|
| 890 | + The projection reads these counts straight off observations, so a payload |
| 891 | + naming a classification outside ``BenchmarkRuntimeClassification``, or a |
| 892 | + factorial count that disagrees with the list beside it, cannot come from this |
| 893 | + reducer. Both invariants are pinned here rather than asserted in prose. |
| 894 | + """ |
| 895 | + |
| 896 | + from loopx.capabilities.benchmark_toolkit.runtime_observation import ( |
| 897 | + BenchmarkRuntimeClassification, |
| 898 | + ) |
| 899 | + |
| 900 | + baseline = _row(arm_id="goal_plain", arm_role="baseline", feature=6, reward=0) |
| 901 | + treatment = _row( |
| 902 | + arm_id="loopx_plain", |
| 903 | + arm_role="treatment", |
| 904 | + feature=9, |
| 905 | + reward=1, |
| 906 | + anchor="goal_plain-case-1", |
| 907 | + ) |
| 908 | + observations = [ |
| 909 | + build_benchmark_runtime_observation( |
| 910 | + admission_active=True, |
| 911 | + job_receipt_state="resolved", |
| 912 | + runner_owner_state="alive", |
| 913 | + ), |
| 914 | + build_benchmark_runtime_observation( |
| 915 | + admission_active=True, |
| 916 | + job_receipt_state="resolved", |
| 917 | + runner_owner_state="absent_after_grace", |
| 918 | + terminal_result_present=True, |
| 919 | + ), |
| 920 | + build_benchmark_runtime_observation( |
| 921 | + admission_active=False, |
| 922 | + job_receipt_state="ambiguous", |
| 923 | + runner_owner_state="unknown", |
| 924 | + ), |
| 925 | + ] |
| 926 | + dashboard = build_benchmark_study_dashboard( |
| 927 | + _manifest(), |
| 928 | + [ |
| 929 | + _envelope(baseline, record_kind="experiment_board_row", key="baseline"), |
| 930 | + _envelope(treatment, record_kind="experiment_board_row", key="treatment"), |
| 931 | + _envelope(_insight(), record_kind="case_insight_projection", key="insight"), |
| 932 | + ] |
| 933 | + + [ |
| 934 | + _envelope( |
| 935 | + observation, |
| 936 | + record_kind="runtime_observation", |
| 937 | + key=f"observation-{index}", |
| 938 | + ) |
| 939 | + for index, observation in enumerate(observations) |
| 940 | + ], |
| 941 | + ) |
| 942 | + campaign = dashboard["campaign"] |
| 943 | + |
| 944 | + assert campaign["runtime_observation_count"] == 3 |
| 945 | + counts = campaign["runtime_classification_counts"] |
| 946 | + assert counts == { |
| 947 | + "not_admitted": 1, |
| 948 | + "running_qualified": 1, |
| 949 | + "terminal_pending_reconcile": 1, |
| 950 | + } |
| 951 | + assert set(counts) <= {item.value for item in BenchmarkRuntimeClassification} |
| 952 | + assert sum(counts.values()) == campaign["runtime_observation_count"] |
| 953 | + assert campaign["factorial_contrast_count"] == len(dashboard["factorial_contrasts"]) |
| 954 | + assert ( |
| 955 | + campaign["factorial_contrast_countable_count"] |
| 956 | + <= campaign["factorial_contrast_count"] |
| 957 | + ) |
| 958 | + assert dashboard["authority"]["factorial_comparison_source"] == ( |
| 959 | + dashboard["factorial_contrasts"][0]["schema_version"] |
| 960 | + if dashboard["factorial_contrasts"] |
| 961 | + else None |
| 962 | + ) |
0 commit comments