From edb92e0424cb7d30caa8b16111e555c432f6b229 Mon Sep 17 00:00:00 2001 From: Yue <108062406+Yue021130@users.noreply.github.com> Date: Tue, 15 Sep 2026 12:36:21 +0800 Subject: [PATCH] test(control-plane): pin the resume_at transport case to an injected clock The parametrized resume_at case built its canonical Todo summary without an evaluated_at, so summary construction re-derived resume readiness from the wall clock. Once real time passed the hard-coded 2026-09-15T00:00:00Z instant the case failed deterministically. - thread evaluated_at through quota_todo_summary into compact_todo_group - evaluate and summarize the case against one shared EVALUATED_AT constant - restore the scheduled-instant literal now that it is pinned to that clock - add a boundary regression covering before, at, and after the instant The boundary regression fails on the pre-fix fixture for both instants that precede the scheduled time and passes once the clock is injected. Fixes #4413 Signed-off-by: Yue <108062406+Yue021130@users.noreply.github.com> --- loopx/control_plane/testing/quota_fixtures.py | 2 + tests/control_plane/test_delivery_response.py | 60 +++++++++++++++++-- 2 files changed, 56 insertions(+), 6 deletions(-) diff --git a/loopx/control_plane/testing/quota_fixtures.py b/loopx/control_plane/testing/quota_fixtures.py index 2cb448da38..3f7fcd3bd7 100644 --- a/loopx/control_plane/testing/quota_fixtures.py +++ b/loopx/control_plane/testing/quota_fixtures.py @@ -67,6 +67,7 @@ def quota_todo_summary( claim_scope_agent_id: str | None = None, item_limit: int | None = None, include_task_orchestration_authority: bool = False, + evaluated_at: str | None = None, ) -> dict[str, Any]: source_section = "Agent Todo" if role == "agent" else "User Todo" summary = compact_todo_group( @@ -75,6 +76,7 @@ def quota_todo_summary( role=role, item_limit=item_limit, include_task_orchestration_authority=include_task_orchestration_authority, + evaluated_at=evaluated_at, ) if summary is None: summary = { diff --git a/tests/control_plane/test_delivery_response.py b/tests/control_plane/test_delivery_response.py index 366aedb05b..b24c06df3c 100644 --- a/tests/control_plane/test_delivery_response.py +++ b/tests/control_plane/test_delivery_response.py @@ -14,6 +14,11 @@ PROFILE = {"outcome_floor": {"outcome_markers": ["outcome"], "surface_only_hints": ["surface"]}} +# Every resume evaluation in this module reads one injected clock instead of the +# wall clock, so the fixtures stay valid before and after any scheduled instant. +EVALUATED_AT = "2026-09-14T00:00:00Z" +RESUME_AT = "2026-09-15T00:00:00Z" + def blocked_run(): return {"agent_id": "agent-a", "todo_id": "todo_delivery", "delivery_outcome": "outcome_gap", @@ -89,10 +94,9 @@ def test_status_compaction_preserves_binding_and_all_consumers_defer_to_current_ ("monitor_changed:todo_dependency", {"baseline_generation": 1}), ("capacity_available:network", {"capability": "other"}), ("pr_merged:#1", {"pr_number": 2}), - # A summary rebuild re-derives resume_ready from the wall clock, so this - # timestamp must stay unsatisfied: a near-future literal makes the case - # start failing the day it passes. - ("resume_at:2099-01-01T00:00:00Z", {"clock_provider": "other"}), + # The scheduled instant is one day ahead of the injected evaluation clock, + # so the condition stays unsatisfied no matter when CI runs. + (f"resume_at:{RESUME_AT}", {"clock_provider": "other"}), ]) def test_real_resume_projection_identity_survives_python_transport(resume, patch): todo = {**waiting_todo(), "resume_when": resume, "resume_monitor_generation": 0, @@ -100,9 +104,10 @@ def test_real_resume_projection_identity_survives_python_transport(resume, patch dependency = {**dependency_todo(), "task_class": "continuous_monitor", "material_change_generation": 0} condition = evaluate_todo_resume_conditions( [todo], source_items=[dependency], available_capabilities=[], - evaluated_at="2026-09-14T00:00:00Z", + evaluated_at=EVALUATED_AT, )[todo["todo_id"]] - summary = quota_todo_summary([todo, dependency], claim_scope_agent_id="agent-a") + summary = quota_todo_summary([todo, dependency], claim_scope_agent_id="agent-a", + evaluated_at=EVALUATED_AT) for key in TODO_PLANNING_SOURCE_KEYS: for row in summary.get(key, []): if row["todo_id"] == todo["todo_id"]: @@ -112,6 +117,49 @@ def test_real_resume_projection_identity_survives_python_transport(resume, patch assert project_delivery_response(blocked_run(), summary)["reason"] == "history_supervision" +def _resume_at_projection(evaluated_at: str, tamper_clock: bool = False) -> tuple[bool, str]: + """Evaluate one `resume_at` wait at an explicit instant and project it.""" + + todo = {**waiting_todo(), "resume_when": f"resume_at:{RESUME_AT}", "resume_monitor_generation": 0, + "task_repository": "git:github.com/example/project"} + dependency = {**dependency_todo(), "task_class": "continuous_monitor", "material_change_generation": 0} + condition = evaluate_todo_resume_conditions( + [todo], source_items=[dependency], available_capabilities=[], + evaluated_at=evaluated_at, + )[todo["todo_id"]] + summary = quota_todo_summary([todo, dependency], claim_scope_agent_id="agent-a", + evaluated_at=evaluated_at) + for key in TODO_PLANNING_SOURCE_KEYS: + for row in summary.get(key, []): + if row["todo_id"] == todo["todo_id"]: + row["resume_condition"] = condition + if tamper_clock: + condition.update({"clock_provider": "other"}) + return bool(condition.get("satisfied")), project_delivery_response(blocked_run(), summary)["reason"] + + +@pytest.mark.parametrize("evaluated_at, expected_reason", [ + ("2026-09-14T00:00:00Z", "canonical_todo_wait"), + ("2026-09-14T23:59:59Z", "canonical_todo_wait"), + ("2026-09-15T00:00:00Z", "history_supervision"), + ("2026-09-15T00:00:01Z", "history_supervision"), + ("2027-09-15T00:00:00Z", "history_supervision"), +]) +def test_resume_at_boundary_projects_from_the_injected_clock_not_the_wall_clock(evaluated_at, expected_reason): + """A `resume_at` boundary must stay stable before and after the scheduled instant. + + The wall clock passed `RESUME_AT` long before this regression was written, so + only an injected evaluation time can keep both sides of the boundary covered. + """ + + satisfied, reason = _resume_at_projection(evaluated_at) + assert reason == expected_reason + assert satisfied is (expected_reason == "history_supervision") + # An unsatisfied wait still yields to a tampered clock provider, which is the + # history-supervision half of the original transport contract. + assert _resume_at_projection(evaluated_at, tamper_clock=True)[1] == "history_supervision" + + def test_surface_supervision_remains_and_invalid_wait_does_not_clear_floor(): summary = quota_todo_summary([waiting_todo()], claim_scope_agent_id="agent-a") asset = {"execution_profile": PROFILE, "agent_todos": summary}