From 79eed79efb225bd373bb5d2bc39d4eeee94eae0d Mon Sep 17 00:00:00 2001 From: huangruiteng Date: Sun, 6 Sep 2026 13:50:37 +0800 Subject: [PATCH 1/3] ci: use capped work-stealing Python test workers Signed-off-by: huangruiteng --- .github/workflows/python-tests.yml | 8 ++++---- .github/workflows/sonarcloud.yml | 3 ++- 2 files changed, 6 insertions(+), 5 deletions(-) diff --git a/.github/workflows/python-tests.yml b/.github/workflows/python-tests.yml index 868559ab98..96b6b7ad73 100644 --- a/.github/workflows/python-tests.yml +++ b/.github/workflows/python-tests.yml @@ -36,9 +36,8 @@ concurrency: jobs: pytest: runs-on: ubuntu-latest - # The parallel suite can take more than 15 minutes through coverage - # collation on the hosted runner. Keep the fail-safe bounded while leaving - # enough headroom for runner variance and post-test coverage reporting. + # Keep a bounded fail-safe for runner variance and coverage collation; + # throughput is controlled by the capped xdist worker pool below. timeout-minutes: 30 steps: - name: Check out repository @@ -87,7 +86,8 @@ jobs: - name: Run fast tests run: >- - python -m pytest -q -n 2 + python -m pytest -q -n auto --maxprocesses=4 --dist=worksteal + --durations=25 --durations-min=1 --cov=loopx --cov-report=term --cov-fail-under=19.6 diff --git a/.github/workflows/sonarcloud.yml b/.github/workflows/sonarcloud.yml index ac0da2084c..3b84b92ba9 100644 --- a/.github/workflows/sonarcloud.yml +++ b/.github/workflows/sonarcloud.yml @@ -75,7 +75,8 @@ jobs: - name: Run fast tests with coverage XML if: steps.sonar-token.outputs.available == 'true' run: >- - python -m pytest -q -n 2 + python -m pytest -q -n auto --maxprocesses=4 --dist=worksteal + --durations=25 --durations-min=1 --cov=loopx --cov-report=term --cov-report=xml:coverage.xml From ba3da4808e676d589bf9d3b64937f39d857260a9 Mon Sep 17 00:00:00 2001 From: huangruiteng Date: Sun, 6 Sep 2026 14:11:33 +0800 Subject: [PATCH 2/3] test(ci): eliminate an unintended real scheduler wait Signed-off-by: huangruiteng --- .github/workflows/python-tests.yml | 7 ++++--- .github/workflows/sonarcloud.yml | 2 +- tests/test_external_scheduler_worker.py | 6 +++--- 3 files changed, 8 insertions(+), 7 deletions(-) diff --git a/.github/workflows/python-tests.yml b/.github/workflows/python-tests.yml index 96b6b7ad73..4bb8e91531 100644 --- a/.github/workflows/python-tests.yml +++ b/.github/workflows/python-tests.yml @@ -36,8 +36,9 @@ concurrency: jobs: pytest: runs-on: ubuntu-latest - # Keep a bounded fail-safe for runner variance and coverage collation; - # throughput is controlled by the capped xdist worker pool below. + # The parallel suite can take more than 15 minutes through coverage + # collation on the hosted runner. Keep the fail-safe bounded while leaving + # enough headroom for runner variance and post-test coverage reporting. timeout-minutes: 30 steps: - name: Check out repository @@ -86,7 +87,7 @@ jobs: - name: Run fast tests run: >- - python -m pytest -q -n auto --maxprocesses=4 --dist=worksteal + python -m pytest -q -n 2 --durations=25 --durations-min=1 --cov=loopx --cov-report=term diff --git a/.github/workflows/sonarcloud.yml b/.github/workflows/sonarcloud.yml index 3b84b92ba9..cd88949f82 100644 --- a/.github/workflows/sonarcloud.yml +++ b/.github/workflows/sonarcloud.yml @@ -75,7 +75,7 @@ jobs: - name: Run fast tests with coverage XML if: steps.sonar-token.outputs.available == 'true' run: >- - python -m pytest -q -n auto --maxprocesses=4 --dist=worksteal + python -m pytest -q -n 2 --durations=25 --durations-min=1 --cov=loopx --cov-report=term diff --git a/tests/test_external_scheduler_worker.py b/tests/test_external_scheduler_worker.py index 4f92e5a19f..e885ea568a 100644 --- a/tests/test_external_scheduler_worker.py +++ b/tests/test_external_scheduler_worker.py @@ -139,7 +139,6 @@ def test_default_invocation_persists_backoff_state( def test_unchanged_limit_runs_final_quota_probe_before_stop( tmp_path: Path, - monkeypatch: pytest.MonkeyPatch, ) -> None: root = tmp_path / "final-probe" fake_cli = root / "fake-loopx" @@ -159,7 +158,7 @@ def test_unchanged_limit_runs_final_quota_probe_before_stop( "counter.write_text(str(count + 1), encoding='utf-8')\n" f"print({payload!r})\n", ) - monkeypatch.setattr(worker.time, "sleep", lambda _seconds: None) + requested_sleeps: list[float] = [] args = _args( fake_cli=fake_cli, state_file=root / "state.json", @@ -167,8 +166,9 @@ def test_unchanged_limit_runs_final_quota_probe_before_stop( ) args.once = False - assert worker.run_worker(args) == 0 + assert worker.run_worker(args, sleep=requested_sleeps.append) == 0 assert counter.read_text(encoding="utf-8") == "3" + assert requested_sleeps == [60] def test_quota_probe_timeout_enters_tick_error(tmp_path: Path) -> None: From 58646c32610afd247dc9c9276454ee0482125841 Mon Sep 17 00:00:00 2001 From: huangruiteng Date: Sun, 6 Sep 2026 14:26:49 +0800 Subject: [PATCH 3/3] ci: shard Python tests across runners and reuse coverage for Sonar Signed-off-by: huangruiteng --- .github/workflows/python-tests.yml | 97 +++++++++++++-- .github/workflows/sonarcloud.yml | 55 ++------- docs/development/testing-and-quality.md | 25 ++++ pyproject.toml | 5 + tests/canary/test_maintainability_ratchet.py | 9 +- tests/control_plane/test_m6_quality_gates.py | 10 -- tests/test_python_ci_workflow.py | 119 +++++++++++++++++++ tests/test_sonarcloud_workflow.py | 18 ++- 8 files changed, 266 insertions(+), 72 deletions(-) create mode 100644 tests/test_python_ci_workflow.py diff --git a/.github/workflows/python-tests.yml b/.github/workflows/python-tests.yml index 4bb8e91531..fe63ae1092 100644 --- a/.github/workflows/python-tests.yml +++ b/.github/workflows/python-tests.yml @@ -4,6 +4,9 @@ on: pull_request: paths: - ".github/workflows/python-tests.yml" + - ".github/workflows/sonarcloud.yml" + - "sonar-project.properties" + - "apps/**" - "loopx/**" - "scripts/**" - "tests/**" @@ -17,6 +20,9 @@ on: - main paths: - ".github/workflows/python-tests.yml" + - ".github/workflows/sonarcloud.yml" + - "sonar-project.properties" + - "apps/**" - "loopx/**" - "scripts/**" - "tests/**" @@ -34,12 +40,9 @@ concurrency: cancel-in-progress: true jobs: - pytest: + checks: runs-on: ubuntu-latest - # The parallel suite can take more than 15 minutes through coverage - # collation on the hosted runner. Keep the fail-safe bounded while leaving - # enough headroom for runner variance and post-test coverage reporting. - timeout-minutes: 30 + timeout-minutes: 15 steps: - name: Check out repository uses: actions/checkout@v7 @@ -85,13 +88,93 @@ jobs: LOOPX_CLI_OUTPUT_BASE_REF: origin/${{ github.event.pull_request.base.ref || 'main' }} run: python examples/control_plane/cli-output-budget-regression-smoke.py - - name: Run fast tests + test-shard: + runs-on: ubuntu-latest + timeout-minutes: 30 + strategy: + fail-fast: false + matrix: + shard: [1, 2] + steps: + - uses: actions/checkout@v7 + with: + fetch-depth: 0 + - uses: actions/setup-python@v6 + with: + python-version: "3.11" + cache: pip + - uses: actions/setup-node@v6 + with: + node-version: "22.6" + cache: npm + cache-dependency-path: package-lock.json + - name: Install test dependencies + run: | + python -m pip install --disable-pip-version-check -e ".[test]" + npm ci --ignore-scripts + - name: Run test shard + # Split the whole collection, not a hand-maintained list of directories. + # Without timing history least_duration alternates equal-weight tests. + # Each runner retains the measured two-worker pool. run: >- python -m pytest -q -n 2 + --splits 2 --group ${{ matrix.shard }} + --splitting-algorithm least_duration --durations=25 --durations-min=1 --cov=loopx --cov-report=term - --cov-fail-under=19.6 + - name: Upload shard coverage + uses: actions/upload-artifact@v7 + with: + name: python-coverage-${{ matrix.shard }} + path: .coverage + include-hidden-files: true + if-no-files-found: error + retention-days: 3 + + pytest: + # Keep the required check name; a skipped/failed shard must not turn it green. + if: always() + needs: [checks, test-shard] + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - name: Require every upstream check + env: + CHECKS_RESULT: ${{ needs.checks.result }} + SHARDS_RESULT: ${{ needs.test-shard.result }} + run: | + test "$CHECKS_RESULT" = success + test "$SHARDS_RESULT" = success + - uses: actions/checkout@v7 + - uses: actions/setup-python@v6 + with: + python-version: "3.11" + cache: pip + - run: python -m pip install --disable-pip-version-check -e ".[test]" + - uses: actions/download-artifact@v7 + with: + pattern: python-coverage-* + path: coverage-shards + - name: Combine complete coverage and enforce the existing floor + run: | + test -s coverage-shards/python-coverage-1/.coverage + test -s coverage-shards/python-coverage-2/.coverage + python -m coverage combine coverage-shards/python-coverage-1/.coverage coverage-shards/python-coverage-2/.coverage + python -m coverage report --fail-under=19.6 + python -m coverage xml -o coverage.xml + - uses: actions/upload-artifact@v7 + with: + name: python-coverage-xml + path: coverage.xml + if-no-files-found: error + retention-days: 3 + + sonar: + needs: pytest + uses: ./.github/workflows/sonarcloud.yml + secrets: + SONAR_TOKEN: ${{ secrets.SONAR_TOKEN }} windows-powershell: runs-on: windows-latest diff --git a/.github/workflows/sonarcloud.yml b/.github/workflows/sonarcloud.yml index cd88949f82..ceb1c04153 100644 --- a/.github/workflows/sonarcloud.yml +++ b/.github/workflows/sonarcloud.yml @@ -1,44 +1,19 @@ name: SonarCloud on: - pull_request: - paths: - - ".github/workflows/sonarcloud.yml" - - "sonar-project.properties" - - "loopx/**" - - "apps/**" - - "scripts/**" - - "tests/**" - - "package.json" - - "package-lock.json" - - "pyproject.toml" - - "tsconfig.control-plane.json" - push: - branches: - - main - paths: - - ".github/workflows/sonarcloud.yml" - - "sonar-project.properties" - - "loopx/**" - - "apps/**" - - "scripts/**" - - "tests/**" - - "package.json" - - "package-lock.json" - - "pyproject.toml" - - "tsconfig.control-plane.json" + workflow_call: + secrets: + SONAR_TOKEN: + required: false permissions: contents: read -concurrency: - group: sonarcloud-${{ github.ref }} - cancel-in-progress: true - jobs: sonar: name: SonarCloud analysis (non-blocking) runs-on: ubuntu-latest + timeout-minutes: 10 steps: - name: Detect whether the token is available id: sonar-token @@ -61,25 +36,11 @@ jobs: with: fetch-depth: 0 - - name: Set up Python + - name: Download this run's combined coverage if: steps.sonar-token.outputs.available == 'true' - uses: actions/setup-python@v6 + uses: actions/download-artifact@v7 with: - python-version: "3.11" - cache: pip - - - name: Install test dependencies - if: steps.sonar-token.outputs.available == 'true' - run: python -m pip install --disable-pip-version-check -e ".[test]" - - - name: Run fast tests with coverage XML - if: steps.sonar-token.outputs.available == 'true' - run: >- - python -m pytest -q -n 2 - --durations=25 --durations-min=1 - --cov=loopx - --cov-report=term - --cov-report=xml:coverage.xml + name: python-coverage-xml - name: SonarCloud scan if: steps.sonar-token.outputs.available == 'true' diff --git a/docs/development/testing-and-quality.md b/docs/development/testing-and-quality.md index c9cf3eacfb..109229166d 100644 --- a/docs/development/testing-and-quality.md +++ b/docs/development/testing-and-quality.md @@ -129,6 +129,31 @@ network latency, provider availability, or a two-hour matrix. 它刻意不包含真实模型调用和 full smoke catalog,因此普通迭代不依赖凭证、网络 时延、模型服务可用性或两小时级测试矩阵。 +The Linux suite uses two hosted runners with two xdist workers each. +`pytest-split` partitions the complete collection using `least_duration`; +without a timing file, tests have equal weight and alternate between shards. +Lint, type checks, and the CLI budget run separately. The required `pytest` +check rejects failed/skipped shards and missing coverage artifacts, then uses +`coverage combine` to enforce the existing 19.6% floor on the union, not on +individual shards. Relative coverage paths make reports portable across runners. +The reusable Sonar workflow consumes that same run's XML and never reruns +pytest or reads cross-run artifacts. Missing Sonar tokens still skip analysis +successfully; test jobs receive no Sonar secret. The trigger is the union of +the former Python and Sonar paths, so app-only and Sonar-configuration changes +also run this lane, including on forks without a token. + +Linux 全套测试分到两台 hosted runner,每台保留两个 xdist worker。`pytest-split` +按完整 collection 分片;没有历史耗时时,等权测试交替分配。lint、类型检查和 CLI +预算独立执行。必需的 `pytest` 汇总检查会拒绝失败/跳过的分片和缺失的 coverage, +合并后再执行原有 19.6% 门槛;不要求单个分片达到全套覆盖率。coverage 使用相对路径, +Sonar 只复用同一次 run 的 XML,不重复测试、不跨 run 取产物。缺少 token 仍成功跳过 +Sonar,测试 job 不接收 Sonar secret。触发范围取原有两套 workflow 的并集,因此仅改 +前端或 Sonar 配置也走此通道,包括没有 token 的 fork。 + +Reproduce one shard locally with `python -m pytest -q -n 2 --splits 2 --group 1 +--splitting-algorithm least_duration --cov=loopx`. Omit the split arguments to +run the complete suite locally. 全量本地测试仍省略分片参数即可。 + ## Smokes And Canary / Smoke 与 Canary A durable smoke should protect shipped behavior, a reusable contract, a diff --git a/pyproject.toml b/pyproject.toml index c8c29f4feb..6a10774de6 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -31,6 +31,7 @@ test = [ "pytest>=8,<9", "pytest-cov>=5,<7", "pytest-xdist>=3,<4", + "pytest-split>=0.10,<0.11", "ruff>=0.12,<0.16", "mypy>=1.18,<2", ] @@ -118,6 +119,10 @@ include = ["loopx*"] [tool.pytest.ini_options] norecursedirs = ["deprecate"] +[tool.coverage.run] +source = ["loopx"] +relative_files = true + [tool.mypy] python_version = "3.11" strict = true diff --git a/tests/canary/test_maintainability_ratchet.py b/tests/canary/test_maintainability_ratchet.py index abd55ade00..46f6ddea1c 100644 --- a/tests/canary/test_maintainability_ratchet.py +++ b/tests/canary/test_maintainability_ratchet.py @@ -31,17 +31,12 @@ def test_current_repository_debt_is_reviewed_without_line_count_pins() -> None: assert set(report["category_counts"]) == {"compatibility_facade"} assert report["category_counts"].get("oversized_decision_function", 0) == 0 assert report["reviewed_exception_count"] == report["finding_count"] - - -def test_current_module_metric_budget_keeps_existing_modules_grandfathered() -> None: - report = build_control_plane_maintainability_report(REPOSITORY_ROOT) - - assert report["ok"] is True, render_control_plane_maintainability_report(report) + # One scan of the immutable checkout owns both debt and metric assertions. + # Mutated temporary repositories below must still be evaluated afresh. assert report["policy"]["module_line_limit"] == 1500 assert report["policy"]["module_any_limit"] == 300 assert report["policy"]["module_dict_any_limit"] == 300 assert report["category_counts"].get("module_metric_budget", 0) == 0 - assert report["unreviewed_count"] == 0 def test_module_metric_ratchet_detects_new_oversized_module(tmp_path: Path) -> None: diff --git a/tests/control_plane/test_m6_quality_gates.py b/tests/control_plane/test_m6_quality_gates.py index 20ec4ca3f8..832f8eff61 100644 --- a/tests/control_plane/test_m6_quality_gates.py +++ b/tests/control_plane/test_m6_quality_gates.py @@ -3,7 +3,6 @@ from pathlib import Path from loopx.canary.maintainability_ratchet import ( - build_control_plane_maintainability_report, module_metric_baseline, module_metrics, ) @@ -32,15 +31,6 @@ def test_m6_rfc_module_budgets_are_enforced_by_the_ratchet_baseline() -> None: assert actual <= ceiling, relative_path -def test_m6_maintainability_ratchet_has_no_unreviewed_debt() -> None: - report = build_control_plane_maintainability_report(REPOSITORY_ROOT) - - assert report["ok"] is True - assert report["unreviewed_count"] == 0 - assert report["stale_exception_count"] == 0 - assert report["category_counts"].get("module_metric_budget", 0) == 0 - - def test_quota_turn_envelope_consumes_effect_turn_at_runtime() -> None: payload = { "ok": True, diff --git a/tests/test_python_ci_workflow.py b/tests/test_python_ci_workflow.py new file mode 100644 index 0000000000..582e5cb65f --- /dev/null +++ b/tests/test_python_ci_workflow.py @@ -0,0 +1,119 @@ +from __future__ import annotations + +import os +from pathlib import Path +import re +import shlex +import subprocess +import sys +import xml.etree.ElementTree as ET + +import pytest + + +WORKFLOW = ( + Path(__file__).resolve().parents[1] / ".github" / "workflows" / "python-tests.yml" +).read_text(encoding="utf-8") + + +@pytest.mark.parametrize("checks", ["success", "failure", "cancelled", "skipped"]) +@pytest.mark.parametrize("shards", ["success", "failure", "cancelled", "skipped"]) +def test_required_pytest_check_rejects_incomplete_upstream_jobs( + checks: str, shards: str, +) -> None: + # Execute the actual gate, including the runner's fail-fast shell behavior. + gate = WORKFLOW.split("name: Require every upstream check", 1)[1] + script = gate.split("run: |", 1)[1].split(" - uses:", 1)[0] + result = subprocess.run( + ["bash", "-e", "-c", script], + env={**os.environ, "CHECKS_RESULT": checks, "SHARDS_RESULT": shards}, + capture_output=True, + check=False, + ) + assert (result.returncode == 0) == (checks == shards == "success") + assert "if: always()\n needs: [checks, test-shard]" in WORKFLOW + + +def test_two_shards_execute_each_test_once_and_merge_portable_coverage( + tmp_path: Path, +) -> None: + # Real pytest-split + xdist + coverage, in two distinct checkout roots. + # Each shard alone misses a function; their union must cover the whole file. + shard_step = WORKFLOW.split("name: Run test shard", 1)[1] + template = shard_step.split("run: >-", 1)[1].split(" - name:", 1)[0] + env = { + key: value for key, value in os.environ.items() + if not key.startswith(("COVERAGE", "COV_CORE", "PYTEST")) + } + seen: list[set[str]] = [] + for shard in (1, 2): + root = tmp_path / f"checkout-{shard}" + root.mkdir() + (root / "ci_subject.py").write_text( + "def first():\n return 1\n\ndef second():\n return 2\n", + encoding="utf-8", + ) + (root / "test_subject.py").write_text( + "from ci_subject import first, second\n" + "def test_first():\n assert first() == 1\n" + "def test_second():\n assert second() == 2\n", + encoding="utf-8", + ) + (root / "pyproject.toml").write_text( + '[tool.coverage.run]\nsource = ["ci_subject"]\nrelative_files = true\n', + encoding="utf-8", + ) + args = shlex.split(template.replace("${{ matrix.shard }}", str(shard))) + args[0] = sys.executable + args[args.index("--cov=loopx")] = "--cov=ci_subject" + result = subprocess.run( + [*args, "--junitxml=results.xml"], cwd=root, env=env, + capture_output=True, text=True, timeout=60, check=False, + ) + assert result.returncode == 0, result.stdout + result.stderr + cases = ET.parse(root / "results.xml").findall(".//testcase") + seen.append({case.attrib["name"] for case in cases}) + partial = subprocess.run( + [sys.executable, "-m", "coverage", "report", "--fail-under=100"], + cwd=root, env=env, capture_output=True, check=False, + ) + assert partial.returncode == 2 + destination = tmp_path / "coverage-shards" / f"python-coverage-{shard}" + destination.mkdir(parents=True) + (root / ".coverage").rename(destination / ".coverage") + + assert seen[0] and seen[1] and seen[0].isdisjoint(seen[1]) + assert seen[0] | seen[1] == {"test_first", "test_second"} + # Reuse the real aggregate shell commands, with a 100% synthetic oracle. + step = WORKFLOW.split("name: Combine complete coverage", 1)[1] + script = step.split("run: |", 1)[1].split(" - uses:", 1)[0] + script = script.replace("python -m", f"{shlex.quote(sys.executable)} -m") + script = script.replace("--fail-under=19.6", "--fail-under=100") + root = tmp_path / "checkout-1" + (tmp_path / "coverage-shards").rename(root / "coverage-shards") + for shard in (1, 2): + data = root / "coverage-shards" / f"python-coverage-{shard}" / ".coverage" + held = data.with_name("held") + data.rename(held) + missing = subprocess.run( + ["bash", "-e", "-c", script], cwd=root, env=env, + capture_output=True, check=False, + ) + assert missing.returncode != 0 + assert not (root / "coverage.xml").exists() + held.rename(data) + result = subprocess.run( + ["bash", "-e", "-c", script], cwd=root, env=env, + capture_output=True, text=True, check=False, + ) + assert result.returncode == 0, result.stdout + result.stderr + assert ET.parse(root / "coverage.xml").getroot().attrib["line-rate"] == "1" + # The combine consumed the inputs: replay with absent artifacts must fail. + missing = subprocess.run( + ["bash", "-e", "-c", script], cwd=root, env=env, + capture_output=True, check=False, + ) + assert missing.returncode != 0 + assert re.search(r"shard: \[1, 2\]", WORKFLOW) + assert "include-hidden-files: true" in WORKFLOW + assert "--cov-fail-under" not in template diff --git a/tests/test_sonarcloud_workflow.py b/tests/test_sonarcloud_workflow.py index c4b8c7a6ea..d98d3398fb 100644 --- a/tests/test_sonarcloud_workflow.py +++ b/tests/test_sonarcloud_workflow.py @@ -19,6 +19,22 @@ def test_missing_sonar_token_reaches_a_successful_skip_step() -> None: def test_sonar_steps_remain_guarded_by_the_token() -> None: workflow = WORKFLOW.read_text(encoding="utf-8") - assert workflow.count("if: steps.sonar-token.outputs.available == 'true'") == 5 + assert workflow.count("if: steps.sonar-token.outputs.available == 'true'") == 3 assert workflow.count("SONAR_TOKEN: ${{ secrets.SONAR_TOKEN }}") == 2 assert "\n env:\n SONAR_TOKEN: ${{ secrets.SONAR_TOKEN }}" not in workflow + + +def test_sonar_reuses_same_run_coverage_without_a_privileged_trigger() -> None: + workflow = WORKFLOW.read_text(encoding="utf-8") + caller = WORKFLOW.with_name("python-tests.yml").read_text(encoding="utf-8") + + assert "workflow_call:" in workflow + assert "workflow_run:" not in workflow + assert "pull_request_target:" not in workflow + caller + assert "python -m pytest" not in workflow + assert "name: python-coverage-xml" in workflow + assert "run-id:" not in workflow + assert "github-token:" not in workflow + assert "needs: pytest\n uses: ./.github/workflows/sonarcloud.yml" in caller + assert '"apps/**"' in caller + assert '"sonar-project.properties"' in caller