Repository navigation
935 lines (896 loc) · 55.4 KB
/
Copy pathshard-timings-refresh.yml
File metadata and controls
935 lines (896 loc) · 55.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
# Regenerate scripts/test-shard-timings.json on a timer, and open the PR.
#
# ══════════════════════════════════════════════════════════════════════════════
# WHY THIS EXISTS: THE BALANCING INPUT WAS THE ONLY PART OF THE LOOP WITH NO
# CLOCK ON IT. (#16464)
# ══════════════════════════════════════════════════════════════════════════════
#
# scripts/test-shard-timings.json is the per-package duration dataset the Test
# Core split is binned from. It is GENERATED — scripts/measure-test-shard-
# timings.mjs turns a green run's six turbo summaries into it — but until now it
# was generated only when a person remembered to. It went 13 days without a
# refresh while ~700 test files were added, and the shard it mis-weighted was
# killed by the job wall twelve times in one day and ejected from the merge queue
# twice (#16173).
#
# The rot is one-directional and silent, which is what makes a timer the fix
# rather than more discipline: suites only get slower, the table stays put, and
# the shard that drifted heavy reads as balanced on paper right up to the moment
# it is killed — and a killed shard produces NO measurement while the rollup
# reads green.
#
# WHAT THIS WORKFLOW IS NOT ALLOWED TO DO, AND WHY EACH ONE IS LOAD-BEARING
# ------------------------------------------------------------------------
# ⛔ It never hand-edits the dataset. Every byte it commits came out of the
# generator. A hand-touched number wearing a generated file's `provenance`
# is worse than a stale one: the staleness is at least visible in
# `measuredAt`.
# ⛔ It never touches `timeout-minutes`, the shard matrix, MAX_SHARD_OVER_MEAN,
# MAX_MEASURED_OVER_PREDICTED or FILE_SHARDED_PACKAGES. The partitioner's
# own header calls raising a bound "the one move that cannot be right", and
# its pin 3c is built to red on exactly that. If an honest refresh breaches
# a bound, that breach is the REPORT, not a problem to be tuned away — it is
# quoted verbatim into the PR body and a human decides. See "THE PINS" below.
# ⛔ It never opens a PR when the regenerated file is byte-identical. A weekly
# no-op PR trains everyone to ignore this PR.
# ⛔ It never moves a suite-duration ceiling (#16468). The dataset also
# carries `ceilings`, which `Test Core` grades every executed package
# against (scripts/check-test-suite-ceilings.mjs). The generator writes
# them in the same pass and HOLDS every ceiling the committed dataset
# already has; a package gets a new one only when it had none (its slowest
# executed run in this window x 1.25). A ceiling rises only by a ruling,
# recorded in that script, so a refresh whose window ran slower REPORTS
# the gap in the PR body instead of absorbing it. The first refresh after
# the check landed writes the first table; until it merges the check reads
# NOT MEASURED.
#
# HOW IT GETS ITS INPUTS, AND WHY IT RUNS HERE RATHER THAN IN AN AGENT WORKTREE
# ----------------------------------------------------------------------------
# The documented refresh path downloads the six `test-core-run-summary-N-of-6`
# artifacts of a green run. `GET /actions/artifacts/{id}/zip` redirects to
# `productionresultssa*.blob.core.windows.net`, and every agent container's
# egress policy answers 403 to CONNECT for that host — reproduced four ways, on
# two artifacts across two runs, on the pre-signed URL as well as the API one, on
# two separate days and containers (#16222, and again on #16173). So the
# documented path cannot be walked from a dev seat at all.
#
# A GitHub-hosted runner reaches that host natively — it is the same host
# `actions/download-artifact` uses. Running the refresh here is therefore not a
# workaround for the denial, it is the removal of the channel from the loop: the
# regeneration happens where the data already is, on a timer, and no seat needs
# reachability it does not have.
#
# CHOOSING THE RUNS IS THE HARD PART — see scripts/ci/select-shard-timings-run.mjs
# ------------------------------------------------------------------------------
# "The newest run" is wrong three different ways here (cancelled runs, cache
# replays, expired artifacts), and a refresh built on the wrong run is worse than
# no refresh because it stamps a fresh `measuredAt` on numbers nobody measured.
# That script carries the argument and the self-test; this file only drives it.
#
# ⚠ WHICH RUNS: THE HOURLY `schedule` RUN OF ci.yml, NEVER A PUSH RUN (#16467).
# Until #16467 a push to `main` re-ran the FULL Test Core battery, so "a green
# push run on main" and "a measurement of the workspace" were the same thing.
# They are not any more: a push run computes its package set with `--affected`
# against `github.event.before`. Its six shards still conclude `success` and
# still upload six run summaries, so NOTHING in the eligibility test would have
# noticed — this lane would have kept regenerating the balancing dataset from
# measurements of whatever the last merge happened to touch, and the coverage
# check would not have caught it either (a package the affected set skipped is a
# cache HIT, carried at its old weight, which is a pass). The selector names the
# event, and `select-shard-timings-run.mjs --self-test` drives a schedule-shaped
# and a push-shaped run through it to prove which one comes back.
#
# ⛔ Do NOT make ci.yml's `Save Turbo cache (main only)` step fire on `schedule`
# as well. It is `github.event_name == 'push'` on purpose and that is now
# load-bearing here: if the hourly run seeded the cache it restores, the next
# hourly run would replay almost the whole workspace, the generator would refuse
# every replayed task, and this lane would measure nothing — by construction,
# every hour, forever. Affected-only pushes seeding a narrower cache is the
# direction that HELPS: it leaves more real misses for the hourly run to time.
#
# ⚠ AND IT IS "RUNS", PLURAL, WHICH THE DOCUMENTED PROCEDURE DOES NOT SAY.
# Measured on this lane's first live run (34083991141), when the runs it read
# were still push runs: of the seven retained green runs on main, the BEST
# measured 52 of the 71 packages the committed dataset holds, and the others
# measured 2, 3, 13, 18, 22 and 49. All seven were partial cache replays. That
# follows from the cache design rather than from luck — turbo's key is
# namespaced per shard and only main pushes write it, so a package whose inputs
# have not changed is a HIT, and the generator refuses hits rather than
# recording a replayed ~0.1s window as a suite's cost. "Download six artifacts
# from any green run" therefore measures a SLICE of the workspace. That number
# should IMPROVE under the hourly run, because a push no longer seeds the whole
# workspace's entries — but it is quoted as it was measured, and the next live
# run is what re-measures it.
#
# So the regeneration step accumulates runs, each under its own `--run <id>`
# group — the grouping #16473 added, which sums a sliced package's slices within
# a run and medians the per-run sums across runs. A package seen in three of the
# accumulated runs gets the median of three observations, which is the property
# the dataset's own merge rule always claimed and could not previously deliver.
#
# ⚠ AND THE ACCUMULATION USED TO STOP AT ONE (#22014). The loop broke at the
# first accumulation that COVERED the workspace — and coverage counts a carried
# weight as covered, so the newest run always covered it. Every refresh since
# the hourly run went live was therefore one run: the dataset on `main` named
# `provenance.runs = ['37262126122']`, and recorded `@objectstack/spec` at
# 1134.86s against 1573-1651s on the executed windows after it. The loop now
# stops only when every package has been EXECUTED by at least the generator's
# MINIMUM_EXECUTED_RUNS (three) of the accumulated runs — executed, not
# replayed: an hourly run replays a different part of the workspace each time,
# so three runs is not three samples of `spec`. Measured over the 20 eligible
# hourly runs retained on 2026-10-06 (17:01 the day before to 15:01): every one
# of the 72 packages was executed by at least three of them, the newest twelve
# already sufficed, and `spec` was the tightest at 7 of 20.
#
# A refresh whose retained runs cannot give some package three executions is
# NOT refused: it is opened PROVISIONAL, with those packages named in the
# dataset's `provisional` list and in the PR body. Refusing would discard every
# other package's measurement over one under-sampled one — the silent-rot
# failure this lane exists to end — while the list keeps the shortfall
# machine-readable for anything that needs a baseline rather than a balancing
# input (a per-package ceiling must refuse a provisional weight).
#
# ⚠ AND ACCUMULATION ALONE STILL FALLS SHORT, so the pass MERGES rather than
# replaces. Measured on the same live runs: the union of all seven retained runs
# reaches 57 of 71 packages and converges there, because 14 packages are cache
# HITs in every one of them. Those are carried at their previous weights through
# `--merge-into`, and the carry is sound for one specific reason — a cache HIT is
# not missing data, it is POSITIVE EVIDENCE that the package's inputs are
# unchanged since the run whose output was replayed, so its last measured weight
# still describes it. Every carried package is named in the dataset's
# `carriedOver` list, and a package absent for any OTHER reason is not carried at
# all: it drops out and the coverage check names it as a refusal. The file's
# invariant is preserved exactly — every number in it is a real measurement of
# code as it stands, never an estimate.
#
# THE PINS, AND THE ONE THING A MACHINE MUST NOT DECIDE
# ----------------------------------------------------
# partition-test-shards.mjs `--self-test` grades the dataset against the
# acceptance bound. On a refresh that makes a package heavier than any split can
# bin, it reds BY DESIGN and names the remedy (raise the slice count — never the
# bound), and its pin 3c additionally demands a decision the day the CLI comes
# back under the bound on its own. Both are judgement, so this workflow REPORTS
# the verdict and never acts on it: the self-test is run on the refreshed file
# and its output — pass or fail, verbatim — goes into the PR body. The PR is
# opened either way, because a refusal to open it would lose the measurement,
# which is the exact silent-rot failure this workflow exists to end.
name: Shard Timings Refresh
on:
schedule:
# Monday 05:30 UTC. After the nightly lanes, and early enough in the week
# that the PR is in front of someone before the week's merge volume builds.
- cron: '30 5 * * 1'
workflow_dispatch:
# Changes to this lane get exercised before they merge — the same posture
# half-state-patrol.yml and required-set-patrol.yml keep, and the reason a
# scheduled-only lane is not an option: `dispatch-gates.mjs` reds when a gate
# family is reachable ONLY on a schedule, because a family no PR-time event
# reaches appears on no card's gate list and is therefore graded by nobody
# until the next sweep.
#
# On a `pull_request` run everything executes — selection, download,
# regeneration, the coverage check, the byte comparison, the partitioner's
# verdict, and the write step's own script with the write step's own `env:`,
# up to and including the local commit — so the transport, the flags and the
# write step's variables are proven on a real runner rather than argued
# about. Only the acts that leave the runner are switched off, behind the one
# `DRY_RUN` switch in that step: no branch is pushed, no PR is opened, no
# label is written, and the body that would have been posted is rendered to
# the run's step summary instead.
pull_request:
paths:
- '.github/workflows/shard-timings-refresh.yml'
- 'scripts/ci/select-shard-timings-run.mjs'
- 'scripts/measure-test-shard-timings.mjs'
# Read-only at the top level; the one job widens to exactly what it writes.
permissions:
contents: read
# One refresh at a time. A second run while the first is mid-push would race on
# the same bot branch; the newer inputs are the better ones, so the in-flight run
# yields.
concurrency:
# Scoped BY REF, not one global group. Two scheduled refreshes cannot overlap
# (they would race on the same bot branch, and the newer inputs are the better
# ones, so the in-flight run yields) — but two PRs touching this lane are
# unrelated runs, and a single global group would have each cancel the other's
# dry run and report a cancellation as though the lane were busy.
group: shard-timings-refresh-${{ github.ref }}
cancel-in-progress: true
jobs:
refresh:
name: Regenerate the shard-timings dataset
runs-on: ubuntu-latest
timeout-minutes: 30
permissions:
# Declaring any `permissions:` block drops every scope not listed, so all
# three are spelled even though only two are writes.
contents: write # push the bot branch
pull-requests: write # open the PR and add its label
actions: read # list runs, read jobs, download run-summary artifacts
steps:
- name: Checkout repository
uses: actions/checkout@v7
with:
# A PAT when one exists, the Actions token otherwise — the same choice
# cut-rc.yml makes, for the same reason, and the difference is stated
# in the PR body rather than left to be discovered: a PR opened with
# the Actions `GITHUB_TOKEN` starts NO workflow runs (GitHub's
# recursion guard), so with the fallback credential the refresh PR
# arrives with no CI of its own until someone pushes to it or reopens
# it. With the PAT its checks run immediately.
token: ${{ secrets.RELEASE_PUSH_TOKEN || github.token }}
- name: Setup Node.js
uses: actions/setup-node@v7
with:
node-version: '22'
- name: Setup pnpm
uses: ./.github/actions/setup-pnpm
- name: Get pnpm store directory
shell: bash
run: echo "STORE_PATH=$(pnpm store path --silent)" >> $GITHUB_ENV
# Restore-only: scheduled runs read main's store cache; the per-push
# workflows own saving it. Same shape as coverage-nightly.yml.
- name: Restore pnpm cache
uses: actions/cache/restore@v6
with:
path: ${{ env.STORE_PATH }}
key: ${{ runner.os }}-pnpm-store-v3-${{ hashFiles('**/pnpm-lock.yaml') }}
restore-keys: |
${{ runner.os }}-pnpm-store-v3-
- name: Install dependencies
run: pnpm install --frozen-lockfile
# Verify the instruments BEFORE trusting their output. Both scripts carry a
# battery floor, so this also catches the case where their assertions
# stopped running — which would otherwise let a wrong dataset through a
# green-looking pipeline.
- name: Self-test the generator and the run selector
run: |
# A COLLECTOR, not a bare sequence. Under `bash -e` the first non-zero
# exit aborts the step, so a plain `a && b` list leaves the second
# self-test neither green nor red -- and this step exists precisely to
# say which instrument is broken before the dataset is trusted.
# ⛔ Never let the collector swallow the exit code: a green step over a
# red self-test looks identical to success.
failed=""
run_self_test() {
echo "-- $*"
if "$@"; then
echo "PASS $*"
else
echo "FAIL $*"
failed="${failed} $*"$'\n'
fi
return 0
}
run_self_test node scripts/measure-test-shard-timings.mjs --self-test
run_self_test node scripts/ci/select-shard-timings-run.mjs --self-test
if [ -n "$failed" ]; then
echo ""
echo "Shard-timings instrument self-tests — the following FAILED:"
printf "%s" "$failed"
exit 1
fi
echo "Shard-timings instrument self-tests — both ran and passed"
# The package list the coverage check judges against. `turbo ls` is
# experimental, so the reader asserts its payload loudly rather than
# defaulting around it (an empty list would make every package look deleted
# and every coverage check pass).
- name: List the workspace
run: pnpm exec turbo ls --output=json > "$RUNNER_TEMP/turbo-ls.json"
- name: Choose a green, uncensored, un-replayed run
id: select
env:
GITHUB_TOKEN: ${{ github.token }}
run: |
# 24, not 15 (#16467). The window that matters is the artifact
# retention window — 1 day — and the runs inside it are now the
# hourly scheduled ones, so 24 is "everything still downloadable"
# rather than an arbitrary depth. Coverage and sample depth are both
# ACCUMULATED across runs (see the next step), so examining fewer
# than the window holds is samples left on the table; the cost is two
# API reads per run examined, and the loop below stops at the first
# accumulation in which every package has been executed by the
# minimum number of runs (#22014).
#
# ⭐ THREE EXITS, THREE READINGS (#16467). The selector answers 0, 1
# or 3, and this step must not flatten them:
#
# 0 eligible runs on stdout — carry on.
# 1 candidates existed and every one was censored, failed or lost
# its artifacts. A FINDING. The job goes red, as it always did.
# 3 EXIT_PREREQUISITE_NOT_MET — the API held NO completed
# `schedule` run of ci.yml at all. NOTHING was measured. The
# dataset is untouched, which is the correct outcome, so the job
# must not go red for it.
#
# Measured, which is why this branch exists: on the pull request that
# landed the hourly run, this lane ran with `event=schedule` before
# that trigger existed on main, printed "NO ELIGIBLE RUN among the 0"
# and named three causes none of which had occurred. It also does not
# clear at merge — there is a bootstrap window of at least an hour
# before the first hourly run finishes.
#
# `|| SELECT_EXIT=$?` and not a bare call: this step runs under
# `bash -e`, where the non-zero would kill it before the code could
# be read at all.
#
# stderr goes to a FILE and is echoed back immediately: the log keeps
# every line it had, and the refusal text becomes quotable into the
# step summary below, where somebody reading a green job will see it.
SELECT_EXIT=0
node scripts/ci/select-shard-timings-run.mjs --candidates --limit 24 \
> "$RUNNER_TEMP/candidates.json" 2> "$RUNNER_TEMP/select-stderr.txt" || SELECT_EXIT=$?
cat "$RUNNER_TEMP/select-stderr.txt"
echo "select_exit=$SELECT_EXIT" >> "$GITHUB_OUTPUT"
if [ "$SELECT_EXIT" -eq 3 ]; then
# ⛔ NEVER QUIET. A 3 that scrolls past in a green job is how this
# lane ends up passing because it never looked — the shape that has
# already cost this repo two cards. It is an annotation, a step
# summary section and a job-level notice, and every one of them says
# that a PERSISTENT 3 is a defect rather than a steady state.
echo "not_measured=true" >> "$GITHUB_OUTPUT"
echo "::warning::Shard timings NOT MEASURED: no completed \`schedule\` run of ci.yml exists on main yet. The dataset was left untouched. If this repeats once the hourly full run has been live for a few hours, the trigger is gone or every hourly run is being cancelled — file it."
{
echo "### Shard timings: NOT MEASURED (exit 3) — nothing was regenerated, and nothing failed"
echo
echo "\`select-shard-timings-run --candidates\` found **no completed \`schedule\` run of"
echo "\`ci.yml\` on \`main\` at all**. That is not \"every candidate was rejected\": there were"
echo "no candidates, so no run was censored, none failed and none lost its artifacts."
echo "\`scripts/test-shard-timings.json\` is untouched, which is the correct outcome for this"
echo "reading, and this job is green because nothing went wrong — not because anything passed."
echo
echo "Expected exactly once: while the hourly full run bootstraps. The trigger has to be on"
echo "\`main\` and one run has to finish, and run-summary artifacts live 1 day."
echo
echo "⛔ **A PERSISTENT NOT MEASURED IS A DEFECT, NOT A STEADY STATE.** If this section is"
echo "still here after the hourly run has been live for a few hours, the \`schedule\` trigger"
echo "has been removed from \`ci.yml\` or every hourly run is being cancelled — and the"
echo "balancing dataset is quietly ageing out while this job reports green. File a card."
echo
echo "The selector's own refusal text, verbatim:"
echo
echo '```'
cat "$RUNNER_TEMP/select-stderr.txt" 2>/dev/null || echo '(the refusal text is in this job log)'
echo '```'
} >> "$GITHUB_STEP_SUMMARY"
exit 0
fi
if [ "$SELECT_EXIT" -ne 0 ]; then
echo "::error::select-shard-timings-run refused with exit $SELECT_EXIT — candidates existed and none was eligible. Nothing was regenerated."
exit "$SELECT_EXIT"
fi
echo "Eligible runs, newest first:"
node -e '
const runs = JSON.parse(require("fs").readFileSync(process.env.RUNNER_TEMP + "/candidates.json", "utf8"));
for (const r of runs) console.log(` ${r.run_id} ${r.created_at} ${r.head_sha.slice(0, 10)}`);
'
# ACCUMULATE runs until the workspace is covered — do not look for one run
# that covers it, because there is no such run.
#
# MEASURED on the first live run of this lane (run 34083991141), when the
# runs read were still push runs, and it is the fact that shapes this
# step: of the seven retained green runs on main, the BEST measured 52 of
# the 71 packages the committed dataset holds, and the rest measured 2, 3,
# 13, 18, 22 and 49. Every one of them was a partial cache replay. That is
# not bad luck, it is the cache design: turbo's key is namespaced per shard
# and only main pushes write it, so a package whose inputs have not changed
# is a HIT — and the generator refuses hits rather than recording a
# replayed ~0.1s window as a suite's cost. "Download six artifacts from any
# green run" therefore measures a SLICE of the workspace, never all of it.
# Since #16467 the runs read here are the HOURLY `schedule` runs, which are
# the full battery; the accumulation stays because the cache argument above
# is unchanged by which event ran the suite.
#
# So runs are accumulated. Each contributes its six summaries under its own
# `--run <id>` group, which is exactly the grouping #16473 added: slices are
# summed within a run and the per-run sums are medianed across runs, so a
# package measured by three of these runs gets a median of three
# observations rather than whichever run happened to be read last. Feeding
# several runs WITHOUT that grouping is refused by the generator by name.
#
# WHEN IT STOPS (#22014). Not at the first accumulation that covers the
# workspace: coverage counts a carried weight as covered, so the newest
# run alone always passed it and every refresh was one run deep. The loop
# stops at the first accumulation that BOTH covers the workspace AND
# leaves the dataset's `provisional` list empty — every package executed
# by at least the generator's MINIMUM_EXECUTED_RUNS of the runs fed.
# When the candidates run out first, the last covering accumulation is
# kept and opened PROVISIONAL, its short packages named; only a coverage
# shortfall is still a refusal.
- name: Regenerate the dataset, accumulating runs until every package is measured enough times
id: generate
# Skipped on the NOT MEASURED reading: there is nothing to download.
# Every step after `compare` is already gated on
# `steps.compare.outputs.changed`, which is the empty string when
# `compare` itself is skipped — so guarding these two guards the tail.
if: steps.select.outputs.not_measured != 'true'
env:
GITHUB_TOKEN: ${{ github.token }}
run: |
set -euo pipefail
WORK="$RUNNER_TEMP/refresh"
mkdir -p "$WORK"
# Both judged on the LAST accumulation the generator accepted, which
# is the one `refreshed.json` holds: a run the generator refuses is
# dropped and leaves the previous accumulation's file in place.
COVERED=''
SHORT=''
ACCEPTED=()
RUN_COUNT=$(node -e 'console.log(JSON.parse(require("fs").readFileSync(process.env.RUNNER_TEMP + "/candidates.json", "utf8")).length)')
for i in $(seq 0 $((RUN_COUNT - 1))); do
RUN_ID=$(node -e 'const r=JSON.parse(require("fs").readFileSync(process.env.RUNNER_TEMP+"/candidates.json","utf8"))[Number(process.argv[1])];console.log(r.run_id)' "$i")
echo "::group::Candidate run $RUN_ID"
SUMDIR="$WORK/$RUN_ID"
rm -rf "$SUMDIR"; mkdir -p "$SUMDIR"
OK=1
for N in 1 2 3 4 5 6; do
ART=$(node -e 'const r=JSON.parse(require("fs").readFileSync(process.env.RUNNER_TEMP+"/candidates.json","utf8"))[Number(process.argv[1])];console.log(r.artifact_ids[process.argv[2]])' "$i" "$N")
# curl drops the Authorization header across the redirect to the
# blob host, which is correct: that URL is pre-signed and the
# header would be rejected.
if ! curl -sSL --fail -H "Authorization: Bearer $GITHUB_TOKEN" \
-H "Accept: application/vnd.github+json" \
"https://api.github.com/repos/$GITHUB_REPOSITORY/actions/artifacts/$ART/zip" \
-o "$SUMDIR/$N.zip"; then
echo "::warning::Could not download artifact $ART (shard $N) of run $RUN_ID; skipping this run."
OK=0; break
fi
unzip -qo "$SUMDIR/$N.zip" -d "$SUMDIR/$N"
done
if [ "$OK" -ne 1 ]; then echo "::endgroup::"; continue; fi
# `find` rather than a fixed-depth glob: `upload-artifact` with
# `path: .turbo/runs/` puts the summaries at the artifact root, but a
# shape assumption here would silently collect NOTHING and hand the
# generator an empty argument list, which reads like a refusal for
# the wrong reason.
find "$SUMDIR" -name '*.json' -type f > "$SUMDIR/summaries.txt"
SUMMARY_COUNT=$(wc -l < "$SUMDIR/summaries.txt")
echo "Collected $SUMMARY_COUNT run summary file(s) from run $RUN_ID."
if [ "$SUMMARY_COUNT" -eq 0 ]; then
echo "::warning::Run $RUN_ID's artifacts contained no run-summary JSON; skipping this run."
echo "::endgroup::"; continue
fi
ACCEPTED+=("$RUN_ID")
# The argument list is REBUILT from the accepted set every round
# rather than appended to, so a run the generator refuses can be
# dropped cleanly instead of poisoning every later attempt. Each run
# is fenced by its own `--run <id>`; an array, not `xargs`, because a
# split invocation would run the generator twice and the second would
# overwrite the first's output with a partial dataset.
ARGS=()
for R in "${ACCEPTED[@]}"; do
ARGS+=( --run "$R" )
mapfile -t RUN_FILES < "$WORK/$R/summaries.txt"
ARGS+=( "${RUN_FILES[@]}" )
done
# `--merge-into` the committed dataset, because no retained run set
# measures the whole workspace (see the header). A package this pass
# did not measure keeps its previous weight ONLY when a turbo cache
# HIT witnesses that its inputs are unchanged; anything else is left
# out for the coverage check below to name. The workflow still writes
# no byte the generator did not emit — the merge happens inside it.
# `--spread-out`: every package's executed-run count and
# min/median/max, for the PR body. Written beside the dataset,
# never into it — nothing machine-reads a spread.
if ! node scripts/measure-test-shard-timings.mjs "${ARGS[@]}" \
--merge-into scripts/test-shard-timings.json \
--out "$WORK/refreshed.json" \
--spread-out "$WORK/spread.md"; then
echo "::warning::The generator refused the set including run $RUN_ID; dropping that run and continuing."
unset 'ACCEPTED[-1]'
echo "::endgroup::"; continue
fi
echo "Accumulated ${#ACCEPTED[@]} run(s): ${ACCEPTED[*]}"
if node scripts/ci/select-shard-timings-run.mjs --check-coverage \
--committed scripts/test-shard-timings.json \
--refreshed "$WORK/refreshed.json" \
--workspace "$RUNNER_TEMP/turbo-ls.json" \
--exclude @objectstack/dogfood; then
COVERED=1
else
COVERED=''
fi
# How many weights still rest on fewer than the minimum executed
# runs. Read from the generator's own `provisional` list rather than
# recounted here, and an absent list is an error, not a zero: a
# zero read from nothing would end the loop at one run again.
SHORT=$(node -e '
const d = JSON.parse(require("fs").readFileSync(process.argv[1], "utf8"));
if (!Array.isArray(d.provisional)) throw new Error(process.argv[1] + " carries no `provisional` list");
console.log(d.provisional.length);
' "$WORK/refreshed.json")
if [ -n "$COVERED" ] && [ "$SHORT" -eq 0 ]; then
echo "Every package is covered and executed by at least the minimum number of these runs; stopping."
echo "::endgroup::"
break
fi
if [ -n "$COVERED" ]; then
echo "Covered, but $SHORT package weight(s) rest on fewer than the minimum executed runs; adding the next older run."
fi
echo "::endgroup::"
done
if [ -z "$COVERED" ]; then
echo "::error::The ${#ACCEPTED[@]} eligible run(s) on main, accumulated and merged with the committed dataset, still leave a package that HAD a measured weight with neither a fresh measurement nor a turbo cache HIT to witness that it is unchanged — the shortfall above names them. Each would drop to the test-file-count ESTIMATE, which is the silent degradation this lane exists to prevent. NOTHING was regenerated and no PR was opened: this is a refusal, not a quiet success. A package that merely went unmeasured is NOT this error — that case is carried on its cache-hit witness — so a shortfall here means a suite failed, a package was renamed or removed, or its slices could not be assembled in any run."
exit 1
fi
# PROVISIONAL, not refused (#22014): the candidates ran out before
# every package reached the minimum. The measurement is kept, the
# short packages are named in the dataset and the PR body, and this
# annotation makes the reading visible on a green job.
if [ "$SHORT" -gt 0 ]; then
echo "::warning::Shard timings PROVISIONAL: $SHORT package weight(s) rest on fewer than the minimum executed runs across all ${#ACCEPTED[@]} retained run(s). They are named in the dataset's \`provisional\` list; read them as balancing inputs, not baselines."
fi
# Candidates arrive newest-first, so ACCEPTED[0] is the most recent run
# in the set and its commit is the one the refresh is dated from. The
# full list travels alongside it: every run in it contributed
# measurements, and the PR body names them all rather than implying one.
NEWEST="${ACCEPTED[0]}"
echo "run_id=$NEWEST" >> "$GITHUB_OUTPUT"
echo "runs=${ACCEPTED[*]}" >> "$GITHUB_OUTPUT"
echo "run_count=${#ACCEPTED[@]}" >> "$GITHUB_OUTPUT"
echo "head_sha=$(node -e 'const rs=JSON.parse(require("fs").readFileSync(process.env.RUNNER_TEMP+"/candidates.json","utf8"));console.log(rs.find(r=>String(r.run_id)===process.argv[1]).head_sha)' "$NEWEST")" >> "$GITHUB_OUTPUT"
# Byte comparison, and it decides everything downstream. `cmp -s` rather
# than a diff of parsed JSON: the committed artefact is the file, so the
# file is what has to differ for a PR to be worth anyone's attention.
- name: Compare against the committed dataset
id: compare
if: steps.select.outputs.not_measured != 'true'
run: |
if cmp -s scripts/test-shard-timings.json "$RUNNER_TEMP/refresh/refreshed.json"; then
echo "changed=false" >> "$GITHUB_OUTPUT"
echo "The regenerated dataset is BYTE-IDENTICAL to the committed one; no PR." | tee -a "$GITHUB_STEP_SUMMARY"
else
echo "changed=true" >> "$GITHUB_OUTPUT"
fi
# The before/after halves of ruling 3, taken from the partitioner's OWN
# verdict line rather than recomputed here — a second implementation of the
# binning would be a second answer to grade against.
#
# ⛔ The AFTER leg's non-zero exit is CAPTURED, not propagated, and that is
# not leniency: a red there is the partitioner refusing an honest
# measurement, which is information the PR must carry rather than a reason
# to withhold the PR. `set +e` around that one command, with `$?` read
# immediately and before any pipe, is what keeps the verdict readable
# without letting the step's own status swallow it; the text goes into the
# body either way and the PR's own lint job grades it again.
- name: Predicted bins, before and after
id: bins
if: steps.compare.outputs.changed == 'true'
run: |
node scripts/partition-test-shards.mjs --self-test > "$RUNNER_TEMP/bins-before.txt" 2>&1 || true
cp "$RUNNER_TEMP/refresh/refreshed.json" scripts/test-shard-timings.json
set +e
node scripts/partition-test-shards.mjs --self-test > "$RUNNER_TEMP/bins-after.txt" 2>&1
echo "partitioner_exit=$?" >> "$GITHUB_OUTPUT"
set -e
echo "BEFORE: $(cat "$RUNNER_TEMP/bins-before.txt")"
echo "AFTER: $(cat "$RUNNER_TEMP/bins-after.txt")"
# Composition is separated from the WRITE on purpose: a `pull_request` run
# of this lane must exercise the body-building — the shard durations, the
# bins, the conditional blocks — without pushing anything. The write step
# below consumes this file on every event: as the PR body when it writes,
# and as the step summary on a `pull_request` dry run.
- name: Compose the pull request body
if: steps.compare.outputs.changed == 'true'
env:
RUN_ID: ${{ steps.generate.outputs.run_id }}
RUNS: ${{ steps.generate.outputs.runs }}
RUN_COUNT: ${{ steps.generate.outputs.run_count }}
HEAD_SHA: ${{ steps.generate.outputs.head_sha }}
PARTITIONER_EXIT: ${{ steps.bins.outputs.partitioner_exit }}
USED_PAT: ${{ secrets.RELEASE_PUSH_TOKEN != '' }}
run: |
set -euo pipefail
# Read out of the generated dataset itself rather than recomputed, so the
# sentence in the PR cannot drift from the file it describes.
CARRY_LINE=$(node -e '
const d = JSON.parse(require("fs").readFileSync(process.env.RUNNER_TEMP + "/refresh/refreshed.json", "utf8"));
const carried = d.carriedOver ?? [];
const total = Object.keys(d.packages).length;
const fresh = total - carried.length;
if (carried.length === 0) {
console.log(`All ${total} package weights were measured in these runs; nothing was carried.`);
} else {
console.log(
`${total} package weights: ${fresh} measured in these runs, and ${carried.length} carried ` +
`forward at their previous values because a turbo cache HIT witnessed that their inputs are ` +
`unchanged (so the old number still describes them). Carried: ${carried.join(", ")}.`
);
}
')
# Sample depth (#22014), read out of the dataset like the carry line
# above. An absent list is an error rather than "none provisional".
DEPTH_LINE=$(node -e '
const d = JSON.parse(require("fs").readFileSync(process.env.RUNNER_TEMP + "/refresh/refreshed.json", "utf8"));
if (!Array.isArray(d.provisional)) throw new Error("the refreshed dataset carries no `provisional` list");
const min = d.provenance.minimumRuns;
const total = Object.keys(d.packages).length;
if (d.provisional.length === 0) {
console.log(`Every one of the ${total} weights rests on at least ${min} executed runs; none is provisional.`);
} else {
console.log(
`⚠️ PROVISIONAL: ${d.provisional.length} of the ${total} weights rest on fewer than ${min} executed ` +
`runs, because the retained runs replayed them more often than they executed them. They are ` +
`named in the dataset \`provisional\` list and are fit to balance on, NOT to set a per-package ` +
`ceiling from: ${d.provisional.join(", ")}.`
);
}
')
# The suite-duration ceilings (#16468), read out of the dataset like the
# two lines above. The prior table is the COMMITTED dataset, read from
# git because the bins step has already copied the refreshed file over
# it. A refresh never moves a held ceiling, so the line counts what is
# new and what is uncapped, and the spread table below marks every
# held ceiling this window would have set differently -- and every one
# a run of this window already went over.
CEILING_LINE=$(node -e '
const fs = require("fs");
const { execFileSync } = require("child_process");
const d = JSON.parse(fs.readFileSync(process.env.RUNNER_TEMP + "/refresh/refreshed.json", "utf8"));
if (!d.ceilings || typeof d.ceilings !== "object" || !d.uncapped || typeof d.uncapped !== "object") {
throw new Error("the refreshed dataset carries no `ceilings` / `uncapped`");
}
const prior = JSON.parse(execFileSync("git", ["show", "HEAD:scripts/test-shard-timings.json"], { encoding: "utf8" }));
const held = prior.ceilings && typeof prior.ceilings === "object" ? prior.ceilings : null;
const names = Object.keys(d.ceilings);
const fresh = held === null ? names : names.filter((n) => !Object.hasOwn(held, n));
const uncapped = Object.entries(d.uncapped).map(([n, why]) => n + " (" + why + ")");
const h = d.provenance.ceilingHeadroom;
const over = (fs.readFileSync(process.env.RUNNER_TEMP + "/refresh/spread.md", "utf8").match(/OVER its held ceiling/g) || []).length;
const lines = [];
if (held === null) {
lines.push(
"⚠️ This refresh writes the FIRST suite-duration ceiling table: " + names.length + " package(s) get a " +
"ceiling, each its slowest executed run in this window x " + h + ". From the merge of this PR the `Test Core` " +
"check (scripts/check-test-suite-ceilings.mjs) grades every executed package against it -- until then it " +
"reads NOT MEASURED. Every later refresh HOLDS these numbers and a ceiling rises only by a ruling, so " +
"review the ceiling column below before merging."
);
} else {
lines.push(
"Suite-duration ceilings: " + (names.length - fresh.length) + " held unchanged (a refresh never moves one; " +
"a raise is a ruling), " + fresh.length + " new" + (fresh.length ? " (" + fresh.join(", ") + ")" : "") +
", each new one its slowest executed run in this window x " + h + "."
);
}
lines.push(uncapped.length ? "No ceiling: " + uncapped.join(", ") + "." : "Every package carries a ceiling.");
if (over > 0) {
lines.push(
"⚠️ " + over + " package(s) ran OVER their held ceiling in this window (marked in the table): the brake " +
"fired on main. This refresh does not raise them; that takes a ruling."
);
}
console.log(lines.join("\n\n"));
')
SHARDS=$(node -e '
const rs = JSON.parse(require("fs").readFileSync(process.env.RUNNER_TEMP + "/candidates.json", "utf8"));
const r = rs.find((x) => String(x.run_id) === process.env.RUN_ID);
const s = r.shard_seconds ?? {};
console.log([1,2,3,4,5,6].map((n) => (s[n] == null ? "?" : `${n}: ${s[n]}s`)).join(" | "));
')
{
echo "Refreshes \`scripts/test-shard-timings.json\`, the balancing input for the Test Core"
echo "shard split. Opened automatically by \`.github/workflows/shard-timings-refresh.yml\`."
echo "Every byte came out of \`scripts/measure-test-shard-timings.mjs\`; nothing here was"
echo "hand-edited, and no bound, timeout or matrix entry was touched."
echo
echo "## Source"
echo
echo "Measured across $RUN_COUNT accumulated run(s) of the HOURLY \`schedule\` run of CI on"
echo "\`main\` — the full-battery run (#16467). A \`push\` run on \`main\` is affected-only and"
echo "is not a measurement of the workspace, so no push run feeds this file."
echo
echo "No single green run measures the whole workspace either — turbo's cache is namespaced"
echo "per shard and only main pushes write it, so a"
echo "package whose inputs have not changed is a HIT and the generator refuses hits rather"
echo "than recording a replay as a duration. Runs are therefore accumulated, each fenced by"
echo "its own \`--run\` group, until every package has been EXECUTED by at least the"
echo "generator's minimum number of them, and each weight is the median of those executions."
echo
for R in $RUNS; do
echo "- https://github.com/$GITHUB_REPOSITORY/actions/runs/$R"
done
echo
echo "- Newest run in the set: \`$RUN_ID\`, commit \`$HEAD_SHA\` — the date this refresh carries."
echo "- Every run above had all six \`Test Core (N/6)\` jobs conclude \`success\` with its six"
echo " run-summary artifacts still retained; runs that were cancelled, failed or had lost"
echo " their artifacts were rejected by name in the log before any of these were used."
echo
echo "$CARRY_LINE"
echo
echo "⚠️ This PR references #16173 and #16222 but does NOT carry a closing keyword for them,"
echo "because a weekly lane cannot know which cards a given run ought to retire. If this is"
echo "the first refresh to land, retire those two by hand as part of merging it."
echo
echo "## Sample depth, run-to-run spread and suite-duration ceilings"
echo
echo "$DEPTH_LINE"
echo
echo "$CEILING_LINE"
echo
cat "$RUNNER_TEMP/refresh/spread.md"
echo
echo "## Measured per-shard suite time on the newest run in the set"
echo
echo "\`\`\`"
echo "$SHARDS"
echo "\`\`\`"
echo
echo "## Predicted bins, before and after"
echo
echo "\`\`\`"
echo "BEFORE $(cat "$RUNNER_TEMP/bins-before.txt")"
echo "AFTER $(cat "$RUNNER_TEMP/bins-after.txt")"
echo "\`\`\`"
echo
if [ "${PARTITIONER_EXIT:-0}" != "0" ]; then
echo "## The partitioner's own pins RED on this refresh — read this before merging"
echo
echo "This is the designed behaviour, not a defect in the refresh: the acceptance bound is"
echo "a ratio, and a package that has grown past what any six-way split can bin makes the"
echo "pins fail with the arithmetic in the message. The remedy the partitioner names is to"
echo "raise the file-level slice count for that package — ⛔ never to raise the bound, and"
echo "⛔ never to hand-edit this dataset. This workflow deliberately does neither: it"
echo "reports and stops, because both are decisions."
echo
fi
if [ "$USED_PAT" != "true" ]; then
echo "## No checks will start on this PR by themselves"
echo
echo "It was opened with the Actions \`GITHUB_TOKEN\`, and GitHub's recursion guard means a"
echo "PR opened that way triggers no workflow runs. Push any commit to the branch, or close"
echo "and reopen the PR, to start CI."
echo
fi
# ⛔ No closing keyword is emitted here, in any form. GitHub's
# parser matches `fixes`/`closes`/`resolves` plus a number and
# ignores every negation around them, so even a sentence saying a
# card is NOT closed would close it. This PR is the workflow's
# OUTPUT; the cards about the workflow are referenced only.
echo "Refs #16464, #16173, #16222."
} > "$RUNNER_TEMP/pr-body.md"
echo "Composed a $(wc -l < "$RUNNER_TEMP/pr-body.md")-line pull request body."
# The WRITE — and, on a `pull_request` run of this lane, its rehearsal.
#
# ONE STEP, ONE `env:`, ONE SCRIPT, ON EVERY EVENT (#18341). The write and
# its dry run used to be two steps behind mutually exclusive `if:`s
# (`github.event_name != 'pull_request'` and `== 'pull_request'`), so no
# pull request ever executed the script that writes. Its first execution
# was scheduled run 34810389734: the dataset was computed, and the step
# then died at `git commit` with `RUN_COUNT: unbound variable`, because
# this step's `env:` exported three of the five variables its script
# expands. Under `set -u` a key missing from `env:` blows up only on the
# leg that runs, and the rehearsal ran the other leg.
#
# So a `pull_request` run now executes THIS script with THIS `env:`. The
# branch, the `git add` and the commit — whose message expands every
# provenance variable — happen for real on the runner, with the repo's own
# commit hook, so a variable missing from the block below reds the pull
# request that dropped it. Only the acts that leave the runner are
# switched, and all of them are switched in ONE place: `outward`, which on
# a dry run prints the command instead of running it. The shell expands a
# command's arguments before `outward` is entered, so `set -u` judges every
# argument on both legs alike.
#
# ⛔ Never call `git push`, `gh` or any other network act outside
# `outward`, and never split this step back into an event-gated pair:
# either one reopens a leg that no pull request can exercise.
- name: Push the refresh branch and open the pull request (dry run on pull_request)
if: steps.compare.outputs.changed == 'true'
env:
# The switch: 'true' on a `pull_request` run and only there. The script
# refuses any other spelling, and refuses 'false' on a `pull_request`
# run, so no edit to this line can make a pull request push.
DRY_RUN: ${{ github.event_name == 'pull_request' }}
GH_TOKEN: ${{ secrets.RELEASE_PUSH_TOKEN || github.token }}
# Every variable the script expands that the runner does not provide.
# The commit message is the dataset's provenance and names all four:
# ⛔ never drop one from the message to get a green run.
RUN_ID: ${{ steps.generate.outputs.run_id }}
RUNS: ${{ steps.generate.outputs.runs }}
RUN_COUNT: ${{ steps.generate.outputs.run_count }}
HEAD_SHA: ${{ steps.generate.outputs.head_sha }}
run: |
set -euo pipefail
case "$DRY_RUN" in
true) DRY_TAG='(dry run, not executed) ' ;;
false) DRY_TAG='' ;;
*)
echo "::error::DRY_RUN must be 'true' or 'false', got '$DRY_RUN'. Nothing was pushed."
exit 1
;;
esac
if [ "$GITHUB_EVENT_NAME" = 'pull_request' ] && [ "$DRY_RUN" != 'true' ]; then
echo "::error::DRY_RUN is '$DRY_RUN' on a pull_request run. A pull_request run of this lane never pushes a branch, opens a PR or writes a label. Nothing was pushed."
exit 1
fi
# `gh` reads GH_TOKEN from the environment, not from an argument, so no
# expansion below would notice it missing and a dry run never starts
# `gh`. Named here, it is judged on both legs like every other key.
: "${GH_TOKEN:?is not set; gh would run unauthenticated. Nothing was pushed.}"
# outward [--stand-in TEXT] COMMAND...
# The one dry-run switch. Live, it runs COMMAND. Dry, it prints COMMAND
# to the log instead, and prints TEXT on stdout where the live call's
# output would have been, so every line after it runs on a value of
# the same shape.
outward() {
local stand_in=''
if [ "$1" = '--stand-in' ]; then
stand_in="$2"
shift 2
fi
if [ "$DRY_RUN" = 'true' ]; then
{ printf 'DRY RUN, not executed:'; printf ' %q' "$@"; printf '\n'; } >&2
if [ -n "$stand_in" ]; then printf '%s\n' "$stand_in"; fi
return 0
fi
"$@"
}
BRANCH="claude/shard-timings-refresh-$RUN_ID"
git config user.name 'github-actions[bot]'
git config user.email '41898282+github-actions[bot]@users.noreply.github.com'
git switch -c "$BRANCH"
git add scripts/test-shard-timings.json
# Separate -m flags rather than one embedded newline: a YAML-indented
# heredoc would carry its own leading whitespace into the message body.
git commit \
-m "chore(ci): refresh the Test Core shard-timings dataset" \
-m "Regenerated by .github/workflows/shard-timings-refresh.yml from the test-core-run-summary artifacts of $RUN_COUNT accumulated run(s) ($RUNS), newest $RUN_ID at $HEAD_SHA. Generated, never hand-edited."
outward git push origin "$BRANCH"
PR_URL=$(outward --stand-in "https://github.com/$GITHUB_REPOSITORY/pull/0" \
gh pr create --base main --head "$BRANCH" \
--title "chore(ci): refresh the Test Core shard-timings dataset" \
--body-file "$RUNNER_TEMP/pr-body.md")
echo "${DRY_TAG}Opened $PR_URL" | tee -a "$GITHUB_STEP_SUMMARY"
# ADDITIVE label write only. A whole-set PUT replaces the PR's labels
# and destroys any that land in between — measured on this repo, one
# second wide (see pr-automation.yml's header). POST names only what it
# adds, so no interleaving can lose another writer's label.
PR_NUMBER=$(outward --stand-in 0 gh pr view "$PR_URL" --json number --jq .number)
outward gh api --method POST "repos/$GITHUB_REPOSITORY/issues/$PR_NUMBER/labels" \
-f "labels[]=skip-changeset" > /dev/null
# Read back, because an additive write is necessary and not sufficient:
# a concurrent whole-set PUT from another workflow can still strip the
# label after a successful POST. `skip-changeset` is this PR's exemption
# from the changeset gate — it publishes nothing — so losing it turns
# the gate red on a PR that legitimately has no changeset.
if outward --stand-in skip-changeset gh api "repos/$GITHUB_REPOSITORY/issues/$PR_NUMBER/labels" --jq '.[].name' \
| grep -qxF 'skip-changeset'; then
echo "${DRY_TAG}skip-changeset confirmed on PR #$PR_NUMBER."
else
echo "::warning::skip-changeset did not survive the write on PR #$PR_NUMBER (a concurrent whole-set label PUT strips it). Re-applying once."
outward gh api --method POST "repos/$GITHUB_REPOSITORY/issues/$PR_NUMBER/labels" \
-f "labels[]=skip-changeset" > /dev/null
outward gh api "repos/$GITHUB_REPOSITORY/issues/$PR_NUMBER/labels" --jq '.[].name' \
| grep -qxF 'skip-changeset' \
|| echo "::error::skip-changeset is still absent from PR #$PR_NUMBER; the changeset gate will demand a changeset this PR legitimately has none of. Apply the label by hand."
fi
# The dry-run half of the `pull_request` posture: render the body that
# WOULD have been posted, so a reviewer of a change to this lane sees the
# actual output rather than the diff of the code that produces it.
if [ "$DRY_RUN" = 'true' ]; then
{
echo "### Shard timings: dry run (no branch pushed, no PR opened, no label written)"
echo
echo "This is a \`pull_request\` run of the refresh lane itself. The run selection, the"
echo "artifact download, the regeneration, the coverage check, the partitioner's verdict"
echo "and the write step's own script, up to and including its local commit, all"
echo "executed for real; only the push, the PR and the label write were skipped. The body"
echo "below is what a scheduled run would have posted."
echo
echo "---"
echo
cat "$RUNNER_TEMP/pr-body.md"
} >> "$GITHUB_STEP_SUMMARY"
fi
- name: Say what happened when nothing changed
if: steps.compare.outputs.changed == 'false'
run: |
{
echo "### Shard timings: byte-identical, no PR"
echo
echo "Run \`${{ steps.generate.outputs.run_id }}\` (\`${{ steps.generate.outputs.head_sha }}\`)"
echo "regenerated \`scripts/test-shard-timings.json\` byte-for-byte identically to the"
echo "committed file, so there is nothing to open a pull request about. The dataset is"
echo "current, and this is the loop working rather than the loop skipping."
} >> "$GITHUB_STEP_SUMMARY"