Skip to content

Commit c62f82f

Browse files
authored
Merge pull request #2405 from Altinity/rebase-cicd-v26.8.7.19-lts
Antalya 26.8 - CI/CD reliability fixes
2 parents aaf0e48 + 287493f commit c62f82f

11 files changed

Lines changed: 73 additions & 49 deletions

File tree

‎.claude/tools/fetch_ci_report.js‎

Lines changed: 9 additions & 10 deletions
Original file line numberDiff line numberDiff line change
@@ -574,30 +574,29 @@ async function getCIReportsFromPR(prUrl) {
574574

575575
console.log(`Fetching CI reports for PR #${prNumber}...\n`);
576576

577-
// Fetch PR comments to find CI bot comment.
577+
// Report URLs are the signal — do not filter by bot login.
578+
// Altinity posts via github-actions[bot] with the virtual-hosted S3 URL;
579+
// older comments used clickhouse-gh[bot] and path-style S3.
578580
// Drop GH_CONFIG_DIR before spawning gh: some agent/runner checkouts set it to a poisoned
579581
// config dir (no/expired auth) that makes `gh api` fail, while the default config is fine.
580582
// Other repo tooling (patch-release-check) does the same via `env -u GH_CONFIG_DIR gh`.
581583
const ghEnv = { ...process.env };
582584
delete ghEnv.GH_CONFIG_DIR;
583585
try {
584-
const commentsJson = execSync(`gh api repos/Altinity/ClickHouse/issues/${prNumber}/comments --paginate --jq '.[] | select(.user.login == "clickhouse-gh[bot]") | {body, created_at}'`, {
586+
const commentsJson = execSync(`gh api repos/Altinity/ClickHouse/issues/${prNumber}/comments --paginate --jq '.[] | {body, created_at}'`, {
585587
encoding: 'utf8',
586588
stdio: ['pipe', 'pipe', 'pipe'],
587589
env: ghEnv
588590
});
589591

590592
const comments = commentsJson.trim().split('\n').filter(l => l.trim()).map(l => JSON.parse(l));
591593
comments.sort((a, b) => (b.created_at || '').localeCompare(a.created_at || ''));
592-
if (!comments || comments.length === 0) {
593-
throw new Error('No CI bot comment found');
594-
}
595594

596-
// Search through all bot comments for CI report URLs (not just the latest). Exclude backtick and
595+
// Search through all comments for CI report URLs (not just the latest). Exclude backtick and
597596
// quote chars so a URL quoted in markdown (e.g. inside the AI-review text) is not captured with
598597
// trailing junk, strip trailing punctuation, and dedupe -- otherwise the same report is fetched
599-
// twice and the summary is doubled.
600-
const reportUrlPattern = /https:\/\/s3\.amazonaws\.com\/altinity-build-artifacts\/json\.html\?[^\s)`'"]+/g;
598+
// twice and the summary is doubled. Match both path-style and virtual-hosted S3 URLs.
599+
const reportUrlPattern = /https:\/\/(?:s3\.amazonaws\.com\/altinity-build-artifacts|altinity-build-artifacts\.s3\.amazonaws\.com)\/json\.html\?[^\s)`'"]+/g;
601600
for (const comment of comments) {
602601
if (!comment.body) continue;
603602
let urls = comment.body.match(reportUrlPattern);
@@ -607,9 +606,9 @@ async function getCIReportsFromPR(prUrl) {
607606
}
608607
}
609608

610-
throw new Error('No CI report URLs found in bot comments');
609+
throw new Error('No CI report URLs found in PR comments');
611610
} catch (error) {
612-
if (error.message.includes('No CI bot comment found') || error.message.includes('No CI report URLs found')) {
611+
if (error.message.includes('No CI report URLs found')) {
613612
throw error;
614613
}
615614
throw new Error(`Failed to fetch PR comments: ${error.message}`);

‎.github/actions/create_workflow_report/create_workflow_report.py‎

Lines changed: 5 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -192,6 +192,9 @@ def _enrich_prs_in_release_merge_prs(df: pd.DataFrame, repo: str) -> pd.DataFram
192192
f"https://api.github.com/repos/{repo}/pulls/{pr_number}",
193193
headers=headers,
194194
)
195+
if response.status_code == 404:
196+
# NOTE (strtgbb): not in this repo — upstream PR merged from a fork
197+
continue
195198
if response.status_code != 200:
196199
raise Exception(
197200
f"Failed to fetch pull request info: {response.status_code} {response.text}"
@@ -207,6 +210,8 @@ def _enrich_prs_in_release_merge_prs(df: pd.DataFrame, repo: str) -> pd.DataFram
207210
"pr_labels": html.escape(", ".join(sorted(label_names)), quote=True),
208211
}
209212
)
213+
if not rows:
214+
return pd.DataFrame(columns=["pr_number", "pr_name", "pr_labels"])
210215
return pd.DataFrame(rows)
211216

212217

‎ci/jobs/integration_test_job.py‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1329,7 +1329,7 @@ def main():
13291329
# hard subprocess backstop). Used below to keep an empty flaky/targeted result a
13301330
# best-effort SKIPPED only when a timeout actually exhausted the budget.
13311331
timed_out = False
1332-
session_timeout_parallel = 3600 * 2
1332+
session_timeout_parallel = 3600 * 2.5
13331333
session_timeout_sequential = 3600
13341334

13351335
if is_llvm_coverage:

‎ci/jobs/scripts/stress/stress.py‎

Lines changed: 9 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -400,7 +400,13 @@ def install_thread_pool_fault_injection() -> None:
400400

401401
logging.info("Installing thread-pool fault-injection config: %s -> %s", src, dst)
402402
subprocess.run(["ln", "-sf", src, dst], check=True)
403-
if not call_with_retry(make_query_command("SYSTEM RELOAD CONFIG"), timeout=30, retry_count=5):
403+
# NOTE (strtgbb): ARM debug + ThreadFuzzer config reload is ~18-23s; the
404+
# default 15s receive_timeout loses the race and fails the job.
405+
if not call_with_retry(
406+
make_query_command("SYSTEM RELOAD CONFIG", receive_timeout=60),
407+
timeout=90,
408+
retry_count=5,
409+
):
404410
# Fail-close before the verify query: a stale non-zero probability left
405411
# over from an earlier reload would otherwise mask the reload failure.
406412
raise RuntimeError(
@@ -655,9 +661,9 @@ def execute_bash(full_command, timeout=120):
655661
raise
656662

657663

658-
def make_query_command(query: str) -> str:
664+
def make_query_command(query: str, receive_timeout: int = 15) -> str:
659665
return (
660-
f'clickhouse client -q "{query}" --receive_timeout=15 --max_untracked_memory=1Gi '
666+
f'clickhouse client -q "{query}" --receive_timeout={receive_timeout} --max_untracked_memory=1Gi '
661667
"--memory_profiler_step=1Gi --max_memory_usage_for_user=0 --max_memory_usage_in_client=1000000000 "
662668
"--enable-progress-table-toggle=0 "
663669
"--ast_fuzzer_runs=0",

‎ci/praktika/native_jobs.py‎

Lines changed: 6 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -292,6 +292,8 @@ def _prepare_submodule_cache(workflow, workflow_config: RunConfig) -> Result:
292292
no_strict=True,
293293
)
294294
Shell.check(f"rm -f {archive_path}")
295+
if not created and not S3.head_object(s3_path):
296+
raise RuntimeError(f"failed to upload submodule cache {s3_path}")
295297
info = (
296298
f"cache miss, created: {cache_hash}"
297299
if created
@@ -302,10 +304,12 @@ def _prepare_submodule_cache(workflow, workflow_config: RunConfig) -> Result:
302304
workflow_config.dump()
303305
status = Result.Status.OK
304306
except Exception as e:
305-
print(f"WARNING: Submodule cache failed: {e}")
307+
print(f"ERROR: Submodule cache failed: {e}")
306308
traceback.print_exc()
307309
info = f"{e}\n{traceback.format_exc()}"
308-
status = Result.Status.OK # non-fatal, jobs fall back to GitHub clone
310+
# Do not continue with an empty submodule_cache_hash. Builds would
311+
# skip the restore and clone the same pins themselves.
312+
status = Result.Status.FAIL
309313

310314
return Result.create_from(
311315
name="Submodule Cache",

‎ci/settings/altinity_overrides.py‎

Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -59,6 +59,10 @@ class RunnerLabels:
5959

6060
INSTALL_PYTHON_REQS_FOR_NATIVE_JOBS = ""
6161

62+
# NOTE (strtgbb): anonymous submodule fetches of non-tip pins get refused
63+
# after the burst of ~150 clones; send the ambient gh token instead.
64+
ENABLE_SUBMODULE_CLONE_AUTH = True
65+
6266
DISABLED_WORKFLOWS = [
6367
"backport_branches.py",
6468
"custom_build_praktika.py",

‎cmake/autogenerated_versions.txt‎

Lines changed: 4 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -7,10 +7,9 @@ SET(VERSION_MAJOR 26)
77
SET(VERSION_MINOR 8)
88
SET(VERSION_PATCH 6)
99
SET(VERSION_GITHASH ec5605431dacfc812affc406cc81ca398ca68174)
10-
SET(VERSION_DESCRIBE v26.8.6.10001.altinitytest)
11-
SET(VERSION_STRING 26.8.6.10001.altinitytest)
10+
SET(VERSION_DESCRIBE v26.8.6.20001.altinityantalya)
11+
SET(VERSION_STRING 26.8.6.20001.altinityantalya)
1212
# end of autochange
1313

14-
SET(VERSION_TWEAK 10001)
15-
SET(VERSION_FLAVOUR altinitytest)
16-
14+
SET(VERSION_TWEAK 20001)
15+
SET(VERSION_FLAVOUR altinityantalya)

‎tests/docker_scripts/stress_runner.sh‎

Lines changed: 6 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -97,7 +97,9 @@ if [ "$cache_policy" = "SLRU" ]; then
9797
sed -i.tmp "s|<cache_policy>LRU</cache_policy>|<cache_policy>SLRU</cache_policy>|" /etc/clickhouse-server/config.d/storage_conf*.xml
9898
fi
9999

100-
start_server || { echo "Failed to start server"; exit 1; }
100+
# Preload writes system logs the restart must load before port 9000 opens.
101+
# The default wait (~70s) expires first on some ARM runners.
102+
start_server 10 || { echo "Failed to start server"; exit 1; }
101103

102104
clickhouse-client --query "SYSTEM STOP THREAD FUZZER"
103105

@@ -302,7 +304,9 @@ fi
302304
# hang the server under sanitizers and trip the hung check.
303305
cp -av --dereference /repo/ci/jobs/scripts/fuzzer/limit-recursion-settings.xml /etc/clickhouse-server/users.d/
304306

305-
start_server || { echo "Failed to start server"; exit 1; }
307+
# Same wait as the other restarts: ARM sanitizer + S3 + async_load_databases=false
308+
# can miss the default ~70s window before port 9000 opens.
309+
start_server 10 || { echo "Failed to start server"; exit 1; }
306310

307311
# clickhouse-test must know which storage backend the server actually uses, or its storage skip
308312
# tags are inert and incompatible tests run on an unsupported backend. Both variables are already

‎tests/integration/test_s3_cluster/test.py‎

Lines changed: 3 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -1729,8 +1729,9 @@ def query_cycle():
17291729

17301730
node_to_shutdown.query("SYSTEM STOP SWARM MODE")
17311731

1732-
# enough time to complete processing of objects, started before "SYSTEM STOP SWARM MODE"
1733-
time.sleep(3)
1732+
# Drain in-flight objects started before STOP SWARM. Query time is 3-4s
1733+
# normally; TSan is several times slower, so 3s is not enough.
1734+
time.sleep(15)
17341735

17351736
node_to_shutdown.stop_clickhouse(kill=True)
17361737

‎tests/queries/0_stateless/02354_vector_search_rescoring_distance_in_select_list.reference‎

Lines changed: 18 additions & 18 deletions
Original file line numberDiff line numberDiff line change
@@ -2,32 +2,32 @@ Create tables with Array(Float32) and Array(BFloat16) column
22
Column: Array(Float32)
33
-- Search vector: Array(Float64)
44
5 0
5-
6 0.09375
6-
7 0.203125
7-
8 0.296875
5+
6 0.1
6+
7 0.2
7+
8 0.3
88
-- Search vector: Array(Float32)
99
5 0
10-
6 0.09375
11-
7 0.203125
12-
8 0.296875
10+
6 0.1
11+
7 0.2
12+
8 0.3
1313
-- Search vector: Array(BFloat16)
1414
5 0
15-
6 0.09375
16-
7 0.203125
17-
8 0.296875
15+
6 0.1
16+
7 0.2
17+
8 0.3
1818
Column: Array(BFloat16)
1919
-- Search vector: Array(Float64)
2020
5 0
21-
6 0.09375
22-
7 0.1875
23-
8 0.296875
21+
6 0.1
22+
7 0.2
23+
8 0.3
2424
-- Search vector: Array(Float32)
2525
5 0
26-
6 0.09375
27-
7 0.1875
28-
8 0.296875
26+
6 0.1
27+
7 0.2
28+
8 0.3
2929
-- Search vector: Array(BFloat16)
3030
5 0
31-
6 0.09375
32-
7 0.1875
33-
8 0.296875
31+
6 0.1
32+
7 0.2
33+
8 0.3

0 commit comments

Comments
 (0)