From 0d979dc889861c4c10a6a4c0e5cb13faad352035 Mon Sep 17 00:00:00 2001 From: Alexey Milovidov Date: Sun, 20 Jul 2025 02:08:35 +0200 Subject: [PATCH 1/4] Faster test `00988_expansion_aliases_limit` (cherry picked from commit a8860b03b5c387ef2f79b0d0defe56f7d4277505) Signed-off-by: CarlosFelipeOR --- tests/queries/0_stateless/00988_expansion_aliases_limit.sql | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/queries/0_stateless/00988_expansion_aliases_limit.sql b/tests/queries/0_stateless/00988_expansion_aliases_limit.sql index 77f2ba2dbd1d..fce55bb68728 100644 --- a/tests/queries/0_stateless/00988_expansion_aliases_limit.sql +++ b/tests/queries/0_stateless/00988_expansion_aliases_limit.sql @@ -1 +1,2 @@ +SET max_expanded_ast_elements = 10000; SELECT 1 AS a, a + a AS b, b + b AS c, c + c AS d, d + d AS e, e + e AS f, f + f AS g, g + g AS h, h + h AS i, i + i AS j, j + j AS k, k + k AS l, l + l AS m, m + m AS n, n + n AS o, o + o AS p, p + p AS q, q + q AS r, r + r AS s, s + s AS t, t + t AS u, u + u AS v, v + v AS w, w + w AS x, x + x AS y, y + y AS z; -- { serverError BAD_ARGUMENTS, 168 } From a68d1d636ad08bd00a6d2cca94f1c37aa6985aac Mon Sep 17 00:00:00 2001 From: CarlosFelipeOR Date: Sun, 20 Sep 2026 14:52:59 -0300 Subject: [PATCH 2/4] Exclude 02354_vector_search_* from TSan Ported by hand from upstream ClickHouse/ClickHouse#84510 (0cba6f50257450c16976d82af85d3f20fe3ae5a1), not cherry-picked: the upstream commit also drops `SET max_execution_time = 600;`, which does not exist on this branch, and the neighbouring context differs (`allow_experimental_vector_similarity_index` here vs `enable_vector_similarity_index` upstream). Only the tag change applies. Both tests sit at the 600s ceiling on `Stateless tests (tsan) [6/6]`: 02354_vector_search_expansion_search timed out in 11 of the last 45 runs of that shard and averages 524s on release runs. Signed-off-by: CarlosFelipeOR --- .../0_stateless/02354_vector_search_expansion_search.sql | 2 +- .../queries/0_stateless/02354_vector_search_multiple_marks.sql | 3 ++- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/tests/queries/0_stateless/02354_vector_search_expansion_search.sql b/tests/queries/0_stateless/02354_vector_search_expansion_search.sql index 091bcec1a254..5450f6b9b1e0 100644 --- a/tests/queries/0_stateless/02354_vector_search_expansion_search.sql +++ b/tests/queries/0_stateless/02354_vector_search_expansion_search.sql @@ -1,4 +1,4 @@ --- Tags: no-fasttest, long, no-asan, no-ubsan, no-debug +-- Tags: no-fasttest, long, no-asan, no-ubsan, no-tsan, no-debug -- ^^ Disable test for slow builds: generating data takes time but a sufficiently large data set -- is necessary for different hnsw_candidate_list_size_for_search settings to make a difference diff --git a/tests/queries/0_stateless/02354_vector_search_multiple_marks.sql b/tests/queries/0_stateless/02354_vector_search_multiple_marks.sql index 5b9bccb23cc5..4c5e8af33dde 100644 --- a/tests/queries/0_stateless/02354_vector_search_multiple_marks.sql +++ b/tests/queries/0_stateless/02354_vector_search_multiple_marks.sql @@ -1,4 +1,5 @@ --- Tags: no-fasttest, no-ordinary-database +-- Tags: no-fasttest, no-ordinary-database, no-tsan +-- no-tsan: generating data takes too long -- Tests correctness of vector similarity index with > 1 mark From 12c136979d49eb6e07de7200630d9643121a6784 Mon Sep 17 00:00:00 2001 From: CarlosFelipeOR Date: Sun, 20 Sep 2026 15:27:50 -0300 Subject: [PATCH 3/4] CI: raise integration test job timeouts to 4h for sanitizers and 3h for release Signed-off-by: CarlosFelipeOR --- tests/ci/ci_config.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/tests/ci/ci_config.py b/tests/ci/ci_config.py index d1570f8ba503..c6ea1afc12e4 100644 --- a/tests/ci/ci_config.py +++ b/tests/ci/ci_config.py @@ -374,16 +374,17 @@ class CI: ), JobNames.INTEGRATION_TEST_ASAN: CommonJobConfigs.INTEGRATION_TEST.with_properties( required_builds=[BuildNames.PACKAGE_ASAN], + timeout=4 * 3600, num_batches=8, ), JobNames.INTEGRATION_TEST_ASAN_OLD_ANALYZER: CommonJobConfigs.INTEGRATION_TEST.with_properties( required_builds=[BuildNames.PACKAGE_ASAN], - timeout=3 * 3600, + timeout=4 * 3600, num_batches=8, ), JobNames.INTEGRATION_TEST_TSAN: CommonJobConfigs.INTEGRATION_TEST.with_properties( required_builds=[BuildNames.PACKAGE_TSAN], - timeout=3 * 3600, + timeout=4 * 3600, num_batches=8, ), JobNames.INTEGRATION_TEST_AARCH64: CommonJobConfigs.INTEGRATION_TEST.with_properties( @@ -393,6 +394,7 @@ class CI: ), JobNames.INTEGRATION_TEST: CommonJobConfigs.INTEGRATION_TEST.with_properties( required_builds=[BuildNames.PACKAGE_RELEASE], + timeout=3 * 3600, num_batches=8, # release_only=True, ), From d4fb00ef749de235b80ec5c932721bc85610f15e Mon Sep 17 00:00:00 2001 From: CarlosFelipeOR Date: Sun, 20 Sep 2026 15:53:07 -0300 Subject: [PATCH 4/4] CI: drop superseded attempts from the workflow report after a rerun Ports the anchor CTE from #1572 and #2142, keeping only rows at or after the latest result batch per check_name. The guard excludes STARTED/COMPLETED instead of praktika's Pre Hooks/Post Hooks, since COMPLETED lands after the tests on the old CI and would otherwise hide every failure. Signed-off-by: CarlosFelipeOR --- .../create_workflow_report.py | 123 +++++++++++------- 1 file changed, 77 insertions(+), 46 deletions(-) diff --git a/.github/actions/create_workflow_report/create_workflow_report.py b/.github/actions/create_workflow_report/create_workflow_report.py index cd9ab7d938b8..c4f9b2b2947d 100755 --- a/.github/actions/create_workflow_report/create_workflow_report.py +++ b/.github/actions/create_workflow_report/create_workflow_report.py @@ -148,28 +148,79 @@ def get_run_details(run_url: str) -> dict: return response.json() +def _checks_latest_test_status_cte(commit_sha: str, branch_name: str) -> str: + """ + Shared filtering for gh-data.checks: anchor time selects the latest result batch + per check_name. + + Rows that must not set the anchor (but are still kept if their time is >= anchor): + - Stateless teardown: check_name LIKE 'Stateless%' AND test_name not matching ^[0-9]{5} + - Empty test_name: CIDB job-level parent rows + - STARTED / COMPLETED: job lifecycle markers with their own stopwatches; COMPLETED + always lands after the tests and would otherwise become the sole surviving batch, + hiding all FAILs + + Keep rows with check_start_time >= anchor so the latest main batch and any later + teardown/marker rows are included. Earlier batches (failed attempts before a rerun + uploaded a newer uniform timestamp) are dropped, including synthetic rows such as + 'Job Timeout Expired' that only exist in the attempt that failed. + """ + return f"""WITH checks_with_anchor AS ( + SELECT + check_name, + test_name, + report_url, + check_status, + test_status, + check_start_time, + maxIf( + check_start_time, + test_name != '' + AND test_name NOT IN ('STARTED', 'COMPLETED') + AND NOT ( + check_name LIKE 'Stateless%' + AND NOT match(test_name, '^[0-9]{{5}}') + ) + ) OVER (PARTITION BY check_name) AS latest_check_start_time + FROM `gh-data`.checks + WHERE commit_sha = '{commit_sha}' + AND head_ref IN ('{branch_name}', 'refs/tags/{branch_name}') + ), + rows_from_latest_check_run AS ( + SELECT + check_name, + test_name, + report_url, + check_status, + test_status, + check_start_time + FROM checks_with_anchor + WHERE check_start_time >= latest_check_start_time + ), + latest_test_status AS ( + SELECT + argMax(check_status, check_start_time) AS job_status, + check_name AS job_name, + argMax(test_status, check_start_time) AS status, + test_name, + report_url AS results_link + FROM rows_from_latest_check_run + GROUP BY check_name, test_name, report_url + )""" + + def get_checks_fails(client: Client, commit_sha: str, branch_name: str): """ Get tests that did not succeed for the given commit and branch. Exclude checks that have status 'error' as they are counted in get_checks_errors. """ - query = f"""SELECT job_status, job_name, status as test_status, test_name, results_link - FROM ( - SELECT - argMax(check_status, check_start_time) as job_status, - check_name as job_name, - argMax(test_status, check_start_time) as status, - test_name, - report_url as results_link, - task_url - FROM `gh-data`.checks - WHERE commit_sha='{commit_sha}' AND head_ref IN ('{branch_name}', 'refs/tags/{branch_name}') - GROUP BY check_name, test_name, report_url, task_url - ) - WHERE test_status IN ('FAIL', 'ERROR') - AND job_status!='error' - ORDER BY job_name, test_name - """ + query = f"""{_checks_latest_test_status_cte(commit_sha, branch_name)} + SELECT job_status, job_name, status AS test_status, test_name, results_link + FROM latest_test_status + WHERE test_status IN ('FAIL', 'ERROR') + AND job_status != 'error' + ORDER BY job_name, test_name + """ return client.query_dataframe(query) @@ -182,19 +233,9 @@ def get_checks_known_fails( if len(known_fails) == 0: return pd.DataFrame() - query = f"""SELECT job_status, job_name, status as test_status, test_name, results_link - FROM ( - SELECT - argMax(check_status, check_start_time) as job_status, - check_name as job_name, - argMax(test_status, check_start_time) as status, - test_name, - report_url as results_link, - task_url - FROM `gh-data`.checks - WHERE commit_sha='{commit_sha}' AND head_ref IN ('{branch_name}', 'refs/tags/{branch_name}') - GROUP BY check_name, test_name, report_url, task_url - ) + query = f"""{_checks_latest_test_status_cte(commit_sha, branch_name)} + SELECT job_status, job_name, status AS test_status, test_name, results_link + FROM latest_test_status WHERE test_status='BROKEN' AND test_name IN ({','.join(f"'{test}'" for test in known_fails.keys())}) ORDER BY job_name, test_name @@ -219,22 +260,12 @@ def get_checks_errors(client: Client, commit_sha: str, branch_name: str): """ Get checks that have status 'error' for the given commit and branch. """ - query = f"""SELECT job_status, job_name, status as test_status, test_name, results_link - FROM ( - SELECT - argMax(check_status, check_start_time) as job_status, - check_name as job_name, - argMax(test_status, check_start_time) as status, - test_name, - report_url as results_link, - task_url - FROM `gh-data`.checks - WHERE commit_sha='{commit_sha}' AND head_ref IN ('{branch_name}', 'refs/tags/{branch_name}') - GROUP BY check_name, test_name, report_url, task_url - ) - WHERE job_status=='error' - ORDER BY job_name, test_name - """ + query = f"""{_checks_latest_test_status_cte(commit_sha, branch_name)} + SELECT job_status, job_name, status AS test_status, test_name, results_link + FROM latest_test_status + WHERE job_status == 'error' + ORDER BY job_name, test_name + """ return client.query_dataframe(query)