Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
123 changes: 77 additions & 46 deletions .github/actions/create_workflow_report/create_workflow_report.py
Original file line number Diff line number Diff line change
Expand Up @@ -148,28 +148,79 @@ def get_run_details(run_url: str) -> dict:
return response.json()


def _checks_latest_test_status_cte(commit_sha: str, branch_name: str) -> str:
"""
Shared filtering for gh-data.checks: anchor time selects the latest result batch
per check_name.

Rows that must not set the anchor (but are still kept if their time is >= anchor):
- Stateless teardown: check_name LIKE 'Stateless%' AND test_name not matching ^[0-9]{5}
- Empty test_name: CIDB job-level parent rows
- STARTED / COMPLETED: job lifecycle markers with their own stopwatches; COMPLETED
always lands after the tests and would otherwise become the sole surviving batch,
hiding all FAILs

Keep rows with check_start_time >= anchor so the latest main batch and any later
teardown/marker rows are included. Earlier batches (failed attempts before a rerun
uploaded a newer uniform timestamp) are dropped, including synthetic rows such as
'Job Timeout Expired' that only exist in the attempt that failed.
"""
return f"""WITH checks_with_anchor AS (
SELECT
check_name,
test_name,
report_url,
check_status,
test_status,
check_start_time,
maxIf(
check_start_time,
test_name != ''
AND test_name NOT IN ('STARTED', 'COMPLETED')
AND NOT (
check_name LIKE 'Stateless%'
AND NOT match(test_name, '^[0-9]{{5}}')
)
) OVER (PARTITION BY check_name) AS latest_check_start_time
FROM `gh-data`.checks
WHERE commit_sha = '{commit_sha}'
AND head_ref IN ('{branch_name}', 'refs/tags/{branch_name}')
),
rows_from_latest_check_run AS (
SELECT
check_name,
test_name,
report_url,
check_status,
test_status,
check_start_time
FROM checks_with_anchor
WHERE check_start_time >= latest_check_start_time
),
latest_test_status AS (
SELECT
argMax(check_status, check_start_time) AS job_status,
check_name AS job_name,
argMax(test_status, check_start_time) AS status,
test_name,
report_url AS results_link
FROM rows_from_latest_check_run
GROUP BY check_name, test_name, report_url
)"""


def get_checks_fails(client: Client, commit_sha: str, branch_name: str):
"""
Get tests that did not succeed for the given commit and branch.
Exclude checks that have status 'error' as they are counted in get_checks_errors.
"""
query = f"""SELECT job_status, job_name, status as test_status, test_name, results_link
FROM (
SELECT
argMax(check_status, check_start_time) as job_status,
check_name as job_name,
argMax(test_status, check_start_time) as status,
test_name,
report_url as results_link,
task_url
FROM `gh-data`.checks
WHERE commit_sha='{commit_sha}' AND head_ref IN ('{branch_name}', 'refs/tags/{branch_name}')
GROUP BY check_name, test_name, report_url, task_url
)
WHERE test_status IN ('FAIL', 'ERROR')
AND job_status!='error'
ORDER BY job_name, test_name
"""
query = f"""{_checks_latest_test_status_cte(commit_sha, branch_name)}
SELECT job_status, job_name, status AS test_status, test_name, results_link
FROM latest_test_status
WHERE test_status IN ('FAIL', 'ERROR')
AND job_status != 'error'
ORDER BY job_name, test_name
"""
return client.query_dataframe(query)


Expand All @@ -182,19 +233,9 @@ def get_checks_known_fails(
if len(known_fails) == 0:
return pd.DataFrame()

query = f"""SELECT job_status, job_name, status as test_status, test_name, results_link
FROM (
SELECT
argMax(check_status, check_start_time) as job_status,
check_name as job_name,
argMax(test_status, check_start_time) as status,
test_name,
report_url as results_link,
task_url
FROM `gh-data`.checks
WHERE commit_sha='{commit_sha}' AND head_ref IN ('{branch_name}', 'refs/tags/{branch_name}')
GROUP BY check_name, test_name, report_url, task_url
)
query = f"""{_checks_latest_test_status_cte(commit_sha, branch_name)}
SELECT job_status, job_name, status AS test_status, test_name, results_link
FROM latest_test_status
WHERE test_status='BROKEN'
AND test_name IN ({','.join(f"'{test}'" for test in known_fails.keys())})
ORDER BY job_name, test_name
Expand All @@ -219,22 +260,12 @@ def get_checks_errors(client: Client, commit_sha: str, branch_name: str):
"""
Get checks that have status 'error' for the given commit and branch.
"""
query = f"""SELECT job_status, job_name, status as test_status, test_name, results_link
FROM (
SELECT
argMax(check_status, check_start_time) as job_status,
check_name as job_name,
argMax(test_status, check_start_time) as status,
test_name,
report_url as results_link,
task_url
FROM `gh-data`.checks
WHERE commit_sha='{commit_sha}' AND head_ref IN ('{branch_name}', 'refs/tags/{branch_name}')
GROUP BY check_name, test_name, report_url, task_url
)
WHERE job_status=='error'
ORDER BY job_name, test_name
"""
query = f"""{_checks_latest_test_status_cte(commit_sha, branch_name)}
SELECT job_status, job_name, status AS test_status, test_name, results_link
FROM latest_test_status
WHERE job_status == 'error'
ORDER BY job_name, test_name
"""
return client.query_dataframe(query)


Expand Down
6 changes: 4 additions & 2 deletions tests/ci/ci_config.py
Original file line number Diff line number Diff line change
Expand Up @@ -374,16 +374,17 @@ class CI:
),
JobNames.INTEGRATION_TEST_ASAN: CommonJobConfigs.INTEGRATION_TEST.with_properties(
required_builds=[BuildNames.PACKAGE_ASAN],
timeout=4 * 3600,
num_batches=8,
),
JobNames.INTEGRATION_TEST_ASAN_OLD_ANALYZER: CommonJobConfigs.INTEGRATION_TEST.with_properties(
required_builds=[BuildNames.PACKAGE_ASAN],
timeout=3 * 3600,
timeout=4 * 3600,
num_batches=8,
),
JobNames.INTEGRATION_TEST_TSAN: CommonJobConfigs.INTEGRATION_TEST.with_properties(
required_builds=[BuildNames.PACKAGE_TSAN],
timeout=3 * 3600,
timeout=4 * 3600,
num_batches=8,
),
JobNames.INTEGRATION_TEST_AARCH64: CommonJobConfigs.INTEGRATION_TEST.with_properties(
Expand All @@ -393,6 +394,7 @@ class CI:
),
JobNames.INTEGRATION_TEST: CommonJobConfigs.INTEGRATION_TEST.with_properties(
required_builds=[BuildNames.PACKAGE_RELEASE],
timeout=3 * 3600,
num_batches=8,
# release_only=True,
),
Expand Down
Original file line number Diff line number Diff line change
@@ -1 +1,2 @@
SET max_expanded_ast_elements = 10000;
SELECT 1 AS a, a + a AS b, b + b AS c, c + c AS d, d + d AS e, e + e AS f, f + f AS g, g + g AS h, h + h AS i, i + i AS j, j + j AS k, k + k AS l, l + l AS m, m + m AS n, n + n AS o, o + o AS p, p + p AS q, q + q AS r, r + r AS s, s + s AS t, t + t AS u, u + u AS v, v + v AS w, w + w AS x, x + x AS y, y + y AS z; -- { serverError BAD_ARGUMENTS, 168 }
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
-- Tags: no-fasttest, long, no-asan, no-ubsan, no-debug
-- Tags: no-fasttest, long, no-asan, no-ubsan, no-tsan, no-debug
-- ^^ Disable test for slow builds: generating data takes time but a sufficiently large data set
-- is necessary for different hnsw_candidate_list_size_for_search settings to make a difference

Expand Down
Original file line number Diff line number Diff line change
@@ -1,4 +1,5 @@
-- Tags: no-fasttest, no-ordinary-database
-- Tags: no-fasttest, no-ordinary-database, no-tsan
-- no-tsan: generating data takes too long

-- Tests correctness of vector similarity index with > 1 mark

Expand Down
Loading