Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 15 additions & 1 deletion .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,21 @@ on:

concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: true
# A push to `main` QUEUES; a pull_request head update still cancels.
#
# Measured 2026-09-21: the Julia job takes ~31 min, and `main` was merged to at
# 14:25, 14:28 and 14:43. Every run was cancelled 17-18 min in by the next one,
# so three consecutive commits to `main` produced NO test verdict at all --
# not a pass, not a failure, nothing. Cancelling is right for a PR, where a
# verdict on a superseded head is worthless; it is wrong for `main`, where each
# commit is a thing we actually want a recorded answer about.
#
# Trade-off, stated plainly: pushes to `main` now run serially, so a burst of
# N merges takes N x ~31 min to drain. That is the cost of getting an answer.
# This does NOT rescue an upstream PR whose head keeps moving (e.g.
# JoshuaJewell#6, whose head IS this fork's `main`) -- only letting `main`
# settle for ~31 min does that.
cancel-in-progress: ${{ github.event_name != 'push' }}

jobs:
# Estate hygiene gates (standards/RSR alignment): cheap, run alongside the
Expand Down
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -265,6 +265,7 @@ docs/*
frontend/tests/results/
frontend/tests/coverage/
frontend/bench/results/
bench/results/

# translation temp files

Expand Down
15 changes: 13 additions & 2 deletions Manifest.toml

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

2 changes: 2 additions & 0 deletions Project.toml
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@ name = "MetaManifold"
uuid = "ea959a01-5458-4cf1-8f0d-4ed45446c396"

[deps]
BenchmarkTools = "6e4b80f9-dd63-53aa-95a3-0cdb28fa8baf"
CSV = "336ed68f-0bac-5ca0-87d4-7b16caf5d00b"
DBInterface = "a10d1c49-ce27-4219-8d33-6db1a4562965"
DataFrames = "a93c6f00-e57d-5684-b7b6-d8193f3e46c0"
Expand All @@ -23,6 +24,7 @@ XLSX = "fdbf4ff8-1666-58a4-91e7-1b58723a45e0"
YAML = "ddb6d928-2868-570f-bddf-ab3f9cf99eb6"

[compat]
BenchmarkTools = "1.8.0"
CSV = "0.10"
DataFrames = "1.8"
PackageCompiler = "2.2.5"
Expand Down
43 changes: 36 additions & 7 deletions bench/duckdb_aggregation/benchmark.jl
Original file line number Diff line number Diff line change
Expand Up @@ -36,36 +36,65 @@ function _create_mock_db(n_samples::Int=20, n_features::Int=1000)
end

function bench_aggregate_by_taxon(con, sample_cols)
@elapsed aggregate_by_taxon(con, "merged", sample_cols, "Genus")
# `aggregate_by_taxon(con, table, sample_cols, rank, where_clause, where_params)`
# -- six arguments. This previously passed four, omitting the trailing filter
# pair. `src/analysis/analysis.jl:141` has required all six since the function
# was introduced, and `test/unit/test_analysis_duckdb.jl` calls it that way.
# The call was unreachable until the bench steps were wired into CI, so it
# failed the moment it first ran. An empty filter benchmarks the unfiltered
# aggregation, which is what the header comment says this measures.
@elapsed aggregate_by_taxon(con, "merged", sample_cols, "Genus", "", [])
end

function bench_venn_taxa_present(con, sample_cols)
# Split samples into 2 groups
g1 = sample_cols[1:div(length(sample_cols),2)]
g2 = sample_cols[div(length(sample_cols),2)+1:end]
@elapsed venn_taxa_present(con, "merged", sample_cols, [g1, g2], "Genus")
# `venn_taxa_present(con, table, sample_cols, rank_col, where_clause, where_params)`
# returns the taxa present in ONE sample set (`src/analysis/analysis.jl:161`).
# It has never accepted a list of groups: the previous call passed `[g1, g2]`
# as a fourth argument in a five-argument form that matches no method. A Venn
# is assembled by calling it once per group, which is what this now measures.
@elapsed begin
venn_taxa_present(con, "merged", g1, "Genus", "", [])
venn_taxa_present(con, "merged", g2, "Genus", "", [])
end
end

function bench_bar_chart()
labels = ["GroupA", "GroupB", "GroupC"]
# `bar_chart(segment_labels, sample_names, counts; top_n, ...)` -- the counts
# matrix is (segments x samples), as `src/analysis/analysis.jl:403` (column
# totals are per-sample) and `test/unit/test_analysis.jl:35` ("2 taxa x 2
# samples") both establish. The previous call passed (labels, counts, names),
# putting the matrix in the `sample_names` position, so no method matched.
# The 100x3 matrix means 100 segments (taxa) across 3 samples (groups), so the
# taxon vector is the segment labels and the group vector the sample names.
segment_labels = ["Taxon$i" for i in 1:100]
sample_names = ["GroupA", "GroupB", "GroupC"]
counts = rand(100, 3) * 1000
@elapsed bar_chart(labels, counts, ["Taxon$i" for i in 1:100], top_n=20)
@elapsed bar_chart(segment_labels, sample_names, counts, top_n=20)
end

function bench_taxa_bar_chart()
labels = ["Taxon$i" for i in 1:50]
counts = rand(50, 10) * 100
sample_names = ["Sample$i" for i in 1:10]
@elapsed taxa_bar_chart(labels, counts, sample_names, top_n=20)
# `taxa_bar_chart(taxon_labels, sample_names, counts; ...)` -- arguments 2 and
# 3 were transposed here. The 50x10 matrix is already (taxa x samples), which
# is the orientation the function wants; only the call order was wrong.
@elapsed taxa_bar_chart(labels, sample_names, counts, top_n=20)
end

function bench_alpha_chart()
sample_names = ["Sample$i" for i in 1:20]
richness = rand(50:500, 20)
shannon = rand(1.0:0.1:5.0, 20)
simpson = rand(0.5:0.01:0.99, 20)
groups = [rand(["Control", "Disease"]) for _ in 1:20]
@elapsed alpha_chart(sample_names, richness, shannon, simpson, groups)
# `alpha_chart(sample_names, richness, shannon, simpson)` takes exactly four
# arguments (`src/analysis/analysis.jl:312`); it has no grouping parameter, and
# the `groups` vector built here was never consumed by any method. Grouped
# alpha display is `alpha_boxplot`'s job, benchmarked in permanova_nmds.
@elapsed alpha_chart(sample_names, richness, shannon, simpson)
end

function run_benchmarks(; n_samples=20, n_features=1000, reps=5)
Expand Down
16 changes: 13 additions & 3 deletions bench/permanova_nmds/benchmark.jl
Original file line number Diff line number Diff line change
Expand Up @@ -50,15 +50,25 @@ function bench_normalise_counts(n::Int=100, n_features::Int=1000)
end

function bench_alpha_boxplot(n_groups::Int=3, n_per_group::Int=10)
groups = []
# Two defects, both latent until the bench steps were wired into CI.
#
# 1. `groups = []` is a `Vector{Any}`, which matches NEITHER `alpha_boxplot`
# method (`src/analysis/analysis.jl:734` and `:840` both dispatch on a
# concrete `Vector{Tuple{...}}`). The elements pushed below are already the
# right 5-tuple shape, so annotating the container is all that is needed.
# 2. `metric=` is not a keyword of either method. The accepted keywords are
# show_points, annotate_significance, pairwise_brackets, paired_samples and
# significance_test. This file's own header says it measures "alpha_boxplot
# with significance", so `annotate_significance=true` is what was meant --
# and it exercises more of the function than the default would.
groups = Tuple{String, Vector{String}, Vector{Int}, Vector{Float64}, Vector{Float64}}[]
for g in 1:n_groups
sample_names = ["Group$(g)_Sample$(i)" for i in 1:n_per_group]
counts = rand(50:500, n_per_group)
shannon_vals = rand(1.0:0.1:5.0, n_per_group)
simpson_vals = rand(0.5:0.01:0.99, n_per_group)
push!(groups, ("Group$g", sample_names, collect(1:n_per_group), shannon_vals, simpson_vals))
end
@elapsed alpha_boxplot(groups, metric="shannon")
@elapsed alpha_boxplot(groups, annotate_significance=true)
end

function bench_nmds_chart(n::Int=20)
Expand Down
11 changes: 8 additions & 3 deletions bench/table_loading/benchmark.jl
Original file line number Diff line number Diff line change
Expand Up @@ -46,8 +46,13 @@ function bench_filtered_counts(con, sample_cols, table::String="merged")
@elapsed filtered_counts(con, table, sample_cols, "", [])
end

function bench_filtered_df(con, sample_cols, table::String="merged")
@elapsed filtered_df(con, table, sample_cols, "", [], 1, 100)
function bench_filtered_df(con, table::String="merged")
# `filtered_df(con, table, where_clause, where_params)` -- four arguments.
# This previously passed seven (sample_cols + an offset/limit pair), a
# paginated signature that has never existed on any commit: the function has
# taken these four arguments since `2987464`. The call was unreachable until
# the bench steps were wired into CI, so it failed the moment it first ran.
@elapsed filtered_df(con, table, "", [])
end

function bench_taxonomy_levels(con, table::String="merged")
Expand All @@ -72,7 +77,7 @@ function run_benchmarks(; n_samples=20, n_features=1000, reps=5)
for _ in 1:reps
push!(results["sample_columns"], bench_sample_columns(con))
push!(results["filtered_counts"], bench_filtered_counts(con, sample_cols))
push!(results["filtered_df"], bench_filtered_df(con, sample_cols))
push!(results["filtered_df"], bench_filtered_df(con))
push!(results["taxonomy_levels"], bench_taxonomy_levels(con))
push!(results["taxon_column"], bench_taxon_column())
end
Expand Down
Loading