Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
98 changes: 95 additions & 3 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -207,12 +207,47 @@ jobs:
working-directory: frontend
run: bun test --coverage --coverage-reporter=lcov --coverage-dir tests/coverage --reporter=junit --reporter-outfile tests/results/junit.xml

# Benchmark harness (proven-tests-and-benches discipline): informational
# medians vs committed baseline; JSON result ships as an artifact.
# Benchmark harness (proven-tests-and-benches discipline): medians vs committed baseline; JSON result ships as an artifact.
# Milestone 2: now includes table_loading, epistemic_parsing, duckdb_aggregation, permanova_nmds, tree_rendering workloads
- name: Benchmark frontend
working-directory: frontend
run: bun run bench -- --json bench/results/results.json

- name: Check frontend benchmark regression >10% (Milestone 2)
working-directory: frontend
run: |
echo "Checking frontend benchmark regression >10% vs baseline"
node -e '
const fs = require("fs");
const baselinePath = "bench/baseline.json";
const resultsPath = "bench/results/results.json";
if (!fs.existsSync(baselinePath) || !fs.existsSync(resultsPath)) {
console.log("No baseline or results — skipping regression check (first run)");
process.exit(0);
}
const baseline = JSON.parse(fs.readFileSync(baselinePath, "utf8"));
const results = JSON.parse(fs.readFileSync(resultsPath, "utf8"));
let failed = false;
for (const r of results.results) {
const b = baseline.results.find(x => x.name === r.name);
if (b) {
const delta = (r.median_ns - b.median_ns) / b.median_ns * 100;
const status = Math.abs(delta) > 10 ? "FAIL" : "PASS";
console.log(`${status} ${r.name}: ${delta.toFixed(1)}% vs baseline ${b.median_ns} ns (current ${r.median_ns} ns)`);
if (Math.abs(delta) > 10) {
console.error(`::error::Regression >10% for ${r.name}: ${delta.toFixed(1)}%`);
failed = true;
}
} else {
console.log(`NEW ${r.name}: no baseline, will be added`);
}
}
if (failed) {
console.error("Failing CI due to >10% regression");
process.exit(1);
}
'

- name: Upload frontend test & benchmark artifacts
if: always()
uses: actions/upload-artifact@v4
Expand All @@ -222,6 +257,7 @@ jobs:
frontend/tests/results/junit.xml
frontend/tests/coverage/lcov.info
frontend/bench/results/results.json
frontend/bench/baseline.json
if-no-files-found: warn

- name: Build frontend
Expand Down Expand Up @@ -253,10 +289,66 @@ jobs:
- name: Process coverage
uses: julia-actions/julia-processcoverage@v1

- name: Upload coverage artifact (local, Codecov removed)
- name: Upload coverage artifact (local, Codecov removed per Milestone 2)
if: always()
uses: actions/upload-artifact@v4
with:
name: julia-coverage-lcov
path: lcov.info
if-no-files-found: warn

# Comprehensive benchmarks (Milestone 2) — table loading, epistemic parsing, DuckDB aggregation, PERMANOVA/NMDS, tree rendering
- name: Benchmark Julia comprehensive
run: |
julia --project=. bench/table_loading/benchmark.jl
julia --project=. bench/epistemic_parsing/benchmark.jl
julia --project=. bench/duckdb_aggregation/benchmark.jl
julia --project=. bench/permanova_nmds/benchmark.jl
julia --project=. bench/tree_rendering/benchmark.jl
julia --project=. bench/comprehensive_benchmark.jl

- name: Check benchmark regression >10%
run: |
echo "Checking for >10% regression in Julia benchmarks (fail if found)"
# Each benchmark script itself fails on >10% when CI=true, so this step is informational
# Here we also check that baseline.json files exist for each category
for cat in table_loading epistemic_parsing duckdb_aggregation permanova_nmds tree_rendering; do
if [ ! -f bench/$cat/baseline.json ]; then
echo "::warning::No baseline.json for $cat — first run will create it"
else
echo "Found baseline for $cat: $(cat bench/$cat/baseline.json | head -c 200)"
fi
done
if [ -f bench/results/comprehensive_results.json ]; then
echo "Comprehensive results: $(cat bench/results/comprehensive_results.json | head -c 500)"
fi

- name: Upload Julia benchmark artifacts
if: always()
uses: actions/upload-artifact@v4
with:
name: julia-benchmarks-comprehensive
path: |
bench/*/baseline.json
bench/results/comprehensive_results.json
bench/**/baseline.json
if-no-files-found: warn

# New test categories: analysis-config and cladistic-explorer (Milestone 2)
- name: Test analysis-config category
run: |
echo "Running analysis-config test category (if present)"
if [ -f test/unit/test_analysis_config.jl ]; then
julia --project=. -e 'using Test; using MetaManifold; include("test/unit/test_analysis_config.jl")'
else
echo "test_analysis_config.jl not present on main — skipping (will be present on feature branches)"
fi

- name: Test cladistic-explorer category
run: |
echo "Running cladistic-explorer test category (if present)"
if [ -f test/unit/test_clade_cumulus.jl ]; then
julia --project=. -e 'using Test; using MetaManifold; include("test/unit/test_clade_cumulus.jl")'
else
echo "test_clade_cumulus.jl not present on main — skipping (will be present on feature branches)"
fi
92 changes: 92 additions & 0 deletions bench/comprehensive_benchmark.jl
Original file line number Diff line number Diff line change
@@ -0,0 +1,92 @@
# SPDX-License-Identifier: AGPL-3.0-only
# SPDX-FileCopyrightText: 2026 Jonathan D.A. Jewell (hyperpolymath) <j.d.a.jewell@open.ac.uk>
"""
Comprehensive benchmark runner for Milestone 2

Runs all benchmark categories:
- table_loading
- epistemic_parsing
- duckdb_aggregation
- permanova_nmds
- tree_rendering

Fails on >10% regression vs committed baselines when CI=true
Uploads artifacts via GitHub Actions (see .github/workflows/ci.yml)
"""

using Logging

const BENCH_DIR = @__DIR__

function run_category(cat::String)
bench_file = joinpath(BENCH_DIR, cat, "benchmark.jl")
if !isfile(bench_file)
@warn "Benchmark file not found" cat bench_file
return nothing
end
println("\n" * "="^60)
println("Running benchmark category: $cat")
println("="^60)
# Include and run
mod = Module()
Base.include(mod, bench_file)
if isdefined(mod, :run_benchmarks)
return Base.invokelatest(mod.run_benchmarks)
else
@warn "No run_benchmarks defined in $bench_file"
return nothing
end
end

function main()
categories = [
"table_loading",
"epistemic_parsing",
"duckdb_aggregation",
"permanova_nmds",
"tree_rendering"
]

all_results = Dict{String, Any}()

for cat in categories
try
results = run_category(cat)
all_results[cat] = results
catch e
@error "Benchmark category failed" cat exception=(e, catch_backtrace())
all_results[cat] = Dict("error" => string(e))
if get(ENV, "CI", "false") == "true"
# Don't exit immediately, continue to run others for full report
# But mark failure
println("::error::Benchmark $cat failed: $e")
end
end
end

# Write combined results
results_path = joinpath(BENCH_DIR, "results", "comprehensive_results.json")
mkpath(dirname(results_path))
try
using JSON3
open(results_path, "w") do io
JSON3.write(io, all_results)
end
println("\nWrote combined results to $results_path")
catch e
@warn "Failed to write JSON results" exception=e
# Fallback: write simple text
open(results_path * ".txt", "w") do io
println(io, all_results)
end
end

println("\n" * "="^60)
println("Comprehensive benchmark complete")
println("="^60)
return all_results
end

if abspath(PROGRAM_FILE) == @__FILE__
main()
end
7 changes: 7 additions & 0 deletions bench/duckdb_aggregation/baseline.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
{
"aggregate_by_taxon": 0.02,
"venn_taxa_present": 0.015,
"bar_chart": 0.005,
"taxa_bar_chart": 0.005,
"alpha_chart": 0.003
}
123 changes: 123 additions & 0 deletions bench/duckdb_aggregation/benchmark.jl
Original file line number Diff line number Diff line change
@@ -0,0 +1,123 @@
# SPDX-License-Identifier: AGPL-3.0-only
# SPDX-FileCopyrightText: 2026 Jonathan D.A. Jewell (hyperpolymath) <j.d.a.jewell@open.ac.uk>
"""
Benchmark for DuckDB aggregation pathways

Measures:
- aggregate_by_taxon (SUM COALESCE, Unclassified fallback)
- combined_counts_across_runs
- venn_taxa_present
- bar_chart and taxa_bar_chart generation
- alpha_chart generation
"""

using DuckDB, DataFrames, DBInterface
using MetaManifold.Analysis: aggregate_by_taxon, venn_taxa_present, alpha_chart, bar_chart, taxa_bar_chart, sample_columns, filtered_counts
using Random

function _create_mock_db(n_samples::Int=20, n_features::Int=1000)
db = DuckDB.DB()
con = DBInterface.connect(db)
sample_cols = ["Sample$(i)" for i in 1:n_samples]
df = DataFrame()
df.SeqName = ["ASV$(i)" for i in 1:n_features]
df.Domain = rand(["Bacteria", "Archaea"], n_features)
df.Phylum = rand(["Firmicutes", "Bacteroidetes", "Proteobacteria"], n_features)
df.Genus = rand(["Bacteroides", "Prevotella", "Faecalibacterium", "Escherichia"], n_features)
df.Species = rand(["B. fragilis", "P. copri", "F. prausnitzii", "E. coli"], n_features)
for sc in sample_cols
df[!, sc] = rand(0:1000, n_features)
end
DuckDB.register_data_frame(con, df, "merged_df")
DBInterface.execute(con, "CREATE TABLE merged AS SELECT * FROM merged_df")
return con, sample_cols
end

function bench_aggregate_by_taxon(con, sample_cols)
@elapsed aggregate_by_taxon(con, "merged", sample_cols, "Genus")
end

function bench_venn_taxa_present(con, sample_cols)
# Split samples into 2 groups
g1 = sample_cols[1:div(length(sample_cols),2)]
g2 = sample_cols[div(length(sample_cols),2)+1:end]
@elapsed venn_taxa_present(con, "merged", sample_cols, [g1, g2], "Genus")
end

function bench_bar_chart()
labels = ["GroupA", "GroupB", "GroupC"]
counts = rand(100, 3) * 1000
@elapsed bar_chart(labels, counts, ["Taxon$i" for i in 1:100], top_n=20)
end

function bench_taxa_bar_chart()
labels = ["Taxon$i" for i in 1:50]
counts = rand(50, 10) * 100
sample_names = ["Sample$i" for i in 1:10]
@elapsed taxa_bar_chart(labels, counts, sample_names, top_n=20)
end

function bench_alpha_chart()
sample_names = ["Sample$i" for i in 1:20]
richness = rand(50:500, 20)
shannon = rand(1.0:0.1:5.0, 20)
simpson = rand(0.5:0.01:0.99, 20)
groups = [rand(["Control", "Disease"]) for _ in 1:20]
@elapsed alpha_chart(sample_names, richness, shannon, simpson, groups)
end

function run_benchmarks(; n_samples=20, n_features=1000, reps=5)
println("=== DuckDB Aggregation Benchmark ===")
con, sample_cols = _create_mock_db(n_samples, n_features)

results = Dict{String, Vector{Float64}}()
for name in ["aggregate_by_taxon", "venn_taxa_present", "bar_chart", "taxa_bar_chart", "alpha_chart"]
results[name] = Float64[]
end

for _ in 1:reps
push!(results["aggregate_by_taxon"], bench_aggregate_by_taxon(con, sample_cols))
push!(results["venn_taxa_present"], bench_venn_taxa_present(con, sample_cols))
push!(results["bar_chart"], bench_bar_chart())
push!(results["taxa_bar_chart"], bench_taxa_bar_chart())
push!(results["alpha_chart"], bench_alpha_chart())
end

for (name, times) in results
med = median(times)
println("$name: median $(round(med*1000, digits=2)) ms over $reps reps")
end

baseline_path = joinpath(@__DIR__, "baseline.json")
if isfile(baseline_path)
using JSON3
baseline = JSON3.read(read(baseline_path, String))
println("\nBaseline comparison (fail on >10% regression):")
for (name, times) in results
med = median(times)
if haskey(baseline, name)
base_med = baseline[name]
delta = (med - base_med) / base_med * 100
status = abs(delta) > 10 ? "FAIL" : "PASS"
println("$status $name: $(round(delta, digits=1))% vs baseline $(round(base_med*1000, digits=2)) ms")
if abs(delta) > 10 && get(ENV, "CI", "false") == "true"
@error "Regression >10% for $name" delta
exit(1)
end
end
end
else
println("\nNo baseline.json — saving current as baseline")
using JSON3
baseline = Dict(name => median(times) for (name, times) in results)
open(baseline_path, "w") do io
JSON3.write(io, baseline)
end
end

return results
end

if abspath(PROGRAM_FILE) == @__FILE__
run_benchmarks()
end
7 changes: 7 additions & 0 deletions bench/epistemic_parsing/baseline.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
{
"avec_fibre_parse": 0.002,
"epistemic_colour": 0.001,
"cloud_size": 0.001,
"present_in_every": 0.005,
"warrant_logic": 0.001
}
Loading
Loading