Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
136 changes: 136 additions & 0 deletions bench/analysis_config/benchmark.jl
Original file line number Diff line number Diff line change
@@ -0,0 +1,136 @@
# SPDX-License-Identifier: AGPL-3.0-only
"""
Benchmark for AnalysisConfig layer

Ensures runtime and memory regression <10% vs baseline.

Measures:
- Config creation + validation
- JSON serialization roundtrip
- DOI bundle creation
- Epistemic validation (present_in_every_admissible_world)
- CladeCumulus tree building

Fail CI on >10% regression.
"""

using BenchmarkTools
using OrderedCollections
using MetaManifold.AnalysisConfig
using MetaManifold.Epistemic
using MetaManifold.CladeCumulus

const SUITE = BenchmarkGroup()

SUITE["config_creation"] = @benchmarkable begin
norm = AnalysisConfig.NormalizationConfig(method="size_factors")
cfg = AnalysisConfig.AnalysisConfigStruct(
method="nb_glm",
formula="~ group + batch",
metadata_columns=["group", "batch", "age"],
normalization=norm,
created_by="benchmark"
)
end

SUITE["config_validation"] = @benchmarkable begin
norm = AnalysisConfig.NormalizationConfig(method="clr", pseudocount=0.5)
cfg = AnalysisConfig.AnalysisConfigStruct(
method="clr_lm",
formula="~ group",
metadata_columns=["group"],
normalization=norm,
)
AnalysisConfig.validate_config(cfg, ["group", "batch"]; strict=false)
end

SUITE["json_roundtrip"] = @benchmarkable begin
norm = AnalysisConfig.NormalizationConfig(method="size_factors")
cfg = AnalysisConfig.AnalysisConfigStruct(
method="nb_glm",
formula="~ group",
metadata_columns=["group"],
normalization=norm,
)
json = AnalysisConfig.to_json(cfg)
AnalysisConfig.from_json(json)
end

SUITE["doi_bundle"] = @benchmarkable begin
norm = AnalysisConfig.NormalizationConfig(method="size_factors")
cfg = AnalysisConfig.AnalysisConfigStruct(
method="nb_glm",
formula="~ group",
metadata_columns=["group"],
normalization=norm,
)
result = AnalysisConfig.AnalysisResult(
config_id=cfg.id,
config_hash=cfg.hash,
method=cfg.method,
results=OrderedDict{String,Any}("taxon1" => OrderedDict("p" => 0.01))
)
mktempdir() do tmp
AnalysisConfig.create_doi_bundle(cfg, result, joinpath(tmp, "bundle"); authors=["Bench"], title="Bench")
end
end

SUITE["epistemic_present_in_every"] = @benchmarkable begin
c1 = Epistemic.Candidate{Tuple{Int,Int},Int}((1,1), true, true)
c2 = Epistemic.Candidate{Tuple{Int,Int},Int}((2,0), true, true)
case_bounded = Epistemic.Case{Tuple{Int,Int},Int}(c1, [c1, c2])
Epistemic.present_in_every_admissible_world(case_bounded, world -> world[1] != 0)
end

SUITE["clade_tree_build"] = @benchmarkable begin
rows = [
Dict{String,Any}("Domain" => "Bacteria", "Phylum" => "Firmicutes", "Genus" => "Lacto$(i)", "total" => Float64(10+i), "avec_fibre" => true, "epistemic_status" => "present_in_every_admissible_world", "residual_count" => i % 5)
for i in 1:100
]
tree = CladeCumulus.build_clade_tree(rows)
CladeCumulus.cumulative_frequencies(tree)
end

# Run and check regression
function run_benchmarks(; baseline_path::String=joinpath(@__DIR__, "baseline.json"))
results = run(SUITE, verbose=true)

# Save current as new baseline if no baseline exists
if !isfile(baseline_path)
BenchmarkTools.save(baseline_path, results)
println("No baseline found, saved current as baseline at $baseline_path")
return results
end

baseline = BenchmarkTools.load(baseline_path)[1]

# Compare, fail on >10% regression in time or memory
for (key, trial) in results
if haskey(baseline, key)
base_trial = baseline[key]
# Median time comparison
curr_time = BenchmarkTools.prettytime(BenchmarkTools.median(trial).time)
base_time = BenchmarkTools.prettytime(BenchmarkTools.median(base_trial).time)

# Ratio
time_ratio = BenchmarkTools.median(trial).time / BenchmarkTools.median(base_trial).time
mem_ratio = BenchmarkTools.median(trial).memory / max(1, BenchmarkTools.median(base_trial).memory)

println("$key: time ratio $(round(time_ratio, digits=3)) (baseline $base_time vs current $curr_time), memory ratio $(round(mem_ratio, digits=3))")

if time_ratio > 1.10
error("Benchmark regression >10% in time for $key: ratio $time_ratio (threshold 1.10)")
end
if mem_ratio > 1.10
error("Benchmark regression >10% in memory for $key: ratio $mem_ratio")
end
end
end

println("All benchmarks within 10% regression threshold — OK")
return results
end

if abspath(PROGRAM_FILE) == @__FILE__
run_benchmarks()
end
243 changes: 243 additions & 0 deletions config/schemas/analysis_config.ncl
Original file line number Diff line number Diff line change
@@ -0,0 +1,243 @@
# SPDX-License-Identifier: AGPL-3.0-only
# AnalysisConfig Nickel contract — hyperpolymath/standards style
# From 1-formats/k9/*.ncl and .machine_readable/contractiles/_base.ncl
# Implements BH mandatory, DANGER banner, advanced validation

let AnalysisMethod = std.enum.TagOrString & [| 'nb_glm, 'clr_lm, 'ilr_lm, 'logistic |] in
let NormalizationMethod = std.enum.TagOrString & [| 'none, 'rarefy, 'relative, 'size_factors, 'clr, 'ilr, 'presence_absence |] in
let CorrectionMethod = std.enum.TagOrString & [| 'BH, 'FDR, 'Benjamini-Hochberg |] in
let DispersionMethod = std.enum.TagOrString & [| 'parametric, 'local, 'mean, 'pooled, 'glmGamPoi |] in
let ZeroHandling = std.enum.TagOrString & [| 'pseudocount, 'multiplicative_replacement, 'bayesian_multiplicative, 'refuse |] in
let IlrBasis = std.enum.TagOrString & [| 'default, 'phylogenetic, 'sequential_binary_partition, 'balance_dendrogram |] in

let DANGER_TOKEN = "I_UNDERSTAND_THE_RISK_AND_WANT_TO_OVERRIDE_BH" in

let ValidFormula = fun label value =>
if std.string.contains ";" value then
'Error { message = "Formula contains forbidden ';' (injection prevention)" }
else if std.string.contains "`" value then
'Error { message = "Formula contains forbidden backtick" }
else if std.string.contains "$" value then
'Error { message = "Formula contains forbidden '$'" }
else if !(std.string.contains "~" value) then
'Error { message = "Formula must contain '~' (R-style), e.g. '~ group'" }
else if std.string.length (std.string.trim value) < 2 then
'Error { message = "Formula must be non-empty and reference at least one metadata column" }
else
'Ok value
in

let PseudocountContract = fun label value =>
if value <= 0 then
'Error { message = "pseudocount must be >0 for CLR/ILR (log(0) undefined). Got %{std.to_string value}" }
else if value >= 1 then
# Warn but allow — typical is 0.5
'Ok value
else
'Ok value
in

let PrevalenceContract = fun label value =>
if value < 0 || value > 1 then
'Error { message = "min_prevalence must be in [0,1], got %{std.to_string value}" }
else
'Ok value
in

let CorrectionContract = fun label value =>
if value.allow_no_correction then
if value.acknowledgment_token != DANGER_TOKEN then
'Error {
message = "DANGER: BH override requires acknowledgment_token = '%{DANGER_TOKEN}'. This will be logged, bannered, and included in DOI bundle. See context-sensitive help for 'correction.method'.",
}
else
'Ok value
else
if value.method != 'BH && value.method != 'FDR && value.method != 'Benjamini-Hochberg then
'Error {
message = "Correction method must be BH in v1 (got %{std.to_string value.method}). BH controls FDR for high-dimensional microbiome data. If you truly want to override, set allow_no_correction=true and acknowledgment_token='%{DANGER_TOKEN}'.",
}
else
'Ok value
in

let ZeroHandlingContract = fun label value =>
if value.zero_handling == 'refuse then
if value.acknowledgment_token != DANGER_TOKEN then
'Error {
message = "DANGER: zero_handling='refuse' will cause log(0) for CLR/ILR and biased handling for NB_GLM. Requires acknowledgment_token='%{DANGER_TOKEN}'. Even then, CLR/ILR + refuse is mathematically invalid and will be refused at runtime.",
}
else
'Ok value
else
'Ok value
in

let MethodNormalizationCompatibility = fun label value =>
let method = value.method in
let norm = value.normalization.method in
if method == 'nb_glm then
if norm == 'clr || norm == 'ilr then
'Error { message = "NB_GLM expects count data, not CLR/ILR transforms. Use clr_lm/ilr_lm for compositional, or change normalization to none/size_factors. Refusing meaningless combination." }
else
'Ok value
else if method == 'clr_lm then
if norm != 'clr then
'Error { message = "CLR_LM requires normalization.method='clr', got '%{std.to_string norm}'" }
else
'Ok value
else if method == 'ilr_lm then
if norm != 'ilr then
'Error { message = "ILR_LM requires normalization.method='ilr', got '%{std.to_string norm}'" }
else
'Ok value
else
'Ok value
in

{
schema_version
| String
| doc "Semver schema version, currently only 1.0.0"
| std.contract.from_predicate (fun v => v == "1.0.0")
= "1.0.0",

id
| String
| doc "UUIDv4, immutable identifier, part of provenance chain",

created_at
| String
| doc "ISO8601 UTC timestamp",

created_by
| String
| doc "User or system that created this config",

method
| AnalysisMethod
| doc m%"
Analysis method — explicit, no auto-selection.

- 'nb_glm: Negative Binomial GLM for raw counts with overdispersion (DESeq2/MASS style)
- 'clr_lm: Centered Log-Ratio + Gaussian LM (compositional, Aitchison geometry)
- 'ilr_lm: Isometric Log-Ratio + Gaussian LM (balances, phylogenetic basis possible)
- 'logistic: Logistic regression for binary outcome (presence/absence)

No silent switching: you must choose one. Changing method changes statistical model and interpretation.
"%m,

formula
| String
| ValidFormula
| doc m%"
R-style formula, e.g. '~ group' or 'disease ~ group + batch'

- Left of ~ is outcome (required for logistic)
- Right of ~ lists metadata columns
- Must reference only columns in metadata_columns
- Forbidden: ; ` $ (injection prevention)
- Must contain at least one covariate

Context: If you have batch effects, include batch: '~ group + batch'. Otherwise p-values may be confounded.
"%m,

outcome_column
| std.option.String
| doc "Binary outcome column for logistic regression. Required for logistic, ignored for others unless formula uses it.",

metadata_columns
| Array String
| std.contract.from_predicate (fun arr => std.array.length arr > 0)
| doc "Explicit list of metadata columns used. Must exist in study metadata. No auto-selection.",

normalization
| {
method
| NormalizationMethod
| doc "Normalization / transform, must be compatible with method",

pseudocount
| Number
| PseudocountContract
| doc "Pseudocount for zero replacement in CLR/ILR. Must be >0. Typical 0.5. Refuses 0 because log(0) undefined.",

ilr_basis
| std.option.String
| doc "ILR basis, only meaningful for ILR method",

multiplicative_replacement_delta
| std.option.Number
| doc "Delta for multiplicative replacement, in (0,1)",
},

correction
| {
method
| [| 'BH, 'FDR, 'Benjamini-Hochberg, 'none, 'bonferroni |]
| doc "BH mandatory in v1. Any override triggers DANGER banner and requires acknowledgment token.",

alpha
| Number
| std.contract.from_predicate (fun a => a > 0 && a < 1)
| doc "FDR threshold, typically 0.05, must be in (0,1)",

allow_no_correction
| Bool
| default = false
| doc "If true, allows non-BH methods, but triggers DANGER banner and requires acknowledgment_token",

acknowledgment_token
| std.option.String
| doc "Must be 'I_UNDERSTAND_THE_RISK_AND_WANT_TO_OVERRIDE_BH' if allow_no_correction=true",
}
| CorrectionContract,

advanced
| {
dispersion_method
| DispersionMethod
| doc "Dispersion estimation for NB_GLM: parametric (DESeq2 default), local, mean, pooled, glmGamPoi",

zero_handling
| ZeroHandling
| doc "Zero handling. 'refuse' is DANGEROUS and requires acknowledgment token.",

min_prevalence
| Number
| PrevalenceContract
| doc "Minimum prevalence filter in [0,1]. 0.1 = present in >=10% samples.",

min_abundance
| Number
| std.contract.from_predicate (fun v => v >= 0)
| doc "Minimum abundance threshold >=0",

max_features
| std.option.Number
| doc "Max features to test. >100k refused as meaningless.",

min_samples_per_group
| Number
| std.contract.from_predicate (fun v => v >= 2)
| doc "Minimum samples per group for variance estimation. <2 refused, <3 triggers DANGER.",

robust
| Bool
| doc "Robust estimation flag",

acknowledgment_token
| std.option.String
| doc "Required for dangerous zero_handling='refuse'",
}
| ZeroHandlingContract,

provenance
| { .. }
| doc "Provenance chain: metamanifold version, host, tools, etc.",

hash
| String
| doc "SHA256 of canonical JSON representation (immutable, content-addressed)",
}
| MethodNormalizationCompatibility
Loading
Loading