From addfc7d29ffc56acd0480441aced8c0490388996 Mon Sep 17 00:00:00 2001 From: hyperpolymath <6759885+hyperpolymath@users.noreply.github.com> Date: Fri, 18 Sep 2026 16:10:30 +0000 Subject: [PATCH 1/2] feat(analysis): safe, explicit, versioned AnalysisConfig layer v1 - AnalysisConfig module with versioned immutable provenance-rich objects - Methods: NB_GLM, CLR_LM, ILR_LM, LOGISTIC with BH mandatory - Hard-stop DANGER banner on overrides with token I_UNDERSTAND_THE_RISK_AND_WANT_TO_OVERRIDE_BH - Advanced options behind AdvancedAnalysis expander with heavy validation - JSON + Nickel + DEED schemas from hyperpolymath/standards - DOI-ready bundles with DataCite - Epistemic bridge (avec_fibre, present_in_every_admissible_world) - Frontend: AnalysisConfigEditor, DangerBanner, AdvancedAnalysisExpander, EvidenceModeToggle - Tests + benchmarks with 10% regression gate --- bench/analysis_config/benchmark.jl | 136 +++ config/schemas/analysis_config.ncl | 243 ++++ config/schemas/analysis_config.schema.json | 251 ++++ config/templates/analysis_config_chora.deed | 70 ++ docs/milestones/00-reconnaissance.md | 153 +++ docs/milestones/01-project-board-graphql.md | 249 ++++ docs/milestones/02-deferred-issues.md | 267 +++++ .../components/AdvancedAnalysisExpander.tsx | 251 ++++ .../src/components/AnalysisConfigEditor.tsx | 342 ++++++ frontend/src/components/CladeCumulus.tsx | 217 ++++ frontend/src/components/DangerBanner.tsx | 90 ++ .../src/components/EvidenceModeToggle.tsx | 81 ++ frontend/src/types/analysis_config.ts | 110 ++ src/MetaManifold.jl | 3 + src/analysis/analysis_config.jl | 1023 +++++++++++++++++ src/analysis/clade_cumulus.jl | 510 ++++++++ src/core/epistemic.jl | 352 ++++++ src/server/routes/analysis_config.jl | 451 ++++++++ src/server/server.jl | 2 + test/runtests.jl | 2 + test/unit/test_analysis_config.jl | 473 ++++++++ 21 files changed, 5276 insertions(+) create mode 100644 bench/analysis_config/benchmark.jl create mode 100644 config/schemas/analysis_config.ncl create mode 100644 config/schemas/analysis_config.schema.json create mode 100644 config/templates/analysis_config_chora.deed create mode 100644 docs/milestones/00-reconnaissance.md create mode 100644 docs/milestones/01-project-board-graphql.md create mode 100644 docs/milestones/02-deferred-issues.md create mode 100644 frontend/src/components/AdvancedAnalysisExpander.tsx create mode 100644 frontend/src/components/AnalysisConfigEditor.tsx create mode 100644 frontend/src/components/CladeCumulus.tsx create mode 100644 frontend/src/components/DangerBanner.tsx create mode 100644 frontend/src/components/EvidenceModeToggle.tsx create mode 100644 frontend/src/types/analysis_config.ts create mode 100644 src/analysis/analysis_config.jl create mode 100644 src/analysis/clade_cumulus.jl create mode 100644 src/core/epistemic.jl create mode 100644 src/server/routes/analysis_config.jl create mode 100644 test/unit/test_analysis_config.jl diff --git a/bench/analysis_config/benchmark.jl b/bench/analysis_config/benchmark.jl new file mode 100644 index 0000000..46440f0 --- /dev/null +++ b/bench/analysis_config/benchmark.jl @@ -0,0 +1,136 @@ +# SPDX-License-Identifier: AGPL-3.0-only +""" + Benchmark for AnalysisConfig layer + +Ensures runtime and memory regression <10% vs baseline. + +Measures: +- Config creation + validation +- JSON serialization roundtrip +- DOI bundle creation +- Epistemic validation (present_in_every_admissible_world) +- CladeCumulus tree building + +Fail CI on >10% regression. +""" + +using BenchmarkTools +using OrderedCollections +using MetaManifold.AnalysisConfig +using MetaManifold.Epistemic +using MetaManifold.CladeCumulus + +const SUITE = BenchmarkGroup() + +SUITE["config_creation"] = @benchmarkable begin + norm = AnalysisConfig.NormalizationConfig(method="size_factors") + cfg = AnalysisConfig.AnalysisConfigStruct( + method="nb_glm", + formula="~ group + batch", + metadata_columns=["group", "batch", "age"], + normalization=norm, + created_by="benchmark" + ) +end + +SUITE["config_validation"] = @benchmarkable begin + norm = AnalysisConfig.NormalizationConfig(method="clr", pseudocount=0.5) + cfg = AnalysisConfig.AnalysisConfigStruct( + method="clr_lm", + formula="~ group", + metadata_columns=["group"], + normalization=norm, + ) + AnalysisConfig.validate_config(cfg, ["group", "batch"]; strict=false) +end + +SUITE["json_roundtrip"] = @benchmarkable begin + norm = AnalysisConfig.NormalizationConfig(method="size_factors") + cfg = AnalysisConfig.AnalysisConfigStruct( + method="nb_glm", + formula="~ group", + metadata_columns=["group"], + normalization=norm, + ) + json = AnalysisConfig.to_json(cfg) + AnalysisConfig.from_json(json) +end + +SUITE["doi_bundle"] = @benchmarkable begin + norm = AnalysisConfig.NormalizationConfig(method="size_factors") + cfg = AnalysisConfig.AnalysisConfigStruct( + method="nb_glm", + formula="~ group", + metadata_columns=["group"], + normalization=norm, + ) + result = AnalysisConfig.AnalysisResult( + config_id=cfg.id, + config_hash=cfg.hash, + method=cfg.method, + results=OrderedDict{String,Any}("taxon1" => OrderedDict("p" => 0.01)) + ) + mktempdir() do tmp + AnalysisConfig.create_doi_bundle(cfg, result, joinpath(tmp, "bundle"); authors=["Bench"], title="Bench") + end +end + +SUITE["epistemic_present_in_every"] = @benchmarkable begin + c1 = Epistemic.Candidate{Tuple{Int,Int},Int}((1,1), true, true) + c2 = Epistemic.Candidate{Tuple{Int,Int},Int}((2,0), true, true) + case_bounded = Epistemic.Case{Tuple{Int,Int},Int}(c1, [c1, c2]) + Epistemic.present_in_every_admissible_world(case_bounded, world -> world[1] != 0) +end + +SUITE["clade_tree_build"] = @benchmarkable begin + rows = [ + Dict{String,Any}("Domain" => "Bacteria", "Phylum" => "Firmicutes", "Genus" => "Lacto$(i)", "total" => Float64(10+i), "avec_fibre" => true, "epistemic_status" => "present_in_every_admissible_world", "residual_count" => i % 5) + for i in 1:100 + ] + tree = CladeCumulus.build_clade_tree(rows) + CladeCumulus.cumulative_frequencies(tree) +end + +# Run and check regression +function run_benchmarks(; baseline_path::String=joinpath(@__DIR__, "baseline.json")) + results = run(SUITE, verbose=true) + + # Save current as new baseline if no baseline exists + if !isfile(baseline_path) + BenchmarkTools.save(baseline_path, results) + println("No baseline found, saved current as baseline at $baseline_path") + return results + end + + baseline = BenchmarkTools.load(baseline_path)[1] + + # Compare, fail on >10% regression in time or memory + for (key, trial) in results + if haskey(baseline, key) + base_trial = baseline[key] + # Median time comparison + curr_time = BenchmarkTools.prettytime(BenchmarkTools.median(trial).time) + base_time = BenchmarkTools.prettytime(BenchmarkTools.median(base_trial).time) + + # Ratio + time_ratio = BenchmarkTools.median(trial).time / BenchmarkTools.median(base_trial).time + mem_ratio = BenchmarkTools.median(trial).memory / max(1, BenchmarkTools.median(base_trial).memory) + + println("$key: time ratio $(round(time_ratio, digits=3)) (baseline $base_time vs current $curr_time), memory ratio $(round(mem_ratio, digits=3))") + + if time_ratio > 1.10 + error("Benchmark regression >10% in time for $key: ratio $time_ratio (threshold 1.10)") + end + if mem_ratio > 1.10 + error("Benchmark regression >10% in memory for $key: ratio $mem_ratio") + end + end + end + + println("All benchmarks within 10% regression threshold — OK") + return results +end + +if abspath(PROGRAM_FILE) == @__FILE__ + run_benchmarks() +end diff --git a/config/schemas/analysis_config.ncl b/config/schemas/analysis_config.ncl new file mode 100644 index 0000000..283e136 --- /dev/null +++ b/config/schemas/analysis_config.ncl @@ -0,0 +1,243 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# AnalysisConfig Nickel contract — hyperpolymath/standards style +# From 1-formats/k9/*.ncl and .machine_readable/contractiles/_base.ncl +# Implements BH mandatory, DANGER banner, advanced validation + +let AnalysisMethod = std.enum.TagOrString & [| 'nb_glm, 'clr_lm, 'ilr_lm, 'logistic |] in +let NormalizationMethod = std.enum.TagOrString & [| 'none, 'rarefy, 'relative, 'size_factors, 'clr, 'ilr, 'presence_absence |] in +let CorrectionMethod = std.enum.TagOrString & [| 'BH, 'FDR, 'Benjamini-Hochberg |] in +let DispersionMethod = std.enum.TagOrString & [| 'parametric, 'local, 'mean, 'pooled, 'glmGamPoi |] in +let ZeroHandling = std.enum.TagOrString & [| 'pseudocount, 'multiplicative_replacement, 'bayesian_multiplicative, 'refuse |] in +let IlrBasis = std.enum.TagOrString & [| 'default, 'phylogenetic, 'sequential_binary_partition, 'balance_dendrogram |] in + +let DANGER_TOKEN = "I_UNDERSTAND_THE_RISK_AND_WANT_TO_OVERRIDE_BH" in + +let ValidFormula = fun label value => + if std.string.contains ";" value then + 'Error { message = "Formula contains forbidden ';' (injection prevention)" } + else if std.string.contains "`" value then + 'Error { message = "Formula contains forbidden backtick" } + else if std.string.contains "$" value then + 'Error { message = "Formula contains forbidden '$'" } + else if !(std.string.contains "~" value) then + 'Error { message = "Formula must contain '~' (R-style), e.g. '~ group'" } + else if std.string.length (std.string.trim value) < 2 then + 'Error { message = "Formula must be non-empty and reference at least one metadata column" } + else + 'Ok value +in + +let PseudocountContract = fun label value => + if value <= 0 then + 'Error { message = "pseudocount must be >0 for CLR/ILR (log(0) undefined). Got %{std.to_string value}" } + else if value >= 1 then + # Warn but allow — typical is 0.5 + 'Ok value + else + 'Ok value +in + +let PrevalenceContract = fun label value => + if value < 0 || value > 1 then + 'Error { message = "min_prevalence must be in [0,1], got %{std.to_string value}" } + else + 'Ok value +in + +let CorrectionContract = fun label value => + if value.allow_no_correction then + if value.acknowledgment_token != DANGER_TOKEN then + 'Error { + message = "DANGER: BH override requires acknowledgment_token = '%{DANGER_TOKEN}'. This will be logged, bannered, and included in DOI bundle. See context-sensitive help for 'correction.method'.", + } + else + 'Ok value + else + if value.method != 'BH && value.method != 'FDR && value.method != 'Benjamini-Hochberg then + 'Error { + message = "Correction method must be BH in v1 (got %{std.to_string value.method}). BH controls FDR for high-dimensional microbiome data. If you truly want to override, set allow_no_correction=true and acknowledgment_token='%{DANGER_TOKEN}'.", + } + else + 'Ok value +in + +let ZeroHandlingContract = fun label value => + if value.zero_handling == 'refuse then + if value.acknowledgment_token != DANGER_TOKEN then + 'Error { + message = "DANGER: zero_handling='refuse' will cause log(0) for CLR/ILR and biased handling for NB_GLM. Requires acknowledgment_token='%{DANGER_TOKEN}'. Even then, CLR/ILR + refuse is mathematically invalid and will be refused at runtime.", + } + else + 'Ok value + else + 'Ok value +in + +let MethodNormalizationCompatibility = fun label value => + let method = value.method in + let norm = value.normalization.method in + if method == 'nb_glm then + if norm == 'clr || norm == 'ilr then + 'Error { message = "NB_GLM expects count data, not CLR/ILR transforms. Use clr_lm/ilr_lm for compositional, or change normalization to none/size_factors. Refusing meaningless combination." } + else + 'Ok value + else if method == 'clr_lm then + if norm != 'clr then + 'Error { message = "CLR_LM requires normalization.method='clr', got '%{std.to_string norm}'" } + else + 'Ok value + else if method == 'ilr_lm then + if norm != 'ilr then + 'Error { message = "ILR_LM requires normalization.method='ilr', got '%{std.to_string norm}'" } + else + 'Ok value + else + 'Ok value +in + +{ + schema_version + | String + | doc "Semver schema version, currently only 1.0.0" + | std.contract.from_predicate (fun v => v == "1.0.0") + = "1.0.0", + + id + | String + | doc "UUIDv4, immutable identifier, part of provenance chain", + + created_at + | String + | doc "ISO8601 UTC timestamp", + + created_by + | String + | doc "User or system that created this config", + + method + | AnalysisMethod + | doc m%" + Analysis method — explicit, no auto-selection. + + - 'nb_glm: Negative Binomial GLM for raw counts with overdispersion (DESeq2/MASS style) + - 'clr_lm: Centered Log-Ratio + Gaussian LM (compositional, Aitchison geometry) + - 'ilr_lm: Isometric Log-Ratio + Gaussian LM (balances, phylogenetic basis possible) + - 'logistic: Logistic regression for binary outcome (presence/absence) + + No silent switching: you must choose one. Changing method changes statistical model and interpretation. + "%m, + + formula + | String + | ValidFormula + | doc m%" + R-style formula, e.g. '~ group' or 'disease ~ group + batch' + + - Left of ~ is outcome (required for logistic) + - Right of ~ lists metadata columns + - Must reference only columns in metadata_columns + - Forbidden: ; ` $ (injection prevention) + - Must contain at least one covariate + + Context: If you have batch effects, include batch: '~ group + batch'. Otherwise p-values may be confounded. + "%m, + + outcome_column + | std.option.String + | doc "Binary outcome column for logistic regression. Required for logistic, ignored for others unless formula uses it.", + + metadata_columns + | Array String + | std.contract.from_predicate (fun arr => std.array.length arr > 0) + | doc "Explicit list of metadata columns used. Must exist in study metadata. No auto-selection.", + + normalization + | { + method + | NormalizationMethod + | doc "Normalization / transform, must be compatible with method", + + pseudocount + | Number + | PseudocountContract + | doc "Pseudocount for zero replacement in CLR/ILR. Must be >0. Typical 0.5. Refuses 0 because log(0) undefined.", + + ilr_basis + | std.option.String + | doc "ILR basis, only meaningful for ILR method", + + multiplicative_replacement_delta + | std.option.Number + | doc "Delta for multiplicative replacement, in (0,1)", + }, + + correction + | { + method + | [| 'BH, 'FDR, 'Benjamini-Hochberg, 'none, 'bonferroni |] + | doc "BH mandatory in v1. Any override triggers DANGER banner and requires acknowledgment token.", + + alpha + | Number + | std.contract.from_predicate (fun a => a > 0 && a < 1) + | doc "FDR threshold, typically 0.05, must be in (0,1)", + + allow_no_correction + | Bool + | default = false + | doc "If true, allows non-BH methods, but triggers DANGER banner and requires acknowledgment_token", + + acknowledgment_token + | std.option.String + | doc "Must be 'I_UNDERSTAND_THE_RISK_AND_WANT_TO_OVERRIDE_BH' if allow_no_correction=true", + } + | CorrectionContract, + + advanced + | { + dispersion_method + | DispersionMethod + | doc "Dispersion estimation for NB_GLM: parametric (DESeq2 default), local, mean, pooled, glmGamPoi", + + zero_handling + | ZeroHandling + | doc "Zero handling. 'refuse' is DANGEROUS and requires acknowledgment token.", + + min_prevalence + | Number + | PrevalenceContract + | doc "Minimum prevalence filter in [0,1]. 0.1 = present in >=10% samples.", + + min_abundance + | Number + | std.contract.from_predicate (fun v => v >= 0) + | doc "Minimum abundance threshold >=0", + + max_features + | std.option.Number + | doc "Max features to test. >100k refused as meaningless.", + + min_samples_per_group + | Number + | std.contract.from_predicate (fun v => v >= 2) + | doc "Minimum samples per group for variance estimation. <2 refused, <3 triggers DANGER.", + + robust + | Bool + | doc "Robust estimation flag", + + acknowledgment_token + | std.option.String + | doc "Required for dangerous zero_handling='refuse'", + } + | ZeroHandlingContract, + + provenance + | { .. } + | doc "Provenance chain: metamanifold version, host, tools, etc.", + + hash + | String + | doc "SHA256 of canonical JSON representation (immutable, content-addressed)", +} +| MethodNormalizationCompatibility diff --git a/config/schemas/analysis_config.schema.json b/config/schemas/analysis_config.schema.json new file mode 100644 index 0000000..6069f65 --- /dev/null +++ b/config/schemas/analysis_config.schema.json @@ -0,0 +1,251 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "https://hyperpolymath.github.io/MetaManifold-WebUI/schemas/analysis_config.schema.json", + "title": "AnalysisConfig — versioned, explicit, provenance-rich analysis configuration", + "description": "Safe, explicit, versioned AnalysisConfig layer for parametric and nonparametric analyses (NB GLM, CLR/ILR+Gaussian LM, logistic in v1; BH mandatory; DANGER banner on overrides; DOI-ready bundles). From hyperpolymath/standards JSON + Nickel + DEED schemes.", + "type": "object", + "required": ["schema_version", "id", "method", "formula", "metadata_columns", "normalization", "correction"], + "properties": { + "schema_version": { + "type": "string", + "pattern": "^[0-9]+\\.[0-9]+\\.[0-9]+$", + "enum": ["1.0.0"], + "description": "Semver schema version. Currently only 1.0.0 is supported." + }, + "id": { + "type": "string", + "format": "uuid", + "description": "Unique immutable identifier for this config (UUIDv4). Part of provenance chain." + }, + "created_at": { + "type": "string", + "format": "date-time", + "description": "ISO8601 UTC timestamp of creation" + }, + "created_by": { + "type": "string", + "minLength": 1, + "description": "User or system that created this config (for provenance)" + }, + "method": { + "type": "string", + "enum": ["nb_glm", "clr_lm", "ilr_lm", "logistic"], + "description": "Analysis method — explicit, no silent switching. NB_GLM for counts, CLR/ILR+LM for compositional, logistic for presence/absence." + }, + "formula": { + "type": "string", + "minLength": 2, + "pattern": "^[^;`$]+$", + "description": "R-style formula containing '~', e.g. '~ group' or 'disease ~ group + batch'. Must reference only metadata_columns. Refuses empty or meaningless formulas." + }, + "outcome_column": { + "type": ["string", "null"], + "description": "Binary outcome column for logistic regression. Required for logistic, ignored for others unless formula uses it." + }, + "metadata_columns": { + "type": "array", + "minItems": 1, + "uniqueItems": true, + "items": { + "type": "string", + "minLength": 1, + "pattern": "^[a-zA-Z0-9_\\.\\-]+$" + }, + "description": "Explicit list of metadata columns used. Must exist in study metadata. No auto-selection." + }, + "normalization": { + "type": "object", + "required": ["method"], + "properties": { + "method": { + "type": "string", + "enum": ["none", "rarefy", "relative", "size_factors", "clr", "ilr", "presence_absence"], + "description": "Normalization / transform. Must be compatible with method: nb_glm allows none/rarefy/size_factors/relative, clr_lm requires clr, ilr_lm requires ilr, logistic allows none/relative/rarefy/presence_absence." + }, + "pseudocount": { + "type": "number", + "exclusiveMinimum": 0, + "description": "Pseudocount for zero replacement in CLR/ILR. Must be >0. Typical 0.5. Refuses 0 because log(0) undefined." + }, + "ilr_basis": { + "type": ["string", "null"], + "enum": ["default", "phylogenetic", "sequential_binary_partition", "balance_dendrogram", null], + "description": "ILR basis. Only meaningful for ilr method. Refuses meaningless use for other methods." + }, + "multiplicative_replacement_delta": { + "type": ["number", "null"], + "exclusiveMinimum": 0, + "exclusiveMaximum": 1, + "description": "Delta for multiplicative replacement, in (0,1)" + } + }, + "allOf": [ + { + "if": { "properties": { "method": { "const": "clr" } } }, + "then": { "required": ["pseudocount"] } + }, + { + "if": { "properties": { "method": { "const": "ilr" } } }, + "then": { "required": ["pseudocount"] } + } + ] + }, + "correction": { + "type": "object", + "required": ["method", "alpha"], + "properties": { + "method": { + "type": "string", + "description": "BH mandatory in v1. Any override triggers DANGER banner and requires acknowledgment token." + }, + "alpha": { + "type": "number", + "exclusiveMinimum": 0, + "exclusiveMaximum": 1, + "description": "FDR threshold, typically 0.05" + }, + "allow_no_correction": { + "type": "boolean", + "default": false, + "description": "If true, allows non-BH methods, but triggers DANGER banner and requires acknowledgment_token" + }, + "acknowledgment_token": { + "type": ["string", "null"], + "description": "Must be 'I_UNDERSTAND_THE_RISK_AND_WANT_TO_OVERRIDE_BH' if allow_no_correction=true" + } + }, + "allOf": [ + { + "if": { + "properties": { "allow_no_correction": { "const": true } } + }, + "then": { + "properties": { + "acknowledgment_token": { "const": "I_UNDERSTAND_THE_RISK_AND_WANT_TO_OVERRIDE_BH" } + }, + "required": ["acknowledgment_token"] + } + }, + { + "if": { + "properties": { "allow_no_correction": { "const": false } } + }, + "then": { + "properties": { + "method": { "enum": ["BH", "FDR", "Benjamini-Hochberg", "benjamini-hochberg"] } + } + } + } + ] + }, + "advanced": { + "type": "object", + "description": "All advanced options behind 'Advanced Analysis' expander, hidden unless Evidence Mode enabled. Heavy validation.", + "properties": { + "dispersion_method": { + "type": "string", + "enum": ["parametric", "local", "mean", "pooled", "glmGamPoi"], + "description": "Dispersion estimation for NB_GLM" + }, + "zero_handling": { + "type": "string", + "enum": ["pseudocount", "multiplicative_replacement", "bayesian_multiplicative", "refuse"], + "description": "Zero handling. 'refuse' is DANGEROUS and requires acknowledgment token." + }, + "min_prevalence": { + "type": "number", + "minimum": 0, + "maximum": 1, + "description": "Minimum prevalence filter in [0,1]. 0.1 = present in >=10% samples." + }, + "min_abundance": { + "type": "number", + "minimum": 0, + "description": "Minimum abundance threshold >=0" + }, + "max_features": { + "type": ["integer", "null"], + "minimum": 1, + "maximum": 100000, + "description": "Max features to test. >100k refused as meaningless." + }, + "min_samples_per_group": { + "type": "integer", + "minimum": 2, + "description": "Minimum samples per group for variance estimation. <2 refused, <3 triggers DANGER." + }, + "robust": { + "type": "boolean", + "description": "Robust estimation flag" + }, + "acknowledgment_token": { + "type": ["string", "null"], + "description": "Required for dangerous zero_handling='refuse'" + } + } + }, + "provenance": { + "type": "object", + "description": "Provenance chain: metamanifold version, host, tools, etc. Enriched automatically." + }, + "hash": { + "type": "string", + "pattern": "^[a-f0-9]{64}$", + "description": "SHA256 of canonical JSON representation (immutable, content-addressed)" + } + }, + "allOf": [ + { + "if": { + "properties": { "method": { "const": "logistic" } } + }, + "then": { + "required": ["outcome_column"], + "properties": { + "outcome_column": { "type": "string", "minLength": 1 } + } + } + }, + { + "if": { + "properties": { "method": { "const": "clr_lm" } } + }, + "then": { + "properties": { + "normalization": { + "properties": { "method": { "const": "clr" } } + } + } + } + }, + { + "if": { + "properties": { "method": { "const": "ilr_lm" } } + }, + "then": { + "properties": { + "normalization": { + "properties": { "method": { "const": "ilr" } } + } + } + } + } + ], + "$defs": { + "epistemic": { + "description": "Epistemic layer integration: avec_fibre column and present_in_every_admissible_world", + "type": "object", + "properties": { + "avec_fibre": { + "type": "boolean", + "description": "True if artefact carries enough semantic fibre to support inferences (Echo Types)" + }, + "epistemic_status": { + "type": "string", + "enum": ["present_in_every_admissible_world", "present_in_some_admissible_world", "absent_in_every_admissible_world", "unknown"], + "description": "Residual evidence status: Holds Present across all admissible worlds?" + } + } + } + } +} diff --git a/config/templates/analysis_config_chora.deed b/config/templates/analysis_config_chora.deed new file mode 100644 index 0000000..cf7c066 --- /dev/null +++ b/config/templates/analysis_config_chora.deed @@ -0,0 +1,70 @@ +;; SPDX-FileCopyrightText: © 2026 Jonathan D.A. Jewell (hyperpolymath) +;; SPDX-License-Identifier: AGPL-3.0-only +;; AnalysisConfig DEED template — from hyperpolymath/standards 1-formats/deed +;; Filename dispatch: *_chora.deed → repo-deed, :schema-version first +;; This file is a template; actual configs are generated via AnalysisConfig.to_deed() +;; See DEED-GRAMMAR-SPEC.adoc v0.2.0: :schema-version structurally first, only () brackets, #t/#f booleans, :kebab-case keywords + +(repo-deed + :schema-version "1.0.0" + :canonical-name "analysis-config-template" + :beholding-chora #u5"estate/chora" + + (method + :name "nb_glm" + :formula "~ group" + :outcome-column "" + :metadata-columns ("group" "batch")) + + (normalization + :method "size_factors" + :pseudocount 0.5 + :ilr-basis "") + + (correction + :method "BH" + :alpha 0.05 + :allow-no-correction #f) + + (advanced + :dispersion-method "parametric" + :zero-handling "pseudocount" + :min-prevalence 0.1 + :min-abundance 0.0 + :max-features 0 + :min-samples-per-group 3 + :robust #f) + + (provenance + :id "00000000-0000-0000-0000-000000000000" + :hash "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855" + :created-at "2026-09-18T00:00:00Z" + :created-by "template" + :dangerous #f) + + (warrant + :evidence-type "AnalysisConfig" + :soundness "BH mandatory, requires acknowledgment token for override" + :fiber "Echo of raw counts through size_factors transform" + :epistemic-status "present_in_every_admissible_world") + + (doi-bundle + :title "MetaManifold Analysis Bundle" + :license "CC-BY-4.0" + :authors ("Anonymous") + :description "Differential abundance analysis with nb_glm, BH correction, DOI-ready") + + (context-help + :method "Analysis method — explicit, no auto-selection. See JSON schema for scientific context." + :formula "R-style formula, e.g. '~ group' or 'disease ~ group + batch'. Must reference only metadata_columns." + :correction "BH mandatory in v1. Any override triggers DANGER banner and requires acknowledgment token I_UNDERSTAND_THE_RISK_AND_WANT_TO_OVERRIDE_BH." + :normalization "Normalization must be compatible with method: nb_glm allows none/rarefy/size_factors/relative, clr_lm requires clr, ilr_lm requires ilr." + :advanced "All advanced options behind Advanced Analysis expander, hidden unless Evidence Mode enabled. Heavy validation, refusal of meaningless inputs.")) + +;; SPDX-License-Identifier: AGPL-3.0-only +;; End of template — generated files will have same structure with filled values +;; DANGER banner example (when allow-no-correction #t): +;; ;; ╔════════════════════════════════════════════════════════════════════════════╗ +;; ;; ║ ⚠️ DANGER — SCIENTIFICALLY RISKY CONFIGURATION DETECTED ⚠️ ║ +;; ;; ║ BH correction disabled — will inflate false discoveries ║ +;; ;; ╚════════════════════════════════════════════════════════════════════════════╝ diff --git a/docs/milestones/00-reconnaissance.md b/docs/milestones/00-reconnaissance.md new file mode 100644 index 0000000..2777f59 --- /dev/null +++ b/docs/milestones/00-reconnaissance.md @@ -0,0 +1,153 @@ +# Milestone 0 — Full Reconnaissance (2026-09-18 Europe/London) + +## Repository State + +- **Origin**: `https://github.com/hyperpolymath/MetaManifold-WebUI.git` +- **Fork of**: `JoshuaJewell/MetaManifold-WebUI` +- **Current branch**: `main` @ `10915ef chore: add SPDX headers to post-rebase ui/migration files` +- **Remote branches**: only `main` (no feature branches yet) +- **Workspace**: empty except cloned `hyperpolymath-MetaManifold` under `/home/user` + +### Codebase Layout (verified) + +``` +src/ + MetaManifold.jl (module entry, includes core/*, annotation/*, pipeline/*, analysis/*) + core/ + types.jl, r_runtime.jl, provenance.jl, log.jl, config.jl, databases.jl, + duckdb_store.jl, validate.jl, project.jl, categories.jl, + composition_library.jl, primers_library.jl, databases_library.jl + analysis/ + analysis.jl (42k lines, Plotly chart builders, alpha, bar, Venn, NMDS, PERMANOVA) + diversity.jl (richness, shannon, simpson, normalise_counts) + annotation/ + funcdb.jl + pipeline/ + tools.jl, merge_taxa.jl, dada2/*, swarm.jl + server/ + server.jl, jobs.jl, routes/* (analysis, annotations, composition, config, databases, duckdb_helpers, events, jobs, pipeline, results, runs, studies) + +frontend/src/ + App.tsx, main.tsx + api/client.ts (typed REST client) + components/ (AnalysisControls, AnnotationPanel, CompositionPanel, DataTable, PipelineStages, etc.) + views/ (RunView 40k, StudyView, GroupView, etc.) + types/domain, state + +config/defaults/ + pipeline.yml, tool_versions.yml (pinned tool archives + sha256), primers.yml, databases.yml, composition.yml, filters/, etc. + +test/ + unit/* (test_analysis, test_categories, test_provenance, test_routes 45k, etc.) + integration/ + fixtures/ +bench/layer1_mock_recovery/ +``` + +### Epistemic Layer Claim vs Reality + +**Task states**: "The epistemic layer (Echo + Epistemic + Residual Evidence with `avec_fibre` column and `Epistemic.jl`) is already implemented." + +**Actual grep** (`grep -R "Epistemic|avec_fibre|present_in_every_admissible_world|Echo" --include="*.jl" --include="*.tsx"`): +- **Zero hits** in Julia or TS sources. +- No `Epistemic.jl` file exists. +- No `avec_fibre` column handling in DuckDB schema. +- No `Evidence Mode` toggle in frontend. + +**Conclusion**: Either the epistemic layer lives on a non-pushed branch, or the task description is aspirational. To avoid overwriting colleagues' work, we will: + +1. Create new modules without touching `categories.jl`, `composition_library.jl`, or `analysis.jl` internals until the epistemic branch is located. +2. Design `AnalysisConfig` and `CladeCumulus` to be **additive**, importing epistemic types if they appear, but not requiring them at compile time (feature-flagged). +3. Log this gap as a risk in the Project board. + +### Analysis Layer — Current State + +- **Alpha diversity**: richness, Shannon, Simpson via `DiversityMetrics` +- **Charts**: bar (rank / category), alpha boxplot, NMDS (R/vegan), PERMANOVA (R/vegan), Venn/Euler/UpSet +- **Normalization**: none / rarefaction / relative sum scaling, auto depth via `auto_min_depth` +- **Filtering**: via `Categories.filter_to_sql_conditions` (pattern/regex, min/max/include, remove_empty) with SQL injection hardening +- **No parametric modelling**: No NB GLM, no CLR/ILR+Gaussian LM, no logistic regression, no BH correction enforcement, no AnalysisConfig object. + +### Provenance Layer — Relevant for DOI bundles + +- `core/provenance.jl` (36k) already implements: + - `ToolRecord`, `JuliaRecord`, `RRecord`, `DatabaseRecord` with SHA256 hashing + - `probe_tool`, `probe_julia`, `probe_r`, `probe_database` with strict release agreement checks + - `probe_host`, `probe_metamanifold` with dirty-tree detection + - Attestation writing (ordered dict, evidence-rich) +- This can be reused for DOI-ready bundles: we need to add `AnalysisConfig` + `AnalysisResult` attestation and bundle export. + +### Frontend Cleanliness + +- `RunView.tsx` (40k) is the heavy orchestrator: config accordion, pipeline stages, QC, results explorer, annotation, composition. +- `AnalysisControls.tsx` is minimal (1.2k) — placeholder for future controls. +- No `Evidence Mode` toggle yet. Requirement: advanced options and cladistic visuals only appear when Evidence Mode enabled. + +### Standards — JSON + Nickel + DEED + +- **Standards repo cloned** to `/tmp/standards` (depth 1) +- **DEED spec**: `1-formats/deed/spec/DEED-GRAMMAR-SPEC.adoc` v0.2.0 DRAFT, normative ABNF at `abnf/deed.anbf` +- **Key DEED rules**: + - Single extension `.deed`, dispatch by stem (`estate_chora.deed` exact, `*_chora.deed`, `ATLAS.deed`, `*_praxis.deed`) + - Header `;; SPDX-*` required, `:schema-version` structurally first + - No `key = value`, no `[section]`, only `(` `)`, booleans `#t/#f`, keywords `:kebab-case` + - Identity maps to `BaseRecord` (Idris2 typed core) — anti-desync primitive + - `lax`/`strict`/`attested` modes, `strict` default for `.deed` +- **JSON schemas**: `1-formats/a2ml/*/spec/schema/*.schema.json` (draft 2020-12), used for STATE, META, etc. +- **Nickel schemas**: `1-formats/k9/*.ncl`, `.machine_readable/contractiles/**/*.ncl` (runner base contracts, trust/must/intend etc.) +- **Implication for AnalysisConfig**: + - JSON schema: draft 2020-12, with `$id`, `$defs`, required fields, `anyOf` for method union + - Nickel: contract with ` | { field | Type, ... }`, plus validators for BH mandatory, dangerous overrides + - DEED: `(repo-deed :schema-version "1.0.0" :canonical-name "analysis-config" ... (method ...) (correction ...) (provenance ...))` — must follow filename dispatch `analysis_config_chora.deed` or similar. + +### CI & Quality Gates + +- `ci.yml`: repo-hygiene (SPDX, format, lint, commit convention) + Julia test matrix (1.12.5 pinned) + R apt pin + bun 1.3.10 + bench checksum +- `codecov.yml` exists, token needed for upload +- `bench/baseline.json` referenced but not yet present? Need to check `bench/` +- `Justfile` is task runner, 15k lines, recipes: `bootstrap`, `ci`, `drift`, `sync-pins`, `setup-full`, `start` +- Tests: 577 pass / 5 todo per ROADMAP, coverage reported but not gated. + +### Missing Secrets / Blockers + +- **GitHub PAT**: required with `repo` + `workflow` + `project` scopes to: + - Create & maintain Project board "Analysis Layer & Cladistics Development" via GraphQL + - Link issues/PRs to board, update status, remove completed + - Push feature branches, open PRs +- **CODECOV_TOKEN**: for coverage upload (from `codecov.yml`) +- **Cache keys**: Julia depot, bun, R renv — CI uses `julia-actions/cache@v2`, but local cache invalidation keys unknown +- **No existing GitHub Project**: verified via `gh` CLI not available, and no `project` data in repo. Must be created via GraphQL `createProjectV2`. + +### Risk Register (for issue bodies) + +1. **Epistemic layer absent** — if it lands on main while we work, merge conflicts. Mitigation: additive modules, feature-flagged imports. +2. **Exact stats layer deferred** — must not implement symbolic engine / exact tests here; keep AnalysisConfig v1 limited to NB GLM, CLR/ILR+LM, logistic. +3. **UI cleanliness** — Advanced Analysis expander must be hidden behind Evidence Mode; otherwise violates rule. +4. **Benchmark regression gate** — need baseline memory/runtime measurement for new analysis methods (R-backed NB GLM may be heavy). +5. **DOI bundles** — need `DataCite` metadata, license, authorship from `CITATION.cff`. + +### Next Steps (Pending Tokens) + +1. **Create feature branch** `feat/analysis-config-v1` (never force-push main) +2. **Implement AnalysisConfig layer**: + - Julia: `src/analysis/analysis_config.jl` — immutable struct, versioned, provenance-rich, with JSON/Nickel/DEED serialization, BH mandatory, DANGER banner logic, validation refusing meaningless inputs + - Tests: `test/unit/test_analysis_config.jl` + benchmarks `bench/analysis_config/` + - Frontend: `EvidenceModeToggle`, `AdvancedAnalysisExpander`, `AnalysisConfigEditor`, DANGER banner component, context-sensitive help + - Schemas: `config/schemas/analysis_config.schema.json`, `config/schemas/analysis_config.ncl`, `analysis_config_chora.deed` template + - DOI bundle: `src/core/doi_bundle.jl` or extension to provenance +3. **Implement CladeCumulus**: + - Backend: cumulative frequencies over cladistic tree, epistemic colour coding (needs epistemic types or fallback), cloud sizing by residual count, `present_in_every_admissible_world` validation + - Frontend: `CladeCumulus.tsx` — tree with D3/Plotly, drag-and-drop, live validation, Evidence Mode toggle +4. **Project board via GraphQL**: create board, columns (Backlog, In Progress, Review, Done), link issues/PRs, auto-update on PR events +5. **Generate deferred feature issue bodies**: exact stats, symbolic engine, etc. + +### Ready-to-Paste GraphQL Mutations (Prepared) + +Will be executed once PAT provided — see `docs/milestones/01-project-board-graphql.md` for full mutation set. + +## Compliance + +- No code changes made in this milestone (recon only) +- No overwrite of colleagues' work +- Branching strategy respected +- UI cleanliness rule noted for future work diff --git a/docs/milestones/01-project-board-graphql.md b/docs/milestones/01-project-board-graphql.md new file mode 100644 index 0000000..341e659 --- /dev/null +++ b/docs/milestones/01-project-board-graphql.md @@ -0,0 +1,249 @@ +# Milestone 1 — GitHub Project Board via GraphQL + +## Board Name +**Analysis Layer & Cladistics Development** + +## Purpose +Track implementation of AnalysisConfig layer (v1: NB GLM, CLR/ILR+LM, logistic, BH mandatory, DANGER banner) and CladeCumulus (cumulative cladistic explorer with epistemic colour coding, cloud sizing, drag-and-drop with present_in_every_admissible_world validation). + +## Required Token +GitHub PAT with scopes: `repo`, `workflow`, `project` (and `org:write` if under hyperpolymath org). + +If PAT not available, mutations below can be run manually via `gh api graphql` or GitHub CLI. + +## GraphQL Mutations — Ready to Paste + +### 1. Create ProjectV2 (org-level or user-level) + +For org `hyperpolymath`: + +```graphql +mutation CreateProject { + createProjectV2(input: { + ownerId: "O_kgDO..." # org node ID, get via query below + title: "Analysis Layer & Cladistics Development" + }) { + projectV2 { + id + title + url + } + } +} +``` + +Get ownerId: + +```graphql +query GetOrgId { + organization(login: "hyperpolymath") { + id + login + } +} +``` + +For user-level (if org not available): + +```graphql +query GetUserId { + viewer { + id + login + } +} +``` + +Then create with ownerId = viewer id. + +### 2. Create Custom Fields + +```graphql +mutation CreateFields($projectId: ID!) { + # Status field (single select) + createStatus: createProjectV2Field(input: { + projectId: $projectId + dataType: SINGLE_SELECT + name: "Status" + singleSelectOptions: [ + { name: "Backlog", color: GRAY, description: "Not started, ready for pickup" }, + { name: "In Progress", color: YELLOW, description: "Actively being worked" }, + { name: "Review", color: PURPLE, description: "PR open, awaiting review" }, + { name: "Done", color: GREEN, description: "Completed and closed" }, + { name: "Blocked", color: RED, description: "Blocked by epistemic layer or other dependency" } + ] + }) { + projectV2Field { id name } + } + + # Method field + createMethod: createProjectV2Field(input: { + projectId: $projectId + dataType: SINGLE_SELECT + name: "Method" + singleSelectOptions: [ + { name: "NB_GLM", color: BLUE, description: "Negative Binomial GLM" }, + { name: "CLR_LM", color: GREEN, description: "CLR + Gaussian LM" }, + { name: "ILR_LM", color: YELLOW, description: "ILR + Gaussian LM" }, + { name: "LOGISTIC", color: ORANGE, description: "Logistic regression" }, + { name: "CladeCumulus", color: PURPLE, description: "Cumulative Cladistic Explorer" }, + { name: "Epistemic", color: PINK, description: "Echo + Epistemic + Residual Evidence" }, + { name: "Infra", color: GRAY, description: "Project board, schemas, DOI bundles, CI" } + ] + }) { + projectV2Field { id name } + } + + # Risk field + createRisk: createProjectV2Field(input: { + projectId: $projectId + dataType: SINGLE_SELECT + name: "Risk" + singleSelectOptions: [ + { name: "Low", color: GREEN }, + { name: "Medium", color: YELLOW }, + { name: "High", color: RED }, + { name: "Scientific", color: ORANGE, description: "Risk of false discoveries, DANGER banner needed" } + ] + }) { + projectV2Field { id name } + } +} +``` + +### 3. Create Issues (linked to board) + +We will create issues via REST then add to board via GraphQL. + +Issue list (see 02-deferred-issues.md for full bodies): + +1. **feat(analysis): AnalysisConfig v1 — NB GLM, CLR/ILR+LM, logistic, BH mandatory** +2. **feat(analysis): Advanced Analysis expander with heavy validation and context help** +3. **feat(analysis): DANGER banner and hard-stop on BH override** +4. **feat(schemas): JSON + Nickel + DEED schemes from hyperpolymath/standards** +5. **feat(doi): DOI-ready bundles with DataCite, provenance, content-addressed hash** +6. **feat(epistemic): Bridge to echo-types, epistemic-types, residual-evidence-types (avec_fibre, present_in_every_admissible_world)** +7. **feat(cladistics): CladeCumulus — cumulative cladistic explorer tree** +8. **feat(cladistics): Epistemic colour coding and cloud sizing by residual count** +9. **feat(cladistics): Drag-and-drop with live present_in_every_admissible_world validation** +10. **feat(ui): Evidence Mode toggle, clean non-cluttered UI** +11. **test: Coverage for AnalysisConfig and CladeCumulus, benchmark regression gate (<10%)** +12. **deferred: Exact statistics layer (exact tests, symbolic engine) — DO NOT IMPLEMENT HERE** +13. **deferred: Symbolic engine for formula manipulation** + +Each issue should be linked to board via: + +```graphql +mutation AddIssueToBoard($projectId: ID!, $contentId: ID!) { + addProjectV2ItemById(input: { projectId: $projectId, contentId: $contentId }) { + item { id } + } +} +``` + +Where contentId is the issue's node ID (get via `gh api repos/hyperpolymath/MetaManifold-WebUI/issues/123 --jq .node_id`). + +### 4. Update Board Status on Every PR + +In PR template, add: + +```graphql +mutation UpdateItemStatus($projectId: ID!, $itemId: ID!, $fieldId: ID!, $optionId: ID!) { + updateProjectV2ItemFieldValue(input: { + projectId: $projectId + itemId: $itemId + fieldId: $fieldId + value: { singleSelectOptionId: $optionId } + }) { + projectV2Item { id } + } +} +``` + +Get optionId via: + +```graphql +query GetFieldOptions($projectId: ID!) { + node(id: $projectId) { + ... on ProjectV2 { + fields(first: 20) { + nodes { + ... on ProjectV2SingleSelectField { + id + name + options { id name } + } + } + } + } + } +} +``` + +### 5. Remove Completed Issues When Closed + +On issue close, via webhook or manual: + +```graphql +mutation DeleteItem($projectId: ID!, $itemId: ID!) { + deleteProjectV2Item(input: { projectId: $projectId, itemId: $itemId }) { + deletedItemId + } +} +``` + +Or archive: + +```graphql +mutation ArchiveItem($projectId: ID!, $itemId: ID!) { + archiveProjectV2Item(input: { projectId: $projectId, itemId: $itemId }) { + item { id } + } +} +``` + +## Automation via GitHub Actions + +Create `.github/workflows/project-board.yml`: + +```yaml +name: Project Board Automation +on: + issues: + types: [opened, closed, reopened] + pull_request: + types: [opened, closed, reopened, synchronize] + +jobs: + update-board: + runs-on: ubuntu-latest + steps: + - uses: actions/add-to-project@v0.5.0 + with: + project-url: https://github.com/orgs/hyperpolymath/projects/XX + github-token: ${{ secrets.PROJECT_PAT }} # PAT with project scope +``` + +## Current Status (2026-09-18) + +- [ ] Board created (pending PAT) +- [ ] Fields created (Status, Method, Risk) +- [ ] Issues created and linked (see 02-deferred-issues.md) +- [ ] PRs linked (feat/analysis-config-v1, feat/clade-cumulus) +- [ ] Automation workflow added + +## Local-Only Mode (No PAT) + +If PAT not provided, we proceed with: + +1. Code on feature branches locally +2. Generate issue bodies as markdown files in `docs/issues/` +3. Prepare GraphQL mutations as above for manual execution when PAT available +4. Milestone reports in `docs/milestones/` + +## Security + +- PAT must have `repo`, `workflow`, `project` scopes +- Store as `PROJECT_PAT` secret in repo settings, not in code +- Never log PAT in CI output +- Rotate after use if exposed diff --git a/docs/milestones/02-deferred-issues.md b/docs/milestones/02-deferred-issues.md new file mode 100644 index 0000000..dffc2a4 --- /dev/null +++ b/docs/milestones/02-deferred-issues.md @@ -0,0 +1,267 @@ +# Deferred Features — Ready-to-Paste GitHub Issue Bodies + +These are for features that must NOT be implemented in the current task (exact statistics layer and symbolic engine), +but have clear scientific value and should be tracked. + +Each issue includes scientific value, difficulty, risks. + +--- + +## Issue 1: Exact Statistics Layer — Exact Tests for Small Samples + +**Title:** `feat(analysis): Exact statistics layer — Fisher's exact, exact NB, permutation tests for small-n microbiome data` + +**Labels:** `enhancement`, `analysis`, `deferred`, `scientific-value:high`, `difficulty:hard` + +**Body:** + +### Scientific Value +Parametric approximations (NB GLM, Gaussian LM) break down with small sample sizes (n<10 per group) or sparse features. Exact tests (Fisher's exact for presence/absence, exact NB test via `exactTest` in edgeR, permutation-based PERMANOVA with exact p-values) provide valid inference when asymptotics fail. Critical for rare biosphere, low-biomass samples, and clinical cohorts with limited n. + +- **Use case:** Pathogen detection where presence in 2/3 cases vs 0/20 controls should be significant even with tiny n. +- **Impact:** Reduces false negatives in small studies, improves reproducibility for low-abundance taxa. + +### Scope (Deferred — DO NOT IMPLEMENT HERE) +- Exact Fisher's test for 2x2 tables (presence/absence vs group) +- Exact negative binomial test (edgeR `exactTest` or `exactTestDoubleTail`) +- Permutation-based exact p-values for NB GLM (parametric bootstrap) +- Exact CLR/ILR via permutation of Aitchison distances +- Integration with AnalysisConfig: new method `exact_fisher`, `exact_nb`, `permutation` + +### Difficulty +**Hard** — requires: +- R integration (exact tests via `stats::fisher.test`, `edgeR`, `permute`) +- Combinatorial explosion for large tables (need network algorithm for Fisher) +- Performance: exact tests O(n!) naive, need optimized implementations +- Memory: permutation stores B=10k resamples per taxon +- Validation: compare against known exact p-values from R + +### Risks +- **Performance regression:** Exact tests 10-100x slower than asymptotic; must be behind Advanced Analysis expander and Evidence Mode, with warning about runtime. +- **Memory:** Permutation matrices for 10k taxa x 10k permutations = 100M entries ~ 800MB +- **Scientific misuse:** Exact does not mean assumption-free; still assumes exchangeability under null. Need context help explaining when exact is appropriate vs when it is overly conservative. +- **Dependency:** Adds `edgeR`, `BiocParallel` to renv.lock — increases install time and fragility + +### Acceptance Criteria +- [ ] New methods in AnalysisConfig v2: `exact_fisher`, `exact_nb`, `permutation_nb` +- [ ] BH still mandatory, DANGER banner if disabled +- [ ] Benchmark: <10% regression for existing methods, new methods benchmarked separately with 100 taxa, n=5 per group +- [ ] Tests: exact p-values match R's `fisher.test` for 10 known tables +- [ ] Docs: context help explains exact vs asymptotic, when to use +- [ ] UI: behind Advanced Analysis expander, shows estimated runtime + +### Related +- Blocked by: AnalysisConfig v1 (this PR) +- Blocks: Symbolic engine (needs exact p-value expressions) + +--- + +## Issue 2: Symbolic Engine — Formula Manipulation and Provenance + +**Title:** `feat(analysis): Symbolic engine for formula manipulation, contrast derivation, and provenance` + +**Labels:** `enhancement`, `analysis`, `deferred`, `scientific-value:high`, `difficulty:very-hard` + +**Body:** + +### Scientific Value +Current AnalysisConfig stores formula as string (`~ group + batch`). Symbolic engine would: +- Parse formula into AST, validate variables exist, derive contrasts (e.g., `groupB - groupA`) +- Automatically generate all pairwise contrasts for multi-level factors +- Prove that BH correction is applied to the correct family (e.g., all taxa, or per-contrast?) +- Generate human-readable report: "Testing 1500 taxa for effect of group, controlling for batch, with BH FDR 0.05" +- Enable DOI bundle to include symbolic derivation of analysis intent + +- **Use case:** Study with 4 groups (control, diseaseA, diseaseB, diseaseC) — should automatically test all 6 pairwise, with BH across 1500*6=9000 tests, not 1500. +- **Impact:** Prevents p-hacking by making analysis intent explicit and auditable. + +### Scope (Deferred — DO NOT IMPLEMENT HERE) +- Formula parser: R-style `~` and `+`, `*`, `:`, `I()`, `(1|batch)` for random effects (future) +- Contrast derivation: for factor with k levels, generate k-1 or k*(k-1)/2 contrasts +- Provenance: symbolic proof that result hash chains to formula AST hash +- Integration with DEED: `(formula (ast ...) (contrasts ...))` +- Nickel contract: formula as structured data, not string + +### Difficulty +**Very Hard** — requires: +- Parser combinators or RCall to `terms.formula` +- Symbolic algebra (like `Symbolics.jl` or custom) +- Handling of R's non-standard evaluation (NSE) for formulas +- Provenance: need to hash AST, not just string (string `~ group + batch` vs `~ batch + group` are same model but different strings — should hash to same? Or not? Decision needed) +- UI: visual formula editor with drag-and-drop of metadata columns, live validation + +### Risks +- **Complexity:** Symbolic engine is a research project itself; may introduce bugs in contrast derivation leading to wrong scientific conclusions (worst risk) +- **Scope creep:** Random effects `(1|batch)` opens mixed models, which is a whole new layer (lme4, glmmTMB) — must be explicitly out of scope for v1 +- **Performance:** Parsing 1500 formulas (one per taxon) x 6 contrasts = 9000 parses — need caching +- **R dependency:** If using R's `terms`, need R runtime lock, which may deadlock with pipeline stages +- **Security:** Formula injection if AST allows arbitrary R code (e.g., `~ group + system('rm -rf /')`) — must whitelist allowed symbols + +### Acceptance Criteria +- [ ] Formula AST type in Julia, with `from_string` and `to_string` roundtrip +- [ ] Contrast derivation for factors, with BH family size correctly computed +- [ ] Provenance: config hash includes AST hash, not just string hash +- [ ] Tests: 20 formulas parsed, contrasts derived, compared to R's `model.matrix` and `contrasts` +- [ ] Security: refuses formulas containing `system`, `eval`, `parse`, etc. +- [ ] Docs: explains symbolic vs string, why AST matters for provenance +- [ ] UI: formula editor with autocomplete of metadata columns, shows derived contrasts + +### Related +- Blocked by: AnalysisConfig v1, exact stats layer (needs exact p-value symbols) +- Blocks: Automated report generation, DOI bundle v2 + +--- + +## Issue 3: Compositional Data Analysis — Advanced Methods (ANCOM-BC, ALDEx2, Songbird) + +**Title:** `feat(analysis): Advanced compositional methods — ANCOM-BC, ALDEx2, Songbird` + +**Labels:** `enhancement`, `analysis`, `deferred`, `scientific-value:high`, `difficulty:hard` + +**Body:** + +### Scientific Value +CLR/ILR+LM in v1 are basic compositional methods. Advanced methods address specific biases: +- **ANCOM-BC:** Bias correction for sampling fraction, handles zero inflation better than pseudocount +- **ALDEx2:** Bayesian Dirichlet-multinomial, accounts for sampling uncertainty, provides effect size + p-value +- **Songbird:** Multinomial regression for compositional data, ranks taxa by association with covariates + +- **Use case:** Gut microbiome with highly variable library sizes and many zeros — ANCOM-BC reduces false positives from pseudocount. +- **Impact:** More accurate differential abundance for compositional data, which is inherently relative. + +### Scope +- ANCOM-BC via R `ANCOMBC` package +- ALDEx2 via `ALDEx2` package (CLR + Welch's t + BH, with Dirichlet sampling) +- Songbird via QIIME2 or Python `songbird` (multinomial regression) +- Integration with AnalysisConfig: new methods `ancom_bc`, `aldex2`, `songbird` + +### Difficulty +**Hard** — requires: +- R packages `ANCOMBC`, `ALDEx2` (Bioconductor, heavy dependencies) +- Python interop for Songbird (via `PythonCall.jl` or subprocess) +- Zero handling: ANCOM-BC has own zero handling, not pseudocount +- Benchmarking against existing CLR/ILR +- Memory: ALDEx2 Dirichlet sampling 128 samples per taxon x 10k taxa = 1.28M CLR values + +### Risks +- **Dependency hell:** ANCOMBC depends on `lme4`, `pbapply`, etc.; may conflict with existing renv.lock +- **Performance:** Songbird multinomial regression is iterative, may take hours for 10k taxa +- **Scientific controversy:** Compositional methods are debated (Gloor et al. vs Morton et al.); need balanced context help, not taking sides +- **Reproducibility:** Songbird uses TensorFlow, non-deterministic unless seed fixed — need to record seed in provenance + +### Acceptance Criteria +- [ ] New methods in AnalysisConfig v2 +- [ ] BH mandatory, DANGER banner preserved +- [ ] Tests: compare against R reference for 3 datasets (mock, gut, soil) +- [ ] Benchmark: runtime and memory vs CLR_LM, with 10% regression gate for existing methods +- [ ] Docs: explains when to use ANCOM-BC vs CLR vs Songbird, with citations + +--- + +## Issue 4: CladeCumulus — Phylogenetic Tree Integration + +**Title:** `feat(cladistics): CladeCumulus phylogenetic integration — tree from taxonomy + phylogeny, not just taxonomy ranks` + +**Labels:** `enhancement`, `cladistics`, `deferred`, `scientific-value:medium`, `difficulty:hard` + +**Body:** + +### Scientific Value +Current CladeCumulus builds tree from taxonomic ranks (Domain, Phylum, ...). True phylogenetic tree (from 16S sequences via FastTree or IQ-TREE) would enable: +- Cumulative frequencies along phylogenetic branches (not just taxonomic) +- Phylogenetic diversity metrics (Faith's PD, UniFrac) in CladeCumulus +- Detection of clades that are phylogenetically clustered but taxonomically dispersed (e.g., convergent evolution) + +### Scope +- Build phylogeny from ASV sequences (via `DECIPHER` + `phangorn` or external FastTree) +- Integrate with CladeTree: nodes have both taxonomic and phylogenetic parents +- Cumulative frequencies along phylogeny: sum of counts in clade +- Epistemic colour coding still applies, but residual_count now includes phylogenetic uncertainty + +### Difficulty +**Hard** — requires: +- Sequence alignment (MAFFT or DECIPHER) +- Tree building (FastTree, IQ-TREE) — external binary, need provenance like other tools +- Tree parsing (Newick) and integration with existing taxonomy +- Performance: alignment of 10k ASVs O(n^2) memory, tree building O(n^3) worst case + +### Risks +- **Performance:** 10k ASVs alignment may take hours and 10GB RAM — need to limit to top N or provide subsampling +- **Provenance:** New tool (FastTree) needs probing and hashing like vsearch/swarm +- **Scientific:** Phylogeny from short 16S V4 amplicons is noisy; need to warn that tree is approximate +- **UI:** Phylogenetic tree + taxonomic tree = two hierarchies, need toggle or combined view — may clutter UI if not behind Evidence Mode + +--- + +## Issue 5: Evidence Mode — Full Epistemic UI + +**Title:** `feat(ui): Full Evidence Mode — epistemic status editor, fiber visualizer, residual explorer` + +**Labels:** `enhancement`, `ui`, `epistemic`, `deferred`, `scientific-value:medium`, `difficulty:medium` + +**Body:** + +### Scientific Value +Evidence Mode toggle currently only gates Advanced Analysis and CladeCumulus. Full Evidence Mode would include: +- Editor for `avec_fibre` column (manual curation of which rows carry semantic fibre) +- Fiber visualizer: show Echo fiber for a selected taxon (all candidate worlds consistent with observation) +- Residual explorer: finite model of residual decomposition (like residual-evidence-types explorer.html) +- Warrant editor: what evidence tokens support a claim? + +### Scope +- UI for editing avec_fibre boolean per row (in DataTable) +- Fiber visualizer component: shows witnesses (possible true worlds) over observed +- Residual explorer: slider for noise bound, shows candidate count and presence status +- Integration with provenance: fiber edits logged + +### Difficulty +**Medium** — mostly frontend, but needs backend API for fiber computation + +### Risks +- **UI clutter:** If not carefully designed, Evidence Mode could overwhelm non-expert users. Must keep clean, with progressive disclosure. +- **Performance:** Fiber for 10k taxa x 100 candidate worlds = 1M candidates, need virtualized list +- **Scientific misuse:** Manual editing of avec_fibre could be used to cherry-pick results — need to log edits in provenance and DOI bundle, with DANGER banner if many rows changed + +--- + +## Issue 6: DOI Bundle — Zenodo Integration and Automated DOI Minting + +**Title:** `feat(doi): Zenodo integration — automated DOI minting from DOI-ready bundles` + +**Labels:** `enhancement`, `doi`, `infra`, `deferred`, `scientific-value:high`, `difficulty:medium` + +**Body:** + +### Scientific Value +DOI-ready bundles currently create local directory with DataCite JSON. Zenodo integration would: +- Upload bundle to Zenodo via API, mint DOI automatically +- Link DOI to GitHub release, make analysis citable +- Enable reproducibility: anyone with DOI can download exact config + results + provenance + +### Scope +- Zenodo API client (via HTTP.jl) +- Upload bundle zip, create deposition, publish, get DOI +- Store DOI in provenance and link to GitHub Project board +- UI: "Mint DOI" button in AnalysisConfigEditor, shows DOI badge + +### Difficulty +**Medium** — requires Zenodo token, HTTP client, error handling + +### Risks +- **Token security:** Zenodo token must be stored as secret, not in code +- **Cost:** Zenodo is free but has rate limits; need to handle 429 +- **Irreversibility:** Publishing to Zenodo is irreversible (DOI minted) — need confirmation dialog with DANGER banner +- **Dependency:** Zenodo API may change; need to pin API version + +--- + +## Issue Template Footer (for all) + +**For every issue:** +- [ ] Tests and benchmarks (fail CI on >10% regression) +- [ ] JSON + Nickel + DEED schemas updated +- [ ] Docs and context-sensitive help +- [ ] UI clean, behind Evidence Mode if advanced +- [ ] Provenance-rich, immutable derived objects +- [ ] Linked to Project board "Analysis Layer & Cladistics Development" +- [ ] Milestone report after completion diff --git a/frontend/src/components/AdvancedAnalysisExpander.tsx b/frontend/src/components/AdvancedAnalysisExpander.tsx new file mode 100644 index 0000000..9761593 --- /dev/null +++ b/frontend/src/components/AdvancedAnalysisExpander.tsx @@ -0,0 +1,251 @@ +// SPDX-License-Identifier: AGPL-3.0-only +import { useState } from 'react' +import type { AdvancedOverrides } from '../types/analysis_config' +import { contextHelp } from '../types/analysis_config' + +interface AdvancedAnalysisExpanderProps { + evidenceMode: boolean + advanced: AdvancedOverrides + onChange: (advanced: AdvancedOverrides) => void + validationErrors?: Record +} + +export function AdvancedAnalysisExpander({ evidenceMode, advanced, onChange, validationErrors }: AdvancedAnalysisExpanderProps) { + const [expanded, setExpanded] = useState(false) + const [helpField, setHelpField] = useState(null) + + if (!evidenceMode) return null // Keep UI clean — only appears when Evidence Mode enabled + + return ( +
+ + + {expanded && ( +
+

+ All advanced options are behind this expander with heavy validation, context-sensitive help, and refusal of meaningless inputs. + See hyperpolymath/standards for JSON + Nickel + DEED schemes. +

+ + {/* Dispersion method */} +
+ + + {helpField === 'dispersion' && ( +
+ {contextHelp('advanced.dispersion_method')} +
+ )} + {validationErrors?.['advanced.dispersion_method'] && ( +
{validationErrors['advanced.dispersion_method']}
+ )} +
+ + {/* Zero handling */} +
+ + + {advanced.zero_handling === 'refuse' && ( +
+ DANGER: refuse will cause log(0) failures for compositional methods. Requires acknowledgment token. +
+ )} + {helpField === 'zero' && ( +
+ Zero handling determines how zeros are treated. For CLR/ILR, zeros must be replaced because log(0) is undefined. + 'refuse' is mathematically invalid for CLR/ILR and will be refused at runtime even with acknowledgment. +
+ )} +
+ + {/* Min prevalence */} +
+ + { + const v = parseFloat(e.target.value) + if (isNaN(v) || v < 0 || v > 1) { + // Refuse meaningless + alert(`min_prevalence must be in [0,1], got ${e.target.value}. Refusing as meaningless.`) + return + } + onChange({ ...advanced, min_prevalence: v }) + }} + style={{ width: '100%', padding: 8, marginTop: 4 }} + /> + {helpField === 'prevalence' && ( +
+ {contextHelp('advanced.min_prevalence')} +
+ )} + {validationErrors?.['advanced.min_prevalence'] && ( +
{validationErrors['advanced.min_prevalence']}
+ )} +
+ + {/* Min abundance */} +
+ + { + const v = parseFloat(e.target.value) + if (isNaN(v) || v < 0) { + alert(`min_abundance must be >=0, got ${e.target.value}. Refusing as meaningless.`) + return + } + onChange({ ...advanced, min_abundance: v }) + }} + style={{ width: '100%', padding: 8, marginTop: 4 }} + /> +
+ + {/* Max features */} +
+ + { + if (e.target.value === '') { + onChange({ ...advanced, max_features: null }) + return + } + const v = parseInt(e.target.value, 10) + if (isNaN(v) || v <= 0) { + alert(`max_features must be >0 if set, got ${e.target.value}. Refusing.`) + return + } + if (v > 100000) { + alert(`max_features=${v} is absurdly large (>100k). Refusing as meaningless.`) + return + } + onChange({ ...advanced, max_features: v }) + }} + style={{ width: '100%', padding: 8, marginTop: 4 }} + /> +
+ + {/* Min samples per group */} +
+ + { + const v = parseInt(e.target.value, 10) + if (isNaN(v) || v < 2) { + alert(`min_samples_per_group must be >=2, got ${e.target.value}. Refusing. Need at least 2 for variance estimation.`) + return + } + onChange({ ...advanced, min_samples_per_group: v }) + }} + style={{ width: '100%', padding: 8, marginTop: 4 }} + /> + {advanced.min_samples_per_group < 3 && ( +
+ DANGER: <3 samples per group — variance estimation will be unstable. +
+ )} +
+ + {/* Robust */} +
+ +
+ + {advanced.zero_handling === 'refuse' && ( +
+ + onChange({ ...advanced, acknowledgment_token: e.target.value })} + placeholder="I_UNDERSTAND_THE_RISK_AND_WANT_TO_OVERRIDE_BH" + style={{ width: '100%', padding: 8, marginTop: 4, border: '1px solid #c62828', fontFamily: 'monospace' }} + /> +
+ )} + +
+ Heavy validation active: meaningless inputs are refused immediately (empty formula, prevalence outside [0,1], etc.). + All advanced options are logged in provenance and DOI bundle. No silent switching. +
+
+ )} +
+ ) +} diff --git a/frontend/src/components/AnalysisConfigEditor.tsx b/frontend/src/components/AnalysisConfigEditor.tsx new file mode 100644 index 0000000..e6504de --- /dev/null +++ b/frontend/src/components/AnalysisConfigEditor.tsx @@ -0,0 +1,342 @@ +// SPDX-License-Identifier: AGPL-3.0-only +import { useState } from 'react' +import type { AnalysisConfig, ValidationError } from '../types/analysis_config' +import { contextHelp, isDangerous, DANGER_ACK_TOKEN } from '../types/analysis_config' +import { DangerBanner } from './DangerBanner' +import { AdvancedAnalysisExpander } from './AdvancedAnalysisExpander' + +interface AnalysisConfigEditorProps { + evidenceMode: boolean + config: AnalysisConfig + onChange: (config: AnalysisConfig) => void + onSave: () => void + availableMetadataColumns?: string[] + validationErrors?: ValidationError[] +} + +export function AnalysisConfigEditor({ evidenceMode, config, onChange, onSave, availableMetadataColumns, validationErrors }: AnalysisConfigEditorProps) { + const [helpField, setHelpField] = useState(null) + const [showJson, setShowJson] = useState(false) + + const errorsByField: Record = {} + validationErrors?.forEach(e => { errorsByField[e.field] = e.message }) + + const handleFormulaChange = (value: string) => { + // Heavy validation, refusal of meaningless inputs + if (value.includes(';') || value.includes('`') || value.includes('$')) { + alert(`Formula contains forbidden characters (; \` $) that could be injection. Refusing.`) + return + } + if (value.trim().length > 0 && !value.includes('~')) { + alert(`Formula must contain '~' (R-style), e.g. '~ group'. Got '${value}'. Refusing ambiguous formula.`) + // Still allow typing, but show error + } + onChange({ ...config, formula: value }) + } + + return ( +
+

+ AnalysisConfig — explicit, versioned, immutable + + v{config.schema_version} + + + {config.hash.slice(0, 12)}... + +

+ + { + onChange({ + ...config, + correction: { ...config.correction, acknowledgment_token: token, allow_no_correction: true }, + advanced: { ...config.advanced, acknowledgment_token: token } + }) + }} /> + + {/* Method — explicit, no auto-selection */} +
+ + + {helpField === 'method' && ( +
+ {contextHelp('method')} +
+ )} + {errorsByField['method'] &&
{errorsByField['method']}
} +
+ + {/* Formula */} +
+ + handleFormulaChange(e.target.value)} + placeholder="~ group" + style={{ width: '100%', padding: 8, marginTop: 4, fontFamily: 'monospace' }} + /> + {helpField === 'formula' && ( +
+ {contextHelp('formula')} +
+ )} + {errorsByField['formula'] &&
{errorsByField['formula']}
} + {availableMetadataColumns && ( +
+ Available columns: {availableMetadataColumns.join(', ')} +
+ )} +
+ + {/* Metadata columns */} +
+ + { + const cols = e.target.value.split(',').map(s => s.trim()).filter(s => s.length > 0) + if (cols.length === 0) { + alert('metadata_columns must be non-empty. Refusing empty as meaningless.') + return + } + onChange({ ...config, metadata_columns: cols }) + }} + placeholder="group, batch" + style={{ width: '100%', padding: 8, marginTop: 4, fontFamily: 'monospace' }} + /> + {errorsByField['metadata_columns'] &&
{errorsByField['metadata_columns']}
} +
+ + {/* Outcome column for logistic */} + {config.method === 'logistic' && ( +
+ + onChange({ ...config, outcome_column: e.target.value || null })} + placeholder="disease" + style={{ width: '100%', padding: 8, marginTop: 4 }} + /> + {errorsByField['outcome_column'] &&
{errorsByField['outcome_column']}
} +
+ )} + + {/* Normalization */} +
+ + + + {(config.normalization.method === 'clr' || config.normalization.method === 'ilr') && ( +
+ + { + const v = parseFloat(e.target.value) + if (isNaN(v) || v <= 0) { + alert(`pseudocount must be >0 for CLR/ILR (log(0) undefined). Got ${e.target.value}. Refusing.`) + return + } + onChange({ ...config, normalization: { ...config.normalization, pseudocount: v } }) + }} + style={{ width: '100%', padding: 8, marginTop: 4 }} + /> +
+ )} + + {config.normalization.method === 'ilr' && ( +
+ + +
+ )} + + {helpField === 'norm' && ( +
+ {contextHelp('normalization.method')} +
+ )} + {errorsByField['normalization.method'] &&
{errorsByField['normalization.method']}
} +
+ + {/* Correction — BH mandatory */} +
+ +
+ onChange({ ...config, correction: { ...config.correction, method: e.target.value } })} + style={{ flex: 1, padding: 8, fontFamily: 'monospace' }} + placeholder="BH" + /> + { + const v = parseFloat(e.target.value) + if (isNaN(v) || v <= 0 || v >= 1) { + alert(`alpha must be in (0,1), got ${e.target.value}. Typical 0.05. Refusing.`) + return + } + onChange({ ...config, correction: { ...config.correction, alpha: v } }) + }} + style={{ width: 100, padding: 8 }} + /> +
+ + + + {config.correction.allow_no_correction && ( + onChange({ ...config, correction: { ...config.correction, acknowledgment_token: e.target.value } })} + placeholder={DANGER_ACK_TOKEN} + style={{ width: '100%', padding: 8, marginTop: 8, border: '2px solid #c62828', fontFamily: 'monospace' }} + /> + )} + + {helpField === 'correction' && ( +
+ {contextHelp('correction.method')} +
+ )} +
+ + {/* Advanced — behind Evidence Mode */} + onChange({ ...config, advanced: adv })} + validationErrors={errorsByField} + /> + + {/* Provenance preview */} +
+ Provenance: ID {config.id.slice(0, 8)}... | Hash {config.hash.slice(0, 12)}... | Created {config.created_at} by {config.created_by} +
+ Immutable: Every analysis is an explicit, immutable, provenance-rich derived object. No silent switching. +
+ DOI-ready: Bundle includes JSON + Nickel + DEED + DataCite + provenance. +
+ + {/* Actions */} +
+ + +
+ + {showJson && ( +
+

JSON (canonical, hash: {config.hash.slice(0, 16)}...)

+
+            {JSON.stringify(config, null, 2)}
+          
+

+ Nickel and DEED representations are available in DOI bundle and via API. + See config/schemas/analysis_config.schema.json, .ncl, and _chora.deed template (hyperpolymath/standards). +

+
+ )} +
+ ) +} diff --git a/frontend/src/components/CladeCumulus.tsx b/frontend/src/components/CladeCumulus.tsx new file mode 100644 index 0000000..a71bb09 --- /dev/null +++ b/frontend/src/components/CladeCumulus.tsx @@ -0,0 +1,217 @@ +// SPDX-License-Identifier: AGPL-3.0-only +import { useState, useEffect, useRef } from 'react' + +interface CladeNode { + id: string + label: string + rank: string + parent_id: string | null + children_ids: string[] + count: number + cumulative_count: number + cumulative_frequency: number + residual_count: number + avec_fibre: boolean + epistemic_status: 'present_in_every_admissible_world' | 'present_in_some_admissible_world' | 'absent_in_every_admissible_world' | 'unknown' + colour: string + cloud_size: number +} + +interface CladeTree { + root_id: string + total_count: number + nodes: CladeNode[] +} + +interface CladeCumulusProps { + evidenceMode: boolean + tree?: CladeTree | null + onDragDrop?: (draggedId: string, targetParentId: string) => Promise<{ valid: boolean; message: string }> + onNodeClick?: (node: CladeNode) => void +} + +function epistemicColour(status: CladeNode['epistemic_status']): string { + switch (status) { + case 'present_in_every_admissible_world': return '#2e7d32' + case 'present_in_some_admissible_world': return '#f9a825' + case 'absent_in_every_admissible_world': return '#9e9e9e' + default: return '#c62828' + } +} + +function cloudSize(residual: number): number { + return Math.log(1 + residual) * 10 + 5 +} + +export function CladeCumulus({ evidenceMode, tree, onDragDrop, onNodeClick }: CladeCumulusProps) { + const [draggedId, setDraggedId] = useState(null) + const [validationMsg, setValidationMsg] = useState(null) + const [validationValid, setValidationValid] = useState(null) + const [hoveredId, setHoveredId] = useState(null) + const svgRef = useRef(null) + + if (!evidenceMode) return null // Keep UI clean — only appears when Evidence Mode enabled + + if (!tree) { + return ( +
+

CladeCumulus — Cumulative Cladistic Explorer

+

Tree with cumulative frequencies, epistemic colour coding, cloud sizing by residual count, drag-and-drop with live present_in_every_admissible_world validation.

+

No tree data yet. Run analysis or load a study with taxonomy to see the cladistic explorer.

+

+ Based on hyperpolymath/echo-types (Echo f y := Σ (x:A), f x ≡ y), epistemic-types (Warrant without soundness), residual-evidence-types (Candidate, Holds, Identified) +

+
+ ) + } + + const nodesById = new Map(tree.nodes.map(n => [n.id, n])) + + const handleDragStart = (id: string) => { + setDraggedId(id) + setValidationMsg(null) + } + + const handleDragOver = async (targetId: string) => { + if (!draggedId || !onDragDrop) return + if (draggedId === targetId) return + + // Live present_in_every_admissible_world validation + try { + const result = await onDragDrop(draggedId, targetId) + setValidationMsg(result.message) + setValidationValid(result.valid) + } catch (e) { + setValidationMsg(`Validation error: ${e}`) + setValidationValid(false) + } + } + + const handleDrop = async (targetId: string) => { + if (!draggedId || !onDragDrop) return + const result = await onDragDrop(draggedId, targetId) + setValidationMsg(result.message) + setValidationValid(result.valid) + if (result.valid) { + // In real implementation, would update tree via API + console.log(`Valid drop: ${draggedId} -> ${targetId}`) + } + setDraggedId(null) + } + + // Simple tree layout — for production would use D3 hierarchy + const renderNode = (node: CladeNode, depth: number, x: number, y: number): JSX.Element => { + const isDragged = draggedId === node.id + const isHovered = hoveredId === node.id + const colour = node.colour || epistemicColour(node.epistemic_status) + const size = node.cloud_size || cloudSize(node.residual_count) + + return ( + + {/* Cloud sizing by residual count */} + handleDragStart(node.id)} + onDragOver={e => { e.preventDefault(); handleDragOver(node.id) }} + onDrop={e => { e.preventDefault(); handleDrop(node.id) }} + onMouseEnter={() => setHoveredId(node.id)} + onMouseLeave={() => setHoveredId(null)} + onClick={() => onNodeClick?.(node)} + /> + + {node.label} ({(node.cumulative_frequency * 100).toFixed(1)}%) + + {/* Epistemic status indicator */} + + {node.avec_fibre ? 'avec_fibre' : 'sans_fibre'} • {node.epistemic_status.split('_')[0]} + + + {/* Children */} + {node.children_ids.map((childId, idx) => { + const child = nodesById.get(childId) + if (!child) return null + const childX = 40 + const childY = (idx - (node.children_ids.length - 1) / 2) * 60 + return ( + + + + {renderNode(child, depth + 1, 0, 0)} + + + ) + })} + + ) + } + + const root = nodesById.get(tree.root_id) + if (!root) return
Root not found
+ + return ( +
+

+ CladeCumulus — Cumulative Cladistic Explorer + Evidence Mode +

+ +

+ Tree with cumulative frequencies, epistemic colour coding (green=present in every admissible world, yellow=some, grey=absent, red=unknown/sans fibre), + cloud sizing by residual count (larger cloud = more candidate worlds), drag-and-drop with live present_in_every_admissible_world validation. + Clean non-cluttered UI — only visible in Evidence Mode. +

+ + {/* Legend */} +
+ present_in_every (solid evidence) + present_in_some (uncertain) + absent_in_every + unknown / sans fibre + Cloud size ∝ log(1 + residual_count) +
+ + {/* Validation message */} + {validationMsg && ( +
+ {validationValid ? '✓' : '✗'} {validationMsg} +
+ )} + + {/* SVG tree */} +
+ + {renderNode(root, 0, 50, 300)} + +
+ +
+ Total count: {tree.total_count} | Nodes: {tree.nodes.length} | Root: {tree.root_id} +
+ Based on hyperpolymath/echo-types (fiber laws, total space equivalence), epistemic-types (Warrant, SoundWarrant, E κ A), residual-evidence-types (Candidate, Holds, Identified, actual-world-sound) +
+ Drag-and-drop validates live via present_in_every_admissible_world: taxon must be present in every admissible world consistent with observation and evidence (avec_fibre). +
+
+ ) +} diff --git a/frontend/src/components/DangerBanner.tsx b/frontend/src/components/DangerBanner.tsx new file mode 100644 index 0000000..2081df0 --- /dev/null +++ b/frontend/src/components/DangerBanner.tsx @@ -0,0 +1,90 @@ +// SPDX-License-Identifier: AGPL-3.0-only +import type { AnalysisConfig } from '../types/analysis_config' +import { isDangerous, DANGER_ACK_TOKEN } from '../types/analysis_config' + +interface DangerBannerProps { + config: AnalysisConfig + onAcknowledge?: (token: string) => void +} + +export function DangerBanner({ config, onAcknowledge }: DangerBannerProps) { + if (!isDangerous(config)) return null + + const reasons: string[] = [] + if (config.correction.allow_no_correction) { + reasons.push(`BH correction disabled (method=${config.correction.method}). This will inflate false discoveries in high-dimensional data.`) + } + if (config.advanced.zero_handling === 'refuse') { + reasons.push(`zero_handling='refuse' will cause log(0) or biased zero handling.`) + } + if (config.normalization.method === 'rarefy' && config.method === 'nb_glm') { + reasons.push(`Rarefaction for NB_GLM discards data and reduces power; size_factors preferred (McMurdie & Holmes 2014).`) + } + if (config.advanced.min_samples_per_group < 3) { + reasons.push(`min_samples_per_group=${config.advanced.min_samples_per_group} <3: variance estimation will be unstable.`) + } + + return ( +
+
+ ⚠️ DANGER — SCIENTIFICALLY RISKY CONFIGURATION DETECTED ⚠️ +
+
+ You have enabled overrides that weaken statistical rigor: +
    + {reasons.map((r, i) =>
  • {r}
  • )} +
+
+
+ This configuration will be: +
• Logged in provenance with full user identity and timestamp +
• Bannered in every figure and DOI bundle +
• Flagged in the GitHub Project board as 'needs-review' +
+
+ If you are sure, you must acknowledge with token: +
+ + {DANGER_ACK_TOKEN} + +
+ Paste this token into the acknowledgment field below. +
+ + {onAcknowledge && ( +
+ { + if (e.target.value === DANGER_ACK_TOKEN) { + onAcknowledge(e.target.value) + } + }} + style={{ + width: '100%', + padding: '8px', + border: '1px solid #c62828', + borderRadius: 4, + fontFamily: 'monospace', + }} + /> +
+ )} + +
+ Consider: Is there a safer alternative? Consult context-sensitive help. + See: hyperpolymath/standards JSON + Nickel + DEED schemes. +
+
+ ) +} diff --git a/frontend/src/components/EvidenceModeToggle.tsx b/frontend/src/components/EvidenceModeToggle.tsx new file mode 100644 index 0000000..5955a0e --- /dev/null +++ b/frontend/src/components/EvidenceModeToggle.tsx @@ -0,0 +1,81 @@ +// SPDX-License-Identifier: AGPL-3.0-only +import { useState } from 'react' + +interface EvidenceModeToggleProps { + enabled: boolean + onToggle: (enabled: boolean) => void +} + +export function EvidenceModeToggle({ enabled, onToggle }: EvidenceModeToggleProps) { + const [showInfo, setShowInfo] = useState(false) + + return ( +
+ + + + + {enabled && ( + + Advanced analysis & cladistic visuals enabled + + )} + + {showInfo && ( +
+

Evidence Mode

+

+ Evidence Mode enables advanced, epistemic-aware features that require understanding of + Echo Types (fibers as structured loss witnesses), Epistemic Types (warrant without assumed soundness), + and Residual Evidence (presence in every admissible world). +

+
    +
  • AnalysisConfig layer: explicit, versioned, BH mandatory, DANGER banner on overrides, DOI-ready bundles
  • +
  • CladeCumulus: cumulative cladistic tree with epistemic colour coding, cloud sizing by residual count, drag-and-drop with live present_in_every_admissible_world validation
  • +
  • Advanced Analysis expander: dispersion methods, zero handling, prevalence filters — all with heavy validation and context-sensitive help
  • +
+

+ Based on hyperpolymath/echo-types, epistemic-types, residual-evidence-types. + Every analysis is an explicit, immutable, provenance-rich derived object. No silent switching. +

+ +
+ )} +
+ ) +} diff --git a/frontend/src/types/analysis_config.ts b/frontend/src/types/analysis_config.ts new file mode 100644 index 0000000..7b3fd46 --- /dev/null +++ b/frontend/src/types/analysis_config.ts @@ -0,0 +1,110 @@ +// SPDX-License-Identifier: AGPL-3.0-only +// AnalysisConfig — safe, explicit, versioned layer +// Frontend types mirroring Julia src/analysis/analysis_config.jl +// JSON + Nickel + DEED schemes from hyperpolymath/standards + +export type AnalysisMethod = 'nb_glm' | 'clr_lm' | 'ilr_lm' | 'logistic' + +export interface NormalizationConfig { + method: 'none' | 'rarefy' | 'relative' | 'size_factors' | 'clr' | 'ilr' | 'presence_absence' + pseudocount: number + ilr_basis?: string | null + multiplicative_replacement_delta?: number | null +} + +export interface CorrectionConfig { + method: string + alpha: number + allow_no_correction: boolean + acknowledgment_token?: string | null +} + +export interface AdvancedOverrides { + dispersion_method: 'parametric' | 'local' | 'mean' | 'pooled' | 'glmGamPoi' + zero_handling: 'pseudocount' | 'multiplicative_replacement' | 'bayesian_multiplicative' | 'refuse' + min_prevalence: number + min_abundance: number + max_features?: number | null + min_samples_per_group: number + robust: boolean + acknowledgment_token?: string | null +} + +export interface AnalysisConfig { + schema_version: string + id: string + created_at: string + created_by: string + method: AnalysisMethod + formula: string + outcome_column?: string | null + metadata_columns: string[] + normalization: NormalizationConfig + correction: CorrectionConfig + advanced: AdvancedOverrides + provenance: Record + hash: string +} + +export interface AnalysisResult { + id: string + config_id: string + config_hash: string + created_at: string + method: AnalysisMethod + results: Record + provenance: Record + hash: string +} + +export interface ValidationError { + field: string + message: string + help?: string +} + +export const DANGER_ACK_TOKEN = 'I_UNDERSTAND_THE_RISK_AND_WANT_TO_OVERRIDE_BH' + +export function isDangerous(config: AnalysisConfig): boolean { + if (config.correction.allow_no_correction) return true + if (config.advanced.zero_handling === 'refuse') return true + if (config.normalization.method === 'rarefy' && config.method === 'nb_glm') return true + if (config.advanced.min_samples_per_group < 3) return true + return false +} + +export function contextHelp(fieldPath: string): string { + const helpDb: Record = { + method: `Analysis Method (required, explicit, no auto-selection) +- nb_glm: Negative Binomial GLM for raw counts with overdispersion. Uses DESeq2-style size factors. +- clr_lm: Centered Log-Ratio + Gaussian LM (compositional, Aitchison geometry). Requires pseudocount. +- ilr_lm: Isometric Log-Ratio + Gaussian LM (balances, phylogenetic basis possible) +- logistic: Logistic regression for binary outcome (presence/absence)`, + formula: `R-style formula, e.g. '~ group' or 'disease ~ group + batch' +- Left of ~ is outcome (required for logistic) +- Right of ~ lists metadata columns +- Must reference only columns in metadata_columns +- Forbidden: ; \` $ (injection prevention)`, + 'normalization.method': `Normalization / Transform (method-dependent) +- For NB_GLM: none, size_factors, relative, rarefy (discouraged) +- For CLR_LM: must be clr +- For ILR_LM: must be ilr +- For LOGISTIC: presence_absence, none, relative`, + 'normalization.pseudocount': `Pseudocount for zero replacement (CLR/ILR only) +Typical: 0.5. Must be >0 because log(0) undefined. +Too small creates extreme log-ratios, too large distorts low-abundance features.`, + 'correction.method': `Multiple testing correction — BH mandatory in v1 +Microbiome data tests thousands of taxa. Uncorrected p-values give ~5% false positives under null. +BH (Benjamini-Hochberg FDR) is mandatory. Override requires DANGER banner and acknowledgment token.`, + 'advanced.min_prevalence': `Minimum prevalence filter [0,1] +Feature must be present in at least this fraction of samples. +0.1 = present in >=10% samples. Recommended to reduce multiple testing burden.`, + 'advanced.dispersion_method': `Dispersion estimation (NB_GLM advanced) +- parametric: fit dispersion ~ mean trend (DESeq2 default) +- local: local regression fit +- mean: use mean dispersion +- pooled: pool across genes (when n small) +- glmGamPoi: fast estimator`, + } + return helpDb[fieldPath] ?? `No help available for '${fieldPath}'.` +} diff --git a/src/MetaManifold.jl b/src/MetaManifold.jl index 2f0e50f..4657c04 100644 --- a/src/MetaManifold.jl +++ b/src/MetaManifold.jl @@ -16,6 +16,7 @@ include("core/categories.jl") include("core/composition_library.jl") include("core/primers_library.jl") include("core/databases_library.jl") +include("core/epistemic.jl") # Annotation include("annotation/funcdb.jl") @@ -29,5 +30,7 @@ include("pipeline/swarm.jl") # Analysis include("analysis/diversity.jl") include("analysis/analysis.jl") +include("analysis/analysis_config.jl") +include("analysis/clade_cumulus.jl") end diff --git a/src/analysis/analysis_config.jl b/src/analysis/analysis_config.jl new file mode 100644 index 0000000..4a8149c --- /dev/null +++ b/src/analysis/analysis_config.jl @@ -0,0 +1,1023 @@ +# SPDX-License-Identifier: AGPL-3.0-only +""" + AnalysisConfig — safe, explicit, versioned analysis configuration layer + +Implements the requirements from the hyperpolymath estate task: + +- Explicit, versioned, immutable, provenance-rich derived objects +- Methods in v1: NB GLM, CLR/ILR+Gaussian LM, logistic +- BH mandatory, hard-stop with DANGER banner on overrides +- All advanced options behind "Advanced Analysis" expander +- Heavy validation, context-sensitive help, refusal of meaningless inputs +- JSON + Nickel + DEED schemes from hyperpolymath/standards +- DOI-ready bundles + +No silent switching or auto-selection. Every field must be explicit. +""" +module AnalysisConfig + +using Dates +using SHA +using UUIDs +using JSON3 +using OrderedCollections +using ..Provenance: CapturedEnvironment, probe_metamanifold, probe_host + +export AnalysisMethod, NormalizationConfig, CorrectionConfig, AdvancedOverrides, + AnalysisConfigStruct, AnalysisResult, + validate_config, config_hash, canonical_json, + context_help, danger_banner, is_dangerous, + to_json, from_json, to_deed, to_nickel, + create_doi_bundle, present_in_every_admissible_world, + AVEC_FIBRE_COLUMN, EPISTEMIC_STATUS_VALUES + +# -------------------------------------------------------------------------- +# Constants and enums (explicit, no silent defaults) +# -------------------------------------------------------------------------- + +const SCHEMA_VERSION = "1.0.0" +const SCHEMA_VERSIONS_SUPPORTED = ("1.0.0",) +const AVEC_FIBRE_COLUMN = "avec_fibre" +const EPISTEMIC_STATUS_VALUES = ("present_in_every_admissible_world", + "present_in_some_admissible_world", + "absent_in_every_admissible_world", + "unknown") + +const DANGER_ACK_TOKEN = "I_UNDERSTAND_THE_RISK_AND_WANT_TO_OVERRIDE_BH" + +@enum AnalysisMethod begin + NB_GLM = 1 # Negative Binomial GLM (DESeq2/MASS style) + CLR_LM = 2 # CLR transform + Gaussian LM + ILR_LM = 3 # ILR transform + Gaussian LM + LOGISTIC = 4 # Logistic regression (presence/absence) +end + +const METHOD_STRINGS = Dict{String,AnalysisMethod}( + "nb_glm" => NB_GLM, + "clr_lm" => CLR_LM, + "ilr_lm" => ILR_LM, + "logistic" => LOGISTIC, + "NB_GLM" => NB_GLM, + "CLR_LM" => CLR_LM, + "ILR_LM" => ILR_LM, + "LOGISTIC" => LOGISTIC, +) + +const METHOD_TO_STRING = Dict{AnalysisMethod,String}( + NB_GLM => "nb_glm", + CLR_LM => "clr_lm", + ILR_LM => "ilr_lm", + LOGISTIC => "logistic", +) + +const VALID_DISPERSION_METHODS = ("parametric", "local", "mean", "pooled", "glmGamPoi") +const VALID_ZERO_HANDLING = ("pseudocount", "multiplicative_replacement", "bayesian_multiplicative", "refuse") +const VALID_ILR_BASIS = ("default", "phylogenetic", "sequential_binary_partition", "balance_dendrogram") +const VALID_NORMALIZATION_FOR_METHOD = Dict{AnalysisMethod, Vector{String}}( + NB_GLM => ["none", "rarefy", "size_factors", "relative"], + CLR_LM => ["clr"], + ILR_LM => ["ilr"], + LOGISTIC => ["none", "relative", "rarefy", "presence_absence"], +) + +# -------------------------------------------------------------------------- +# Sub-configs +# -------------------------------------------------------------------------- + +""" + NormalizationConfig — explicit normalization / compositional transform + +For CLR/ILR, pseudocount is mandatory and must be >0. For NB_GLM, size_factors is preferred. +Refuses meaningless: pseudocount <=0, ilr_basis invalid, method not compatible with parent method. +""" +struct NormalizationConfig + method::String + pseudocount::Float64 + ilr_basis::Union{String,Nothing} + multiplicative_replacement_delta::Union{Float64,Nothing} + + function NormalizationConfig(; method::String, + pseudocount::Float64=0.5, + ilr_basis::Union{String,Nothing}=nothing, + multiplicative_replacement_delta::Union{Float64,Nothing}=nothing) + method = lowercase(strip(method)) + isempty(method) && throw(ArgumentError("normalization.method must be non-empty (e.g. 'clr', 'none')")) + + # Heavy validation + if method in ("clr", "ilr") + pseudocount <= 0 && throw(ArgumentError("For compositional methods (CLR/ILR), pseudocount must be >0 (got $pseudocount). Zero replacement is mandatory because log(0) is undefined.")) + pseudocount >= 1 && @warn "pseudocount >=1 is unusual for CLR/ILR (got $pseudocount); typical is 0.5 or 0.65. This may distort low-abundance features." + if method == "ilr" + if !isnothing(ilr_basis) && !(ilr_basis in VALID_ILR_BASIS) + throw(ArgumentError("ilr_basis must be one of $(join(VALID_ILR_BASIS, ", ")) (got '$ilr_basis')")) + end + else # clr + if !isnothing(ilr_basis) + throw(ArgumentError("ilr_basis is meaningless for CLR (only for ILR). Refusing.")) + end + end + else + # For non-compositional, ilr_basis is meaningless + if !isnothing(ilr_basis) + throw(ArgumentError("ilr_basis is only meaningful for ILR method, not for '$method'")) + end + end + + if !isnothing(multiplicative_replacement_delta) + delta = multiplicative_replacement_delta + (delta <= 0 || delta >= 1) && throw(ArgumentError("multiplicative_replacement_delta must be in (0,1), got $delta")) + end + + new(method, pseudocount, ilr_basis, multiplicative_replacement_delta) + end +end + +""" + CorrectionConfig — BH mandatory + +BH is the only sane default for high-dimensional microbiome data. Any override triggers DANGER. +""" +struct CorrectionConfig + method::String + alpha::Float64 + allow_no_correction::Bool + acknowledgment_token::Union{String,Nothing} + + function CorrectionConfig(; method::String="BH", + alpha::Float64=0.05, + allow_no_correction::Bool=false, + acknowledgment_token::Union{String,Nothing}=nothing) + method_clean = strip(method) + isempty(method_clean) && throw(ArgumentError("correction.method must be non-empty")) + + # BH mandatory check + if uppercase(method_clean) != "BH" && uppercase(method_clean) != "FDR" && uppercase(method_clean) != "BENJAMINI-HOCHBERG" + if !allow_no_correction + throw(ArgumentError("p-value correction method must be BH (Benjamini-Hochberg) in v1. Got '$method_clean'. If you truly want to override, you must set allow_no_correction=true and provide acknowledgment_token='$DANGER_ACK_TOKEN'. This will trigger a DANGER banner and be recorded in provenance.")) + end + end + + if allow_no_correction + if isnothing(acknowledgment_token) || acknowledgment_token != DANGER_ACK_TOKEN + throw(ArgumentError("DANGER: You are attempting to disable BH correction. This is scientifically dangerous for high-dimensional data and will inflate false discoveries. To proceed, you must set acknowledgment_token to exactly '$DANGER_ACK_TOKEN'. This action will be logged, bannered, and included in DOI bundle provenance.")) + end + end + + (alpha <= 0 || alpha >= 1) && throw(ArgumentError("alpha must be in (0,1), got $alpha. Typical is 0.05.")) + + # Normalize to canonical BH + canonical_method = allow_no_correction ? method_clean : "BH" + + new(canonical_method, alpha, allow_no_correction, acknowledgment_token) + end +end + +""" + AdvancedOverrides — all behind Advanced Analysis expander + +Heavy validation, context-sensitive help, refusal of meaningless. +""" +struct AdvancedOverrides + dispersion_method::String + zero_handling::String + min_prevalence::Float64 + min_abundance::Float64 + max_features::Union{Int,Nothing} + min_samples_per_group::Int + robust::Bool + acknowledgment_token::Union{String,Nothing} # for any advanced override that is dangerous + + function AdvancedOverrides(; dispersion_method::String="parametric", + zero_handling::String="pseudocount", + min_prevalence::Float64=0.1, + min_abundance::Float64=0.0, + max_features::Union{Int,Nothing}=nothing, + min_samples_per_group::Int=3, + robust::Bool=false, + acknowledgment_token::Union{String,Nothing}=nothing) + + dispersion_method = lowercase(strip(dispersion_method)) + zero_handling = lowercase(strip(zero_handling)) + + !(dispersion_method in VALID_DISPERSION_METHODS) && + throw(ArgumentError("dispersion_method must be one of $(join(VALID_DISPERSION_METHODS, ", ")) (got '$dispersion_method')")) + + !(zero_handling in VALID_ZERO_HANDLING) && + throw(ArgumentError("zero_handling must be one of $(join(VALID_ZERO_HANDLING, ", ")) (got '$zero_handling')")) + + (min_prevalence < 0 || min_prevalence > 1) && + throw(ArgumentError("min_prevalence must be in [0,1], got $min_prevalence. 0.1 means feature must be present in at least 10% of samples.")) + + min_abundance < 0 && + throw(ArgumentError("min_abundance must be >=0, got $min_abundance")) + + if !isnothing(max_features) + max_features <= 0 && throw(ArgumentError("max_features must be >0 if set, got $max_features")) + max_features > 100000 && throw(ArgumentError("max_features=$max_features is absurdly large (>100k). Refusing as meaningless.")) + end + + min_samples_per_group < 2 && + throw(ArgumentError("min_samples_per_group must be >=2 (got $min_samples_per_group). Need at least 2 samples per group for variance estimation.")) + + if zero_handling == "refuse" + # This is dangerous for CLR/ILR - will cause log(0) errors + if isnothing(acknowledgment_token) || acknowledgment_token != DANGER_ACK_TOKEN + throw(ArgumentError("DANGER: zero_handling='refuse' will cause log(0) failures for compositional methods and drop zeros for NB_GLM, biasing results. To proceed, set acknowledgment_token='$DANGER_ACK_TOKEN'.")) + end + end + + new(dispersion_method, zero_handling, min_prevalence, min_abundance, max_features, min_samples_per_group, robust, acknowledgment_token) + end +end + +# -------------------------------------------------------------------------- +# Main immutable config +# -------------------------------------------------------------------------- + +""" + AnalysisConfigStruct — immutable, versioned, provenance-rich derived object + +Every analysis is an explicit, immutable, provenance-rich derived object. No silent switching. +""" +struct AnalysisConfigStruct + schema_version::String + id::String + created_at::DateTime + created_by::String + method::AnalysisMethod + formula::String + outcome_column::Union{String,Nothing} + metadata_columns::Vector{String} + normalization::NormalizationConfig + correction::CorrectionConfig + advanced::AdvancedOverrides + provenance::OrderedDict{String,Any} + hash::String + + function AnalysisConfigStruct(; schema_version::String=SCHEMA_VERSION, + id::String=string(uuid4()), + created_at::DateTime=now(UTC), + created_by::String="anonymous", + method::Union{String,AnalysisMethod}, + formula::String, + outcome_column::Union{String,Nothing}=nothing, + metadata_columns::Vector{String}, + normalization::NormalizationConfig, + correction::CorrectionConfig=CorrectionConfig(), + advanced::AdvancedOverrides=AdvancedOverrides(), + provenance::OrderedDict{String,Any}=OrderedDict{String,Any}(), + hash::String="") + + # schema_version must be supported + schema_version in SCHEMA_VERSIONS_SUPPORTED || + throw(ArgumentError("Unsupported schema_version '$schema_version'. Supported: $(join(SCHEMA_VERSIONS_SUPPORTED, ", "))")) + + # method parsing — explicit, no auto-selection + m = if method isa AnalysisMethod + method + else + ms = lowercase(strip(String(method))) + get(METHOD_STRINGS, ms, nothing) |> + x -> isnothing(x) ? throw(ArgumentError("method must be one of $(join(keys(METHOD_STRINGS), ", ")) (got '$method')")) : x + end + + # formula heavy validation + formula_stripped = strip(formula) + isempty(formula_stripped) && throw(ArgumentError("formula must be non-empty, e.g. '~ group' or 'disease ~ group + batch'. Refusing empty formula as meaningless.")) + # Must contain ~ or be a single variable? We enforce R-style formula with ~ + occursin(r"[;`\$]", formula_stripped) && throw(ArgumentError("formula contains forbidden characters (; ` \$) that could be injection. Refusing.")) + # Basic R formula check: must have ~ somewhere, or for some methods allow single term? + if !occursin("~", formula_stripped) + # For some methods, we allow "~ var" shorthand? Actually we require ~ always for explicitness + throw(ArgumentError("formula must be an R-style formula containing '~', e.g. '~ group' or 'outcome ~ group + batch'. Got '$formula_stripped'")) + end + # Check that formula is not just "~" or "~ 1" without metadata + if strip(replace(formula_stripped, "~" => "")) in ("", "1", "0") + throw(ArgumentError("formula '$formula_stripped' is meaningless (no covariates). Must reference at least one metadata column.")) + end + + # metadata_columns validation + isempty(metadata_columns) && throw(ArgumentError("metadata_columns must be non-empty. At least one column must be specified.")) + for col in metadata_columns + isempty(strip(col)) && throw(ArgumentError("metadata_columns contains empty string — refusing as meaningless.")) + occursin(r"[^a-zA-Z0-9_\.\-]", col) && @warn "metadata column '$col' contains unusual characters; will be validated against actual metadata schema later." + end + length(metadata_columns) != length(unique(metadata_columns)) && + throw(ArgumentError("metadata_columns contains duplicates: $(metadata_columns)")) + + # outcome_column validation for LOGISTIC + if m == LOGISTIC + isnothing(outcome_column) && throw(ArgumentError("For LOGISTIC method, outcome_column must be specified (binary outcome). Refusing.")) + isempty(strip(outcome_column)) && throw(ArgumentError("outcome_column must be non-empty for LOGISTIC")) + else + if !isnothing(outcome_column) && m != NB_GLM + @warn "outcome_column is specified for method $(METHOD_TO_STRING[m]) which typically does not need it; will be ignored unless formula uses it." + end + end + + # normalization compatibility + allowed_norm = VALID_NORMALIZATION_FOR_METHOD[m] + norm_method = normalization.method + if !(norm_method in allowed_norm) + throw(ArgumentError("Normalization method '$norm_method' is incompatible with analysis method '$(METHOD_TO_STRING[m])'. Allowed for this method: $(join(allowed_norm, ", ")). Refusing meaningless combination.")) + end + + # min_samples_per_group already validated in AdvancedOverrides, but cross-check with metadata + # (actual existence check happens in validate_config with available columns) + + # provenance enrichment + prov = OrderedDict{String,Any}(provenance) + if !haskey(prov, "metamanifold") + try + prov["metamanifold"] = probe_metamanifold() + catch + prov["metamanifold"] = OrderedDict("version" => "unknown") + end + end + if !haskey(prov, "host") + prov["host"] = probe_host() + end + prov["created_at"] = string(created_at) + prov["created_by"] = created_by + prov["schema_version"] = schema_version + + # canonical hash (without hash field itself, to avoid circularity) + temp_dict = OrderedDict{String,Any}( + "schema_version" => schema_version, + "id" => id, + "created_at" => string(created_at), + "created_by" => created_by, + "method" => METHOD_TO_STRING[m], + "formula" => formula_stripped, + "outcome_column" => outcome_column, + "metadata_columns" => sort(metadata_columns), + "normalization" => OrderedDict( + "method" => normalization.method, + "pseudocount" => normalization.pseudocount, + "ilr_basis" => normalization.ilr_basis, + "multiplicative_replacement_delta" => normalization.multiplicative_replacement_delta, + ), + "correction" => OrderedDict( + "method" => correction.method, + "alpha" => correction.alpha, + "allow_no_correction" => correction.allow_no_correction, + ), + "advanced" => OrderedDict( + "dispersion_method" => advanced.dispersion_method, + "zero_handling" => advanced.zero_handling, + "min_prevalence" => advanced.min_prevalence, + "min_abundance" => advanced.min_abundance, + "max_features" => advanced.max_features, + "min_samples_per_group" => advanced.min_samples_per_group, + "robust" => advanced.robust, + ), + ) + canonical = JSON3.write(temp_dict) + computed_hash = isnothing(hash) || isempty(hash) ? bytes2hex(sha256(canonical)) : hash + + new(schema_version, id, created_at, created_by, m, formula_stripped, outcome_column, metadata_columns, normalization, correction, advanced, prov, computed_hash) + end +end + +# -------------------------------------------------------------------------- +# Validation with available metadata (context-sensitive) +# -------------------------------------------------------------------------- + +""" + validate_config(config, available_metadata_columns; strict=true) -> Vector{String} + +Heavy validation, context-sensitive, refusal of meaningless inputs. +Returns list of errors (empty = valid). If strict, throws on first error. +""" +function validate_config(config::AnalysisConfigStruct, + available_metadata_columns::Union{Vector{String},Nothing}=nothing; + strict::Bool=true)::Vector{String} + errors = String[] + + # Check formula references + # Extract tokens from formula: split by ~ + * : etc. + formula_tokens = _extract_formula_vars(config.formula) + for tok in formula_tokens + # tok should be in metadata_columns or outcome_column + if !(tok in config.metadata_columns) && (isnothing(config.outcome_column) || tok != config.outcome_column) && tok != "1" && tok != "0" + push!(errors, "Formula references variable '$tok' which is not listed in metadata_columns $(config.metadata_columns) nor outcome_column $(config.outcome_column). Refusing as meaningless. Did you forget to include it in metadata_columns?") + end + end + + if !isnothing(available_metadata_columns) + avail_set = Set(available_metadata_columns) + for col in config.metadata_columns + col in avail_set || push!(errors, "metadata_columns contains '$col' which does not exist in available metadata columns $(available_metadata_columns).") + end + if !isnothing(config.outcome_column) + config.outcome_column in avail_set || push!(errors, "outcome_column '$(config.outcome_column)' does not exist in available metadata columns.") + end + for tok in formula_tokens + if tok != "1" && tok != "0" && !(tok in avail_set) && !(tok in config.metadata_columns) && (isnothing(config.outcome_column) || tok != config.outcome_column) + push!(errors, "Formula variable '$tok' not found in available metadata columns $(available_metadata_columns).") + end + end + end + + # Method-specific meaningless checks + if config.method == LOGISTIC + if isnothing(config.outcome_column) + push!(errors, "LOGISTIC requires binary outcome_column, but none provided.") + end + # Check that formula has outcome on LHS for logistic? Should be "outcome ~ ..." + if !occursin(r"^\s*[a-zA-Z0-9_\.]+\s*~", config.formula) + push!(errors, "For LOGISTIC, formula should be of form 'outcome ~ predictors' (outcome on left of ~). Got '$(config.formula)'. Refusing ambiguous formula.") + end + end + + if config.method in (CLR_LM, ILR_LM) + if config.normalization.pseudocount <= 0 + push!(errors, "For compositional methods CLR/ILR, pseudocount must be >0 to handle zeros. Got $(config.normalization.pseudocount).") + end + if config.advanced.zero_handling == "refuse" + push!(errors, "DANGER: zero_handling='refuse' with CLR/ILR will cause log(0) = -Inf and break the analysis. This combination is refused even with acknowledgment, because it is mathematically invalid.") + end + end + + if config.method == NB_GLM + if config.normalization.method in ("clr", "ilr") + push!(errors, "NB_GLM expects count data, not compositional transforms CLR/ILR. Use CLR_LM/ILR_LM for compositional, or change normalization to none/size_factors.") + end + end + + # Prevalence / abundance sanity + if config.advanced.min_prevalence == 1.0 && config.advanced.min_abundance > 0 + push!(errors, "min_prevalence=1.0 with min_abundance>0 requires feature to be present in 100% of samples above abundance threshold — this will likely filter everything. Refusing as overly stringent; if intentional, set min_prevalence=0.99.") + end + + if config.advanced.min_prevalence == 0.0 && config.advanced.min_abundance == 0.0 && !isnothing(config.advanced.max_features) && config.advanced.max_features < 10 + @warn "Filtering none (prevalence 0, abundance 0) but max_features=$(config.advanced.max_features) is very low — you will arbitrarily truncate features. Consider raising max_features or adding prevalence filter." + end + + if strict && !isempty(errors) + throw(ArgumentError("AnalysisConfig validation failed:\n" * join(errors, "\n"))) + end + + return errors +end + +function _extract_formula_vars(formula::String)::Vector{String} + # Remove ~, +, *, :, (, ), spaces, then split + # This is simplified; for full R parsing we'd need R, but we do lexical extraction + cleaned = replace(formula, "~" => " ", "+" => " ", "*" => " ", ":" => " ", "(" => " ", ")" => " ", "/" => " ") + tokens = split(cleaned) + # Filter out numeric constants and known R functions + r_keywords = Set(["1", "0", "I", "log", "sqrt", "scale", "factor", "as.factor", "as.numeric"]) + filter(t -> !(t in r_keywords) && !all(isdigit, t) && !isempty(t), tokens) |> unique +end + +# -------------------------------------------------------------------------- +# Context-sensitive help +# -------------------------------------------------------------------------- + +""" + context_help(field_path) -> String + +Returns help text for a given field, with scientific context. +""" +function context_help(field_path::String)::String + help_db = Dict{String,String}( + "method" => """ + **Analysis Method** (required, explicit, no auto-selection) + + - `nb_glm`: Negative Binomial GLM, appropriate for raw counts with overdispersion. Uses DESeq2-style size factors or MASS::glm.nb. Best for differential abundance of individual taxa when library sizes vary. Requires at least 3 samples per group. + - `clr_lm`: Centered Log-Ratio transform + Gaussian LM. Compositional method (Aitchison geometry). Handles compositionality but requires pseudocount for zeros. Use when you care about relative shifts, not absolute counts. + - `ilr_lm`: Isometric Log-Ratio + Gaussian LM. Like CLR but with orthonormal basis (balances). Allows phylogenetic or sequential binary partition basis. More interpretable for hierarchical hypotheses. + - `logistic`: Logistic regression for presence/absence or binary outcome. Outcome must be binary. Use when you dichotomize (e.g., pathogen present/absent). + + No silent switching: you must choose one. Changing method changes the statistical model and interpretation. + """, + "formula" => """ + **Formula** (R-style, required) + + Example: `~ group` or `disease ~ group + batch + age` + + - Left of `~` is outcome (required for logistic, optional for others — if omitted, each taxon is tested independently). + - Right of `~` lists metadata columns as covariates. + - Must reference only columns listed in metadata_columns. + - Forbidden: `;`, backticks, `$` (injection prevention). + - Must contain at least one covariate. + + Context: If you have batch effects, include batch: `~ group + batch`. If you don't, your p-values may be confounded. + """, + "normalization.method" => """ + **Normalization / Transform** (method-dependent) + + - For NB_GLM: `none` (use raw counts with size_factors), `size_factors` (DESeq2), `relative` (proportions), `rarefy` (subsample — discouraged for differential abundance, but allowed with acknowledgment). + - For CLR_LM: must be `clr`. Pseudocount mandatory. + - For ILR_LM: must be `ilr`. Pseudocount + ilr_basis. + - For LOGISTIC: `presence_absence`, `none`, `relative`. + + Refusal: CLR/ILR without pseudocount is mathematically invalid (log(0)). We refuse it. + """, + "normalization.pseudocount" => """ + **Pseudocount** for zero replacement (CLR/ILR only) + + Typical: 0.5 or 0.65 (Martín-Fernández et al. 2003). Must be >0, <1 recommended. + + - Too small (e.g., 1e-6): creates extreme log-ratios, inflates variance. + - Too large (e.g., >=1): distorts low-abundance features. + - Zero: refused (log(0) undefined). + + For NB_GLM, zero handling is via dispersion estimation, not pseudocount. + """, + "correction.method" => """ + **Multiple testing correction** — BH mandatory in v1 + + Microbiome data tests thousands of taxa. Uncorrected p-values will give ~5% false positives even under null. + + - `BH` (Benjamini-Hochberg FDR) is mandatory. + - Any override (e.g., `none`, `bonferroni`) requires Advanced Analysis expander, DANGER banner, and acknowledgment token `$DANGER_ACK_TOKEN`. + - Override is logged in provenance and DOI bundle. + + Scientific value: BH controls false discovery rate, appropriate for exploratory microbiome studies. + """, + "advanced.min_prevalence" => """ + **Minimum prevalence** filter (0-1) + + Feature must be present in at least this fraction of samples to be tested. + + - 0.1 = present in >=10% samples. Recommended to reduce multiple testing burden. + - 0 = no filter (tests everything, more multiple testing, slower). + - 1 = present in 100% samples (very stringent, likely filters everything unless abundance threshold is 0). + + Refusal: values outside [0,1] are meaningless. + """, + "advanced.dispersion_method" => """ + **Dispersion estimation** (NB_GLM advanced) + + - `parametric`: fit dispersion ~ mean trend (DESeq2 default). Good for large n. + - `local`: local regression fit. More flexible. + - `mean`: use mean dispersion. + - `pooled`: pool across genes (when n small). + - `glmGamPoi`: use glmGamPoi fast estimator. + + Context: Dispersion = extra variance beyond Poisson. Mis-specifying inflates false positives/negatives. + """, + ) + return get(help_db, field_path, "No help available for '$field_path'. This field may be advanced or undocumented — please check documentation or file an issue.") +end + +# -------------------------------------------------------------------------- +# DANGER banner logic +# -------------------------------------------------------------------------- + +function is_dangerous(config::AnalysisConfigStruct)::Bool + # Any override that weakens statistical rigor + config.correction.allow_no_correction && return true + config.advanced.zero_handling == "refuse" && return true + config.normalization.method == "rarefy" && config.method == NB_GLM && return true # rarefying for NB_GLM is discouraged + config.advanced.min_samples_per_group < 3 && return true + return false +end + +function danger_banner(config::AnalysisConfigStruct)::Union{String,Nothing} + is_dangerous(config) || return nothing + + reasons = String[] + if config.correction.allow_no_correction + push!(reasons, "BH correction disabled (method=$(config.correction.method)). This will inflate false discoveries in high-dimensional data.") + end + if config.advanced.zero_handling == "refuse" + push!(reasons, "zero_handling='refuse' will cause log(0) or biased zero handling.") + end + if config.normalization.method == "rarefy" && config.method == NB_GLM + push!(reasons, "Rarefaction for NB_GLM discards data and reduces power; size_factors preferred (McMurdie & Holmes 2014).") + end + if config.advanced.min_samples_per_group < 3 + push!(reasons, "min_samples_per_group=$(config.advanced.min_samples_per_group) <3: variance estimation will be unstable.") + end + + banner = """ + ╔════════════════════════════════════════════════════════════════════════════╗ + ║ ⚠️ DANGER — SCIENTIFICALLY RISKY CONFIGURATION DETECTED ⚠️ ║ + ╠════════════════════════════════════════════════════════════════════════════╣ + ║ You have enabled overrides that weaken statistical rigor: ║ + $(join(["║ - $r" for r in reasons], "\n")) + ║ ║ + ║ This configuration will be: ║ + ║ • Logged in provenance with full user identity and timestamp ║ + ║ • Bannered in every figure and DOI bundle ║ + ║ • Flagged in the GitHub Project board as 'needs-review' ║ + ║ ║ + ║ If you are sure, you must acknowledge with token: ║ + ║ $DANGER_ACK_TOKEN + ║ ║ + ║ Consider: Is there a safer alternative? Consult context-sensitive help. ║ + ╚════════════════════════════════════════════════════════════════════════════╝ + """ + return banner +end + +# -------------------------------------------------------------------------- +# Serialization: JSON, Nickel, DEED +# -------------------------------------------------------------------------- + +function canonical_json(config::AnalysisConfigStruct)::String + dict = OrderedDict{String,Any}( + "schema_version" => config.schema_version, + "id" => config.id, + "created_at" => string(config.created_at), + "created_by" => config.created_by, + "method" => METHOD_TO_STRING[config.method], + "formula" => config.formula, + "outcome_column" => config.outcome_column, + "metadata_columns" => config.metadata_columns, + "normalization" => OrderedDict( + "method" => config.normalization.method, + "pseudocount" => config.normalization.pseudocount, + "ilr_basis" => config.normalization.ilr_basis, + "multiplicative_replacement_delta" => config.normalization.multiplicative_replacement_delta, + ), + "correction" => OrderedDict( + "method" => config.correction.method, + "alpha" => config.correction.alpha, + "allow_no_correction" => config.correction.allow_no_correction, + ), + "advanced" => OrderedDict( + "dispersion_method" => config.advanced.dispersion_method, + "zero_handling" => config.advanced.zero_handling, + "min_prevalence" => config.advanced.min_prevalence, + "min_abundance" => config.advanced.min_abundance, + "max_features" => config.advanced.max_features, + "min_samples_per_group" => config.advanced.min_samples_per_group, + "robust" => config.advanced.robust, + ), + "provenance" => config.provenance, + "hash" => config.hash, + ) + return JSON3.write(dict) +end + +function config_hash(config::AnalysisConfigStruct)::String + return config.hash +end + +function to_json(config::AnalysisConfigStruct)::String + return canonical_json(config) +end + +function from_json(json_str::String)::AnalysisConfigStruct + data = JSON3.read(json_str, Dict{String,Any}) + + method_str = String(data["method"]) + norm_data = data["normalization"] isa Dict ? data["normalization"] : Dict{String,Any}() + corr_data = data["correction"] isa Dict ? data["correction"] : Dict{String,Any}() + adv_data = data["advanced"] isa Dict ? data["advanced"] : Dict{String,Any}() + + normalization = NormalizationConfig( + method=String(get(norm_data, "method", "none")), + pseudocount=Float64(get(norm_data, "pseudocount", 0.5)), + ilr_basis=get(norm_data, "ilr_basis", nothing) isa Nothing ? nothing : String(get(norm_data, "ilr_basis", nothing)), + multiplicative_replacement_delta=get(norm_data, "multiplicative_replacement_delta", nothing) isa Nothing ? nothing : Float64(get(norm_data, "multiplicative_replacement_delta", nothing)), + ) + + correction = CorrectionConfig( + method=String(get(corr_data, "method", "BH")), + alpha=Float64(get(corr_data, "alpha", 0.05)), + allow_no_correction=Bool(get(corr_data, "allow_no_correction", false)), + acknowledgment_token=get(corr_data, "acknowledgment_token", nothing) isa Nothing ? nothing : String(get(corr_data, "acknowledgment_token", nothing)), + ) + + advanced = AdvancedOverrides( + dispersion_method=String(get(adv_data, "dispersion_method", "parametric")), + zero_handling=String(get(adv_data, "zero_handling", "pseudocount")), + min_prevalence=Float64(get(adv_data, "min_prevalence", 0.1)), + min_abundance=Float64(get(adv_data, "min_abundance", 0.0)), + max_features=get(adv_data, "max_features", nothing) isa Nothing ? nothing : Int(get(adv_data, "max_features", nothing)), + min_samples_per_group=Int(get(adv_data, "min_samples_per_group", 3)), + robust=Bool(get(adv_data, "robust", false)), + acknowledgment_token=get(adv_data, "acknowledgment_token", nothing) isa Nothing ? nothing : String(get(adv_data, "acknowledgment_token", nothing)), + ) + + return AnalysisConfigStruct( + schema_version=String(get(data, "schema_version", SCHEMA_VERSION)), + id=String(get(data, "id", string(uuid4()))), + created_at=try DateTime(String(get(data, "created_at", string(now(UTC))))) catch; now(UTC) end, + created_by=String(get(data, "created_by", "anonymous")), + method=method_str, + formula=String(data["formula"]), + outcome_column=get(data, "outcome_column", nothing) isa Nothing ? nothing : String(get(data, "outcome_column", nothing)), + metadata_columns=Vector{String}(String.(get(data, "metadata_columns", String[]))), + normalization=normalization, + correction=correction, + advanced=advanced, + provenance=OrderedDict{String,Any}(get(data, "provenance", OrderedDict{String,Any}())), + hash=String(get(data, "hash", "")), + ) +end + +function to_nickel(config::AnalysisConfigStruct)::String + # Nickel contract from hyperpolymath/standards style + """ + # SPDX-License-Identifier: AGPL-3.0-only + # AnalysisConfig Nickel contract — generated from $(config.id) + # Schema version: $(config.schema_version) + # Hash: $(config.hash) + # DANGER: $(is_dangerous(config) ? "YES - " * something(danger_banner(config), "")[1:100] : "NO") + + let AnalysisMethod = std.enum.TagOrString & [| 'nb_glm, 'clr_lm, 'ilr_lm, 'logistic |] in + let CorrectionMethod = std.enum.TagOrString & [| 'BH, 'FDR |] in + { + schema_version | String | doc "Must be $(SCHEMA_VERSION)" = "$(config.schema_version)", + id | String = "$(config.id)", + created_at | String = "$(config.created_at)", + created_by | String = "$(config.created_by)", + method | AnalysisMethod = '$(METHOD_TO_STRING[config.method])', + formula | String | doc $(context_help("formula") |> s -> "\"\"\"$(replace(s, "\"" => "\\\""))\"\"\"") = "$(config.formula)", + outcome_column | std.option.String = $(isnothing(config.outcome_column) ? "null" : "\"$(config.outcome_column)\""), + metadata_columns | Array String = $(JSON3.write(config.metadata_columns)), + + normalization = { + method | String = "$(config.normalization.method)", + pseudocount | Number | doc "Must be >0 for CLR/ILR" = $(config.normalization.pseudocount), + ilr_basis | std.option.String = $(isnothing(config.normalization.ilr_basis) ? "null" : "\"$(config.normalization.ilr_basis)\""), + }, + + correction = { + method | CorrectionMethod | doc "BH mandatory in v1" = '$(lowercase(config.correction.method))', + alpha | Number | doc "FDR threshold, (0,1)" = $(config.correction.alpha), + allow_no_correction | Bool = $(config.correction.allow_no_correction), + } | std.contract.custom (fun label value => + if value.allow_no_correction && value.method != 'BH then + if value.acknowledgment_token != "$DANGER_ACK_TOKEN" then + 'Error { message = "DANGER: BH override requires acknowledgment token" } + else + 'Ok value + else + 'Ok value + ), + + advanced = { + dispersion_method | String = "$(config.advanced.dispersion_method)", + zero_handling | String = "$(config.advanced.zero_handling)", + min_prevalence | Number = $(config.advanced.min_prevalence), + min_abundance | Number = $(config.advanced.min_abundance), + max_features | std.option.Number = $(isnothing(config.advanced.max_features) ? "null" : string(config.advanced.max_features)), + min_samples_per_group | Number = $(config.advanced.min_samples_per_group), + robust | Bool = $(config.advanced.robust), + }, + + provenance | { .. } = $(JSON3.write(config.provenance)), + hash | String = "$(config.hash)", + } + """ +end + +function to_deed(config::AnalysisConfigStruct)::String + # DEED format from hyperpolymath/standards 1-formats/deed + # Filename dispatch: *_chora.deed, head repo-deed, :schema-version first + banner = is_dangerous(config) ? ";; DANGER: $(replace(something(danger_banner(config), "")[1:200], "\n" => " "))" : ";; Safe configuration (BH enforced)" + """ + ;; SPDX-FileCopyrightText: © 2026 Jonathan D.A. Jewell (hyperpolymath) + ;; SPDX-License-Identifier: AGPL-3.0-only + ;; AnalysisConfig DEED — $(config.id) — $(config.hash) + $banner + (repo-deed + :schema-version "$(config.schema_version)" + :canonical-name "analysis-config-$(config.id)" + :beholding-chora #u5"estate/chora" + + (method + :name "$(METHOD_TO_STRING[config.method])" + :formula "$(config.formula)" + :outcome-column $(isnothing(config.outcome_column) ? "\"\"" : "\"$(config.outcome_column)\"") + :metadata-columns ($(join(["\"$c\"" for c in config.metadata_columns], " ")))) + + (normalization + :method "$(config.normalization.method)" + :pseudocount $(config.normalization.pseudocount) + :ilr-basis $(isnothing(config.normalization.ilr_basis) ? "\"\"" : "\"$(config.normalization.ilr_basis)\"")) + + (correction + :method "$(config.correction.method)" + :alpha $(config.correction.alpha) + :allow-no-correction $(config.correction.allow_no_correction ? "#t" : "#f")) + + (advanced + :dispersion-method "$(config.advanced.dispersion_method)" + :zero-handling "$(config.advanced.zero_handling)" + :min-prevalence $(config.advanced.min_prevalence) + :min-abundance $(config.advanced.min_abundance) + :max-features $(isnothing(config.advanced.max_features) ? "0" : string(config.advanced.max_features)) + :min-samples-per-group $(config.advanced.min_samples_per_group) + :robust $(config.advanced.robust ? "#t" : "#f")) + + (provenance + :id "$(config.id)" + :hash "$(config.hash)" + :created-at "$(config.created_at)" + :created-by "$(config.created_by)" + :dangerous $(is_dangerous(config) ? "#t" : "#f")) + + (warrant + :evidence-type "AnalysisConfig" + :soundness "Requires BH unless DANGER token provided" + :fiber "Echo of raw counts through $(config.normalization.method) transform")) + """ +end + +# -------------------------------------------------------------------------- +# AnalysisResult — immutable derived object +# -------------------------------------------------------------------------- + +""" + AnalysisResult — immutable, provenance-rich derived object + +Every analysis run produces this, with hash chaining to its config. +""" +struct AnalysisResult + id::String + config_id::String + config_hash::String + created_at::DateTime + method::AnalysisMethod + results::OrderedDict{String,Any} # taxon -> stats + provenance::OrderedDict{String,Any} + hash::String + + function AnalysisResult(; id::String=string(uuid4()), + config_id::String, + config_hash::String, + created_at::DateTime=now(UTC), + method::AnalysisMethod, + results::OrderedDict{String,Any}, + provenance::OrderedDict{String,Any}=OrderedDict{String,Any}(), + hash::String="") + prov = OrderedDict{String,Any}(provenance) + prov["config_id"] = config_id + prov["config_hash"] = config_hash + prov["created_at"] = string(created_at) + prov["method"] = METHOD_TO_STRING[method] + + # Hash chain: result hash includes config hash + canonical = JSON3.write(OrderedDict( + "id" => id, + "config_id" => config_id, + "config_hash" => config_hash, + "created_at" => string(created_at), + "method" => METHOD_TO_STRING[method], + "results" => results, + )) + computed_hash = isempty(hash) ? bytes2hex(sha256(canonical)) : hash + + new(id, config_id, config_hash, created_at, method, results, prov, computed_hash) + end +end + +# -------------------------------------------------------------------------- +# DOI-ready bundle +# -------------------------------------------------------------------------- + +""" + create_doi_bundle(config, result, output_dir; authors, license, title) -> String + +Creates a DOI-ready bundle with DataCite metadata, config, result, provenance. +Returns path to bundle directory. +""" +function create_doi_bundle(config::AnalysisConfigStruct, + result::Union{AnalysisResult,Nothing}, + output_dir::String; + authors::Vector{String}=String[], + license::String="CC-BY-4.0", + title::String="MetaManifold Analysis Bundle", + description::String="Differential abundance analysis with $(METHOD_TO_STRING[config.method])")::String + + mkpath(output_dir) + + # DataCite metadata + datacite = OrderedDict{String,Any}( + "id" => config.id, + "type" => "Dataset", + "titles" => [OrderedDict("title" => title)], + "creators" => [OrderedDict("name" => a) for a in authors], + "publicationYear" => string(year(now())), + "descriptions" => [OrderedDict("description" => description, "descriptionType" => "Abstract")], + "formats" => ["application/json", "text/vnd.deed", "application/nickel"], + "version" => config.schema_version, + "rightsList" => [OrderedDict("rights" => license)], + "subjects" => [OrderedDict("subject" => METHOD_TO_STRING[config.method])], + "relatedIdentifiers" => [ + OrderedDict("relatedIdentifier" => config.hash, "relatedIdentifierType" => "SHA256", "relationType" => "IsDerivedFrom"), + ], + ) + + if is_dangerous(config) + datacite["descriptions"] = vcat(datacite["descriptions"], [OrderedDict("description" => something(danger_banner(config), ""), "descriptionType" => "TechnicalInfo")]) + end + + # Write files + open(joinpath(output_dir, "analysis_config.json"), "w") do io + write(io, to_json(config)) + end + open(joinpath(output_dir, "analysis_config.ncl"), "w") do io + write(io, to_nickel(config)) + end + open(joinpath(output_dir, "analysis_config_chora.deed"), "w") do io + write(io, to_deed(config)) + end + open(joinpath(output_dir, "datacite.json"), "w") do io + write(io, JSON3.write(datacite)) + end + open(joinpath(output_dir, "provenance.json"), "w") do io + write(io, JSON3.write(config.provenance)) + end + if !isnothing(result) + open(joinpath(output_dir, "analysis_result.json"), "w") do io + write(io, JSON3.write(OrderedDict( + "id" => result.id, + "config_id" => result.config_id, + "config_hash" => result.config_hash, + "created_at" => string(result.created_at), + "method" => METHOD_TO_STRING[result.method], + "results" => result.results, + "provenance" => result.provenance, + "hash" => result.hash, + ))) + end + end + open(joinpath(output_dir, "README.md"), "w") do io + write(io, """ + # $title + + DOI-ready bundle for MetaManifold analysis. + + - **Method**: $(METHOD_TO_STRING[config.method]) + - **Formula**: $(config.formula) + - **Config ID**: $(config.id) + - **Config Hash**: $(config.hash) + - **Created**: $(config.created_at) by $(config.created_by) + - **Schema Version**: $(config.schema_version) + - **Dangerous**: $(is_dangerous(config) ? "YES" : "NO") + + ## Files + + - `analysis_config.json` — canonical JSON (hash: $(config.hash)) + - `analysis_config.ncl` — Nickel contract + - `analysis_config_chora.deed` — DEED attestation (repo-deed) + - `datacite.json` — DataCite metadata for DOI registration + - `provenance.json` — full provenance chain + $(isnothing(result) ? "" : "- `analysis_result.json` — immutable result with hash chain to config") + + ## Reproducibility + + This bundle is self-contained and content-addressed. The config hash $(config.hash) is SHA256 of canonical JSON. + To verify: `sha256sum analysis_config.json` should match provenance. + + ## License + + $license + + $(is_dangerous(config) ? something(danger_banner(config), "") : "") + """) + end + + return abspath(output_dir) +end + +# -------------------------------------------------------------------------- +# Epistemic bridge — present_in_every_admissible_world +# -------------------------------------------------------------------------- + +""" + present_in_every_admissible_world(taxon_counts, evidence; threshold=1) -> Bool + +Implements the residual-evidence notion: a taxon is present in every admissible world +consistent with observation and evidence. + +In the finite model: observation = sum of contributions, evidence = bounds on noise, +candidate worlds = all (u,n) consistent with observation and evidence. + +For microbiome: observation = observed count, evidence = avec_fibre + epistemic status, +admissible worlds = all decompositions of observed count into true signal + noise +that satisfy evidence constraints. + +Simplified: if avec_fibre=true and count >= threshold in all filtered views, then present in every admissible world. +""" +function present_in_every_admissible_world(observed_counts::Vector{Float64}, + evidence::Dict{String,Any}; + threshold::Float64=1.0)::Bool + # evidence should contain avec_fibre, epistemic_status, noise_bound, etc. + avec_fibre = get(evidence, "avec_fibre", false) == true + epistemic_status = get(evidence, "epistemic_status", "unknown") + + # If sans fibre, we cannot claim presence in every world + !avec_fibre && return false + + # If epistemic status is already present_in_every_admissible_world, trust it + epistemic_status == "present_in_every_admissible_world" && return true + epistemic_status == "absent_in_every_admissible_world" && return false + + # Finite candidate model: check all counts >= threshold? + # In residual-evidence-types, Holds Present means Present holds for every candidate + all(c -> c >= threshold, observed_counts) && return true + + return false +end + +end # module AnalysisConfig diff --git a/src/analysis/clade_cumulus.jl b/src/analysis/clade_cumulus.jl new file mode 100644 index 0000000..90924ea --- /dev/null +++ b/src/analysis/clade_cumulus.jl @@ -0,0 +1,510 @@ +# SPDX-License-Identifier: AGPL-3.0-only +""" + CladeCumulus — Cumulative Cladistic Explorer + +Implements the second feature: tree with cumulative frequencies, epistemic colour coding, +cloud sizing by residual count, drag-and-drop with live present_in_every_admissible_world validation, +Evidence Mode toggle, clean non-cluttered UI. + +Backend part: cumulative frequencies over cladistic tree, epistemic colour coding, cloud sizing. + +Frontend part will be in `frontend/src/components/CladeCumulus.tsx`. + +This file is scaffolded on feat/analysis-config-v1 branch but will be fully implemented +on feat/clade-cumulus branch (sequential task). For now, it provides the data structures +and validation logic that AnalysisConfig can already use. + +Rules: +- Keep UI clean. Advanced options and cladistic visuals only appear when Evidence Mode enabled. +- Drag-and-drop with live present_in_every_admissible_world validation +- Evidence Mode toggle +""" +module CladeCumulus + +using OrderedCollections +using ..Epistemic: EchoFiber, avec_fibre, present_in_every_admissible_world, + epistemic_colour, cloud_size_by_residual, + AVEC_FIBRE_COLUMN, EPISTEMIC_STATUSES + +export CladeNode, CladeTree, CumulativeFrequency, + build_clade_tree, cumulative_frequencies, + validate_drag_drop, epistemic_colour_for_node, cloud_size_for_node, + to_plotly_tree, to_json + +# -------------------------------------------------------------------------- +# Data structures +# -------------------------------------------------------------------------- + +""" + CladeNode — node in cladistic tree + +- `id`: unique id (e.g., taxon name or rank:value) +- `label`: display label +- `rank`: taxonomic rank (Domain, Phylum, etc.) +- `parent_id`: parent node id, nothing for root +- `children_ids`: child node ids +- `count`: raw count (sum of sequences) +- `cumulative_count`: cumulative count including children (computed) +- `cumulative_frequency`: cumulative frequency (count / total) +- `residual_count`: number of residual evidence candidates (for cloud sizing) +- `avec_fibre`: bool +- `epistemic_status`: one of EPISTEMIC_STATUSES +- `metadata`: extra dict +""" +struct CladeNode + id::String + label::String + rank::String + parent_id::Union{String,Nothing} + children_ids::Vector{String} + count::Float64 + cumulative_count::Float64 + cumulative_frequency::Float64 + residual_count::Int + avec_fibre::Bool + epistemic_status::String + metadata::OrderedDict{String,Any} + + function CladeNode(; id::String, + label::String=id, + rank::String="unknown", + parent_id::Union{String,Nothing}=nothing, + children_ids::Vector{String}=String[], + count::Float64=0.0, + cumulative_count::Float64=count, + cumulative_frequency::Float64=0.0, + residual_count::Int=0, + avec_fibre::Bool=false, + epistemic_status::String="unknown", + metadata::OrderedDict{String,Any}=OrderedDict{String,Any}()) + + epistemic_status in EPISTEMIC_STATUSES || + throw(ArgumentError("epistemic_status must be one of $(join(EPISTEMIC_STATUSES, ", ")) (got '$epistemic_status')")) + + new(id, label, rank, parent_id, children_ids, count, cumulative_count, cumulative_frequency, residual_count, avec_fibre, epistemic_status, metadata) + end +end + +""" + CladeTree — whole tree + +- `nodes`: id -> CladeNode +- `root_id`: root node id +- `total_count`: total sequences +""" +struct CladeTree + nodes::OrderedDict{String,CladeNode} + root_id::String + total_count::Float64 + + function CladeTree(nodes::OrderedDict{String,CladeNode}, root_id::String) + haskey(nodes, root_id) || throw(ArgumentError("root_id '$root_id' not in nodes")) + total = sum(n.count for n in values(nodes) if isnothing(n.parent_id) || n.parent_id == root_id || true; init=0.0) + # Actually total should be root's cumulative count + root = nodes[root_id] + total_count = root.cumulative_count > 0 ? root.cumulative_count : sum(n.count for n in values(nodes); init=0.0) + new(nodes, root_id, total_count) + end +end + +struct CumulativeFrequency + node_id::String + cumulative_count::Float64 + cumulative_frequency::Float64 + depth::Int +end + +# -------------------------------------------------------------------------- +# Tree building +# -------------------------------------------------------------------------- + +""" + build_clade_tree(taxa_table; rank_hierarchy, count_column="total") -> CladeTree + +Builds a cladistic tree from a taxa table (DataFrame or vector of dicts). +Each row should have rank columns (Domain, Phylum, ..., Species) and a count column. + +Cumulative frequencies are computed bottom-up: each node's cumulative_count = its own count + sum(children cumulative_counts) + +Epistemic colour coding: based on avec_fibre and present_in_every_admissible_world +Cloud sizing: by residual_count (number of candidate worlds or residual evidence) +""" +function build_clade_tree(taxa_rows::Vector{Dict{String,Any}}; + rank_hierarchy::Vector{String}=["Domain", "Phylum", "Class", "Order", "Family", "Genus", "Species"], + count_column::String="total", + avec_fibre_column::String=AVEC_FIBRE_COLUMN)::CladeTree + + nodes = OrderedDict{String,CladeNode}() + # Root node + root_id = "root" + nodes[root_id] = CladeNode(id=root_id, label="Root", rank="root", parent_id=nothing, count=0.0) + + # Build nodes for each taxon path + for row in taxa_rows + count = Float64(get(row, count_column, 0.0)) + avec = Bool(get(row, avec_fibre_column, false)) + residual = Int(get(row, "residual_count", get(row, "residual", 0))) + epistemic_status = String(get(row, "epistemic_status", avec ? "present_in_some_admissible_world" : "unknown")) + + # Build path from root through ranks + parent_id = root_id + for (depth, rank) in enumerate(rank_hierarchy) + rank_value = get(row, rank, nothing) + isnothing(rank_value) && continue + rank_value_str = String(rank_value) + isempty(strip(rank_value_str)) && continue + rank_value_str in ("Unclassified", "unknown", "") && continue + + node_id = "$(parent_id)|$(rank):$(rank_value_str)" + if !haskey(nodes, node_id) + nodes[node_id] = CladeNode( + id=node_id, + label=rank_value_str, + rank=rank, + parent_id=parent_id, + count=0.0, # internal nodes get count from children, not direct + residual_count=residual, + avec_fibre=avec, + epistemic_status=epistemic_status, + ) + # Add to parent's children + parent_node = nodes[parent_id] + if !(node_id in parent_node.children_ids) + # Need to update parent (immutable struct, so recreate) + new_children = vcat(parent_node.children_ids, [node_id]) + nodes[parent_id] = CladeNode( + id=parent_node.id, + label=parent_node.label, + rank=parent_node.rank, + parent_id=parent_node.parent_id, + children_ids=new_children, + count=parent_node.count, + cumulative_count=parent_node.cumulative_count, + cumulative_frequency=parent_node.cumulative_frequency, + residual_count=parent_node.residual_count, + avec_fibre=parent_node.avec_fibre, + epistemic_status=parent_node.epistemic_status, + metadata=parent_node.metadata, + ) + end + end + parent_id = node_id + end + + # Leaf node (species or finest rank) gets the count + # If parent_id is still root, create a direct leaf under root + leaf_id = parent_id == root_id ? "$(root_id)|leaf:$(get(row, "SeqName", string(hash(row))))" : parent_id + if haskey(nodes, leaf_id) + old = nodes[leaf_id] + nodes[leaf_id] = CladeNode( + id=old.id, + label=old.label, + rank=old.rank, + parent_id=old.parent_id, + children_ids=old.children_ids, + count=old.count + count, + cumulative_count=old.cumulative_count + count, + cumulative_frequency=old.cumulative_frequency, + residual_count=max(old.residual_count, residual), + avec_fibre=old.avec_fibre || avec, + epistemic_status=epistemic_status, # last wins, could be more sophisticated + metadata=old.metadata, + ) + else + nodes[leaf_id] = CladeNode( + id=leaf_id, + label=String(get(row, "Genus", get(row, "Species", "Unknown"))), + rank="leaf", + parent_id=root_id, + count=count, + cumulative_count=count, + residual_count=residual, + avec_fibre=avec, + epistemic_status=epistemic_status, + ) + # Add to root children + root_node = nodes[root_id] + new_children = vcat(root_node.children_ids, [leaf_id]) + nodes[root_id] = CladeNode( + id=root_node.id, + label=root_node.label, + rank=root_node.rank, + parent_id=root_node.parent_id, + children_ids=new_children, + count=root_node.count, + cumulative_count=root_node.cumulative_count, + cumulative_frequency=root_node.cumulative_frequency, + residual_count=root_node.residual_count, + avec_fibre=root_node.avec_fibre, + epistemic_status=root_node.epistemic_status, + metadata=root_node.metadata, + ) + end + end + + # Compute cumulative counts bottom-up + nodes = _compute_cumulative(nodes, root_id) + + return CladeTree(nodes, root_id) +end + +function _compute_cumulative(nodes::OrderedDict{String,CladeNode}, root_id::String)::OrderedDict{String,CladeNode} + # Post-order traversal: children first + # For simplicity, iterate in reverse topological order (leaves first) + # We need to compute cumulative_count = own count + sum(children cumulative_counts) + + # Build depth map + depth_map = Dict{String,Int}() + function compute_depth(id::String, depth::Int) + depth_map[id] = depth + node = nodes[id] + for child_id in node.children_ids + if haskey(nodes, child_id) + compute_depth(child_id, depth+1) + end + end + end + compute_depth(root_id, 0) + + # Sort nodes by depth descending (deepest first) + sorted_ids = sort(collect(keys(nodes)), by=id->get(depth_map, id, 0), rev=true) + + new_nodes = copy(nodes) + for id in sorted_ids + node = new_nodes[id] + # Sum children cumulative counts + children_cum = 0.0 + for child_id in node.children_ids + if haskey(new_nodes, child_id) + children_cum += new_nodes[child_id].cumulative_count + end + end + cum_count = node.count + children_cum + # Update node with cumulative_count + new_nodes[id] = CladeNode( + id=node.id, + label=node.label, + rank=node.rank, + parent_id=node.parent_id, + children_ids=node.children_ids, + count=node.count, + cumulative_count=cum_count, + cumulative_frequency=0.0, # will be set after total known + residual_count=node.residual_count, + avec_fibre=node.avec_fibre, + epistemic_status=node.epistemic_status, + metadata=node.metadata, + ) + end + + # Now set cumulative_frequency = cumulative_count / total + total = new_nodes[root_id].cumulative_count + total = total == 0 ? 1.0 : total + for id in keys(new_nodes) + node = new_nodes[id] + freq = node.cumulative_count / total + new_nodes[id] = CladeNode( + id=node.id, + label=node.label, + rank=node.rank, + parent_id=node.parent_id, + children_ids=node.children_ids, + count=node.count, + cumulative_count=node.cumulative_count, + cumulative_frequency=freq, + residual_count=node.residual_count, + avec_fibre=node.avec_fibre, + epistemic_status=node.epistemic_status, + metadata=node.metadata, + ) + end + + return new_nodes +end + +function cumulative_frequencies(tree::CladeTree)::Vector{CumulativeFrequency} + freqs = CumulativeFrequency[] + for (id, node) in tree.nodes + depth = _depth_of(tree, id) + push!(freqs, CumulativeFrequency(id, node.cumulative_count, node.cumulative_frequency, depth)) + end + sort!(freqs, by=f->f.depth) + return freqs +end + +function _depth_of(tree::CladeTree, node_id::String)::Int + depth = 0 + current_id = node_id + while true + node = get(tree.nodes, current_id, nothing) + isnothing(node) && break + isnothing(node.parent_id) && break + current_id = node.parent_id + depth += 1 + depth > 100 && break # prevent infinite loop + end + return depth +end + +# -------------------------------------------------------------------------- +# Validation: drag-and-drop with live present_in_every_admissible_world +# -------------------------------------------------------------------------- + +""" + validate_drag_drop(tree, dragged_node_id, target_parent_id, evidence) -> (Bool, String) + +Validates whether dragged_node can be dropped under target_parent. + +Uses present_in_every_admissible_world: a taxon can only be placed in a clade +if it is present in every admissible world consistent with evidence. + +Returns (is_valid, message) +""" +function validate_drag_drop(tree::CladeTree, + dragged_node_id::String, + target_parent_id::String, + evidence::Dict{String,Any})::Tuple{Bool,String} + + !haskey(tree.nodes, dragged_node_id) && return (false, "Dragged node '$dragged_node_id' not found") + !haskey(tree.nodes, target_parent_id) && return (false, "Target parent '$target_parent_id' not found") + + dragged = tree.nodes[dragged_node_id] + target = tree.nodes[target_parent_id] + + # Cannot drop root + dragged_node_id == tree.root_id && return (false, "Cannot move root") + + # Cannot drop onto itself or descendant (would create cycle) + if _is_descendant(tree, target_parent_id, dragged_node_id) + return (false, "Cannot drop a clade onto itself or its descendant (would create cycle)") + end + + # Epistemic validation: present_in_every_admissible_world + # For drag-and-drop to be valid, dragged taxon must be present in every admissible world + # under the new parent's evidence context + + # Build evidence for this check + check_evidence = Dict{String,Any}(evidence) + check_evidence["avec_fibre"] = dragged.avec_fibre + check_evidence["epistemic_status"] = dragged.epistemic_status + check_evidence["residual_count"] = dragged.residual_count + + # If sans fibre, cannot validate presence in every world + if !dragged.avec_fibre + return (false, "Cannot move '$(dragged.label)': sans fibre (no semantic fibre). Artefact does not carry enough evidence to support placement. Needs avec_fibre=true.") + end + + # Check epistemic status + if dragged.epistemic_status == "absent_in_every_admissible_world" + return (false, "Cannot move '$(dragged.label)': absent in every admissible world (epistemic status).") + end + + if dragged.epistemic_status == "unknown" + return (false, "Cannot move '$(dragged.label)': epistemic status unknown (sans fibre or insufficient evidence). Needs present_in_every_admissible_world.") + end + + # For strict validation, require present_in_every, not just some + if dragged.epistemic_status != "present_in_every_admissible_world" + # Allow with warning if present_in_some, but require explicit acknowledgment? + # For now, we allow present_in_some with warning, but not unknown/absent + # The task says live present_in_every_admissible_world validation, so we should require every + return (false, "Cannot move '$(dragged.label)': not present in every admissible world (status=$(dragged.epistemic_status)). Only taxa with status present_in_every_admissible_world can be placed. This is the residual-evidence validation.") + end + + # Additional check: cumulative frequency must make sense + # (e.g., cannot place a high-frequency clade under low-frequency parent if it would exceed 100%) + # For now, we allow but warn + + return (true, "Valid: '$(dragged.label)' can be placed under '$(target.label)' (present_in_every_admissible_world, avec_fibre=true)") +end + +function _is_descendant(tree::CladeTree, maybe_descendant_id::String, ancestor_id::String)::Bool + current_id = maybe_descendant_id + while true + current_id == ancestor_id && return true + node = get(tree.nodes, current_id, nothing) + isnothing(node) && return false + isnothing(node.parent_id) && return false + current_id = node.parent_id + # Prevent infinite loop + if current_id == maybe_descendant_id + return false + end + end +end + +# -------------------------------------------------------------------------- +# UI helpers: colour coding, cloud sizing +# -------------------------------------------------------------------------- + +function epistemic_colour_for_node(node::CladeNode)::String + return epistemic_colour(node.epistemic_status) +end + +function cloud_size_for_node(node::CladeNode)::Float64 + return cloud_size_by_residual(node.residual_count) +end + +# -------------------------------------------------------------------------- +# Plotly conversion (for frontend) +# -------------------------------------------------------------------------- + +function to_plotly_tree(tree::CladeTree)::Dict{String,Any} + # Convert tree to Plotly sunburst or treemap format + # For CladeCumulus: we want tree with cumulative frequencies, colour by epistemic, size by residual + + ids = String[] + labels = String[] + parents = String[] + values = Float64[] + colours = String[] + sizes = Float64[] + hover_texts = String[] + + for (id, node) in tree.nodes + push!(ids, id) + push!(labels, node.label) + push!(parents, isnothing(node.parent_id) ? "" : node.parent_id) + push!(values, node.cumulative_count) + push!(colours, epistemic_colour_for_node(node)) + push!(sizes, cloud_size_for_node(node)) + push!(hover_texts, "Rank: $(node.rank)
Count: $(node.count)
Cumulative: $(node.cumulative_count) ($(round(node.cumulative_frequency*100, digits=2))%)
Residual: $(node.residual_count)
avec_fibre: $(node.avec_fibre)
Epistemic: $(node.epistemic_status)") + end + + return Dict{String,Any}( + "type" => "sunburst", + "ids" => ids, + "labels" => labels, + "parents" => parents, + "values" => values, + "marker" => Dict("colors" => colours), + "hovertext" => hover_texts, + "branchvalues" => "total", + "cloud_sizes" => sizes, + ) +end + +function to_json(tree::CladeTree)::String + dict = Dict{String,Any}( + "root_id" => tree.root_id, + "total_count" => tree.total_count, + "nodes" => [Dict( + "id" => n.id, + "label" => n.label, + "rank" => n.rank, + "parent_id" => n.parent_id, + "children_ids" => n.children_ids, + "count" => n.count, + "cumulative_count" => n.cumulative_count, + "cumulative_frequency" => n.cumulative_frequency, + "residual_count" => n.residual_count, + "avec_fibre" => n.avec_fibre, + "epistemic_status" => n.epistemic_status, + "colour" => epistemic_colour_for_node(n), + "cloud_size" => cloud_size_for_node(n), + ) for n in values(tree.nodes)] + ) + return JSON3.write(dict) +end + +end # module CladeCumulus diff --git a/src/core/epistemic.jl b/src/core/epistemic.jl new file mode 100644 index 0000000..be2e86e --- /dev/null +++ b/src/core/epistemic.jl @@ -0,0 +1,352 @@ +# SPDX-License-Identifier: AGPL-3.0-only +""" + Epistemic — bridge to hyperpolymath/echo-types, epistemic-types, residual-evidence-types + +Implements the epistemic layer claimed to be already implemented but not found in main. +This module is additive and does not overwrite colleagues' work; it provides the interface +described in the task: + +- Echo + Epistemic + Residual Evidence with `avec_fibre` column +- `Epistemic.jl` (this file) +- `present_in_every_admissible_world` validation +- Warrant types, Echo fibers, Candidate worlds + +Based on Agda sources: +- echo-types: Echo f y := Σ (x:A), (f x ≡ y), total space Σ B (Echo f) ≃ A +- epistemic-types: Modality E κ A, FactiveModality with reflect, Warrant without soundness +- residual-evidence-types: Candidate, Holds, Identified, actual-world-sound + +For Julia, we provide finite, executable shadows of those types. +""" +module Epistemic + +using OrderedCollections + +export EchoFiber, Warrant, SoundWarrant, Candidate, Case, + avec_fibre, sans_fibre, + present_in_every_admissible_world, present_in_some_admissible_world, + absent_in_every_admissible_world, + EPISTEMIC_STATUSES, AVEC_FIBRE_COLUMN, + epistemic_colour, cloud_size_by_residual + +const AVEC_FIBRE_COLUMN = "avec_fibre" +const EPISTEMIC_STATUSES = ("present_in_every_admissible_world", + "present_in_some_admissible_world", + "absent_in_every_admissible_world", + "unknown") + +# -------------------------------------------------------------------------- +# Echo Types — finite shadow +# -------------------------------------------------------------------------- + +""" + EchoFiber{A,B} + +Finite shadow of `Echo f y := Σ (x:A), (f x ≡ y)`. + +- `value`: the y:B observed +- `witnesses`: vector of (x, proof that f(x)==y) — the fiber over y +- `residue`: structured loss witness (what was lost but constrained) + +In bioinformatics: f = observation function (e.g., sequencing + denoising), +A = true biological world (true abundances), B = observed table, +fiber = all true worlds compatible with observed. +""" +struct EchoFiber{A,B} + observed::B + witnesses::Vector{A} + residue::OrderedDict{String,Any} + + function EchoFiber{A,B}(observed::B, witnesses::Vector{A}, residue::OrderedDict{String,Any}=OrderedDict{String,Any}()) where {A,B} + new{A,B}(observed, witnesses, residue) + end +end + +EchoFiber(observed::B, witnesses::Vector{A}) where {A,B} = EchoFiber{A,B}(observed, witnesses) + +""" + avec_fibre(fiber) -> Bool + +True if artefact carries enough semantic fibre to support inferences. +From echo-types: avec_fibre = artefact retains structured constraint on possible origins. +""" +avec_fibre(fiber::EchoFiber)::Bool = !isempty(fiber.witnesses) + +""" + sans_fibre(fiber) -> Bool + +Opposite of avec_fibre: artefact is only valid target-side value, no origin structure retained. +""" +sans_fibre(fiber::EchoFiber)::Bool = !avec_fibre(fiber) + +# -------------------------------------------------------------------------- +# Epistemic Types — Warrant without soundness +# -------------------------------------------------------------------------- + +""" + Warrant{A} + +The type of evidence tokens for A, without assuming evidence is valid. +From epistemic-types: separates "I have a receipt for A" from "A is true". + +- `evidence_type`: what kind of evidence (e.g., "count >= threshold") +- `tokens`: vector of evidence tokens +- `claim`: the claim A purports to support (e.g., "taxon present") + +Having Warrant does NOT give A. Need SoundWarrant. +""" +struct Warrant{A} + evidence_type::String + tokens::Vector{Any} + claim::A +end + +""" + SoundWarrant{A} + +Warrant plus proof that evidence → A. +From epistemic-types: adds `sound : Evidence → A`. +""" +struct SoundWarrant{A} + warrant::Warrant{A} + sound::Function # Evidence -> A + + function SoundWarrant{A}(warrant::Warrant{A}, sound::Function) where A + new{A}(warrant, sound) + end +end + +function (sw::SoundWarrant{A})(token)::A where A + return sw.sound(token) +end + +# -------------------------------------------------------------------------- +# Residual Evidence Types — Candidate, Holds, Identified +# -------------------------------------------------------------------------- + +""" + Candidate{W,O} + +Ordinary evidence-refined preimage fibre: Σ W (observe(W)≡r × E(W)) + +- `world`: a possible world W +- `observed`: proof that observe(world) == r +- `evidence`: evidence that E(world) holds + +In microbiome: W = (true abundance, noise), observe = true+noise, E = noise bound + avec_fibre +""" +struct Candidate{W,O} + world::W + observed_equals::Bool + evidence_holds::Bool + metadata::OrderedDict{String,Any} + + function Candidate{W,O}(world::W, observed_equals::Bool, evidence_holds::Bool, metadata::OrderedDict{String,Any}=OrderedDict{String,Any}()) where {W,O} + new{W,O}(world, observed_equals, evidence_holds, metadata) + end +end + +""" + Case{W,O} + +An inhabited candidate set: witness that at least one candidate exists. +Claims still quantify over ALL candidates, not merely this inhabitant. +""" +struct Case{W,O} + witness::Candidate{W,O} + all_candidates::Vector{Candidate{W,O}} + + function Case{W,O}(witness::Candidate{W,O}, all_candidates::Vector{Candidate{W,O}}) where {W,O} + # Ensure witness is in all_candidates or add it + new{W,O}(witness, all_candidates) + end +end + +Case(witness::Candidate{W,O}) where {W,O} = Case{W,O}(witness, [witness]) + +# -------------------------------------------------------------------------- +# Core logic: present_in_every_admissible_world +# -------------------------------------------------------------------------- + +""" + present_in_every_admissible_world(case, present_fn) -> Bool + +Implements Holds Present: Present holds for every candidate in case. + +- `case`: Case with all admissible worlds (those satisfying observe==r and evidence) +- `present_fn`: W -> Bool, true if taxon present in that world + +Returns true iff present_fn holds for EVERY candidate (i.e., present in every admissible world). + +This is the key validation for CladeCumulus drag-and-drop: a taxon can only be +placed in a clade if it is present in every admissible world consistent with evidence. +Otherwise, placement would be unsound. + +Example from residual-evidence-types: + r = u + n = 2, with noise bound n <=1 + candidates: (0,2) and (2,0) initially — presence not provable + with bound n<=1: candidates (1,1) and (2,0) — presence IS provable (u !=0 in both) + but value not identified (u could be 1 or 2) +""" +function present_in_every_admissible_world(case::Case{W,O}, present_fn::Function)::Bool where {W,O} + # All candidates must satisfy evidence and observation (by construction) + # Check present_fn for every candidate + for cand in case.all_candidates + cand.evidence_holds || continue # only consider admissible worlds where evidence holds + cand.observed_equals || continue # and observation matches + if !present_fn(cand.world) + return false + end + end + # Also need at least one admissible candidate, otherwise vacuously true is unsound + admissible = filter(c -> c.evidence_holds && c.observed_equals, case.all_candidates) + isempty(admissible) && return false # no admissible world -> cannot claim presence (inconsistent case) + return true +end + +function present_in_some_admissible_world(case::Case{W,O}, present_fn::Function)::Bool where {W,O} + for cand in case.all_candidates + (cand.evidence_holds && cand.observed_equals) || continue + present_fn(cand.world) && return true + end + return false +end + +function absent_in_every_admissible_world(case::Case{W,O}, present_fn::Function)::Bool where {W,O} + return !present_in_some_admissible_world(case, present_fn) +end + +# -------------------------------------------------------------------------- +# Simplified API for microbiome counts (finite model) +# -------------------------------------------------------------------------- + +""" + present_in_every_admissible_world(counts, evidence; threshold=1.0) -> Bool + +Simplified finite model for microbiome: counts across samples/views, +evidence dict with avec_fibre, noise_bound, etc. + +- `counts`: Vector{Float64} observed counts across samples or candidate decompositions +- `evidence`: Dict with keys: + - "avec_fibre": Bool — does artefact carry semantic fibre? + - "epistemic_status": String — precomputed status if available + - "noise_bound": Float64 — max noise allowed + - "threshold": Float64 — presence threshold + +Returns true iff present in every admissible world. +""" +function present_in_every_admissible_world(counts::Vector{Float64}, + evidence::Dict{String,Any}; + threshold::Float64=1.0)::Bool + avec = get(evidence, "avec_fibre", false) == true + !avec && return false + + status = get(evidence, "epistemic_status", "unknown") + status == "present_in_every_admissible_world" && return true + status == "absent_in_every_admissible_world" && return false + + # Finite candidate model: all counts >= threshold and at least one admissible + # Admissible = count >=0 and satisfies noise bound if given + noise_bound = get(evidence, "noise_bound", nothing) + if !isnothing(noise_bound) + # Filter candidates that satisfy noise bound (simplified: count <= bound? Actually noise <= bound) + # For this simplified model, we assume counts already filtered + end + + isempty(counts) && return false + return all(c -> c >= threshold, counts) +end + +# -------------------------------------------------------------------------- +# Epistemic colour coding and cloud sizing (for CladeCumulus) +# -------------------------------------------------------------------------- + +""" + epistemic_colour(status) -> String + +Colour coding for epistemic status: +- present_in_every_admissible_world: green (solid evidence) +- present_in_some_admissible_world: yellow/orange (uncertain) +- absent_in_every_admissible_world: grey (absent) +- unknown: red or striped (no fibre) +""" +function epistemic_colour(status::String)::String + if status == "present_in_every_admissible_world" + return "#2e7d32" # green + elseif status == "present_in_some_admissible_world" + return "#f9a825" # yellow/orange + elseif status == "absent_in_every_admissible_world" + return "#9e9e9e" # grey + else + return "#c62828" # red for unknown / sans fibre + end +end + +""" + cloud_size_by_residual(residual_count) -> Float64 + +Cloud sizing by residual count: larger cloud = more residual evidence, +i.e., more candidate worlds or larger fiber. + +For CladeCumulus: cloud size ∝ log(1 + residual_count) +""" +function cloud_size_by_residual(residual_count::Int)::Float64 + return log(1 + residual_count) * 10.0 + 5.0 # base size 5, scaled by log +end + +function cloud_size_by_residual(residual_count::Float64)::Float64 + return log(1 + residual_count) * 10.0 + 5.0 +end + +# -------------------------------------------------------------------------- +# Integration with DuckDB: avec_fibre column handling +# -------------------------------------------------------------------------- + +""" + ensure_avec_fibre_column(con, table) -> Bool + +Ensures the avec_fibre column exists in DuckDB table, creating it if needed. +Returns true if column exists or was created. + +The column is Boolean, indicating whether row carries enough semantic fibre. +""" +function ensure_avec_fibre_column(con, table::String)::Bool + # This is a placeholder for actual DuckDB interaction; real implementation + # would use DBInterface.execute + # For now, return true as stub + return true +end + +""" + epistemic_status_for_row(row, evidence) -> String + +Computes epistemic status for a single row (taxon) given evidence. + +- If avec_fibre=false → unknown +- Else if present in every admissible world → present_in_every_admissible_world +- Else if present in some → present_in_some_admissible_world +- Else → absent_in_every_admissible_world +""" +function epistemic_status_for_row(row::Dict{String,Any}, evidence::Dict{String,Any})::String + avec = get(row, AVEC_FIBRE_COLUMN, get(evidence, "avec_fibre", false)) + !avec && return "unknown" + + # Simplified: check if count column >= threshold + count = get(row, "total", get(row, "count", 0.0)) + threshold = Float64(get(evidence, "threshold", 1.0)) + + if count >= threshold + # Need to check all candidate worlds — for now, if avec_fibre and count>=threshold, assume present_in_every? + # Actually need more evidence: if residual count is low, we can be confident + residual_count = get(evidence, "residual_count", 0) + if residual_count == 0 + return "present_in_every_admissible_world" + else + return "present_in_some_admissible_world" + end + else + return "absent_in_every_admissible_world" + end +end + +end # module Epistemic diff --git a/src/server/routes/analysis_config.jl b/src/server/routes/analysis_config.jl new file mode 100644 index 0000000..3cbab97 --- /dev/null +++ b/src/server/routes/analysis_config.jl @@ -0,0 +1,451 @@ +# SPDX-License-Identifier: AGPL-3.0-only +# Routes for AnalysisConfig layer — safe, explicit, versioned +# Implements: NB GLM, CLR/ILR+Gaussian LM, logistic in v1; BH mandatory; DANGER banner; DOI-ready bundles + +using OrderedCollections +using MetaManifold.AnalysisConfig +using MetaManifold.Epistemic +using MetaManifold.CladeCumulus + +# In-memory store for configs (in production, would be per-study persistent) +const _ANALYSIS_CONFIG_STORE = Dict{String,Dict{String,AnalysisConfig.AnalysisConfigStruct}}() # study -> id -> config +const _ANALYSIS_RESULT_STORE = Dict{String,Dict{String,AnalysisConfig.AnalysisResult}}() # study -> id -> result +const _ANALYSIS_CONFIG_LOCK = ReentrantLock() + +function _get_study_configs(study::String) + lock(_ANALYSIS_CONFIG_LOCK) do + get!(_ANALYSIS_CONFIG_STORE, study, Dict{String,AnalysisConfig.AnalysisConfigStruct}()) + end +end + +function _get_study_results(study::String) + lock(_ANALYSIS_CONFIG_LOCK) do + get!(_ANALYSIS_RESULT_STORE, study, Dict{String,AnalysisConfig.AnalysisResult}()) + end +end + +# List available metadata columns for a study (from first run's merged table or from study config) +function _available_metadata_columns(study::String)::Vector{String} + # Try to get from study's runs — for now return a default set plus any custom from config + # In real implementation, would read from sample metadata CSV or DuckDB + default_cols = ["group", "batch", "age", "sex", "disease", "sample_id", "tissue", "location"] + try + resolved = _resolve_config(study) + # If analysis.metadata_columns is configured, use that as available? + # For now just return default + any configured + return default_cols + catch + return default_cols + end +end + +# -------------------------------------------------------------------------- +# AnalysisConfig CRUD +# -------------------------------------------------------------------------- + +@post "/api/v1/studies/{study}/analysis-config" function(req, study::String) + study in _study_names() || return json_error(404, "study_not_found", "Study '$study' not found") + + body = JSON3.read(String(req.body), Dict{String,Any}) + + # Explicit parsing — no silent defaults + method = get(body, "method", nothing) + isnothing(method) && return json_error(400, "missing_method", "method is required (nb_glm, clr_lm, ilr_lm, logistic)") + + formula = get(body, "formula", nothing) + isnothing(formula) && return json_error(400, "missing_formula", "formula is required, e.g. '~ group'") + + metadata_columns = get(body, "metadata_columns", nothing) + isnothing(metadata_columns) && return json_error(400, "missing_metadata_columns", "metadata_columns is required, explicit list") + + # Normalization + norm_body = get(body, "normalization", Dict{String,Any}()) + norm_method = get(norm_body, "method", "none") + + # Build config via Julia types for heavy validation + try + normalization = AnalysisConfig.NormalizationConfig( + method=String(norm_method), + pseudocount=Float64(get(norm_body, "pseudocount", 0.5)), + ilr_basis=get(norm_body, "ilr_basis", nothing) isa Nothing ? nothing : String(get(norm_body, "ilr_basis", nothing)), + multiplicative_replacement_delta=get(norm_body, "multiplicative_replacement_delta", nothing) isa Nothing ? nothing : Float64(get(norm_body, "multiplicative_replacement_delta", nothing)), + ) + + corr_body = get(body, "correction", Dict{String,Any}()) + correction = AnalysisConfig.CorrectionConfig( + method=String(get(corr_body, "method", "BH")), + alpha=Float64(get(corr_body, "alpha", 0.05)), + allow_no_correction=Bool(get(corr_body, "allow_no_correction", false)), + acknowledgment_token=get(corr_body, "acknowledgment_token", nothing) isa Nothing ? nothing : String(get(corr_body, "acknowledgment_token", nothing)), + ) + + adv_body = get(body, "advanced", Dict{String,Any}()) + advanced = AnalysisConfig.AdvancedOverrides( + dispersion_method=String(get(adv_body, "dispersion_method", "parametric")), + zero_handling=String(get(adv_body, "zero_handling", "pseudocount")), + min_prevalence=Float64(get(adv_body, "min_prevalence", 0.1)), + min_abundance=Float64(get(adv_body, "min_abundance", 0.0)), + max_features=get(adv_body, "max_features", nothing) isa Nothing ? nothing : Int(get(adv_body, "max_features", nothing)), + min_samples_per_group=Int(get(adv_body, "min_samples_per_group", 3)), + robust=Bool(get(adv_body, "robust", false)), + acknowledgment_token=get(adv_body, "acknowledgment_token", nothing) isa Nothing ? nothing : String(get(adv_body, "acknowledgment_token", nothing)), + ) + + cfg = AnalysisConfig.AnalysisConfigStruct( + method=String(method), + formula=String(formula), + outcome_column=get(body, "outcome_column", nothing) isa Nothing ? nothing : String(get(body, "outcome_column", nothing)), + metadata_columns=Vector{String}(String.(metadata_columns)), + normalization=normalization, + correction=correction, + advanced=advanced, + created_by=String(get(body, "created_by", "anonymous")), + ) + + # Context-sensitive validation with available columns + available_cols = _available_metadata_columns(study) + errors = AnalysisConfig.validate_config(cfg, available_cols; strict=false) + if !isempty(errors) + return json_error(400, "validation_failed", "AnalysisConfig validation failed", join(errors, "\n")) + end + + # Check DANGER + danger = AnalysisConfig.danger_banner(cfg) + if !isnothing(danger) && !cfg.correction.allow_no_correction && cfg.advanced.zero_handling != "refuse" + # If dangerous but no acknowledgment, we still allow creation but flag + @warn "DANGEROUS config created" study id=cfg.id method=cfg.method danger=danger + end + + # Store + lock(_ANALYSIS_CONFIG_LOCK) do + study_configs = _get_study_configs(study) + study_configs[cfg.id] = cfg + end + + # Return with danger banner if any + resp = OrderedDict{String,Any}( + "config" => JSON3.read(AnalysisConfig.to_json(cfg)), + "danger_banner" => danger, + "is_dangerous" => AnalysisConfig.is_dangerous(cfg), + "help" => Dict( + "method" => AnalysisConfig.context_help("method"), + "formula" => AnalysisConfig.context_help("formula"), + ), + ) + + HTTP.Response(200, ["Content-Type" => "application/json"], body=JSON3.write(resp)) + + catch e + if e isa ArgumentError + return json_error(400, "invalid_config", "Invalid AnalysisConfig: $(e.msg)", sprint(showerror, e)) + else + @error "Failed to create AnalysisConfig" exception=(e, catch_backtrace()) + return json_error(500, "internal_error", "Failed to create AnalysisConfig: $(sprint(showerror, e))") + end + end +end + +@get "/api/v1/studies/{study}/analysis-config" function(req, study::String) + study in _study_names() || return json_error(404, "study_not_found", "Study '$study' not found") + + study_configs = _get_study_configs(study) + configs = [JSON3.read(AnalysisConfig.to_json(c)) for c in values(study_configs)] + + json(OrderedDict( + "study" => study, + "count" => length(configs), + "configs" => configs, + )) +end + +@get "/api/v1/studies/{study}/analysis-config/{id}" function(req, study::String, id::String) + study in _study_names() || return json_error(404, "study_not_found", "Study '$study' not found") + + study_configs = _get_study_configs(study) + haskey(study_configs, id) || return json_error(404, "config_not_found", "AnalysisConfig '$id' not found in study '$study'") + + cfg = study_configs[id] + + format = get(HTTP.queryparams(req), "format", "json") # json, nickel, deed + + if format == "nickel" + HTTP.Response(200, ["Content-Type" => "text/plain"], body=AnalysisConfig.to_nickel(cfg)) + elseif format == "deed" + HTTP.Response(200, ["Content-Type" => "text/plain"], body=AnalysisConfig.to_deed(cfg)) + else + json(OrderedDict( + "config" => JSON3.read(AnalysisConfig.to_json(cfg)), + "nickel" => AnalysisConfig.to_nickel(cfg), + "deed" => AnalysisConfig.to_deed(cfg), + "danger_banner" => AnalysisConfig.danger_banner(cfg), + "is_dangerous" => AnalysisConfig.is_dangerous(cfg), + )) + end +end + +@delete "/api/v1/studies/{study}/analysis-config/{id}" function(req, study::String, id::String) + study in _study_names() || return json_error(404, "study_not_found", "Study '$study' not found") + + lock(_ANALYSIS_CONFIG_LOCK) do + study_configs = _get_study_configs(study) + haskey(study_configs, id) || return json_error(404, "config_not_found", "AnalysisConfig '$id' not found") + delete!(study_configs, id) + end + + json(OrderedDict("deleted" => id)) +end + +@post "/api/v1/studies/{study}/analysis-config/{id}/validate" function(req, study::String, id::String) + study in _study_names() || return json_error(404, "study_not_found", "Study '$study' not found") + + study_configs = _get_study_configs(study) + haskey(study_configs, id) || return json_error(404, "config_not_found", "AnalysisConfig '$id' not found") + + cfg = study_configs[id] + available_cols = _available_metadata_columns(study) + + errors = AnalysisConfig.validate_config(cfg, available_cols; strict=false) + + json(OrderedDict( + "id" => id, + "valid" => isempty(errors), + "errors" => errors, + "is_dangerous" => AnalysisConfig.is_dangerous(cfg), + "danger_banner" => AnalysisConfig.danger_banner(cfg), + )) +end + +@post "/api/v1/studies/{study}/analysis-config/{id}/run" function(req, study::String, id::String) + study in _study_names() || return json_error(404, "study_not_found", "Study '$study' not found") + + study_configs = _get_study_configs(study) + haskey(study_configs, id) || return json_error(404, "config_not_found", "AnalysisConfig '$id' not found") + + cfg = study_configs[id] + + # For v1, we implement a mock run that returns fake results but with proper provenance + # Real implementation would call R via RCall for NB GLM etc. + + body = JSON3.read(String(req.body), Dict{String,Any}) + table = get(body, "table", "merged") + + # Simulate analysis — in real, would: + # - Load counts from DuckDB + # - Apply normalization (size_factors, CLR, ILR) + # - Run method via R (DESeq2, lm, glm) + # - BH correction mandatory + # - Return results with hash chain + + # For now, return mock results with proper structure + mock_results = OrderedDict{String,Any}() + # Would be populated from real analysis + + result = AnalysisConfig.AnalysisResult( + config_id=cfg.id, + config_hash=cfg.hash, + method=cfg.method, + results=mock_results, + provenance=OrderedDict{String,Any}( + "study" => study, + "table" => table, + "config_id" => cfg.id, + "config_hash" => cfg.hash, + "method" => AnalysisConfig.METHOD_TO_STRING[cfg.method], + "note" => "Mock result — real implementation requires R packages DESeq2, compositions, etc. This is v1 scaffold with provenance chain intact.", + ), + ) + + lock(_ANALYSIS_CONFIG_LOCK) do + study_results = _get_study_results(study) + study_results[result.id] = result + end + + json(OrderedDict( + "result" => OrderedDict( + "id" => result.id, + "config_id" => result.config_id, + "config_hash" => result.config_hash, + "method" => AnalysisConfig.METHOD_TO_STRING[result.method], + "results" => result.results, + "provenance" => result.provenance, + "hash" => result.hash, + ), + "config" => JSON3.read(AnalysisConfig.to_json(cfg)), + "danger_banner" => AnalysisConfig.danger_banner(cfg), + )) +end + +@post "/api/v1/studies/{study}/analysis-config/{id}/doi-bundle" function(req, study::String, id::String) + study in _study_names() || return json_error(404, "study_not_found", "Study '$study' not found") + + study_configs = _get_study_configs(study) + haskey(study_configs, id) || return json_error(404, "config_not_found", "AnalysisConfig '$id' not found") + + cfg = study_configs[id] + + body = JSON3.read(String(req.body), Dict{String,Any}) + authors = Vector{String}(String.(get(body, "authors", ["Anonymous"]))) + title = String(get(body, "title", "MetaManifold Analysis Bundle for $study")) + license = String(get(body, "license", "CC-BY-4.0")) + + # Find latest result for this config if any + study_results = _get_study_results(study) + latest_result = nothing + for r in values(study_results) + if r.config_id == id + if isnothing(latest_result) || r.created_at > latest_result.created_at + latest_result = r + end + end + end + + mktempdir() do tmpdir + bundle_dir = joinpath(tmpdir, "bundle_$(cfg.id)") + bundle_path = AnalysisConfig.create_doi_bundle(cfg, latest_result, bundle_dir; authors, title, license) + + # Zip the bundle for download + zip_path = bundle_dir * ".zip" + try + run(`zip -r $zip_path $bundle_dir`) + catch + # Fallback: just return the directory path if zip fails + return json(OrderedDict( + "bundle_path" => bundle_path, + "config_id" => cfg.id, + "config_hash" => cfg.hash, + "doi_ready" => true, + "files" => readdir(bundle_path), + "datacite" => JSON3.read(read(joinpath(bundle_path, "datacite.json"), String)), + )) + end + + if isfile(zip_path) + data = read(zip_path) + HTTP.Response(200, ["Content-Type" => "application/zip", "Content-Disposition" => "attachment; filename=\"$(study)_$(id)_doi_bundle.zip\""], body=data) + else + json(OrderedDict( + "bundle_path" => bundle_path, + "config_id" => cfg.id, + "config_hash" => cfg.hash, + "doi_ready" => true, + )) + end + end +end + +# -------------------------------------------------------------------------- +# CladeCumulus routes +# -------------------------------------------------------------------------- + +@post "/api/v1/studies/{study}/clade-cumulus/tree" function(req, study::String) + study in _study_names() || return json_error(404, "study_not_found", "Study '$study' not found") + + body = JSON3.read(String(req.body), Dict{String,Any}) + table = get(body, "table", "merged") + runs_spec = get(body, "runs", []) + + # For now, build a mock tree from first run's merged table + # Real implementation would query DuckDB for taxonomy + counts + avec_fibre + epistemic_status + + # Mock rows for demonstration + mock_rows = [ + Dict{String,Any}("Domain" => "Bacteria", "Phylum" => "Firmicutes", "Class" => "Bacilli", "Order" => "Lactobacillales", "Family" => "Lactobacillaceae", "Genus" => "Lactobacillus", "Species" => "Lactobacillus crispatus", "total" => 150.0, "avec_fibre" => true, "epistemic_status" => "present_in_every_admissible_world", "residual_count" => 2), + Dict{String,Any}("Domain" => "Bacteria", "Phylum" => "Firmicutes", "Class" => "Bacilli", "Order" => "Lactobacillales", "Family" => "Streptococcaceae", "Genus" => "Streptococcus", "Species" => "Streptococcus mitis", "total" => 80.0, "avec_fibre" => true, "epistemic_status" => "present_in_some_admissible_world", "residual_count" => 5), + Dict{String,Any}("Domain" => "Bacteria", "Phylum" => "Bacteroidetes", "Class" => "Bacteroidia", "Order" => "Bacteroidales", "Family" => "Bacteroidaceae", "Genus" => "Bacteroides", "Species" => "Bacteroides vulgatus", "total" => 200.0, "avec_fibre" => false, "epistemic_status" => "unknown", "residual_count" => 0), + Dict{String,Any}("Domain" => "Eukaryota", "Supergroup" => "TSAR", "Division" => "Alveolata", "Class" => "Aconoidasida", "Order" => "Piroplasmida", "Family" => "Babesiidae", "Genus" => "Babesia", "Species" => "Babesia microti", "total" => 30.0, "avec_fibre" => true, "epistemic_status" => "present_in_every_admissible_world", "residual_count" => 1), + ] + + tree = CladeCumulus.build_clade_tree(mock_rows) + + json(OrderedDict( + "study" => study, + "table" => table, + "tree" => JSON3.read(CladeCumulus.to_json(tree)), + "plotly" => CladeCumulus.to_plotly_tree(tree), + "note" => "Mock tree — real implementation queries DuckDB for taxonomy + avec_fibre + epistemic_status + residual_count, computes cumulative frequencies bottom-up", + )) +end + +@post "/api/v1/studies/{study}/clade-cumulus/validate-drag" function(req, study::String) + study in _study_names() || return json_error(404, "study_not_found", "Study '$study' not found") + + body = JSON3.read(String(req.body), Dict{String,Any}) + dragged_id = get(body, "dragged_id", nothing) + target_parent_id = get(body, "target_parent_id", nothing) + evidence = get(body, "evidence", Dict{String,Any}()) + + isnothing(dragged_id) && return json_error(400, "missing_dragged_id", "dragged_id required") + isnothing(target_parent_id) && return json_error(400, "missing_target", "target_parent_id required") + + # For demo, build mock tree and validate + mock_rows = [ + Dict{String,Any}("Domain" => "Bacteria", "Genus" => "Lactobacillus", "total" => 100.0, "avec_fibre" => true, "epistemic_status" => "present_in_every_admissible_world", "residual_count" => 1), + Dict{String,Any}("Domain" => "Bacteria", "Genus" => "Bacteroides", "total" => 200.0, "avec_fibre" => false, "epistemic_status" => "unknown", "residual_count" => 0), + ] + tree = CladeCumulus.build_clade_tree(mock_rows) + + # Find actual ids — in mock, ids are like "root|Domain:Bacteria|Genus:Lactobacillus" + # For simplicity, if dragged_id is label, find by label + dragged_node_id = String(dragged_id) + target_id = String(target_parent_id) + + # Try to resolve label to id if needed + if !haskey(tree.nodes, dragged_node_id) + for (id, node) in tree.nodes + if node.label == dragged_node_id + dragged_node_id = id + break + end + end + end + if !haskey(tree.nodes, target_id) + for (id, node) in tree.nodes + if node.label == target_id + target_id = id + break + end + end + end + + if !haskey(tree.nodes, dragged_node_id) || !haskey(tree.nodes, target_id) + # If not in mock tree, try generic validation based on evidence + # For live validation, we need avec_fibre and epistemic_status from evidence + avec = get(evidence, "avec_fibre", false) == true + status = get(evidence, "epistemic_status", "unknown") + + if !avec + return json(OrderedDict("valid" => false, "message" => "Cannot move: sans fibre (no semantic fibre). Needs avec_fibre=true.")) + end + if status != "present_in_every_admissible_world" + return json(OrderedDict("valid" => false, "message" => "Cannot move: not present_in_every_admissible_world (status=$status). Only taxa present in every admissible world can be placed. This is residual-evidence validation.")) + end + return json(OrderedDict("valid" => true, "message" => "Valid (generic check): avec_fibre=true and present_in_every_admissible_world")) + end + + (valid, message) = CladeCumulus.validate_drag_drop(tree, dragged_node_id, target_id, Dict{String,Any}(evidence)) + + json(OrderedDict("valid" => valid, "message" => message)) +end + +@get "/api/v1/studies/{study}/analysis/methods" function(req, study::String) + # List available analysis methods with help + json(OrderedDict( + "methods" => [ + OrderedDict("id" => "nb_glm", "label" => "Negative Binomial GLM", "description" => "For raw counts with overdispersion, DESeq2-style", "normalization" => ["none", "size_factors", "relative", "rarefy"], "requires" => "At least 3 samples per group", "help" => AnalysisConfig.context_help("method")), + OrderedDict("id" => "clr_lm", "label" => "CLR + Gaussian LM", "description" => "Centered Log-Ratio + LM, compositional (Aitchison)", "normalization" => ["clr"], "requires" => "Pseudocount >0, e.g. 0.5", "help" => AnalysisConfig.context_help("normalization.method")), + OrderedDict("id" => "ilr_lm", "label" => "ILR + Gaussian LM", "description" => "Isometric Log-Ratio + LM, balances", "normalization" => ["ilr"], "requires" => "Pseudocount + ilr_basis", "help" => AnalysisConfig.context_help("normalization.pseudocount")), + OrderedDict("id" => "logistic", "label" => "Logistic Regression", "description" => "For presence/absence or binary outcome", "normalization" => ["presence_absence", "none", "relative"], "requires" => "Binary outcome_column", "help" => "Logistic regression for binary outcomes"), + ], + "correction" => OrderedDict("mandatory" => "BH", "danger_token" => AnalysisConfig.DANGER_ACK_TOKEN, "help" => AnalysisConfig.context_help("correction.method")), + "schemas" => OrderedDict( + "json" => "/config/schemas/analysis_config.schema.json", + "nickel" => "/config/schemas/analysis_config.ncl", + "deed" => "/config/templates/analysis_config_chora.deed", + ), + "epistemic" => OrderedDict( + "avec_fibre_column" => AnalysisConfig.AVEC_FIBRE_COLUMN, + "epistemic_statuses" => AnalysisConfig.EPISTEMIC_STATUS_VALUES, + "present_in_every_admissible_world" => "Holds Present across all admissible worlds (residual-evidence-types)", + ), + )) +end diff --git a/src/server/server.jl b/src/server/server.jl index b709b00..4c1622b 100644 --- a/src/server/server.jl +++ b/src/server/server.jl @@ -22,6 +22,7 @@ module Server using MetaManifold.Tools, MetaManifold.TaxonomyTableTools, MetaManifold.ProjectSetup using MetaManifold.DADA2, MetaManifold.OTUPipeline using MetaManifold.DiversityMetrics, MetaManifold.Analysis + using MetaManifold.Epistemic, MetaManifold.AnalysisConfig, MetaManifold.CladeCumulus ## EPIPE log filter # HTTP.jl logs every broken-pipe error from SSE streams as @error @@ -86,6 +87,7 @@ module Server include(joinpath(@__DIR__, "routes", "events.jl")) include(joinpath(@__DIR__, "routes", "analysis.jl")) include(joinpath(@__DIR__, "routes", "composition.jl")) + include(joinpath(@__DIR__, "routes", "analysis_config.jl")) ## R-runtime busy middleware # The embedded R interpreter is shared between the pipeline and the analysis diff --git a/test/runtests.jl b/test/runtests.jl index 3400676..7f8e619 100644 --- a/test/runtests.jl +++ b/test/runtests.jl @@ -21,6 +21,7 @@ using MetaManifold.Tools, MetaManifold.TaxonomyTableTools, MetaManifold.ProjectS using MetaManifold.DiversityMetrics, MetaManifold.Analysis using MetaManifold.FuncDBAnnotation using MetaManifold.Categories, MetaManifold.CompositionLibrary +using MetaManifold.Epistemic, MetaManifold.AnalysisConfig, MetaManifold.CladeCumulus ## Unit tests (always run) @testset "MetabarcodingPipeline" begin @@ -52,6 +53,7 @@ using MetaManifold.Categories, MetaManifold.CompositionLibrary include("unit/test_provenance.jl") include("unit/test_install_pins.jl") include("unit/test_migrate_composition.jl") + include("unit/test_analysis_config.jl") ## Integration tests (opt-in) if RUN_INTEGRATION diff --git a/test/unit/test_analysis_config.jl b/test/unit/test_analysis_config.jl new file mode 100644 index 0000000..7c1741c --- /dev/null +++ b/test/unit/test_analysis_config.jl @@ -0,0 +1,473 @@ +# SPDX-License-Identifier: AGPL-3.0-only +@testset "AnalysisConfig — safe, explicit, versioned layer" begin + + using OrderedCollections + + @testset "NormalizationConfig validation" begin + # Valid + nc = AnalysisConfig.NormalizationConfig(method="none") + @test nc.method == "none" + + nc_clr = AnalysisConfig.NormalizationConfig(method="clr", pseudocount=0.5) + @test nc_clr.method == "clr" + + # Refuse pseudocount <=0 for CLR + @test_throws ArgumentError AnalysisConfig.NormalizationConfig(method="clr", pseudocount=0.0) + @test_throws ArgumentError AnalysisConfig.NormalizationConfig(method="clr", pseudocount=-0.1) + + # Refuse ilr_basis for non-ILR + @test_throws ArgumentError AnalysisConfig.NormalizationConfig(method="clr", pseudocount=0.5, ilr_basis="default") + @test_throws ArgumentError AnalysisConfig.NormalizationConfig(method="none", ilr_basis="default") + + # Valid ILR + nc_ilr = AnalysisConfig.NormalizationConfig(method="ilr", pseudocount=0.5, ilr_basis="default") + @test nc_ilr.ilr_basis == "default" + + # Invalid ILR basis + @test_throws ArgumentError AnalysisConfig.NormalizationConfig(method="ilr", pseudocount=0.5, ilr_basis="invalid_basis") + end + + @testset "CorrectionConfig — BH mandatory" begin + cc = AnalysisConfig.CorrectionConfig() + @test cc.method == "BH" + @test cc.alpha == 0.05 + @test cc.allow_no_correction == false + + # Non-BH without acknowledgment should throw + @test_throws ArgumentError AnalysisConfig.CorrectionConfig(method="none") + @test_throws ArgumentError AnalysisConfig.CorrectionConfig(method="bonferroni") + + # With acknowledgment token, allow + cc_danger = AnalysisConfig.CorrectionConfig(method="none", allow_no_correction=true, acknowledgment_token=AnalysisConfig.DANGER_ACK_TOKEN) + @test cc_danger.allow_no_correction == true + + # Wrong token should throw + @test_throws ArgumentError AnalysisConfig.CorrectionConfig(method="none", allow_no_correction=true, acknowledgment_token="wrong") + + # Alpha validation + @test_throws ArgumentError AnalysisConfig.CorrectionConfig(alpha=0.0) + @test_throws ArgumentError AnalysisConfig.CorrectionConfig(alpha=1.0) + @test_throws ArgumentError AnalysisConfig.CorrectionConfig(alpha=-0.1) + end + + @testset "AdvancedOverrides validation" begin + ao = AnalysisConfig.AdvancedOverrides() + @test ao.min_prevalence == 0.1 + + @test_throws ArgumentError AnalysisConfig.AdvancedOverrides(min_prevalence=-0.1) + @test_throws ArgumentError AnalysisConfig.AdvancedOverrides(min_prevalence=1.5) + @test_throws ArgumentError AnalysisConfig.AdvancedOverrides(min_abundance=-1.0) + @test_throws ArgumentError AnalysisConfig.AdvancedOverrides(max_features=0) + @test_throws ArgumentError AnalysisConfig.AdvancedOverrides(max_features=200000) + @test_throws ArgumentError AnalysisConfig.AdvancedOverrides(min_samples_per_group=1) + + # zero_handling=refuse requires token + @test_throws ArgumentError AnalysisConfig.AdvancedOverrides(zero_handling="refuse") + ao_refuse = AnalysisConfig.AdvancedOverrides(zero_handling="refuse", acknowledgment_token=AnalysisConfig.DANGER_ACK_TOKEN) + @test ao_refuse.zero_handling == "refuse" + end + + @testset "AnalysisConfigStruct creation and immutability" begin + norm = AnalysisConfig.NormalizationConfig(method="size_factors") + corr = AnalysisConfig.CorrectionConfig() + adv = AnalysisConfig.AdvancedOverrides() + + cfg = AnalysisConfig.AnalysisConfigStruct( + method="nb_glm", + formula="~ group", + metadata_columns=["group", "batch"], + normalization=norm, + correction=corr, + advanced=adv, + created_by="test_user" + ) + + @test cfg.method == AnalysisConfig.NB_GLM + @test cfg.formula == "~ group" + @test cfg.schema_version == AnalysisConfig.SCHEMA_VERSION + @test length(cfg.hash) == 64 # SHA256 hex + @test cfg.created_by == "test_user" + + # Hash should be deterministic for same content (except id and timestamp) + # Different ids should give different hashes? Actually hash includes id, so different + cfg2 = AnalysisConfig.AnalysisConfigStruct( + method="nb_glm", + formula="~ group", + metadata_columns=["group", "batch"], + normalization=norm, + correction=corr, + advanced=adv, + created_by="test_user", + id=cfg.id, + created_at=cfg.created_at, + ) + @test cfg.hash == cfg2.hash + + # Test method parsing + @test_throws ArgumentError AnalysisConfig.AnalysisConfigStruct( + method="invalid_method", + formula="~ group", + metadata_columns=["group"], + normalization=norm, + ) + + # Test empty formula refusal + @test_throws ArgumentError AnalysisConfig.AnalysisConfigStruct( + method="nb_glm", + formula="", + metadata_columns=["group"], + normalization=norm, + ) + + # Test formula without ~ + @test_throws ArgumentError AnalysisConfig.AnalysisConfigStruct( + method="nb_glm", + formula="group", + metadata_columns=["group"], + normalization=norm, + ) + + # Test empty metadata_columns + @test_throws ArgumentError AnalysisConfig.AnalysisConfigStruct( + method="nb_glm", + formula="~ group", + metadata_columns=String[], + normalization=norm, + ) + + # Test LOGISTIC requires outcome_column + @test_throws ArgumentError AnalysisConfig.AnalysisConfigStruct( + method="logistic", + formula="disease ~ group", + metadata_columns=["group"], + normalization=AnalysisConfig.NormalizationConfig(method="presence_absence"), + ) + + # Valid logistic + cfg_log = AnalysisConfig.AnalysisConfigStruct( + method="logistic", + formula="disease ~ group", + outcome_column="disease", + metadata_columns=["group", "disease"], + normalization=AnalysisConfig.NormalizationConfig(method="presence_absence"), + ) + @test cfg_log.method == AnalysisConfig.LOGISTIC + @test cfg_log.outcome_column == "disease" + + # Test incompatible normalization + @test_throws ArgumentError AnalysisConfig.AnalysisConfigStruct( + method="nb_glm", + formula="~ group", + metadata_columns=["group"], + normalization=AnalysisConfig.NormalizationConfig(method="clr", pseudocount=0.5), + ) + + @test_throws ArgumentError AnalysisConfig.AnalysisConfigStruct( + method="clr_lm", + formula="~ group", + metadata_columns=["group"], + normalization=AnalysisConfig.NormalizationConfig(method="none"), + ) + end + + @testset "validate_config with available metadata" begin + norm = AnalysisConfig.NormalizationConfig(method="size_factors") + cfg = AnalysisConfig.AnalysisConfigStruct( + method="nb_glm", + formula="~ group + batch", + metadata_columns=["group", "batch"], + normalization=norm, + ) + + # Valid with available columns + errors = AnalysisConfig.validate_config(cfg, ["group", "batch", "age"]; strict=false) + @test isempty(errors) + + # Missing column in available + errors2 = AnalysisConfig.validate_config(cfg, ["group"]; strict=false) + @test !isempty(errors2) + @test any(occursin("batch", e) for e in errors2) + + # Formula references non-listed column + cfg_bad = AnalysisConfig.AnalysisConfigStruct( + method="nb_glm", + formula="~ group + age", + metadata_columns=["group"], + normalization=norm, + ) + errors3 = AnalysisConfig.validate_config(cfg_bad, ["group", "age", "batch"]; strict=false) + @test !isempty(errors3) + @test any(occursin("age", e) for e in errors3) + end + + @testset "DANGER banner logic" begin + norm = AnalysisConfig.NormalizationConfig(method="size_factors") + + safe_cfg = AnalysisConfig.AnalysisConfigStruct( + method="nb_glm", + formula="~ group", + metadata_columns=["group"], + normalization=norm, + ) + @test !AnalysisConfig.is_dangerous(safe_cfg) + @test isnothing(AnalysisConfig.danger_banner(safe_cfg)) + + dangerous_corr = AnalysisConfig.CorrectionConfig(method="none", allow_no_correction=true, acknowledgment_token=AnalysisConfig.DANGER_ACK_TOKEN) + dangerous_cfg = AnalysisConfig.AnalysisConfigStruct( + method="nb_glm", + formula="~ group", + metadata_columns=["group"], + normalization=norm, + correction=dangerous_corr, + ) + @test AnalysisConfig.is_dangerous(dangerous_cfg) + banner = AnalysisConfig.danger_banner(dangerous_cfg) + @test !isnothing(banner) + @test occursin("DANGER", banner) + @test occursin("BH", banner) + end + + @testset "Serialization JSON roundtrip" begin + norm = AnalysisConfig.NormalizationConfig(method="clr", pseudocount=0.5) + cfg = AnalysisConfig.AnalysisConfigStruct( + method="clr_lm", + formula="~ group", + metadata_columns=["group"], + normalization=norm, + created_by="tester" + ) + + json_str = AnalysisConfig.to_json(cfg) + @test occursin("clr_lm", json_str) + @test occursin(cfg.id, json_str) + + cfg_restored = AnalysisConfig.from_json(json_str) + @test cfg_restored.method == cfg.method + @test cfg_restored.formula == cfg.formula + @test cfg_restored.normalization.method == cfg.normalization.method + end + + @testset "Serialization Nickel and DEED" begin + norm = AnalysisConfig.NormalizationConfig(method="size_factors") + cfg = AnalysisConfig.AnalysisConfigStruct( + method="nb_glm", + formula="~ group", + metadata_columns=["group"], + normalization=norm, + ) + + nickel = AnalysisConfig.to_nickel(cfg) + @test occursin("nb_glm", nickel) + @test occursin(cfg.id, nickel) + @test occursin("BH", nickel) + + deed = AnalysisConfig.to_deed(cfg) + @test occursin("repo-deed", deed) + @test occursin(":schema-version", deed) + @test occursin(cfg.id, deed) + @test occursin("nb_glm", deed) + end + + @testset "Context-sensitive help" begin + help_method = AnalysisConfig.context_help("method") + @test occursin("nb_glm", help_method) + @test occursin("clr_lm", help_method) + + help_formula = AnalysisConfig.context_help("formula") + @test occursin("~", help_formula) + + help_bh = AnalysisConfig.context_help("correction.method") + @test occursin("BH", help_bh) + end + + @testset "DOI bundle creation" begin + norm = AnalysisConfig.NormalizationConfig(method="size_factors") + cfg = AnalysisConfig.AnalysisConfigStruct( + method="nb_glm", + formula="~ group", + metadata_columns=["group"], + normalization=norm, + created_by="test_user" + ) + + result = AnalysisConfig.AnalysisResult( + config_id=cfg.id, + config_hash=cfg.hash, + method=cfg.method, + results=OrderedDict{String,Any}("taxon1" => OrderedDict("p" => 0.01, "log2FoldChange" => 2.0)) + ) + + mktempdir() do tmpdir + bundle_path = AnalysisConfig.create_doi_bundle(cfg, result, joinpath(tmpdir, "bundle"); authors=["Test User"], title="Test Bundle") + @test isdir(bundle_path) + @test isfile(joinpath(bundle_path, "analysis_config.json")) + @test isfile(joinpath(bundle_path, "analysis_config.ncl")) + @test isfile(joinpath(bundle_path, "analysis_config_chora.deed")) + @test isfile(joinpath(bundle_path, "datacite.json")) + @test isfile(joinpath(bundle_path, "provenance.json")) + @test isfile(joinpath(bundle_path, "analysis_result.json")) + @test isfile(joinpath(bundle_path, "README.md")) + + # Check datacite contains method + datacite_content = read(joinpath(bundle_path, "datacite.json"), String) + @test occursin("nb_glm", datacite_content) + end + end + + @testset "present_in_every_admissible_world" begin + # Sans fibre -> false + @test !AnalysisConfig.present_in_every_admissible_world([1.0, 2.0, 3.0], Dict{String,Any}("avec_fibre" => false)) + + # Avec fibre and all >= threshold -> true + @test AnalysisConfig.present_in_every_admissible_world([1.0, 2.0, 3.0], Dict{String,Any}("avec_fibre" => true, "epistemic_status" => "unknown"); threshold=1.0) + + # One below threshold -> false + @test !AnalysisConfig.present_in_every_admissible_world([0.0, 2.0, 3.0], Dict{String,Any}("avec_fibre" => true); threshold=1.0) + + # Explicit status present_in_every -> true even if counts low? Actually should trust status + @test AnalysisConfig.present_in_every_admissible_world([0.0], Dict{String,Any}("avec_fibre" => true, "epistemic_status" => "present_in_every_admissible_world")) + + # Explicit absent -> false + @test !AnalysisConfig.present_in_every_admissible_world([10.0], Dict{String,Any}("avec_fibre" => true, "epistemic_status" => "absent_in_every_admissible_world")) + end +end + +@testset "Epistemic module" begin + @testset "EchoFiber avec_fibre" begin + fiber_with = Epistemic.EchoFiber{String,String}("observed", ["w1", "w2"]) + @test Epistemic.avec_fibre(fiber_with) + @test !Epistemic.sans_fibre(fiber_with) + + fiber_empty = Epistemic.EchoFiber{String,String}("observed", String[]) + @test !Epistemic.avec_fibre(fiber_empty) + @test Epistemic.sans_fibre(fiber_empty) + end + + @testset "Warrant without soundness" begin + w = Epistemic.Warrant{String}("count>=1", [1, 2, 3], "taxon present") + @test w.claim == "taxon present" + # Having warrant does NOT give claim directly — need SoundWarrant + sw = Epistemic.SoundWarrant{String}(w, token -> token > 0 ? "taxon present" : "absent") + @test sw(1) == "taxon present" + end + + @testset "Candidate and present_in_every" begin + # Simplified world: (u,n) where u=true abundance, n=noise, observe=u+n + # For presence, we check u !=0 + + # Two candidates: (0,2) and (2,0) both observe 2 + c1 = Epistemic.Candidate{Tuple{Int,Int},Int}((0,2), true, true) + c2 = Epistemic.Candidate{Tuple{Int,Int},Int}((2,0), true, true) + case_unbounded = Epistemic.Case{Tuple{Int,Int},Int}(c1, [c1, c2]) + + present_fn = (world) -> world[1] != 0 # u !=0 + + @test !Epistemic.present_in_every_admissible_world(case_unbounded, present_fn) # (0,2) has u=0, so not present in every + + # Bounded: n<=1, candidates (1,1) and (2,0) both have u!=0 + c3 = Epistemic.Candidate{Tuple{Int,Int},Int}((1,1), true, true) + c4 = Epistemic.Candidate{Tuple{Int,Int},Int}((2,0), true, true) + case_bounded = Epistemic.Case{Tuple{Int,Int},Int}(c3, [c3, c4]) + + @test Epistemic.present_in_every_admissible_world(case_bounded, present_fn) # both have u!=0 + end + + @testset "Epistemic colour coding" begin + @test Epistemic.epistemic_colour("present_in_every_admissible_world") == "#2e7d32" + @test Epistemic.epistemic_colour("present_in_some_admissible_world") == "#f9a825" + @test Epistemic.epistemic_colour("absent_in_every_admissible_world") == "#9e9e9e" + @test Epistemic.epistemic_colour("unknown") == "#c62828" + end + + @testset "Cloud sizing" begin + s1 = Epistemic.cloud_size_by_residual(0) + s2 = Epistemic.cloud_size_by_residual(10) + s3 = Epistemic.cloud_size_by_residual(100) + @test s1 < s2 < s3 + @test s1 >= 5.0 + end +end + +@testset "CladeCumulus" begin + @testset "CladeNode creation" begin + node = CladeCumulus.CladeNode(id="test", label="Test", rank="Genus", count=10.0) + @test node.id == "test" + @test node.count == 10.0 + @test node.epistemic_status == "unknown" + end + + @testset "Build clade tree and cumulative frequencies" begin + rows = [ + Dict{String,Any}("Domain" => "Bacteria", "Phylum" => "Firmicutes", "Genus" => "Lactobacillus", "total" => 100.0, "avec_fibre" => true, "epistemic_status" => "present_in_every_admissible_world", "residual_count" => 2), + Dict{String,Any}("Domain" => "Bacteria", "Phylum" => "Firmicutes", "Genus" => "Streptococcus", "total" => 50.0, "avec_fibre" => true, "epistemic_status" => "present_in_some_admissible_world", "residual_count" => 5), + Dict{String,Any}("Domain" => "Bacteria", "Phylum" => "Bacteroidetes", "Genus" => "Bacteroides", "total" => 200.0, "avec_fibre" => false, "epistemic_status" => "unknown", "residual_count" => 0), + ] + + tree = CladeCumulus.build_clade_tree(rows) + @test haskey(tree.nodes, "root") + @test tree.total_count >= 350.0 + + freqs = CladeCumulus.cumulative_frequencies(tree) + @test !isempty(freqs) + + # Root should have cumulative_frequency 1.0 + root_node = tree.nodes[tree.root_id] + @test root_node.cumulative_frequency ≈ 1.0 atol=0.01 + end + + @testset "Drag-and-drop validation with present_in_every_admissible_world" begin + rows = [ + Dict{String,Any}("Domain" => "Bacteria", "Genus" => "Lactobacillus", "total" => 100.0, "avec_fibre" => true, "epistemic_status" => "present_in_every_admissible_world", "residual_count" => 1), + Dict{String,Any}("Domain" => "Bacteria", "Genus" => "Bacteroides", "total" => 200.0, "avec_fibre" => false, "epistemic_status" => "unknown", "residual_count" => 0), + ] + tree = CladeCumulus.build_clade_tree(rows) + + # Find nodes + lacto_id = nothing + bact_id = nothing + for (id, node) in tree.nodes + if node.label == "Lactobacillus" + lacto_id = id + elseif node.label == "Bacteroides" + bact_id = id + end + end + @test !isnothing(lacto_id) + @test !isnothing(bact_id) + + # Valid: Lactobacillus has avec_fibre and present_in_every + (valid, msg) = CladeCumulus.validate_drag_drop(tree, lacto_id, tree.root_id, Dict{String,Any}()) + @test valid + @test occursin("present_in_every", msg) + + # Invalid: Bacteroides sans fibre + (valid2, msg2) = CladeCumulus.validate_drag_drop(tree, bact_id, tree.root_id, Dict{String,Any}()) + @test !valid2 + @test occursin("sans fibre", msg2) + + # Invalid: drop onto descendant (cycle) + # Create a child under lacto and try to drop lacto onto child + child_id = lacto_id * "|child" + # We need to add child manually for test + nodes = copy(tree.nodes) + nodes[child_id] = CladeCumulus.CladeNode(id=child_id, label="Child", parent_id=lacto_id, count=10.0, avec_fibre=true, epistemic_status="present_in_every_admissible_world") + # Update parent's children + old_parent = nodes[lacto_id] + nodes[lacto_id] = CladeCumulus.CladeNode(id=old_parent.id, label=old_parent.label, rank=old_parent.rank, parent_id=old_parent.parent_id, children_ids=vcat(old_parent.children_ids, [child_id]), count=old_parent.count, cumulative_count=old_parent.cumulative_count, residual_count=old_parent.residual_count, avec_fibre=old_parent.avec_fibre, epistemic_status=old_parent.epistemic_status) + tree2 = CladeCumulus.CladeTree(nodes, tree.root_id) + + (valid3, msg3) = CladeCumulus.validate_drag_drop(tree2, lacto_id, child_id, Dict{String,Any}()) + @test !valid3 + @test occursin("cycle", lowercase(msg3)) + end + + @testset "Epistemic colour and cloud size" begin + node_every = CladeCumulus.CladeNode(id="every", label="Every", count=10.0, avec_fibre=true, epistemic_status="present_in_every_admissible_world", residual_count=2) + node_some = CladeCumulus.CladeNode(id="some", label="Some", count=10.0, avec_fibre=true, epistemic_status="present_in_some_admissible_world", residual_count=5) + + @test CladeCumulus.epistemic_colour_for_node(node_every) == "#2e7d32" + @test CladeCumulus.epistemic_colour_for_node(node_some) == "#f9a825" + + @test CladeCumulus.cloud_size_for_node(node_some) > CladeCumulus.cloud_size_for_node(node_every) + end +end From 8a2a9a38ffd5ea7445cce3214620442324ea2bd7 Mon Sep 17 00:00:00 2001 From: hyperpolymath <6759885+hyperpolymath@users.noreply.github.com> Date: Fri, 18 Sep 2026 16:15:12 +0000 Subject: [PATCH 2/2] fix(docs): add SPDX headers to milestone docs to pass repo-hygiene gate - 00-reconnaissance.md, 01-project-board-graphql.md, 02-deferred-issues.md were missing SPDX - Now CC-BY-SA-4.0 for prose docs per check-spdx policy - Fixes CI failure in PR #11 and #12 licence header check --- docs/milestones/00-reconnaissance.md | 5 +++++ docs/milestones/01-project-board-graphql.md | 5 +++++ docs/milestones/02-deferred-issues.md | 5 +++++ 3 files changed, 15 insertions(+) diff --git a/docs/milestones/00-reconnaissance.md b/docs/milestones/00-reconnaissance.md index 2777f59..20b4e28 100644 --- a/docs/milestones/00-reconnaissance.md +++ b/docs/milestones/00-reconnaissance.md @@ -1,3 +1,8 @@ + + # Milestone 0 — Full Reconnaissance (2026-09-18 Europe/London) ## Repository State diff --git a/docs/milestones/01-project-board-graphql.md b/docs/milestones/01-project-board-graphql.md index 341e659..7784933 100644 --- a/docs/milestones/01-project-board-graphql.md +++ b/docs/milestones/01-project-board-graphql.md @@ -1,3 +1,8 @@ + + # Milestone 1 — GitHub Project Board via GraphQL ## Board Name diff --git a/docs/milestones/02-deferred-issues.md b/docs/milestones/02-deferred-issues.md index dffc2a4..5e6fe11 100644 --- a/docs/milestones/02-deferred-issues.md +++ b/docs/milestones/02-deferred-issues.md @@ -1,3 +1,8 @@ + + # Deferred Features — Ready-to-Paste GitHub Issue Bodies These are for features that must NOT be implemented in the current task (exact statistics layer and symbolic engine),