From a211b03ae8ac86663e90a4135d857878341e2a0e Mon Sep 17 00:00:00 2001 From: Daniel Nachun Date: Mon, 21 Sep 2026 13:19:58 -0700 Subject: [PATCH 1/3] update to jupyter-book 2.0 --- .github/workflows/build_site.yml | 25 +- .github/workflows/dispatch_build_site.yml | 24 +- .gitignore | 1 + CONTRIBUTORS.md | 2 +- README.md | 2 +- .../TensorQTL/TensorQTL.ipynb | 34 +- .../qtl_association_postprocessing.ipynb | 15 +- .../qtl_association_testing.ipynb | 6 +- .../quantile_models/qr_and_twas.ipynb | 345 ++++++++++++++++- .../covariate/covariate_formatting.ipynb | 19 +- .../covariate/covariate_hidden_factor.ipynb | 30 +- .../covariate_preprocessing.ipynb | 86 ++--- .../data_preprocessing/genotype/GWAS_QC.ipynb | 44 ++- .../SoS/data_preprocessing/genotype/PCA.ipynb | 65 +++- .../data_preprocessing/genotype/VCF_QC.ipynb | 48 ++- .../genotype/genotype_formatting.ipynb | 67 +++- .../genotype_preprocessing.ipynb | 66 +++- .../phenotype/gene_annotation.ipynb | 43 ++- .../phenotype/phenotype_formatting.ipynb | 40 +- .../phenotype/phenotype_imputation.ipynb | 48 ++- .../phenotype_preprocessing.ipynb | 144 +++---- .../enrichment/enrichment_validation.ipynb | 22 +- code/SoS/enrichment/eoo_enrichment.ipynb | 20 +- code/SoS/enrichment/gregor.ipynb | 44 ++- code/SoS/enrichment/gsea.ipynb | 18 +- code/SoS/enrichment/sldsc_enrichment.ipynb | 39 +- code/SoS/graveyard/APEX/APEX.ipynb | 19 +- code/SoS/graveyard/GRM.ipynb | 22 +- code/SoS/graveyard/MMQTL/MMQTL.ipynb | 14 +- code/SoS/graveyard/SuSiE_slides.ipynb | 2 +- .../association_scan_post_processing.ipynb | 2 +- code/SoS/graveyard/bam_to_bw.ipynb | 17 +- .../graveyard/bulk_expression_commands.ipynb | 4 +- .../graveyard/eQTL_analysis_commands.ipynb | 2 +- code/SoS/graveyard/fastenloc_susie.ipynb | 40 +- code/SoS/graveyard/genotype_alignment.ipynb | 15 +- .../graveyard/ld_reference_generation.ipynb | 23 +- code/SoS/graveyard/polyfun.ipynb | 2 +- ...bulk_expression_QC_and_normalization.ipynb | 24 +- ...obulk_expression_aggregation_QC_norm.ipynb | 29 +- ...mega_expression_QC_and_normalization.ipynb | 18 +- .../1_twas_mr_colocboost.ipynb | 363 +++++++++++++++++- .../misc/advanced_analysis/2_enrichment.ipynb | 64 ++- code/SoS/misc/build_container.ipynb | 8 +- .../1_phenotype_preprocessing.ipynb | 17 +- .../2_genotype_preprocessing.ipynb | 21 +- .../data_preprocessing/3_genotype_pca.ipynb | 23 +- .../4_covariates_preprocessing.ipynb | 14 +- code/SoS/misc/mini-protocol-example.ipynb | 14 +- code/SoS/misc/module-example.ipynb | 290 +++++++------- .../1_xqtl_association.ipynb | 36 +- .../2_finemapping.ipynb | 16 +- .../mnm_analysis/mnm_methods/colocboost.ipynb | 36 +- .../mnm_methods/mnm_regression.ipynb | 66 +++- .../mnm_methods/qtl_rss_analysis.ipynb | 252 +++++++++++- .../mnm_methods/rss_analysis.ipynb | 22 +- code/SoS/mnm_analysis/mnm_miniprotocol.ipynb | 16 +- .../SoS/mnm_analysis/mnm_postprocessing.ipynb | 2 +- .../multivariate_fine_mapping_vignette.ipynb | 4 +- ...iate_multigene_fine_mapping_vignette.ipynb | 37 +- .../summary_stats_finemapping_vignette.ipynb | 37 +- ...variate_fine_mapping_fsusie_vignette.ipynb | 37 +- ...nivariate_fine_mapping_twas_vignette.ipynb | 33 +- .../molecular_phenotypes/QC/apa_impute.ipynb | 24 +- .../QC/bulk_expression_QC.ipynb | 21 +- .../QC/bulk_expression_normalization.ipynb | 16 +- .../QC/pseudobulk_preprocessing.ipynb | 37 +- .../QC/splicing_normalization.ipynb | 40 +- code/SoS/molecular_phenotypes/apa.ipynb | 39 +- .../bulk_expression.ipynb | 52 ++- .../calling/RNA_calling.ipynb | 74 +++- .../calling/apa_calling.ipynb | 37 +- .../calling/methylation_calling.ipynb | 4 +- .../molecular_phenotypes/methylation.ipynb | 12 +- .../molecular_phenotypes/single_cell.ipynb | 25 +- .../snRNAseq_preprocessing.ipynb | 22 +- code/SoS/molecular_phenotypes/splicing.ipynb | 10 +- code/SoS/multivariate_genome/MASH.ipynb | 66 +++- .../multivariate_genome/MASH/mash_fit.ipynb | 23 +- .../MASH/mash_posterior.ipynb | 66 +++- .../MASH/mash_preprocessing.ipynb | 24 +- .../MASH/mixture_prior.ipynb | 73 +++- .../SoS/multivariate_genome/METAL/METAL.ipynb | 28 +- .../multivariate_mixture_vignette.ipynb | 6 +- .../SoS/pecotmr_integration/SuSiE_enloc.ipynb | 2 +- .../gwas_integration.ipynb | 86 ++++- code/SoS/pecotmr_integration/intact.ipynb | 16 +- code/SoS/pecotmr_integration/twas_ctwas.ipynb | 36 +- .../pecotmr_integration/twas_vignette.ipynb | 6 +- code/SoS/rare_xqtl/watershed.ipynb | 8 +- .../SoS/reference_data/generalized_TADB.ipynb | 2 +- .../reference_data/ld_prune_reference.ipynb | 2 +- code/SoS/reference_data/reference_data.ipynb | 64 ++- .../reference_data_preparation.ipynb | 72 +++- code/SoS/reference_data/rss_ld_sketch.ipynb | 103 ++++- .../xqtl_modifier_score/ems_prediction.ipynb | 278 +++++++++++++- .../xqtl_modifier_score/ems_training.ipynb | 272 ++++++++++++- code/SoS/xqtl_protocol_demo.ipynb | 4 +- code/SoS/xqtl_protocol_draft.ipynb | 2 +- code/SoS/xqtl_protocol_workflow_builder.html | 2 +- code/SoS/xqtl_protocol_workflow_builder.ipynb | 13 +- myst.yml | 127 ++++++ recipe.yaml | 0 .../calling/test_methylation_calling.py | 1 - .../scripts/test_documented_command_paths.py | 32 +- tests/scripts/test_notebook_format.py | 74 ++++ tests/scripts/test_website_sources.py | 47 +++ website/_config.yml | 40 -- website/_toc.yml | 110 ------ website/build_flat_book.py | 260 ------------- website/build_support.py | 222 +++++++++++ .../nature_protocol/conversion_notebook.ipynb | 69 ++-- .../nature_protocol/example_manuscript.ipynb | 2 +- .../example_miniprotocol.ipynb | 7 +- website/nature_protocol/example_module.ipynb | 4 +- .../xqtl_flowchart.html | 0 116 files changed, 4450 insertions(+), 1150 deletions(-) create mode 100644 myst.yml delete mode 100644 recipe.yaml create mode 100644 tests/scripts/test_notebook_format.py create mode 100644 tests/scripts/test_website_sources.py delete mode 100644 website/_config.yml delete mode 100644 website/_toc.yml delete mode 100644 website/build_flat_book.py create mode 100644 website/build_support.py rename xqtl_flowchart.html => website/xqtl_flowchart.html (100%) diff --git a/.github/workflows/build_site.yml b/.github/workflows/build_site.yml index 59f06ccfb..6b18400a5 100644 --- a/.github/workflows/build_site.yml +++ b/.github/workflows/build_site.yml @@ -6,7 +6,6 @@ on: paths: - 'code/**/*.ipynb' - 'website/**' - - 'xqtl_flowchart.html' jobs: build_website: @@ -21,28 +20,20 @@ jobs: token: ${{ secrets.CI_TOKEN }} fetch-depth: 0 - - name: Set up Python - uses: actions/setup-python@v7 + - name: Setup pixi + uses: prefix-dev/setup-pixi@v0.10.2 with: - python-version: '3.12' - - - name: Install dependencies - run: | - pip install jupyter-book==1.0.4 ghp-import nbconvert + run-install: false - name: Build website and manuscript + shell: pixi exec --spec jupyter-book ghp-import nbconvert -- bash -e {0} run: | - # Build website from a temporary flat source tree. The source notebooks - # stay under code/SoS, while public pages are emitted as .html. rm -f pipeline/*.ipynb - BOOK_SRC="$(mktemp -d)" - python website/build_flat_book.py --output "$BOOK_SRC" - jupyter-book build "$BOOK_SRC" --path-output . --config "$BOOK_SRC/website/_config.yml" --toc "$BOOK_SRC/website/_toc.yml" - python website/build_flat_book.py --redirects _build/html - rsync -auzP code/images/* _build/html/_images/ - cp xqtl_flowchart.html _build/html/ + python website/build_support.py --stage-widget + BASE_URL=/xqtl-protocol jupyter-book build --html + python website/build_support.py --widget _build/html --redirects _build/html + cp website/xqtl_flowchart.html _build/html/ ghp-import -n -p -f _build/html - # Link notebooks find code/SoS -name '*.ipynb' -not -path '*/.ipynb_checkpoints/*' -print0 | while IFS= read -r -d '' notebook; do ln -s "../${notebook}" "pipeline/$(basename "$notebook")" diff --git a/.github/workflows/dispatch_build_site.yml b/.github/workflows/dispatch_build_site.yml index 58b6e9435..e932a899c 100644 --- a/.github/workflows/dispatch_build_site.yml +++ b/.github/workflows/dispatch_build_site.yml @@ -16,28 +16,20 @@ jobs: token: ${{ secrets.CI_TOKEN }} fetch-depth: 0 - - name: Set up Python - uses: actions/setup-python@v7 + - name: Setup pixi + uses: prefix-dev/setup-pixi@v0.10.2 with: - python-version: '3.12' - - - name: Install dependencies - run: | - pip install jupyter-book==1.0.4 ghp-import nbconvert + run-install: false - name: Build website and manuscript + shell: pixi exec --spec jupyter-book ghp-import nbconvert -- bash -e {0} run: | - # Build website from a temporary flat source tree. The source notebooks - # stay under code/SoS, while public pages are emitted as .html. rm -f pipeline/*.ipynb - BOOK_SRC="$(mktemp -d)" - python website/build_flat_book.py --output "$BOOK_SRC" - jupyter-book build "$BOOK_SRC" --path-output . --config "$BOOK_SRC/website/_config.yml" --toc "$BOOK_SRC/website/_toc.yml" - python website/build_flat_book.py --redirects _build/html - rsync -auzP code/images/* _build/html/_images/ - cp xqtl_flowchart.html _build/html/ + python website/build_support.py --stage-widget + BASE_URL=/xqtl-protocol jupyter-book build --html + python website/build_support.py --widget _build/html --redirects _build/html + cp website/xqtl_flowchart.html _build/html/ ghp-import -n -p -f _build/html - # Link notebooks find code/SoS -name '*.ipynb' -not -path '*/.ipynb_checkpoints/*' -print0 | while IFS= read -r -d '' notebook; do ln -s "../${notebook}" "pipeline/$(basename "$notebook")" diff --git a/.gitignore b/.gitignore index ec547a5dd..eba0f1f95 100644 --- a/.gitignore +++ b/.gitignore @@ -69,3 +69,4 @@ pixi.lock output_*/ input/ *.log +website/_generated diff --git a/CONTRIBUTORS.md b/CONTRIBUTORS.md index 8fb86ebf1..710745cdb 100644 --- a/CONTRIBUTORS.md +++ b/CONTRIBUTORS.md @@ -54,4 +54,4 @@ maintain it. --- -New to the protocol? See [Environment Setup](https://statfungen.github.io/xqtl-protocol/xqtl_protocol_demo.html) to get running, or the [xQTL Analysis Workflow Builder](https://statfungen.github.io/xqtl-protocol/xqtl_protocol_workflow_builder.html) to find the pipelines for your study. +New to the protocol? See [Environment Setup](https://statfungen.github.io/xqtl-protocol/xqtl-protocol-demo) to get running, or the [xQTL Analysis Workflow Builder](https://statfungen.github.io/xqtl-protocol/xqtl_protocol_workflow_builder.html) to find the pipelines for your study. diff --git a/README.md b/README.md index 68c7b5128..fb784e24c 100644 --- a/README.md +++ b/README.md @@ -25,7 +25,7 @@ that route with the commands to run them. | I want to... | Go to | |---|---| -| **Set up my computing environment** | [Environment Setup](https://statfungen.github.io/xqtl-protocol/xqtl_protocol_demo.html) | +| **Set up my computing environment** | [Environment Setup](https://statfungen.github.io/xqtl-protocol/xqtl-protocol-demo) | | **Work out which pipelines I need** | [xQTL Analysis Workflow Builder](https://statfungen.github.io/xqtl-protocol/xqtl_protocol_workflow_builder.html) | ## Overview of the protocol diff --git a/code/SoS/association_scan/TensorQTL/TensorQTL.ipynb b/code/SoS/association_scan/TensorQTL/TensorQTL.ipynb index 714450d48..fbac9bb5f 100644 --- a/code/SoS/association_scan/TensorQTL/TensorQTL.ipynb +++ b/code/SoS/association_scan/TensorQTL/TensorQTL.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "c57f2ed4", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "f9a4a145", "metadata": { "kernel": "SoS" }, @@ -32,6 +34,7 @@ }, { "cell_type": "markdown", + "id": "280fd7c1", "metadata": { "kernel": "SoS" }, @@ -75,6 +78,7 @@ }, { "cell_type": "markdown", + "id": "ef210a58", "metadata": { "kernel": "SoS" }, @@ -85,6 +89,7 @@ { "cell_type": "code", "execution_count": 6, + "id": "17baa000", "metadata": { "kernel": "R" }, @@ -211,6 +216,7 @@ }, { "cell_type": "markdown", + "id": "48374a8e", "metadata": { "kernel": "SoS" }, @@ -221,6 +227,7 @@ { "cell_type": "code", "execution_count": 7, + "id": "72ddf454", "metadata": { "kernel": "R" }, @@ -375,6 +382,7 @@ }, { "cell_type": "markdown", + "id": "431b57e2", "metadata": { "kernel": "SoS" }, @@ -385,6 +393,7 @@ { "cell_type": "code", "execution_count": 9, + "id": "b92a2458", "metadata": { "kernel": "R" }, @@ -578,6 +587,7 @@ }, { "cell_type": "markdown", + "id": "f5c7902c", "metadata": { "kernel": "SoS" }, @@ -588,6 +598,7 @@ { "cell_type": "code", "execution_count": 10, + "id": "fbca14a5", "metadata": { "kernel": "R" }, @@ -665,6 +676,7 @@ }, { "cell_type": "markdown", + "id": "a11f0598", "metadata": { "kernel": "SoS" }, @@ -675,6 +687,7 @@ { "cell_type": "code", "execution_count": 11, + "id": "203e2fb0", "metadata": { "kernel": "R" }, @@ -703,6 +716,7 @@ }, { "cell_type": "markdown", + "id": "2dd2644e", "metadata": { "kernel": "SoS" }, @@ -765,6 +779,7 @@ }, { "cell_type": "markdown", + "id": "c76f813b", "metadata": { "kernel": "SoS" }, @@ -867,6 +882,7 @@ }, { "cell_type": "markdown", + "id": "4c25577a", "metadata": { "kernel": "SoS" }, @@ -878,6 +894,7 @@ }, { "cell_type": "markdown", + "id": "613fc631", "metadata": { "kernel": "SoS" }, @@ -889,6 +906,7 @@ }, { "cell_type": "markdown", + "id": "6abf80d7", "metadata": { "kernel": "SoS" }, @@ -899,6 +917,7 @@ { "cell_type": "code", "execution_count": null, + "id": "57eb46a4", "metadata": { "kernel": "Bash" }, @@ -913,6 +932,7 @@ }, { "cell_type": "markdown", + "id": "8d378186", "metadata": { "kernel": "SoS" }, @@ -924,6 +944,7 @@ }, { "cell_type": "markdown", + "id": "c3b5fb81", "metadata": { "kernel": "SoS" }, @@ -934,6 +955,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f40a4504", "metadata": { "kernel": "Bash" }, @@ -949,6 +971,7 @@ }, { "cell_type": "markdown", + "id": "22ce6ae7", "metadata": { "kernel": "SoS" }, @@ -959,6 +982,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e75feaee", "metadata": { "kernel": "Bash" }, @@ -969,6 +993,7 @@ }, { "cell_type": "markdown", + "id": "a9807227", "metadata": { "kernel": "SoS" }, @@ -1079,6 +1104,7 @@ }, { "cell_type": "markdown", + "id": "8df2d07f", "metadata": { "kernel": "SoS" }, @@ -1091,6 +1117,7 @@ { "cell_type": "code", "execution_count": null, + "id": "d079ab12", "metadata": { "kernel": "SoS" }, @@ -1244,6 +1271,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5e19da07", "metadata": { "kernel": "SoS" }, @@ -1303,6 +1331,7 @@ { "cell_type": "code", "execution_count": null, + "id": "cdf89360", "metadata": { "kernel": "SoS" }, @@ -1324,6 +1353,7 @@ { "cell_type": "code", "execution_count": null, + "id": "861aefc3", "metadata": { "kernel": "SoS" }, @@ -1413,5 +1443,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/association_scan/qtl_association_postprocessing.ipynb b/code/SoS/association_scan/qtl_association_postprocessing.ipynb index 9d7e402cf..2117e4052 100644 --- a/code/SoS/association_scan/qtl_association_postprocessing.ipynb +++ b/code/SoS/association_scan/qtl_association_postprocessing.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "32e9381c", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "6d59f2f4", "metadata": { "kernel": "SoS" }, @@ -40,6 +42,7 @@ }, { "cell_type": "markdown", + "id": "e4a5a5cc", "metadata": { "kernel": "SoS" }, @@ -86,6 +89,7 @@ }, { "cell_type": "markdown", + "id": "45c09740", "metadata": { "kernel": "SoS" }, @@ -102,6 +106,7 @@ }, { "cell_type": "markdown", + "id": "e668fe12", "metadata": { "kernel": "SoS" }, @@ -113,6 +118,7 @@ }, { "cell_type": "markdown", + "id": "506d365f", "metadata": { "kernel": "SoS" }, @@ -123,6 +129,7 @@ { "cell_type": "code", "execution_count": null, + "id": "dd0001f0", "metadata": { "kernel": "Bash" }, @@ -138,6 +145,7 @@ }, { "cell_type": "markdown", + "id": "bd90e957", "metadata": { "kernel": "SoS" }, @@ -148,6 +156,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4950b621", "metadata": { "kernel": "SoS" }, @@ -158,6 +167,7 @@ }, { "cell_type": "markdown", + "id": "239519a1", "metadata": { "kernel": "SoS" }, @@ -208,6 +218,7 @@ }, { "cell_type": "markdown", + "id": "fab76b4f", "metadata": { "kernel": "SoS" }, @@ -220,6 +231,7 @@ { "cell_type": "code", "execution_count": null, + "id": "fca638c7", "metadata": { "kernel": "SoS" }, @@ -256,6 +268,7 @@ { "cell_type": "code", "execution_count": null, + "id": "dcb0e78c", "metadata": { "kernel": "SoS" }, @@ -310,5 +323,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 + "nbformat_minor": 5 } diff --git a/code/SoS/association_scan/qtl_association_testing.ipynb b/code/SoS/association_scan/qtl_association_testing.ipynb index ba4692bca..f1e78301a 100644 --- a/code/SoS/association_scan/qtl_association_testing.ipynb +++ b/code/SoS/association_scan/qtl_association_testing.ipynb @@ -64,6 +64,7 @@ { "cell_type": "code", "execution_count": null, + "id": "25aa3e0e", "metadata": { "kernel": "Bash" }, @@ -89,6 +90,7 @@ { "cell_type": "code", "execution_count": null, + "id": "834eca34", "metadata": { "kernel": "Bash" }, @@ -115,6 +117,7 @@ { "cell_type": "code", "execution_count": null, + "id": "70f40332", "metadata": { "kernel": "Bash" }, @@ -171,6 +174,7 @@ { "cell_type": "code", "execution_count": null, + "id": "507ebce6", "metadata": { "kernel": "Bash" }, @@ -207,4 +211,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/association_scan/quantile_models/qr_and_twas.ipynb b/code/SoS/association_scan/quantile_models/qr_and_twas.ipynb index efdbbe0bb..ca1bcbd72 100644 --- a/code/SoS/association_scan/quantile_models/qr_and_twas.ipynb +++ b/code/SoS/association_scan/quantile_models/qr_and_twas.ipynb @@ -2,13 +2,21 @@ "cells": [ { "cell_type": "markdown", + "id": "8803b471", "metadata": { "kernel": "SoS" }, - "source": "# Quantile regression for QTL association testing\n\n> **Under active development.** This module is a work in progress: its interface, parameters and outputs may still change, and it is not yet covered by the automated test suite.\n\nFits quantile regression of a molecular phenotype on genotype across a grid of quantiles, and turns the fit into TWAS weights." + "source": [ + "# Quantile regression for QTL association testing\n", + "\n", + "> **Under active development.** This module is a work in progress: its interface, parameters and outputs may still change, and it is not yet covered by the automated test suite.\n", + "\n", + "Fits quantile regression of a molecular phenotype on genotype across a grid of quantiles, and turns the fit into TWAS weights." + ] }, { "cell_type": "markdown", + "id": "e8979282", "metadata": { "kernel": "SoS" }, @@ -24,6 +32,7 @@ }, { "cell_type": "markdown", + "id": "b86aafa9", "metadata": { "kernel": "SoS", "tags": [] @@ -75,6 +84,7 @@ }, { "cell_type": "markdown", + "id": "653513ea", "metadata": { "jp-MarkdownHeadingCollapsed": true, "kernel": "SoS", @@ -101,6 +111,7 @@ }, { "cell_type": "markdown", + "id": "fdd87fe2", "metadata": { "kernel": "SoS" }, @@ -112,6 +123,7 @@ }, { "cell_type": "markdown", + "id": "252c5b28", "metadata": { "kernel": "SoS" }, @@ -123,6 +135,7 @@ }, { "cell_type": "markdown", + "id": "a60b60a3", "metadata": { "kernel": "SoS" }, @@ -133,6 +146,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4456ee4f", "metadata": { "kernel": "Bash" }, @@ -152,6 +166,7 @@ }, { "cell_type": "markdown", + "id": "2f00ed49", "metadata": { "kernel": "SoS" }, @@ -163,6 +178,7 @@ }, { "cell_type": "markdown", + "id": "4450f96c", "metadata": { "kernel": "SoS" }, @@ -173,6 +189,7 @@ { "cell_type": "code", "execution_count": null, + "id": "0eb4ae86", "metadata": { "kernel": "Bash" }, @@ -191,6 +208,7 @@ }, { "cell_type": "markdown", + "id": "bd843827", "metadata": { "kernel": "SoS" }, @@ -201,6 +219,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a2b7e811", "metadata": { "kernel": "SoS" }, @@ -211,6 +230,7 @@ }, { "cell_type": "markdown", + "id": "1c4c700d", "metadata": { "kernel": "SoS" }, @@ -316,6 +336,7 @@ }, { "cell_type": "markdown", + "id": "0e50e6fc", "metadata": { "kernel": "SoS" }, @@ -326,6 +347,7 @@ { "cell_type": "code", "execution_count": null, + "id": "913f1b3f", "metadata": { "kernel": "SoS" }, @@ -460,6 +482,7 @@ { "cell_type": "code", "execution_count": null, + "id": "83a4d080", "metadata": { "kernel": "SoS" }, @@ -654,24 +677,336 @@ { "cell_type": "code", "execution_count": null, + "id": "888c1025", "metadata": { "kernel": "SoS" }, "outputs": [], "source": [ - "[qtl_dataset_construct]\n# Build one pecotmr::QtlDataset from the phenotype manifest derived from\n# regional_data, replacing the per-region file I/O in quantile_qtl_twas_weight.\n# Writes a manifest TSV (one row per context: cond, path, cov_path) then calls\n# pecotmr::loadQtlDatasetFromManifest() to build and serialize the QtlDataset.\n# Downstream quantile_qtl_twas_weight loads this single RDS and uses the\n# QtlDataset accessors (getGenotypes, getPhenotypes, getPhenotypeCovariates)\n# instead of re-reading phenotype files per region.\ndepends: sos_variable(\"regional_data\"), sos_variable(\"meta_data\")\noutput: f\"{cwd:a}/qtl_dataset/{name}.qtl_dataset.rds\"\n# All code that references shared sos_variables must appear AFTER output: so that\n# SoS backward analysis (which only evaluates up to the output: directive) never\n# tries to execute it with regional_data / meta_data undefined.\nif len(regional_data['data']) == 0:\n from sos.utils import StopInputGroup\n raise StopInputGroup('No phenotype data available to build QtlDataset.')\n\nimport os, pandas as pd\nmanifest_rows = []\nfor _, row in meta_data.iterrows():\n conds = str(row['cond']).split(',')\n paths = str(row['path']).split(',')\n cov_paths = str(row['cov_path']).split(',')\n for cond, pheno_path, cov_path in zip(conds, paths, cov_paths):\n manifest_rows.append({'cond': cond.strip(), 'path': pheno_path.strip(), 'cov_path': cov_path.strip()})\nmanifest_df = pd.DataFrame(manifest_rows, columns=['cond', 'path', 'cov_path']).drop_duplicates(subset=['cond', 'path', 'cov_path'])\nos.makedirs(f\"{cwd:a}/qtl_dataset\", exist_ok=True)\nmanifest_path = f\"{cwd:a}/qtl_dataset/{name}.manifest.tsv\"\nmanifest_df.to_csv(manifest_path, sep='\\t', index=False)\n\ntask: trunk_workers = 1, trunk_size = 1, walltime = walltime, mem = mem, cores = numThreads, tags = f\"{step_name}_{_output:bn}\"\nR: expand = '${ }', stdout = f\"{_output:n}.stdout\", stderr = f\"{_output:n}.stderr\", container = container\n library(pecotmr)\n manifest_path <- \"${manifest_path}\"\n geno_path <- \"${genoFile:a}\"\n genotype_prefix <- if (grepl(\"\\\\.bed$\", geno_path)) {\n tools::file_path_sans_ext(geno_path)\n } else {\n geno_list <- read.table(geno_path, header = TRUE, stringsAsFactors = FALSE)\n tools::file_path_sans_ext(as.character(geno_list[1, 2]))\n }\n keep_samples_vec <- character(0)\n if (${\"TRUE\" if keep_samples.is_file() else \"FALSE\"}) {\n keep_samples_vec <- unique(trimws(unlist(strsplit(\n readLines(${keep_samples:ar}), \"[[:space:]]+\"))))\n }\n qd <- loadQtlDatasetFromManifest(\n manifest = manifest_path,\n study = \"${name}\",\n genotypes = genotype_prefix,\n mafCutoff = ${maf},\n macCutoff = ${mac},\n imissCutoff = ${imiss},\n keepIndel = ${\"TRUE\" if indel else \"FALSE\"},\n keepSamples = keep_samples_vec,\n keepVariants = character(0),\n transposeCovariates = TRUE,\n scaleResiduals = FALSE)\n dir.create(dirname(\"${_output:a}\"), showWarnings = FALSE, recursive = TRUE)\n saveRDS(qd, \"${_output:a}\")\n message(paste0(\"QtlDataset built: \", length(getContexts(qd)),\n \" context(s) saved to ${_output:a}\"))\n" + "[qtl_dataset_construct]\n", + "# Build one pecotmr::QtlDataset from the phenotype manifest derived from\n", + "# regional_data, replacing the per-region file I/O in quantile_qtl_twas_weight.\n", + "# Writes a manifest TSV (one row per context: cond, path, cov_path) then calls\n", + "# pecotmr::loadQtlDatasetFromManifest() to build and serialize the QtlDataset.\n", + "# Downstream quantile_qtl_twas_weight loads this single RDS and uses the\n", + "# QtlDataset accessors (getGenotypes, getPhenotypes, getPhenotypeCovariates)\n", + "# instead of re-reading phenotype files per region.\n", + "depends: sos_variable(\"regional_data\"), sos_variable(\"meta_data\")\n", + "output: f\"{cwd:a}/qtl_dataset/{name}.qtl_dataset.rds\"\n", + "# All code that references shared sos_variables must appear AFTER output: so that\n", + "# SoS backward analysis (which only evaluates up to the output: directive) never\n", + "# tries to execute it with regional_data / meta_data undefined.\n", + "if len(regional_data['data']) == 0:\n", + " from sos.utils import StopInputGroup\n", + " raise StopInputGroup('No phenotype data available to build QtlDataset.')\n", + "\n", + "import os, pandas as pd\n", + "manifest_rows = []\n", + "for _, row in meta_data.iterrows():\n", + " conds = str(row['cond']).split(',')\n", + " paths = str(row['path']).split(',')\n", + " cov_paths = str(row['cov_path']).split(',')\n", + " for cond, pheno_path, cov_path in zip(conds, paths, cov_paths):\n", + " manifest_rows.append({'cond': cond.strip(), 'path': pheno_path.strip(), 'cov_path': cov_path.strip()})\n", + "manifest_df = pd.DataFrame(manifest_rows, columns=['cond', 'path', 'cov_path']).drop_duplicates(subset=['cond', 'path', 'cov_path'])\n", + "os.makedirs(f\"{cwd:a}/qtl_dataset\", exist_ok=True)\n", + "manifest_path = f\"{cwd:a}/qtl_dataset/{name}.manifest.tsv\"\n", + "manifest_df.to_csv(manifest_path, sep='\\t', index=False)\n", + "\n", + "task: trunk_workers = 1, trunk_size = 1, walltime = walltime, mem = mem, cores = numThreads, tags = f\"{step_name}_{_output:bn}\"\n", + "R: expand = '${ }', stdout = f\"{_output:n}.stdout\", stderr = f\"{_output:n}.stderr\", container = container\n", + " library(pecotmr)\n", + " manifest_path <- \"${manifest_path}\"\n", + " geno_path <- \"${genoFile:a}\"\n", + " genotype_prefix <- if (grepl(\"\\\\.bed$\", geno_path)) {\n", + " tools::file_path_sans_ext(geno_path)\n", + " } else {\n", + " geno_list <- read.table(geno_path, header = TRUE, stringsAsFactors = FALSE)\n", + " tools::file_path_sans_ext(as.character(geno_list[1, 2]))\n", + " }\n", + " keep_samples_vec <- character(0)\n", + " if (${\"TRUE\" if keep_samples.is_file() else \"FALSE\"}) {\n", + " keep_samples_vec <- unique(trimws(unlist(strsplit(\n", + " readLines(${keep_samples:ar}), \"[[:space:]]+\"))))\n", + " }\n", + " qd <- loadQtlDatasetFromManifest(\n", + " manifest = manifest_path,\n", + " study = \"${name}\",\n", + " genotypes = genotype_prefix,\n", + " mafCutoff = ${maf},\n", + " macCutoff = ${mac},\n", + " imissCutoff = ${imiss},\n", + " keepIndel = ${\"TRUE\" if indel else \"FALSE\"},\n", + " keepSamples = keep_samples_vec,\n", + " keepVariants = character(0),\n", + " transposeCovariates = TRUE,\n", + " scaleResiduals = FALSE)\n", + " dir.create(dirname(\"${_output:a}\"), showWarnings = FALSE, recursive = TRUE)\n", + " saveRDS(qd, \"${_output:a}\")\n", + " message(paste0(\"QtlDataset built: \", length(getContexts(qd)),\n", + " \" context(s) saved to ${_output:a}\"))\n" ] }, { "cell_type": "code", "execution_count": null, + "id": "d73e3591", "metadata": { "kernel": "SoS", "tags": [] }, "outputs": [], "source": [ - "[quantile_qtl_twas_weight]\ndepends: sos_variable(\"regional_data\"), sos_variable(\"meta_data\"), path(f\"{cwd:a}/qtl_dataset/{name}.qtl_dataset.rds\")\n# Check if both 'data' and 'meta_info' are empty lists\nif len(regional_data['data']) == 0:\n from sos.utils import StopInputGroup\n raise StopInputGroup(f'No data for region(s): {region_name}')\n\nmeta_info = regional_data[\"meta_info\"]\ninput: path(f\"{cwd:a}/qtl_dataset/{name}.qtl_dataset.rds\"), for_each = \"meta_info\"\noutput: f'{cwd:a}/{step_name}/{name}.{_meta_info[0].split(\":\")[0]}_{_meta_info[2]}.univariate_qr_twas_weights.rds'\ntask: trunk_workers = 1, trunk_size = job_size, walltime = walltime, mem = mem, cores = numThreads, tags = f'{step_name}_{_output:bn}'\nR: expand = '${ }', stdout = f\"{_output:n}.stdout\", stderr = f\"{_output:n}.stderr\", container = container\n options(warn=1)\n library(pecotmr)\n library(qQTLR)\n library(SummarizedExperiment)\n library(GenomicRanges)\n library(readr)\n start_time_total <- proc.time()\n\n # Load the pre-built QtlDataset (built by qtl_dataset_construct)\n qd <- readRDS(${_input:ar})\n\n conditions = c(${\",\".join(['\"%s\"' % x for x in _meta_info[4:]])})\n region = ${(\"'%s'\" % _meta_info[0]) if int(_meta_info[0].split('-')[-1])>0 else 'NULL'} # if the end position is zero return NULL\n association_window = \"${_meta_info[1]}\"\n extract_region_name = list(${\",\".join([(\"c('\"+x+\"')\") if isinstance(x, str) else (\"c\"+ str(x)) for x in _meta_info[3]])})\n phenotype_header = ${\"4\" if int(_meta_info[0].split('-')[-1])>0 else \"1\"}\n region_name_col = ${\"4\" if int(_meta_info[0].split('-')[-1])>0 else \"1\"}\n\n # Parse association window as GRanges for getGenotypes / getMaf\n assoc_parts <- strsplit(\"${_meta_info[1]}\", \"[:-]\")[[1]]\n assoc_gr <- GenomicRanges::GRanges(\n seqnames = assoc_parts[1],\n ranges = IRanges::IRanges(start = as.integer(assoc_parts[2]),\n end = as.integer(assoc_parts[3]))\n )\n\n # Extract genotype block once; shared across all contexts for this region\n X_full <- pecotmr::getGenotypes(qd, region = assoc_gr) # samples x variants\n maf_full <- pecotmr::getMaf(qd, region = assoc_gr) # named by variant ID\n\n # Normalize variant IDs to colon format (chr:pos:ref:alt) to match\n # the keep_variants file which uses colons rather than underscores.\n colnames(X_full) <- pecotmr:::normalizeVariantId(colnames(X_full))\n names(maf_full) <- pecotmr:::normalizeVariantId(names(maf_full))\n\n if (ncol(X_full) == 0) {\n message(\"No SNPs in association window ${_meta_info[1]}\")\n saveRDS(list(${_meta_info[2]} = \"No SNPs in window\"), ${_output:ar}, compress='xz')\n quit(save=\"no\")\n }\n\n # extract subset of samples\n keep_samples = NULL\n if (${\"TRUE\" if keep_samples.is_file() else \"FALSE\"}) {\n keep_samples = unlist(strsplit(readLines(${keep_samples:ar}), \"\\\\s+\"))\n message(paste(length(keep_samples), \"samples are selected to be loaded for analysis\"))\n }\n\n # Load variant filter list if provided\n keep_variants_list = NULL\n keep_variants_per_context = FALSE\n keep_variants_region_data = NULL\n if (${\"TRUE\" if keep_variants.is_file() else \"FALSE\"}) {\n keep_variants_path = ${keep_variants:ar}\n\n # Read input: RDS or text/tsv\n if (grepl(\"\\\\.rds$\", keep_variants_path, ignore.case = TRUE)) {\n keep_variants_data = readRDS(keep_variants_path)\n # Ensure it's a data.frame/data.table\n if (!is.data.frame(keep_variants_data)) {\n # If RDS contains a vector, treat as simple variant list\n keep_variants_list = as.character(keep_variants_data)\n keep_variants_list = keep_variants_list[keep_variants_list != \"\"]\n message(paste(length(keep_variants_list), \"variants are specified to be kept for analysis (from RDS vector)\"))\n keep_variants_data = NULL\n }\n } else {\n keep_variants_data = data.table::fread(keep_variants_path, header = TRUE)\n }\n\n if (!is.null(keep_variants_data)) {\n # Fix column name if first column is \"#chr\"\n if (\"#chr\" %in% names(keep_variants_data)) {\n names(keep_variants_data)[names(keep_variants_data) == \"#chr\"] <- \"chr\"\n }\n\n if (ncol(keep_variants_data) == 1) {\n # Single column: treat as simple variant_id list (old behavior)\n keep_variants_list = trimws(as.character(keep_variants_data[[1]]))\n keep_variants_list = keep_variants_list[keep_variants_list != \"\"]\n message(paste(length(keep_variants_list), \"variants are specified to be kept for analysis\"))\n } else if (all(c(\"context\", \"molecular_trait_object_id\", \"variant_id\") %in% names(keep_variants_data))) {\n # Multi-column format: filter by context and molecular_trait_object_id\n current_region = \"${_meta_info[2]}\"\n original_region_ids = c(${\",\".join([(\"'%s'\" % x) if isinstance(x, str) else \",\".join([(\"'%s'\" % i) for i in x]) for x in _meta_info[3]])})\n all_region_ids = unique(c(current_region, original_region_ids))\n current_contexts = conditions\n keep_variants_region_data = keep_variants_data[\n keep_variants_data$molecular_trait_object_id %in% all_region_ids &\n keep_variants_data$context %in% current_contexts, ]\n keep_variants_list = unique(as.character(keep_variants_region_data$variant_id))\n keep_variants_list = keep_variants_list[keep_variants_list != \"\"]\n keep_variants_per_context = TRUE\n message(paste(length(keep_variants_list), \"variants matched for region\", current_region,\n \"across\", length(current_contexts), \"contexts (per-context filtering enabled)\"))\n if (length(keep_variants_list) == 0) {\n message(\"No matching variants found for this region, marginal coef will be skipped for contexts without variants\")\n }\n } else {\n # Fallback: treat first column as variant_id list\n keep_variants_list = trimws(as.character(keep_variants_data[[1]]))\n keep_variants_list = keep_variants_list[keep_variants_list != \"\"]\n message(paste(length(keep_variants_list), \"variants are specified to be kept for analysis (from first column)\"))\n }\n }\n }\n\n # setup univariate analysis pipeline options\n if (\"${_meta_info[2]}\" != \"${_meta_info[3]}\") {\n region_name = c(\"${_meta_info[2]}\", c(${\",\".join([(\"c('\"+x+\"')\") if isinstance(x, str) else (\"c\"+ str(x)) for x in _meta_info[3]])}))\n } else {\n region_name = \"${_meta_info[2]}\"\n }\n\n region_info = list(region_coord=parseRegion(\"${_meta_info[0]}\"), grange=parseRegion(\"${_meta_info[1]}\"), region_name=region_name)\n\n fitted = list()\n condition_names = vector()\n r = 1L\n while (r <= length(conditions)) {\n ctx <- conditions[r]\n original_ids <- extract_region_name[[r]]\n\n # Get phenotype for this context and gene/protein IDs from QtlDataset\n se_pheno <- tryCatch(\n pecotmr::getPhenotypes(qd, contexts = ctx, traitId = original_ids),\n error = function(e) {\n message(\"getPhenotypes error for ctx=\", ctx, \": \", e$message)\n NULL\n })\n if (is.null(se_pheno)) {\n message(\"Skipping context \", ctx, \" (trait not found in QtlDataset)\")\n r <- r + 1L\n next\n }\n\n Y_mat <- t(SummarizedExperiment::assay(se_pheno, 1L)) # samples x traits\n Z_mat <- pecotmr::getPhenotypeCovariates(qd, contexts = ctx)[[ctx]] # samples x covariates\n\n # Intersect samples among genotypes, phenotypes, and covariates\n common_samps <- intersect(rownames(X_full),\n intersect(rownames(Y_mat), rownames(Z_mat)))\n if (!is.null(keep_samples)) {\n common_samps <- intersect(common_samps, keep_samples)\n }\n if (length(common_samps) == 0) {\n message(\"No common samples for context \", ctx, \", skipping.\")\n r <- r + 1L\n next\n }\n X <- X_full[common_samps, , drop = FALSE]\n Y <- Y_mat[common_samps, , drop = FALSE]\n Z <- Z_mat[common_samps, , drop = FALSE]\n maf <- maf_full # MAF named by variant ID, same for all contexts\n\n if (is.null(dim(Y))) {\n Y <- matrix(Y, nrow = length(Y), ncol = 1L)\n }\n colnames(Y) <- original_ids\n\n # Update condition names\n ctx_name <- conditions[r]\n new_col_names <- extract_region_name[[r]]\n if (!identical(ctx_name, new_col_names)) {\n new_names <- paste(ctx_name, new_col_names, sep = \"_\")\n } else {\n new_names <- new_col_names\n }\n\n column_results <- lapply(1:ncol(Y), function(i) {\n Y_col <- matrix(Y[,i], ncol=1)\n colnames(Y_col) <- colnames(Y)[i]\n # Determine per-context variant list\n current_keep_variants = keep_variants_list\n if (keep_variants_per_context && !is.null(keep_variants_region_data)) {\n ctx_filtered = keep_variants_region_data[keep_variants_region_data$context == ctx_name, ]\n if (nrow(ctx_filtered) > 0) {\n current_keep_variants = unique(as.character(ctx_filtered$variant_id))\n current_keep_variants = current_keep_variants[current_keep_variants != \"\"]\n message(paste(length(current_keep_variants), \"variants for context\", ctx_name))\n } else {\n current_keep_variants = character(0)\n message(paste(\"No context-specific variants for\", ctx_name, \"- skipping marginal coef fitting\"))\n }\n }\n\n qr_results = quantile_twas_weight_pipeline(\n X = X,\n Y = Y_col,\n Z = Z,\n ld_reference_meta_file=${('\"%s\"' % ld_reference_meta_file) if not ld_reference_meta_file.is_dir() else \"NULL\"},\n maf = maf,\n twas_maf_cutoff = ${min_twas_maf},\n region_id = paste0(colnames(Y_col), \"_\", ctx_name),\n quantile_qtl_tau_list = seq(0.05, 0.95, by = 0.05),\n quantile_twas_tau_list = seq(0.01, 0.99, by = 0.01),\n screen_method = \"${screen_method}\",\n screen_threshold = ${screen_threshold},\n screen_significant = ${\"TRUE\" if screen_significant else \"FALSE\"},\n pre_filter_by_pqr = ${\"TRUE\" if pre_filter_by_pqr else \"FALSE\"},\n initial_corr_filter_cutoff = ${initial_corr_filter_cutoff},\n full_rank_corr_filter_cutoff = ${full_rank_corr_filter_cutoff},\n keep_variants = current_keep_variants,\n marginal_beta_calculate = ${\"TRUE\" if marginal_beta_calculate else \"FALSE\"},\n twas_weight_calculate = ${\"TRUE\" if twas_weight_calculate else \"FALSE\"},\n qrank_screen_calculate = ${\"TRUE\" if qrank_screen_calculate else \"FALSE\"},\n vqtl_calculate = ${\"TRUE\" if vqtl_calculate else \"FALSE\"}\n )\n\n if (!is.null(qr_results$message)) {\n message(qr_results$message)\n }\n\n qr_results$region_info = region_info\n qr_results$maf = maf\n\n return(qr_results)\n })\n\n fitted <- c(fitted, column_results)\n condition_names <- c(condition_names, new_names)\n\n if (length(new_names) > 0) {\n message(\"Analysis completed for: \", paste(new_names, collapse=\",\"))\n }\n\n r = r + 1L\n }\n\n # Set names for the final results\n if (length(fitted) > 0) {\n names(fitted) <- condition_names\n }\n\n saveRDS(list(\"${_meta_info[2]}\" = fitted), ${_output:ar}, compress='xz')\n end_time_total <- proc.time()\n total_time <- end_time_total - start_time_total\n print(total_time)\n" + "[quantile_qtl_twas_weight]\n", + "depends: sos_variable(\"regional_data\"), sos_variable(\"meta_data\"), path(f\"{cwd:a}/qtl_dataset/{name}.qtl_dataset.rds\")\n", + "# Check if both 'data' and 'meta_info' are empty lists\n", + "if len(regional_data['data']) == 0:\n", + " from sos.utils import StopInputGroup\n", + " raise StopInputGroup(f'No data for region(s): {region_name}')\n", + "\n", + "meta_info = regional_data[\"meta_info\"]\n", + "input: path(f\"{cwd:a}/qtl_dataset/{name}.qtl_dataset.rds\"), for_each = \"meta_info\"\n", + "output: f'{cwd:a}/{step_name}/{name}.{_meta_info[0].split(\":\")[0]}_{_meta_info[2]}.univariate_qr_twas_weights.rds'\n", + "task: trunk_workers = 1, trunk_size = job_size, walltime = walltime, mem = mem, cores = numThreads, tags = f'{step_name}_{_output:bn}'\n", + "R: expand = '${ }', stdout = f\"{_output:n}.stdout\", stderr = f\"{_output:n}.stderr\", container = container\n", + " options(warn=1)\n", + " library(pecotmr)\n", + " library(qQTLR)\n", + " library(SummarizedExperiment)\n", + " library(GenomicRanges)\n", + " library(readr)\n", + " start_time_total <- proc.time()\n", + "\n", + " # Load the pre-built QtlDataset (built by qtl_dataset_construct)\n", + " qd <- readRDS(${_input:ar})\n", + "\n", + " conditions = c(${\",\".join(['\"%s\"' % x for x in _meta_info[4:]])})\n", + " region = ${(\"'%s'\" % _meta_info[0]) if int(_meta_info[0].split('-')[-1])>0 else 'NULL'} # if the end position is zero return NULL\n", + " association_window = \"${_meta_info[1]}\"\n", + " extract_region_name = list(${\",\".join([(\"c('\"+x+\"')\") if isinstance(x, str) else (\"c\"+ str(x)) for x in _meta_info[3]])})\n", + " phenotype_header = ${\"4\" if int(_meta_info[0].split('-')[-1])>0 else \"1\"}\n", + " region_name_col = ${\"4\" if int(_meta_info[0].split('-')[-1])>0 else \"1\"}\n", + "\n", + " # Parse association window as GRanges for getGenotypes / getMaf\n", + " assoc_parts <- strsplit(\"${_meta_info[1]}\", \"[:-]\")[[1]]\n", + " assoc_gr <- GenomicRanges::GRanges(\n", + " seqnames = assoc_parts[1],\n", + " ranges = IRanges::IRanges(start = as.integer(assoc_parts[2]),\n", + " end = as.integer(assoc_parts[3]))\n", + " )\n", + "\n", + " # Extract genotype block once; shared across all contexts for this region\n", + " X_full <- pecotmr::getGenotypes(qd, region = assoc_gr) # samples x variants\n", + " maf_full <- pecotmr::getMaf(qd, region = assoc_gr) # named by variant ID\n", + "\n", + " # Normalize variant IDs to colon format (chr:pos:ref:alt) to match\n", + " # the keep_variants file which uses colons rather than underscores.\n", + " colnames(X_full) <- pecotmr:::normalizeVariantId(colnames(X_full))\n", + " names(maf_full) <- pecotmr:::normalizeVariantId(names(maf_full))\n", + "\n", + " if (ncol(X_full) == 0) {\n", + " message(\"No SNPs in association window ${_meta_info[1]}\")\n", + " saveRDS(list(${_meta_info[2]} = \"No SNPs in window\"), ${_output:ar}, compress='xz')\n", + " quit(save=\"no\")\n", + " }\n", + "\n", + " # extract subset of samples\n", + " keep_samples = NULL\n", + " if (${\"TRUE\" if keep_samples.is_file() else \"FALSE\"}) {\n", + " keep_samples = unlist(strsplit(readLines(${keep_samples:ar}), \"\\\\s+\"))\n", + " message(paste(length(keep_samples), \"samples are selected to be loaded for analysis\"))\n", + " }\n", + "\n", + " # Load variant filter list if provided\n", + " keep_variants_list = NULL\n", + " keep_variants_per_context = FALSE\n", + " keep_variants_region_data = NULL\n", + " if (${\"TRUE\" if keep_variants.is_file() else \"FALSE\"}) {\n", + " keep_variants_path = ${keep_variants:ar}\n", + "\n", + " # Read input: RDS or text/tsv\n", + " if (grepl(\"\\\\.rds$\", keep_variants_path, ignore.case = TRUE)) {\n", + " keep_variants_data = readRDS(keep_variants_path)\n", + " # Ensure it's a data.frame/data.table\n", + " if (!is.data.frame(keep_variants_data)) {\n", + " # If RDS contains a vector, treat as simple variant list\n", + " keep_variants_list = as.character(keep_variants_data)\n", + " keep_variants_list = keep_variants_list[keep_variants_list != \"\"]\n", + " message(paste(length(keep_variants_list), \"variants are specified to be kept for analysis (from RDS vector)\"))\n", + " keep_variants_data = NULL\n", + " }\n", + " } else {\n", + " keep_variants_data = data.table::fread(keep_variants_path, header = TRUE)\n", + " }\n", + "\n", + " if (!is.null(keep_variants_data)) {\n", + " # Fix column name if first column is \"#chr\"\n", + " if (\"#chr\" %in% names(keep_variants_data)) {\n", + " names(keep_variants_data)[names(keep_variants_data) == \"#chr\"] <- \"chr\"\n", + " }\n", + "\n", + " if (ncol(keep_variants_data) == 1) {\n", + " # Single column: treat as simple variant_id list (old behavior)\n", + " keep_variants_list = trimws(as.character(keep_variants_data[[1]]))\n", + " keep_variants_list = keep_variants_list[keep_variants_list != \"\"]\n", + " message(paste(length(keep_variants_list), \"variants are specified to be kept for analysis\"))\n", + " } else if (all(c(\"context\", \"molecular_trait_object_id\", \"variant_id\") %in% names(keep_variants_data))) {\n", + " # Multi-column format: filter by context and molecular_trait_object_id\n", + " current_region = \"${_meta_info[2]}\"\n", + " original_region_ids = c(${\",\".join([(\"'%s'\" % x) if isinstance(x, str) else \",\".join([(\"'%s'\" % i) for i in x]) for x in _meta_info[3]])})\n", + " all_region_ids = unique(c(current_region, original_region_ids))\n", + " current_contexts = conditions\n", + " keep_variants_region_data = keep_variants_data[\n", + " keep_variants_data$molecular_trait_object_id %in% all_region_ids &\n", + " keep_variants_data$context %in% current_contexts, ]\n", + " keep_variants_list = unique(as.character(keep_variants_region_data$variant_id))\n", + " keep_variants_list = keep_variants_list[keep_variants_list != \"\"]\n", + " keep_variants_per_context = TRUE\n", + " message(paste(length(keep_variants_list), \"variants matched for region\", current_region,\n", + " \"across\", length(current_contexts), \"contexts (per-context filtering enabled)\"))\n", + " if (length(keep_variants_list) == 0) {\n", + " message(\"No matching variants found for this region, marginal coef will be skipped for contexts without variants\")\n", + " }\n", + " } else {\n", + " # Fallback: treat first column as variant_id list\n", + " keep_variants_list = trimws(as.character(keep_variants_data[[1]]))\n", + " keep_variants_list = keep_variants_list[keep_variants_list != \"\"]\n", + " message(paste(length(keep_variants_list), \"variants are specified to be kept for analysis (from first column)\"))\n", + " }\n", + " }\n", + " }\n", + "\n", + " # setup univariate analysis pipeline options\n", + " if (\"${_meta_info[2]}\" != \"${_meta_info[3]}\") {\n", + " region_name = c(\"${_meta_info[2]}\", c(${\",\".join([(\"c('\"+x+\"')\") if isinstance(x, str) else (\"c\"+ str(x)) for x in _meta_info[3]])}))\n", + " } else {\n", + " region_name = \"${_meta_info[2]}\"\n", + " }\n", + "\n", + " region_info = list(region_coord=parseRegion(\"${_meta_info[0]}\"), grange=parseRegion(\"${_meta_info[1]}\"), region_name=region_name)\n", + "\n", + " fitted = list()\n", + " condition_names = vector()\n", + " r = 1L\n", + " while (r <= length(conditions)) {\n", + " ctx <- conditions[r]\n", + " original_ids <- extract_region_name[[r]]\n", + "\n", + " # Get phenotype for this context and gene/protein IDs from QtlDataset\n", + " se_pheno <- tryCatch(\n", + " pecotmr::getPhenotypes(qd, contexts = ctx, traitId = original_ids),\n", + " error = function(e) {\n", + " message(\"getPhenotypes error for ctx=\", ctx, \": \", e$message)\n", + " NULL\n", + " })\n", + " if (is.null(se_pheno)) {\n", + " message(\"Skipping context \", ctx, \" (trait not found in QtlDataset)\")\n", + " r <- r + 1L\n", + " next\n", + " }\n", + "\n", + " Y_mat <- t(SummarizedExperiment::assay(se_pheno, 1L)) # samples x traits\n", + " Z_mat <- pecotmr::getPhenotypeCovariates(qd, contexts = ctx)[[ctx]] # samples x covariates\n", + "\n", + " # Intersect samples among genotypes, phenotypes, and covariates\n", + " common_samps <- intersect(rownames(X_full),\n", + " intersect(rownames(Y_mat), rownames(Z_mat)))\n", + " if (!is.null(keep_samples)) {\n", + " common_samps <- intersect(common_samps, keep_samples)\n", + " }\n", + " if (length(common_samps) == 0) {\n", + " message(\"No common samples for context \", ctx, \", skipping.\")\n", + " r <- r + 1L\n", + " next\n", + " }\n", + " X <- X_full[common_samps, , drop = FALSE]\n", + " Y <- Y_mat[common_samps, , drop = FALSE]\n", + " Z <- Z_mat[common_samps, , drop = FALSE]\n", + " maf <- maf_full # MAF named by variant ID, same for all contexts\n", + "\n", + " if (is.null(dim(Y))) {\n", + " Y <- matrix(Y, nrow = length(Y), ncol = 1L)\n", + " }\n", + " colnames(Y) <- original_ids\n", + "\n", + " # Update condition names\n", + " ctx_name <- conditions[r]\n", + " new_col_names <- extract_region_name[[r]]\n", + " if (!identical(ctx_name, new_col_names)) {\n", + " new_names <- paste(ctx_name, new_col_names, sep = \"_\")\n", + " } else {\n", + " new_names <- new_col_names\n", + " }\n", + "\n", + " column_results <- lapply(1:ncol(Y), function(i) {\n", + " Y_col <- matrix(Y[,i], ncol=1)\n", + " colnames(Y_col) <- colnames(Y)[i]\n", + " # Determine per-context variant list\n", + " current_keep_variants = keep_variants_list\n", + " if (keep_variants_per_context && !is.null(keep_variants_region_data)) {\n", + " ctx_filtered = keep_variants_region_data[keep_variants_region_data$context == ctx_name, ]\n", + " if (nrow(ctx_filtered) > 0) {\n", + " current_keep_variants = unique(as.character(ctx_filtered$variant_id))\n", + " current_keep_variants = current_keep_variants[current_keep_variants != \"\"]\n", + " message(paste(length(current_keep_variants), \"variants for context\", ctx_name))\n", + " } else {\n", + " current_keep_variants = character(0)\n", + " message(paste(\"No context-specific variants for\", ctx_name, \"- skipping marginal coef fitting\"))\n", + " }\n", + " }\n", + "\n", + " qr_results = quantile_twas_weight_pipeline(\n", + " X = X,\n", + " Y = Y_col,\n", + " Z = Z,\n", + " ld_reference_meta_file=${('\"%s\"' % ld_reference_meta_file) if not ld_reference_meta_file.is_dir() else \"NULL\"},\n", + " maf = maf,\n", + " twas_maf_cutoff = ${min_twas_maf},\n", + " region_id = paste0(colnames(Y_col), \"_\", ctx_name),\n", + " quantile_qtl_tau_list = seq(0.05, 0.95, by = 0.05),\n", + " quantile_twas_tau_list = seq(0.01, 0.99, by = 0.01),\n", + " screen_method = \"${screen_method}\",\n", + " screen_threshold = ${screen_threshold},\n", + " screen_significant = ${\"TRUE\" if screen_significant else \"FALSE\"},\n", + " pre_filter_by_pqr = ${\"TRUE\" if pre_filter_by_pqr else \"FALSE\"},\n", + " initial_corr_filter_cutoff = ${initial_corr_filter_cutoff},\n", + " full_rank_corr_filter_cutoff = ${full_rank_corr_filter_cutoff},\n", + " keep_variants = current_keep_variants,\n", + " marginal_beta_calculate = ${\"TRUE\" if marginal_beta_calculate else \"FALSE\"},\n", + " twas_weight_calculate = ${\"TRUE\" if twas_weight_calculate else \"FALSE\"},\n", + " qrank_screen_calculate = ${\"TRUE\" if qrank_screen_calculate else \"FALSE\"},\n", + " vqtl_calculate = ${\"TRUE\" if vqtl_calculate else \"FALSE\"}\n", + " )\n", + "\n", + " if (!is.null(qr_results$message)) {\n", + " message(qr_results$message)\n", + " }\n", + "\n", + " qr_results$region_info = region_info\n", + " qr_results$maf = maf\n", + "\n", + " return(qr_results)\n", + " })\n", + "\n", + " fitted <- c(fitted, column_results)\n", + " condition_names <- c(condition_names, new_names)\n", + "\n", + " if (length(new_names) > 0) {\n", + " message(\"Analysis completed for: \", paste(new_names, collapse=\",\"))\n", + " }\n", + "\n", + " r = r + 1L\n", + " }\n", + "\n", + " # Set names for the final results\n", + " if (length(fitted) > 0) {\n", + " names(fitted) <- condition_names\n", + " }\n", + "\n", + " saveRDS(list(\"${_meta_info[2]}\" = fitted), ${_output:ar}, compress='xz')\n", + " end_time_total <- proc.time()\n", + " total_time <- end_time_total - start_time_total\n", + " print(total_time)\n" ] } ], @@ -717,5 +1052,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/data_preprocessing/covariate/covariate_formatting.ipynb b/code/SoS/data_preprocessing/covariate/covariate_formatting.ipynb index 9209d49ed..e3d7d08ac 100644 --- a/code/SoS/data_preprocessing/covariate/covariate_formatting.ipynb +++ b/code/SoS/data_preprocessing/covariate/covariate_formatting.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "5f141fdb", "metadata": { "kernel": "SoS", "tags": [] @@ -14,6 +15,7 @@ }, { "cell_type": "markdown", + "id": "dd6c56bd", "metadata": { "kernel": "SoS" }, @@ -32,6 +34,7 @@ }, { "cell_type": "markdown", + "id": "1f66ed0b", "metadata": { "kernel": "SoS" }, @@ -74,6 +77,7 @@ }, { "cell_type": "markdown", + "id": "b7da351e", "metadata": { "kernel": "SoS" }, @@ -94,6 +98,7 @@ }, { "cell_type": "markdown", + "id": "f5a03beb", "metadata": { "kernel": "SoS" }, @@ -103,6 +108,7 @@ }, { "cell_type": "markdown", + "id": "400ddb1f", "metadata": { "kernel": "SoS" }, @@ -114,6 +120,7 @@ }, { "cell_type": "markdown", + "id": "729e85d9", "metadata": { "kernel": "SoS", "tags": [] @@ -125,6 +132,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e78e798b", "metadata": { "kernel": "Bash", "vscode": { @@ -144,6 +152,7 @@ }, { "cell_type": "markdown", + "id": "af3c9395", "metadata": { "kernel": "SoS" }, @@ -154,6 +163,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e7faa9c3", "metadata": { "kernel": "SoS" }, @@ -164,6 +174,7 @@ }, { "cell_type": "markdown", + "id": "f8dd7601", "metadata": { "kernel": "SoS" }, @@ -222,6 +233,7 @@ }, { "cell_type": "markdown", + "id": "a8a06a15", "metadata": { "kernel": "SoS" }, @@ -232,6 +244,7 @@ { "cell_type": "code", "execution_count": null, + "id": "44aae970", "metadata": { "kernel": "SoS" }, @@ -260,6 +273,7 @@ { "cell_type": "code", "execution_count": 1, + "id": "f3684834", "metadata": { "kernel": "SoS" }, @@ -300,6 +314,7 @@ { "cell_type": "code", "execution_count": null, + "id": "64477bfe", "metadata": { "kernel": "SoS" }, @@ -334,5 +349,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/data_preprocessing/covariate/covariate_hidden_factor.ipynb b/code/SoS/data_preprocessing/covariate/covariate_hidden_factor.ipynb index b6c78bc5c..334d95b90 100644 --- a/code/SoS/data_preprocessing/covariate/covariate_hidden_factor.ipynb +++ b/code/SoS/data_preprocessing/covariate/covariate_hidden_factor.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "eb7a9017", "metadata": { "kernel": "SoS", "tags": [] @@ -14,6 +15,7 @@ }, { "cell_type": "markdown", + "id": "9be57145", "metadata": { "kernel": "SoS" }, @@ -29,6 +31,7 @@ }, { "cell_type": "markdown", + "id": "3518bd20", "metadata": { "kernel": "SoS" }, @@ -55,6 +58,7 @@ }, { "cell_type": "markdown", + "id": "14ffc85c", "metadata": { "kernel": "SoS" }, @@ -80,6 +84,7 @@ }, { "cell_type": "markdown", + "id": "18e974fc", "metadata": { "kernel": "SoS" }, @@ -91,6 +96,7 @@ }, { "cell_type": "markdown", + "id": "4e8ae0a4", "metadata": { "kernel": "SoS" }, @@ -102,6 +108,7 @@ }, { "cell_type": "markdown", + "id": "bd227eb3", "metadata": { "kernel": "SoS", "tags": [] @@ -113,6 +120,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f3232998", "metadata": { "kernel": "Bash" }, @@ -127,6 +135,7 @@ }, { "cell_type": "markdown", + "id": "4d115e8a", "metadata": { "kernel": "SoS" }, @@ -138,6 +147,7 @@ }, { "cell_type": "markdown", + "id": "1408000f", "metadata": { "kernel": "SoS", "tags": [] @@ -149,6 +159,7 @@ { "cell_type": "code", "execution_count": null, + "id": "bed3eada", "metadata": { "kernel": "Bash" }, @@ -163,6 +174,7 @@ }, { "cell_type": "markdown", + "id": "cf61ef79", "metadata": { "kernel": "SoS" }, @@ -174,6 +186,7 @@ }, { "cell_type": "markdown", + "id": "b7e87d95", "metadata": { "kernel": "SoS", "tags": [] @@ -185,6 +198,7 @@ { "cell_type": "code", "execution_count": null, + "id": "484a5c2d", "metadata": { "kernel": "Bash" }, @@ -200,6 +214,7 @@ }, { "cell_type": "markdown", + "id": "28c39d8a", "metadata": { "collapsed": true, "jupyter": { @@ -215,6 +230,7 @@ { "cell_type": "code", "execution_count": null, + "id": "52dec1fd", "metadata": { "kernel": "SoS" }, @@ -225,6 +241,7 @@ }, { "cell_type": "markdown", + "id": "1bac1318", "metadata": { "kernel": "SoS" }, @@ -303,6 +320,7 @@ }, { "cell_type": "markdown", + "id": "2955cf37", "metadata": { "kernel": "R" }, @@ -313,6 +331,7 @@ { "cell_type": "code", "execution_count": null, + "id": "ce0fea5d", "metadata": { "kernel": "SoS", "tags": [] @@ -344,6 +363,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a4445238", "metadata": { "kernel": "SoS" }, @@ -367,6 +387,7 @@ }, { "cell_type": "markdown", + "id": "11ca4190", "metadata": { "kernel": "SoS" }, @@ -377,6 +398,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9d79e187", "metadata": { "kernel": "SoS" }, @@ -402,6 +424,7 @@ }, { "cell_type": "markdown", + "id": "65adb374", "metadata": { "kernel": "SoS" }, @@ -412,6 +435,7 @@ { "cell_type": "code", "execution_count": null, + "id": "702607b4", "metadata": { "kernel": "SoS" }, @@ -460,6 +484,7 @@ { "cell_type": "code", "execution_count": null, + "id": "177720b5", "metadata": { "kernel": "SoS" }, @@ -479,6 +504,7 @@ }, { "cell_type": "markdown", + "id": "924ea0d2", "metadata": { "kernel": "Bash" }, @@ -532,5 +558,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/data_preprocessing/covariate_preprocessing.ipynb b/code/SoS/data_preprocessing/covariate_preprocessing.ipynb index 0cf6552ea..5fa95f924 100644 --- a/code/SoS/data_preprocessing/covariate_preprocessing.ipynb +++ b/code/SoS/data_preprocessing/covariate_preprocessing.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "25e9e86a", "metadata": { "kernel": "SoS" }, @@ -9,11 +10,11 @@ "# Covariate data preprocessing\n", "\n", "Combine known covariates with genotype principal components and optionally infer hidden factors for xQTL association testing.\n" - ], - "id": "25e9e86a" + ] }, { "cell_type": "markdown", + "id": "c9f10ef3", "metadata": { "kernel": "SoS" }, @@ -23,25 +24,25 @@ "This is the total duration for one complete toy-data route; module-specific timings appear on their respective pages.\n", "\n", "Timing: <3 min (on the toy dataset)\n" - ], - "id": "c9f10ef3" + ] }, { "cell_type": "markdown", + "id": "5d626526", "metadata": { "kernel": "SoS" }, "source": [ "## Overview\n", "\n", - "This mini-protocol walks through construction of an association-ready covariate matrix. [`covariate_formatting.ipynb`](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/covariate/covariate_formatting.html) first merges sample covariates with genotype principal components. [`covariate_hidden_factor.ipynb`](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/covariate/covariate_hidden_factor.html) then residualizes the molecular phenotype against those observed covariates and optionally estimates latent factors using Marchenko–Pastur selection, a user-specified PCA dimension, PEER, or bi-cross-validation (BiCV).\n", + "This mini-protocol walks through construction of an association-ready covariate matrix. [`covariate_formatting.ipynb`](https://statfungen.github.io/xqtl-protocol/covariate-formatting) first merges sample covariates with genotype principal components. [`covariate_hidden_factor.ipynb`](https://statfungen.github.io/xqtl-protocol/covariate-hidden-factor) then residualizes the molecular phenotype against those observed covariates and optionally estimates latent factors using Marchenko–Pastur selection, a user-specified PCA dimension, PEER, or bi-cross-validation (BiCV).\n", "\n", "Run step 1 once, then choose one hidden-factor method from steps 2–5. These alternatives are not a four-step chain. Use the resulting matrix as the covariate input to association testing.\n" - ], - "id": "5d626526" + ] }, { "cell_type": "markdown", + "id": "e88d7b48", "metadata": { "kernel": "SoS" }, @@ -59,24 +60,24 @@ "| Select hidden factors by bi-cross-validation | 1 → 5 | Step 1 inputs; `tests/fixtures/phenotype_formatting/protocol_example.rnaseq.bed.bed.gz` |\n", "\n", "Run only the row matching the intended analysis goal." - ], - "id": "e88d7b48" + ] }, { "cell_type": "markdown", + "id": "0c4ac9c6", "metadata": { "kernel": "SoS" }, "source": [ - "### 1. [Merge observed covariates and genotype PCs](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/covariate/covariate_formatting.html)\n", + "### 1. [Merge observed covariates and genotype PCs](https://statfungen.github.io/xqtl-protocol/covariate-formatting)\n", "\n", "**What it does:** Combine the base covariate table with the selected genotype principal components.\n" - ], - "id": "0c4ac9c6" + ] }, { "cell_type": "code", "execution_count": null, + "id": "f5e2ad67", "metadata": { "kernel": "Bash" }, @@ -89,24 +90,24 @@ " --name protocol_example.covariates.protocol_example.genotype.merged.plink_qc.plink_qc.prune.pca \\\n", " --tol-cov 0.4 \\\n", " --k `awk '$3 < 0.8' output/genotype/genotype_pca/protocol_example.genotype.merged.plink_qc.plink_qc.prune.pca.scree.txt | tail -1 | cut -f 1`\n" - ], - "id": "f5e2ad67" + ] }, { "cell_type": "markdown", + "id": "b0523bf6", "metadata": { "kernel": "SoS" }, "source": [ - "### 2. [Infer PCA factors with Marchenko–Pastur selection](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/covariate/covariate_hidden_factor.html)\n", + "### 2. [Infer PCA factors with Marchenko–Pastur selection](https://statfungen.github.io/xqtl-protocol/covariate-hidden-factor)\n", "\n", "**What it does:** Residualize the phenotype and automatically retain PCA factors above the Marchenko–Pastur noise threshold.\n" - ], - "id": "b0523bf6" + ] }, { "cell_type": "code", "execution_count": null, + "id": "a10fb6ad", "metadata": { "kernel": "Bash" }, @@ -117,24 +118,24 @@ " --phenoFile tests/fixtures/phenotype_formatting/protocol_example.rnaseq.bed.bed.gz \\\n", " --covFile output/covariate/protocol_example.covariates.protocol_example.genotype.merged.plink_qc.plink_qc.prune.pca.gz \\\n", " --mean-impute-missing\n" - ], - "id": "a10fb6ad" + ] }, { "cell_type": "markdown", + "id": "89e6dabb", "metadata": { "kernel": "SoS" }, "source": [ - "### 3. [Infer PEER factors](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/covariate/covariate_hidden_factor.html)\n", + "### 3. [Infer PEER factors](https://statfungen.github.io/xqtl-protocol/covariate-hidden-factor)\n", "\n", "**What it does:** Residualize the phenotype and estimate the requested number of probabilistic PEER factors.\n" - ], - "id": "89e6dabb" + ] }, { "cell_type": "code", "execution_count": null, + "id": "afa6aaa5", "metadata": { "kernel": "Bash" }, @@ -145,24 +146,24 @@ " --phenoFile tests/fixtures/phenotype_formatting/protocol_example.rnaseq.bed.bed.gz \\\n", " --covFile output/covariate/protocol_example.covariates.protocol_example.genotype.merged.plink_qc.plink_qc.prune.pca.gz \\\n", " --N 3\n" - ], - "id": "afa6aaa5" + ] }, { "cell_type": "markdown", + "id": "0fa9e389", "metadata": { "kernel": "SoS" }, "source": [ - "### 4. [Infer configurable PCA factors](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/covariate/covariate_hidden_factor.html)\n", + "### 4. [Infer configurable PCA factors](https://statfungen.github.io/xqtl-protocol/covariate-hidden-factor)\n", "\n", "**What it does:** Residualize the phenotype and estimate PCA factors using the selected dimension rule.\n" - ], - "id": "0fa9e389" + ] }, { "cell_type": "code", "execution_count": null, + "id": "60317089", "metadata": { "kernel": "Bash" }, @@ -174,24 +175,24 @@ " --covFile output/covariate/protocol_example.covariates.protocol_example.genotype.merged.plink_qc.plink_qc.prune.pca.gz \\\n", " --choose_k_method Marchenko \\\n", " --mean-impute-missing\n" - ], - "id": "60317089" + ] }, { "cell_type": "markdown", + "id": "c5d0f178", "metadata": { "kernel": "SoS" }, "source": [ - "### 5. [Infer factors with bi-cross-validation](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/covariate/covariate_hidden_factor.html)\n", + "### 5. [Infer factors with bi-cross-validation](https://statfungen.github.io/xqtl-protocol/covariate-hidden-factor)\n", "\n", "**What it does:** Residualize the phenotype and select latent structure using the BiCV workflow.\n" - ], - "id": "c5d0f178" + ] }, { "cell_type": "code", "execution_count": null, + "id": "a104f40d", "metadata": { "kernel": "Bash" }, @@ -202,11 +203,11 @@ " --phenoFile tests/fixtures/phenotype_formatting/protocol_example.rnaseq.bed.bed.gz \\\n", " --covFile output/covariate/protocol_example.covariates.protocol_example.genotype.merged.plink_qc.plink_qc.prune.pca.gz \\\n", " --N 3\n" - ], - "id": "a104f40d" + ] }, { "cell_type": "markdown", + "id": "5bd240a3", "metadata": { "kernel": "SoS" }, @@ -220,11 +221,11 @@ "| Step 3 | `output/covariate/protocol_example.rnaseq.bed.PEER.gz`; `output/covariate/protocol_example.rnaseq.bed.PEER.diag.pdf`; `output/covariate/protocol_example.rnaseq.bed.PEER_MODEL.hd5` |\n", "| Step 4 | `output/covariate/protocol_example.rnaseq.Marchenko_PC.gz` when `--choose_k_method Marchenko` is used |\n", "| Step 5 | `output/covariate/protocol_example.rnaseq.bed.BiCV.gz` |" - ], - "id": "5bd240a3" + ] }, { "cell_type": "markdown", + "id": "eefe4f67", "metadata": { "kernel": "SoS" }, @@ -234,11 +235,11 @@ "Step 1 produces a sample-by-covariate matrix containing the observed covariates and selected genotype principal components. A selected hidden-factor route adds latent factors estimated after residualizing the molecular phenotype against those observed covariates.\n", "\n", "Use exactly one final matrix in association testing; do not concatenate results from alternative hidden-factor methods.\n" - ], - "id": "eefe4f67" + ] }, { "cell_type": "markdown", + "id": "af8ffbf0", "metadata": { "kernel": "SoS" }, @@ -246,12 +247,12 @@ "## Command Interface\n", "\n", "List the workflows and parameters available in each module used by this mini-protocol.\n" - ], - "id": "af8ffbf0" + ] }, { "cell_type": "code", "execution_count": null, + "id": "cc7345e3", "metadata": { "kernel": "Bash" }, @@ -259,8 +260,7 @@ "source": [ "sos run pipeline/covariate_formatting.ipynb -h\n", "sos run pipeline/covariate_hidden_factor.ipynb -h\n" - ], - "id": "cc7345e3" + ] } ], "metadata": { @@ -299,4 +299,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/data_preprocessing/genotype/GWAS_QC.ipynb b/code/SoS/data_preprocessing/genotype/GWAS_QC.ipynb index b20f55803..29fe73383 100644 --- a/code/SoS/data_preprocessing/genotype/GWAS_QC.ipynb +++ b/code/SoS/data_preprocessing/genotype/GWAS_QC.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "afdb8e65", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "fc738e94", "metadata": { "kernel": "SoS" }, @@ -47,6 +49,7 @@ }, { "cell_type": "markdown", + "id": "a6b3049e", "metadata": { "kernel": "SoS" }, @@ -102,6 +105,7 @@ }, { "cell_type": "markdown", + "id": "468bb89e", "metadata": { "kernel": "SoS" }, @@ -152,6 +156,7 @@ }, { "cell_type": "markdown", + "id": "df9fa532", "metadata": { "kernel": "SoS", "tags": [] @@ -162,17 +167,19 @@ }, { "cell_type": "markdown", + "id": "d25e1487", "metadata": { "kernel": "SoS" }, "source": [ "### Genotype QC of the protocol example data\n", "\n", - "The `chr1_chr6` set was merged from the `chr1` and `chr6` data with the `merge_plink` command in [genotype formatting](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/genotype_formatting.html). The steps below run in order, each consuming what the earlier ones produced." + "The `chr1_chr6` set was merged from the `chr1` and `chr6` data with the `merge_plink` command in [genotype formatting](https://statfungen.github.io/xqtl-protocol/genotype-formatting). The steps below run in order, each consuming what the earlier ones produced." ] }, { "cell_type": "markdown", + "id": "c501ebe9", "metadata": { "kernel": "SoS" }, @@ -184,6 +191,7 @@ }, { "cell_type": "markdown", + "id": "273d6721", "metadata": { "kernel": "SoS" }, @@ -194,6 +202,7 @@ { "cell_type": "code", "execution_count": null, + "id": "28b1a68b", "metadata": { "jp-MarkdownHeadingCollapsed": true, "kernel": "Bash", @@ -213,6 +222,7 @@ }, { "cell_type": "markdown", + "id": "e1efd1b8", "metadata": { "kernel": "SoS" }, @@ -224,6 +234,7 @@ }, { "cell_type": "markdown", + "id": "b723c081", "metadata": { "kernel": "SoS" }, @@ -234,6 +245,7 @@ { "cell_type": "code", "execution_count": null, + "id": "49c13f75", "metadata": { "kernel": "Bash", "vscode": { @@ -251,6 +263,7 @@ }, { "cell_type": "markdown", + "id": "7444d15e", "metadata": { "kernel": "SoS", "tags": [] @@ -263,6 +276,7 @@ }, { "cell_type": "markdown", + "id": "32b8a20b", "metadata": { "kernel": "SoS" }, @@ -273,6 +287,7 @@ { "cell_type": "code", "execution_count": null, + "id": "1d44e7c4", "metadata": { "jp-MarkdownHeadingCollapsed": true, "kernel": "Bash", @@ -289,6 +304,7 @@ }, { "cell_type": "markdown", + "id": "0eb2e55f", "metadata": { "kernel": "SoS" }, @@ -300,6 +316,7 @@ }, { "cell_type": "markdown", + "id": "fc6cfda9", "metadata": { "kernel": "SoS" }, @@ -310,6 +327,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e651d0d9", "metadata": { "kernel": "Bash", "vscode": { @@ -327,6 +345,7 @@ }, { "cell_type": "markdown", + "id": "02075066", "metadata": { "kernel": "SoS", "tags": [] @@ -339,6 +358,7 @@ }, { "cell_type": "markdown", + "id": "4f71fa85", "metadata": { "kernel": "SoS" }, @@ -349,6 +369,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5f4bae76", "metadata": { "kernel": "Bash" }, @@ -363,6 +384,7 @@ }, { "cell_type": "markdown", + "id": "297d5cde", "metadata": { "kernel": "Bash" }, @@ -374,6 +396,7 @@ }, { "cell_type": "markdown", + "id": "a94e3ae4", "metadata": { "kernel": "SoS" }, @@ -384,6 +407,7 @@ { "cell_type": "code", "execution_count": null, + "id": "60f091d7", "metadata": { "kernel": "Bash" }, @@ -401,6 +425,7 @@ }, { "cell_type": "markdown", + "id": "b4a6d4a3", "metadata": { "kernel": "Bash", "tags": [] @@ -412,6 +437,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2d432d51", "metadata": { "kernel": "Bash" }, @@ -422,6 +448,7 @@ }, { "cell_type": "markdown", + "id": "31108992", "metadata": { "kernel": "SoS" }, @@ -545,6 +572,7 @@ }, { "cell_type": "markdown", + "id": "e7b3778c", "metadata": { "kernel": "SoS" }, @@ -557,6 +585,7 @@ { "cell_type": "code", "execution_count": null, + "id": "8c70535a", "metadata": { "kernel": "SoS" }, @@ -683,6 +712,7 @@ }, { "cell_type": "markdown", + "id": "cd7a15ad", "metadata": { "kernel": "SoS" }, @@ -695,6 +725,7 @@ { "cell_type": "code", "execution_count": null, + "id": "dc2c843b", "metadata": { "kernel": "SoS" }, @@ -725,6 +756,7 @@ { "cell_type": "code", "execution_count": null, + "id": "3e999bfd", "metadata": { "kernel": "SoS" }, @@ -749,6 +781,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e4be6730", "metadata": { "kernel": "SoS" }, @@ -777,6 +810,7 @@ }, { "cell_type": "markdown", + "id": "8b5b3005", "metadata": { "kernel": "SoS" }, @@ -789,6 +823,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2f10cefe", "metadata": { "kernel": "SoS" }, @@ -861,6 +896,7 @@ { "cell_type": "code", "execution_count": null, + "id": "52f710f9", "metadata": { "kernel": "SoS" }, @@ -906,6 +942,7 @@ }, { "cell_type": "markdown", + "id": "d092bf41", "metadata": { "kernel": "SoS" }, @@ -918,6 +955,7 @@ { "cell_type": "code", "execution_count": null, + "id": "1c241d8e", "metadata": { "kernel": "SoS" }, @@ -1039,5 +1077,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/data_preprocessing/genotype/PCA.ipynb b/code/SoS/data_preprocessing/genotype/PCA.ipynb index 01cb8699b..ca703c939 100644 --- a/code/SoS/data_preprocessing/genotype/PCA.ipynb +++ b/code/SoS/data_preprocessing/genotype/PCA.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "3f6939cf", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "e550e8c4", "metadata": {}, "source": [ "## Overview\n", @@ -52,6 +54,7 @@ }, { "cell_type": "markdown", + "id": "97e6de5f", "metadata": { "kernel": "SoS" }, @@ -81,6 +84,7 @@ }, { "cell_type": "markdown", + "id": "4b72954f", "metadata": { "kernel": "SoS" }, @@ -114,6 +118,7 @@ }, { "cell_type": "markdown", + "id": "f0d12065", "metadata": { "kernel": "SoS" }, @@ -125,6 +130,7 @@ }, { "cell_type": "markdown", + "id": "1270842b", "metadata": { "kernel": "SoS" }, @@ -136,6 +142,7 @@ }, { "cell_type": "markdown", + "id": "90130c5e", "metadata": { "kernel": "SoS" }, @@ -146,6 +153,7 @@ { "cell_type": "code", "execution_count": null, + "id": "408f3f17", "metadata": { "kernel": "Bash", "vscode": { @@ -163,6 +171,7 @@ }, { "cell_type": "markdown", + "id": "ae033c47", "metadata": { "kernel": "SoS" }, @@ -174,6 +183,7 @@ }, { "cell_type": "markdown", + "id": "4ca1dffc", "metadata": { "kernel": "SoS" }, @@ -184,6 +194,7 @@ }, { "cell_type": "markdown", + "id": "61011198", "metadata": { "kernel": "SoS" }, @@ -194,6 +205,7 @@ { "cell_type": "code", "execution_count": null, + "id": "8ef13d9d", "metadata": { "kernel": "Bash", "vscode": { @@ -210,6 +222,7 @@ }, { "cell_type": "markdown", + "id": "ff250d63", "metadata": { "kernel": "SoS" }, @@ -219,6 +232,7 @@ }, { "cell_type": "markdown", + "id": "268f2be8", "metadata": { "kernel": "SoS" }, @@ -229,6 +243,7 @@ { "cell_type": "code", "execution_count": null, + "id": "3ae307a3", "metadata": { "kernel": "Bash", "vscode": { @@ -247,6 +262,7 @@ }, { "cell_type": "markdown", + "id": "e27cfba0", "metadata": { "kernel": "SoS" }, @@ -258,6 +274,7 @@ }, { "cell_type": "markdown", + "id": "1256090c", "metadata": { "kernel": "SoS" }, @@ -267,6 +284,7 @@ }, { "cell_type": "markdown", + "id": "9a68a758", "metadata": { "kernel": "SoS" }, @@ -277,6 +295,7 @@ { "cell_type": "code", "execution_count": null, + "id": "d10f9da7", "metadata": { "kernel": "Bash", "vscode": { @@ -293,6 +312,7 @@ }, { "cell_type": "markdown", + "id": "c11e4801", "metadata": { "kernel": "SoS" }, @@ -303,6 +323,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e6f2275e", "metadata": { "kernel": "SoS" }, @@ -313,6 +334,7 @@ }, { "cell_type": "markdown", + "id": "437a46d2", "metadata": { "kernel": "SoS" }, @@ -324,6 +346,7 @@ }, { "cell_type": "markdown", + "id": "ac01d8d7", "metadata": { "kernel": "SoS" }, @@ -334,6 +357,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2b3d74d5", "metadata": { "kernel": "Bash", "vscode": { @@ -356,6 +380,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b00f62ec", "metadata": { "kernel": "SoS" }, @@ -366,6 +391,7 @@ }, { "cell_type": "markdown", + "id": "cade47db", "metadata": { "kernel": "SoS" }, @@ -376,6 +402,7 @@ { "cell_type": "code", "execution_count": null, + "id": "22c9e960", "metadata": { "kernel": "Bash", "vscode": { @@ -393,6 +420,7 @@ }, { "cell_type": "markdown", + "id": "1bc6f4f0", "metadata": { "kernel": "SoS" }, @@ -404,6 +432,7 @@ }, { "cell_type": "markdown", + "id": "6e45d3a4", "metadata": { "kernel": "SoS" }, @@ -413,6 +442,7 @@ }, { "cell_type": "markdown", + "id": "d8384f66", "metadata": { "kernel": "SoS" }, @@ -423,6 +453,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b143664a", "metadata": { "kernel": "Bash", "vscode": { @@ -440,6 +471,7 @@ }, { "cell_type": "markdown", + "id": "f9569f8f", "metadata": { "kernel": "SoS" }, @@ -449,6 +481,7 @@ }, { "cell_type": "markdown", + "id": "5eaef054", "metadata": { "kernel": "SoS" }, @@ -461,6 +494,7 @@ { "cell_type": "code", "execution_count": 2, + "id": "8f576f8f", "metadata": { "kernel": "R", "tags": [] @@ -482,6 +516,7 @@ }, { "cell_type": "markdown", + "id": "9a24f1e3", "metadata": { "kernel": "SoS" }, @@ -494,6 +529,7 @@ { "cell_type": "code", "execution_count": null, + "id": "cda6c37d", "metadata": { "kernel": "Bash", "vscode": { @@ -515,6 +551,7 @@ }, { "cell_type": "markdown", + "id": "7a741e54", "metadata": { "kernel": "SoS" }, @@ -529,6 +566,7 @@ { "cell_type": "code", "execution_count": null, + "id": "8f06c4dc", "metadata": { "kernel": "Bash", "vscode": { @@ -550,6 +588,7 @@ }, { "cell_type": "markdown", + "id": "d7d82322", "metadata": { "kernel": "SoS" }, @@ -562,6 +601,7 @@ { "cell_type": "code", "execution_count": null, + "id": "eb05c559", "metadata": { "kernel": "Bash", "vscode": { @@ -585,6 +625,7 @@ }, { "cell_type": "markdown", + "id": "f5647fcb", "metadata": { "kernel": "SoS" }, @@ -597,6 +638,7 @@ { "cell_type": "code", "execution_count": null, + "id": "888b4e61", "metadata": { "kernel": "Bash", "vscode": { @@ -621,6 +663,7 @@ }, { "cell_type": "markdown", + "id": "152a6a0c", "metadata": { "kernel": "SoS" }, @@ -631,6 +674,7 @@ { "cell_type": "code", "execution_count": null, + "id": "c47fe27c", "metadata": { "kernel": "SoS", "scrolled": true @@ -642,6 +686,7 @@ }, { "cell_type": "markdown", + "id": "bafa9751", "metadata": { "kernel": "SoS" }, @@ -651,6 +696,7 @@ }, { "cell_type": "markdown", + "id": "0a6c3fbe", "metadata": { "kernel": "SoS" }, @@ -661,6 +707,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f783b132", "metadata": {}, "outputs": [], "source": [ @@ -669,6 +716,7 @@ }, { "cell_type": "markdown", + "id": "54455ebf", "metadata": {}, "source": [ "```\n", @@ -779,6 +827,7 @@ }, { "cell_type": "markdown", + "id": "d576d2fd", "metadata": {}, "source": [ "## Workflow implementation" @@ -787,6 +836,7 @@ { "cell_type": "code", "execution_count": null, + "id": "10d5be9f", "metadata": { "kernel": "SoS" }, @@ -912,6 +962,7 @@ { "cell_type": "code", "execution_count": null, + "id": "813b142d", "metadata": { "kernel": "SoS" }, @@ -931,6 +982,7 @@ }, { "cell_type": "markdown", + "id": "22e062a2", "metadata": { "kernel": "SoS" }, @@ -941,6 +993,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6e522e74", "metadata": { "kernel": "SoS" }, @@ -981,6 +1034,7 @@ }, { "cell_type": "markdown", + "id": "55eb9b17", "metadata": { "kernel": "SoS" }, @@ -991,6 +1045,7 @@ { "cell_type": "code", "execution_count": null, + "id": "882e0ef9", "metadata": { "kernel": "SoS" }, @@ -1028,6 +1083,7 @@ }, { "cell_type": "markdown", + "id": "02322eac", "metadata": { "kernel": "SoS" }, @@ -1038,6 +1094,7 @@ { "cell_type": "code", "execution_count": null, + "id": "75d819ae", "metadata": { "kernel": "SoS" }, @@ -1075,6 +1132,7 @@ }, { "cell_type": "markdown", + "id": "0bc3554b", "metadata": { "kernel": "SoS" }, @@ -1085,6 +1143,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e2b517f5", "metadata": { "kernel": "SoS" }, @@ -1126,6 +1185,7 @@ }, { "cell_type": "markdown", + "id": "319a0b46", "metadata": { "kernel": "SoS" }, @@ -1136,6 +1196,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f3db0ebd", "metadata": { "kernel": "SoS" }, @@ -1270,5 +1331,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/data_preprocessing/genotype/VCF_QC.ipynb b/code/SoS/data_preprocessing/genotype/VCF_QC.ipynb index d9225ed6c..b0fefa73f 100644 --- a/code/SoS/data_preprocessing/genotype/VCF_QC.ipynb +++ b/code/SoS/data_preprocessing/genotype/VCF_QC.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "8b65a4f6", "metadata": { "kernel": "SoS", "tags": [] @@ -14,6 +15,7 @@ }, { "cell_type": "markdown", + "id": "632a2a71", "metadata": { "kernel": "SoS" }, @@ -36,6 +38,7 @@ }, { "cell_type": "markdown", + "id": "4fa791ec", "metadata": { "kernel": "SoS", "tags": [] @@ -83,6 +86,7 @@ }, { "cell_type": "markdown", + "id": "26cc92eb", "metadata": { "kernel": "SoS" }, @@ -104,6 +108,7 @@ }, { "cell_type": "markdown", + "id": "83582784", "metadata": { "kernel": "SoS" }, @@ -139,6 +144,7 @@ }, { "cell_type": "markdown", + "id": "0df1d09f", "metadata": { "kernel": "SoS" }, @@ -150,6 +156,7 @@ }, { "cell_type": "markdown", + "id": "48137099", "metadata": { "kernel": "SoS" }, @@ -161,6 +168,7 @@ }, { "cell_type": "markdown", + "id": "ed726478", "metadata": { "kernel": "SoS" }, @@ -170,6 +178,7 @@ }, { "cell_type": "markdown", + "id": "7bff0a35", "metadata": { "kernel": "SoS", "tags": [] @@ -181,6 +190,7 @@ { "cell_type": "code", "execution_count": null, + "id": "3097a62e", "metadata": { "kernel": "Bash", "vscode": { @@ -196,6 +206,7 @@ }, { "cell_type": "markdown", + "id": "fb8cd712", "metadata": { "kernel": "SoS" }, @@ -205,6 +216,7 @@ }, { "cell_type": "markdown", + "id": "1f61c200", "metadata": { "kernel": "SoS" }, @@ -214,6 +226,7 @@ }, { "cell_type": "markdown", + "id": "7a0bfffb", "metadata": { "kernel": "SoS", "tags": [] @@ -225,6 +238,7 @@ { "cell_type": "code", "execution_count": null, + "id": "22e17754", "metadata": { "kernel": "Bash", "vscode": { @@ -240,6 +254,7 @@ }, { "cell_type": "markdown", + "id": "6c9eee35", "metadata": { "kernel": "Bash" }, @@ -253,6 +268,7 @@ }, { "cell_type": "markdown", + "id": "631c6366", "metadata": { "kernel": "SoS" }, @@ -264,6 +280,7 @@ }, { "cell_type": "markdown", + "id": "145424fc", "metadata": { "kernel": "SoS" }, @@ -273,6 +290,7 @@ }, { "cell_type": "markdown", + "id": "f53ba9f2", "metadata": { "kernel": "SoS", "tags": [] @@ -284,6 +302,7 @@ { "cell_type": "code", "execution_count": null, + "id": "8724d552", "metadata": { "kernel": "Bash", "vscode": { @@ -302,6 +321,7 @@ }, { "cell_type": "markdown", + "id": "dafa15fc", "metadata": { "kernel": "SoS" }, @@ -313,6 +333,7 @@ }, { "cell_type": "markdown", + "id": "e5f54f2f", "metadata": { "kernel": "SoS", "tags": [] @@ -324,6 +345,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4b3538c1", "metadata": { "kernel": "Bash", "vscode": { @@ -343,6 +365,7 @@ }, { "cell_type": "markdown", + "id": "41c409b4", "metadata": { "kernel": "SoS" }, @@ -354,6 +377,7 @@ }, { "cell_type": "markdown", + "id": "b56d7cc1", "metadata": { "kernel": "SoS", "tags": [] @@ -365,6 +389,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a73a1191", "metadata": { "kernel": "Bash" }, @@ -380,6 +405,7 @@ }, { "cell_type": "markdown", + "id": "31bb9389", "metadata": { "kernel": "SoS" }, @@ -395,6 +421,7 @@ { "cell_type": "code", "execution_count": null, + "id": "39420ac7", "metadata": { "kernel": "SoS" }, @@ -405,6 +432,7 @@ }, { "cell_type": "markdown", + "id": "5934f8c4", "metadata": { "kernel": "SoS" }, @@ -415,6 +443,7 @@ { "cell_type": "code", "execution_count": null, + "id": "8123a64f", "metadata": { "kernel": "SoS" }, @@ -425,6 +454,7 @@ }, { "cell_type": "markdown", + "id": "968a7a6c", "metadata": { "kernel": "SoS" }, @@ -435,6 +465,7 @@ { "cell_type": "code", "execution_count": null, + "id": "634ddacb", "metadata": { "kernel": "SoS" }, @@ -445,6 +476,7 @@ }, { "cell_type": "markdown", + "id": "6988109d", "metadata": { "kernel": "SoS" }, @@ -455,6 +487,7 @@ { "cell_type": "code", "execution_count": null, + "id": "ab76c85a", "metadata": { "kernel": "SoS" }, @@ -465,6 +498,7 @@ }, { "cell_type": "markdown", + "id": "94f2f5b3", "metadata": { "kernel": "SoS" }, @@ -556,6 +590,7 @@ }, { "cell_type": "markdown", + "id": "5f310359", "metadata": { "kernel": "Bash" }, @@ -566,6 +601,7 @@ { "cell_type": "code", "execution_count": null, + "id": "04d94d59", "metadata": { "kernel": "SoS" }, @@ -653,6 +689,7 @@ }, { "cell_type": "markdown", + "id": "55fd94cf", "metadata": { "kernel": "SoS", "tags": [] @@ -668,6 +705,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9f7f400c", "metadata": { "kernel": "SoS" }, @@ -687,6 +725,7 @@ { "cell_type": "code", "execution_count": null, + "id": "da9dabb9", "metadata": { "kernel": "SoS" }, @@ -704,6 +743,7 @@ }, { "cell_type": "markdown", + "id": "cc54b421", "metadata": { "kernel": "SoS" }, @@ -716,6 +756,7 @@ { "cell_type": "code", "execution_count": null, + "id": "468e0a95", "metadata": { "kernel": "SoS" }, @@ -747,6 +788,7 @@ }, { "cell_type": "markdown", + "id": "a348588c", "metadata": { "kernel": "SoS" }, @@ -757,6 +799,7 @@ { "cell_type": "code", "execution_count": null, + "id": "24631654", "metadata": { "kernel": "SoS" }, @@ -801,6 +844,7 @@ { "cell_type": "code", "execution_count": null, + "id": "79a563b9", "metadata": { "kernel": "SoS" }, @@ -867,5 +911,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/data_preprocessing/genotype/genotype_formatting.ipynb b/code/SoS/data_preprocessing/genotype/genotype_formatting.ipynb index 77d5541fc..05b819244 100644 --- a/code/SoS/data_preprocessing/genotype/genotype_formatting.ipynb +++ b/code/SoS/data_preprocessing/genotype/genotype_formatting.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "8413f361", "metadata": { "kernel": "SoS", "tags": [] @@ -14,6 +15,7 @@ }, { "cell_type": "markdown", + "id": "5d8af5b9", "metadata": { "kernel": "SoS", "tags": [] @@ -32,6 +34,7 @@ }, { "cell_type": "markdown", + "id": "397713af", "metadata": { "kernel": "SoS" }, @@ -50,6 +53,7 @@ }, { "cell_type": "markdown", + "id": "9a3542e9", "metadata": { "kernel": "SoS", "tags": [] @@ -100,6 +104,7 @@ }, { "cell_type": "markdown", + "id": "6cb3cf12", "metadata": { "kernel": "SoS" }, @@ -154,6 +159,7 @@ }, { "cell_type": "markdown", + "id": "37eda720", "metadata": { "kernel": "SoS" }, @@ -163,6 +169,7 @@ }, { "cell_type": "markdown", + "id": "4e6e2fd8", "metadata": { "kernel": "SoS" }, @@ -172,6 +179,7 @@ }, { "cell_type": "markdown", + "id": "adb9093e", "metadata": { "kernel": "SoS" }, @@ -181,6 +189,7 @@ }, { "cell_type": "markdown", + "id": "ebac0c48", "metadata": { "kernel": "SoS" }, @@ -190,6 +199,7 @@ }, { "cell_type": "markdown", + "id": "65598688", "metadata": { "kernel": "SoS", "tags": [] @@ -201,6 +211,7 @@ { "cell_type": "code", "execution_count": null, + "id": "7f715157", "metadata": { "kernel": "Bash", "vscode": { @@ -218,6 +229,7 @@ }, { "cell_type": "markdown", + "id": "1d9fb08b", "metadata": { "kernel": "SoS" }, @@ -227,6 +239,7 @@ }, { "cell_type": "markdown", + "id": "09507b2b", "metadata": { "kernel": "SoS" }, @@ -236,6 +249,7 @@ }, { "cell_type": "markdown", + "id": "d3c3d999", "metadata": { "kernel": "SoS", "tags": [] @@ -247,6 +261,7 @@ { "cell_type": "code", "execution_count": null, + "id": "069182b7", "metadata": { "kernel": "Bash", "vscode": { @@ -264,6 +279,7 @@ }, { "cell_type": "markdown", + "id": "23a45875", "metadata": { "kernel": "SoS" }, @@ -273,6 +289,7 @@ }, { "cell_type": "markdown", + "id": "f9c8f4f8", "metadata": { "kernel": "SoS" }, @@ -282,6 +299,7 @@ }, { "cell_type": "markdown", + "id": "885c5153", "metadata": { "kernel": "SoS", "tags": [] @@ -293,6 +311,7 @@ { "cell_type": "code", "execution_count": null, + "id": "c84cc933", "metadata": { "kernel": "Bash", "vscode": { @@ -310,6 +329,7 @@ }, { "cell_type": "markdown", + "id": "732334d3", "metadata": { "kernel": "SoS" }, @@ -319,6 +339,7 @@ }, { "cell_type": "markdown", + "id": "324e0801", "metadata": { "kernel": "SoS" }, @@ -330,6 +351,7 @@ }, { "cell_type": "markdown", + "id": "2c6c4337", "metadata": { "kernel": "SoS" }, @@ -340,6 +362,7 @@ { "cell_type": "code", "execution_count": null, + "id": "edf8b117", "metadata": { "kernel": "Bash", "vscode": { @@ -356,6 +379,7 @@ }, { "cell_type": "markdown", + "id": "d523ff8d", "metadata": { "kernel": "SoS" }, @@ -367,6 +391,7 @@ }, { "cell_type": "markdown", + "id": "c60295a5", "metadata": { "kernel": "SoS" }, @@ -377,6 +402,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6926979d", "metadata": { "kernel": "Bash", "vscode": { @@ -393,6 +419,7 @@ }, { "cell_type": "markdown", + "id": "b1f1791a", "metadata": { "kernel": "SoS" }, @@ -404,6 +431,7 @@ }, { "cell_type": "markdown", + "id": "7e01bf24", "metadata": { "kernel": "SoS" }, @@ -416,6 +444,7 @@ }, { "cell_type": "markdown", + "id": "5acd794c", "metadata": { "kernel": "SoS" }, @@ -426,6 +455,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2f0fce00", "metadata": { "kernel": "Bash", "vscode": { @@ -444,6 +474,7 @@ }, { "cell_type": "markdown", + "id": "469ecd92", "metadata": { "kernel": "SoS" }, @@ -455,6 +486,7 @@ }, { "cell_type": "markdown", + "id": "b4fdb520", "metadata": { "kernel": "SoS" }, @@ -465,6 +497,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2042afec", "metadata": { "kernel": "Bash", "vscode": { @@ -483,6 +516,7 @@ }, { "cell_type": "markdown", + "id": "b40a219c", "metadata": { "kernel": "SoS" }, @@ -492,6 +526,7 @@ }, { "cell_type": "markdown", + "id": "f0ff8dcc", "metadata": { "kernel": "SoS" }, @@ -503,6 +538,7 @@ }, { "cell_type": "markdown", + "id": "9cf8a8d2", "metadata": { "kernel": "SoS" }, @@ -513,6 +549,7 @@ { "cell_type": "code", "execution_count": null, + "id": "93f00799", "metadata": { "kernel": "Bash", "vscode": { @@ -531,6 +568,7 @@ }, { "cell_type": "markdown", + "id": "0af85321", "metadata": { "kernel": "SoS" }, @@ -541,6 +579,7 @@ { "cell_type": "code", "execution_count": null, + "id": "be6cbf71", "metadata": { "kernel": "SoS" }, @@ -551,6 +590,7 @@ }, { "cell_type": "markdown", + "id": "4d5a6cc7", "metadata": { "kernel": "SoS" }, @@ -643,6 +683,7 @@ }, { "cell_type": "markdown", + "id": "26ccf4aa", "metadata": { "kernel": "SoS" }, @@ -653,6 +694,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f98335e0", "metadata": { "kernel": "SoS" }, @@ -800,6 +842,7 @@ }, { "cell_type": "markdown", + "id": "4d340884", "metadata": { "kernel": "SoS", "tags": [] @@ -811,6 +854,7 @@ { "cell_type": "code", "execution_count": null, + "id": "25804f2c", "metadata": { "kernel": "SoS" }, @@ -838,6 +882,7 @@ }, { "cell_type": "markdown", + "id": "c31a29d0", "metadata": { "kernel": "SoS" }, @@ -852,6 +897,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f7560d77", "metadata": { "kernel": "SoS" }, @@ -884,6 +930,7 @@ }, { "cell_type": "markdown", + "id": "f9e5c24b", "metadata": { "kernel": "SoS", "tags": [] @@ -895,6 +942,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e4772c44", "metadata": { "kernel": "SoS" }, @@ -927,6 +975,7 @@ }, { "cell_type": "markdown", + "id": "357018ee", "metadata": { "kernel": "SoS" }, @@ -939,6 +988,7 @@ { "cell_type": "code", "execution_count": null, + "id": "c42334e0", "metadata": { "kernel": "SoS" }, @@ -963,6 +1013,7 @@ }, { "cell_type": "markdown", + "id": "d9c27020", "metadata": { "kernel": "SoS" }, @@ -982,6 +1033,7 @@ }, { "cell_type": "markdown", + "id": "96510b44", "metadata": { "kernel": "SoS" }, @@ -994,6 +1046,7 @@ }, { "cell_type": "markdown", + "id": "063e1dd7", "metadata": { "kernel": "SoS", "vscode": { @@ -1007,6 +1060,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9c768c5f", "metadata": { "kernel": "SoS" }, @@ -1041,6 +1095,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6a713cf5", "metadata": { "kernel": "SoS" }, @@ -1057,6 +1112,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e808bc29", "metadata": { "kernel": "SoS" }, @@ -1071,6 +1127,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b7c1ed67", "metadata": { "kernel": "SoS" }, @@ -1085,6 +1142,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6865c9f2", "metadata": { "kernel": "SoS" }, @@ -1100,6 +1158,7 @@ { "cell_type": "code", "execution_count": null, + "id": "36f06329", "metadata": { "kernel": "SoS" }, @@ -1123,6 +1182,7 @@ }, { "cell_type": "markdown", + "id": "e0fe937b", "metadata": { "kernel": "SoS" }, @@ -1133,6 +1193,7 @@ { "cell_type": "code", "execution_count": null, + "id": "cb1389e7", "metadata": { "kernel": "SoS" }, @@ -1160,6 +1221,7 @@ }, { "cell_type": "markdown", + "id": "643be9f6", "metadata": { "kernel": "SoS" }, @@ -1170,6 +1232,7 @@ { "cell_type": "code", "execution_count": null, + "id": "bc899b6f", "metadata": { "kernel": "SoS" }, @@ -1229,5 +1292,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/data_preprocessing/genotype_preprocessing.ipynb b/code/SoS/data_preprocessing/genotype_preprocessing.ipynb index 343d37b90..caada08f4 100644 --- a/code/SoS/data_preprocessing/genotype_preprocessing.ipynb +++ b/code/SoS/data_preprocessing/genotype_preprocessing.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "9319685b", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "69165e40", "metadata": { "kernel": "SoS" }, @@ -26,19 +28,21 @@ }, { "cell_type": "markdown", + "id": "f861c886", "metadata": { "kernel": "SoS" }, "source": [ "## Overview\n", "\n", - "This mini-protocol walks through how raw genotype VCFs are normalized, converted to PLINK, quality-controlled, matched to molecular-phenotype samples, and prepared for population-structure adjustment. Each step calls a workflow from [`VCF_QC.ipynb`](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/VCF_QC.html), [`genotype_formatting.ipynb`](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/genotype_formatting.html), [`GWAS_QC.ipynb`](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/GWAS_QC.html), or [`PCA.ipynb`](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/PCA.html).\n", + "This mini-protocol walks through how raw genotype VCFs are normalized, converted to PLINK, quality-controlled, matched to molecular-phenotype samples, and prepared for population-structure adjustment. Each step calls a workflow from [`VCF_QC.ipynb`](https://statfungen.github.io/xqtl-protocol/vcf-qc), [`genotype_formatting.ipynb`](https://statfungen.github.io/xqtl-protocol/genotype-formatting), [`GWAS_QC.ipynb`](https://statfungen.github.io/xqtl-protocol/gwas-qc), or [`PCA.ipynb`](https://statfungen.github.io/xqtl-protocol/pca).\n", "\n", "The commands are organized as selectable routes rather than one mandatory 12-step chain. The standard route estimates principal components in unrelated individuals. Use the conditional extension only when related individuals will remain in the analysis and must be projected into the same PCA space before the datasets are recombined.\n" ] }, { "cell_type": "markdown", + "id": "c95b935b", "metadata": { "kernel": "SoS" }, @@ -58,11 +62,12 @@ }, { "cell_type": "markdown", + "id": "30cae79c", "metadata": { "kernel": "SoS" }, "source": [ - "### 1. [Quality-control the input VCF](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/VCF_QC.html)\n", + "### 1. [Quality-control the input VCF](https://statfungen.github.io/xqtl-protocol/vcf-qc)\n", "\n", "**What it does:** Normalize and filter the toy VCF against dbSNP and GRCh38.\n" ] @@ -70,6 +75,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4c8c83c6", "metadata": { "kernel": "Bash" }, @@ -85,11 +91,12 @@ }, { "cell_type": "markdown", + "id": "ede35027", "metadata": { "kernel": "SoS" }, "source": [ - "### 2. [Convert the QC-passed VCF to PLINK and merge chromosomes](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/genotype_formatting.html)\n", + "### 2. [Convert the QC-passed VCF to PLINK and merge chromosomes](https://statfungen.github.io/xqtl-protocol/genotype-formatting)\n", "\n", "**What it does:** Convert the output of step 1 to PLINK. The merge command generalizes to multiple chromosome-level files.\n" ] @@ -97,6 +104,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f7063546", "metadata": { "kernel": "Bash" }, @@ -117,11 +125,12 @@ }, { "cell_type": "markdown", + "id": "070aeb5a", "metadata": { "kernel": "SoS" }, "source": [ - "### 3. [Apply PLINK-level quality control](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/GWAS_QC.html)\n", + "### 3. [Apply PLINK-level quality control](https://statfungen.github.io/xqtl-protocol/gwas-qc)\n", "\n", "**What it does:** Apply genotype-, sample- and Hardy-Weinberg-equilibrium filters to the merged PLINK dataset.\n" ] @@ -129,6 +138,7 @@ { "cell_type": "code", "execution_count": null, + "id": "dd73532a", "metadata": { "kernel": "Bash" }, @@ -145,11 +155,12 @@ }, { "cell_type": "markdown", + "id": "edb56c29", "metadata": { "kernel": "SoS" }, "source": [ - "### 4. [Partition the QC-passed genotype data by chromosome](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/genotype_formatting.html)\n", + "### 4. [Partition the QC-passed genotype data by chromosome](https://statfungen.github.io/xqtl-protocol/genotype-formatting)\n", "\n", "**What it does:** Create chromosome-specific PLINK files required by chromosome-oriented downstream workflows.\n" ] @@ -157,6 +168,7 @@ { "cell_type": "code", "execution_count": null, + "id": "52c88fa6", "metadata": { "kernel": "Bash" }, @@ -171,11 +183,12 @@ }, { "cell_type": "markdown", + "id": "88ea29ec", "metadata": { "kernel": "SoS" }, "source": [ - "### 5. [Match genotype and molecular-phenotype samples](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/GWAS_QC.html)\n", + "### 5. [Match genotype and molecular-phenotype samples](https://statfungen.github.io/xqtl-protocol/gwas-qc)\n", "\n", "**What it does:** Retain the sample intersection between the QC-passed genotype data and molecular phenotype.\n" ] @@ -183,6 +196,7 @@ { "cell_type": "code", "execution_count": null, + "id": "214841e2", "metadata": { "kernel": "Bash" }, @@ -196,11 +210,12 @@ }, { "cell_type": "markdown", + "id": "a1e71e68", "metadata": { "kernel": "SoS" }, "source": [ - "### 6. [Estimate kinship and separate related individuals](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/GWAS_QC.html)\n", + "### 6. [Estimate kinship and separate related individuals](https://statfungen.github.io/xqtl-protocol/gwas-qc)\n", "\n", "**What it does:** Use KING to identify related pairs and produce related and unrelated subsets.\n" ] @@ -208,6 +223,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a1a78e55", "metadata": { "kernel": "Bash" }, @@ -222,11 +238,12 @@ }, { "cell_type": "markdown", + "id": "dc700021", "metadata": { "kernel": "SoS" }, "source": [ - "### 7. [Prepare the unrelated, LD-pruned PCA subset](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/GWAS_QC.html)\n", + "### 7. [Prepare the unrelated, LD-pruned PCA subset](https://statfungen.github.io/xqtl-protocol/gwas-qc)\n", "\n", "**What it does:** Apply the minor-allele-count filter and LD pruning used to estimate ancestry axes.\n" ] @@ -234,6 +251,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a632cb34", "metadata": { "kernel": "Bash" }, @@ -247,11 +265,12 @@ }, { "cell_type": "markdown", + "id": "8f2a5b45", "metadata": { "kernel": "SoS" }, "source": [ - "### 8. [Estimate principal components in unrelated individuals](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/PCA.html)\n", + "### 8. [Estimate principal components in unrelated individuals](https://statfungen.github.io/xqtl-protocol/pca)\n", "\n", "**What it does:** Estimate the PCA model and scores in unrelated individuals.\n" ] @@ -259,6 +278,7 @@ { "cell_type": "code", "execution_count": null, + "id": "1d7cb166", "metadata": { "kernel": "Bash" }, @@ -272,11 +292,12 @@ }, { "cell_type": "markdown", + "id": "b78af0aa", "metadata": { "kernel": "SoS" }, "source": [ - "### 9. [Extract the related samples at the PCA variants](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/GWAS_QC.html)\n", + "### 9. [Extract the related samples at the PCA variants](https://statfungen.github.io/xqtl-protocol/gwas-qc)\n", "\n", "**What it does:** Restrict the related subset to the variants used by the unrelated-sample PCA model.\n" ] @@ -284,6 +305,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a52360b8", "metadata": { "kernel": "Bash", "tags": [] @@ -300,11 +322,12 @@ }, { "cell_type": "markdown", + "id": "3af3e8e1", "metadata": { "kernel": "SoS" }, "source": [ - "### 10. [Project related samples and detect PCA outliers](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/PCA.html)\n", + "### 10. [Project related samples and detect PCA outliers](https://statfungen.github.io/xqtl-protocol/pca)\n", "\n", "**What it does:** Project related individuals into the PCA space and identify ancestry-space outliers.\n" ] @@ -312,6 +335,7 @@ { "cell_type": "code", "execution_count": null, + "id": "3bb2d3c0", "metadata": { "kernel": "Bash", "tags": [] @@ -328,11 +352,12 @@ }, { "cell_type": "markdown", + "id": "e485d3c7", "metadata": { "kernel": "SoS" }, "source": [ - "### 11. [Remove projected PCA outliers](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/GWAS_QC.html)\n", + "### 11. [Remove projected PCA outliers](https://statfungen.github.io/xqtl-protocol/gwas-qc)\n", "\n", "**What it does:** Remove projected outliers before recombining samples.\n" ] @@ -340,6 +365,7 @@ { "cell_type": "code", "execution_count": null, + "id": "c18e4f51", "metadata": { "kernel": "Bash", "tags": [] @@ -355,11 +381,12 @@ }, { "cell_type": "markdown", + "id": "f8b45b3c", "metadata": { "kernel": "SoS" }, "source": [ - "### 12. [Recombine unrelated and projected related samples](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/genotype_formatting.html)\n", + "### 12. [Recombine unrelated and projected related samples](https://statfungen.github.io/xqtl-protocol/genotype-formatting)\n", "\n", "**What it does:** Merge the unrelated PCA subset with the retained projected related samples for downstream analysis.\n" ] @@ -367,6 +394,7 @@ { "cell_type": "code", "execution_count": null, + "id": "d3f20592", "metadata": { "kernel": "Bash", "tags": [] @@ -382,6 +410,7 @@ }, { "cell_type": "markdown", + "id": "a9cd4016", "metadata": { "kernel": "SoS" }, @@ -403,6 +432,7 @@ }, { "cell_type": "markdown", + "id": "412d3457", "metadata": { "kernel": "SoS" }, @@ -418,6 +448,7 @@ }, { "cell_type": "markdown", + "id": "d48efec1", "metadata": { "kernel": "SoS" }, @@ -425,12 +456,12 @@ "## Command Interface\n", "\n", "List the workflows and parameters available in each module used by this mini-protocol.\n" - ], - "id": "d48efec1" + ] }, { "cell_type": "code", "execution_count": null, + "id": "56a944bc", "metadata": { "kernel": "Bash" }, @@ -440,8 +471,7 @@ "sos run pipeline/genotype_formatting.ipynb -h\n", "sos run pipeline/GWAS_QC.ipynb -h\n", "sos run pipeline/PCA.ipynb -h\n" - ], - "id": "56a944bc" + ] } ], "metadata": { @@ -480,4 +510,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/data_preprocessing/phenotype/gene_annotation.ipynb b/code/SoS/data_preprocessing/phenotype/gene_annotation.ipynb index a32e73ec8..95e13d12d 100644 --- a/code/SoS/data_preprocessing/phenotype/gene_annotation.ipynb +++ b/code/SoS/data_preprocessing/phenotype/gene_annotation.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "f443aa0b", "metadata": { "kernel": "SoS", "tags": [] @@ -14,6 +15,7 @@ }, { "cell_type": "markdown", + "id": "1c9d9bcc", "metadata": { "kernel": "SoS" }, @@ -33,6 +35,7 @@ }, { "cell_type": "markdown", + "id": "c8791923", "metadata": { "kernel": "SoS", "tags": [] @@ -68,6 +71,7 @@ }, { "cell_type": "markdown", + "id": "b9829a01", "metadata": { "kernel": "SoS", "tags": [] @@ -100,6 +104,7 @@ }, { "cell_type": "markdown", + "id": "73f76baa", "metadata": { "kernel": "SoS" }, @@ -111,6 +116,7 @@ }, { "cell_type": "markdown", + "id": "1260f2f7", "metadata": { "kernel": "SoS" }, @@ -120,6 +126,7 @@ }, { "cell_type": "markdown", + "id": "5748ddf3", "metadata": { "kernel": "SoS" }, @@ -129,6 +136,7 @@ }, { "cell_type": "markdown", + "id": "4a8022ba", "metadata": { "kernel": "SoS", "tags": [] @@ -140,6 +148,7 @@ { "cell_type": "code", "execution_count": null, + "id": "7082fa88", "metadata": { "kernel": "Bash" }, @@ -154,6 +163,7 @@ }, { "cell_type": "markdown", + "id": "301163ca", "metadata": { "kernel": "SoS" }, @@ -163,6 +173,7 @@ }, { "cell_type": "markdown", + "id": "42b28b32", "metadata": { "kernel": "SoS" }, @@ -172,6 +183,7 @@ }, { "cell_type": "markdown", + "id": "87a0b690", "metadata": { "kernel": "SoS" }, @@ -181,6 +193,7 @@ }, { "cell_type": "markdown", + "id": "99529459", "metadata": { "kernel": "SoS" }, @@ -191,6 +204,7 @@ { "cell_type": "code", "execution_count": null, + "id": "ec6ec830", "metadata": { "kernel": "Bash" }, @@ -206,6 +220,7 @@ }, { "cell_type": "markdown", + "id": "fcdd4925", "metadata": { "kernel": "SoS" }, @@ -215,6 +230,7 @@ }, { "cell_type": "markdown", + "id": "30217c5b", "metadata": { "kernel": "SoS" }, @@ -224,6 +240,7 @@ }, { "cell_type": "markdown", + "id": "9beb2f4d", "metadata": { "kernel": "SoS" }, @@ -234,6 +251,7 @@ { "cell_type": "code", "execution_count": null, + "id": "741a8923", "metadata": { "kernel": "Bash" }, @@ -249,6 +267,7 @@ }, { "cell_type": "markdown", + "id": "6521594f", "metadata": { "kernel": "SoS" }, @@ -258,6 +277,7 @@ }, { "cell_type": "markdown", + "id": "9a39aab9", "metadata": { "kernel": "SoS" }, @@ -290,6 +310,7 @@ }, { "cell_type": "markdown", + "id": "4383d22b", "metadata": { "kernel": "SoS" }, @@ -300,6 +321,7 @@ { "cell_type": "code", "execution_count": null, + "id": "de473bfb", "metadata": { "kernel": "Bash" }, @@ -315,6 +337,7 @@ }, { "cell_type": "markdown", + "id": "c572cd84", "metadata": { "kernel": "SoS" }, @@ -324,6 +347,7 @@ }, { "cell_type": "markdown", + "id": "27d38b15", "metadata": { "kernel": "SoS" }, @@ -333,6 +357,7 @@ }, { "cell_type": "markdown", + "id": "9246ca7c", "metadata": { "kernel": "SoS" }, @@ -343,6 +368,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5156e591", "metadata": { "kernel": "Bash" }, @@ -356,6 +382,7 @@ }, { "cell_type": "markdown", + "id": "521b2337", "metadata": { "kernel": "Markdown", "tags": [] @@ -376,6 +403,7 @@ }, { "cell_type": "markdown", + "id": "9054d269", "metadata": { "kernel": "SoS" }, @@ -398,6 +426,7 @@ }, { "cell_type": "markdown", + "id": "1d044a9a", "metadata": { "kernel": "SoS" }, @@ -408,6 +437,7 @@ { "cell_type": "code", "execution_count": null, + "id": "13ec4c67", "metadata": { "kernel": "Bash" }, @@ -418,6 +448,7 @@ }, { "cell_type": "markdown", + "id": "71d727dc", "metadata": { "kernel": "SoS" }, @@ -506,6 +537,7 @@ }, { "cell_type": "markdown", + "id": "591f819c", "metadata": { "kernel": "Bash" }, @@ -516,6 +548,7 @@ { "cell_type": "code", "execution_count": null, + "id": "c2024628", "metadata": { "kernel": "SoS" }, @@ -541,6 +574,7 @@ }, { "cell_type": "markdown", + "id": "8d4db6ea", "metadata": { "kernel": "SoS" }, @@ -555,6 +589,7 @@ { "cell_type": "code", "execution_count": null, + "id": "58d683ce", "metadata": { "kernel": "SoS" }, @@ -596,6 +631,7 @@ }, { "cell_type": "markdown", + "id": "ee793441", "metadata": { "kernel": "SoS", "tags": [] @@ -608,6 +644,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2158a0bd", "metadata": { "kernel": "SoS" }, @@ -630,6 +667,7 @@ }, { "cell_type": "markdown", + "id": "f8ddf9e7", "metadata": { "kernel": "SoS" }, @@ -639,6 +677,7 @@ }, { "cell_type": "markdown", + "id": "30e8e672", "metadata": { "kernel": "SoS" }, @@ -649,6 +688,7 @@ { "cell_type": "code", "execution_count": null, + "id": "3386f5fb", "metadata": { "kernel": "SoS" }, @@ -682,6 +722,7 @@ { "cell_type": "code", "execution_count": null, + "id": "0eb6b936", "metadata": { "kernel": "SoS" }, @@ -741,5 +782,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 + "nbformat_minor": 5 } diff --git a/code/SoS/data_preprocessing/phenotype/phenotype_formatting.ipynb b/code/SoS/data_preprocessing/phenotype/phenotype_formatting.ipynb index 564e9726a..0f773691d 100644 --- a/code/SoS/data_preprocessing/phenotype/phenotype_formatting.ipynb +++ b/code/SoS/data_preprocessing/phenotype/phenotype_formatting.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "bf572ae5", "metadata": { "kernel": "SoS", "tags": [] @@ -14,6 +15,7 @@ }, { "cell_type": "markdown", + "id": "32e3b6a0", "metadata": { "kernel": "SoS" }, @@ -29,6 +31,7 @@ }, { "cell_type": "markdown", + "id": "c0bbcc68", "metadata": { "kernel": "SoS" }, @@ -67,6 +70,7 @@ }, { "cell_type": "markdown", + "id": "faeec82b", "metadata": { "kernel": "SoS" }, @@ -90,6 +94,7 @@ }, { "cell_type": "markdown", + "id": "cf992056", "metadata": { "kernel": "SoS" }, @@ -99,6 +104,7 @@ }, { "cell_type": "markdown", + "id": "88f57abf", "metadata": { "kernel": "SoS" }, @@ -110,6 +116,7 @@ }, { "cell_type": "markdown", + "id": "6ee6ad73", "metadata": { "kernel": "SoS", "tags": [] @@ -121,6 +128,7 @@ { "cell_type": "code", "execution_count": null, + "id": "0bb08d67", "metadata": { "kernel": "Bash", "vscode": { @@ -138,6 +146,7 @@ }, { "cell_type": "markdown", + "id": "326e6ccc", "metadata": { "kernel": "SoS" }, @@ -149,6 +158,7 @@ }, { "cell_type": "markdown", + "id": "97059420", "metadata": { "kernel": "SoS" }, @@ -159,6 +169,7 @@ { "cell_type": "code", "execution_count": null, + "id": "97d2ff1a", "metadata": { "kernel": "SoS" }, @@ -172,6 +183,7 @@ }, { "cell_type": "markdown", + "id": "1771892b", "metadata": { "kernel": "SoS" }, @@ -183,6 +195,7 @@ }, { "cell_type": "markdown", + "id": "4ca711e2", "metadata": { "kernel": "SoS" }, @@ -193,6 +206,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4191adde", "metadata": { "kernel": "SoS" }, @@ -206,6 +220,7 @@ }, { "cell_type": "markdown", + "id": "b7200e4c", "metadata": { "kernel": "SoS" }, @@ -217,6 +232,7 @@ }, { "cell_type": "markdown", + "id": "a6d9be1d", "metadata": { "kernel": "SoS" }, @@ -227,6 +243,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f813fc12", "metadata": { "kernel": "SoS" }, @@ -241,6 +258,7 @@ }, { "cell_type": "markdown", + "id": "8c5953cd", "metadata": { "kernel": "SoS" }, @@ -252,6 +270,7 @@ }, { "cell_type": "markdown", + "id": "d2ee4e51", "metadata": { "kernel": "SoS" }, @@ -262,6 +281,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2958114f", "metadata": { "kernel": "SoS" }, @@ -275,6 +295,7 @@ }, { "cell_type": "markdown", + "id": "2b3e53c5", "metadata": { "kernel": "SoS" }, @@ -286,6 +307,7 @@ }, { "cell_type": "markdown", + "id": "3cef7a25", "metadata": { "kernel": "SoS" }, @@ -296,6 +318,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2e102029", "metadata": { "kernel": "SoS" }, @@ -309,6 +332,7 @@ }, { "cell_type": "markdown", + "id": "7adb0e7c", "metadata": { "kernel": "SoS" }, @@ -319,6 +343,7 @@ { "cell_type": "code", "execution_count": null, + "id": "93d5ad28", "metadata": { "kernel": "SoS" }, @@ -329,6 +354,7 @@ }, { "cell_type": "markdown", + "id": "63fe8232", "metadata": { "kernel": "SoS" }, @@ -404,6 +430,7 @@ }, { "cell_type": "markdown", + "id": "16ced609", "metadata": { "kernel": "SoS" }, @@ -414,6 +441,7 @@ { "cell_type": "code", "execution_count": null, + "id": "dffc9f87", "metadata": { "kernel": "SoS" }, @@ -445,6 +473,7 @@ { "cell_type": "code", "execution_count": null, + "id": "c3b3650d", "metadata": { "kernel": "SoS" }, @@ -471,6 +500,7 @@ { "cell_type": "code", "execution_count": null, + "id": "424088bc", "metadata": { "kernel": "SoS" }, @@ -492,6 +522,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2a028724", "metadata": { "kernel": "SoS" }, @@ -516,6 +547,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b17a6837", "metadata": { "kernel": "SoS" }, @@ -542,6 +574,7 @@ { "cell_type": "code", "execution_count": null, + "id": "08ecb981", "metadata": { "kernel": "SoS" }, @@ -568,6 +601,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f7ecfcb9", "metadata": { "kernel": "SoS" }, @@ -586,6 +620,7 @@ { "cell_type": "code", "execution_count": null, + "id": "728e8eca", "metadata": { "kernel": "SoS" }, @@ -609,6 +644,7 @@ { "cell_type": "code", "execution_count": null, + "id": "36cba9d6", "metadata": { "kernel": "SoS", "vscode": { @@ -661,5 +697,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/data_preprocessing/phenotype/phenotype_imputation.ipynb b/code/SoS/data_preprocessing/phenotype/phenotype_imputation.ipynb index 8e0e92e97..44439da4f 100644 --- a/code/SoS/data_preprocessing/phenotype/phenotype_imputation.ipynb +++ b/code/SoS/data_preprocessing/phenotype/phenotype_imputation.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "d7740f95", "metadata": { "kernel": "SoS", "tags": [] @@ -14,6 +15,7 @@ }, { "cell_type": "markdown", + "id": "af7af950", "metadata": { "kernel": "SoS" }, @@ -29,6 +31,7 @@ }, { "cell_type": "markdown", + "id": "7230f534", "metadata": { "kernel": "SoS" }, @@ -51,6 +54,7 @@ }, { "cell_type": "markdown", + "id": "21cb6b70", "metadata": { "kernel": "SoS" }, @@ -69,6 +73,7 @@ }, { "cell_type": "markdown", + "id": "b2443f53", "metadata": { "kernel": "SoS" }, @@ -80,6 +85,7 @@ }, { "cell_type": "markdown", + "id": "3a8399c2", "metadata": { "kernel": "SoS" }, @@ -91,6 +97,7 @@ }, { "cell_type": "markdown", + "id": "52e3e82e", "metadata": { "kernel": "SoS", "tags": [] @@ -102,6 +109,7 @@ { "cell_type": "code", "execution_count": null, + "id": "38277a02", "metadata": { "kernel": "Bash" }, @@ -115,6 +123,7 @@ }, { "cell_type": "markdown", + "id": "b88ae028", "metadata": { "kernel": "SoS" }, @@ -126,6 +135,7 @@ }, { "cell_type": "markdown", + "id": "24b3bd81", "metadata": { "kernel": "SoS", "tags": [] @@ -137,6 +147,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9c848841", "metadata": { "kernel": "Bash" }, @@ -150,6 +161,7 @@ }, { "cell_type": "markdown", + "id": "8aa28f78", "metadata": { "kernel": "SoS" }, @@ -161,6 +173,7 @@ }, { "cell_type": "markdown", + "id": "08c94c52", "metadata": { "kernel": "SoS", "tags": [] @@ -172,6 +185,7 @@ { "cell_type": "code", "execution_count": null, + "id": "62bd4464", "metadata": { "kernel": "Bash" }, @@ -185,6 +199,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4481f101", "metadata": { "kernel": "Bash" }, @@ -197,6 +212,7 @@ }, { "cell_type": "markdown", + "id": "e63d171e", "metadata": { "kernel": "SoS" }, @@ -208,6 +224,7 @@ }, { "cell_type": "markdown", + "id": "5d4714f6", "metadata": { "kernel": "SoS", "tags": [] @@ -219,6 +236,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2ff7b06b", "metadata": { "kernel": "Bash" }, @@ -231,6 +249,7 @@ }, { "cell_type": "markdown", + "id": "3ba19eff", "metadata": { "kernel": "SoS" }, @@ -242,6 +261,7 @@ }, { "cell_type": "markdown", + "id": "a3918c84", "metadata": { "kernel": "SoS", "tags": [] @@ -253,6 +273,7 @@ { "cell_type": "code", "execution_count": null, + "id": "7f70c248", "metadata": { "kernel": "Bash" }, @@ -265,6 +286,7 @@ }, { "cell_type": "markdown", + "id": "71e42cc4", "metadata": { "kernel": "SoS" }, @@ -276,6 +298,7 @@ }, { "cell_type": "markdown", + "id": "5388aebd", "metadata": { "kernel": "SoS", "tags": [] @@ -287,6 +310,7 @@ { "cell_type": "code", "execution_count": null, + "id": "83447b29", "metadata": { "kernel": "Bash" }, @@ -299,6 +323,7 @@ }, { "cell_type": "markdown", + "id": "693e214f", "metadata": { "kernel": "SoS" }, @@ -310,6 +335,7 @@ }, { "cell_type": "markdown", + "id": "49840c7e", "metadata": { "kernel": "SoS", "tags": [] @@ -321,6 +347,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6e0db9b8", "metadata": { "kernel": "Bash" }, @@ -333,6 +360,7 @@ }, { "cell_type": "markdown", + "id": "2b8b5d5c", "metadata": { "kernel": "SoS" }, @@ -344,6 +372,7 @@ }, { "cell_type": "markdown", + "id": "6cb16051", "metadata": { "kernel": "SoS" }, @@ -354,6 +383,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5ddbbc7d", "metadata": { "kernel": "SoS" }, @@ -366,6 +396,7 @@ }, { "cell_type": "markdown", + "id": "f3f51567", "metadata": { "kernel": "SoS" }, @@ -376,6 +407,7 @@ { "cell_type": "code", "execution_count": null, + "id": "d2b18573", "metadata": { "kernel": "SoS" }, @@ -386,6 +418,7 @@ }, { "cell_type": "markdown", + "id": "c22ed2f7", "metadata": { "kernel": "SoS" }, @@ -462,6 +495,7 @@ }, { "cell_type": "markdown", + "id": "c133cb7e", "metadata": { "kernel": "SoS" }, @@ -472,6 +506,7 @@ { "cell_type": "code", "execution_count": null, + "id": "59519bc0", "metadata": { "kernel": "SoS" }, @@ -501,6 +536,7 @@ { "cell_type": "code", "execution_count": null, + "id": "86e83add", "metadata": { "kernel": "SoS" }, @@ -530,6 +566,7 @@ { "cell_type": "code", "execution_count": null, + "id": "c10ca93f", "metadata": { "kernel": "SoS" }, @@ -562,6 +599,7 @@ { "cell_type": "code", "execution_count": null, + "id": "735d07b6", "metadata": { "kernel": "SoS" }, @@ -584,6 +622,7 @@ { "cell_type": "code", "execution_count": null, + "id": "dd407028", "metadata": { "kernel": "SoS" }, @@ -606,6 +645,7 @@ { "cell_type": "code", "execution_count": null, + "id": "c76960fc", "metadata": { "kernel": "SoS" }, @@ -628,6 +668,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f4d8210c", "metadata": { "kernel": "SoS" }, @@ -650,6 +691,7 @@ { "cell_type": "code", "execution_count": null, + "id": "ca80d965", "metadata": { "kernel": "SoS" }, @@ -672,6 +714,7 @@ { "cell_type": "code", "execution_count": null, + "id": "00ac4a20", "metadata": { "kernel": "SoS" }, @@ -694,6 +737,7 @@ { "cell_type": "code", "execution_count": null, + "id": "8e678174", "metadata": { "kernel": "SoS", "vscode": { @@ -753,5 +797,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/data_preprocessing/phenotype_preprocessing.ipynb b/code/SoS/data_preprocessing/phenotype_preprocessing.ipynb index d17bbf09d..6e3ae139b 100644 --- a/code/SoS/data_preprocessing/phenotype_preprocessing.ipynb +++ b/code/SoS/data_preprocessing/phenotype_preprocessing.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "4189c8e9", "metadata": { "kernel": "SoS" }, @@ -9,11 +10,11 @@ "# Phenotype preprocessing\n", "\n", "Prepare molecular-phenotype matrices for xQTL analysis by imputing missing values when needed, adding genomic annotations, and formatting data for chromosome-, region-, TAD-, or sample-specific analyses.\n" - ], - "id": "4189c8e9" + ] }, { "cell_type": "markdown", + "id": "9baa6527", "metadata": { "kernel": "SoS" }, @@ -23,25 +24,25 @@ "This is the total duration for the standard toy-data route; module-specific timings appear on their respective pages.\n", "\n", "Timing: <12 min (on the toy dataset)\n" - ], - "id": "9baa6527" + ] }, { "cell_type": "markdown", + "id": "416d15c6", "metadata": { "kernel": "SoS" }, "source": [ "## Overview\n", "\n", - "This mini-protocol walks through how molecular-phenotype matrices are prepared for downstream xQTL analysis. Each step calls a workflow from [`phenotype_imputation.ipynb`](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/phenotype/phenotype_imputation.html), [`gene_annotation.ipynb`](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/phenotype/gene_annotation.html), or [`phenotype_formatting.ipynb`](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/phenotype/phenotype_formatting.html).\n", + "This mini-protocol walks through how molecular-phenotype matrices are prepared for downstream xQTL analysis. Each step calls a workflow from [`phenotype_imputation.ipynb`](https://statfungen.github.io/xqtl-protocol/phenotype-imputation), [`gene_annotation.ipynb`](https://statfungen.github.io/xqtl-protocol/gene-annotation), or [`phenotype_formatting.ipynb`](https://statfungen.github.io/xqtl-protocol/phenotype-formatting).\n", "\n", "The commands are selectable routes rather than one mandatory chain. Impute only matrices containing missing values. Choose the annotation workflow matching gene, protein, or LeafCutter identifiers, then choose the formatting workflow required by the downstream analysis unit. GCT sample extraction and BAM subsetting are independent utilities.\n" - ], - "id": "416d15c6" + ] }, { "cell_type": "markdown", + "id": "6628e5e0", "metadata": { "kernel": "SoS" }, @@ -64,24 +65,24 @@ "| Subset BAM files to selected genomic regions | 11 | `input/rnaseq/bam_file_list.txt`; referenced BAM files; no example BAM is currently bundled |\n", "\n", "Run only the row matching the intended analysis goal, following its commands in numerical order. Other imputation methods are available through `phenotype_imputation.ipynb` and its Command Interface." - ], - "id": "6628e5e0" + ] }, { "cell_type": "markdown", + "id": "a7c33119", "metadata": { "kernel": "SoS" }, "source": [ - "### 1. [Impute missing phenotype values](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/phenotype/phenotype_imputation.html)\n", + "### 1. [Impute missing phenotype values](https://statfungen.github.io/xqtl-protocol/phenotype-imputation)\n", "\n", "**What it does:** Use generalized empirical Bayes matrix factorization to complete the example protein matrix.\n" - ], - "id": "a7c33119" + ] }, { "cell_type": "code", "execution_count": null, + "id": "c79f6386", "metadata": { "kernel": "Bash" }, @@ -91,24 +92,24 @@ " --phenoFile tests/fixtures/phenotype_imputation/protocol_example.protein.missing.bed.gz \\\n", " --cwd output/phenotype_imputation_uf \\\n", " --num_factor 30\n" - ], - "id": "c79f6386" + ] }, { "cell_type": "markdown", + "id": "6c5a94f3", "metadata": { "kernel": "SoS" }, "source": [ - "### 2. [Add genomic coordinates to gene or protein phenotypes](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/phenotype/gene_annotation.html)\n", + "### 2. [Add genomic coordinates to gene or protein phenotypes](https://statfungen.github.io/xqtl-protocol/gene-annotation)\n", "\n", "**What it does:** Join phenotype identifiers to a supplied coordinate annotation and write a coordinate-aware BED matrix.\n" - ], - "id": "6c5a94f3" + ] }, { "cell_type": "code", "execution_count": null, + "id": "dc78fd7a", "metadata": { "kernel": "Bash" }, @@ -119,24 +120,24 @@ " --phenoFile tests/fixtures/gene_annotation/protocol_example.rnaseq.bed.gz \\\n", " --coordinate-annotation tests/fixtures/gene_annotation/Homo_sapiens.GRCh38.103.collapse_only.gene.chr22.gtf.gz \\\n", " --phenotype-id-column gene_id\n" - ], - "id": "dc78fd7a" + ] }, { "cell_type": "markdown", + "id": "4a64a987", "metadata": { "kernel": "SoS" }, "source": [ - "### 3. [Retrieve gene coordinates from Ensembl BioMart](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/phenotype/gene_annotation.html)\n", + "### 3. [Retrieve gene coordinates from Ensembl BioMart](https://statfungen.github.io/xqtl-protocol/gene-annotation)\n", "\n", "**What it does:** Query the selected Ensembl release when a local coordinate annotation is unavailable.\n" - ], - "id": "4a64a987" + ] }, { "cell_type": "code", "execution_count": null, + "id": "7401792f", "metadata": { "kernel": "Bash" }, @@ -146,24 +147,24 @@ " --cwd output/gene_annotation \\\n", " --phenoFile tests/fixtures/gene_annotation/protocol_example.rnaseq.gene_ID.tsv \\\n", " --ensembl-version 115\n" - ], - "id": "7401792f" + ] }, { "cell_type": "markdown", + "id": "e7409492", "metadata": { "kernel": "SoS" }, "source": [ - "### 4. [Map LeafCutter clusters to genes](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/phenotype/gene_annotation.html)\n", + "### 4. [Map LeafCutter clusters to genes](https://statfungen.github.io/xqtl-protocol/gene-annotation)\n", "\n", "**What it does:** Assign LeafCutter clusters to genes using splice-site overlap with the gene annotation.\n" - ], - "id": "e7409492" + ] }, { "cell_type": "code", "execution_count": null, + "id": "355635fc", "metadata": { "kernel": "Bash" }, @@ -175,24 +176,24 @@ " --intron-count tests/fixtures/gene_annotation/protocol_example.leafcutter.intron_count.tsv \\\n", " --coordinate-annotation \\\n", " --map-stra site\n" - ], - "id": "355635fc" + ] }, { "cell_type": "markdown", + "id": "16afacb2", "metadata": { "kernel": "SoS" }, "source": [ - "### 5. [Annotate LeafCutter isoforms](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/phenotype/gene_annotation.html)\n", + "### 5. [Annotate LeafCutter isoforms](https://statfungen.github.io/xqtl-protocol/gene-annotation)\n", "\n", "**What it does:** Convert LeafCutter intron-cluster phenotypes into annotated isoform features for QTL analysis.\n" - ], - "id": "16afacb2" + ] }, { "cell_type": "code", "execution_count": null, + "id": "72809d29", "metadata": { "kernel": "Bash" }, @@ -204,24 +205,24 @@ " --intron-count tests/fixtures/gene_annotation/protocol_example.leafcutter.intron_count.tsv \\\n", " --coordinate-annotation \\\n", " --map-stra site\n" - ], - "id": "72809d29" + ] }, { "cell_type": "markdown", + "id": "557f48cf", "metadata": { "kernel": "SoS" }, "source": [ - "### 6. [Partition a BED phenotype by chromosome](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/phenotype/phenotype_formatting.html)\n", + "### 6. [Partition a BED phenotype by chromosome](https://statfungen.github.io/xqtl-protocol/phenotype-formatting)\n", "\n", "**What it does:** Split a coordinate-annotated BED phenotype into chromosome-specific files.\n" - ], - "id": "557f48cf" + ] }, { "cell_type": "code", "execution_count": null, + "id": "7c3143df", "metadata": { "kernel": "Bash" }, @@ -232,24 +233,24 @@ " --phenoFile tests/fixtures/phenotype_formatting/protocol_example.rnaseq.bed.bed.gz \\\n", " --name protocol_example \\\n", " --chrom chr22\n" - ], - "id": "7c3143df" + ] }, { "cell_type": "markdown", + "id": "ff8a1cbe", "metadata": { "kernel": "SoS" }, "source": [ - "### 7. [Partition a GCT phenotype by chromosome](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/phenotype/phenotype_formatting.html)\n", + "### 7. [Partition a GCT phenotype by chromosome](https://statfungen.github.io/xqtl-protocol/phenotype-formatting)\n", "\n", "**What it does:** Split a coordinate-aware GCT matrix into chromosome-specific GCT files.\n" - ], - "id": "ff8a1cbe" + ] }, { "cell_type": "code", "execution_count": null, + "id": "7f114e77", "metadata": { "kernel": "Bash" }, @@ -259,24 +260,24 @@ " --cwd output/phenotype_gct \\\n", " --phenoFile output/phenotype/phenotype_by_chrom_for_cis/protocol_example.rnaseq.gene_tpm.gct.gz \\\n", " --chrom chr21 chr22\n" - ], - "id": "7f114e77" + ] }, { "cell_type": "markdown", + "id": "90267404", "metadata": { "kernel": "SoS" }, "source": [ - "### 8. [Partition a phenotype by predefined regions](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/phenotype/phenotype_formatting.html)\n", + "### 8. [Partition a phenotype by predefined regions](https://statfungen.github.io/xqtl-protocol/phenotype-formatting)\n", "\n", "**What it does:** Extract phenotype features falling within each region in a supplied region list.\n" - ], - "id": "90267404" + ] }, { "cell_type": "code", "execution_count": null, + "id": "cb9b0cde", "metadata": { "kernel": "Bash" }, @@ -286,24 +287,24 @@ " --cwd output/phenotype_by_region \\\n", " --phenoFile tests/fixtures/phenotype_formatting/protocol_example.rnaseq.bed.bed.gz \\\n", " --region-list output/phenotype/phenotype_by_chrom_for_cis/protocol_example_protein.enhanced_cis_chr22.bed\n" - ], - "id": "cb9b0cde" + ] }, { "cell_type": "markdown", + "id": "cfce416b", "metadata": { "kernel": "SoS" }, "source": [ - "### 9. [Define TAD-based phenotype regions](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/phenotype/phenotype_formatting.html)\n", + "### 9. [Define TAD-based phenotype regions](https://statfungen.github.io/xqtl-protocol/phenotype-formatting)\n", "\n", "**What it does:** Assign phenotype features to TAD windows and generate a region list for downstream analysis.\n" - ], - "id": "cfce416b" + ] }, { "cell_type": "code", "execution_count": null, + "id": "107d5715", "metadata": { "kernel": "Bash" }, @@ -314,24 +315,24 @@ " --phenoFile tests/fixtures/phenotype_formatting/protocol_example.rnaseq.bed.bed.gz \\\n", " --TAD-list tests/fixtures/generalized_TADB/expected/TADB_enhanced_cis.bed \\\n", " --phenotype-per-tad 2\n" - ], - "id": "107d5715" + ] }, { "cell_type": "markdown", + "id": "03c6b4ce", "metadata": { "kernel": "SoS" }, "source": [ - "### 10. [Extract selected samples from a GCT matrix](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/phenotype/phenotype_formatting.html)\n", + "### 10. [Extract selected samples from a GCT matrix](https://statfungen.github.io/xqtl-protocol/phenotype-formatting)\n", "\n", "**What it does:** Retain only samples listed in a supplied keep file.\n" - ], - "id": "03c6b4ce" + ] }, { "cell_type": "code", "execution_count": null, + "id": "addabe04", "metadata": { "kernel": "Bash" }, @@ -341,24 +342,24 @@ " --cwd output/phenotype_gct \\\n", " --phenoFile output/phenotype/phenotype_by_chrom_for_cis/protocol_example.rnaseq.gene_tpm.gct.gz \\\n", " --keep-samples tests/fixtures/phenotype_formatting/keep_samples.txt\n" - ], - "id": "addabe04" + ] }, { "cell_type": "markdown", + "id": "720706d3", "metadata": { "kernel": "SoS" }, "source": [ - "### 11. [Subset BAM files by genomic region](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/phenotype/phenotype_formatting.html)\n", + "### 11. [Subset BAM files by genomic region](https://statfungen.github.io/xqtl-protocol/phenotype-formatting)\n", "\n", "**What it does:** Extract selected chromosomes or regions from every BAM listed in the input manifest.\n" - ], - "id": "720706d3" + ] }, { "cell_type": "code", "execution_count": null, + "id": "d5c0f358", "metadata": { "kernel": "Bash" }, @@ -368,11 +369,11 @@ " --cwd output/bam_subset \\\n", " --phenoFile output/phenotype/phenotype_by_chrom_for_cis/bam_file_list.txt \\\n", " --region chr21 chr22\n" - ], - "id": "d5c0f358" + ] }, { "cell_type": "markdown", + "id": "48f58267", "metadata": { "kernel": "SoS" }, @@ -392,11 +393,11 @@ "| Step 9 | `output/phenotype_by_region/*_pheno_per_region.region_list`; TAD-annotated region list under `output/phenotype_by_region/` |\n", "| Step 10 | `output/phenotype_gct/protocol_example.rnaseq.gene_tpm.sample_matched.gct.gz` |\n", "| Step 11 | `output/bam_subset/*.subsetted.bam` |" - ], - "id": "48f58267" + ] }, { "cell_type": "markdown", + "id": "c7bd9755", "metadata": { "kernel": "SoS" }, @@ -406,11 +407,11 @@ "The selected route produces a molecular-phenotype matrix with the genomic annotation and layout required by its downstream xQTL analysis. Standard gene-expression routes typically end with chromosome-specific BED files; splicing routes end with gene- or isoform-annotated phenotypes; region and TAD routes end with per-region files and their manifest.\n", "\n", "Continue with covariate preprocessing and then the appropriate association-testing workflow.\n" - ], - "id": "c7bd9755" + ] }, { "cell_type": "markdown", + "id": "08582e7a", "metadata": { "kernel": "SoS" }, @@ -418,12 +419,12 @@ "## Command Interface\n", "\n", "List the workflows and parameters available in each module used by this mini-protocol.\n" - ], - "id": "08582e7a" + ] }, { "cell_type": "code", "execution_count": null, + "id": "a71cd0bd", "metadata": { "kernel": "Bash" }, @@ -432,8 +433,7 @@ "sos run pipeline/phenotype_imputation.ipynb -h\n", "sos run pipeline/gene_annotation.ipynb -h\n", "sos run pipeline/phenotype_formatting.ipynb -h\n" - ], - "id": "a71cd0bd" + ] } ], "metadata": { diff --git a/code/SoS/enrichment/enrichment_validation.ipynb b/code/SoS/enrichment/enrichment_validation.ipynb index bb66cbf2d..2bae78c06 100644 --- a/code/SoS/enrichment/enrichment_validation.ipynb +++ b/code/SoS/enrichment/enrichment_validation.ipynb @@ -31,7 +31,7 @@ }, "source": [ "## Overview\n", - "This mini-protocol provides alternative functional follow-up routes. Step 1 calls [`gsea.ipynb`](https://statfungen.github.io/xqtl-protocol/code/enrichment/gsea.html), step 2 calls [`eoo_enrichment.ipynb`](https://statfungen.github.io/xqtl-protocol/code/enrichment/eoo_enrichment.html), steps 3–5 call [`gregor.ipynb`](https://statfungen.github.io/xqtl-protocol/code/enrichment/gregor.html), and steps 6–9 call [`sldsc_enrichment.ipynb`](https://statfungen.github.io/xqtl-protocol/code/enrichment/sldsc_enrichment.html).\n", + "This mini-protocol provides alternative functional follow-up routes. Step 1 calls [`gsea.ipynb`](https://statfungen.github.io/xqtl-protocol/gsea), step 2 calls [`eoo_enrichment.ipynb`](https://statfungen.github.io/xqtl-protocol/eoo-enrichment), steps 3–5 call [`gregor.ipynb`](https://statfungen.github.io/xqtl-protocol/gregor), and steps 6–9 call [`sldsc_enrichment.ipynb`](https://statfungen.github.io/xqtl-protocol/sldsc-enrichment).\n", "Pathway analysis, enrichment-over-odds, GREGOR, and S-LDSC use different inputs and null models. Select the route matching the scientific question; only the numbered commands within the GREGOR and S-LDSC routes form ordered chains." ] }, @@ -63,7 +63,7 @@ "kernel": "SoS" }, "source": [ - "### 1. [Test pathway and GO enrichment](https://statfungen.github.io/xqtl-protocol/code/enrichment/gsea.html)\n", + "### 1. [Test pathway and GO enrichment](https://statfungen.github.io/xqtl-protocol/gsea)\n", "\n", "**What it does:** Maps grouped genes to ENTREZ identifiers and tests KEGG and GO BP/CC/MF over-representation for each group." ] @@ -101,7 +101,7 @@ "kernel": "SoS" }, "source": [ - "### 2. [Estimate enrichment over odds](https://statfungen.github.io/xqtl-protocol/code/enrichment/eoo_enrichment.html)\n", + "### 2. [Estimate enrichment over odds](https://statfungen.github.io/xqtl-protocol/eoo-enrichment)\n", "\n", "**What it does:** Estimates annotation odds ratios and enrichment with chromosome block-jackknife uncertainty." ] @@ -140,7 +140,7 @@ "kernel": "SoS" }, "source": [ - "### 3. [Create a GREGOR configuration](https://statfungen.github.io/xqtl-protocol/code/enrichment/gregor.html)\n", + "### 3. [Create a GREGOR configuration](https://statfungen.github.io/xqtl-protocol/gregor)\n", "\n", "**What it does:** Writes the configuration connecting index SNPs, annotation BED files, population settings, and the GREGOR reference database." ] @@ -179,7 +179,7 @@ "kernel": "SoS" }, "source": [ - "### 4. [Run GREGOR enrichment](https://statfungen.github.io/xqtl-protocol/code/enrichment/gregor.html)\n", + "### 4. [Run GREGOR enrichment](https://statfungen.github.io/xqtl-protocol/gregor)\n", "\n", "**What it does:** Compares annotation overlap for index SNPs with overlap among LD- and frequency-matched control variants." ] @@ -218,7 +218,7 @@ "kernel": "SoS" }, "source": [ - "### 5. [Plot GREGOR Fisher enrichment](https://statfungen.github.io/xqtl-protocol/code/enrichment/gregor.html)\n", + "### 5. [Plot GREGOR Fisher enrichment](https://statfungen.github.io/xqtl-protocol/gregor)\n", "\n", "**What it does:** Compares annotation odds ratios from two GREGOR result sets." ] @@ -255,7 +255,7 @@ "kernel": "SoS" }, "source": [ - "### 6. [Build annotation LD scores](https://statfungen.github.io/xqtl-protocol/code/enrichment/sldsc_enrichment.html)\n", + "### 6. [Build annotation LD scores](https://statfungen.github.io/xqtl-protocol/sldsc-enrichment)\n", "\n", "**What it does:** Converts genomic annotations into chromosome-level annotation and LD-score files for stratified LD-score regression." ] @@ -295,7 +295,7 @@ "kernel": "SoS" }, "source": [ - "### 7. [Estimate stratified SNP heritability](https://statfungen.github.io/xqtl-protocol/code/enrichment/sldsc_enrichment.html)\n", + "### 7. [Estimate stratified SNP heritability](https://statfungen.github.io/xqtl-protocol/sldsc-enrichment)\n", "\n", "**What it does:** Runs S-LDSC for each trait and annotation target to estimate annotation-specific heritability enrichment." ] @@ -338,7 +338,7 @@ "kernel": "SoS" }, "source": [ - "### 8. [Postprocess and meta-analyze S-LDSC results](https://statfungen.github.io/xqtl-protocol/code/enrichment/sldsc_enrichment.html)\n", + "### 8. [Postprocess and meta-analyze S-LDSC results](https://statfungen.github.io/xqtl-protocol/sldsc-enrichment)\n", "\n", "**What it does:** Standardizes per-trait results and computes random-effects meta-analyses across traits." ] @@ -378,7 +378,7 @@ "kernel": "SoS" }, "source": [ - "### 9. [Re-meta-analyze a trait subset](https://statfungen.github.io/xqtl-protocol/code/enrichment/sldsc_enrichment.html)\n", + "### 9. [Re-meta-analyze a trait subset](https://statfungen.github.io/xqtl-protocol/sldsc-enrichment)\n", "\n", "**What it does:** Reuses postprocessed S-LDSC results to estimate enrichment for a selected subset without rerunning regression." ] @@ -536,4 +536,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/enrichment/eoo_enrichment.ipynb b/code/SoS/enrichment/eoo_enrichment.ipynb index 367e1f67d..ea9938a98 100644 --- a/code/SoS/enrichment/eoo_enrichment.ipynb +++ b/code/SoS/enrichment/eoo_enrichment.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "18f7b84b", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "4d1a4f91", "metadata": { "kernel": "SoS" }, @@ -28,6 +30,7 @@ }, { "cell_type": "markdown", + "id": "79b46029", "metadata": { "kernel": "SoS", "vscode": { @@ -60,6 +63,7 @@ { "cell_type": "code", "execution_count": 1, + "id": "34dc19eb", "metadata": { "kernel": "Python3", "tags": [] @@ -175,6 +179,7 @@ }, { "cell_type": "markdown", + "id": "5112d409", "metadata": { "kernel": "SoS" }, @@ -184,6 +189,7 @@ }, { "cell_type": "markdown", + "id": "d61d1675", "metadata": { "kernel": "SoS" }, @@ -214,6 +220,7 @@ }, { "cell_type": "markdown", + "id": "438d5a6d", "metadata": { "kernel": "SoS" }, @@ -241,6 +248,7 @@ }, { "cell_type": "markdown", + "id": "9943d1cb", "metadata": { "kernel": "SoS" }, @@ -252,6 +260,7 @@ }, { "cell_type": "markdown", + "id": "35116962", "metadata": { "kernel": "SoS" }, @@ -262,6 +271,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e4574071", "metadata": { "kernel": "Bash" }, @@ -277,6 +287,7 @@ }, { "cell_type": "markdown", + "id": "6a1e0d47", "metadata": { "kernel": "SoS" }, @@ -287,6 +298,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9b5e7bfa", "metadata": { "kernel": "Bash" }, @@ -297,6 +309,7 @@ }, { "cell_type": "markdown", + "id": "c00177e2", "metadata": { "kernel": "SoS" }, @@ -337,6 +350,7 @@ }, { "cell_type": "markdown", + "id": "b42d0f96", "metadata": { "kernel": "SoS" }, @@ -347,6 +361,7 @@ { "cell_type": "code", "execution_count": null, + "id": "78762186", "metadata": { "kernel": "SoS" }, @@ -373,6 +388,7 @@ { "cell_type": "code", "execution_count": null, + "id": "23abedd4", "metadata": { "kernel": "SoS" }, @@ -434,5 +450,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/enrichment/gregor.ipynb b/code/SoS/enrichment/gregor.ipynb index 097170728..4c7f7cacd 100644 --- a/code/SoS/enrichment/gregor.ipynb +++ b/code/SoS/enrichment/gregor.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "c865e8a7", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "dd3b8ab6", "metadata": { "kernel": "SoS" }, @@ -28,6 +30,7 @@ }, { "cell_type": "markdown", + "id": "c154b31c", "metadata": { "kernel": "SoS" }, @@ -66,6 +69,7 @@ }, { "cell_type": "markdown", + "id": "20d88223", "metadata": { "kernel": "SoS" }, @@ -92,6 +96,7 @@ }, { "cell_type": "markdown", + "id": "35f565f3", "metadata": { "kernel": "SoS" }, @@ -103,6 +108,7 @@ }, { "cell_type": "markdown", + "id": "258bd8d2", "metadata": { "kernel": "SoS" }, @@ -112,6 +118,7 @@ }, { "cell_type": "markdown", + "id": "6d07fd96", "metadata": { "kernel": "SoS" }, @@ -123,6 +130,7 @@ }, { "cell_type": "markdown", + "id": "264db894", "metadata": { "kernel": "SoS" }, @@ -133,6 +141,7 @@ { "cell_type": "code", "execution_count": null, + "id": "647507c1", "metadata": { "kernel": "Bash" }, @@ -148,6 +157,7 @@ }, { "cell_type": "markdown", + "id": "f6558c67", "metadata": { "kernel": "SoS" }, @@ -159,6 +169,7 @@ }, { "cell_type": "markdown", + "id": "078d0c7f", "metadata": { "kernel": "SoS" }, @@ -169,6 +180,7 @@ { "cell_type": "code", "execution_count": null, + "id": "562de313", "metadata": { "kernel": "Bash" }, @@ -184,6 +196,7 @@ }, { "cell_type": "markdown", + "id": "d03f631a", "metadata": { "kernel": "SoS" }, @@ -195,6 +208,7 @@ }, { "cell_type": "markdown", + "id": "16fc9ba8", "metadata": { "kernel": "SoS" }, @@ -205,6 +219,7 @@ { "cell_type": "code", "execution_count": null, + "id": "775cf82a", "metadata": { "kernel": "Bash" }, @@ -218,6 +233,7 @@ }, { "cell_type": "markdown", + "id": "bb511f90", "metadata": { "kernel": "SoS" }, @@ -228,6 +244,7 @@ { "cell_type": "code", "execution_count": null, + "id": "449ce3e2", "metadata": { "kernel": "SoS" }, @@ -238,6 +255,7 @@ }, { "cell_type": "markdown", + "id": "7e0d002f", "metadata": { "kernel": "SoS" }, @@ -283,6 +301,7 @@ }, { "cell_type": "markdown", + "id": "992c23f2", "metadata": { "kernel": "SoS" }, @@ -293,6 +312,7 @@ { "cell_type": "code", "execution_count": 6, + "id": "e74d01e3", "metadata": { "kernel": "SoS" }, @@ -308,6 +328,7 @@ { "cell_type": "code", "execution_count": null, + "id": "42500f47", "metadata": { "kernel": "SoS" }, @@ -359,29 +380,26 @@ }, { "cell_type": "markdown", + "id": "53210f12", "metadata": { "kernel": "SoS" }, "source": [ - "GREGOR is written in `perl`. If you don't use containers, some libraries are required before one can run GREGOR:\n", + "GREGOR is an external `perl` tool that is not ported to R; it is installed from the `gregor` conda package and is on `PATH` as `GREGOR`, so `gregor_2` below simply calls it. The package brings its own `perl` and the modules GREGOR needs (`DBI`, `Switch`, `DBD::SQLite`), so no separate library installation is required — note that this means a bare `perl -MDBI` in your shell can still fail while GREGOR itself works.\n", + "\n", + "To run a scan by hand, generate the configuration file with the `gregor_conf` workflow above and pass it to GREGOR directly:\n", "\n", "```\n", - "sudo apt-get install libdbi-perl libswitch-perl libdbd-sqlite3-perl\n", + "GREGOR --conf output/gregor/index.gregor.conf\n", "```\n", "\n", - "With the docker image:\n", - "\n", - "```\n", - "cd GREGOR_folder\n", - "docker run -v \"$PWD:/usr/src/myapp\" -it custom-perl\n", - "perl script/GREGOR --conf example/mvsusie_annotation.conf\n", - "perl script/GREGOR --conf example/susie_annotation.conf\n", - "```" + "`--ld_window_size` must match the window the reference database's r2 tables were built with, and `--min_neighbor` must be small enough for the matched-control bins that reference actually contains. A mismatch does not raise an error: GREGOR keeps searching for matched controls it will never find, so the run simply appears to hang.\n" ] }, { "cell_type": "code", "execution_count": null, + "id": "6f0966e9", "metadata": { "kernel": "SoS" }, @@ -397,6 +415,7 @@ { "cell_type": "code", "execution_count": 1, + "id": "862eae81", "metadata": { "kernel": "SoS" }, @@ -494,6 +513,7 @@ { "cell_type": "code", "execution_count": 1, + "id": "3e8ffaa0", "metadata": { "kernel": "SoS" }, @@ -564,5 +584,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/enrichment/gsea.ipynb b/code/SoS/enrichment/gsea.ipynb index 46e853b3b..e97c8c0ec 100644 --- a/code/SoS/enrichment/gsea.ipynb +++ b/code/SoS/enrichment/gsea.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "eb744614", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "01a77d05", "metadata": { "kernel": "SoS" }, @@ -30,6 +32,7 @@ }, { "cell_type": "markdown", + "id": "45214eea", "metadata": { "kernel": "SoS" }, @@ -56,6 +59,7 @@ }, { "cell_type": "markdown", + "id": "8ab2f7f8", "metadata": { "kernel": "SoS" }, @@ -87,6 +91,7 @@ }, { "cell_type": "markdown", + "id": "4cc7ea13", "metadata": { "kernel": "SoS" }, @@ -96,6 +101,7 @@ }, { "cell_type": "markdown", + "id": "02e90f9b", "metadata": { "kernel": "SoS" }, @@ -105,6 +111,7 @@ }, { "cell_type": "markdown", + "id": "c35ca82d", "metadata": { "kernel": "SoS" }, @@ -115,6 +122,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6d43f9a2", "metadata": { "kernel": "Bash" }, @@ -129,6 +137,7 @@ }, { "cell_type": "markdown", + "id": "11b068e8", "metadata": { "kernel": "SoS" }, @@ -139,6 +148,7 @@ { "cell_type": "code", "execution_count": null, + "id": "dbd07121", "metadata": { "kernel": "Bash" }, @@ -149,6 +159,7 @@ }, { "cell_type": "markdown", + "id": "4ee233b8", "metadata": {}, "source": [ "```\n", @@ -185,6 +196,7 @@ }, { "cell_type": "markdown", + "id": "7d2556a5", "metadata": { "kernel": "SoS" }, @@ -195,6 +207,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f4d94e45", "metadata": { "kernel": "SoS" }, @@ -218,6 +231,7 @@ { "cell_type": "code", "execution_count": null, + "id": "d4d026bb", "metadata": { "kernel": "SoS" }, @@ -273,5 +287,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/enrichment/sldsc_enrichment.ipynb b/code/SoS/enrichment/sldsc_enrichment.ipynb index 21c99fab5..46c6d6bfc 100644 --- a/code/SoS/enrichment/sldsc_enrichment.ipynb +++ b/code/SoS/enrichment/sldsc_enrichment.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "56aafa68", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "14539361", "metadata": { "kernel": "SoS" }, @@ -48,6 +50,7 @@ }, { "cell_type": "markdown", + "id": "020a7283", "metadata": { "kernel": "SoS" }, @@ -73,6 +76,7 @@ }, { "cell_type": "markdown", + "id": "80d0628b", "metadata": { "kernel": "SoS" }, @@ -110,6 +114,7 @@ }, { "cell_type": "markdown", + "id": "15a68b4c", "metadata": { "kernel": "SoS" }, @@ -219,6 +224,7 @@ }, { "cell_type": "markdown", + "id": "eb0f12cd", "metadata": { "kernel": "SoS" }, @@ -228,6 +234,7 @@ }, { "cell_type": "markdown", + "id": "7e26e34e", "metadata": { "kernel": "SoS" }, @@ -327,6 +334,7 @@ }, { "cell_type": "markdown", + "id": "05710933", "metadata": { "kernel": "SoS" }, @@ -374,6 +382,7 @@ }, { "cell_type": "markdown", + "id": "10b7bb5f", "metadata": { "kernel": "SoS" }, @@ -383,6 +392,7 @@ }, { "cell_type": "markdown", + "id": "05c6b08e", "metadata": { "kernel": "SoS" }, @@ -394,6 +404,7 @@ }, { "cell_type": "markdown", + "id": "05c6161f", "metadata": { "kernel": "SoS" }, @@ -404,6 +415,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b9cebce2", "metadata": { "kernel": "SoS" }, @@ -420,6 +432,7 @@ }, { "cell_type": "markdown", + "id": "c0a7e878", "metadata": { "kernel": "SoS" }, @@ -432,6 +445,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e7209a62", "metadata": { "kernel": "SoS" }, @@ -448,6 +462,7 @@ }, { "cell_type": "markdown", + "id": "45d007e7", "metadata": { "kernel": "SoS" }, @@ -457,6 +472,7 @@ }, { "cell_type": "markdown", + "id": "d624a292", "metadata": { "kernel": "SoS" }, @@ -467,6 +483,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6ada30fd", "metadata": { "kernel": "SoS" }, @@ -486,6 +503,7 @@ }, { "cell_type": "markdown", + "id": "b9f68c6c", "metadata": { "kernel": "SoS" }, @@ -505,6 +523,7 @@ }, { "cell_type": "markdown", + "id": "b1190b63", "metadata": { "kernel": "SoS" }, @@ -515,6 +534,7 @@ { "cell_type": "code", "execution_count": null, + "id": "03c0bfda", "metadata": { "kernel": "SoS" }, @@ -531,6 +551,7 @@ }, { "cell_type": "markdown", + "id": "3949c1e8", "metadata": { "kernel": "SoS" }, @@ -553,6 +574,7 @@ }, { "cell_type": "markdown", + "id": "0586b199", "metadata": { "kernel": "SoS" }, @@ -563,6 +585,7 @@ { "cell_type": "code", "execution_count": null, + "id": "79cb2f19", "metadata": { "kernel": "SoS" }, @@ -578,6 +601,7 @@ }, { "cell_type": "markdown", + "id": "6380336a", "metadata": { "kernel": "SoS" }, @@ -588,6 +612,7 @@ { "cell_type": "code", "execution_count": null, + "id": "729f81e0", "metadata": { "kernel": "SoS" }, @@ -598,6 +623,7 @@ }, { "cell_type": "markdown", + "id": "aab45759", "metadata": { "kernel": "SoS" }, @@ -756,6 +782,7 @@ }, { "cell_type": "markdown", + "id": "5ed932ce", "metadata": { "kernel": "SoS" }, @@ -768,6 +795,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a3acd526", "metadata": { "kernel": "SoS" }, @@ -821,6 +849,7 @@ }, { "cell_type": "markdown", + "id": "644c37ba", "metadata": { "kernel": "Python 3 (ipykernel)" }, @@ -831,6 +860,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f5a4b943", "metadata": { "kernel": "SoS" }, @@ -1052,6 +1082,7 @@ }, { "cell_type": "markdown", + "id": "6df09656", "metadata": { "kernel": "Python 3 (ipykernel)" }, @@ -1062,6 +1093,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b1c03692", "metadata": { "kernel": "SoS" }, @@ -1184,6 +1216,7 @@ { "cell_type": "code", "execution_count": null, + "id": "45a2c84b", "metadata": { "kernel": "SoS" }, @@ -1213,6 +1246,7 @@ { "cell_type": "code", "execution_count": null, + "id": "892e1918", "metadata": { "kernel": "SoS" }, @@ -1258,6 +1292,7 @@ { "cell_type": "code", "execution_count": null, + "id": "7448205f", "metadata": { "kernel": "SoS" }, @@ -1335,5 +1370,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/graveyard/APEX/APEX.ipynb b/code/SoS/graveyard/APEX/APEX.ipynb index 40541baf8..01eea150b 100644 --- a/code/SoS/graveyard/APEX/APEX.ipynb +++ b/code/SoS/graveyard/APEX/APEX.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "4422f502", "metadata": { "kernel": "SoS" }, @@ -22,6 +23,7 @@ }, { "cell_type": "markdown", + "id": "88ac99c7", "metadata": { "kernel": "SoS" }, @@ -64,6 +66,7 @@ }, { "cell_type": "markdown", + "id": "c62aedd3", "metadata": { "kernel": "SoS" }, @@ -74,6 +77,7 @@ { "cell_type": "code", "execution_count": 1, + "id": "0c862d82", "metadata": { "kernel": "Bash" }, @@ -145,6 +149,7 @@ }, { "cell_type": "markdown", + "id": "fc6b66a2", "metadata": { "kernel": "SoS" }, @@ -157,6 +162,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b526e906", "metadata": { "kernel": "SoS" }, @@ -173,6 +179,7 @@ }, { "cell_type": "markdown", + "id": "4dcd625f", "metadata": { "kernel": "SoS" }, @@ -184,6 +191,7 @@ { "cell_type": "code", "execution_count": 5, + "id": "9572dd69", "metadata": { "kernel": "SoS", "tags": [] @@ -232,6 +240,7 @@ }, { "cell_type": "markdown", + "id": "e2e63deb", "metadata": { "kernel": "SoS" }, @@ -245,6 +254,7 @@ { "cell_type": "code", "execution_count": null, + "id": "20df91fa", "metadata": { "kernel": "SoS" }, @@ -268,6 +278,7 @@ }, { "cell_type": "markdown", + "id": "22bc7ba5", "metadata": { "kernel": "SoS" }, @@ -279,6 +290,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f1bced97", "metadata": { "kernel": "SoS" }, @@ -305,6 +317,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a2737fcb", "metadata": { "kernel": "SoS" }, @@ -333,6 +346,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a1ff572f", "metadata": { "kernel": "SoS" }, @@ -354,6 +368,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5dca83de", "metadata": { "kernel": "SoS" }, @@ -418,5 +433,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/graveyard/GRM.ipynb b/code/SoS/graveyard/GRM.ipynb index 25eb5f56e..a96cc9ba6 100644 --- a/code/SoS/graveyard/GRM.ipynb +++ b/code/SoS/graveyard/GRM.ipynb @@ -3,6 +3,7 @@ { "attachments": {}, "cell_type": "markdown", + "id": "59667864", "metadata": { "kernel": "SoS", "tags": [] @@ -16,6 +17,7 @@ { "attachments": {}, "cell_type": "markdown", + "id": "6ff3ceaa", "metadata": { "kernel": "SoS" }, @@ -34,6 +36,7 @@ { "attachments": {}, "cell_type": "markdown", + "id": "110584e1", "metadata": { "kernel": "SoS", "tags": [] @@ -57,6 +60,7 @@ { "attachments": {}, "cell_type": "markdown", + "id": "7910715e", "metadata": { "kernel": "SoS", "tags": [] @@ -85,6 +89,7 @@ { "attachments": {}, "cell_type": "markdown", + "id": "cd8d6685", "metadata": { "kernel": "SoS" }, @@ -95,6 +100,7 @@ { "attachments": {}, "cell_type": "markdown", + "id": "ff8f355a", "metadata": { "kernel": "SoS", "tags": [] @@ -106,6 +112,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2bd274f9", "metadata": { "kernel": "Bash" }, @@ -119,6 +126,7 @@ { "attachments": {}, "cell_type": "markdown", + "id": "e7755f0c", "metadata": { "kernel": "SoS" }, @@ -129,6 +137,7 @@ { "cell_type": "code", "execution_count": null, + "id": "d753ab8c", "metadata": { "kernel": "Bash" }, @@ -139,6 +148,7 @@ }, { "cell_type": "markdown", + "id": "d7e9191e", "metadata": { "kernel": "SoS" }, @@ -181,6 +191,7 @@ }, { "cell_type": "markdown", + "id": "fa854715", "metadata": { "kernel": "SoS" }, @@ -191,6 +202,7 @@ { "cell_type": "code", "execution_count": null, + "id": "14cb5e4d", "metadata": { "kernel": "SoS" }, @@ -267,6 +279,7 @@ { "attachments": {}, "cell_type": "markdown", + "id": "9b55c1ef", "metadata": { "kernel": "SoS" }, @@ -277,6 +290,7 @@ { "cell_type": "code", "execution_count": null, + "id": "3274b3eb", "metadata": { "kernel": "SoS" }, @@ -294,6 +308,7 @@ { "attachments": {}, "cell_type": "markdown", + "id": "b0845513", "metadata": { "kernel": "SoS" }, @@ -304,6 +319,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6da2203f", "metadata": { "kernel": "SoS" }, @@ -323,6 +339,7 @@ { "attachments": {}, "cell_type": "markdown", + "id": "c3c5c8a7", "metadata": { "kernel": "SoS" }, @@ -333,6 +350,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2fd60872", "metadata": { "kernel": "SoS" }, @@ -352,6 +370,7 @@ { "attachments": {}, "cell_type": "markdown", + "id": "aaf816be", "metadata": { "kernel": "SoS" }, @@ -362,6 +381,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6f7ab041", "metadata": { "kernel": "SoS" }, @@ -414,5 +434,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 + "nbformat_minor": 5 } diff --git a/code/SoS/graveyard/MMQTL/MMQTL.ipynb b/code/SoS/graveyard/MMQTL/MMQTL.ipynb index aad7ae94c..753414323 100644 --- a/code/SoS/graveyard/MMQTL/MMQTL.ipynb +++ b/code/SoS/graveyard/MMQTL/MMQTL.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "cc25c4b2", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "e1d61169", "metadata": { "kernel": "SoS", "tags": [] @@ -31,6 +33,7 @@ }, { "cell_type": "markdown", + "id": "7659c179", "metadata": { "kernel": "SoS", "tags": [] @@ -65,6 +68,7 @@ }, { "cell_type": "markdown", + "id": "6e4e9c12", "metadata": { "jp-MarkdownHeadingCollapsed": true, "kernel": "SoS", @@ -77,6 +81,7 @@ }, { "cell_type": "markdown", + "id": "12ecb78a", "metadata": { "kernel": "SoS", "tags": [] @@ -104,6 +109,7 @@ }, { "cell_type": "markdown", + "id": "b2056b9b", "metadata": { "kernel": "SoS" }, @@ -114,6 +120,7 @@ { "cell_type": "code", "execution_count": 1, + "id": "099325b2", "metadata": { "kernel": "SoS" }, @@ -154,6 +161,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9b6ca2b3", "metadata": { "kernel": "SoS" }, @@ -233,6 +241,7 @@ { "cell_type": "code", "execution_count": null, + "id": "bef92d3a", "metadata": {}, "outputs": [], "source": [ @@ -262,6 +271,7 @@ { "cell_type": "code", "execution_count": null, + "id": "28656c86", "metadata": {}, "outputs": [], "source": [ @@ -305,5 +315,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/graveyard/SuSiE_slides.ipynb b/code/SoS/graveyard/SuSiE_slides.ipynb index 5db1e315a..4f95cf73c 100644 --- a/code/SoS/graveyard/SuSiE_slides.ipynb +++ b/code/SoS/graveyard/SuSiE_slides.ipynb @@ -257,4 +257,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/graveyard/association_scan_post_processing.ipynb b/code/SoS/graveyard/association_scan_post_processing.ipynb index 75551a7c7..463c57644 100644 --- a/code/SoS/graveyard/association_scan_post_processing.ipynb +++ b/code/SoS/graveyard/association_scan_post_processing.ipynb @@ -186,4 +186,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/graveyard/bam_to_bw.ipynb b/code/SoS/graveyard/bam_to_bw.ipynb index bac9dc7cf..aaffc7871 100644 --- a/code/SoS/graveyard/bam_to_bw.ipynb +++ b/code/SoS/graveyard/bam_to_bw.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "fea0558b", "metadata": { "kernel": "SoS" }, @@ -11,6 +12,7 @@ }, { "cell_type": "markdown", + "id": "c66984ff", "metadata": { "kernel": "SoS" }, @@ -22,6 +24,7 @@ }, { "cell_type": "markdown", + "id": "be1de70a", "metadata": { "kernel": "SoS" }, @@ -35,6 +38,7 @@ }, { "cell_type": "markdown", + "id": "46883fb3", "metadata": { "kernel": "SoS" }, @@ -46,6 +50,7 @@ }, { "cell_type": "markdown", + "id": "c212e841", "metadata": { "kernel": "SoS" }, @@ -57,6 +62,7 @@ }, { "cell_type": "markdown", + "id": "fbb3bab0", "metadata": { "kernel": "SoS" }, @@ -67,6 +73,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5f396792", "metadata": { "kernel": "Bash" }, @@ -79,6 +86,7 @@ }, { "cell_type": "markdown", + "id": "a27161eb", "metadata": { "kernel": "SoS" }, @@ -89,6 +97,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6b3eda61", "metadata": { "kernel": "Bash" }, @@ -99,6 +108,7 @@ }, { "cell_type": "markdown", + "id": "11cc11c8", "metadata": { "kernel": "SoS" }, @@ -108,6 +118,7 @@ }, { "cell_type": "markdown", + "id": "cce436b3", "metadata": { "kernel": "SoS" }, @@ -120,6 +131,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9ab1349a", "metadata": { "kernel": "SoS" }, @@ -143,6 +155,7 @@ { "cell_type": "code", "execution_count": null, + "id": "85bd60c3", "metadata": { "kernel": "SoS" }, @@ -186,5 +199,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/graveyard/bulk_expression_commands.ipynb b/code/SoS/graveyard/bulk_expression_commands.ipynb index c76afb7f6..3dc0177d9 100644 --- a/code/SoS/graveyard/bulk_expression_commands.ipynb +++ b/code/SoS/graveyard/bulk_expression_commands.ipynb @@ -12,8 +12,8 @@ "\n", "This document shows the use of various modules to prepare reference data, perform RNA-seq calling, expression level quantification and quality control. In particular,\n", "\n", - "1. [`reference_data.ipynb`](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/reference_data.html)\n", - "2. [`RNA_calling.ipynb`](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/RNA_calling.html)\n", + "1. [`reference_data.ipynb`](https://statfungen.github.io/xqtl-protocol/reference-data)\n", + "2. [`RNA_calling.ipynb`](https://statfungen.github.io/xqtl-protocol/rna-calling)\n", "3. [`readCount_QC.ipynb`](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/QC/readCount_QC.html)\n", "\n", "A minimal working example is available on [Google Drive](https://drive.google.com/drive/u/0/folders/11kQv7PXozsKkgeqADH-28bC_kZ-w_oHo)." diff --git a/code/SoS/graveyard/eQTL_analysis_commands.ipynb b/code/SoS/graveyard/eQTL_analysis_commands.ipynb index 1b8439d02..1a96fe7f5 100644 --- a/code/SoS/graveyard/eQTL_analysis_commands.ipynb +++ b/code/SoS/graveyard/eQTL_analysis_commands.ipynb @@ -1040,4 +1040,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/graveyard/fastenloc_susie.ipynb b/code/SoS/graveyard/fastenloc_susie.ipynb index ce995e93c..d6b1fc753 100644 --- a/code/SoS/graveyard/fastenloc_susie.ipynb +++ b/code/SoS/graveyard/fastenloc_susie.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "931e6e75", "metadata": { "kernel": "SoS", "tags": [] @@ -17,6 +18,7 @@ }, { "cell_type": "markdown", + "id": "a1c487ed", "metadata": { "kernel": "SoS" }, @@ -26,6 +28,7 @@ }, { "cell_type": "markdown", + "id": "e47e76ad", "metadata": { "kernel": "SoS" }, @@ -36,6 +39,7 @@ }, { "cell_type": "markdown", + "id": "a40be1b3", "metadata": { "kernel": "SoS" }, @@ -49,6 +53,7 @@ }, { "cell_type": "markdown", + "id": "a23e9946", "metadata": { "kernel": "SoS", "tags": [] @@ -56,7 +61,7 @@ "source": [ "## Input\n", "\n", - "1. QTL susie table\uff1a\n", + "1. QTL susie table:\n", " - This table has two columns for `molecular_trait_id` and `susie_file`: target gene and corresponding susie output rds respectively.\n", "2. GWAS susie table: \n", " - This table has two columns for `ld_block` and `susie_object_file`: LD block and corresponding susie output rds respectively.\n" @@ -64,6 +69,7 @@ }, { "cell_type": "markdown", + "id": "790f83d5", "metadata": { "jp-MarkdownHeadingCollapsed": true, "kernel": "SoS", @@ -130,6 +136,7 @@ }, { "cell_type": "markdown", + "id": "ab20cba8", "metadata": { "kernel": "SoS" }, @@ -143,6 +150,7 @@ }, { "cell_type": "markdown", + "id": "39d6c3c0", "metadata": { "kernel": "SoS" }, @@ -152,6 +160,7 @@ }, { "cell_type": "markdown", + "id": "cdbaad2a", "metadata": { "kernel": "SoS" }, @@ -181,6 +190,7 @@ }, { "cell_type": "markdown", + "id": "0ffb54e6", "metadata": { "kernel": "SoS" }, @@ -192,6 +202,7 @@ { "cell_type": "code", "execution_count": null, + "id": "affc12b5", "metadata": { "kernel": "SoS" }, @@ -207,6 +218,7 @@ { "cell_type": "code", "execution_count": null, + "id": "da930848", "metadata": { "kernel": "SoS" }, @@ -224,6 +236,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e6546a6e", "metadata": { "kernel": "SoS" }, @@ -233,6 +246,7 @@ { "cell_type": "code", "execution_count": 1, + "id": "8ad7832d", "metadata": { "kernel": "SoS" }, @@ -273,6 +287,7 @@ }, { "cell_type": "markdown", + "id": "8859b15d", "metadata": { "kernel": "SoS" }, @@ -286,6 +301,7 @@ { "cell_type": "code", "execution_count": 17, + "id": "03235165", "metadata": { "kernel": "Bash" }, @@ -309,6 +325,7 @@ { "cell_type": "code", "execution_count": null, + "id": "05322ad7", "metadata": { "kernel": "SoS" }, @@ -382,6 +399,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5249cb21", "metadata": { "kernel": "SoS" }, @@ -402,6 +420,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5739f23b", "metadata": { "kernel": "SoS" }, @@ -436,6 +455,7 @@ }, { "cell_type": "markdown", + "id": "8bf48f07", "metadata": { "kernel": "SoS" }, @@ -446,6 +466,7 @@ { "cell_type": "code", "execution_count": 19, + "id": "92237fd1", "metadata": { "kernel": "Bash" }, @@ -470,6 +491,7 @@ }, { "cell_type": "markdown", + "id": "b7be20bc", "metadata": { "kernel": "SoS" }, @@ -479,6 +501,7 @@ }, { "cell_type": "markdown", + "id": "c52f6c5d", "metadata": { "kernel": "Markdown" }, @@ -489,6 +512,7 @@ { "cell_type": "code", "execution_count": 18, + "id": "aec371db", "metadata": { "kernel": "Bash" }, @@ -511,6 +535,7 @@ }, { "cell_type": "markdown", + "id": "5c877f0c", "metadata": { "kernel": "SoS" }, @@ -520,6 +545,7 @@ }, { "cell_type": "markdown", + "id": "7e04d512", "metadata": { "kernel": "SoS" }, @@ -530,6 +556,7 @@ { "cell_type": "code", "execution_count": 2, + "id": "05ecf3d3", "metadata": { "kernel": "SoS" }, @@ -570,6 +597,7 @@ { "cell_type": "code", "execution_count": 3, + "id": "7f6849a6", "metadata": { "kernel": "SoS" }, @@ -591,6 +619,7 @@ { "cell_type": "code", "execution_count": 3, + "id": "984b2037", "metadata": { "kernel": "SoS" }, @@ -609,6 +638,7 @@ }, { "cell_type": "markdown", + "id": "0366a850", "metadata": { "kernel": "SoS" }, @@ -619,6 +649,7 @@ { "cell_type": "code", "execution_count": 23, + "id": "8d34cdaa", "metadata": { "kernel": "Bash", "tags": [] @@ -644,6 +675,7 @@ }, { "cell_type": "markdown", + "id": "3ea003a6", "metadata": { "kernel": "Bash" }, @@ -655,6 +687,7 @@ }, { "cell_type": "markdown", + "id": "64d131d5", "metadata": { "kernel": "Bash" }, @@ -665,6 +698,7 @@ { "cell_type": "code", "execution_count": 6, + "id": "d3809607", "metadata": { "kernel": "Bash" }, @@ -795,5 +829,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/graveyard/genotype_alignment.ipynb b/code/SoS/graveyard/genotype_alignment.ipynb index 453abb6cb..94f6006bb 100644 --- a/code/SoS/graveyard/genotype_alignment.ipynb +++ b/code/SoS/graveyard/genotype_alignment.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "78e495a1", "metadata": { "kernel": "SoS" }, @@ -15,6 +16,7 @@ }, { "cell_type": "markdown", + "id": "2906e4f2", "metadata": { "kernel": "SoS" }, @@ -31,6 +33,7 @@ }, { "cell_type": "markdown", + "id": "92878e0b", "metadata": { "kernel": "SoS" }, @@ -55,6 +58,7 @@ }, { "cell_type": "markdown", + "id": "989602df", "metadata": { "kernel": "SoS" }, @@ -64,6 +68,7 @@ }, { "cell_type": "markdown", + "id": "dc4a31c2", "metadata": { "kernel": "SoS" }, @@ -73,6 +78,7 @@ }, { "cell_type": "markdown", + "id": "84226ef1", "metadata": { "kernel": "SoS" }, @@ -83,6 +89,7 @@ { "cell_type": "code", "execution_count": null, + "id": "d42f90df", "metadata": { "kernel": "Bash" }, @@ -96,6 +103,7 @@ }, { "cell_type": "markdown", + "id": "1d74238b", "metadata": { "kernel": "SoS" }, @@ -106,6 +114,7 @@ { "cell_type": "code", "execution_count": null, + "id": "390c2b46", "metadata": { "kernel": "Bash" }, @@ -116,6 +125,7 @@ }, { "cell_type": "markdown", + "id": "3fd2af77", "metadata": { "kernel": "SoS" }, @@ -125,6 +135,7 @@ }, { "cell_type": "markdown", + "id": "83d8bbe7", "metadata": { "kernel": "SoS" }, @@ -137,6 +148,7 @@ { "cell_type": "code", "execution_count": null, + "id": "70af538e", "metadata": { "kernel": "SoS" }, @@ -167,6 +179,7 @@ { "cell_type": "code", "execution_count": 1, + "id": "8636a567", "metadata": { "kernel": "SoS" }, @@ -249,5 +262,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 + "nbformat_minor": 5 } diff --git a/code/SoS/graveyard/ld_reference_generation.ipynb b/code/SoS/graveyard/ld_reference_generation.ipynb index b9ec0e8a7..ca333d8fc 100644 --- a/code/SoS/graveyard/ld_reference_generation.ipynb +++ b/code/SoS/graveyard/ld_reference_generation.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "ae0b1688", "metadata": { "kernel": "SoS" }, @@ -11,6 +12,7 @@ }, { "cell_type": "markdown", + "id": "61d80829", "metadata": { "kernel": "SoS" }, @@ -26,6 +28,7 @@ }, { "cell_type": "markdown", + "id": "22cc5fbb", "metadata": { "kernel": "SoS" }, @@ -41,6 +44,7 @@ }, { "cell_type": "markdown", + "id": "ab78dbcd", "metadata": { "kernel": "SoS" }, @@ -54,6 +58,7 @@ }, { "cell_type": "markdown", + "id": "f25d7604", "metadata": { "kernel": "SoS" }, @@ -63,6 +68,7 @@ }, { "cell_type": "markdown", + "id": "d15929f0", "metadata": { "kernel": "SoS" }, @@ -72,6 +78,7 @@ }, { "cell_type": "markdown", + "id": "b47dfb19", "metadata": {}, "source": [ "**Timing:** ~10-30 min on typical compute infrastructure." @@ -80,6 +87,7 @@ { "cell_type": "code", "execution_count": null, + "id": "75a45c8b", "metadata": { "kernel": "Bash" }, @@ -95,6 +103,7 @@ { "cell_type": "code", "execution_count": null, + "id": "05768193", "metadata": { "kernel": "Bash" }, @@ -110,6 +119,7 @@ }, { "cell_type": "markdown", + "id": "f11a7568", "metadata": { "kernel": "SoS" }, @@ -120,6 +130,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9850e026", "metadata": { "kernel": "Bash" }, @@ -131,6 +142,7 @@ { "cell_type": "code", "execution_count": null, + "id": "db09cb0c", "metadata": { "kernel": "Bash" }, @@ -148,6 +160,7 @@ }, { "cell_type": "markdown", + "id": "175871dd", "metadata": { "kernel": "SoS" }, @@ -160,6 +173,7 @@ { "cell_type": "code", "execution_count": null, + "id": "c3609492", "metadata": { "kernel": "SoS" }, @@ -179,6 +193,7 @@ { "cell_type": "code", "execution_count": null, + "id": "040da1b5", "metadata": { "kernel": "SoS" }, @@ -219,6 +234,7 @@ }, { "cell_type": "markdown", + "id": "65081adc", "metadata": { "kernel": "SoS" }, @@ -233,9 +249,12 @@ }, { "cell_type": "markdown", + "id": "befb0b6f", "metadata": {}, "source": [ - "## Anticipated Results\n\nThe pipeline produces output files in the `output/` subdirectory named after the workflow step. Verify success by checking that output files exist and are non-empty. See the **Output** section above for the expected file names and formats." + "## Anticipated Results\n", + "\n", + "The pipeline produces output files in the `output/` subdirectory named after the workflow step. Verify success by checking that output files exist and are non-empty. See the **Output** section above for the expected file names and formats." ] }, { @@ -477,5 +496,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 + "nbformat_minor": 5 } diff --git a/code/SoS/graveyard/polyfun.ipynb b/code/SoS/graveyard/polyfun.ipynb index 6768be118..da3db3aed 100644 --- a/code/SoS/graveyard/polyfun.ipynb +++ b/code/SoS/graveyard/polyfun.ipynb @@ -1025,4 +1025,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/graveyard/pseudobulk_expression_QC_and_normalization.ipynb b/code/SoS/graveyard/pseudobulk_expression_QC_and_normalization.ipynb index 6020f0e42..02ee0c2fa 100644 --- a/code/SoS/graveyard/pseudobulk_expression_QC_and_normalization.ipynb +++ b/code/SoS/graveyard/pseudobulk_expression_QC_and_normalization.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "bb4641db", "metadata": { "kernel": "SoS" }, @@ -11,6 +12,7 @@ }, { "cell_type": "markdown", + "id": "433b3ae6", "metadata": { "kernel": "SoS" }, @@ -20,6 +22,7 @@ }, { "cell_type": "markdown", + "id": "26a3a3f6", "metadata": { "kernel": "SoS" }, @@ -42,6 +45,7 @@ }, { "cell_type": "markdown", + "id": "eb1d798d", "metadata": { "kernel": "SoS" }, @@ -61,6 +65,7 @@ }, { "cell_type": "markdown", + "id": "7b58f0d8", "metadata": { "kernel": "SoS" }, @@ -80,6 +85,7 @@ }, { "cell_type": "markdown", + "id": "084b4062", "metadata": { "kernel": "SoS" }, @@ -89,6 +95,7 @@ }, { "cell_type": "markdown", + "id": "b24827df", "metadata": { "kernel": "SoS" }, @@ -98,6 +105,7 @@ }, { "cell_type": "markdown", + "id": "57c002e5", "metadata": { "kernel": "SoS" }, @@ -107,6 +115,7 @@ }, { "cell_type": "markdown", + "id": "567bc63d", "metadata": { "kernel": "SoS" }, @@ -116,6 +125,7 @@ }, { "cell_type": "markdown", + "id": "380c0345", "metadata": { "kernel": "SoS" }, @@ -126,6 +136,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4d7560ba", "metadata": { "kernel": "Bash" }, @@ -139,6 +150,7 @@ }, { "cell_type": "markdown", + "id": "c7782f3b", "metadata": { "kernel": "SoS" }, @@ -148,6 +160,7 @@ }, { "cell_type": "markdown", + "id": "90fa054c", "metadata": { "kernel": "SoS" }, @@ -158,6 +171,7 @@ { "cell_type": "code", "execution_count": null, + "id": "109c4909", "metadata": { "kernel": "Bash" }, @@ -173,6 +187,7 @@ }, { "cell_type": "markdown", + "id": "3a16a434", "metadata": { "kernel": "SoS" }, @@ -183,6 +198,7 @@ { "cell_type": "code", "execution_count": 1, + "id": "663d704e", "metadata": { "kernel": "Bash" }, @@ -234,6 +250,7 @@ }, { "cell_type": "markdown", + "id": "b68bbd4e", "metadata": { "kernel": "Bash" }, @@ -244,6 +261,7 @@ { "cell_type": "code", "execution_count": 1, + "id": "f2e091f4", "metadata": { "kernel": "SoS", "tags": [] @@ -271,6 +289,7 @@ { "cell_type": "code", "execution_count": 2, + "id": "ff7a5d89", "metadata": { "kernel": "SoS" }, @@ -340,6 +359,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f8b002bd", "metadata": { "kernel": "SoS" }, @@ -417,5 +437,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/graveyard/pseudobulk_expression_aggregation_QC_norm.ipynb b/code/SoS/graveyard/pseudobulk_expression_aggregation_QC_norm.ipynb index 939be8c6b..07051440d 100644 --- a/code/SoS/graveyard/pseudobulk_expression_aggregation_QC_norm.ipynb +++ b/code/SoS/graveyard/pseudobulk_expression_aggregation_QC_norm.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "2704cb69", "metadata": { "kernel": "SoS" }, @@ -11,6 +12,7 @@ }, { "cell_type": "markdown", + "id": "e83f606e", "metadata": { "kernel": "SoS" }, @@ -30,6 +32,7 @@ }, { "cell_type": "markdown", + "id": "05ded710", "metadata": {}, "source": [ "**Timing:** ~5-10 min on typical compute infrastructure." @@ -37,6 +40,7 @@ }, { "cell_type": "markdown", + "id": "aa3a70cb", "metadata": { "kernel": "SoS" }, @@ -57,6 +61,7 @@ }, { "cell_type": "markdown", + "id": "0df21eec", "metadata": { "kernel": "SoS" }, @@ -72,6 +77,7 @@ }, { "cell_type": "markdown", + "id": "666d075a", "metadata": { "kernel": "SoS" }, @@ -83,6 +89,7 @@ }, { "cell_type": "markdown", + "id": "acd7ef45", "metadata": { "kernel": "SoS" }, @@ -95,6 +102,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f8a5ddd6", "metadata": { "kernel": "Bash" }, @@ -108,6 +116,7 @@ }, { "cell_type": "markdown", + "id": "61d3bafd", "metadata": { "kernel": "SoS" }, @@ -120,6 +129,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b450bf03", "metadata": { "kernel": "Bash" }, @@ -133,6 +143,7 @@ }, { "cell_type": "markdown", + "id": "9a314a11", "metadata": { "kernel": "SoS" }, @@ -145,6 +156,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b7c48eb7", "metadata": { "kernel": "Bash" }, @@ -158,6 +170,7 @@ }, { "cell_type": "markdown", + "id": "563e4e19", "metadata": { "kernel": "SoS" }, @@ -168,6 +181,7 @@ { "cell_type": "code", "execution_count": null, + "id": "3b52020b", "metadata": { "kernel": "Bash" }, @@ -178,6 +192,7 @@ }, { "cell_type": "markdown", + "id": "bc3ec997", "metadata": { "kernel": "SoS" }, @@ -187,14 +202,18 @@ }, { "cell_type": "markdown", + "id": "00123e2b", "metadata": {}, "source": [ - "## Anticipated Results\n\nThe pipeline produces output files in the `output/` subdirectory named after the workflow step. Verify success by checking that output files exist and are non-empty. See the **Output** section above for the expected file names and formats." + "## Anticipated Results\n", + "\n", + "The pipeline produces output files in the `output/` subdirectory named after the workflow step. Verify success by checking that output files exist and are non-empty. See the **Output** section above for the expected file names and formats." ] }, { "cell_type": "code", "execution_count": null, + "id": "5d44d50a", "metadata": { "kernel": "SoS", "vscode": { @@ -221,6 +240,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2025f81b", "metadata": { "kernel": "SoS", "vscode": { @@ -317,6 +337,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b194093a", "metadata": { "kernel": "SoS", "vscode": { @@ -416,6 +437,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a2850f7e", "metadata": { "kernel": "SoS", "vscode": { @@ -566,6 +588,7 @@ { "cell_type": "code", "execution_count": null, + "id": "28caac27", "metadata": { "kernel": "SoS", "vscode": { @@ -633,5 +656,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/graveyard/pseudobulk_mega_expression_QC_and_normalization.ipynb b/code/SoS/graveyard/pseudobulk_mega_expression_QC_and_normalization.ipynb index b4763b367..f91616e7c 100644 --- a/code/SoS/graveyard/pseudobulk_mega_expression_QC_and_normalization.ipynb +++ b/code/SoS/graveyard/pseudobulk_mega_expression_QC_and_normalization.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "01ea8d6c", "metadata": { "kernel": "SoS" }, @@ -11,6 +12,7 @@ }, { "cell_type": "markdown", + "id": "554a4ef8", "metadata": { "kernel": "SoS" }, @@ -22,6 +24,7 @@ }, { "cell_type": "markdown", + "id": "b43b9158", "metadata": {}, "source": [ "**Timing:** ~10-20 min on typical compute infrastructure." @@ -29,6 +32,7 @@ }, { "cell_type": "markdown", + "id": "f8b2f8c0", "metadata": { "kernel": "SoS" }, @@ -43,6 +47,7 @@ }, { "cell_type": "markdown", + "id": "9262660f", "metadata": { "kernel": "SoS" }, @@ -56,6 +61,7 @@ }, { "cell_type": "markdown", + "id": "b02ea62c", "metadata": { "kernel": "SoS" }, @@ -67,6 +73,7 @@ }, { "cell_type": "markdown", + "id": "d4a67f1e", "metadata": { "kernel": "SoS" }, @@ -76,6 +83,7 @@ }, { "cell_type": "markdown", + "id": "70c91683", "metadata": { "kernel": "SoS" }, @@ -86,6 +94,7 @@ { "cell_type": "code", "execution_count": null, + "id": "613cb4ae", "metadata": { "kernel": "Bash" }, @@ -99,6 +108,7 @@ }, { "cell_type": "markdown", + "id": "3236e80b", "metadata": { "kernel": "SoS" }, @@ -109,6 +119,7 @@ { "cell_type": "code", "execution_count": null, + "id": "0ce8fcc9", "metadata": { "kernel": "Bash" }, @@ -119,6 +130,7 @@ }, { "cell_type": "markdown", + "id": "d7677af9", "metadata": { "kernel": "SoS" }, @@ -129,6 +141,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4b584266", "metadata": { "kernel": "SoS", "vscode": { @@ -155,6 +168,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b4bebf22", "metadata": { "kernel": "SoS", "vscode": { @@ -361,5 +375,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/misc/advanced_analysis/1_twas_mr_colocboost.ipynb b/code/SoS/misc/advanced_analysis/1_twas_mr_colocboost.ipynb index 580b5ee14..4856b9cd1 100644 --- a/code/SoS/misc/advanced_analysis/1_twas_mr_colocboost.ipynb +++ b/code/SoS/misc/advanced_analysis/1_twas_mr_colocboost.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "cbfd9a14", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "88bd27b3", "metadata": { "kernel": "SoS" }, @@ -34,6 +36,7 @@ }, { "cell_type": "markdown", + "id": "c7993dcc", "metadata": { "kernel": "SoS" }, @@ -50,13 +53,64 @@ }, { "cell_type": "markdown", + "id": "600c0ecc", "metadata": { "kernel": "SoS" }, - "source": "## 1. TWAS\n\n### 1.1 Transcriptome-Wide Association Study (TWAS)\n\n**TWAS** is a method used to identify genes associated with complex traits by leveraging predicted gene expression levels. It combines precomputed molecular phenotype prediction weights with GWAS summary statistics to perform gene-trait association tests. \n\n\n#### 1.2 Input File Formats\n\n- GWAS Metadata (`--gwas_meta_data`) \nA TSV file specifying study ID, chromosome, path to summary statistics, and a column mapping file. \n- LD Reference Metadata (`--ld_meta_data`) \nA TSV file with chromosome, start and end positions, and paths to LD matrix files. \n- xQTL Weight Metadata (`--xqtl_meta_data`) \nMetadata file containing gene region coordinates and paths to weight files stored in RDS format. \n- Genomic Region Definitions (`--regions`) \nA BED file defining analysis regions. \n- xQTL Type Table (`--xqtl_type_table`) \nA mapping file defining the molecular phenotype types and their associated biological context.\n\n\n#### 1.3 Key Parameters\n\n- `--rsq_cutoff 0.01`: Cross-validation R² threshold used to determine whether a gene is predictable \n- `--rsq_pval_cutoff 0.05`: Cross-validation p-value threshold \n\n\n#### 1.4 Analysis Workflow \n\n1). **Weight Loading**: Load TWAS prediction weights from RDS files. \n2). **Predictable Gene Identification**: Determine which genes are reliably predictable based on cross-validation performance. \n3). **TWAS Testing**: Compute TWAS Z-scores by combining predicted expression with GWAS statistics. \n4). **Mendelian Randomization (MR)**: Perform MR analysis on candidate genes.\n\n### Input Files\n\n| File | Produced by | Used as |\n|------|-------------|---------|\n| `input/twas/protocol_example.twas.gwas_meta.tsv` | toy data prep | GWAS summary-statistics metadata |\n| `input/twas/protocol_example.twas.xqtl_meta.tsv` | toy data prep | xQTL weight metadata (gene region + RDS weights) |\n| `input/twas/protocol_example.twas.LD_blocks.chr22.bed` | toy data prep | analysis region definitions |\n| `input/twas/protocol_example.twas.data_type_table.txt` | toy data prep | xQTL molecular phenotype type table |\n| `input/LD_sketch/meta.tsv` | RSS LD sketch step | LD reference metadata |\n| `output/finemapping/susie_twas/twas_weights/*.univariate_twas_weights.rds` | fine-mapping (2_finemapping) | TWAS prediction weights |\n\nKey parameters: `--rsq_cutoff 0.01` (cross-validation R² threshold for predictable genes) and `--rsq_pval_cutoff` (cross-validation p-value threshold).\n\nThe TWAS step proceeds in four stages. First it **loads** the precomputed prediction weights from the RDS files. It then identifies **predictable genes** — those whose expression is reliably imputable based on cross-validation performance (controlled by `--rsq_cutoff` and `--rsq_pval_cutoff`). For those genes it computes **TWAS Z-scores** by combining the predicted expression with the GWAS summary statistics, and finally performs **Mendelian Randomization (MR)** on the candidate genes to assess potential causal effects of expression on the trait.\n" + "source": [ + "## 1. TWAS\n", + "\n", + "### 1.1 Transcriptome-Wide Association Study (TWAS)\n", + "\n", + "**TWAS** is a method used to identify genes associated with complex traits by leveraging predicted gene expression levels. It combines precomputed molecular phenotype prediction weights with GWAS summary statistics to perform gene-trait association tests. \n", + "\n", + "\n", + "#### 1.2 Input File Formats\n", + "\n", + "- GWAS Metadata (`--gwas_meta_data`) \n", + "A TSV file specifying study ID, chromosome, path to summary statistics, and a column mapping file. \n", + "- LD Reference Metadata (`--ld_meta_data`) \n", + "A TSV file with chromosome, start and end positions, and paths to LD matrix files. \n", + "- xQTL Weight Metadata (`--xqtl_meta_data`) \n", + "Metadata file containing gene region coordinates and paths to weight files stored in RDS format. \n", + "- Genomic Region Definitions (`--regions`) \n", + "A BED file defining analysis regions. \n", + "- xQTL Type Table (`--xqtl_type_table`) \n", + "A mapping file defining the molecular phenotype types and their associated biological context.\n", + "\n", + "\n", + "#### 1.3 Key Parameters\n", + "\n", + "- `--rsq_cutoff 0.01`: Cross-validation R² threshold used to determine whether a gene is predictable \n", + "- `--rsq_pval_cutoff 0.05`: Cross-validation p-value threshold \n", + "\n", + "\n", + "#### 1.4 Analysis Workflow \n", + "\n", + "1). **Weight Loading**: Load TWAS prediction weights from RDS files. \n", + "2). **Predictable Gene Identification**: Determine which genes are reliably predictable based on cross-validation performance. \n", + "3). **TWAS Testing**: Compute TWAS Z-scores by combining predicted expression with GWAS statistics. \n", + "4). **Mendelian Randomization (MR)**: Perform MR analysis on candidate genes.\n", + "\n", + "### Input Files\n", + "\n", + "| File | Produced by | Used as |\n", + "|------|-------------|---------|\n", + "| `input/twas/protocol_example.twas.gwas_meta.tsv` | toy data prep | GWAS summary-statistics metadata |\n", + "| `input/twas/protocol_example.twas.xqtl_meta.tsv` | toy data prep | xQTL weight metadata (gene region + RDS weights) |\n", + "| `input/twas/protocol_example.twas.LD_blocks.chr22.bed` | toy data prep | analysis region definitions |\n", + "| `input/twas/protocol_example.twas.data_type_table.txt` | toy data prep | xQTL molecular phenotype type table |\n", + "| `input/LD_sketch/meta.tsv` | RSS LD sketch step | LD reference metadata |\n", + "| `output/finemapping/susie_twas/twas_weights/*.univariate_twas_weights.rds` | fine-mapping (2_finemapping) | TWAS prediction weights |\n", + "\n", + "Key parameters: `--rsq_cutoff 0.01` (cross-validation R² threshold for predictable genes) and `--rsq_pval_cutoff` (cross-validation p-value threshold).\n", + "\n", + "The TWAS step proceeds in four stages. First it **loads** the precomputed prediction weights from the RDS files. It then identifies **predictable genes** — those whose expression is reliably imputable based on cross-validation performance (controlled by `--rsq_cutoff` and `--rsq_pval_cutoff`). For those genes it computes **TWAS Z-scores** by combining the predicted expression with the GWAS summary statistics, and finally performs **Mendelian Randomization (MR)** on the candidate genes to assess potential causal effects of expression on the trait.\n" + ] }, { "cell_type": "markdown", + "id": "72b5d56b", "metadata": { "kernel": "SoS" }, @@ -69,6 +123,7 @@ }, { "cell_type": "markdown", + "id": "81a0f0e2", "metadata": { "kernel": "SoS" }, @@ -79,6 +134,7 @@ { "cell_type": "code", "execution_count": 16, + "id": "c0f9239b", "metadata": { "kernel": "R" }, @@ -102,6 +158,7 @@ { "cell_type": "code", "execution_count": 17, + "id": "64f3a003", "metadata": { "kernel": "R" }, @@ -123,6 +180,7 @@ }, { "cell_type": "markdown", + "id": "9b97e55f", "metadata": { "kernel": "SoS" }, @@ -133,6 +191,7 @@ { "cell_type": "code", "execution_count": null, + "id": "8243c66f", "metadata": { "kernel": "SoS" }, @@ -153,6 +212,7 @@ }, { "cell_type": "markdown", + "id": "53db58ea", "metadata": { "kernel": "SoS" }, @@ -163,6 +223,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6b4e0532", "metadata": { "kernel": "SoS" }, @@ -175,6 +236,7 @@ { "cell_type": "code", "execution_count": null, + "id": "dbb81249", "metadata": { "kernel": "SoS" }, @@ -186,6 +248,7 @@ }, { "cell_type": "markdown", + "id": "2419e8d6", "metadata": { "kernel": "SoS" }, @@ -200,13 +263,74 @@ }, { "cell_type": "markdown", + "id": "ce5cc0d0", "metadata": { "kernel": "SoS" }, - "source": "## 2. xQTL–GWAS enrichment\n\nThis command executes the **`xqtl_gwas_enrichment`** workflow, which analyzes enrichment relationships between xQTL and GWAS data\n\n#### **Purpose**\n\nThe purpose of this step is to calculate enrichment parameters (a0, a1) between xQTL and GWAS data and generate prior probabilities (p1, p2, p12) for subsequent colocalization analysi\n\n#### **Method**\n\nThe workflow uses the **`xqtl_enrichment_wrapper`** function to perform enrichment analysis . The method:\n\n1. Identifies GWAS blocks with top loci using single variant regression methods\n2. Maps analysis regions to overlapping gene regions with corresponding QTL files containing top loci tables\n3. Finds contexts within xQTL metadata that include top loci results for each gene\n4. Executes enrichment analysis to generate parameters SuSiE_enloc.ipynb:102-106\n\n#### **Input Data**\n\n**Metadata files:**\n\n- **`-gwas-meta-data`**: Metadata file for GWAS fine-mapping results\n- **`-xqtl-meta-data`**: Metadata file for xQTL fine-mapping results\n- **`-context_meta`**: Meta file showing analysis names and contained contexts\n\n**Data paths:**\n\n- **`-qtl-path`** and **`-gwas_path`**: Directory paths for original fine-mapping results\n\n#### **Parameter Interpretation**\n\n**Object access parameters:**\n\n- **`-xqtl_finemapping_obj`**: Table name in xQTL RDS files to get fine-mapping results\n- **`-xqtl_varname_obj`**: Table name to get variable names\n- **`-gwas_finemapping_obj`** and **`-gwas_varname_obj`**: Corresponding parameters for GWAS data\n- **`-xqtl_region_obj`**: Table name to get region information\n\n#### **Output**\n\nGenerates enrichment analysis result files: **`*.enrichment.rds`**, containing global enrichment estimates that combine all input data for each context. The output file path format is: **`{cwd}/{name}.{context}.enrichment.rds`** \n\n#### **Relationship to Downstream Analysis**\n\nThis enrichment analysis is a prerequisite step for colocalization analysis. The generated enrichment parameters serve as prior probability inputs for the **`susie_coloc`** workflow. In colocalization analysis, the system automatically reads enrichment result files to set p1, p2, and p12 prior probabilities.\n\n#### **Notes**\n\nThis workflow is the first stage of the SuSiE-enloc framework, specifically handling enrichment analysis between fine-mapping results from different genomic regions. The workflow automatically handles region overlap identification and context matching, providing statistically sound prior probabilities for subsequent colocalization analysis.\n\n### Input Files\n\n| File | Produced by | Used as |\n|------|-------------|---------|\n| `input/susie_enloc_data/protocol_example.enloc.gwas_meta.tsv` | SuSiE-enloc demo | GWAS fine-mapping metadata |\n| `input/susie_enloc_data/protocol_example.enloc.xqtl_meta.tsv` | SuSiE-enloc demo | xQTL fine-mapping metadata |\n| `input/susie_enloc_data/protocol_example.enloc.context_meta.tsv` | SuSiE-enloc demo | context / analysis-name map |\n| `input/susie_enloc_data/*` | SuSiE-enloc demo | fine-mapping result objects |\n" + "source": [ + "## 2. xQTL–GWAS enrichment\n", + "\n", + "This command executes the **`xqtl_gwas_enrichment`** workflow, which analyzes enrichment relationships between xQTL and GWAS data\n", + "\n", + "#### **Purpose**\n", + "\n", + "The purpose of this step is to calculate enrichment parameters (a0, a1) between xQTL and GWAS data and generate prior probabilities (p1, p2, p12) for subsequent colocalization analysi\n", + "\n", + "#### **Method**\n", + "\n", + "The workflow uses the **`xqtl_enrichment_wrapper`** function to perform enrichment analysis . The method:\n", + "\n", + "1. Identifies GWAS blocks with top loci using single variant regression methods\n", + "2. Maps analysis regions to overlapping gene regions with corresponding QTL files containing top loci tables\n", + "3. Finds contexts within xQTL metadata that include top loci results for each gene\n", + "4. Executes enrichment analysis to generate parameters SuSiE_enloc.ipynb:102-106\n", + "\n", + "#### **Input Data**\n", + "\n", + "**Metadata files:**\n", + "\n", + "- **`-gwas-meta-data`**: Metadata file for GWAS fine-mapping results\n", + "- **`-xqtl-meta-data`**: Metadata file for xQTL fine-mapping results\n", + "- **`-context_meta`**: Meta file showing analysis names and contained contexts\n", + "\n", + "**Data paths:**\n", + "\n", + "- **`-qtl-path`** and **`-gwas_path`**: Directory paths for original fine-mapping results\n", + "\n", + "#### **Parameter Interpretation**\n", + "\n", + "**Object access parameters:**\n", + "\n", + "- **`-xqtl_finemapping_obj`**: Table name in xQTL RDS files to get fine-mapping results\n", + "- **`-xqtl_varname_obj`**: Table name to get variable names\n", + "- **`-gwas_finemapping_obj`** and **`-gwas_varname_obj`**: Corresponding parameters for GWAS data\n", + "- **`-xqtl_region_obj`**: Table name to get region information\n", + "\n", + "#### **Output**\n", + "\n", + "Generates enrichment analysis result files: **`*.enrichment.rds`**, containing global enrichment estimates that combine all input data for each context. The output file path format is: **`{cwd}/{name}.{context}.enrichment.rds`** \n", + "\n", + "#### **Relationship to Downstream Analysis**\n", + "\n", + "This enrichment analysis is a prerequisite step for colocalization analysis. The generated enrichment parameters serve as prior probability inputs for the **`susie_coloc`** workflow. In colocalization analysis, the system automatically reads enrichment result files to set p1, p2, and p12 prior probabilities.\n", + "\n", + "#### **Notes**\n", + "\n", + "This workflow is the first stage of the SuSiE-enloc framework, specifically handling enrichment analysis between fine-mapping results from different genomic regions. The workflow automatically handles region overlap identification and context matching, providing statistically sound prior probabilities for subsequent colocalization analysis.\n", + "\n", + "### Input Files\n", + "\n", + "| File | Produced by | Used as |\n", + "|------|-------------|---------|\n", + "| `input/susie_enloc_data/protocol_example.enloc.gwas_meta.tsv` | SuSiE-enloc demo | GWAS fine-mapping metadata |\n", + "| `input/susie_enloc_data/protocol_example.enloc.xqtl_meta.tsv` | SuSiE-enloc demo | xQTL fine-mapping metadata |\n", + "| `input/susie_enloc_data/protocol_example.enloc.context_meta.tsv` | SuSiE-enloc demo | context / analysis-name map |\n", + "| `input/susie_enloc_data/*` | SuSiE-enloc demo | fine-mapping result objects |\n" + ] }, { "cell_type": "markdown", + "id": "3261c3a0", "metadata": { "kernel": "SoS" }, @@ -217,6 +341,7 @@ { "cell_type": "code", "execution_count": null, + "id": "3a543cc5", "metadata": { "kernel": "SoS" }, @@ -238,6 +363,7 @@ }, { "cell_type": "markdown", + "id": "6ca65fbb", "metadata": { "kernel": "SoS" }, @@ -248,6 +374,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9142dbaf", "metadata": { "kernel": "SoS" }, @@ -261,6 +388,7 @@ }, { "cell_type": "markdown", + "id": "896e7c5b", "metadata": { "kernel": "SoS" }, @@ -274,13 +402,75 @@ }, { "cell_type": "markdown", + "id": "fd37598f", "metadata": { "kernel": "SoS" }, - "source": "## 3. Pairwise colocalization (SuSiE-coloc)\n\nThis command executes the **`susie_coloc`** workflow, which performs colocalization analysis between xQTL and GWAS data to identify shared causal variants.\n\n#### **Purpose**\n\nThe purpose of this step is to perform pairwise colocalization analysis between xQTL and GWAS fine-mapping results to determine whether observed associations share causal variants in overlapping genomic regions. This analysis identifies contexts with top loci results for each gene and applies **`susie_coloc`** to analyze each gene under each identified condition.\n\n#### **Method**\n\nThe workflow uses the **`coloc_wrapper`** function to perform colocalization analysis. The method:\n\n1. **Region Overlap Detection**: Identifies GWAS blocks that contain overlapping top loci variants for each gene using genomic coordinate matching\n2. **Prior Probability Setting**: Either loads enrichment-derived priors from previous analysis or uses default values when **`-skip-enrich`** is specified\n3. **Colocalization Analysis**: Applies **`coloc_wrapper`** for each xQTL-GWAS file pair with appropriate prior probabilities\n\n#### **Input Data**\n\n**Metadata files:**\n\n- **`-gwas-meta-data`**: GWAS fine-mapping results metadata\n- **`-xqtl-meta-data`**: xQTL fine-mapping results metadata with overlap information\n- **`-context_meta`**: Analysis names and context mappings\n\n**Data paths:**\n\n- **`-qtl-path`** and **`-gwas_path`**: Directory paths for original fine-mapping results\n\n#### **Parameter Interpretation**\n\n**Object access parameters:**\n\n- **`-xqtl_finemapping_obj`**: Table name in xQTL RDS files for fine-mapping results\n- **`-xqtl_varname_obj`**: Table name for variable names\n- **`-gwas_finemapping_obj`** and **`-gwas_varname_obj`**: Corresponding GWAS parameters\n- **`-xqtl_region_obj`**: Table name for region information\n\n**Analysis control parameters:**\n\n- **`-skip-enrich`**: Skips enrichment analysis and uses default prior probabilities (p1=1e-4, p2=1e-4, p12=5e-6)\n- **`-ld_meta_file_path`**: LD reference metadata for post-processing credible sets\n\n#### **Output**\n\nGenerates colocalization result files: **`*.coloc.rds`** containing colocalization statistics and posterior probabilities for each gene-context pair. When LD metadata is provided, also outputs credible set files (**`*.coloc_res`**) with variant-level results SuSiE_enloc.ipynb:851 .\n\n#### **Notes**\n\nThis workflow represents the second stage of the SuSiE-enloc framework, performing gene-by-gene colocalization analysis on regions identified to have overlapping variants. The **`--skip-enrich`** parameter allows bypassing enrichment analysis when using default priors, making the analysis faster but potentially less statistically informed. The inclusion of LD metadata enables variant-level credible set reporting for significant colocalization results.\n\n### Input Files\n\n| File | Produced by | Used as |\n|------|-------------|---------|\n| `input/susie_enloc_data/protocol_example.enloc.gwas_meta.tsv` | SuSiE-enloc demo | GWAS fine-mapping metadata |\n| `input/susie_enloc_data/protocol_example.enloc.xqtl_meta.tsv` | SuSiE-enloc demo | xQTL fine-mapping metadata (with overlap info) |\n| `input/susie_enloc_data/protocol_example.enloc.context_meta.tsv` | SuSiE-enloc demo | context / analysis-name map |\n| `input/ld_reference/protocol_example.ld_meta_file.tsv` | toy data prep | LD reference metadata for credible sets |\n| `input/susie_enloc_data/*` | SuSiE-enloc demo | fine-mapping result objects |\n" + "source": [ + "## 3. Pairwise colocalization (SuSiE-coloc)\n", + "\n", + "This command executes the **`susie_coloc`** workflow, which performs colocalization analysis between xQTL and GWAS data to identify shared causal variants.\n", + "\n", + "#### **Purpose**\n", + "\n", + "The purpose of this step is to perform pairwise colocalization analysis between xQTL and GWAS fine-mapping results to determine whether observed associations share causal variants in overlapping genomic regions. This analysis identifies contexts with top loci results for each gene and applies **`susie_coloc`** to analyze each gene under each identified condition.\n", + "\n", + "#### **Method**\n", + "\n", + "The workflow uses the **`coloc_wrapper`** function to perform colocalization analysis. The method:\n", + "\n", + "1. **Region Overlap Detection**: Identifies GWAS blocks that contain overlapping top loci variants for each gene using genomic coordinate matching\n", + "2. **Prior Probability Setting**: Either loads enrichment-derived priors from previous analysis or uses default values when **`-skip-enrich`** is specified\n", + "3. **Colocalization Analysis**: Applies **`coloc_wrapper`** for each xQTL-GWAS file pair with appropriate prior probabilities\n", + "\n", + "#### **Input Data**\n", + "\n", + "**Metadata files:**\n", + "\n", + "- **`-gwas-meta-data`**: GWAS fine-mapping results metadata\n", + "- **`-xqtl-meta-data`**: xQTL fine-mapping results metadata with overlap information\n", + "- **`-context_meta`**: Analysis names and context mappings\n", + "\n", + "**Data paths:**\n", + "\n", + "- **`-qtl-path`** and **`-gwas_path`**: Directory paths for original fine-mapping results\n", + "\n", + "#### **Parameter Interpretation**\n", + "\n", + "**Object access parameters:**\n", + "\n", + "- **`-xqtl_finemapping_obj`**: Table name in xQTL RDS files for fine-mapping results\n", + "- **`-xqtl_varname_obj`**: Table name for variable names\n", + "- **`-gwas_finemapping_obj`** and **`-gwas_varname_obj`**: Corresponding GWAS parameters\n", + "- **`-xqtl_region_obj`**: Table name for region information\n", + "\n", + "**Analysis control parameters:**\n", + "\n", + "- **`-skip-enrich`**: Skips enrichment analysis and uses default prior probabilities (p1=1e-4, p2=1e-4, p12=5e-6)\n", + "- **`-ld_meta_file_path`**: LD reference metadata for post-processing credible sets\n", + "\n", + "#### **Output**\n", + "\n", + "Generates colocalization result files: **`*.coloc.rds`** containing colocalization statistics and posterior probabilities for each gene-context pair. When LD metadata is provided, also outputs credible set files (**`*.coloc_res`**) with variant-level results SuSiE_enloc.ipynb:851 .\n", + "\n", + "#### **Notes**\n", + "\n", + "This workflow represents the second stage of the SuSiE-enloc framework, performing gene-by-gene colocalization analysis on regions identified to have overlapping variants. The **`--skip-enrich`** parameter allows bypassing enrichment analysis when using default priors, making the analysis faster but potentially less statistically informed. The inclusion of LD metadata enables variant-level credible set reporting for significant colocalization results.\n", + "\n", + "### Input Files\n", + "\n", + "| File | Produced by | Used as |\n", + "|------|-------------|---------|\n", + "| `input/susie_enloc_data/protocol_example.enloc.gwas_meta.tsv` | SuSiE-enloc demo | GWAS fine-mapping metadata |\n", + "| `input/susie_enloc_data/protocol_example.enloc.xqtl_meta.tsv` | SuSiE-enloc demo | xQTL fine-mapping metadata (with overlap info) |\n", + "| `input/susie_enloc_data/protocol_example.enloc.context_meta.tsv` | SuSiE-enloc demo | context / analysis-name map |\n", + "| `input/ld_reference/protocol_example.ld_meta_file.tsv` | toy data prep | LD reference metadata for credible sets |\n", + "| `input/susie_enloc_data/*` | SuSiE-enloc demo | fine-mapping result objects |\n" + ] }, { "cell_type": "markdown", + "id": "7e8287d3", "metadata": { "kernel": "SoS" }, @@ -291,6 +481,7 @@ { "cell_type": "code", "execution_count": null, + "id": "acf57b68", "metadata": { "kernel": "SoS" }, @@ -314,6 +505,7 @@ }, { "cell_type": "markdown", + "id": "27554521", "metadata": { "kernel": "SoS" }, @@ -324,6 +516,7 @@ { "cell_type": "code", "execution_count": null, + "id": "17be8b68", "metadata": { "kernel": "SoS" }, @@ -337,6 +530,7 @@ }, { "cell_type": "markdown", + "id": "d8da2c10", "metadata": { "kernel": "SoS" }, @@ -355,6 +549,7 @@ }, { "cell_type": "markdown", + "id": "d8ecf043", "metadata": { "kernel": "SoS" }, @@ -465,13 +660,80 @@ }, { "cell_type": "markdown", + "id": "4616c501", "metadata": { "kernel": "SoS" }, - "source": "## 4. ColocBoost\n\n**ColocBoost** is a multi-trait colocalization analysis tool designed to identify shared genetic variants influencing multiple molecular traits.\n\n\n### Core Features\n\n- Performs colocalization analysis within genomic regions accounting for **multiple causal variants**\n- Scales to the analysis of **hundreds of traits**\n- Supports inclusion or exclusion of **GWAS summary statistics**\n- Requires **individual-level xQTL data from the same cohort**\n\n### Analysis Modes\n\nColocBoost supports three major analysis modes:\n\n- xQTL-only Analysis (`--xqtl-coloc`): \n Performs colocalization only among molecular traits.\n\n- Joint GWAS Analysis (`--joint-gwas`): \n Integrates all GWAS traits in a combined colocalization analysis.\n\n- Separate GWAS Analysis (`--separate-gwas`): \n Performs colocalization analysis separately for each GWAS trait.\n\n### `--separate-gwas` vs `--no-separate-gwas`\n\n#### `--separate-gwas` (Enabled by default)\n\nWhen this flag is active, ColocBoost performs **individual colocalization analyses** for each GWAS trait.\n\n- Generates **one output file per GWAS study**\n- Output format: `{base_filename}.cb_xqtl_{study_name}.rds`\n- Supports **trait-specific** analysis and result inspection\n\n#### `--no-separate-gwas`\n\nWhen this flag is used, separate GWAS analyses are **not performed**.\n\n- Skips individual trait analysis\n- Reduces computational burden and the number of output files\n- Typically used **in combination with** `--joint-gwas`\n\n\n### Practical Examples\n\n#### xQTL-only Analysis\n\n- Use `--no-separate-gwas` to focus solely on **colocalization between molecular traits**\n\n#### GWAS-xQTL Joint Analysis\n\n- Use `--separate-gwas` to perform **trait-specific** colocalization for each GWAS dataset\n\n### Input Files\n\n| File | Produced by | Used as |\n|------|-------------|---------|\n| `output/plink/protocol_example.genotype.merged.plink_qc.bed` | genotype preprocessing | merged genotypes |\n| `output/phenotype/.../bulk_rnaseq.phenotype_by_chrom_files.region_list.txt` | phenotype preprocessing | phenotype region list |\n| `output/covariate/protocol_example...Marchenko_PC.gz` | covariate preprocessing | covariates |\n| `input/reference_data/TAD/TADB_enhanced_cis.bed` | reference data | cis association windows |\n| `input/LD_sketch/meta.tsv` | RSS LD sketch step | LD reference metadata (joint mode) |\n| `input/colocboost/gwas_meta.txt` | toy data prep | GWAS metadata, tab-separated (joint mode) |\n" + "source": [ + "## 4. ColocBoost\n", + "\n", + "**ColocBoost** is a multi-trait colocalization analysis tool designed to identify shared genetic variants influencing multiple molecular traits.\n", + "\n", + "\n", + "### Core Features\n", + "\n", + "- Performs colocalization analysis within genomic regions accounting for **multiple causal variants**\n", + "- Scales to the analysis of **hundreds of traits**\n", + "- Supports inclusion or exclusion of **GWAS summary statistics**\n", + "- Requires **individual-level xQTL data from the same cohort**\n", + "\n", + "### Analysis Modes\n", + "\n", + "ColocBoost supports three major analysis modes:\n", + "\n", + "- xQTL-only Analysis (`--xqtl-coloc`): \n", + " Performs colocalization only among molecular traits.\n", + "\n", + "- Joint GWAS Analysis (`--joint-gwas`): \n", + " Integrates all GWAS traits in a combined colocalization analysis.\n", + "\n", + "- Separate GWAS Analysis (`--separate-gwas`): \n", + " Performs colocalization analysis separately for each GWAS trait.\n", + "\n", + "### `--separate-gwas` vs `--no-separate-gwas`\n", + "\n", + "#### `--separate-gwas` (Enabled by default)\n", + "\n", + "When this flag is active, ColocBoost performs **individual colocalization analyses** for each GWAS trait.\n", + "\n", + "- Generates **one output file per GWAS study**\n", + "- Output format: `{base_filename}.cb_xqtl_{study_name}.rds`\n", + "- Supports **trait-specific** analysis and result inspection\n", + "\n", + "#### `--no-separate-gwas`\n", + "\n", + "When this flag is used, separate GWAS analyses are **not performed**.\n", + "\n", + "- Skips individual trait analysis\n", + "- Reduces computational burden and the number of output files\n", + "- Typically used **in combination with** `--joint-gwas`\n", + "\n", + "\n", + "### Practical Examples\n", + "\n", + "#### xQTL-only Analysis\n", + "\n", + "- Use `--no-separate-gwas` to focus solely on **colocalization between molecular traits**\n", + "\n", + "#### GWAS-xQTL Joint Analysis\n", + "\n", + "- Use `--separate-gwas` to perform **trait-specific** colocalization for each GWAS dataset\n", + "\n", + "### Input Files\n", + "\n", + "| File | Produced by | Used as |\n", + "|------|-------------|---------|\n", + "| `output/plink/protocol_example.genotype.merged.plink_qc.bed` | genotype preprocessing | merged genotypes |\n", + "| `output/phenotype/.../bulk_rnaseq.phenotype_by_chrom_files.region_list.txt` | phenotype preprocessing | phenotype region list |\n", + "| `output/covariate/protocol_example...Marchenko_PC.gz` | covariate preprocessing | covariates |\n", + "| `input/reference_data/TAD/TADB_enhanced_cis.bed` | reference data | cis association windows |\n", + "| `input/LD_sketch/meta.tsv` | RSS LD sketch step | LD reference metadata (joint mode) |\n", + "| `input/colocboost/gwas_meta.txt` | toy data prep | GWAS metadata, tab-separated (joint mode) |\n" + ] }, { "cell_type": "markdown", + "id": "79a784e9", "metadata": { "kernel": "SoS" }, @@ -482,6 +744,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9b2f0946", "metadata": { "kernel": "SoS" }, @@ -501,6 +764,7 @@ }, { "cell_type": "markdown", + "id": "e44dfd69", "metadata": { "kernel": "SoS" }, @@ -511,6 +775,7 @@ { "cell_type": "code", "execution_count": null, + "id": "32783e03", "metadata": { "kernel": "SoS" }, @@ -532,6 +797,7 @@ }, { "cell_type": "markdown", + "id": "9d2b7853", "metadata": { "kernel": "SoS" }, @@ -542,6 +808,7 @@ { "cell_type": "code", "execution_count": null, + "id": "cfa9726b", "metadata": { "kernel": "SoS" }, @@ -555,6 +822,7 @@ }, { "cell_type": "markdown", + "id": "92b5fb0e", "metadata": { "kernel": "SoS" }, @@ -571,15 +839,96 @@ }, { "cell_type": "markdown", + "id": "2ce2252d", "metadata": { "kernel": "SoS" }, "source": [ - "### Colocboost Output File Structure\n\nEach `.rds` holds the `colocboost` result for one gene: a named list of class `\"colocboost\"` whose top-level element corresponds to that gene (e.g. `ENSG00000049246`). In joint xQTL–GWAS mode (`--separate-gwas`) one such file is written per GWAS trait, named `{base_filename}.cb_xqtl_{study_name}.rds`. The internal structure is organized into the components below.\n\n#### Top-Level Components\n\n| Field | Type | Description |\n| --- | --- | --- |\n| `cos_summary` | NULL or list | Summary of colocalized signals (empty if none) |\n| `vcp` | NULL or list | Variant colocalization probabilities |\n| `cos_details` | NULL or list | Details for colocalized signals |\n| `data_info` | list | Information on traits, SNPs, z-scores, and coefficients |\n| `model_info` | list | Model convergence and log-likelihood tracking |\n| `ucos_details` | list | Results for **uncolocalized signals** (unique to colocboost) |\n| `region_info` | list | Genomic coordinates and extended window used in analysis |\n| `computing_time` | list | Timing for loading, QC, and model fitting steps |\n\n---\n\n#### data_info\n\n| Field | Description |\n| --- | --- |\n| `n_outcomes` | Number of traits analyzed |\n| `n_variables` | Number of SNPs analyzed |\n| `outcome_info` | Data frame of each trait’s name, sample size, is_sumstats, and is_focal |\n| `variables` | Character vector of SNP IDs in `chr:pos:ref:alt` format |\n| `coef`, `z` | Trait-wise regression coefficients and z-scores (lists with length = number of traits) |\n\n---\n\n#### model_info\n\nTracks model convergence and optimization metrics for each trait:\n\n| Field | Description |\n| --- | --- |\n| `model_coveraged` | Whether the joint model converged |\n| `outcome_model_coveraged` | Per-trait convergence status |\n| `n_updates` | Total number of updates |\n| `outcome_n_updates` | Number of updates per trait |\n| `profile_loglik` | Log-likelihood per iteration (joint) |\n| `outcome_profile_loglik` | Log-likelihood per trait |\n| `outcome_proximity_obj` | Proximity statistics per trait per iteration |\n| `outcome_coupled_best_update_obj` | Coupled update quality per trait |\n| `jk_star` | Jackknife matrix (optional) |\n\n---\n\n#### ucos_details\n\nDescribes **uncolocalized effect sets** when no coloc is detected:\n\n| Field | Description |\n| --- | --- |\n| `ucos_index` | Indices of SNPs in each uncolocalized set |\n| `ucos_variables` | SNP IDs in each set |\n| `ucos_outcomes` | Trait name associated with each set (e.g., `\"Wightman\"`) |\n| `ucos_weight` | Per-SNP weights in each set |\n| `ucos_top_variables` | Top SNP per set |\n| `ucos_purity` | Pairwise correlation (min, max, median) across sets |\n| `ucos_outcomes_delta` | Delta value (importance) per set |\n\n> Interpretation:\n> \n> - Large number of `ucos` suggests GWAS has signals independent of molecular QTLs.\n> - `ucos_delta` indicates signal strength; higher = more trait variance explained.\n\n---\n\n#### region_info\n\n| Field | Description |\n| --- | --- |\n| `region_coord` | Original gene start/end coordinates |\n| `grange` | Extended region used for coloc (typically ±1.5Mb) |\n| `region_name` | Gene or region ID (can repeat if multiple traits mapped) |\n\n#### computing_time\n---" + "### Colocboost Output File Structure\n", + "\n", + "Each `.rds` holds the `colocboost` result for one gene: a named list of class `\"colocboost\"` whose top-level element corresponds to that gene (e.g. `ENSG00000049246`). In joint xQTL–GWAS mode (`--separate-gwas`) one such file is written per GWAS trait, named `{base_filename}.cb_xqtl_{study_name}.rds`. The internal structure is organized into the components below.\n", + "\n", + "#### Top-Level Components\n", + "\n", + "| Field | Type | Description |\n", + "| --- | --- | --- |\n", + "| `cos_summary` | NULL or list | Summary of colocalized signals (empty if none) |\n", + "| `vcp` | NULL or list | Variant colocalization probabilities |\n", + "| `cos_details` | NULL or list | Details for colocalized signals |\n", + "| `data_info` | list | Information on traits, SNPs, z-scores, and coefficients |\n", + "| `model_info` | list | Model convergence and log-likelihood tracking |\n", + "| `ucos_details` | list | Results for **uncolocalized signals** (unique to colocboost) |\n", + "| `region_info` | list | Genomic coordinates and extended window used in analysis |\n", + "| `computing_time` | list | Timing for loading, QC, and model fitting steps |\n", + "\n", + "---\n", + "\n", + "#### data_info\n", + "\n", + "| Field | Description |\n", + "| --- | --- |\n", + "| `n_outcomes` | Number of traits analyzed |\n", + "| `n_variables` | Number of SNPs analyzed |\n", + "| `outcome_info` | Data frame of each trait’s name, sample size, is_sumstats, and is_focal |\n", + "| `variables` | Character vector of SNP IDs in `chr:pos:ref:alt` format |\n", + "| `coef`, `z` | Trait-wise regression coefficients and z-scores (lists with length = number of traits) |\n", + "\n", + "---\n", + "\n", + "#### model_info\n", + "\n", + "Tracks model convergence and optimization metrics for each trait:\n", + "\n", + "| Field | Description |\n", + "| --- | --- |\n", + "| `model_coveraged` | Whether the joint model converged |\n", + "| `outcome_model_coveraged` | Per-trait convergence status |\n", + "| `n_updates` | Total number of updates |\n", + "| `outcome_n_updates` | Number of updates per trait |\n", + "| `profile_loglik` | Log-likelihood per iteration (joint) |\n", + "| `outcome_profile_loglik` | Log-likelihood per trait |\n", + "| `outcome_proximity_obj` | Proximity statistics per trait per iteration |\n", + "| `outcome_coupled_best_update_obj` | Coupled update quality per trait |\n", + "| `jk_star` | Jackknife matrix (optional) |\n", + "\n", + "---\n", + "\n", + "#### ucos_details\n", + "\n", + "Describes **uncolocalized effect sets** when no coloc is detected:\n", + "\n", + "| Field | Description |\n", + "| --- | --- |\n", + "| `ucos_index` | Indices of SNPs in each uncolocalized set |\n", + "| `ucos_variables` | SNP IDs in each set |\n", + "| `ucos_outcomes` | Trait name associated with each set (e.g., `\"Wightman\"`) |\n", + "| `ucos_weight` | Per-SNP weights in each set |\n", + "| `ucos_top_variables` | Top SNP per set |\n", + "| `ucos_purity` | Pairwise correlation (min, max, median) across sets |\n", + "| `ucos_outcomes_delta` | Delta value (importance) per set |\n", + "\n", + "> Interpretation:\n", + "> \n", + "> - Large number of `ucos` suggests GWAS has signals independent of molecular QTLs.\n", + "> - `ucos_delta` indicates signal strength; higher = more trait variance explained.\n", + "\n", + "---\n", + "\n", + "#### region_info\n", + "\n", + "| Field | Description |\n", + "| --- | --- |\n", + "| `region_coord` | Original gene start/end coordinates |\n", + "| `grange` | Extended region used for coloc (typically ±1.5Mb) |\n", + "| `region_name` | Gene or region ID (can repeat if multiple traits mapped) |\n", + "\n", + "#### computing_time\n", + "---" ] }, { "cell_type": "markdown", + "id": "bbf898ee", "metadata": { "kernel": "SoS" }, @@ -681,5 +1030,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/misc/advanced_analysis/2_enrichment.ipynb b/code/SoS/misc/advanced_analysis/2_enrichment.ipynb index 543fddb56..2afb73d55 100644 --- a/code/SoS/misc/advanced_analysis/2_enrichment.ipynb +++ b/code/SoS/misc/advanced_analysis/2_enrichment.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "0fe4c82d", "metadata": {}, "source": [ "# Enrichment Analysis\n", @@ -11,6 +12,7 @@ }, { "cell_type": "markdown", + "id": "dfa6da20", "metadata": {}, "source": [ "## 1. EOO — Enrichment Of Overlap (block jackknife)\n", @@ -20,6 +22,7 @@ }, { "cell_type": "markdown", + "id": "376538d4", "metadata": {}, "source": [ "### Input Files\n", @@ -32,6 +35,7 @@ }, { "cell_type": "markdown", + "id": "8fd978be", "metadata": {}, "source": [ "### **Step 1.** Run the EOO enrichment pipeline\n" @@ -40,6 +44,7 @@ { "cell_type": "code", "execution_count": null, + "id": "8d8633e6", "metadata": { "vscode": { "languageId": "r" @@ -56,6 +61,7 @@ }, { "cell_type": "markdown", + "id": "924c8ddd", "metadata": {}, "source": [ "### Inspect the EOO results\n" @@ -64,6 +70,7 @@ { "cell_type": "code", "execution_count": null, + "id": "432a866f", "metadata": { "vscode": { "languageId": "r" @@ -78,6 +85,7 @@ }, { "cell_type": "markdown", + "id": "83dd8d83", "metadata": {}, "source": [ "### Output\n", @@ -87,6 +95,7 @@ }, { "cell_type": "markdown", + "id": "9e0a0a0c", "metadata": {}, "source": [ "## 2. Pathway enrichment (KEGG / clusterProfiler)\n", @@ -96,6 +105,7 @@ }, { "cell_type": "markdown", + "id": "87a19ae0", "metadata": {}, "source": [ "### Input Files\n", @@ -107,6 +117,7 @@ }, { "cell_type": "markdown", + "id": "a6d8a6d9", "metadata": {}, "source": [ "### **Step 1.** Run the pathway (GSEA/KEGG) pipeline\n" @@ -115,6 +126,7 @@ { "cell_type": "code", "execution_count": null, + "id": "97e632e9", "metadata": { "vscode": { "languageId": "r" @@ -130,6 +142,7 @@ }, { "cell_type": "markdown", + "id": "255ac48f", "metadata": {}, "source": [ "### Inspect the pathway results\n" @@ -138,6 +151,7 @@ { "cell_type": "code", "execution_count": null, + "id": "881ac7ff", "metadata": { "vscode": { "languageId": "r" @@ -152,6 +166,7 @@ }, { "cell_type": "markdown", + "id": "43db5b48", "metadata": {}, "source": [ "### Output\n", @@ -161,6 +176,7 @@ }, { "cell_type": "markdown", + "id": "dfe4cec4", "metadata": {}, "source": [ "## 3. sLDSC enrichment (PolyFun)\n", @@ -170,6 +186,7 @@ }, { "cell_type": "markdown", + "id": "8ea773e6", "metadata": {}, "source": [ "### Input Files\n", @@ -186,6 +203,7 @@ }, { "cell_type": "markdown", + "id": "aa621099", "metadata": {}, "source": [ "### **Step 1.** Make annotation files and calculate LD-scores\n" @@ -194,6 +212,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f8dfb986", "metadata": { "vscode": { "languageId": "r" @@ -217,6 +236,7 @@ }, { "cell_type": "markdown", + "id": "3b08a3e0", "metadata": {}, "source": [ "### **Step 2.** LDSC regression: calculate heritability\n" @@ -225,16 +245,34 @@ { "cell_type": "code", "execution_count": null, + "id": "9897c54a", "metadata": { "vscode": { "languageId": "r" } }, "outputs": [], - "source": "sos run pipeline/sldsc_enrichment.ipynb get_heritability \\\n --maf_cutoff 0 \\\n --target_anno_dirs output/polyfun/test_colocboost_single_1 \\\n --sumstat_dir data/polyfun/example_data \\\n --baseline_ld_dir data/polyfun/example_data \\\n --python_exec python \\\n --polyfun_path data/github/polyfun \\\n --weights_dir data/polyfun/example_data \\\n --plink_name reference. \\\n --baseline_name annotations. \\\n --weight_name weights. \\\n --annotation_name test_colocboost \\\n --cwd output/polyfun/test_colocboost/sumstats/ \\\n --all_traits_file data/polyfun/input/sumstats_test_all.txt \\\n -s build -j 2\n" + "source": [ + "sos run pipeline/sldsc_enrichment.ipynb get_heritability \\\n", + " --maf_cutoff 0 \\\n", + " --target_anno_dirs output/polyfun/test_colocboost_single_1 \\\n", + " --sumstat_dir data/polyfun/example_data \\\n", + " --baseline_ld_dir data/polyfun/example_data \\\n", + " --python_exec python \\\n", + " --polyfun_path data/github/polyfun \\\n", + " --weights_dir data/polyfun/example_data \\\n", + " --plink_name reference. \\\n", + " --baseline_name annotations. \\\n", + " --weight_name weights. \\\n", + " --annotation_name test_colocboost \\\n", + " --cwd output/polyfun/test_colocboost/sumstats/ \\\n", + " --all_traits_file data/polyfun/input/sumstats_test_all.txt \\\n", + " -s build -j 2\n" + ] }, { "cell_type": "markdown", + "id": "9c948e75", "metadata": {}, "source": [ "### **Step 3.** Summarise per trait and meta-analyse across GWAS groups\n" @@ -243,16 +281,30 @@ { "cell_type": "code", "execution_count": null, + "id": "ebccd89d", "metadata": { "vscode": { "languageId": "r" } }, "outputs": [], - "source": "sos run pipeline/sldsc_enrichment.ipynb postprocess \\\n --traits_file data/polyfun/input/sumstats_test_all.txt \\\n --heritability_cwd output/polyfun/test_colocboost/sumstats \\\n --target_categories ANNOT_0 \\\n --target_categories_label test_colocboost_annotation \\\n --target_anno_dir output/polyfun/test_colocboost_single_1 \\\n --annotation_name test_colocboost \\\n --python_exec python \\\n --polyfun_path data/github/polyfun \\\n --maf_cutoff 0 \\\n --cwd output/polyfun/test_colocboost/postprocess -j 4" + "source": [ + "sos run pipeline/sldsc_enrichment.ipynb postprocess \\\n", + " --traits_file data/polyfun/input/sumstats_test_all.txt \\\n", + " --heritability_cwd output/polyfun/test_colocboost/sumstats \\\n", + " --target_categories ANNOT_0 \\\n", + " --target_categories_label test_colocboost_annotation \\\n", + " --target_anno_dir output/polyfun/test_colocboost_single_1 \\\n", + " --annotation_name test_colocboost \\\n", + " --python_exec python \\\n", + " --polyfun_path data/github/polyfun \\\n", + " --maf_cutoff 0 \\\n", + " --cwd output/polyfun/test_colocboost/postprocess -j 4" + ] }, { "cell_type": "markdown", + "id": "c63d94f4", "metadata": {}, "source": [ "### Inspect per-GWAS sLDSC results\n" @@ -261,6 +313,7 @@ { "cell_type": "code", "execution_count": null, + "id": "61907c74", "metadata": { "vscode": { "languageId": "r" @@ -275,6 +328,7 @@ }, { "cell_type": "markdown", + "id": "5daa95b8", "metadata": {}, "source": [ "### Inspect the meta-analysis results\n" @@ -283,6 +337,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4bf0771d", "metadata": { "vscode": { "languageId": "r" @@ -297,6 +352,7 @@ }, { "cell_type": "markdown", + "id": "5dd9dbda", "metadata": {}, "source": [ "### Output\n", @@ -321,5 +377,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/misc/build_container.ipynb b/code/SoS/misc/build_container.ipynb index 4b25ed248..bfc7448b3 100644 --- a/code/SoS/misc/build_container.ipynb +++ b/code/SoS/misc/build_container.ipynb @@ -3,6 +3,7 @@ { "cell_type": "code", "execution_count": null, + "id": "3821b056", "metadata": { "kernel": "SoS", "tags": [] @@ -17,6 +18,7 @@ { "cell_type": "code", "execution_count": null, + "id": "8d3098e7", "metadata": { "kernel": "SoS" }, @@ -30,6 +32,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4c2b9da8", "metadata": { "kernel": "SoS", "tags": [] @@ -46,6 +49,7 @@ { "cell_type": "code", "execution_count": null, + "id": "0fdf266f", "metadata": { "kernel": "SoS" }, @@ -95,6 +99,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2b84a939", "metadata": { "kernel": "SoS" }, @@ -120,6 +125,7 @@ { "cell_type": "code", "execution_count": null, + "id": "46f0c9e6", "metadata": { "kernel": "SoS" }, @@ -173,5 +179,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 + "nbformat_minor": 5 } diff --git a/code/SoS/misc/data_preprocessing/1_phenotype_preprocessing.ipynb b/code/SoS/misc/data_preprocessing/1_phenotype_preprocessing.ipynb index 1a64ddabc..38330ce17 100644 --- a/code/SoS/misc/data_preprocessing/1_phenotype_preprocessing.ipynb +++ b/code/SoS/misc/data_preprocessing/1_phenotype_preprocessing.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "c38b8cfc", "metadata": { "kernel": "SoS" }, @@ -23,6 +24,7 @@ }, { "cell_type": "markdown", + "id": "11c41e17", "metadata": { "kernel": "SoS" }, @@ -38,6 +40,7 @@ }, { "cell_type": "markdown", + "id": "16421b39", "metadata": { "kernel": "SoS" }, @@ -52,6 +55,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6ff864eb", "metadata": { "kernel": "SoS" }, @@ -66,6 +70,7 @@ }, { "cell_type": "markdown", + "id": "982d5e02", "metadata": { "kernel": "SoS" }, @@ -76,6 +81,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f0857b1e", "metadata": { "kernel": "SoS" }, @@ -86,6 +92,7 @@ }, { "cell_type": "markdown", + "id": "b4cd1b66", "metadata": { "kernel": "SoS" }, @@ -106,6 +113,7 @@ }, { "cell_type": "markdown", + "id": "0217b101", "metadata": { "kernel": "SoS" }, @@ -136,6 +144,7 @@ { "cell_type": "code", "execution_count": null, + "id": "34d59022", "metadata": { "kernel": "SoS" }, @@ -150,6 +159,7 @@ }, { "cell_type": "markdown", + "id": "4d722dd9", "metadata": { "kernel": "SoS" }, @@ -162,6 +172,7 @@ { "cell_type": "code", "execution_count": null, + "id": "1e2cbe6b", "metadata": { "kernel": "SoS" }, @@ -176,6 +187,7 @@ }, { "cell_type": "markdown", + "id": "2e254bf3", "metadata": { "kernel": "SoS" }, @@ -191,6 +203,7 @@ }, { "cell_type": "markdown", + "id": "0b4c8fb2", "metadata": { "kernel": "SoS" }, @@ -228,5 +241,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/misc/data_preprocessing/2_genotype_preprocessing.ipynb b/code/SoS/misc/data_preprocessing/2_genotype_preprocessing.ipynb index 5a36a5004..4a3977867 100644 --- a/code/SoS/misc/data_preprocessing/2_genotype_preprocessing.ipynb +++ b/code/SoS/misc/data_preprocessing/2_genotype_preprocessing.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "95fe253e", "metadata": { "kernel": "SoS" }, @@ -23,6 +24,7 @@ }, { "cell_type": "markdown", + "id": "1b12fd57", "metadata": { "kernel": "SoS" }, @@ -39,6 +41,7 @@ }, { "cell_type": "markdown", + "id": "9db2333d", "metadata": { "kernel": "SoS" }, @@ -53,6 +56,7 @@ { "cell_type": "code", "execution_count": null, + "id": "54fd9809", "metadata": { "kernel": "Bash" }, @@ -64,6 +68,7 @@ }, { "cell_type": "markdown", + "id": "c92cb4d5", "metadata": { "kernel": "Bash" }, @@ -85,6 +90,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9867417e", "metadata": { "kernel": "Bash" }, @@ -96,6 +102,7 @@ }, { "cell_type": "markdown", + "id": "fef49a26", "metadata": { "kernel": "Bash" }, @@ -116,6 +123,7 @@ }, { "cell_type": "markdown", + "id": "fdb17988", "metadata": { "kernel": "SoS" }, @@ -134,6 +142,7 @@ { "cell_type": "code", "execution_count": null, + "id": "eee52370", "metadata": { "kernel": "Bash" }, @@ -149,6 +158,7 @@ }, { "cell_type": "markdown", + "id": "3c036331", "metadata": { "kernel": "SoS" }, @@ -161,6 +171,7 @@ { "cell_type": "code", "execution_count": null, + "id": "439ad7fc", "metadata": { "kernel": "Bash" }, @@ -178,6 +189,7 @@ }, { "cell_type": "markdown", + "id": "722cb01a", "metadata": { "kernel": "SoS" }, @@ -202,6 +214,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9c15be5f", "metadata": { "kernel": "Bash" }, @@ -218,6 +231,7 @@ }, { "cell_type": "markdown", + "id": "e17ad2d9", "metadata": { "kernel": "SoS" }, @@ -230,6 +244,7 @@ { "cell_type": "code", "execution_count": null, + "id": "63dea193", "metadata": { "kernel": "Bash" }, @@ -243,6 +258,7 @@ }, { "cell_type": "markdown", + "id": "450747d1", "metadata": { "kernel": "SoS" }, @@ -260,6 +276,7 @@ }, { "cell_type": "markdown", + "id": "910e1e1a", "metadata": { "kernel": "SoS" }, @@ -303,5 +320,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/misc/data_preprocessing/3_genotype_pca.ipynb b/code/SoS/misc/data_preprocessing/3_genotype_pca.ipynb index 7f3f1d015..d693d1478 100644 --- a/code/SoS/misc/data_preprocessing/3_genotype_pca.ipynb +++ b/code/SoS/misc/data_preprocessing/3_genotype_pca.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "06809d62", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "dd7ec507", "metadata": { "kernel": "SoS" }, @@ -28,6 +30,7 @@ }, { "cell_type": "markdown", + "id": "f8e6904a", "metadata": { "kernel": "SoS" }, @@ -42,6 +45,7 @@ }, { "cell_type": "markdown", + "id": "64134b49", "metadata": { "kernel": "SoS" }, @@ -51,6 +55,7 @@ }, { "cell_type": "markdown", + "id": "4b83d55b", "metadata": { "kernel": "SoS" }, @@ -63,6 +68,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e4d27952", "metadata": { "kernel": "SoS" }, @@ -76,6 +82,7 @@ }, { "cell_type": "markdown", + "id": "98acc4fc", "metadata": { "kernel": "SoS" }, @@ -90,6 +97,7 @@ { "cell_type": "code", "execution_count": null, + "id": "845adca6", "metadata": { "kernel": "SoS" }, @@ -104,6 +112,7 @@ }, { "cell_type": "markdown", + "id": "4d82cfce", "metadata": { "kernel": "SoS" }, @@ -118,6 +127,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e9281ebf", "metadata": { "kernel": "SoS" }, @@ -132,6 +142,7 @@ }, { "cell_type": "markdown", + "id": "6d0edb2b", "metadata": { "kernel": "SoS" }, @@ -144,6 +155,7 @@ { "cell_type": "code", "execution_count": null, + "id": "643f1071", "metadata": { "kernel": "SoS" }, @@ -156,6 +168,7 @@ }, { "cell_type": "markdown", + "id": "fe0eacc8", "metadata": { "kernel": "SoS" }, @@ -168,6 +181,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a556c638", "metadata": { "kernel": "SoS" }, @@ -183,6 +197,7 @@ }, { "cell_type": "markdown", + "id": "b049076c", "metadata": { "kernel": "SoS" }, @@ -195,6 +210,7 @@ { "cell_type": "code", "execution_count": null, + "id": "d3ccc197", "metadata": { "kernel": "SoS" }, @@ -208,6 +224,7 @@ }, { "cell_type": "markdown", + "id": "4978afaf", "metadata": { "kernel": "SoS" }, @@ -218,6 +235,7 @@ }, { "cell_type": "markdown", + "id": "a5f021fd", "metadata": { "kernel": "SoS" }, @@ -237,6 +255,7 @@ }, { "cell_type": "markdown", + "id": "34f9c060", "metadata": { "kernel": "SoS" }, @@ -275,5 +294,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/misc/data_preprocessing/4_covariates_preprocessing.ipynb b/code/SoS/misc/data_preprocessing/4_covariates_preprocessing.ipynb index 68cc83124..3b9a05955 100644 --- a/code/SoS/misc/data_preprocessing/4_covariates_preprocessing.ipynb +++ b/code/SoS/misc/data_preprocessing/4_covariates_preprocessing.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "640e5253", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "7a9fb38c", "metadata": { "kernel": "SoS" }, @@ -30,6 +32,7 @@ }, { "cell_type": "markdown", + "id": "d8de4953", "metadata": { "kernel": "SoS" }, @@ -46,6 +49,7 @@ }, { "cell_type": "markdown", + "id": "6132cc15", "metadata": { "kernel": "SoS" }, @@ -58,6 +62,7 @@ { "cell_type": "code", "execution_count": null, + "id": "7ca2d9ea", "metadata": { "kernel": "SoS" }, @@ -73,6 +78,7 @@ }, { "cell_type": "markdown", + "id": "3cbf6840", "metadata": { "kernel": "SoS" }, @@ -85,6 +91,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f5337e22", "metadata": { "kernel": "SoS" }, @@ -99,6 +106,7 @@ }, { "cell_type": "markdown", + "id": "8e0bf180", "metadata": { "kernel": "SoS" }, @@ -110,6 +118,7 @@ }, { "cell_type": "markdown", + "id": "d1247190", "metadata": { "kernel": "SoS" }, @@ -125,6 +134,7 @@ }, { "cell_type": "markdown", + "id": "b2a50a89", "metadata": { "kernel": "SoS" }, @@ -163,5 +173,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/misc/mini-protocol-example.ipynb b/code/SoS/misc/mini-protocol-example.ipynb index 33522e5d5..17f186b03 100644 --- a/code/SoS/misc/mini-protocol-example.ipynb +++ b/code/SoS/misc/mini-protocol-example.ipynb @@ -3,6 +3,7 @@ { "attachments": {}, "cell_type": "markdown", + "id": "55309812", "metadata": {}, "source": [ "# Mini-protocol Name\n", @@ -13,6 +14,7 @@ { "attachments": {}, "cell_type": "markdown", + "id": "314db31c", "metadata": {}, "source": [ "#### Miniprotocol Timing\n", @@ -24,6 +26,7 @@ { "attachments": {}, "cell_type": "markdown", + "id": "22e460dd", "metadata": {}, "source": [ "## Overview\n", @@ -35,6 +38,7 @@ { "attachments": {}, "cell_type": "markdown", + "id": "f64b8890", "metadata": {}, "source": [ "## Steps\n" @@ -43,6 +47,7 @@ { "attachments": {}, "cell_type": "markdown", + "id": "3fb6bef5", "metadata": {}, "source": [ "### i. [Step 1 description]" @@ -51,6 +56,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a6999cc5", "metadata": { "vscode": { "languageId": "plaintext" @@ -64,6 +70,7 @@ { "attachments": {}, "cell_type": "markdown", + "id": "54cc58dd", "metadata": {}, "source": [ "### ii. [Step 2 description]" @@ -72,6 +79,7 @@ { "cell_type": "code", "execution_count": null, + "id": "20ecaee4", "metadata": { "vscode": { "languageId": "plaintext" @@ -85,6 +93,7 @@ { "attachments": {}, "cell_type": "markdown", + "id": "5d936eb0", "metadata": {}, "source": [ "## Anticipated Results" @@ -94,9 +103,8 @@ "metadata": { "language_info": { "name": "python" - }, - "orig_nbformat": 4 + } }, "nbformat": 4, - "nbformat_minor": 2 + "nbformat_minor": 5 } diff --git a/code/SoS/misc/module-example.ipynb b/code/SoS/misc/module-example.ipynb index 019563393..043a5fe41 100644 --- a/code/SoS/misc/module-example.ipynb +++ b/code/SoS/misc/module-example.ipynb @@ -1,143 +1,155 @@ { - "cells": [ - { - "attachments": {}, - "cell_type": "markdown", - "metadata": {}, - "source": [ - "# Module Name\n", - "\n", - "[module description]" - ] - }, - { - "attachments": {}, - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Overview\n", - "\n", - "[more in-depth description of the module]" - ] - }, - { - "attachments": {}, - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Methods [optional]\n", - "\n", - "[methods descriptions]" - ] - }, - { - "attachments": {}, - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Workflow [optional]\n", - "[workflow description]" - ] - }, - { - "attachments": {}, - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Default Parameters: ____ [optional]\n", - "[workflow description]" - ] - }, - { - "attachments": {}, - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## [Anything else can be added here]\n", - "[description ]" - ] - }, - { - "attachments": {}, - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Input\n", - "\n", - "list your input files with descriptions" - ] - }, - { - "attachments": {}, - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Output\n", - "\n", - "list output files with descriptions" - ] - }, - { - "attachments": {}, - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Minimal Working Example\n", - "\n", - "Tell us where your data can be found (and use hyperlink)" - ] - }, - { - "attachments": {}, - "cell_type": "markdown", - "metadata": {}, - "source": [ - "### Step 1: name of step 1\n", - "Timing: < X minutes/hours" - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": { - "vscode": { - "languageId": "plaintext" - } - }, - "outputs": [], - "source": [ - "# step 1 sample code" - ] - }, - { - "attachments": {}, - "cell_type": "markdown", - "metadata": {}, - "source": [ - "```\n", - "Step 1 sample output\n", - "[1] \"Gene expression profiles loaded successfully!\"\n", - "[1] \"19 genes and 1118 samples are loaded from data/mwe.TPM.gct\"\n", - "[1] \"0 genes are filtered, because > 20 % samples have expression values < 0.1\"\n", - "[1] \"19 genes left, saving output.\"\n", - "```" - ] - }, - { - "attachments": {}, - "cell_type": "markdown", - "metadata": {}, - "source": [ - "## Command Interface and other pipeline code below" - ] + "cells": [ + { + "attachments": {}, + "cell_type": "markdown", + "id": "dad9f1fd", + "metadata": {}, + "source": [ + "# Module Name\n", + "\n", + "[module description]" + ] + }, + { + "attachments": {}, + "cell_type": "markdown", + "id": "66d5c73e", + "metadata": {}, + "source": [ + "## Overview\n", + "\n", + "[more in-depth description of the module]" + ] + }, + { + "attachments": {}, + "cell_type": "markdown", + "id": "acb46183", + "metadata": {}, + "source": [ + "## Methods [optional]\n", + "\n", + "[methods descriptions]" + ] + }, + { + "attachments": {}, + "cell_type": "markdown", + "id": "df71f583", + "metadata": {}, + "source": [ + "## Workflow [optional]\n", + "[workflow description]" + ] + }, + { + "attachments": {}, + "cell_type": "markdown", + "id": "1a2466da", + "metadata": {}, + "source": [ + "## Default Parameters: ____ [optional]\n", + "[workflow description]" + ] + }, + { + "attachments": {}, + "cell_type": "markdown", + "id": "d3c743bc", + "metadata": {}, + "source": [ + "## [Anything else can be added here]\n", + "[description ]" + ] + }, + { + "attachments": {}, + "cell_type": "markdown", + "id": "c8c3a61a", + "metadata": {}, + "source": [ + "## Input\n", + "\n", + "list your input files with descriptions" + ] + }, + { + "attachments": {}, + "cell_type": "markdown", + "id": "f372a378", + "metadata": {}, + "source": [ + "## Output\n", + "\n", + "list output files with descriptions" + ] + }, + { + "attachments": {}, + "cell_type": "markdown", + "id": "ba14ea11", + "metadata": {}, + "source": [ + "## Minimal Working Example\n", + "\n", + "Tell us where your data can be found (and use hyperlink)" + ] + }, + { + "attachments": {}, + "cell_type": "markdown", + "id": "dd42c6a7", + "metadata": {}, + "source": [ + "### Step 1: name of step 1\n", + "Timing: < X minutes/hours" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "b215d8a1", + "metadata": { + "vscode": { + "languageId": "plaintext" } - ], - "metadata": { - "language_info": { - "name": "python" - }, - "orig_nbformat": 4 + }, + "outputs": [], + "source": [ + "# step 1 sample code" + ] + }, + { + "attachments": {}, + "cell_type": "markdown", + "id": "d1385261", + "metadata": {}, + "source": [ + "```\n", + "Step 1 sample output\n", + "[1] \"Gene expression profiles loaded successfully!\"\n", + "[1] \"19 genes and 1118 samples are loaded from data/mwe.TPM.gct\"\n", + "[1] \"0 genes are filtered, because > 20 % samples have expression values < 0.1\"\n", + "[1] \"19 genes left, saving output.\"\n", + "```" + ] }, - "nbformat": 4, - "nbformat_minor": 2 + { + "attachments": {}, + "cell_type": "markdown", + "id": "66f7a3a3", + "metadata": {}, + "source": [ + "## Command Interface and other pipeline code below" + ] + } + ], + "metadata": { + "language_info": { + "name": "python" + } + }, + "nbformat": 4, + "nbformat_minor": 5 } diff --git a/code/SoS/misc/qtl_association_finemapping/1_xqtl_association.ipynb b/code/SoS/misc/qtl_association_finemapping/1_xqtl_association.ipynb index 559ba0d2e..3ef014679 100644 --- a/code/SoS/misc/qtl_association_finemapping/1_xqtl_association.ipynb +++ b/code/SoS/misc/qtl_association_finemapping/1_xqtl_association.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "0f156014", "metadata": { "kernel": "SoS" }, @@ -11,6 +12,7 @@ }, { "cell_type": "markdown", + "id": "b54e2f1c", "metadata": { "kernel": "SoS" }, @@ -24,6 +26,7 @@ }, { "cell_type": "markdown", + "id": "cf07ad07", "metadata": { "kernel": "SoS" }, @@ -42,6 +45,7 @@ }, { "cell_type": "markdown", + "id": "a753ba73", "metadata": { "kernel": "SoS" }, @@ -51,6 +55,7 @@ }, { "cell_type": "markdown", + "id": "e54b6204", "metadata": { "kernel": "SoS" }, @@ -63,6 +68,7 @@ { "cell_type": "code", "execution_count": null, + "id": "33f20fb0", "metadata": { "kernel": "SoS" }, @@ -76,6 +82,7 @@ }, { "cell_type": "markdown", + "id": "6d902c43", "metadata": { "kernel": "SoS" }, @@ -88,6 +95,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5d6d7d97", "metadata": { "kernel": "SoS" }, @@ -102,6 +110,7 @@ }, { "cell_type": "markdown", + "id": "73b799bb", "metadata": { "kernel": "SoS" }, @@ -114,6 +123,7 @@ { "cell_type": "code", "execution_count": null, + "id": "08c8212c", "metadata": { "kernel": "SoS" }, @@ -131,6 +141,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9ba0fccd", "metadata": { "kernel": "SoS" }, @@ -142,6 +153,7 @@ }, { "cell_type": "markdown", + "id": "55cea1f5", "metadata": { "kernel": "SoS" }, @@ -160,6 +172,7 @@ }, { "cell_type": "markdown", + "id": "888a07a9", "metadata": { "kernel": "SoS" }, @@ -170,6 +183,7 @@ { "cell_type": "code", "execution_count": null, + "id": "ad5e5cf2", "metadata": { "kernel": "SoS" }, @@ -180,6 +194,7 @@ }, { "cell_type": "markdown", + "id": "5669eb4d", "metadata": { "kernel": "SoS" }, @@ -194,6 +209,7 @@ }, { "cell_type": "markdown", + "id": "c93bff3a", "metadata": { "kernel": "SoS" }, @@ -203,6 +219,7 @@ }, { "cell_type": "markdown", + "id": "7dec41ad", "metadata": { "kernel": "SoS" }, @@ -215,6 +232,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f1537df3", "metadata": { "kernel": "SoS" }, @@ -229,6 +247,7 @@ }, { "cell_type": "markdown", + "id": "365168b8", "metadata": { "kernel": "SoS" }, @@ -244,6 +263,7 @@ { "cell_type": "code", "execution_count": null, + "id": "93eede5e", "metadata": { "kernel": "SoS" }, @@ -262,6 +282,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2cff5767", "metadata": { "kernel": "SoS" }, @@ -273,6 +294,7 @@ }, { "cell_type": "markdown", + "id": "c2d7cd6f", "metadata": { "kernel": "SoS" }, @@ -282,6 +304,7 @@ }, { "cell_type": "markdown", + "id": "5c1b5fc5", "metadata": { "kernel": "SoS" }, @@ -291,6 +314,7 @@ }, { "cell_type": "markdown", + "id": "088f8e90", "metadata": { "kernel": "SoS" }, @@ -303,6 +327,7 @@ { "cell_type": "code", "execution_count": null, + "id": "47cf36ac", "metadata": { "kernel": "SoS" }, @@ -325,6 +350,7 @@ { "cell_type": "code", "execution_count": null, + "id": "df267890", "metadata": { "kernel": "SoS" }, @@ -336,6 +362,7 @@ }, { "cell_type": "markdown", + "id": "0ea6ff8b", "metadata": { "kernel": "SoS" }, @@ -356,6 +383,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9a3fb8f1", "metadata": { "kernel": "SoS" }, @@ -366,6 +394,7 @@ }, { "cell_type": "markdown", + "id": "40cbb7a8", "metadata": { "kernel": "SoS" }, @@ -381,6 +410,7 @@ }, { "cell_type": "markdown", + "id": "74c54404", "metadata": { "kernel": "SoS" }, @@ -390,6 +420,7 @@ }, { "cell_type": "markdown", + "id": "ada4d8ab", "metadata": { "kernel": "SoS" }, @@ -407,6 +438,7 @@ }, { "cell_type": "markdown", + "id": "8c2a51d7", "metadata": { "kernel": "SoS" }, @@ -445,5 +477,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/misc/qtl_association_finemapping/2_finemapping.ipynb b/code/SoS/misc/qtl_association_finemapping/2_finemapping.ipynb index 8dd0e471f..15a10b583 100644 --- a/code/SoS/misc/qtl_association_finemapping/2_finemapping.ipynb +++ b/code/SoS/misc/qtl_association_finemapping/2_finemapping.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "1da4a2d2", "metadata": { "kernel": "SoS" }, @@ -28,6 +29,7 @@ }, { "cell_type": "markdown", + "id": "4c4405cf", "metadata": { "kernel": "SoS" }, @@ -43,6 +45,7 @@ }, { "cell_type": "markdown", + "id": "770b97e7", "metadata": { "kernel": "SoS" }, @@ -61,6 +64,7 @@ }, { "cell_type": "markdown", + "id": "2afb5634", "metadata": { "kernel": "SoS" }, @@ -70,6 +74,7 @@ }, { "cell_type": "markdown", + "id": "64c42afe", "metadata": { "kernel": "SoS" }, @@ -86,6 +91,7 @@ { "cell_type": "code", "execution_count": null, + "id": "d52e6e9c", "metadata": { "kernel": "SoS" }, @@ -108,6 +114,7 @@ }, { "cell_type": "markdown", + "id": "84c1cfe3", "metadata": { "kernel": "SoS" }, @@ -124,6 +131,7 @@ { "cell_type": "code", "execution_count": null, + "id": "11021d86", "metadata": { "kernel": "SoS" }, @@ -143,6 +151,7 @@ }, { "cell_type": "markdown", + "id": "7f2efc95", "metadata": { "kernel": "SoS" }, @@ -157,6 +166,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2b466a77", "metadata": { "kernel": "SoS" }, @@ -177,6 +187,7 @@ }, { "cell_type": "markdown", + "id": "75973cf4", "metadata": { "kernel": "SoS" }, @@ -193,6 +204,7 @@ }, { "cell_type": "markdown", + "id": "52925880", "metadata": { "kernel": "SoS" }, @@ -234,5 +246,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/mnm_analysis/mnm_methods/colocboost.ipynb b/code/SoS/mnm_analysis/mnm_methods/colocboost.ipynb index 093432ae2..8e072a02f 100644 --- a/code/SoS/mnm_analysis/mnm_methods/colocboost.ipynb +++ b/code/SoS/mnm_analysis/mnm_methods/colocboost.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "0af8bc2a", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "37e1529e", "metadata": { "kernel": "SoS" }, @@ -41,6 +43,7 @@ }, { "cell_type": "markdown", + "id": "20ea5d09", "metadata": { "kernel": "SoS", "tags": [] @@ -84,6 +87,7 @@ }, { "cell_type": "markdown", + "id": "df8b8b76", "metadata": { "jp-MarkdownHeadingCollapsed": true, "kernel": "SoS", @@ -165,6 +169,7 @@ }, { "cell_type": "markdown", + "id": "aee98940", "metadata": { "kernel": "SoS" }, @@ -180,6 +185,7 @@ }, { "cell_type": "markdown", + "id": "63fddc6a", "metadata": { "kernel": "SoS" }, @@ -191,17 +197,18 @@ }, { "cell_type": "markdown", + "id": "1e51346c", "metadata": { "kernel": "SoS" }, "source": [ "**Timing**: TBD (on toy dataset)" - ], - "id": "1e51346c" + ] }, { "cell_type": "code", "execution_count": null, + "id": "4681e465", "metadata": { "kernel": "SoS", "vscode": { @@ -222,6 +229,7 @@ }, { "cell_type": "markdown", + "id": "b678cd0e", "metadata": { "kernel": "SoS" }, @@ -233,17 +241,18 @@ }, { "cell_type": "markdown", + "id": "73abb479", "metadata": { "kernel": "SoS" }, "source": [ "**Timing**: TBD (on toy dataset)" - ], - "id": "73abb479" + ] }, { "cell_type": "code", "execution_count": null, + "id": "c4c9c24a", "metadata": { "kernel": "SoS" }, @@ -263,6 +272,7 @@ }, { "cell_type": "markdown", + "id": "a923cec2", "metadata": { "kernel": "SoS" }, @@ -273,17 +283,18 @@ }, { "cell_type": "markdown", + "id": "85dd4373", "metadata": { "kernel": "SoS" }, "source": [ "**Timing**: TBD (on toy dataset)" - ], - "id": "85dd4373" + ] }, { "cell_type": "code", "execution_count": null, + "id": "18b3e186", "metadata": { "kernel": "SoS" }, @@ -303,6 +314,7 @@ }, { "cell_type": "markdown", + "id": "17577bbb", "metadata": { "kernel": "SoS" }, @@ -313,6 +325,7 @@ { "cell_type": "code", "execution_count": null, + "id": "45b4dfdc", "metadata": { "kernel": "SoS" }, @@ -323,6 +336,7 @@ }, { "cell_type": "markdown", + "id": "3e333abb", "metadata": { "kernel": "SoS" }, @@ -439,6 +453,7 @@ }, { "cell_type": "markdown", + "id": "4ebf3cb3", "metadata": { "kernel": "SoS" }, @@ -449,6 +464,7 @@ { "cell_type": "code", "execution_count": null, + "id": "441b142a", "metadata": { "kernel": "SoS" }, @@ -535,6 +551,7 @@ { "cell_type": "code", "execution_count": null, + "id": "acfcc48a", "metadata": { "kernel": "SoS" }, @@ -573,6 +590,7 @@ { "cell_type": "code", "execution_count": null, + "id": "41479945", "metadata": { "kernel": "SoS" }, @@ -599,6 +617,7 @@ }, { "cell_type": "markdown", + "id": "bef16c54", "metadata": { "kernel": "SoS" }, @@ -609,6 +628,7 @@ { "cell_type": "code", "execution_count": null, + "id": "3d7a0ac7", "metadata": { "kernel": "SoS", "tags": [] @@ -704,5 +724,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/mnm_analysis/mnm_methods/mnm_regression.ipynb b/code/SoS/mnm_analysis/mnm_methods/mnm_regression.ipynb index 543bd322f..e9dc2f16e 100644 --- a/code/SoS/mnm_analysis/mnm_methods/mnm_regression.ipynb +++ b/code/SoS/mnm_analysis/mnm_methods/mnm_regression.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "bafd0692", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "16629d77", "metadata": { "kernel": "SoS" }, @@ -54,6 +56,7 @@ }, { "cell_type": "markdown", + "id": "11d747c9", "metadata": { "kernel": "SoS" }, @@ -148,6 +151,7 @@ }, { "cell_type": "markdown", + "id": "58ca74d0", "metadata": { "kernel": "SoS" }, @@ -245,6 +249,7 @@ { "cell_type": "code", "execution_count": null, + "id": "60850943", "metadata": { "kernel": "SoS" }, @@ -259,6 +264,7 @@ }, { "cell_type": "markdown", + "id": "dc2b43aa", "metadata": { "kernel": "SoS" }, @@ -271,6 +277,7 @@ }, { "cell_type": "markdown", + "id": "8e68639b", "metadata": { "kernel": "SoS" }, @@ -281,6 +288,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4c21ada3", "metadata": { "kernel": "SoS" }, @@ -297,6 +305,7 @@ }, { "cell_type": "markdown", + "id": "c7605cfe", "metadata": { "kernel": "SoS" }, @@ -307,17 +316,18 @@ }, { "cell_type": "markdown", + "id": "08156b31", "metadata": { "kernel": "SoS" }, "source": [ "**Timing**: TBD (on toy dataset)" - ], - "id": "08156b31" + ] }, { "cell_type": "code", "execution_count": null, + "id": "1973a618", "metadata": { "kernel": "SoS" }, @@ -334,6 +344,7 @@ }, { "cell_type": "markdown", + "id": "ebece135", "metadata": { "kernel": "SoS" }, @@ -343,6 +354,7 @@ }, { "cell_type": "markdown", + "id": "d7f0414e", "metadata": { "kernel": "SoS" }, @@ -354,17 +366,18 @@ }, { "cell_type": "markdown", + "id": "44f99ba1", "metadata": { "kernel": "SoS" }, "source": [ "**Timing**: TBD (on toy dataset)" - ], - "id": "44f99ba1" + ] }, { "cell_type": "code", "execution_count": null, + "id": "15e7b746", "metadata": { "kernel": "SoS" }, @@ -382,6 +395,7 @@ }, { "cell_type": "markdown", + "id": "c57af4de", "metadata": { "kernel": "SoS" }, @@ -391,17 +405,18 @@ }, { "cell_type": "markdown", + "id": "6abc252f", "metadata": { "kernel": "SoS" }, "source": [ "**Timing**: TBD (on toy dataset)" - ], - "id": "6abc252f" + ] }, { "cell_type": "code", "execution_count": null, + "id": "31da83a1", "metadata": { "kernel": "SoS" }, @@ -419,6 +434,7 @@ }, { "cell_type": "markdown", + "id": "406e386c", "metadata": { "kernel": "SoS" }, @@ -430,17 +446,18 @@ }, { "cell_type": "markdown", + "id": "c9ce1e11", "metadata": { "kernel": "SoS" }, "source": [ "**Timing**: TBD (on toy dataset)" - ], - "id": "c9ce1e11" + ] }, { "cell_type": "code", "execution_count": null, + "id": "d90f1282", "metadata": { "kernel": "Bash" }, @@ -457,6 +474,7 @@ }, { "cell_type": "markdown", + "id": "e33ba320", "metadata": { "kernel": "SoS" }, @@ -468,17 +486,18 @@ }, { "cell_type": "markdown", + "id": "0075bbe0", "metadata": { "kernel": "SoS" }, "source": [ "**Timing**: TBD (on toy dataset)" - ], - "id": "0075bbe0" + ] }, { "cell_type": "code", "execution_count": null, + "id": "e835a797", "metadata": { "kernel": "Bash" }, @@ -497,6 +516,7 @@ }, { "cell_type": "markdown", + "id": "d24edf49", "metadata": { "kernel": "SoS" }, @@ -508,17 +528,18 @@ }, { "cell_type": "markdown", + "id": "c0c5fe72", "metadata": { "kernel": "SoS" }, "source": [ "**Timing**: TBD (on toy dataset)" - ], - "id": "c0c5fe72" + ] }, { "cell_type": "code", "execution_count": null, + "id": "44f78b99", "metadata": { "kernel": "Bash" }, @@ -535,6 +556,7 @@ }, { "cell_type": "markdown", + "id": "84d4949c", "metadata": { "kernel": "SoS" }, @@ -546,17 +568,18 @@ }, { "cell_type": "markdown", + "id": "62d2817e", "metadata": { "kernel": "SoS" }, "source": [ "**Timing**: TBD (on toy dataset)" - ], - "id": "62d2817e" + ] }, { "cell_type": "code", "execution_count": null, + "id": "74f42e63", "metadata": { "kernel": "Bash" }, @@ -573,6 +596,7 @@ }, { "cell_type": "markdown", + "id": "85311f83", "metadata": { "kernel": "SoS" }, @@ -583,6 +607,7 @@ { "cell_type": "code", "execution_count": null, + "id": "3a73d7d6", "metadata": { "kernel": "SoS" }, @@ -593,6 +618,7 @@ }, { "cell_type": "markdown", + "id": "f2f2204c", "metadata": { "kernel": "SoS" }, @@ -847,6 +873,7 @@ }, { "cell_type": "markdown", + "id": "0a4ac0c6", "metadata": { "kernel": "SoS" }, @@ -859,6 +886,7 @@ { "cell_type": "code", "execution_count": null, + "id": "7b34e50f", "metadata": { "kernel": "SoS" }, @@ -971,6 +999,7 @@ { "cell_type": "code", "execution_count": null, + "id": "cbe06ae1", "metadata": { "kernel": "SoS" }, @@ -1027,6 +1056,7 @@ { "cell_type": "code", "execution_count": null, + "id": "85f579a1", "metadata": { "kernel": "SoS", "tags": [] @@ -1133,6 +1163,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e42770bd", "metadata": { "kernel": "SoS" }, @@ -1183,6 +1214,7 @@ { "cell_type": "code", "execution_count": null, + "id": "017d4806", "metadata": { "kernel": "SoS" }, @@ -1246,6 +1278,7 @@ { "cell_type": "code", "execution_count": null, + "id": "62f20d88", "metadata": { "kernel": "SoS" }, @@ -1312,6 +1345,7 @@ }, { "cell_type": "markdown", + "id": "767f3c9f", "metadata": { "kernel": "SoS" }, @@ -1324,6 +1358,7 @@ { "cell_type": "code", "execution_count": null, + "id": "19fcdeed", "metadata": { "kernel": "SoS" }, @@ -1364,6 +1399,7 @@ }, { "cell_type": "markdown", + "id": "1f78df73", "metadata": { "kernel": "SoS" }, @@ -1421,5 +1457,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 + "nbformat_minor": 5 } diff --git a/code/SoS/mnm_analysis/mnm_methods/qtl_rss_analysis.ipynb b/code/SoS/mnm_analysis/mnm_methods/qtl_rss_analysis.ipynb index d036f710f..4b2b9534b 100644 --- a/code/SoS/mnm_analysis/mnm_methods/qtl_rss_analysis.ipynb +++ b/code/SoS/mnm_analysis/mnm_methods/qtl_rss_analysis.ipynb @@ -2,80 +2,304 @@ "cells": [ { "cell_type": "markdown", + "id": "cbda1850", "metadata": {}, - "source": "# RSS Fine-mapping and TWAS Weights with QTL Summary Statistics\n\nFine-maps a cis window and learns TWAS weights from cis-QTL **summary statistics** plus an\nLD reference panel, without individual-level genotypes." + "source": [ + "# RSS Fine-mapping and TWAS Weights with QTL Summary Statistics\n", + "\n", + "Fine-maps a cis window and learns TWAS weights from cis-QTL **summary statistics** plus an\n", + "LD reference panel, without individual-level genotypes." + ] }, { "cell_type": "markdown", + "id": "df6137b8", "metadata": {}, - "source": "## Overview\n\n`mnm_regression.ipynb` fits a cis window from individual-level data: it needs the genotype\nmatrix and the phenotype for every sample. That is not always possible -- summary statistics\nare shareable where genotypes are not, and a scan that has already run need not be repeated.\n\nThis module takes the other route. It reads a cis-QTL nominal association table (effect,\nstandard error and the variant id per SNP), pairs it with an LD reference panel, and hands\nthe resulting `QtlSumStats` to the same two pipelines the individual-level route uses:\n\n- **`qtl_rss_fine_mapping`** runs SuSiE-RSS (`susieR::susie_rss`) over the window. The\n regression-with-summary-statistics likelihood replaces the individual-level one; credible\n sets and PIPs mean what they always did.\n- **`qtl_rss_twas_weights`** learns predictive weights from the same object. Not every\n method has a summary-statistics implementation: `mrash`, `lasso`, `scad`, `mcp`,\n `l0learn`, `mrmash` and `dpr_gibbs` do; `enet` and the `bayes_*` family are\n individual-level only and will be rejected here.\n\n**The LD panel is the load-bearing input.** RSS reconstructs the joint fit from marginal\nstatistics and LD, so the panel must cover the window's variants and must be on the same\nallele orientation. `summaryStatsQc()` harmonizes the two and reports what it corrected --\nread that line. If it says it sign- or strand-flipped everything, the alleles were declared\nwrong, not fixed (see `--variant-id-alleles` below).\n\n**When to run it.** After a cis scan (`TensorQTL.ipynb`), in place of\n`mnm_regression.ipynb`'s `susie_twas` when only summary statistics are available." + "source": [ + "## Overview\n", + "\n", + "`mnm_regression.ipynb` fits a cis window from individual-level data: it needs the genotype\n", + "matrix and the phenotype for every sample. That is not always possible -- summary statistics\n", + "are shareable where genotypes are not, and a scan that has already run need not be repeated.\n", + "\n", + "This module takes the other route. It reads a cis-QTL nominal association table (effect,\n", + "standard error and the variant id per SNP), pairs it with an LD reference panel, and hands\n", + "the resulting `QtlSumStats` to the same two pipelines the individual-level route uses:\n", + "\n", + "- **`qtl_rss_fine_mapping`** runs SuSiE-RSS (`susieR::susie_rss`) over the window. The\n", + " regression-with-summary-statistics likelihood replaces the individual-level one; credible\n", + " sets and PIPs mean what they always did.\n", + "- **`qtl_rss_twas_weights`** learns predictive weights from the same object. Not every\n", + " method has a summary-statistics implementation: `mrash`, `lasso`, `scad`, `mcp`,\n", + " `l0learn`, `mrmash` and `dpr_gibbs` do; `enet` and the `bayes_*` family are\n", + " individual-level only and will be rejected here.\n", + "\n", + "**The LD panel is the load-bearing input.** RSS reconstructs the joint fit from marginal\n", + "statistics and LD, so the panel must cover the window's variants and must be on the same\n", + "allele orientation. `summaryStatsQc()` harmonizes the two and reports what it corrected --\n", + "read that line. If it says it sign- or strand-flipped everything, the alleles were declared\n", + "wrong, not fixed (see `--variant-id-alleles` below).\n", + "\n", + "**When to run it.** After a cis scan (`TensorQTL.ipynb`), in place of\n", + "`mnm_regression.ipynb`'s `susie_twas` when only summary statistics are available." + ] }, { "cell_type": "markdown", + "id": "d36a27bd", "metadata": {}, - "source": "## Input\n\n- `--sumstats` **`output/cis/example.cis_qtl.pairs.tsv.gz`**\n(the cis-QTL nominal table from `TensorQTL.ipynb`: one row per (trait, variant) with\n`bhat`/`sebhat`, `pvalue`, `af` and `n`. A `z` column is used if present, otherwise the Wald\nz `bhat/sebhat` is derived.)\n\n- `--ld-sketch` **`tests/fixtures/qtl_mini/protocol_example.genotype.chr22.bed`**\n(the LD reference panel: a genotype path/prefix, or a per-chromosome LD meta file. Its\nvariants must cover the window.)\n\n- `--study` / `--context` / `--trait` -- the tuple this collection describes. `--trait` is\nthe gene id, and is also what the trait filter matches.\n\n- `--trait-column` (default `molecular_trait_id`) -- a cis scan writes *every* gene into one\ntable, so the rows for `--trait` are selected by this column. Set it empty for a file that\nalready holds one gene.\n\n- `--variant-id-alleles` (`none` | `A2A1` | `A1A2`) -- where the alleles come from when the\ntable has no `A1`/`A2` columns, as a cis scan typically does not: they sit inside the variant\nid. **The order cannot be inferred from the string** -- a `.pvar` writes `REF:ALT` (pecotmr's\ncanonical `A2:A1`) while a PLINK `.bim` writes `A1:A2` -- so declare which one your ids use.\nA real `A1`/`A2` column always wins. Declaring it wrong is silent: QC will \"correct\" the\napparent mismatch by flipping every variant against the panel.\n\n- `--genome` (default `GRCh38`), `--region`, `--n-sample`, `--column-mapping` -- optional.\n\nQC knobs are forwarded to `summaryStatsQc()`: `--maf`, `--mac`, `--imiss`,\n`--z-mismatch-qc`, `--pip-cutoff-to-skip`, and `--skip-qc` for diagnostics." + "source": [ + "## Input\n", + "\n", + "- `--sumstats` **`output/cis/example.cis_qtl.pairs.tsv.gz`**\n", + "(the cis-QTL nominal table from `TensorQTL.ipynb`: one row per (trait, variant) with\n", + "`bhat`/`sebhat`, `pvalue`, `af` and `n`. A `z` column is used if present, otherwise the Wald\n", + "z `bhat/sebhat` is derived.)\n", + "\n", + "- `--ld-sketch` **`tests/fixtures/qtl_mini/protocol_example.genotype.chr22.bed`**\n", + "(the LD reference panel: a genotype path/prefix, or a per-chromosome LD meta file. Its\n", + "variants must cover the window.)\n", + "\n", + "- `--study` / `--context` / `--trait` -- the tuple this collection describes. `--trait` is\n", + "the gene id, and is also what the trait filter matches.\n", + "\n", + "- `--trait-column` (default `molecular_trait_id`) -- a cis scan writes *every* gene into one\n", + "table, so the rows for `--trait` are selected by this column. Set it empty for a file that\n", + "already holds one gene.\n", + "\n", + "- `--variant-id-alleles` (`none` | `A2A1` | `A1A2`) -- where the alleles come from when the\n", + "table has no `A1`/`A2` columns, as a cis scan typically does not: they sit inside the variant\n", + "id. **The order cannot be inferred from the string** -- a `.pvar` writes `REF:ALT` (pecotmr's\n", + "canonical `A2:A1`) while a PLINK `.bim` writes `A1:A2` -- so declare which one your ids use.\n", + "A real `A1`/`A2` column always wins. Declaring it wrong is silent: QC will \"correct\" the\n", + "apparent mismatch by flipping every variant against the panel.\n", + "\n", + "- `--genome` (default `GRCh38`), `--region`, `--n-sample`, `--column-mapping` -- optional.\n", + "\n", + "QC knobs are forwarded to `summaryStatsQc()`: `--maf`, `--mac`, `--imiss`,\n", + "`--z-mismatch-qc`, `--pip-cutoff-to-skip`, and `--skip-qc` for diagnostics." + ] }, { "cell_type": "markdown", + "id": "5b2a2e3a", "metadata": {}, - "source": "## Output\n\n- `{cwd}/sumstats/{study}.{context}.{trait}.qtl_sumstats.rds` -- the `QtlSumStats`: the\nwindow's variants with `SNP`/`A1`/`A2`/`Z`/`N` (plus `BETA`/`SE`/`P`/`AF` when supplied), the\nLD panel attached as the `ldSketch`, and a `qcInfo` audit of what QC did.\n- `{cwd}/fine_mapping/{...}.qtl_rss_finemap.rds` -- a `QtlFineMappingResult`: credible sets\nand PIPs from SuSiE-RSS, the same class the individual-level route produces.\n- `{cwd}/twas_weights/{...}.qtl_rss_twas_weights.rds` -- a `TwasWeights` collection.\n\nEach step also writes `.stdout` / `.stderr` beside its output. The QC line in the sumstats\nlog is worth reading every time:\n\n```\n[study/context/gene] QC summary: 200 in -> 200 out | corrected: sign-flip 0, strand-flip 0\n```" + "source": [ + "## Output\n", + "\n", + "- `{cwd}/sumstats/{study}.{context}.{trait}.qtl_sumstats.rds` -- the `QtlSumStats`: the\n", + "window's variants with `SNP`/`A1`/`A2`/`Z`/`N` (plus `BETA`/`SE`/`P`/`AF` when supplied), the\n", + "LD panel attached as the `ldSketch`, and a `qcInfo` audit of what QC did.\n", + "- `{cwd}/fine_mapping/{...}.qtl_rss_finemap.rds` -- a `QtlFineMappingResult`: credible sets\n", + "and PIPs from SuSiE-RSS, the same class the individual-level route produces.\n", + "- `{cwd}/twas_weights/{...}.qtl_rss_twas_weights.rds` -- a `TwasWeights` collection.\n", + "\n", + "Each step also writes `.stdout` / `.stderr` beside its output. The QC line in the sumstats\n", + "log is worth reading every time:\n", + "\n", + "```\n", + "[study/context/gene] QC summary: 200 in -> 200 out | corrected: sign-flip 0, strand-flip 0\n", + "```" + ] }, { "cell_type": "markdown", + "id": "d3c7d32a", "metadata": {}, - "source": "## Minimal Working Example\n\nRuns on the committed chr22 toy data: the TensorQTL nominal table for 16 genes, with the\nsame 49-sample genotypes used as the LD reference (in-sample LD -- fine for a smoke test,\noptimistic for real inference, where a separate reference panel belongs)." + "source": [ + "## Minimal Working Example\n", + "\n", + "Runs on the committed chr22 toy data: the TensorQTL nominal table for 16 genes, with the\n", + "same 49-sample genotypes used as the LD reference (in-sample LD -- fine for a smoke test,\n", + "optimistic for real inference, where a separate reference panel belongs)." + ] }, { "cell_type": "code", "execution_count": null, + "id": "e6e5df2c", "metadata": {}, "outputs": [], - "source": "sos run pipeline/qtl_rss_analysis.ipynb qtl_rss \\\n --cwd output/qtl_rss \\\n --sumstats tests/fixtures/tensorqtl/expected/cis_qtl.pairs.tsv.gz \\\n --ld-sketch tests/fixtures/qtl_mini/protocol_example.genotype.chr22.bed \\\n --study test_study --context context1 --trait ENSG00000283047 \\\n --variant-id-alleles A1A2 \\\n --methods susie --twas-methods lasso -j1" + "source": [ + "sos run pipeline/qtl_rss_analysis.ipynb qtl_rss \\\n", + " --cwd output/qtl_rss \\\n", + " --sumstats tests/fixtures/tensorqtl/expected/cis_qtl.pairs.tsv.gz \\\n", + " --ld-sketch tests/fixtures/qtl_mini/protocol_example.genotype.chr22.bed \\\n", + " --study test_study --context context1 --trait ENSG00000283047 \\\n", + " --variant-id-alleles A1A2 \\\n", + " --methods susie --twas-methods lasso -j1" + ] }, { "cell_type": "markdown", + "id": "5da6277a", "metadata": {}, - "source": "## Command Interface" + "source": [ + "## Command Interface" + ] }, { "cell_type": "code", "execution_count": null, + "id": "495af7a4", "metadata": {}, "outputs": [], - "source": "sos run pipeline/qtl_rss_analysis.ipynb -h" + "source": [ + "sos run pipeline/qtl_rss_analysis.ipynb -h" + ] }, { "cell_type": "markdown", + "id": "d25086fe", "metadata": {}, - "source": "## Workflow implementation" + "source": [ + "## Workflow implementation" + ] }, { "cell_type": "code", "execution_count": null, + "id": "4e568b32", "metadata": {}, "outputs": [], - "source": "[global]\nparameter: cwd = path('output')\nparameter: modular_script_dir = path('code/script')\n# --- the (study, context, trait) this run describes ------------------\nparameter: study = str\nparameter: context = str\nparameter: trait = str\n# --- inputs ----------------------------------------------------------\nparameter: sumstats = path\nparameter: ld_sketch = path\nparameter: genome = 'GRCh38'\nparameter: region = ''\nparameter: n_sample = -1.0 # study-level total N; <0 = take it from the file\nparameter: column_mapping = ''\n# A cis scan writes every gene into one table; empty = the file holds one trait.\nparameter: trait_column = 'molecular_trait_id'\n# Where the alleles live when there is no A1/A2 column: none | A2A1 | A1A2.\n# Declaring this wrong is silent -- see the Input section.\nparameter: variant_id_alleles = 'none'\n# --- QC knobs (forwarded to summaryStatsQc) --------------------------\nparameter: maf = 0.0\nparameter: mac = 0.0\nparameter: imiss = 1.0\nparameter: z_mismatch_qc = 'none' # none | slalom | dentist\nparameter: pip_cutoff_to_skip = 0.0\nparameter: skip_qc = False\n# --- fine-mapping knobs (forwarded to fine_mapping.R) ----------------\nparameter: methods = 'susie'\nparameter: coverage = 0.95\nparameter: secondary_coverage = '0.7,0.5'\nparameter: min_abs_corr = 0.5\nparameter: pip_cutoff = 0.025\nparameter: L = 10\nparameter: L_greedy = 'none' # 'none'/'off' = greedy off (default); a positive int enables greedy-L\nparameter: ser_fallback = True\nparameter: r_mismatch = 'none' # none | eb | eb_mix\nparameter: method_args = '' # JSON {token: {kwarg: value}}\n# --- TWAS-weight knobs (forwarded to twas_weights.R) -----------------\n# Summary-statistics implementations only: mrash, lasso, scad, mcp, l0learn,\n# mrmash, dpr_gibbs. enet and the bayes_* family are individual-level only.\nparameter: twas_methods = 'lasso'\nparameter: seed = 999\n# --- cluster resources -----------------------------------------------\nparameter: job_size = 1\nparameter: walltime = '5h'\nparameter: mem = '16G'\nparameter: numThreads = 1\nparameter: container = ''\nparameter: entrypoint = ''\n\nprefix = f'{study}.{context}.{trait}'" + "source": [ + "[global]\n", + "parameter: cwd = path('output')\n", + "parameter: modular_script_dir = path('code/script')\n", + "# --- the (study, context, trait) this run describes ------------------\n", + "parameter: study = str\n", + "parameter: context = str\n", + "parameter: trait = str\n", + "# --- inputs ----------------------------------------------------------\n", + "parameter: sumstats = path\n", + "parameter: ld_sketch = path\n", + "parameter: genome = 'GRCh38'\n", + "parameter: region = ''\n", + "parameter: n_sample = -1.0 # study-level total N; <0 = take it from the file\n", + "parameter: column_mapping = ''\n", + "# A cis scan writes every gene into one table; empty = the file holds one trait.\n", + "parameter: trait_column = 'molecular_trait_id'\n", + "# Where the alleles live when there is no A1/A2 column: none | A2A1 | A1A2.\n", + "# Declaring this wrong is silent -- see the Input section.\n", + "parameter: variant_id_alleles = 'none'\n", + "# --- QC knobs (forwarded to summaryStatsQc) --------------------------\n", + "parameter: maf = 0.0\n", + "parameter: mac = 0.0\n", + "parameter: imiss = 1.0\n", + "parameter: z_mismatch_qc = 'none' # none | slalom | dentist\n", + "parameter: pip_cutoff_to_skip = 0.0\n", + "parameter: skip_qc = False\n", + "# --- fine-mapping knobs (forwarded to fine_mapping.R) ----------------\n", + "parameter: methods = 'susie'\n", + "parameter: coverage = 0.95\n", + "parameter: secondary_coverage = '0.7,0.5'\n", + "parameter: min_abs_corr = 0.5\n", + "parameter: pip_cutoff = 0.025\n", + "parameter: L = 10\n", + "parameter: L_greedy = 'none' # 'none'/'off' = greedy off (default); a positive int enables greedy-L\n", + "parameter: ser_fallback = True\n", + "parameter: r_mismatch = 'none' # none | eb | eb_mix\n", + "parameter: method_args = '' # JSON {token: {kwarg: value}}\n", + "# --- TWAS-weight knobs (forwarded to twas_weights.R) -----------------\n", + "# Summary-statistics implementations only: mrash, lasso, scad, mcp, l0learn,\n", + "# mrmash, dpr_gibbs. enet and the bayes_* family are individual-level only.\n", + "parameter: twas_methods = 'lasso'\n", + "parameter: seed = 999\n", + "# --- cluster resources -----------------------------------------------\n", + "parameter: job_size = 1\n", + "parameter: walltime = '5h'\n", + "parameter: mem = '16G'\n", + "parameter: numThreads = 1\n", + "parameter: container = ''\n", + "parameter: entrypoint = ''\n", + "\n", + "prefix = f'{study}.{context}.{trait}'" + ] }, { "cell_type": "code", "execution_count": null, + "id": "a4fc66cf", "metadata": {}, "outputs": [], - "source": "[qtl_rss_1, generate_qtl_sumstats]\n# Read the cis-QTL nominal table, restrict it to this trait, attach the LD panel\n# and run summaryStatsQc -> one QtlSumStats for the window.\noutput: f'{cwd:a}/sumstats/{prefix}.qtl_sumstats.rds'\ntask: trunk_workers = 1, trunk_size = job_size, walltime = walltime, mem = mem, cores = numThreads, tags = f'{step_name}_{_output:bn}'\nbash: expand = '${ }', stderr = f'{_output:n}.stderr', stdout = f'{_output:n}.stdout', container = container, entrypoint = entrypoint\n Rscript ${modular_script_dir}/pecotmr_integration/qtl_sumstats_construct.R \\\n --sumstats ${sumstats:a} \\\n --study ${study} \\\n --context ${context} \\\n --trait ${trait} \\\n --trait-column '${trait_column}' \\\n --variant-id-alleles ${variant_id_alleles} \\\n --ld-sketch ${ld_sketch:a} \\\n --genome ${genome} \\\n ${('--region ' + region) if region else ''} \\\n ${('--n-sample ' + str(n_sample)) if n_sample >= 0 else ''} \\\n ${('--column-mapping ' + column_mapping) if column_mapping else ''} \\\n --maf ${maf} \\\n --mac ${mac} \\\n --imiss ${imiss} \\\n --z-mismatch-qc ${z_mismatch_qc} \\\n --pip-cutoff-to-skip ${pip_cutoff_to_skip} \\\n ${'--skip-qc' if skip_qc else ''} \\\n --output ${_output}" + "source": [ + "[qtl_rss_1, generate_qtl_sumstats]\n", + "# Read the cis-QTL nominal table, restrict it to this trait, attach the LD panel\n", + "# and run summaryStatsQc -> one QtlSumStats for the window.\n", + "output: f'{cwd:a}/sumstats/{prefix}.qtl_sumstats.rds'\n", + "task: trunk_workers = 1, trunk_size = job_size, walltime = walltime, mem = mem, cores = numThreads, tags = f'{step_name}_{_output:bn}'\n", + "bash: expand = '${ }', stderr = f'{_output:n}.stderr', stdout = f'{_output:n}.stdout', container = container, entrypoint = entrypoint\n", + " Rscript ${modular_script_dir}/pecotmr_integration/qtl_sumstats_construct.R \\\n", + " --sumstats ${sumstats:a} \\\n", + " --study ${study} \\\n", + " --context ${context} \\\n", + " --trait ${trait} \\\n", + " --trait-column '${trait_column}' \\\n", + " --variant-id-alleles ${variant_id_alleles} \\\n", + " --ld-sketch ${ld_sketch:a} \\\n", + " --genome ${genome} \\\n", + " ${('--region ' + region) if region else ''} \\\n", + " ${('--n-sample ' + str(n_sample)) if n_sample >= 0 else ''} \\\n", + " ${('--column-mapping ' + column_mapping) if column_mapping else ''} \\\n", + " --maf ${maf} \\\n", + " --mac ${mac} \\\n", + " --imiss ${imiss} \\\n", + " --z-mismatch-qc ${z_mismatch_qc} \\\n", + " --pip-cutoff-to-skip ${pip_cutoff_to_skip} \\\n", + " ${'--skip-qc' if skip_qc else ''} \\\n", + " --output ${_output}" + ] }, { "cell_type": "code", "execution_count": null, + "id": "f4deb69f", "metadata": {}, "outputs": [], - "source": "[qtl_rss_2, qtl_rss_fine_mapping]\n# SuSiE-RSS over the window: the same fine_mapping.R the individual-level and\n# GWAS routes use, dispatching on the QtlSumStats class.\ninput: f'{cwd:a}/sumstats/{prefix}.qtl_sumstats.rds'\noutput: f'{cwd:a}/fine_mapping/{prefix}.qtl_rss_finemap.rds'\ntask: trunk_workers = 1, trunk_size = job_size, walltime = walltime, mem = mem, cores = numThreads, tags = f'{step_name}_{_output:bn}'\nbash: expand = '${ }', stderr = f'{_output:n}.stderr', stdout = f'{_output:n}.stdout', container = container, entrypoint = entrypoint\n Rscript ${modular_script_dir}/pecotmr_integration/fine_mapping.R \\\n --qtl-sumstats ${_input} \\\n --methods ${methods} \\\n --coverage ${coverage} \\\n --secondary-coverage ${secondary_coverage} \\\n --min-abs-corr ${min_abs_corr} \\\n --pip-cutoff ${pip_cutoff} \\\n --L ${L} \\\n --L-greedy ${L_greedy} \\\n --ser-fallback ${'TRUE' if ser_fallback else 'FALSE'} \\\n --r-mismatch ${r_mismatch} \\\n ${('--method-args ' + repr(method_args)) if method_args else ''} \\\n --seed ${seed} \\\n --output ${_output}" + "source": [ + "[qtl_rss_2, qtl_rss_fine_mapping]\n", + "# SuSiE-RSS over the window: the same fine_mapping.R the individual-level and\n", + "# GWAS routes use, dispatching on the QtlSumStats class.\n", + "input: f'{cwd:a}/sumstats/{prefix}.qtl_sumstats.rds'\n", + "output: f'{cwd:a}/fine_mapping/{prefix}.qtl_rss_finemap.rds'\n", + "task: trunk_workers = 1, trunk_size = job_size, walltime = walltime, mem = mem, cores = numThreads, tags = f'{step_name}_{_output:bn}'\n", + "bash: expand = '${ }', stderr = f'{_output:n}.stderr', stdout = f'{_output:n}.stdout', container = container, entrypoint = entrypoint\n", + " Rscript ${modular_script_dir}/pecotmr_integration/fine_mapping.R \\\n", + " --qtl-sumstats ${_input} \\\n", + " --methods ${methods} \\\n", + " --coverage ${coverage} \\\n", + " --secondary-coverage ${secondary_coverage} \\\n", + " --min-abs-corr ${min_abs_corr} \\\n", + " --pip-cutoff ${pip_cutoff} \\\n", + " --L ${L} \\\n", + " --L-greedy ${L_greedy} \\\n", + " --ser-fallback ${'TRUE' if ser_fallback else 'FALSE'} \\\n", + " --r-mismatch ${r_mismatch} \\\n", + " ${('--method-args ' + repr(method_args)) if method_args else ''} \\\n", + " --seed ${seed} \\\n", + " --output ${_output}" + ] }, { "cell_type": "code", "execution_count": null, + "id": "f6401974", "metadata": {}, "outputs": [], - "source": "[qtl_rss_3, qtl_rss_twas_weights]\n# RSS TWAS weights from the same collection. A QtlSumStats already spans one\n# window, so no --gene-id / --region selection applies.\ninput: f'{cwd:a}/sumstats/{prefix}.qtl_sumstats.rds'\noutput: f'{cwd:a}/twas_weights/{prefix}.qtl_rss_twas_weights.rds'\ntask: trunk_workers = 1, trunk_size = job_size, walltime = walltime, mem = mem, cores = numThreads, tags = f'{step_name}_{_output:bn}'\nbash: expand = '${ }', stderr = f'{_output:n}.stderr', stdout = f'{_output:n}.stdout', container = container, entrypoint = entrypoint\n Rscript ${modular_script_dir}/pecotmr_integration/twas_weights.R \\\n --qtl-sumstats ${_input} \\\n --methods ${twas_methods} \\\n --seed ${seed} \\\n --output ${_output}" + "source": [ + "[qtl_rss_3, qtl_rss_twas_weights]\n", + "# RSS TWAS weights from the same collection. A QtlSumStats already spans one\n", + "# window, so no --gene-id / --region selection applies.\n", + "input: f'{cwd:a}/sumstats/{prefix}.qtl_sumstats.rds'\n", + "output: f'{cwd:a}/twas_weights/{prefix}.qtl_rss_twas_weights.rds'\n", + "task: trunk_workers = 1, trunk_size = job_size, walltime = walltime, mem = mem, cores = numThreads, tags = f'{step_name}_{_output:bn}'\n", + "bash: expand = '${ }', stderr = f'{_output:n}.stderr', stdout = f'{_output:n}.stdout', container = container, entrypoint = entrypoint\n", + " Rscript ${modular_script_dir}/pecotmr_integration/twas_weights.R \\\n", + " --qtl-sumstats ${_input} \\\n", + " --methods ${twas_methods} \\\n", + " --seed ${seed} \\\n", + " --output ${_output}" + ] } ], "metadata": { @@ -106,5 +330,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 + "nbformat_minor": 5 } diff --git a/code/SoS/mnm_analysis/mnm_methods/rss_analysis.ipynb b/code/SoS/mnm_analysis/mnm_methods/rss_analysis.ipynb index 5ed9fd0b6..f723ce6af 100644 --- a/code/SoS/mnm_analysis/mnm_methods/rss_analysis.ipynb +++ b/code/SoS/mnm_analysis/mnm_methods/rss_analysis.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "92ef0327", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "3626f35c", "metadata": { "kernel": "SoS" }, @@ -44,6 +46,7 @@ }, { "cell_type": "markdown", + "id": "dd166609", "metadata": { "kernel": "SoS" }, @@ -113,6 +116,7 @@ }, { "cell_type": "markdown", + "id": "0bd14aec", "metadata": { "kernel": "SoS" }, @@ -155,6 +159,7 @@ }, { "cell_type": "markdown", + "id": "d18f9a37", "metadata": { "kernel": "SoS" }, @@ -165,6 +170,7 @@ }, { "cell_type": "markdown", + "id": "67768dff", "metadata": { "kernel": "SoS" }, @@ -174,6 +180,7 @@ }, { "cell_type": "markdown", + "id": "b2f47cef", "metadata": { "kernel": "SoS" }, @@ -184,6 +191,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f2af6234", "metadata": { "kernel": "SoS" }, @@ -199,6 +207,7 @@ }, { "cell_type": "markdown", + "id": "a7be2fc9", "metadata": { "kernel": "SoS" }, @@ -208,6 +217,7 @@ }, { "cell_type": "markdown", + "id": "fdd47aab", "metadata": { "kernel": "SoS" }, @@ -218,6 +228,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e0cdb07f", "metadata": { "kernel": "SoS" }, @@ -235,6 +246,7 @@ }, { "cell_type": "markdown", + "id": "4540c8a5", "metadata": { "kernel": "SoS" }, @@ -245,6 +257,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b557081c", "metadata": { "kernel": "SoS" }, @@ -255,6 +268,7 @@ }, { "cell_type": "markdown", + "id": "f64f8a61", "metadata": { "kernel": "SoS" }, @@ -349,6 +363,7 @@ }, { "cell_type": "markdown", + "id": "6889c8c4", "metadata": { "kernel": "SoS" }, @@ -359,6 +374,7 @@ { "cell_type": "code", "execution_count": null, + "id": "baf7a4ba", "metadata": { "kernel": "SoS" }, @@ -421,6 +437,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9221bd01", "metadata": { "kernel": "SoS" }, @@ -445,6 +462,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e4d571a4", "metadata": { "kernel": "SoS" }, @@ -486,6 +504,7 @@ { "cell_type": "code", "execution_count": null, + "id": "90de2edc", "metadata": { "kernel": "SoS" }, @@ -518,6 +537,7 @@ { "cell_type": "code", "execution_count": null, + "id": "1cf1b0da", "metadata": { "kernel": "SoS" }, @@ -562,5 +582,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 + "nbformat_minor": 5 } diff --git a/code/SoS/mnm_analysis/mnm_miniprotocol.ipynb b/code/SoS/mnm_analysis/mnm_miniprotocol.ipynb index 688e77420..15b5d693e 100644 --- a/code/SoS/mnm_analysis/mnm_miniprotocol.ipynb +++ b/code/SoS/mnm_analysis/mnm_miniprotocol.ipynb @@ -27,9 +27,9 @@ "source": [ "## Overview\n", "\n", - "The [MNM regression module](https://statfungen.github.io/xqtl-protocol/code/mnm_analysis/mnm_methods/mnm_regression.html) analyzes individual-level genotype and molecular-phenotype data. `qtl_dataset_construct+susie_twas` performs univariate SuSiE fine-mapping and estimates TWAS weights; `mnm` jointly analyzes multiple molecular traits with multivariate priors; `mnm_genes` extends the multivariate model across genes; and `fsusie` fine-maps functional or epigenomic phenotypes.\n", + "The [MNM regression module](https://statfungen.github.io/xqtl-protocol/mnm-regression) analyzes individual-level genotype and molecular-phenotype data. `qtl_dataset_construct+susie_twas` performs univariate SuSiE fine-mapping and estimates TWAS weights; `mnm` jointly analyzes multiple molecular traits with multivariate priors; `mnm_genes` extends the multivariate model across genes; and `fsusie` fine-maps functional or epigenomic phenotypes.\n", "\n", - "The [RSS module](https://statfungen.github.io/xqtl-protocol/code/mnm_analysis/mnm_methods/rss_analysis.html) instead combines GWAS summary statistics with an external LD reference. These five commands are alternative analysis routes rather than a mandatory chain." + "The [RSS module](https://statfungen.github.io/xqtl-protocol/rss-analysis) instead combines GWAS summary statistics with an external LD reference. These five commands are alternative analysis routes rather than a mandatory chain." ] }, { @@ -55,7 +55,7 @@ "id": "14f54de1", "metadata": {}, "source": [ - "### [1. Univariate fine-mapping and TWAS](https://statfungen.github.io/xqtl-protocol/code/mnm_analysis/mnm_methods/mnm_regression.html)\n", + "### [1. Univariate fine-mapping and TWAS](https://statfungen.github.io/xqtl-protocol/mnm-regression)\n", "\n", "**What it does:** `qtl_dataset_construct+susie_twas` builds the regional dataset, fits SuSiE and saves fine-mapping results and cross-validated TWAS weights." ] @@ -75,7 +75,7 @@ "id": "bb824285", "metadata": {}, "source": [ - "### [2. Multivariate fine-mapping](https://statfungen.github.io/xqtl-protocol/code/mnm_analysis/mnm_methods/mnm_regression.html)\n", + "### [2. Multivariate fine-mapping](https://statfungen.github.io/xqtl-protocol/mnm-regression)\n", "\n", "**What it does:** `mnm` jointly fine-maps multiple molecular traits using the analyses and prior settings specified by the fine-mapping metadata." ] @@ -95,7 +95,7 @@ "id": "16a9100b", "metadata": {}, "source": [ - "### [3. Multigene multivariate fine-mapping](https://statfungen.github.io/xqtl-protocol/code/mnm_analysis/mnm_methods/mnm_regression.html)\n", + "### [3. Multigene multivariate fine-mapping](https://statfungen.github.io/xqtl-protocol/mnm-regression)\n", "\n", "**What it does:** `mnm_genes` coordinates multivariate fine-mapping across genes while preserving phenotype identifiers and a common retained-sample set." ] @@ -115,7 +115,7 @@ "id": "30bd5556", "metadata": {}, "source": [ - "### [4. Functional fine-mapping](https://statfungen.github.io/xqtl-protocol/code/mnm_analysis/mnm_methods/mnm_regression.html)\n", + "### [4. Functional fine-mapping](https://statfungen.github.io/xqtl-protocol/mnm-regression)\n", "\n", "**What it does:** `fsusie` applies functional SuSiE to epigenomic or other functional phenotypes and saves posterior fine-mapping results and residual data." ] @@ -135,7 +135,7 @@ "id": "28ac41b4", "metadata": {}, "source": [ - "### [5. Summary-statistic fine-mapping](https://statfungen.github.io/xqtl-protocol/code/mnm_analysis/mnm_methods/rss_analysis.html)\n", + "### [5. Summary-statistic fine-mapping](https://statfungen.github.io/xqtl-protocol/rss-analysis)\n", "\n", "**What it does:** the RSS chain harmonizes regional GWAS summary statistics with the LD reference, runs SuSiE-RSS fine-mapping and produces a regional diagnostic plot." ] @@ -226,4 +226,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/mnm_analysis/mnm_postprocessing.ipynb b/code/SoS/mnm_analysis/mnm_postprocessing.ipynb index cdfe2bfe2..f4711e866 100644 --- a/code/SoS/mnm_analysis/mnm_postprocessing.ipynb +++ b/code/SoS/mnm_analysis/mnm_postprocessing.ipynb @@ -1868,4 +1868,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/mnm_analysis/multivariate_fine_mapping_vignette.ipynb b/code/SoS/mnm_analysis/multivariate_fine_mapping_vignette.ipynb index 3b5ea6724..0694627ea 100644 --- a/code/SoS/mnm_analysis/multivariate_fine_mapping_vignette.ipynb +++ b/code/SoS/mnm_analysis/multivariate_fine_mapping_vignette.ipynb @@ -63,7 +63,7 @@ "id": "16e87087", "metadata": {}, "source": [ - "### [Run cross-context fine-mapping](https://statfungen.github.io/xqtl-protocol/code/mnm_analysis/mnm_methods/mnm_regression.html#multivariate-fine-mapping)\n", + "### [Run cross-context fine-mapping](https://statfungen.github.io/xqtl-protocol/mnm-regression#multivariate-fine-mapping)\n", "\n", "`qtl_dataset_construct+mnm` constructs the harmonized dataset and performs one joint mvSuSiE fit for the selected trait." ] @@ -233,4 +233,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/mnm_analysis/multivariate_multigene_fine_mapping_vignette.ipynb b/code/SoS/mnm_analysis/multivariate_multigene_fine_mapping_vignette.ipynb index 1d658a123..e7fafe74d 100644 --- a/code/SoS/mnm_analysis/multivariate_multigene_fine_mapping_vignette.ipynb +++ b/code/SoS/mnm_analysis/multivariate_multigene_fine_mapping_vignette.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "9bb05509", "metadata": {}, "source": [ "# Cross-gene fine-mapping with mvSuSiE\n", @@ -11,6 +12,7 @@ }, { "cell_type": "markdown", + "id": "dc36fe77", "metadata": {}, "source": [ "## Learning goals\n", @@ -25,6 +27,7 @@ }, { "cell_type": "markdown", + "id": "86734511", "metadata": {}, "source": [ "## Background and method\n", @@ -38,6 +41,7 @@ }, { "cell_type": "markdown", + "id": "2b79d7e5", "metadata": {}, "source": [ "## Worked example\n", @@ -56,15 +60,17 @@ }, { "cell_type": "markdown", + "id": "ac741db1", "metadata": {}, "source": [ - "### [Run cross-gene fine-mapping](https://statfungen.github.io/xqtl-protocol/code/mnm_analysis/mnm_methods/mnm_regression.html#cross-gene-multivariate-fine-mapping)\n", + "### [Run cross-gene fine-mapping](https://statfungen.github.io/xqtl-protocol/mnm-regression#cross-gene-multivariate-fine-mapping)\n", "\n", "`qtl_dataset_construct+mnm_genes` constructs the shared dataset, identifies the genes in each selected locus and fits mvSuSiE jointly." ] }, { "cell_type": "markdown", + "id": "84d4fd8c", "metadata": {}, "source": [ "**Timing**: TBD" @@ -72,7 +78,10 @@ }, { "cell_type": "code", + "execution_count": null, + "id": "8a2cdbe3", "metadata": {}, + "outputs": [], "source": [ "sos run pipeline/mnm_regression.ipynb qtl_dataset_construct+mnm_genes \\\n", " --name protocol_example \\\n", @@ -84,12 +93,11 @@ " --region-name ENSG00000130538 \\\n", " --transpose-covariates \\\n", " -j 1" - ], - "execution_count": null, - "outputs": [] + ] }, { "cell_type": "markdown", + "id": "84dd9eb0", "metadata": {}, "source": [ "The command is expected to create:\n", @@ -103,6 +111,7 @@ }, { "cell_type": "markdown", + "id": "7cd2a588", "metadata": {}, "source": [ "### Command reference\n", @@ -112,15 +121,17 @@ }, { "cell_type": "code", + "execution_count": null, + "id": "61337190", "metadata": {}, + "outputs": [], "source": [ "sos run pipeline/mnm_regression.ipynb -h" - ], - "execution_count": null, - "outputs": [] + ] }, { "cell_type": "markdown", + "id": "20177656", "metadata": {}, "source": [ "## Results and interpretation\n", @@ -139,7 +150,10 @@ }, { "cell_type": "code", + "execution_count": null, + "id": "9a4b093f", "metadata": {}, + "outputs": [], "source": [ "suppressPackageStartupMessages(library(pecotmr))\n", "\n", @@ -153,12 +167,11 @@ "top_loci <- top_loci[order(top_loci$pip, decreasing = TRUE), ]\n", "keep <- intersect(c(\"variant_id\", \"pip\", \"cs_95\", \"cs_95_purity\", \"method\"), names(top_loci))\n", "head(top_loci[, keep, drop = FALSE], 8)" - ], - "execution_count": null, - "outputs": [] + ] }, { "cell_type": "markdown", + "id": "16fb7c35", "metadata": {}, "source": [ "A high joint PIP identifies a variant supported by the multivariate regional model; it does not by itself identify which gene mediates the signal. Interpret the variant-level evidence together with posterior effects for each gene. If several genes retain similar effects, the data may not separate shared regulation from correlated phenotypes. A broad or low-purity credible set indicates limited localization." @@ -166,6 +179,7 @@ }, { "cell_type": "markdown", + "id": "1e787afb", "metadata": {}, "source": [ "## Limitations and common pitfalls\n", @@ -180,6 +194,7 @@ }, { "cell_type": "markdown", + "id": "8f13cd3e", "metadata": {}, "source": [ "## Next steps\n", @@ -209,4 +224,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/mnm_analysis/summary_stats_finemapping_vignette.ipynb b/code/SoS/mnm_analysis/summary_stats_finemapping_vignette.ipynb index 382e05fd9..6cd4b9ccc 100644 --- a/code/SoS/mnm_analysis/summary_stats_finemapping_vignette.ipynb +++ b/code/SoS/mnm_analysis/summary_stats_finemapping_vignette.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "f0979ca2", "metadata": {}, "source": [ "# Fine-mapping GWAS summary statistics with SuSiE RSS\n", @@ -11,6 +12,7 @@ }, { "cell_type": "markdown", + "id": "9e1cde41", "metadata": {}, "source": [ "## Learning goals\n", @@ -20,6 +22,7 @@ }, { "cell_type": "markdown", + "id": "480a08a7", "metadata": {}, "source": [ "## Background and method\n", @@ -31,6 +34,7 @@ }, { "cell_type": "markdown", + "id": "6fe4e045", "metadata": {}, "source": [ "## Worked example\n", @@ -49,15 +53,17 @@ }, { "cell_type": "markdown", + "id": "15f3b18e", "metadata": {}, "source": [ - "### [Run summary-statistics fine-mapping](https://statfungen.github.io/xqtl-protocol/code/mnm_analysis/mnm_methods/rss_analysis.html#minimal-working-example)\n", + "### [Run summary-statistics fine-mapping](https://statfungen.github.io/xqtl-protocol/rss-analysis#minimal-working-example)\n", "\n", "The four chained workflows generate the analysis manifest, harmonize GWAS statistics, fit SuSiE RSS and draw the PIP plot." ] }, { "cell_type": "markdown", + "id": "d8d47719", "metadata": {}, "source": [ "**Timing**: TBD" @@ -65,7 +71,10 @@ }, { "cell_type": "code", + "execution_count": null, + "id": "e96bafd3", "metadata": {}, + "outputs": [], "source": [ "sos run pipeline/rss_analysis.ipynb \\\n", " generate_manifest+generate_gwas_sumstats+gwas_fine_mapping+gwas_rss_plot \\\n", @@ -74,12 +83,11 @@ " --gwas-meta tests/fixtures/rss_analysis/protocol_example.rss_mwe.gwas_meta.tsv \\\n", " --regions chr22:49355984-50799822 \\\n", " --ld-meta tests/fixtures/ld_reference/ld_meta_file.tsv" - ], - "execution_count": null, - "outputs": [] + ] }, { "cell_type": "markdown", + "id": "fbf85155", "metadata": {}, "source": [ "The command creates:\n", @@ -93,6 +101,7 @@ }, { "cell_type": "markdown", + "id": "931322d7", "metadata": {}, "source": [ "### Command reference\n", @@ -102,15 +111,17 @@ }, { "cell_type": "code", + "execution_count": null, + "id": "8c2e23b3", "metadata": {}, + "outputs": [], "source": [ "sos run pipeline/rss_analysis.ipynb -h" - ], - "execution_count": null, - "outputs": [] + ] }, { "cell_type": "markdown", + "id": "797014c4", "metadata": {}, "source": [ "## Results and interpretation\n", @@ -130,7 +141,10 @@ }, { "cell_type": "code", + "execution_count": null, + "id": "0654216b", "metadata": {}, + "outputs": [], "source": [ "suppressPackageStartupMessages(library(pecotmr))\n", "\n", @@ -144,12 +158,11 @@ "top_loci <- attr(entry, \"topLoci\")\n", "top_loci <- top_loci[order(top_loci$pip, decreasing = TRUE), ]\n", "head(top_loci[, c(\"variant_id\", \"af\", \"marginal_z\", \"pip\", \"cs_95\", \"cs_95_purity\")], 8)" - ], - "execution_count": null, - "outputs": [] + ] }, { "cell_type": "markdown", + "id": "412235c7", "metadata": {}, "source": [ "The toy result ranks variants by PIP but does not retain a high-purity credible set. That means the example demonstrates the object and workflow without localizing a causal variant. In a substantive analysis, prioritize high-PIP variants only after checking credible-set purity, allele alignment, GWAS/LD ancestry matching and sensitivity to QC. Several variants in one credible set represent unresolved alternatives, not multiple confirmed causal variants." @@ -157,6 +170,7 @@ }, { "cell_type": "markdown", + "id": "f3f6b5fd", "metadata": {}, "source": [ "## Limitations and common pitfalls\n", @@ -171,6 +185,7 @@ }, { "cell_type": "markdown", + "id": "491b36a7", "metadata": {}, "source": [ "## Next steps\n", @@ -200,4 +215,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/mnm_analysis/univariate_fine_mapping_fsusie_vignette.ipynb b/code/SoS/mnm_analysis/univariate_fine_mapping_fsusie_vignette.ipynb index beade799d..dff981697 100644 --- a/code/SoS/mnm_analysis/univariate_fine_mapping_fsusie_vignette.ipynb +++ b/code/SoS/mnm_analysis/univariate_fine_mapping_fsusie_vignette.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "4797237f", "metadata": {}, "source": [ "# Functional fine-mapping with fSuSiE\n", @@ -11,6 +12,7 @@ }, { "cell_type": "markdown", + "id": "6df87bcc", "metadata": {}, "source": [ "## Learning goals\n", @@ -25,6 +27,7 @@ }, { "cell_type": "markdown", + "id": "e00a937d", "metadata": {}, "source": [ "## Background and method\n", @@ -38,6 +41,7 @@ }, { "cell_type": "markdown", + "id": "85148b80", "metadata": {}, "source": [ "## Worked example\n", @@ -56,15 +60,17 @@ }, { "cell_type": "markdown", + "id": "7126e642", "metadata": {}, "source": [ - "### [Run fSuSiE](https://statfungen.github.io/xqtl-protocol/code/mnm_analysis/mnm_methods/mnm_regression.html#fsusie)\n", + "### [Run fSuSiE](https://statfungen.github.io/xqtl-protocol/mnm-regression#fsusie)\n", "\n", "`qtl_dataset_construct+fsusie` builds the regional dataset and then performs the joint fSuSiE fit." ] }, { "cell_type": "markdown", + "id": "a97298ac", "metadata": {}, "source": [ "**Timing**: TBD" @@ -72,7 +78,10 @@ }, { "cell_type": "code", + "execution_count": null, + "id": "69d451d9", "metadata": {}, + "outputs": [], "source": [ "sos run pipeline/mnm_regression.ipynb qtl_dataset_construct+fsusie \\\n", " --name protocol_example \\\n", @@ -84,12 +93,11 @@ " --region-name ENSG00000130538 \\\n", " --transpose-covariates \\\n", " -j 1" - ], - "execution_count": null, - "outputs": [] + ] }, { "cell_type": "markdown", + "id": "4b322c2e", "metadata": {}, "source": [ "The command is expected to create:\n", @@ -111,6 +119,7 @@ }, { "cell_type": "markdown", + "id": "541b690f", "metadata": {}, "source": [ "### Command reference\n", @@ -120,15 +129,17 @@ }, { "cell_type": "code", + "execution_count": null, + "id": "5fa05e4c", "metadata": {}, + "outputs": [], "source": [ "sos run pipeline/mnm_regression.ipynb -h" - ], - "execution_count": null, - "outputs": [] + ] }, { "cell_type": "markdown", + "id": "6526a158", "metadata": {}, "source": [ "## Results and interpretation\n", @@ -147,7 +158,10 @@ }, { "cell_type": "code", + "execution_count": null, + "id": "9dbc00ad", "metadata": {}, + "outputs": [], "source": [ "suppressPackageStartupMessages(library(pecotmr))\n", "\n", @@ -167,12 +181,11 @@ " names(top_loci)\n", ")\n", "head(top_loci[, keep, drop = FALSE], 8)" - ], - "execution_count": null, - "outputs": [] + ] }, { "cell_type": "markdown", + "id": "4efe1430", "metadata": {}, "source": [ "Interpret variants by PIP first, then use credible-set membership and purity to judge localization. A high-PIP variant in a compact, high-purity credible set is more strongly localized than a similarly ranked variant in a large or low-purity set. Lack of a retained credible set in a toy dataset is not evidence that the locus has no functional effect; small sample size, weak signal, phenotype definition and LD can all prevent localization." @@ -180,6 +193,7 @@ }, { "cell_type": "markdown", + "id": "556aee2e", "metadata": {}, "source": [ "## Limitations and common pitfalls\n", @@ -194,6 +208,7 @@ }, { "cell_type": "markdown", + "id": "0d0896c7", "metadata": {}, "source": [ "## Next steps\n", @@ -223,4 +238,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/mnm_analysis/univariate_fine_mapping_twas_vignette.ipynb b/code/SoS/mnm_analysis/univariate_fine_mapping_twas_vignette.ipynb index a05099784..2879325a9 100644 --- a/code/SoS/mnm_analysis/univariate_fine_mapping_twas_vignette.ipynb +++ b/code/SoS/mnm_analysis/univariate_fine_mapping_twas_vignette.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "26657285", "metadata": {}, "source": [ "# Univariate fine-mapping and TWAS with SuSiE\n", @@ -10,6 +11,7 @@ }, { "cell_type": "markdown", + "id": "8cc62ae9", "metadata": {}, "source": [ "## Learning goals\n", @@ -22,10 +24,11 @@ }, { "cell_type": "markdown", + "id": "e75a82c7", "metadata": {}, "source": [ "## Background and method\n", - "`qtl_dataset_construct+susie_twas` in [`mnm_regression.ipynb`](https://statfungen.github.io/xqtl-protocol/code/mnm_analysis/mnm_methods/mnm_regression.html) builds a `QtlDataset` from genotype, phenotype, and covariate files. For each region, it residualizes genotype and phenotype matrices, fits SuSiE, and trains ten TWAS prediction models (`mrash`, `susie`, `susie_inf`, `enet`, `lasso`, `mcp`, `scad`, `l0learn`, `bayes_r`, and `bayes_c`) plus a cross-validation-based ensemble.\n", + "`qtl_dataset_construct+susie_twas` in [`mnm_regression.ipynb`](https://statfungen.github.io/xqtl-protocol/mnm-regression) builds a `QtlDataset` from genotype, phenotype, and covariate files. For each region, it residualizes genotype and phenotype matrices, fits SuSiE, and trains ten TWAS prediction models (`mrash`, `susie`, `susie_inf`, `enet`, `lasso`, `mcp`, `scad`, `l0learn`, `bayes_r`, and `bayes_c`) plus a cross-validation-based ensemble.\n", "\n", "SuSiE represents the regional effect as a sum of single effects and reports posterior inclusion probabilities (PIPs) and credible sets. In this example, the maximum number of effects is `L = 5`. TWAS weights solve a different problem: they predict the molecular trait from regional genotypes. Fine-mapping evidence and predictive performance should therefore be interpreted separately.\n", "\n", @@ -34,6 +37,7 @@ }, { "cell_type": "markdown", + "id": "eb9a5660", "metadata": {}, "source": [ "## Worked example\n", @@ -54,9 +58,10 @@ }, { "cell_type": "markdown", + "id": "548dbfbe", "metadata": {}, "source": [ - "### [Run univariate fine-mapping and estimate TWAS weights](https://statfungen.github.io/xqtl-protocol/code/mnm_analysis/mnm_methods/mnm_regression.html)\n", + "### [Run univariate fine-mapping and estimate TWAS weights](https://statfungen.github.io/xqtl-protocol/mnm-regression)\n", "\n", "This command constructs the analysis dataset, fits SuSiE separately in each context, and estimates TWAS weights. `--save-data` retains the residualized matrices for analyses that reuse the fitted dataset." ] @@ -64,6 +69,7 @@ { "cell_type": "code", "execution_count": null, + "id": "63f99b75", "metadata": {}, "outputs": [], "source": [ @@ -82,6 +88,7 @@ }, { "cell_type": "markdown", + "id": "14ec3fc5", "metadata": {}, "source": [ "The worked example produces:\n", @@ -93,9 +100,10 @@ }, { "cell_type": "markdown", + "id": "3bc4e429", "metadata": {}, "source": [ - "### Optional: [adapt the analysis to peak-based traits](https://statfungen.github.io/xqtl-protocol/code/mnm_analysis/mnm_methods/mnm_regression.html)\n", + "### Optional: [adapt the analysis to peak-based traits](https://statfungen.github.io/xqtl-protocol/mnm-regression)\n", "\n", "Peak- or CpG-based traits use the same fine-mapping workflow, but first require a phenotype manifest and biologically appropriate association windows. For peak data, the helper below splits a multi-sample BED into indexed region files and assigns each feature to its smallest containing extended TAD. Replace the placeholders with your peak BED, extended-TAD reference, and output directory, then use the generated manifest and windows in the main command." ] @@ -103,6 +111,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4c3498ab", "metadata": {}, "outputs": [], "source": [ @@ -119,6 +128,7 @@ }, { "cell_type": "markdown", + "id": "6ed5ee1c", "metadata": {}, "source": [ "### Parameters used in the example\n", @@ -137,6 +147,7 @@ }, { "cell_type": "markdown", + "id": "474a8e8c", "metadata": {}, "source": [ "### Command reference" @@ -145,6 +156,7 @@ { "cell_type": "code", "execution_count": null, + "id": "7230fd6e", "metadata": {}, "outputs": [], "source": [ @@ -153,6 +165,7 @@ }, { "cell_type": "markdown", + "id": "9d7232e8", "metadata": {}, "source": [ "## Results and interpretation" @@ -160,6 +173,7 @@ }, { "cell_type": "markdown", + "id": "619033ba", "metadata": {}, "source": [ "### Fine-mapping result\n", @@ -180,6 +194,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6c5ff6fa", "metadata": { "kernel": "R", "tags": [] @@ -212,6 +227,7 @@ }, { "cell_type": "markdown", + "id": "b5640656", "metadata": {}, "source": [ "The largest PIPs in context 1 are approximately 0.03, every displayed candidate-set label has purity 0, and SuSiE retains no credible set after filtering. This toy example therefore does not support a localized fine-mapped signal. It demonstrates how to inspect PIPs and credible-set diagnostics, not a biological discovery." @@ -219,6 +235,7 @@ }, { "cell_type": "markdown", + "id": "8697cf36", "metadata": {}, "source": [ "### TWAS prediction result\n", @@ -240,6 +257,7 @@ { "cell_type": "code", "execution_count": null, + "id": "0adc3e18", "metadata": {}, "outputs": [], "source": [ @@ -280,6 +298,7 @@ }, { "cell_type": "markdown", + "id": "8ab1d23c", "metadata": {}, "source": [ "For context 1, the SuSiE predictor has cross-validated R2 = 0.960. The ensemble assigns approximately 0.955 of its coefficient weight to `bayes_r` and 0.045 to `enet`. These values show how to assess prediction performance and ensemble composition. Because the example contains only 49 samples, they should not be treated as stable estimates of out-of-sample accuracy.\n", @@ -289,6 +308,7 @@ }, { "cell_type": "markdown", + "id": "93b4ed5c", "metadata": {}, "source": [ "## Limitations and common pitfalls\n", @@ -307,10 +327,11 @@ }, { "cell_type": "markdown", + "id": "ea73d5d2", "metadata": {}, "source": [ "## Next steps\n", - "Use the [fine-mapping and TWAS mini-protocol](https://statfungen.github.io/xqtl-protocol/mnm_miniprotocol.html) to choose among univariate, multivariate, functional, multi-gene, and summary-statistic routes. Refer to [`mnm_regression.ipynb`](https://statfungen.github.io/xqtl-protocol/code/mnm_analysis/mnm_methods/mnm_regression.html) for the complete command interface and parameter defaults.\n", + "Use the [fine-mapping and TWAS mini-protocol](https://statfungen.github.io/xqtl-protocol/mnm-miniprotocol) to choose among univariate, multivariate, functional, multi-gene, and summary-statistic routes. Refer to [`mnm_regression.ipynb`](https://statfungen.github.io/xqtl-protocol/mnm-regression) for the complete command interface and parameter defaults.\n", "\n", "After validating credible sets and cross-validated prediction performance, the TWAS weights can enter the GWAS-integration workflows. Fine-mapping PIPs and TWAS associations answer different questions and should not be treated as interchangeable evidence." ] @@ -330,5 +351,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/molecular_phenotypes/QC/apa_impute.ipynb b/code/SoS/molecular_phenotypes/QC/apa_impute.ipynb index 331187593..b9588c20d 100644 --- a/code/SoS/molecular_phenotypes/QC/apa_impute.ipynb +++ b/code/SoS/molecular_phenotypes/QC/apa_impute.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "0edb1745", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "c9996958", "metadata": { "kernel": "SoS" }, @@ -28,6 +30,7 @@ }, { "cell_type": "markdown", + "id": "98258e38", "metadata": { "kernel": "SoS" }, @@ -50,6 +53,7 @@ }, { "cell_type": "markdown", + "id": "b8b7d3e0", "metadata": { "kernel": "SoS" }, @@ -67,6 +71,7 @@ }, { "cell_type": "markdown", + "id": "8e31e5b3", "metadata": { "kernel": "SoS" }, @@ -78,6 +83,7 @@ }, { "cell_type": "markdown", + "id": "a96867d8", "metadata": { "kernel": "SoS" }, @@ -87,6 +93,7 @@ }, { "cell_type": "markdown", + "id": "af499ef3", "metadata": { "kernel": "SoS" }, @@ -97,6 +104,7 @@ { "cell_type": "code", "execution_count": null, + "id": "847d873e", "metadata": { "kernel": "Bash" }, @@ -109,6 +117,7 @@ }, { "cell_type": "markdown", + "id": "f565feae", "metadata": { "kernel": "SoS" }, @@ -120,6 +129,7 @@ }, { "cell_type": "markdown", + "id": "874ec1a8", "metadata": { "kernel": "SoS" }, @@ -130,6 +140,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e2bf24e6", "metadata": { "kernel": "Bash" }, @@ -143,6 +154,7 @@ }, { "cell_type": "markdown", + "id": "1f600ef6", "metadata": { "kernel": "SoS" }, @@ -153,6 +165,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a61fce1f", "metadata": { "kernel": "Bash" }, @@ -163,6 +176,7 @@ }, { "cell_type": "markdown", + "id": "dfebfdb7", "metadata": { "kernel": "SoS" }, @@ -211,6 +225,7 @@ }, { "cell_type": "markdown", + "id": "884ee521", "metadata": { "kernel": "SoS" }, @@ -221,6 +236,7 @@ { "cell_type": "code", "execution_count": null, + "id": "dad88283", "metadata": { "kernel": "SoS" }, @@ -240,6 +256,7 @@ { "cell_type": "code", "execution_count": null, + "id": "894a0595", "metadata": { "kernel": "SoS" }, @@ -259,6 +276,7 @@ { "cell_type": "code", "execution_count": null, + "id": "8f6ff06a", "metadata": { "kernel": "SoS" }, @@ -280,6 +298,7 @@ { "cell_type": "code", "execution_count": null, + "id": "daf62cf7", "metadata": { "kernel": "SoS" }, @@ -299,6 +318,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2504d53f", "metadata": { "kernel": "SoS" }, @@ -358,5 +378,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/molecular_phenotypes/QC/bulk_expression_QC.ipynb b/code/SoS/molecular_phenotypes/QC/bulk_expression_QC.ipynb index ba4c47aae..788ec28b3 100644 --- a/code/SoS/molecular_phenotypes/QC/bulk_expression_QC.ipynb +++ b/code/SoS/molecular_phenotypes/QC/bulk_expression_QC.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "8eddf046", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "edb0b4fa", "metadata": { "kernel": "SoS" }, @@ -35,6 +37,7 @@ }, { "cell_type": "markdown", + "id": "e6c02cff", "metadata": { "kernel": "SoS" }, @@ -75,6 +78,7 @@ }, { "cell_type": "markdown", + "id": "07250385", "metadata": { "kernel": "SoS" }, @@ -101,6 +105,7 @@ }, { "cell_type": "markdown", + "id": "fc59a7c3", "metadata": { "kernel": "SoS" }, @@ -110,6 +115,7 @@ }, { "cell_type": "markdown", + "id": "570c3cda", "metadata": { "kernel": "SoS" }, @@ -120,6 +126,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9c5830fd", "metadata": { "kernel": "Bash" }, @@ -133,6 +140,7 @@ }, { "cell_type": "markdown", + "id": "ef16d0e4", "metadata": { "kernel": "SoS" }, @@ -143,6 +151,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f53b0795", "metadata": { "kernel": "SoS" }, @@ -153,6 +162,7 @@ }, { "cell_type": "markdown", + "id": "b1f36707", "metadata": { "kernel": "SoS" }, @@ -207,6 +217,7 @@ }, { "cell_type": "markdown", + "id": "14408e99", "metadata": { "kernel": "SoS" }, @@ -217,6 +228,7 @@ { "cell_type": "code", "execution_count": null, + "id": "d1b6c7da", "metadata": { "kernel": "SoS" }, @@ -247,6 +259,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4c031846", "metadata": { "kernel": "SoS" }, @@ -270,6 +283,7 @@ }, { "cell_type": "markdown", + "id": "3b66233d", "metadata": { "kernel": "SoS" }, @@ -280,6 +294,7 @@ { "cell_type": "code", "execution_count": null, + "id": "d453bd58", "metadata": { "kernel": "SoS" }, @@ -310,6 +325,7 @@ }, { "cell_type": "markdown", + "id": "17a7c401", "metadata": { "kernel": "SoS" }, @@ -320,6 +336,7 @@ { "cell_type": "code", "execution_count": null, + "id": "cffb0710", "metadata": { "kernel": "SoS" }, @@ -379,5 +396,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/molecular_phenotypes/QC/bulk_expression_normalization.ipynb b/code/SoS/molecular_phenotypes/QC/bulk_expression_normalization.ipynb index f30e4bd14..49312b5d1 100644 --- a/code/SoS/molecular_phenotypes/QC/bulk_expression_normalization.ipynb +++ b/code/SoS/molecular_phenotypes/QC/bulk_expression_normalization.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "cf04b5b1", "metadata": { "kernel": "SoS", "tags": [] @@ -14,6 +15,7 @@ }, { "cell_type": "markdown", + "id": "5b6ec6ed", "metadata": { "kernel": "SoS" }, @@ -31,6 +33,7 @@ }, { "cell_type": "markdown", + "id": "7b0834fe", "metadata": { "kernel": "SoS" }, @@ -74,6 +77,7 @@ }, { "cell_type": "markdown", + "id": "fe73c2fd", "metadata": { "kernel": "SoS" }, @@ -93,6 +97,7 @@ }, { "cell_type": "markdown", + "id": "c3e24d70", "metadata": { "kernel": "SoS" }, @@ -102,6 +107,7 @@ }, { "cell_type": "markdown", + "id": "f7b5479a", "metadata": { "kernel": "SoS" }, @@ -113,6 +119,7 @@ }, { "cell_type": "markdown", + "id": "3b3ada79", "metadata": { "kernel": "SoS", "tags": [] @@ -124,6 +131,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b5a58b92", "metadata": { "kernel": "Bash" }, @@ -140,6 +148,7 @@ }, { "cell_type": "markdown", + "id": "df8ca4b0", "metadata": { "kernel": "SoS" }, @@ -150,6 +159,7 @@ { "cell_type": "code", "execution_count": 4, + "id": "625135aa", "metadata": { "kernel": "Bash" }, @@ -214,6 +224,7 @@ }, { "cell_type": "markdown", + "id": "fbe06b1f", "metadata": { "kernel": "SoS" }, @@ -272,6 +283,7 @@ }, { "cell_type": "markdown", + "id": "8613d694", "metadata": { "kernel": "SoS" }, @@ -284,6 +296,7 @@ { "cell_type": "code", "execution_count": 9, + "id": "6021db9b", "metadata": { "kernel": "SoS" }, @@ -324,6 +337,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a9b66d99", "metadata": { "kernel": "SoS" }, @@ -386,5 +400,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 + "nbformat_minor": 5 } diff --git a/code/SoS/molecular_phenotypes/QC/pseudobulk_preprocessing.ipynb b/code/SoS/molecular_phenotypes/QC/pseudobulk_preprocessing.ipynb index b05c16cad..e447e4f25 100644 --- a/code/SoS/molecular_phenotypes/QC/pseudobulk_preprocessing.ipynb +++ b/code/SoS/molecular_phenotypes/QC/pseudobulk_preprocessing.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "f4201fd0", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "4f57014f", "metadata": { "kernel": "SoS" }, @@ -33,6 +35,7 @@ }, { "cell_type": "markdown", + "id": "d03e2486", "metadata": {}, "source": [ "**When to run it.** After single-nuclei data have been clustered and cell types assigned, and before\n", @@ -41,6 +44,7 @@ }, { "cell_type": "markdown", + "id": "1d7ff263", "metadata": { "kernel": "SoS" }, @@ -106,6 +110,7 @@ }, { "cell_type": "markdown", + "id": "f006a8b5", "metadata": { "kernel": "SoS" }, @@ -149,6 +154,7 @@ }, { "cell_type": "markdown", + "id": "56b6e1aa", "metadata": {}, "source": [ "## Minimal Working Example" @@ -156,6 +162,7 @@ }, { "cell_type": "markdown", + "id": "2c39759c", "metadata": {}, "source": [ "#### Step 1. Pseudobulk count matrix generation\n", @@ -165,6 +172,7 @@ }, { "cell_type": "markdown", + "id": "52b61e46", "metadata": { "kernel": "SoS" }, @@ -175,6 +183,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4c17890b", "metadata": { "kernel": "Bash" }, @@ -188,6 +197,7 @@ }, { "cell_type": "markdown", + "id": "046065ef", "metadata": { "kernel": "SoS" }, @@ -199,6 +209,7 @@ }, { "cell_type": "markdown", + "id": "d36ffe81", "metadata": { "kernel": "SoS" }, @@ -209,6 +220,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5ee5184a", "metadata": { "kernel": "Bash" }, @@ -222,6 +234,7 @@ }, { "cell_type": "markdown", + "id": "f2ee776c", "metadata": { "kernel": "SoS" }, @@ -233,6 +246,7 @@ }, { "cell_type": "markdown", + "id": "8bcf068f", "metadata": { "kernel": "SoS" }, @@ -243,6 +257,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2bd2b813", "metadata": { "kernel": "Bash" }, @@ -257,6 +272,7 @@ }, { "cell_type": "markdown", + "id": "6d20a3ca", "metadata": { "kernel": "SoS" }, @@ -268,6 +284,7 @@ }, { "cell_type": "markdown", + "id": "20bff092", "metadata": { "kernel": "SoS" }, @@ -278,6 +295,7 @@ { "cell_type": "code", "execution_count": null, + "id": "50b6cb38", "metadata": { "kernel": "Bash" }, @@ -291,6 +309,7 @@ }, { "cell_type": "markdown", + "id": "b43aff49", "metadata": { "kernel": "SoS" }, @@ -301,6 +320,7 @@ { "cell_type": "code", "execution_count": null, + "id": "bada3815", "metadata": {}, "outputs": [], "source": [ @@ -309,6 +329,7 @@ }, { "cell_type": "markdown", + "id": "0c0da1e7", "metadata": {}, "source": [ "```\n", @@ -373,6 +394,7 @@ }, { "cell_type": "markdown", + "id": "aea41410", "metadata": {}, "source": [ "## Workflow implementation\n", @@ -381,6 +403,7 @@ }, { "cell_type": "markdown", + "id": "deaceaa0", "metadata": { "kernel": "SoS" }, @@ -391,6 +414,7 @@ { "cell_type": "code", "execution_count": null, + "id": "cf1ee10b", "metadata": { "kernel": "SoS" }, @@ -412,6 +436,7 @@ }, { "cell_type": "markdown", + "id": "4831282e", "metadata": {}, "source": [ "```\n", @@ -465,6 +490,7 @@ }, { "cell_type": "markdown", + "id": "14eff80a", "metadata": {}, "source": [ "### `pseudobulk_counts`" @@ -473,6 +499,7 @@ { "cell_type": "code", "execution_count": null, + "id": "69ccf3fa", "metadata": { "kernel": "SoS" }, @@ -503,6 +530,7 @@ }, { "cell_type": "markdown", + "id": "e0ac3248", "metadata": { "kernel": "SoS" }, @@ -513,6 +541,7 @@ { "cell_type": "code", "execution_count": null, + "id": "89ebeafa", "metadata": { "kernel": "SoS" }, @@ -541,6 +570,7 @@ }, { "cell_type": "markdown", + "id": "09ad65ba", "metadata": { "kernel": "SoS" }, @@ -551,6 +581,7 @@ { "cell_type": "code", "execution_count": null, + "id": "ead3bae6", "metadata": { "kernel": "SoS" }, @@ -603,6 +634,7 @@ }, { "cell_type": "markdown", + "id": "12d13898", "metadata": { "kernel": "SoS" }, @@ -613,6 +645,7 @@ { "cell_type": "code", "execution_count": null, + "id": "01f35b3c", "metadata": { "kernel": "SoS" }, @@ -670,5 +703,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/molecular_phenotypes/QC/splicing_normalization.ipynb b/code/SoS/molecular_phenotypes/QC/splicing_normalization.ipynb index da4fcb443..3a9ee8cac 100644 --- a/code/SoS/molecular_phenotypes/QC/splicing_normalization.ipynb +++ b/code/SoS/molecular_phenotypes/QC/splicing_normalization.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "0fde36f0", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "cecaee36", "metadata": { "kernel": "SoS" }, @@ -37,6 +39,7 @@ }, { "cell_type": "markdown", + "id": "389ccf6d", "metadata": { "kernel": "SoS" }, @@ -63,6 +66,7 @@ }, { "cell_type": "markdown", + "id": "ecedba0c", "metadata": { "kernel": "SoS" }, @@ -100,6 +104,7 @@ }, { "cell_type": "markdown", + "id": "37c6ee62", "metadata": { "kernel": "SoS" }, @@ -109,6 +114,7 @@ }, { "cell_type": "markdown", + "id": "60b0f511", "metadata": { "kernel": "SoS" }, @@ -122,6 +128,7 @@ }, { "cell_type": "markdown", + "id": "41a31b78", "metadata": { "kernel": "SoS" }, @@ -132,6 +139,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b8007645", "metadata": { "kernel": "SoS" }, @@ -146,6 +154,7 @@ }, { "cell_type": "markdown", + "id": "bcacab91", "metadata": { "kernel": "SoS" }, @@ -157,6 +166,7 @@ }, { "cell_type": "markdown", + "id": "d3f99143", "metadata": { "kernel": "SoS" }, @@ -167,6 +177,7 @@ { "cell_type": "code", "execution_count": null, + "id": "1336aa19", "metadata": { "kernel": "SoS" }, @@ -181,6 +192,7 @@ }, { "cell_type": "markdown", + "id": "a344a50c", "metadata": { "kernel": "SoS" }, @@ -194,6 +206,7 @@ }, { "cell_type": "markdown", + "id": "6ee6dda7", "metadata": { "kernel": "SoS" }, @@ -204,6 +217,7 @@ { "cell_type": "code", "execution_count": null, + "id": "bd606cc0", "metadata": { "kernel": "Bash" }, @@ -217,6 +231,7 @@ }, { "cell_type": "markdown", + "id": "3c44618b", "metadata": { "kernel": "SoS" }, @@ -234,6 +249,7 @@ }, { "cell_type": "markdown", + "id": "02ebd688", "metadata": { "kernel": "SoS" }, @@ -243,6 +259,7 @@ }, { "cell_type": "markdown", + "id": "39a656c6", "metadata": { "kernel": "SoS" }, @@ -252,6 +269,7 @@ }, { "cell_type": "markdown", + "id": "187d0f7f", "metadata": { "kernel": "SoS" }, @@ -262,6 +280,7 @@ { "cell_type": "code", "execution_count": null, + "id": "8ff0cd13", "metadata": { "kernel": "SoS" }, @@ -276,6 +295,7 @@ }, { "cell_type": "markdown", + "id": "17306f6a", "metadata": { "kernel": "SoS" }, @@ -285,6 +305,7 @@ }, { "cell_type": "markdown", + "id": "dafe25fa", "metadata": { "kernel": "SoS" }, @@ -295,6 +316,7 @@ { "cell_type": "code", "execution_count": null, + "id": "490eb73f", "metadata": { "kernel": "SoS" }, @@ -310,6 +332,7 @@ }, { "cell_type": "markdown", + "id": "c1a41d88", "metadata": { "kernel": "SoS" }, @@ -319,6 +342,7 @@ }, { "cell_type": "markdown", + "id": "50fcc69f", "metadata": { "kernel": "SoS" }, @@ -329,6 +353,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5e70d9d0", "metadata": { "kernel": "SoS" }, @@ -342,6 +367,7 @@ }, { "cell_type": "markdown", + "id": "a19b16c3", "metadata": { "kernel": "SoS" }, @@ -351,6 +377,7 @@ }, { "cell_type": "markdown", + "id": "a3b8a816", "metadata": { "kernel": "SoS" }, @@ -361,6 +388,7 @@ { "cell_type": "code", "execution_count": null, + "id": "926c6990", "metadata": { "kernel": "SoS" }, @@ -375,6 +403,7 @@ }, { "cell_type": "markdown", + "id": "ea081a6e", "metadata": { "kernel": "SoS" }, @@ -385,6 +414,7 @@ { "cell_type": "code", "execution_count": null, + "id": "ed13e02e", "metadata": { "kernel": "SoS" }, @@ -395,6 +425,7 @@ }, { "cell_type": "markdown", + "id": "f71cdb89", "metadata": { "kernel": "SoS" }, @@ -468,6 +499,7 @@ }, { "cell_type": "markdown", + "id": "030a590e", "metadata": { "kernel": "SoS" }, @@ -478,6 +510,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b033db9d", "metadata": { "kernel": "SoS" }, @@ -507,6 +540,7 @@ { "cell_type": "code", "execution_count": null, + "id": "250532e1", "metadata": { "kernel": "SoS" }, @@ -528,6 +562,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e0bc751b", "metadata": { "kernel": "SoS" }, @@ -550,6 +585,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2039e104", "metadata": { "kernel": "SoS" }, @@ -575,6 +611,7 @@ { "cell_type": "code", "execution_count": null, + "id": "adcc093d", "metadata": { "kernel": "SoS" }, @@ -602,6 +639,7 @@ { "cell_type": "code", "execution_count": null, + "id": "492da6c2", "metadata": { "kernel": "SoS" }, @@ -689,5 +727,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 + "nbformat_minor": 5 } diff --git a/code/SoS/molecular_phenotypes/apa.ipynb b/code/SoS/molecular_phenotypes/apa.ipynb index aeed27f24..c9c4b1ea6 100644 --- a/code/SoS/molecular_phenotypes/apa.ipynb +++ b/code/SoS/molecular_phenotypes/apa.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "0d2421da", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "fea1155b", "metadata": { "kernel": "SoS" }, @@ -24,19 +26,21 @@ }, { "cell_type": "markdown", + "id": "688f91e7", "metadata": { "kernel": "SoS" }, "source": [ "## Overview\n", "\n", - "[APA calling](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/apa_calling.html) derives 3′UTR regions, converts transcriptome BAM files to per-base coverage, and runs DaPars2 to estimate percentage of distal polyadenylation site usage (PDUI). [APA imputation and QC](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/QC/apa_impute.html) imputes missing PDUI values, quantile-normalizes the matrix, and can replace internal sample identifiers with analysis identifiers.\n", + "[APA calling](https://statfungen.github.io/xqtl-protocol/apa-calling) derives 3′UTR regions, converts transcriptome BAM files to per-base coverage, and runs DaPars2 to estimate percentage of distal polyadenylation site usage (PDUI). [APA imputation and QC](https://statfungen.github.io/xqtl-protocol/apa-impute) imputes missing PDUI values, quantile-normalizes the matrix, and can replace internal sample identifiers with analysis identifiers.\n", "\n", "The full calling route requires transcriptome BAM files, which are not included with the example data. A precomputed chromosome-22 DaPars result is provided so that the imputation and renaming routes can be run independently." ] }, { "cell_type": "markdown", + "id": "01615588", "metadata": { "kernel": "SoS" }, @@ -57,11 +61,12 @@ }, { "cell_type": "markdown", + "id": "5d14a380", "metadata": { "kernel": "SoS" }, "source": [ - "### [1. Build the 3′UTR reference](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/apa_calling.html)\n", + "### [1. Build the 3′UTR reference](https://statfungen.github.io/xqtl-protocol/apa-calling)\n", "\n", "**What it does:** `UTR_reference` extracts transcript-level 3′UTR intervals from the reference GTF for DaPars2." ] @@ -69,6 +74,7 @@ { "cell_type": "code", "execution_count": null, + "id": "28e2337d", "metadata": { "kernel": "Bash" }, @@ -81,11 +87,12 @@ }, { "cell_type": "markdown", + "id": "7d4d3b74", "metadata": { "kernel": "SoS" }, "source": [ - "### [2. Generate coverage and read-depth files](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/apa_calling.html)\n", + "### [2. Generate coverage and read-depth files](https://statfungen.github.io/xqtl-protocol/apa-calling)\n", "\n", "**What it does:** `bam2tools` converts each transcriptome BAM into a per-base WIG file and a matching read-depth summary." ] @@ -93,6 +100,7 @@ { "cell_type": "code", "execution_count": null, + "id": "05f4eb0f", "metadata": { "kernel": "Bash" }, @@ -105,11 +113,12 @@ }, { "cell_type": "markdown", + "id": "518f4bfe", "metadata": { "kernel": "SoS" }, "source": [ - "### [3. Build the DaPars2 configuration](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/apa_calling.html)\n", + "### [3. Build the DaPars2 configuration](https://statfungen.github.io/xqtl-protocol/apa-calling)\n", "\n", "**What it does:** `APAconfig` creates the sample mapping and DaPars2 configuration files from the coverage directory and 3′UTR annotation." ] @@ -117,6 +126,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b35c6850", "metadata": { "kernel": "Bash" }, @@ -130,11 +140,12 @@ }, { "cell_type": "markdown", + "id": "cd0fdfb1", "metadata": { "kernel": "SoS" }, "source": [ - "### [4. Estimate chromosome-level PDUI](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/apa_calling.html)\n", + "### [4. Estimate chromosome-level PDUI](https://statfungen.github.io/xqtl-protocol/apa-calling)\n", "\n", "**What it does:** `APAmain` runs DaPars2 for chromosome 22 and writes the unprocessed PDUI result used by the QC module." ] @@ -142,6 +153,7 @@ { "cell_type": "code", "execution_count": null, + "id": "ee933d09", "metadata": { "kernel": "Bash" }, @@ -156,11 +168,12 @@ }, { "cell_type": "markdown", + "id": "1cabdccb", "metadata": { "kernel": "SoS" }, "source": [ - "### [5. Impute and normalize PDUI](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/QC/apa_impute.html)\n", + "### [5. Impute and normalize PDUI](https://statfungen.github.io/xqtl-protocol/apa-impute)\n", "\n", "**What it does:** `APAimpute` imputes missing chromosome-level PDUI values and applies quantile normalization." ] @@ -168,6 +181,7 @@ { "cell_type": "code", "execution_count": null, + "id": "12a3a872", "metadata": { "kernel": "Bash" }, @@ -180,11 +194,12 @@ }, { "cell_type": "markdown", + "id": "945c22e4", "metadata": { "kernel": "SoS" }, "source": [ - "### [6. Rename samples and assemble the final matrix](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/QC/apa_impute.html)\n", + "### [6. Rename samples and assemble the final matrix](https://statfungen.github.io/xqtl-protocol/apa-impute)\n", "\n", "**What it does:** `APArename` maps sample identifiers, combines requested chromosomes, and bgzip-indexes the final APA phenotype matrix." ] @@ -192,6 +207,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a25f633d", "metadata": { "kernel": "Bash" }, @@ -205,6 +221,7 @@ }, { "cell_type": "markdown", + "id": "55f6d9f3", "metadata": { "kernel": "SoS" }, @@ -223,6 +240,7 @@ }, { "cell_type": "markdown", + "id": "120f4dfc", "metadata": { "kernel": "SoS" }, @@ -236,6 +254,7 @@ }, { "cell_type": "markdown", + "id": "b40d30a4", "metadata": { "kernel": "SoS" }, @@ -246,6 +265,7 @@ { "cell_type": "code", "execution_count": null, + "id": "1c358bfd", "metadata": { "kernel": "SoS" }, @@ -257,6 +277,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f9e52cdc", "metadata": { "kernel": "SoS" }, @@ -301,5 +322,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/molecular_phenotypes/bulk_expression.ipynb b/code/SoS/molecular_phenotypes/bulk_expression.ipynb index 20d72ba96..583e13994 100644 --- a/code/SoS/molecular_phenotypes/bulk_expression.ipynb +++ b/code/SoS/molecular_phenotypes/bulk_expression.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "0e4fed22", "metadata": { "kernel": "SoS", "tags": [] @@ -14,6 +15,7 @@ }, { "cell_type": "markdown", + "id": "b7aab672", "metadata": { "kernel": "SoS", "tags": [] @@ -26,6 +28,7 @@ }, { "cell_type": "markdown", + "id": "b3c39a6e", "metadata": { "kernel": "SoS", "tags": [] @@ -33,13 +36,14 @@ "source": [ "## Overview\n", "\n", - "This mini-protocol walks through bulk RNA-seq expression processing. [`RNA_calling.ipynb`](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/RNA_calling.html) performs read-level quality assessment, optional adapter trimming, STAR alignment, and gene- or transcript-level quantification. [`bulk_expression_QC.ipynb`](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/QC/bulk_expression_QC.html) filters low-expression features and detects sample outliers, and [`bulk_expression_normalization.ipynb`](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/QC/bulk_expression_normalization.html) produces normalized expression matrices for downstream xQTL analysis.\n", + "This mini-protocol walks through bulk RNA-seq expression processing. [`RNA_calling.ipynb`](https://statfungen.github.io/xqtl-protocol/rna-calling) performs read-level quality assessment, optional adapter trimming, STAR alignment, and gene- or transcript-level quantification. [`bulk_expression_QC.ipynb`](https://statfungen.github.io/xqtl-protocol/bulk-expression-qc) filters low-expression features and detects sample outliers, and [`bulk_expression_normalization.ipynb`](https://statfungen.github.io/xqtl-protocol/bulk-expression-normalization) produces normalized expression matrices for downstream xQTL analysis.\n", "\n", "The commands form selectable routes rather than one mandatory chain. Gene-level RNA-SeQC and transcript-level RSEM quantification are alternatives after alignment. The bundled matrix-QC example starts from existing count and TPM matrices, so steps 6–7 can be run independently of the toy FASTQ example in steps 1–5.\n" ] }, { "cell_type": "markdown", + "id": "b415b044", "metadata": { "kernel": "SoS", "tags": [] @@ -63,12 +67,13 @@ }, { "cell_type": "markdown", + "id": "0f8f4385", "metadata": { "kernel": "SoS", "tags": [] }, "source": [ - "### 1. [Assess FASTQ read quality](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/RNA_calling.html)\n", + "### 1. [Assess FASTQ read quality](https://statfungen.github.io/xqtl-protocol/rna-calling)\n", "\n", "**What it does:** Generate per-sample FastQC reports before alignment.\n" ] @@ -76,6 +81,7 @@ { "cell_type": "code", "execution_count": null, + "id": "8e4cc3e5", "metadata": { "kernel": "Bash" }, @@ -89,12 +95,13 @@ }, { "cell_type": "markdown", + "id": "39f4b56e", "metadata": { "kernel": "SoS", "tags": [] }, "source": [ - "### 2. [Trim sequencing adapters](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/RNA_calling.html)\n", + "### 2. [Trim sequencing adapters](https://statfungen.github.io/xqtl-protocol/rna-calling)\n", "\n", "**What it does:** Remove configured adapter sequences and write trimmed FASTQ files for alignment.\n" ] @@ -102,6 +109,7 @@ { "cell_type": "code", "execution_count": null, + "id": "c1c685f9", "metadata": { "kernel": "Bash" }, @@ -117,12 +125,13 @@ }, { "cell_type": "markdown", + "id": "9a495ba5", "metadata": { "kernel": "SoS", "tags": [] }, "source": [ - "### 3. [Align reads with STAR and run Picard QC](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/RNA_calling.html)\n", + "### 3. [Align reads with STAR and run Picard QC](https://statfungen.github.io/xqtl-protocol/rna-calling)\n", "\n", "**What it does:** Map reads to the reference genome and calculate alignment-level RNA-seq quality metrics.\n" ] @@ -130,6 +139,7 @@ { "cell_type": "code", "execution_count": null, + "id": "1e50d9bc", "metadata": { "kernel": "Bash" }, @@ -147,12 +157,13 @@ }, { "cell_type": "markdown", + "id": "c78aabdd", "metadata": { "kernel": "SoS", "tags": [] }, "source": [ - "### 4. [Quantify gene-level expression with RNA-SeQC](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/RNA_calling.html)\n", + "### 4. [Quantify gene-level expression with RNA-SeQC](https://statfungen.github.io/xqtl-protocol/rna-calling)\n", "\n", "**What it does:** Summarize aligned reads into gene-level expression measurements.\n" ] @@ -160,6 +171,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4170a44a", "metadata": { "kernel": "Bash" }, @@ -176,12 +188,13 @@ }, { "cell_type": "markdown", + "id": "6b2fbba0", "metadata": { "kernel": "SoS", "tags": [] }, "source": [ - "### 5. [Quantify transcript-level expression with RSEM](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/RNA_calling.html)\n", + "### 5. [Quantify transcript-level expression with RSEM](https://statfungen.github.io/xqtl-protocol/rna-calling)\n", "\n", "**What it does:** Estimate transcript- and gene-level abundance using the RSEM reference index.\n" ] @@ -189,6 +202,7 @@ { "cell_type": "code", "execution_count": null, + "id": "799d09ee", "metadata": { "kernel": "Bash" }, @@ -208,12 +222,13 @@ }, { "cell_type": "markdown", + "id": "36389c09", "metadata": { "kernel": "SoS", "tags": [] }, "source": [ - "### 6. [Perform cohort-level expression QC](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/QC/bulk_expression_QC.html)\n", + "### 6. [Perform cohort-level expression QC](https://statfungen.github.io/xqtl-protocol/bulk-expression-qc)\n", "\n", "**What it does:** Filter low-expression features and identify expression outlier samples across the cohort.\n" ] @@ -221,6 +236,7 @@ { "cell_type": "code", "execution_count": null, + "id": "eb39df2a", "metadata": { "kernel": "Bash" }, @@ -234,12 +250,13 @@ }, { "cell_type": "markdown", + "id": "ed2f19c1", "metadata": { "kernel": "SoS", "tags": [] }, "source": [ - "### 7. [Normalize the QC-passed expression matrices](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/QC/bulk_expression_normalization.html)\n", + "### 7. [Normalize the QC-passed expression matrices](https://statfungen.github.io/xqtl-protocol/bulk-expression-normalization)\n", "\n", "**What it does:** Normalize the retained count and TPM matrices and write association-ready expression phenotypes.\n" ] @@ -247,6 +264,7 @@ { "cell_type": "code", "execution_count": null, + "id": "600e9b13", "metadata": { "kernel": "Bash" }, @@ -262,6 +280,7 @@ }, { "cell_type": "markdown", + "id": "d4866a5a", "metadata": { "kernel": "SoS", "tags": [] @@ -283,6 +302,7 @@ }, { "cell_type": "markdown", + "id": "672f5098", "metadata": { "kernel": "SoS", "tags": [] @@ -297,6 +317,7 @@ }, { "cell_type": "markdown", + "id": "f92c5520", "metadata": { "kernel": "SoS", "tags": [] @@ -307,11 +328,11 @@ "\n", "\n", "**Figure 1A. Bulk RNA-seq quality-control D-statistic distribution.**\n" - ], - "id": "f92c5520" + ] }, { "cell_type": "markdown", + "id": "c0636815", "metadata": { "kernel": "SoS", "tags": [] @@ -322,11 +343,11 @@ "\n", "\n", "**Figure 1B. Bulk RNA-seq quality-control relative log-expression residuals.**\n" - ], - "id": "c0636815" + ] }, { "cell_type": "markdown", + "id": "39b0865e", "metadata": { "kernel": "SoS", "tags": [] @@ -337,11 +358,11 @@ "\n", "\n", "**Figure 1C. Bulk RNA-seq quality-control Mahalanobis-distance p-value clustering.**\n" - ], - "id": "39b0865e" + ] }, { "cell_type": "markdown", + "id": "1f9761e7", "metadata": { "kernel": "SoS", "tags": [] @@ -355,6 +376,7 @@ { "cell_type": "code", "execution_count": null, + "id": "104ff03a", "metadata": { "kernel": "Bash" }, @@ -395,4 +417,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/molecular_phenotypes/calling/RNA_calling.ipynb b/code/SoS/molecular_phenotypes/calling/RNA_calling.ipynb index da89cda77..800c7a31f 100644 --- a/code/SoS/molecular_phenotypes/calling/RNA_calling.ipynb +++ b/code/SoS/molecular_phenotypes/calling/RNA_calling.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "20d2845c", "metadata": { "kernel": "SoS", "tags": [] @@ -14,6 +15,7 @@ }, { "cell_type": "markdown", + "id": "db56d4e5", "metadata": { "kernel": "SoS" }, @@ -35,6 +37,7 @@ }, { "cell_type": "markdown", + "id": "3b55755b", "metadata": { "kernel": "SoS" }, @@ -93,6 +96,7 @@ }, { "cell_type": "markdown", + "id": "0c76c56e", "metadata": { "kernel": "SoS" }, @@ -125,6 +129,7 @@ }, { "cell_type": "markdown", + "id": "cdaf274d", "metadata": { "kernel": "SoS" }, @@ -136,6 +141,7 @@ }, { "cell_type": "markdown", + "id": "42f47505", "metadata": { "kernel": "SoS" }, @@ -147,6 +153,7 @@ }, { "cell_type": "markdown", + "id": "d2129848", "metadata": { "kernel": "SoS" }, @@ -157,6 +164,7 @@ { "cell_type": "code", "execution_count": null, + "id": "85b3fa04", "metadata": { "kernel": "SoS" }, @@ -170,6 +178,7 @@ }, { "cell_type": "markdown", + "id": "1433ef7f", "metadata": { "kernel": "SoS" }, @@ -179,6 +188,7 @@ }, { "cell_type": "markdown", + "id": "e24c1e1d", "metadata": { "kernel": "SoS", "tags": [] @@ -190,6 +200,7 @@ { "cell_type": "code", "execution_count": null, + "id": "999a3f68", "metadata": { "kernel": "Bash" }, @@ -203,6 +214,7 @@ }, { "cell_type": "markdown", + "id": "b3623458", "metadata": { "kernel": "SoS" }, @@ -214,6 +226,7 @@ }, { "cell_type": "markdown", + "id": "d26a52c0", "metadata": { "kernel": "SoS", "tags": [] @@ -225,6 +238,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5f3c675f", "metadata": { "kernel": "Bash" }, @@ -238,6 +252,7 @@ }, { "cell_type": "markdown", + "id": "598514c5", "metadata": { "kernel": "SoS" }, @@ -249,6 +264,7 @@ }, { "cell_type": "markdown", + "id": "03508f15", "metadata": { "kernel": "SoS" }, @@ -259,6 +275,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5ee52b49", "metadata": { "kernel": "SoS" }, @@ -272,6 +289,7 @@ }, { "cell_type": "markdown", + "id": "28f91200", "metadata": { "kernel": "SoS" }, @@ -283,6 +301,7 @@ }, { "cell_type": "markdown", + "id": "cfd84855", "metadata": { "kernel": "SoS", "tags": [] @@ -294,6 +313,7 @@ { "cell_type": "code", "execution_count": null, + "id": "20fb45c6", "metadata": { "kernel": "Bash" }, @@ -313,6 +333,7 @@ }, { "cell_type": "markdown", + "id": "bf432158", "metadata": { "kernel": "SoS" }, @@ -324,6 +345,7 @@ }, { "cell_type": "markdown", + "id": "3fe1a2eb", "metadata": { "kernel": "SoS" }, @@ -334,6 +356,7 @@ { "cell_type": "code", "execution_count": null, + "id": "09f7fabd", "metadata": { "kernel": "SoS" }, @@ -347,6 +370,7 @@ }, { "cell_type": "markdown", + "id": "7822502b", "metadata": { "kernel": "SoS" }, @@ -358,6 +382,7 @@ }, { "cell_type": "markdown", + "id": "9e1b4c7d", "metadata": { "kernel": "SoS" }, @@ -368,6 +393,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a405a71f", "metadata": { "kernel": "SoS" }, @@ -384,6 +410,7 @@ }, { "cell_type": "markdown", + "id": "980bfcb4", "metadata": { "kernel": "SoS" }, @@ -395,6 +422,7 @@ }, { "cell_type": "markdown", + "id": "474994af", "metadata": { "kernel": "SoS", "tags": [] @@ -406,6 +434,7 @@ { "cell_type": "code", "execution_count": null, + "id": "7b7d30d1", "metadata": { "kernel": "Bash" }, @@ -422,6 +451,7 @@ }, { "cell_type": "markdown", + "id": "9b222505", "metadata": { "kernel": "SoS" }, @@ -433,6 +463,7 @@ }, { "cell_type": "markdown", + "id": "8dc02a31", "metadata": { "kernel": "SoS", "tags": [] @@ -444,6 +475,7 @@ { "cell_type": "code", "execution_count": null, + "id": "cd3eee35", "metadata": { "kernel": "Bash" }, @@ -459,6 +491,7 @@ }, { "cell_type": "markdown", + "id": "ccaa8e81", "metadata": { "kernel": "SoS" }, @@ -470,6 +503,7 @@ }, { "cell_type": "markdown", + "id": "78d3c9bb", "metadata": { "kernel": "SoS" }, @@ -480,6 +514,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a0381a5b", "metadata": { "kernel": "SoS" }, @@ -493,6 +528,7 @@ }, { "cell_type": "markdown", + "id": "f0aeb3c5", "metadata": { "kernel": "SoS" }, @@ -503,6 +539,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b2fe9b72", "metadata": { "kernel": "Bash" }, @@ -513,6 +550,7 @@ }, { "cell_type": "markdown", + "id": "f6cf2b31", "metadata": { "kernel": "SoS" }, @@ -674,6 +712,7 @@ }, { "cell_type": "markdown", + "id": "f0ced14f", "metadata": { "kernel": "SoS" }, @@ -684,6 +723,7 @@ { "cell_type": "code", "execution_count": null, + "id": "c83ea9b8", "metadata": { "kernel": "SoS" }, @@ -776,6 +816,7 @@ }, { "cell_type": "markdown", + "id": "54c8edea", "metadata": { "kernel": "SoS", "vscode": { @@ -791,6 +832,7 @@ { "cell_type": "code", "execution_count": null, + "id": "95fe09f8", "metadata": { "kernel": "SoS", "vscode": { @@ -814,6 +856,7 @@ }, { "cell_type": "markdown", + "id": "e809b8e7", "metadata": { "kernel": "SoS" }, @@ -826,6 +869,7 @@ { "cell_type": "code", "execution_count": null, + "id": "c088bfef", "metadata": { "kernel": "SoS" }, @@ -848,6 +892,7 @@ }, { "cell_type": "markdown", + "id": "820d2e6f", "metadata": { "kernel": "SoS", "tags": [] @@ -861,6 +906,7 @@ { "cell_type": "code", "execution_count": null, + "id": "fe559b40", "metadata": { "kernel": "SoS" }, @@ -898,6 +944,7 @@ { "cell_type": "code", "execution_count": null, + "id": "23b16de4", "metadata": { "kernel": "SoS" }, @@ -917,6 +964,7 @@ }, { "cell_type": "markdown", + "id": "32ea6cc5", "metadata": { "kernel": "SoS", "tags": [], @@ -931,6 +979,7 @@ { "cell_type": "code", "execution_count": null, + "id": "224369a5", "metadata": { "kernel": "SoS", "tags": [] @@ -970,6 +1019,7 @@ }, { "cell_type": "markdown", + "id": "3bd79592", "metadata": { "kernel": "SoS", "toc-hr-collapsed": true @@ -983,6 +1033,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b9e634de", "metadata": { "kernel": "SoS" }, @@ -1078,6 +1129,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e041e170", "metadata": { "kernel": "SoS" }, @@ -1096,6 +1148,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4b1a321c", "metadata": { "kernel": "SoS" }, @@ -1123,6 +1176,7 @@ }, { "cell_type": "markdown", + "id": "e3024e82", "metadata": { "kernel": "SoS", "toc-hr-collapsed": true @@ -1136,6 +1190,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b4f506ea", "metadata": { "kernel": "SoS" }, @@ -1188,6 +1243,7 @@ { "cell_type": "code", "execution_count": null, + "id": "3c293a27", "metadata": { "kernel": "SoS" }, @@ -1210,6 +1266,7 @@ }, { "cell_type": "markdown", + "id": "46c4ad39", "metadata": { "kernel": "SoS", "toc-hr-collapsed": true @@ -1223,6 +1280,7 @@ { "cell_type": "code", "execution_count": null, + "id": "ffabeb62", "metadata": { "kernel": "SoS" }, @@ -1276,6 +1334,7 @@ }, { "cell_type": "markdown", + "id": "9123e02f", "metadata": { "kernel": "SoS" }, @@ -1286,6 +1345,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b0f8ff50", "metadata": { "kernel": "SoS" }, @@ -1309,6 +1369,7 @@ }, { "cell_type": "markdown", + "id": "776927c4", "metadata": { "kernel": "SoS", "toc-hr-collapsed": true @@ -1322,6 +1383,7 @@ { "cell_type": "code", "execution_count": null, + "id": "bdd1bc15", "metadata": { "kernel": "SoS" }, @@ -1364,6 +1426,7 @@ }, { "cell_type": "markdown", + "id": "76bb1921", "metadata": { "kernel": "SoS" }, @@ -1374,6 +1437,7 @@ { "cell_type": "code", "execution_count": null, + "id": "3986d78e", "metadata": { "kernel": "SoS" }, @@ -1402,6 +1466,7 @@ }, { "cell_type": "markdown", + "id": "eb507dbc", "metadata": { "kernel": "SoS" }, @@ -1414,6 +1479,7 @@ { "cell_type": "code", "execution_count": null, + "id": "67340731", "metadata": { "kernel": "SoS" }, @@ -1433,6 +1499,7 @@ { "cell_type": "code", "execution_count": null, + "id": "08ed7c0f", "metadata": { "kernel": "SoS" }, @@ -1454,6 +1521,7 @@ }, { "cell_type": "markdown", + "id": "ee112b03", "metadata": { "kernel": "SoS" }, @@ -1466,6 +1534,7 @@ { "cell_type": "code", "execution_count": null, + "id": "3c7badf5", "metadata": { "kernel": "SoS" }, @@ -1494,6 +1563,7 @@ }, { "cell_type": "markdown", + "id": "ea071a16", "metadata": { "kernel": "SoS" }, @@ -1541,5 +1611,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/molecular_phenotypes/calling/apa_calling.ipynb b/code/SoS/molecular_phenotypes/calling/apa_calling.ipynb index 02d6e4fdc..7cb60a68d 100644 --- a/code/SoS/molecular_phenotypes/calling/apa_calling.ipynb +++ b/code/SoS/molecular_phenotypes/calling/apa_calling.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "412bd5cd", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "f81cbf37", "metadata": { "kernel": "SoS" }, @@ -37,6 +39,7 @@ { "cell_type": "code", "execution_count": null, + "id": "965195c1", "metadata": { "kernel": "SoS" }, @@ -47,6 +50,7 @@ }, { "cell_type": "markdown", + "id": "0299ef01", "metadata": { "kernel": "SoS" }, @@ -60,7 +64,7 @@ "**Step 1, `UTR_reference`.**\n", "\n", "- `--hg-gtf` **`output/apa/chr22.gtf`**\n", - "(a gene annotation GTF matching the genome build used for alignment. The step reads transcript records from it and derives the 3'UTR intervals that DaPars2 will test. Built from the transcriptome-level gene feature file [prepared by the reference data module](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/reference_data.html).)\n", + "(a gene annotation GTF matching the genome build used for alignment. The step reads transcript records from it and derives the 3'UTR intervals that DaPars2 will test. Built from the transcriptome-level gene feature file [prepared by the reference data module](https://statfungen.github.io/xqtl-protocol/reference-data).)\n", "\n", "**Step 2, `bam2tools`.**\n", "\n", @@ -102,6 +106,7 @@ }, { "cell_type": "markdown", + "id": "e54e3be7", "metadata": { "kernel": "SoS" }, @@ -161,6 +166,7 @@ }, { "cell_type": "markdown", + "id": "afa34f06", "metadata": { "kernel": "SoS" }, @@ -175,6 +181,7 @@ }, { "cell_type": "markdown", + "id": "0f887a74", "metadata": { "kernel": "SoS" }, @@ -189,6 +196,7 @@ }, { "cell_type": "markdown", + "id": "49ddce7f", "metadata": { "kernel": "SoS" }, @@ -199,6 +207,7 @@ { "cell_type": "code", "execution_count": null, + "id": "44d23d84", "metadata": { "kernel": "Bash" }, @@ -211,6 +220,7 @@ }, { "cell_type": "markdown", + "id": "fb52d163", "metadata": { "kernel": "SoS" }, @@ -223,6 +233,7 @@ }, { "cell_type": "markdown", + "id": "ee38b830", "metadata": { "kernel": "SoS" }, @@ -233,6 +244,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5b96c41a", "metadata": { "kernel": "Bash" }, @@ -245,6 +257,7 @@ }, { "cell_type": "markdown", + "id": "b009841f", "metadata": { "kernel": "SoS" }, @@ -258,6 +271,7 @@ }, { "cell_type": "markdown", + "id": "21ada85c", "metadata": { "kernel": "SoS" }, @@ -268,6 +282,7 @@ { "cell_type": "code", "execution_count": null, + "id": "cdea1120", "metadata": { "kernel": "Bash" }, @@ -281,6 +296,7 @@ }, { "cell_type": "markdown", + "id": "c8774f27", "metadata": { "kernel": "SoS" }, @@ -296,6 +312,7 @@ }, { "cell_type": "markdown", + "id": "8074e6a0", "metadata": { "kernel": "SoS" }, @@ -306,6 +323,7 @@ { "cell_type": "code", "execution_count": null, + "id": "cc1d76e4", "metadata": { "kernel": "Bash" }, @@ -320,6 +338,7 @@ }, { "cell_type": "markdown", + "id": "7148a554", "metadata": { "kernel": "SoS" }, @@ -330,6 +349,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b2e9d83c", "metadata": { "kernel": "SoS" }, @@ -340,6 +360,7 @@ }, { "cell_type": "markdown", + "id": "9c1d9f5d", "metadata": { "kernel": "SoS" }, @@ -407,6 +428,7 @@ }, { "cell_type": "markdown", + "id": "a86557e7", "metadata": { "kernel": "SoS" }, @@ -417,6 +439,7 @@ { "cell_type": "code", "execution_count": null, + "id": "73acf3bd", "metadata": { "kernel": "SoS" }, @@ -437,6 +460,7 @@ }, { "cell_type": "markdown", + "id": "26be1ac3", "metadata": { "kernel": "SoS" }, @@ -448,6 +472,7 @@ { "cell_type": "code", "execution_count": null, + "id": "c0750719", "metadata": { "kernel": "SoS" }, @@ -469,6 +494,7 @@ }, { "cell_type": "markdown", + "id": "b1f7e7cb", "metadata": { "kernel": "SoS" }, @@ -480,6 +506,7 @@ { "cell_type": "code", "execution_count": null, + "id": "84698d2a", "metadata": { "kernel": "SoS" }, @@ -500,6 +527,7 @@ }, { "cell_type": "markdown", + "id": "5b87fb2d", "metadata": { "kernel": "SoS" }, @@ -511,6 +539,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a7c9fed9", "metadata": { "kernel": "SoS" }, @@ -532,6 +561,7 @@ }, { "cell_type": "markdown", + "id": "89efe75c", "metadata": { "kernel": "SoS" }, @@ -543,6 +573,7 @@ { "cell_type": "code", "execution_count": null, + "id": "062c35f4", "metadata": { "kernel": "SoS" }, @@ -601,5 +632,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/molecular_phenotypes/calling/methylation_calling.ipynb b/code/SoS/molecular_phenotypes/calling/methylation_calling.ipynb index 02e8398c3..c8841054b 100644 --- a/code/SoS/molecular_phenotypes/calling/methylation_calling.ipynb +++ b/code/SoS/molecular_phenotypes/calling/methylation_calling.ipynb @@ -73,7 +73,7 @@ "2. The methylation data samples will first be filtered based on [bisulphite conversation rate](https://www.ncbi.nlm.nih.gov/pmc/articles/PMC4527772/). This operation is done using the [bscon function from watermelon package](http://www.bioconductor.org/packages/release/bioc/vignettes/wateRmelon/inst/doc/wateRmelon.html#introduction) \n", "3. samples will then be filtered based on a [detection pvalue](https://www.rdocumentation.org/packages/minfi/versions/1.18.4/topics/detectionP), which indicates the quality of the signal at each genomics position\n", "4. [Stratified Quantile Normalization](https://rdrr.io/bioc/minfi/man/preprocessQuantile.html) will then be applied.\n", - "5. features will be filtered if they are on sex chr, known to be [cross-reactive,maping to multiple regions in the genome.](https://academic.oup.com/nargab/article/2/4/lqaa105/6040968), overlapping with snps, or having too low a detection P. The list of cross-reactive probe can be found as `/opt/cross_reactive_probe_Hop2020.txt` in the docker image and [here](https://raw.githubusercontent.com/hsun3163/xqtl-protocol/main/data/cross_reactive_probe_Hop2020.txt).\n", + "5. features will be filtered if they are on sex chr, known to be [cross-reactive,maping to multiple regions in the genome.](https://academic.oup.com/nargab/article/2/4/lqaa105/6040968), overlapping with snps, or having too low a detection P. The list of cross-reactive probes ships with this repository as [`data/cross_reactive_probe_Hop2020.txt`](https://raw.githubusercontent.com/StatFunGen/xqtl-protocol/main/data/cross_reactive_probe_Hop2020.txt), which is the default for `--cross-reactive-probes`.\n", "6. Beta and M value will for all the probes/samples will then each be saved to a indexed bed.gz file.\n", "\n", "[As documented here](https://github.com/statfungen/xqtl-protocol/issues/312) when the batch of IDAT data are different, there will be a problem reading the IDAT file without specifing the force = TRUE option in the `read.metharray.exp(targets = targets,force = TRUE)`\n", @@ -524,4 +524,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/molecular_phenotypes/methylation.ipynb b/code/SoS/molecular_phenotypes/methylation.ipynb index 7e20e37ed..414df1173 100644 --- a/code/SoS/molecular_phenotypes/methylation.ipynb +++ b/code/SoS/molecular_phenotypes/methylation.ipynb @@ -27,9 +27,9 @@ "source": [ "## Overview\n", "\n", - "Choose either [SeSAMe](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/methylation_calling.html) or [minfi](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/methylation_calling.html) to preprocess paired red/green IDAT intensities. Both workflows produce beta-value and M-value matrices, convert probe coordinates to BED phenotype files, and annotate probes with nearby genes. M values are generally used as the quantitative phenotype for association testing, whereas beta values remain useful for interpretation.\n", + "Choose either [SeSAMe](https://statfungen.github.io/xqtl-protocol/methylation-calling) or [minfi](https://statfungen.github.io/xqtl-protocol/methylation-calling) to preprocess paired red/green IDAT intensities. Both workflows produce beta-value and M-value matrices, convert probe coordinates to BED phenotype files, and annotate probes with nearby genes. M values are generally used as the quantitative phenotype for association testing, whereas beta values remain useful for interpretation.\n", "\n", - "After calling, [phenotype imputation](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/phenotype/phenotype_imputation.html) removes probes above the missingness threshold and imputes the remaining missing values. Steps 1 and 2 are alternative starting routes; run only one, then run step 3 on its M-value BED output." + "After calling, [phenotype imputation](https://statfungen.github.io/xqtl-protocol/phenotype-imputation) removes probes above the missingness threshold and imputes the remaining missing values. Steps 1 and 2 are alternative starting routes; run only one, then run step 3 on its M-value BED output." ] }, { @@ -52,7 +52,7 @@ "id": "e82bf49d", "metadata": {}, "source": [ - "### [1. Call methylation levels with SeSAMe](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/methylation_calling.html)\n", + "### [1. Call methylation levels with SeSAMe](https://statfungen.github.io/xqtl-protocol/methylation-calling)\n", "\n", "**What it does:** `sesame` performs probe- and sample-level detection filtering, calculates beta and M values, and formats both matrices as coordinate-sorted BED files." ] @@ -72,7 +72,7 @@ "id": "f3f67af0", "metadata": {}, "source": [ - "### [2. Call methylation levels with minfi](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/methylation_calling.html)\n", + "### [2. Call methylation levels with minfi](https://statfungen.github.io/xqtl-protocol/methylation-calling)\n", "\n", "**What it does:** `minfi` applies detection-p-value filtering and functional normalization, then writes beta and M values in the same downstream-compatible BED format." ] @@ -92,7 +92,7 @@ "id": "d3517c97", "metadata": {}, "source": [ - "### [3. Filter and impute the M-value matrix](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/phenotype/phenotype_imputation.html)\n", + "### [3. Filter and impute the M-value matrix](https://statfungen.github.io/xqtl-protocol/phenotype-imputation)\n", "\n", "**What it does:** `bed_filter_na` removes probes exceeding the missingness threshold and uses soft-impute for the remaining missing entries." ] @@ -203,4 +203,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/molecular_phenotypes/single_cell.ipynb b/code/SoS/molecular_phenotypes/single_cell.ipynb index ffdfb8b6a..327308cf6 100644 --- a/code/SoS/molecular_phenotypes/single_cell.ipynb +++ b/code/SoS/molecular_phenotypes/single_cell.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "20c5a20e", "metadata": {}, "source": [ "# Single-cell and single-nucleus molecular phenotype preprocessing\n", @@ -11,6 +12,7 @@ }, { "cell_type": "markdown", + "id": "d27a1e77", "metadata": {}, "source": [ "#### Miniprotocol Timing\n", @@ -22,6 +24,7 @@ }, { "cell_type": "markdown", + "id": "cfe0d9b0", "metadata": {}, "source": [ "## Overview\n", @@ -33,6 +36,7 @@ }, { "cell_type": "markdown", + "id": "5aface33", "metadata": {}, "source": [ "## Steps\n", @@ -51,6 +55,7 @@ }, { "cell_type": "markdown", + "id": "f4421cab", "metadata": {}, "source": [ "### 1. [Cell-level quality control](snRNAseq_preprocessing.ipynb)\n", @@ -61,6 +66,7 @@ { "cell_type": "code", "execution_count": null, + "id": "aaa30f44", "metadata": {}, "outputs": [], "source": [ @@ -69,6 +75,7 @@ }, { "cell_type": "markdown", + "id": "eaa28c1c", "metadata": {}, "source": [ "### 2. [Cell-type annotation](snRNAseq_preprocessing.ipynb)\n", @@ -79,6 +86,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5c221960", "metadata": {}, "outputs": [], "source": [ @@ -87,6 +95,7 @@ }, { "cell_type": "markdown", + "id": "6638117d", "metadata": {}, "source": [ "### 3. [Pseudobulk count aggregation](QC/pseudobulk_preprocessing.ipynb)\n", @@ -97,6 +106,7 @@ { "cell_type": "code", "execution_count": null, + "id": "16a5e7fc", "metadata": {}, "outputs": [], "source": [ @@ -105,6 +115,7 @@ }, { "cell_type": "markdown", + "id": "b693a2bc", "metadata": {}, "source": [ "### 4. [Sample-ID harmonization](QC/pseudobulk_preprocessing.ipynb)\n", @@ -115,6 +126,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6d390393", "metadata": {}, "outputs": [], "source": [ @@ -123,6 +135,7 @@ }, { "cell_type": "markdown", + "id": "dad1cb8b", "metadata": {}, "source": [ "### 5. [Pseudobulk quality control and residualization](QC/pseudobulk_preprocessing.ipynb)\n", @@ -133,6 +146,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a361334b", "metadata": {}, "outputs": [], "source": [ @@ -141,6 +155,7 @@ }, { "cell_type": "markdown", + "id": "1a034b79", "metadata": {}, "source": [ "### 6. [QTL phenotype formatting](QC/pseudobulk_preprocessing.ipynb)\n", @@ -151,6 +166,7 @@ { "cell_type": "code", "execution_count": null, + "id": "cf85549e", "metadata": {}, "outputs": [], "source": [ @@ -159,6 +175,7 @@ }, { "cell_type": "markdown", + "id": "2d344fc8", "metadata": {}, "source": [ "## Output\n", @@ -177,6 +194,7 @@ }, { "cell_type": "markdown", + "id": "37b42eac", "metadata": {}, "source": [ "## Anticipated Results\n", @@ -188,6 +206,7 @@ }, { "cell_type": "markdown", + "id": "f59bd2d0", "metadata": {}, "source": [ "## Command Interface\n", @@ -198,6 +217,7 @@ { "cell_type": "code", "execution_count": null, + "id": "1fd0ca9d", "metadata": {}, "outputs": [], "source": [ @@ -207,6 +227,7 @@ { "cell_type": "code", "execution_count": null, + "id": "10f0ad03", "metadata": {}, "outputs": [], "source": [ @@ -234,5 +255,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/molecular_phenotypes/snRNAseq_preprocessing.ipynb b/code/SoS/molecular_phenotypes/snRNAseq_preprocessing.ipynb index 5b1cfb239..a33698172 100644 --- a/code/SoS/molecular_phenotypes/snRNAseq_preprocessing.ipynb +++ b/code/SoS/molecular_phenotypes/snRNAseq_preprocessing.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "3a6b8983", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "ec07b065", "metadata": { "kernel": "SoS" }, @@ -28,6 +30,7 @@ }, { "cell_type": "markdown", + "id": "aa9a7be1", "metadata": { "kernel": "SoS" }, @@ -51,6 +54,7 @@ }, { "cell_type": "markdown", + "id": "d25633ca", "metadata": { "kernel": "SoS" }, @@ -71,6 +75,7 @@ }, { "cell_type": "markdown", + "id": "d56b42b2", "metadata": { "kernel": "SoS" }, @@ -82,6 +87,7 @@ }, { "cell_type": "markdown", + "id": "87e63dff", "metadata": { "kernel": "SoS" }, @@ -93,6 +99,7 @@ }, { "cell_type": "markdown", + "id": "9b10d208", "metadata": { "kernel": "SoS" }, @@ -103,6 +110,7 @@ { "cell_type": "code", "execution_count": null, + "id": "307a159a", "metadata": { "kernel": "SoS" }, @@ -117,6 +125,7 @@ }, { "cell_type": "markdown", + "id": "30a36be6", "metadata": { "kernel": "SoS" }, @@ -126,6 +135,7 @@ }, { "cell_type": "markdown", + "id": "b16b9fb1", "metadata": { "kernel": "SoS" }, @@ -136,6 +146,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5cf8a473", "metadata": { "kernel": "SoS" }, @@ -150,6 +161,7 @@ }, { "cell_type": "markdown", + "id": "5dc30658", "metadata": { "kernel": "SoS" }, @@ -160,6 +172,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a89f306c", "metadata": { "kernel": "SoS" }, @@ -170,6 +183,7 @@ }, { "cell_type": "markdown", + "id": "5c812df4", "metadata": { "kernel": "SoS" }, @@ -224,6 +238,7 @@ }, { "cell_type": "markdown", + "id": "0dba8f81", "metadata": { "kernel": "SoS" }, @@ -234,6 +249,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f719dc3c", "metadata": { "kernel": "SoS" }, @@ -256,6 +272,7 @@ { "cell_type": "code", "execution_count": null, + "id": "1550f2ee", "metadata": { "kernel": "SoS" }, @@ -299,6 +316,7 @@ { "cell_type": "code", "execution_count": null, + "id": "3e5ddc5f", "metadata": { "kernel": "SoS" }, @@ -355,5 +373,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/molecular_phenotypes/splicing.ipynb b/code/SoS/molecular_phenotypes/splicing.ipynb index 2c59e83a5..4e253a37d 100644 --- a/code/SoS/molecular_phenotypes/splicing.ipynb +++ b/code/SoS/molecular_phenotypes/splicing.ipynb @@ -33,9 +33,9 @@ "source": [ "## Overview\n", "\n", - "The [splicing-calling module](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/splicing_calling.html) quantifies splicing with LeafCutter, which derives intron excision ratios from STAR splice-junction files without relying on a transcript annotation.\n", + "The [splicing-calling module](https://statfungen.github.io/xqtl-protocol/splicing-calling) quantifies splicing with LeafCutter, which derives intron excision ratios from STAR splice-junction files without relying on a transcript annotation.\n", "\n", - "The [normalization module](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/QC/splicing_normalization.html) performs missingness and variability filtering followed by quantile normalization. The [gene-annotation module](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/phenotype/gene_annotation.html) then assigns coordinates and phenotype groups required by TensorQTL." + "The [normalization module](https://statfungen.github.io/xqtl-protocol/splicing-normalization) performs missingness and variability filtering followed by quantile normalization. The [gene-annotation module](https://statfungen.github.io/xqtl-protocol/gene-annotation) then assigns coordinates and phenotype groups required by TensorQTL." ] }, { @@ -62,7 +62,7 @@ "kernel": "SoS" }, "source": [ - "### [1. Quantify intron usage with LeafCutter](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/splicing_calling.html)\n", + "### [1. Quantify intron usage with LeafCutter](https://statfungen.github.io/xqtl-protocol/splicing-calling)\n", "\n", "**What it does:** `leafcutter` extracts splice junctions and clusters introns to calculate per-sample intron excision ratios." ] @@ -86,7 +86,7 @@ "kernel": "SoS" }, "source": [ - "### [2. Normalize LeafCutter ratios](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/QC/splicing_normalization.html)\n", + "### [2. Normalize LeafCutter ratios](https://statfungen.github.io/xqtl-protocol/splicing-normalization)\n", "\n", "**What it does:** `leafcutter_norm` filters introns and clusters, mean-imputes retained missing values, and quantile-normalizes the ratio matrix." ] @@ -110,7 +110,7 @@ "kernel": "SoS" }, "source": [ - "### [3. Annotate LeafCutter phenotypes](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/phenotype/gene_annotation.html)\n", + "### [3. Annotate LeafCutter phenotypes](https://statfungen.github.io/xqtl-protocol/gene-annotation)\n", "\n", "**What it does:** `annotate_leafcutter_isoforms` maps introns to genes and writes the coordinate-sorted phenotype matrix and phenotype-group file used by TensorQTL." ] diff --git a/code/SoS/multivariate_genome/MASH.ipynb b/code/SoS/multivariate_genome/MASH.ipynb index b36dca20d..24b40e233 100644 --- a/code/SoS/multivariate_genome/MASH.ipynb +++ b/code/SoS/multivariate_genome/MASH.ipynb @@ -4,31 +4,57 @@ "cell_type": "markdown", "id": "242c9400", "metadata": {}, - "source": "# Multivariate adaptive shrinkage (MASH) mini-protocol\nPrepare cross-condition summary statistics, fit a multivariate adaptive-shrinkage model, and estimate condition-specific effects and contrasts." + "source": [ + "# Multivariate adaptive shrinkage (MASH) mini-protocol\n", + "Prepare cross-condition summary statistics, fit a multivariate adaptive-shrinkage model, and estimate condition-specific effects and contrasts." + ] }, { "cell_type": "markdown", "id": "a0485e0e", "metadata": {}, - "source": "#### Miniprotocol Timing\nThis is the total duration for the selected route; module-specific timings appear on their respective pages.\nTiming: TBD" + "source": [ + "#### Miniprotocol Timing\n", + "This is the total duration for the selected route; module-specific timings appear on their respective pages.\n", + "Timing: TBD" + ] }, { "cell_type": "markdown", "id": "b4a06433", "metadata": {}, - "source": "## Overview\nThis mini-protocol organizes the MASH workflow into preprocessing, model fitting, and posterior analysis. Steps 1–2 call [`mash_preprocessing.ipynb`](https://statfungen.github.io/xqtl-protocol/code/multivariate_genome/MASH/mash_preprocessing.html), step 3 calls [`mash_fit.ipynb`](https://statfungen.github.io/xqtl-protocol/code/multivariate_genome/MASH/mash_fit.html), and steps 4–5 call [`mash_posterior.ipynb`](https://statfungen.github.io/xqtl-protocol/code/multivariate_genome/MASH/mash_posterior.html). The covariance prior and residual variance used for fitting can be learned with [`mixture_prior.ipynb`](https://statfungen.github.io/xqtl-protocol/code/multivariate_genome/MASH/mixture_prior.html).\nSteps 1 and 2 are alternative preprocessing routes. Step 3 fits a model from prepared inputs, while steps 4–5 apply an existing model and summarize posterior contrasts." + "source": [ + "## Overview\n", + "This mini-protocol organizes the MASH workflow into preprocessing, model fitting, and posterior analysis. Steps 1–2 call [`mash_preprocessing.ipynb`](https://statfungen.github.io/xqtl-protocol/mash-preprocessing), step 3 calls [`mash_fit.ipynb`](https://statfungen.github.io/xqtl-protocol/mash-fit), and steps 4–5 call [`mash_posterior.ipynb`](https://statfungen.github.io/xqtl-protocol/mash-posterior). The covariance prior and residual variance used for fitting can be learned with [`mixture_prior.ipynb`](https://statfungen.github.io/xqtl-protocol/mixture-prior).\n", + "Steps 1 and 2 are alternative preprocessing routes. Step 3 fits a model from prepared inputs, while steps 4–5 apply an existing model and summarize posterior contrasts." + ] }, { "cell_type": "markdown", "id": "615550c9", "metadata": {}, - "source": "## Steps\nChoose a route before running commands; the commands are not one mandatory chain.\n| Analysis goal | Commands to run, in order | Inputs |\n|---|---|---|\n| Build MASH-ready data from fine-mapping results and fit a model | 1 → 3 | `tests/fixtures/qtl_mini/fine_mapping_meta.tsv`; prepared covariance prior and residual-variance RDS files |\n| Build random and null sets from tensorQTL results and fit a model | 2 → 3 | `input/finemapping/protocol_example.region`; `input/protocol_example.sumstats_list.txt`; prepared covariance prior and residual-variance RDS files |\n| Apply an existing MASH model | 4 | `input/finemapping/protocol_example.analysis_units.txt`; `input/mash/protocol_example.mash_model.rds`; `input/twas/protocol_example.posterior_vhat.rds` |\n| Plot posterior contrasts | 4 → 5 | Inputs for step 4 |\n\nThe fitting command uses `input/mash/protocol_example.EE.mash.rds`, `input/mash/protocol_example.EE.V_simple.rds`, and `input/mash/protocol_example.EE.prior.rds`. The prior and residual variance can be learned with the [mixture-prior module](https://statfungen.github.io/xqtl-protocol/mixture_prior.html); its covariance-component and variance-estimation workflows are alternatives and should be selected deliberately rather than run as a single chain." + "source": [ + "## Steps\n", + "Choose a route before running commands; the commands are not one mandatory chain.\n", + "| Analysis goal | Commands to run, in order | Inputs |\n", + "|---|---|---|\n", + "| Build MASH-ready data from fine-mapping results and fit a model | 1 → 3 | `tests/fixtures/qtl_mini/fine_mapping_meta.tsv`; prepared covariance prior and residual-variance RDS files |\n", + "| Build random and null sets from tensorQTL results and fit a model | 2 → 3 | `input/finemapping/protocol_example.region`; `input/protocol_example.sumstats_list.txt`; prepared covariance prior and residual-variance RDS files |\n", + "| Apply an existing MASH model | 4 | `input/finemapping/protocol_example.analysis_units.txt`; `input/mash/protocol_example.mash_model.rds`; `input/twas/protocol_example.posterior_vhat.rds` |\n", + "| Plot posterior contrasts | 4 → 5 | Inputs for step 4 |\n", + "\n", + "The fitting command uses `input/mash/protocol_example.EE.mash.rds`, `input/mash/protocol_example.EE.V_simple.rds`, and `input/mash/protocol_example.EE.prior.rds`. The prior and residual variance can be learned with the [mixture-prior module](https://statfungen.github.io/xqtl-protocol/mixture-prior); its covariance-component and variance-estimation workflows are alternatives and should be selected deliberately rather than run as a single chain." + ] }, { "cell_type": "markdown", "id": "6cf8a734", "metadata": {}, - "source": "### 1. [Prepare strong effects from fine-mapping results](https://statfungen.github.io/xqtl-protocol/code/multivariate_genome/MASH/mash_preprocessing.html)\n\n**What it does:** Converts fine-mapping posterior estimates into the strong-effect data used to learn or fit a MASH model." + "source": [ + "### 1. [Prepare strong effects from fine-mapping results](https://statfungen.github.io/xqtl-protocol/mash-preprocessing)\n", + "\n", + "**What it does:** Converts fine-mapping posterior estimates into the strong-effect data used to learn or fit a MASH model." + ] }, { "cell_type": "markdown", @@ -57,7 +83,11 @@ "cell_type": "markdown", "id": "57319d6e", "metadata": {}, - "source": "### 2. [Prepare random and null effects from tensorQTL results](https://statfungen.github.io/xqtl-protocol/code/multivariate_genome/MASH/mash_preprocessing.html)\n\n**What it does:** Samples random regions and extracts null effects from tensorQTL summary statistics to complement the strong-effect set. Use this route instead of step 1 when tensorQTL summary statistics are the starting point." + "source": [ + "### 2. [Prepare random and null effects from tensorQTL results](https://statfungen.github.io/xqtl-protocol/mash-preprocessing)\n", + "\n", + "**What it does:** Samples random regions and extracts null effects from tensorQTL summary statistics to complement the strong-effect set. Use this route instead of step 1 when tensorQTL summary statistics are the starting point." + ] }, { "cell_type": "markdown", @@ -86,7 +116,11 @@ "cell_type": "markdown", "id": "cb1835c7", "metadata": {}, - "source": "### 3. [Fit the MASH model](https://statfungen.github.io/xqtl-protocol/code/multivariate_genome/MASH/mash_fit.html)\n\n**What it does:** Fits the multivariate adaptive-shrinkage model using prepared effects, a residual-variance estimate, and a covariance prior; `--compute-posterior` also stores posterior summaries for the fitted data." + "source": [ + "### 3. [Fit the MASH model](https://statfungen.github.io/xqtl-protocol/mash-fit)\n", + "\n", + "**What it does:** Fits the multivariate adaptive-shrinkage model using prepared effects, a residual-variance estimate, and a covariance prior; `--compute-posterior` also stores posterior summaries for the fitted data." + ] }, { "cell_type": "markdown", @@ -117,7 +151,11 @@ "cell_type": "markdown", "id": "d90d08a4", "metadata": {}, - "source": "### 4. [Apply the fitted model](https://statfungen.github.io/xqtl-protocol/code/multivariate_genome/MASH/mash_posterior.html)\n\n**What it does:** Computes posterior effect estimates for each analysis unit using an existing MASH model. Adjust `--exclude-condition` only when specific conditions must be omitted." + "source": [ + "### 4. [Apply the fitted model](https://statfungen.github.io/xqtl-protocol/mash-posterior)\n", + "\n", + "**What it does:** Computes posterior effect estimates for each analysis unit using an existing MASH model. Adjust `--exclude-condition` only when specific conditions must be omitted." + ] }, { "cell_type": "markdown", @@ -147,7 +185,11 @@ "cell_type": "markdown", "id": "a47d3670", "metadata": {}, - "source": "### 5. [Plot posterior contrasts](https://statfungen.github.io/xqtl-protocol/code/multivariate_genome/MASH/mash_posterior.html)\n\n**What it does:** Summarizes and plots the condition contrasts produced by step 4." + "source": [ + "### 5. [Plot posterior contrasts](https://statfungen.github.io/xqtl-protocol/mash-posterior)\n", + "\n", + "**What it does:** Summarizes and plots the condition contrasts produced by step 4." + ] }, { "cell_type": "markdown", @@ -205,7 +247,9 @@ "cell_type": "markdown", "id": "6ce34c31", "metadata": {}, - "source": "## Command Interface" + "source": [ + "## Command Interface" + ] }, { "cell_type": "code", @@ -265,4 +309,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/multivariate_genome/MASH/mash_fit.ipynb b/code/SoS/multivariate_genome/MASH/mash_fit.ipynb index df26b9aac..9cdff1c64 100644 --- a/code/SoS/multivariate_genome/MASH/mash_fit.ipynb +++ b/code/SoS/multivariate_genome/MASH/mash_fit.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "f0ee37f2", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "875480d9", "metadata": { "kernel": "SoS" }, @@ -32,6 +34,7 @@ }, { "cell_type": "markdown", + "id": "345aae16", "metadata": { "kernel": "SoS" }, @@ -126,6 +129,7 @@ }, { "cell_type": "markdown", + "id": "6d297565", "metadata": { "kernel": "SoS" }, @@ -202,6 +206,7 @@ }, { "cell_type": "markdown", + "id": "8d228c8e", "metadata": { "kernel": "SoS" }, @@ -215,17 +220,18 @@ }, { "cell_type": "markdown", + "id": "02058627", "metadata": { "kernel": "SoS" }, "source": [ - "**Timing**: ~1-2 min (on toy dataset)\n", - "" + "**Timing**: ~1-2 min (on toy dataset)\n" ] }, { "cell_type": "code", "execution_count": null, + "id": "77fcea6b", "metadata": { "kernel": "Bash" }, @@ -243,6 +249,7 @@ }, { "cell_type": "markdown", + "id": "0981f0f4", "metadata": { "kernel": "SoS" }, @@ -253,6 +260,7 @@ { "cell_type": "code", "execution_count": null, + "id": "c7326da9", "metadata": { "kernel": "SoS" }, @@ -263,6 +271,7 @@ }, { "cell_type": "markdown", + "id": "51ee641e", "metadata": { "kernel": "SoS" }, @@ -321,6 +330,7 @@ }, { "cell_type": "markdown", + "id": "0a46fef6", "metadata": { "kernel": "SoS" }, @@ -333,6 +343,7 @@ { "cell_type": "code", "execution_count": null, + "id": "39cc1360", "metadata": { "kernel": "SoS" }, @@ -373,6 +384,7 @@ }, { "cell_type": "markdown", + "id": "bcbef439", "metadata": { "kernel": "SoS" }, @@ -387,6 +399,7 @@ { "cell_type": "code", "execution_count": null, + "id": "554e2537", "metadata": { "kernel": "SoS" }, @@ -411,6 +424,7 @@ }, { "cell_type": "markdown", + "id": "c31423e1", "metadata": { "kernel": "SoS" }, @@ -423,6 +437,7 @@ { "cell_type": "code", "execution_count": null, + "id": "eaf7b274", "metadata": { "kernel": "SoS" }, @@ -498,5 +513,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/multivariate_genome/MASH/mash_posterior.ipynb b/code/SoS/multivariate_genome/MASH/mash_posterior.ipynb index 177399078..3630eafdd 100644 --- a/code/SoS/multivariate_genome/MASH/mash_posterior.ipynb +++ b/code/SoS/multivariate_genome/MASH/mash_posterior.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "e8e28a92", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "cf8143e1", "metadata": { "kernel": "SoS" }, @@ -37,6 +39,7 @@ }, { "cell_type": "markdown", + "id": "4657a80b", "metadata": { "kernel": "SoS" }, @@ -148,6 +151,7 @@ }, { "cell_type": "markdown", + "id": "348a7268", "metadata": { "kernel": "SoS" }, @@ -184,6 +188,7 @@ }, { "cell_type": "markdown", + "id": "99fadb96", "metadata": { "kernel": "SoS" }, @@ -193,6 +198,7 @@ }, { "cell_type": "markdown", + "id": "2c2a79a6", "metadata": { "kernel": "SoS" }, @@ -202,6 +208,7 @@ }, { "cell_type": "markdown", + "id": "c50bf659", "metadata": { "kernel": "SoS" }, @@ -212,6 +219,7 @@ { "cell_type": "code", "execution_count": null, + "id": "87a5d5ad", "metadata": { "kernel": "Bash" }, @@ -228,6 +236,7 @@ }, { "cell_type": "markdown", + "id": "d3678518", "metadata": { "kernel": "SoS" }, @@ -238,6 +247,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5ca60bb7", "metadata": { "kernel": "Bash" }, @@ -255,6 +265,7 @@ }, { "cell_type": "markdown", + "id": "372bb0f2", "metadata": { "kernel": "SoS" }, @@ -264,17 +275,18 @@ }, { "cell_type": "markdown", + "id": "381ff5e1", "metadata": { "kernel": "SoS" }, "source": [ "**Timing**: TBD (on toy dataset)" - ], - "id": "381ff5e1" + ] }, { "cell_type": "code", "execution_count": null, + "id": "da1f23e4", "metadata": { "kernel": "Bash" }, @@ -287,6 +299,7 @@ }, { "cell_type": "markdown", + "id": "510942cb", "metadata": { "kernel": "SoS" }, @@ -296,17 +309,18 @@ }, { "cell_type": "markdown", + "id": "7e8c0637", "metadata": { "kernel": "SoS" }, "source": [ "**Timing**: TBD (on toy dataset)" - ], - "id": "7e8c0637" + ] }, { "cell_type": "code", "execution_count": null, + "id": "e36bbdfa", "metadata": { "kernel": "Bash" }, @@ -321,6 +335,7 @@ }, { "cell_type": "markdown", + "id": "ecef424f", "metadata": { "kernel": "SoS" }, @@ -330,17 +345,18 @@ }, { "cell_type": "markdown", + "id": "2573b8e6", "metadata": { "kernel": "SoS" }, "source": [ "**Timing**: TBD (on toy dataset)" - ], - "id": "2573b8e6" + ] }, { "cell_type": "code", "execution_count": null, + "id": "c8eaf132", "metadata": { "kernel": "Bash" }, @@ -355,6 +371,7 @@ }, { "cell_type": "markdown", + "id": "1c012cdc", "metadata": { "kernel": "SoS" }, @@ -364,17 +381,18 @@ }, { "cell_type": "markdown", + "id": "2fd7579b", "metadata": { "kernel": "SoS" }, "source": [ "**Timing**: TBD (on toy dataset)" - ], - "id": "2fd7579b" + ] }, { "cell_type": "code", "execution_count": null, + "id": "528db6dc", "metadata": { "kernel": "Bash" }, @@ -389,6 +407,7 @@ }, { "cell_type": "markdown", + "id": "1b5455b0", "metadata": { "kernel": "SoS" }, @@ -398,17 +417,18 @@ }, { "cell_type": "markdown", + "id": "27251ba3", "metadata": { "kernel": "SoS" }, "source": [ "**Timing**: TBD (on toy dataset)" - ], - "id": "27251ba3" + ] }, { "cell_type": "code", "execution_count": null, + "id": "dccb4e0c", "metadata": { "kernel": "Bash" }, @@ -423,6 +443,7 @@ }, { "cell_type": "markdown", + "id": "bae29765", "metadata": { "kernel": "SoS" }, @@ -433,6 +454,7 @@ { "cell_type": "code", "execution_count": null, + "id": "0626a823", "metadata": { "kernel": "SoS" }, @@ -443,6 +465,7 @@ }, { "cell_type": "markdown", + "id": "43d637a8", "metadata": { "kernel": "SoS" }, @@ -564,6 +587,7 @@ }, { "cell_type": "markdown", + "id": "18bbc438", "metadata": { "kernel": "SoS" }, @@ -574,6 +598,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2f73ea50", "metadata": { "kernel": "SoS" }, @@ -615,6 +640,7 @@ }, { "cell_type": "markdown", + "id": "4b1cc854", "metadata": { "kernel": "SoS", "tags": [] @@ -628,6 +654,7 @@ { "cell_type": "code", "execution_count": null, + "id": "30ddd6f6", "metadata": { "kernel": "SoS" }, @@ -661,6 +688,7 @@ { "cell_type": "code", "execution_count": null, + "id": "91285957", "metadata": { "kernel": "SoS" }, @@ -676,6 +704,7 @@ }, { "cell_type": "markdown", + "id": "a119cefb", "metadata": { "kernel": "SoS", "tags": [] @@ -690,6 +719,7 @@ { "cell_type": "code", "execution_count": null, + "id": "562c1e3e", "metadata": { "kernel": "SoS" }, @@ -723,6 +753,7 @@ { "cell_type": "code", "execution_count": null, + "id": "c20d9acf", "metadata": { "kernel": "SoS" }, @@ -743,6 +774,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e70c7304", "metadata": { "kernel": "SoS", "tags": [] @@ -761,6 +793,7 @@ }, { "cell_type": "markdown", + "id": "9b9c5bd2", "metadata": { "kernel": "SoS", "tags": [] @@ -773,6 +806,7 @@ { "cell_type": "code", "execution_count": null, + "id": "ce6d61f4", "metadata": { "kernel": "SoS" }, @@ -798,6 +832,7 @@ }, { "cell_type": "markdown", + "id": "079a9551", "metadata": { "kernel": "SoS" }, @@ -811,6 +846,7 @@ { "cell_type": "code", "execution_count": null, + "id": "90c54757", "metadata": { "kernel": "SoS" }, @@ -839,6 +875,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e4d6ba7d", "metadata": { "kernel": "SoS" }, @@ -856,6 +893,7 @@ }, { "cell_type": "markdown", + "id": "3cc01e95", "metadata": { "kernel": "SoS" }, @@ -867,6 +905,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5e723fac", "metadata": { "kernel": "SoS" }, @@ -892,6 +931,7 @@ }, { "cell_type": "markdown", + "id": "daf09eee", "metadata": { "kernel": "SoS" }, @@ -903,6 +943,7 @@ { "cell_type": "code", "execution_count": null, + "id": "ea64f36b", "metadata": { "kernel": "SoS" }, @@ -931,6 +972,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e1729dcf", "metadata": { "kernel": "SoS" }, @@ -987,5 +1029,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/multivariate_genome/MASH/mash_preprocessing.ipynb b/code/SoS/multivariate_genome/MASH/mash_preprocessing.ipynb index 06e6ce7b5..a83a15c04 100644 --- a/code/SoS/multivariate_genome/MASH/mash_preprocessing.ipynb +++ b/code/SoS/multivariate_genome/MASH/mash_preprocessing.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "0bf5c536", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "d7222667", "metadata": { "kernel": "SoS" }, @@ -34,6 +36,7 @@ }, { "cell_type": "markdown", + "id": "da17ce6d", "metadata": { "kernel": "SoS" }, @@ -82,6 +85,7 @@ }, { "cell_type": "markdown", + "id": "de59df71", "metadata": { "kernel": "SoS" }, @@ -258,6 +262,7 @@ }, { "cell_type": "markdown", + "id": "3ead301f", "metadata": { "kernel": "SoS" }, @@ -269,6 +274,7 @@ }, { "cell_type": "markdown", + "id": "ac112e99", "metadata": { "kernel": "SoS" }, @@ -280,6 +286,7 @@ }, { "cell_type": "markdown", + "id": "d479ee23", "metadata": { "kernel": "SoS" }, @@ -290,6 +297,7 @@ { "cell_type": "code", "execution_count": null, + "id": "d9a05283", "metadata": { "kernel": "Bash" }, @@ -305,6 +313,7 @@ }, { "cell_type": "markdown", + "id": "df82af6d", "metadata": { "kernel": "Bash" }, @@ -316,6 +325,7 @@ }, { "cell_type": "markdown", + "id": "3608770b", "metadata": { "kernel": "SoS" }, @@ -326,6 +336,7 @@ { "cell_type": "code", "execution_count": null, + "id": "65a58c1d", "metadata": { "kernel": "Bash" }, @@ -341,6 +352,7 @@ }, { "cell_type": "markdown", + "id": "8bf7deb5", "metadata": { "kernel": "SoS" }, @@ -351,6 +363,7 @@ { "cell_type": "code", "execution_count": null, + "id": "827ef89d", "metadata": { "kernel": "SoS" }, @@ -361,6 +374,7 @@ }, { "cell_type": "markdown", + "id": "aa41bddb", "metadata": { "kernel": "SoS" }, @@ -446,6 +460,7 @@ }, { "cell_type": "markdown", + "id": "726d4932", "metadata": { "kernel": "SoS" }, @@ -458,6 +473,7 @@ { "cell_type": "code", "execution_count": null, + "id": "32c2a4ef", "metadata": { "kernel": "SoS" }, @@ -500,6 +516,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e5f886cc", "metadata": { "kernel": "SoS" }, @@ -542,6 +559,7 @@ }, { "cell_type": "markdown", + "id": "72ca164d", "metadata": { "kernel": "SoS" }, @@ -552,6 +570,7 @@ { "cell_type": "code", "execution_count": null, + "id": "21002c4e", "metadata": { "kernel": "SoS" }, @@ -592,6 +611,7 @@ { "cell_type": "code", "execution_count": null, + "id": "cafa3767", "metadata": { "kernel": "SoS" }, @@ -650,5 +670,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/multivariate_genome/MASH/mixture_prior.ipynb b/code/SoS/multivariate_genome/MASH/mixture_prior.ipynb index 64284e73f..4855d502f 100644 --- a/code/SoS/multivariate_genome/MASH/mixture_prior.ipynb +++ b/code/SoS/multivariate_genome/MASH/mixture_prior.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "01291ec0", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "5dff8b62", "metadata": { "kernel": "SoS" }, @@ -35,6 +37,7 @@ }, { "cell_type": "markdown", + "id": "43000362", "metadata": { "kernel": "SoS" }, @@ -78,6 +81,7 @@ }, { "cell_type": "markdown", + "id": "2a5878f9", "metadata": { "kernel": "SoS" }, @@ -104,6 +108,7 @@ }, { "cell_type": "markdown", + "id": "e4961525", "metadata": { "kernel": "SoS" }, @@ -113,6 +118,7 @@ }, { "cell_type": "markdown", + "id": "51952d0b", "metadata": { "kernel": "SoS" }, @@ -124,6 +130,7 @@ }, { "cell_type": "markdown", + "id": "76309c0f", "metadata": { "kernel": "SoS" }, @@ -135,6 +142,7 @@ }, { "cell_type": "markdown", + "id": "2137da15", "metadata": { "kernel": "SoS" }, @@ -145,6 +153,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4ff8271a", "metadata": { "kernel": "Bash" }, @@ -158,6 +167,7 @@ }, { "cell_type": "markdown", + "id": "00eafecf", "metadata": { "kernel": "SoS" }, @@ -169,6 +179,7 @@ }, { "cell_type": "markdown", + "id": "e36acd9a", "metadata": { "kernel": "SoS" }, @@ -179,6 +190,7 @@ { "cell_type": "code", "execution_count": null, + "id": "07b5596b", "metadata": { "kernel": "Bash" }, @@ -192,6 +204,7 @@ }, { "cell_type": "markdown", + "id": "b583a5cc", "metadata": { "kernel": "SoS" }, @@ -203,6 +216,7 @@ }, { "cell_type": "markdown", + "id": "b327aae0", "metadata": { "kernel": "SoS" }, @@ -213,6 +227,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f79ab394", "metadata": { "kernel": "Bash" }, @@ -226,6 +241,7 @@ }, { "cell_type": "markdown", + "id": "ce2f83da", "metadata": { "kernel": "SoS" }, @@ -237,6 +253,7 @@ }, { "cell_type": "markdown", + "id": "36ff6360", "metadata": { "kernel": "SoS" }, @@ -247,6 +264,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e576919a", "metadata": { "kernel": "Bash" }, @@ -260,6 +278,7 @@ }, { "cell_type": "markdown", + "id": "da71a9d7", "metadata": { "kernel": "SoS" }, @@ -271,6 +290,7 @@ }, { "cell_type": "markdown", + "id": "e3cd9506", "metadata": { "kernel": "SoS" }, @@ -282,6 +302,7 @@ }, { "cell_type": "markdown", + "id": "7cd740ff", "metadata": { "kernel": "SoS" }, @@ -292,6 +313,7 @@ { "cell_type": "code", "execution_count": null, + "id": "67ac347e", "metadata": { "kernel": "Bash" }, @@ -305,6 +327,7 @@ }, { "cell_type": "markdown", + "id": "255f44fc", "metadata": { "kernel": "SoS" }, @@ -316,6 +339,7 @@ }, { "cell_type": "markdown", + "id": "68f39335", "metadata": { "kernel": "SoS" }, @@ -326,6 +350,7 @@ { "cell_type": "code", "execution_count": null, + "id": "dfd5c549", "metadata": { "kernel": "Bash" }, @@ -339,6 +364,7 @@ }, { "cell_type": "markdown", + "id": "94346cbd", "metadata": { "kernel": "SoS" }, @@ -350,6 +376,7 @@ }, { "cell_type": "markdown", + "id": "e3f2de49", "metadata": { "kernel": "SoS" }, @@ -360,6 +387,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e4b7ac06", "metadata": { "kernel": "Bash" }, @@ -373,6 +401,7 @@ }, { "cell_type": "markdown", + "id": "8512e399", "metadata": { "kernel": "SoS" }, @@ -384,6 +413,7 @@ }, { "cell_type": "markdown", + "id": "60d1e648", "metadata": { "kernel": "SoS" }, @@ -394,6 +424,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5e00657d", "metadata": { "kernel": "Bash" }, @@ -407,6 +438,7 @@ }, { "cell_type": "markdown", + "id": "ffc6ecfc", "metadata": { "kernel": "SoS" }, @@ -418,6 +450,7 @@ }, { "cell_type": "markdown", + "id": "53026ecb", "metadata": { "kernel": "SoS" }, @@ -428,6 +461,7 @@ { "cell_type": "code", "execution_count": null, + "id": "7e48714f", "metadata": { "kernel": "Bash" }, @@ -441,6 +475,7 @@ }, { "cell_type": "markdown", + "id": "298486a5", "metadata": { "kernel": "SoS" }, @@ -452,6 +487,7 @@ }, { "cell_type": "markdown", + "id": "47deb4c5", "metadata": { "kernel": "SoS" }, @@ -463,6 +499,7 @@ }, { "cell_type": "markdown", + "id": "b5bddfab", "metadata": { "kernel": "SoS" }, @@ -473,6 +510,7 @@ { "cell_type": "code", "execution_count": null, + "id": "ba2d7195", "metadata": { "kernel": "Bash" }, @@ -486,6 +524,7 @@ }, { "cell_type": "markdown", + "id": "13fc2954", "metadata": { "kernel": "SoS" }, @@ -497,6 +536,7 @@ }, { "cell_type": "markdown", + "id": "fd8d0b07", "metadata": { "kernel": "SoS" }, @@ -507,6 +547,7 @@ { "cell_type": "code", "execution_count": null, + "id": "c0db82e3", "metadata": { "kernel": "Bash" }, @@ -520,6 +561,7 @@ }, { "cell_type": "markdown", + "id": "943218f6", "metadata": { "kernel": "SoS" }, @@ -531,6 +573,7 @@ }, { "cell_type": "markdown", + "id": "4a9c48a0", "metadata": { "kernel": "SoS" }, @@ -541,6 +584,7 @@ { "cell_type": "code", "execution_count": null, + "id": "d8bbb0cb", "metadata": { "kernel": "Bash" }, @@ -554,6 +598,7 @@ }, { "cell_type": "markdown", + "id": "5754efc6", "metadata": { "kernel": "SoS" }, @@ -565,6 +610,7 @@ }, { "cell_type": "markdown", + "id": "27e2be45", "metadata": { "kernel": "SoS" }, @@ -576,6 +622,7 @@ }, { "cell_type": "markdown", + "id": "f88fe829", "metadata": { "kernel": "SoS" }, @@ -586,6 +633,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b03a9bc2", "metadata": { "kernel": "Bash" }, @@ -599,6 +647,7 @@ }, { "cell_type": "markdown", + "id": "6712a10b", "metadata": { "kernel": "SoS" }, @@ -609,6 +658,7 @@ { "cell_type": "code", "execution_count": null, + "id": "388e931f", "metadata": { "kernel": "Bash" }, @@ -619,6 +669,7 @@ }, { "cell_type": "markdown", + "id": "05ef61d6", "metadata": { "kernel": "SoS" }, @@ -718,6 +769,7 @@ }, { "cell_type": "markdown", + "id": "c880bb77", "metadata": { "kernel": "SoS" }, @@ -727,6 +779,7 @@ }, { "cell_type": "markdown", + "id": "cabab930", "metadata": { "kernel": "SoS", "tags": [] @@ -738,6 +791,7 @@ { "cell_type": "code", "execution_count": 2, + "id": "d8926180", "metadata": { "kernel": "SoS" }, @@ -778,6 +832,7 @@ }, { "cell_type": "markdown", + "id": "64abbeda", "metadata": { "kernel": "Bash" }, @@ -788,6 +843,7 @@ { "cell_type": "code", "execution_count": null, + "id": "8aa711c8", "metadata": { "kernel": "SoS" }, @@ -808,6 +864,7 @@ { "cell_type": "code", "execution_count": null, + "id": "cd6d53d0", "metadata": { "kernel": "SoS" }, @@ -828,6 +885,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2fdf5a18", "metadata": { "kernel": "SoS" }, @@ -852,6 +910,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b3dfdff9", "metadata": { "kernel": "SoS" }, @@ -871,6 +930,7 @@ }, { "cell_type": "markdown", + "id": "51ffb527", "metadata": { "kernel": "SoS" }, @@ -881,6 +941,7 @@ { "cell_type": "code", "execution_count": 6, + "id": "0e905980", "metadata": { "kernel": "SoS" }, @@ -902,6 +963,7 @@ { "cell_type": "code", "execution_count": 7, + "id": "17b068af", "metadata": { "kernel": "SoS" }, @@ -923,6 +985,7 @@ { "cell_type": "code", "execution_count": 8, + "id": "cd2e3c4c", "metadata": { "kernel": "SoS" }, @@ -951,6 +1014,7 @@ { "cell_type": "code", "execution_count": null, + "id": "92f0e445", "metadata": { "kernel": "SoS" }, @@ -972,6 +1036,7 @@ { "cell_type": "code", "execution_count": null, + "id": "51ebefec", "metadata": { "kernel": "SoS" }, @@ -992,6 +1057,7 @@ }, { "cell_type": "markdown", + "id": "2c1992cc", "metadata": { "kernel": "SoS", "tags": [] @@ -1003,6 +1069,7 @@ { "cell_type": "code", "execution_count": null, + "id": "08192614", "metadata": { "kernel": "SoS" }, @@ -1027,6 +1094,7 @@ { "cell_type": "code", "execution_count": null, + "id": "246c78b9", "metadata": { "kernel": "SoS" }, @@ -1051,6 +1119,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9831015a", "metadata": { "kernel": "SoS" }, @@ -1074,6 +1143,7 @@ }, { "cell_type": "markdown", + "id": "c4ed1227", "metadata": { "kernel": "SoS", "tags": [] @@ -1087,6 +1157,7 @@ { "cell_type": "code", "execution_count": null, + "id": "14c7b07f", "metadata": { "kernel": "SoS" }, @@ -1142,5 +1213,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 + "nbformat_minor": 5 } diff --git a/code/SoS/multivariate_genome/METAL/METAL.ipynb b/code/SoS/multivariate_genome/METAL/METAL.ipynb index 5eebc9b8a..b5735bbd1 100644 --- a/code/SoS/multivariate_genome/METAL/METAL.ipynb +++ b/code/SoS/multivariate_genome/METAL/METAL.ipynb @@ -6,7 +6,19 @@ "metadata": { "kernel": "SoS" }, - "source": "# Meta-analysis with METAL\n\n> **Under active development.** This module is a work in progress: its interface, parameters and outputs may still change, and it is not yet covered by the automated test suite.\n\nThis notebook runs cross-cohort meta-analysis of summary statistics with **METAL** on the toy `protocol_example` dataset. METAL is a command-line tool that takes a script documenting the input summary-statistic files, the field mapping for each, and the analysis settings. Meta-analysis here is essentially a weighted sum of Z-scores, so the same set of variants must be present across the input cohorts; the input is a list of paths to the per-cohort summary statistics to analyse together.\n\nBy default the weighting **SCHEME** is set to standard error (`STDERR`); switching to sample-size weighting would require a sample-size column in the upstream input. The `AVERAGEFREQ`, `MINMAXFREQ`, and `GENOMICCONTROL` options are off by default — enabling any of them requires the corresponding additional column.\n\nThe workflow is split into four numbered steps (`METAL_1` … `METAL_4`); the commands below run them in order on the toy input `protocol_example.sumstat_list.tsv` (a tab-separated list with a `#chr` column and one cohort column `protocol_example`). The referenced sumstats carry `chr, pos, A1, A2, beta, se, z, p`, so effect/SE/p-value/alleles are mapped to `beta`/`se`/`p`/`A1` `A2`; with no frequency or sample-size column the default STDERR scheme is used. Results go to `output/metal`, no container.\n\n**Role in the protocol.** METAL is the meta-analysis module: it consumes the per-cohort summary statistics produced by QTL association and combines them. It produces two outputs: (1) a list of meta-analysed summary statistics for downstream MASH and mvSuSiE-RSS analysis, and (2) a list of summary statistics in VCF format that serves as the end product of the analysis." + "source": [ + "# Meta-analysis with METAL\n", + "\n", + "> **Under active development.** This module is a work in progress: its interface, parameters and outputs may still change, and it is not yet covered by the automated test suite.\n", + "\n", + "This notebook runs cross-cohort meta-analysis of summary statistics with **METAL** on the toy `protocol_example` dataset. METAL is a command-line tool that takes a script documenting the input summary-statistic files, the field mapping for each, and the analysis settings. Meta-analysis here is essentially a weighted sum of Z-scores, so the same set of variants must be present across the input cohorts; the input is a list of paths to the per-cohort summary statistics to analyse together.\n", + "\n", + "By default the weighting **SCHEME** is set to standard error (`STDERR`); switching to sample-size weighting would require a sample-size column in the upstream input. The `AVERAGEFREQ`, `MINMAXFREQ`, and `GENOMICCONTROL` options are off by default — enabling any of them requires the corresponding additional column.\n", + "\n", + "The workflow is split into four numbered steps (`METAL_1` … `METAL_4`); the commands below run them in order on the toy input `protocol_example.sumstat_list.tsv` (a tab-separated list with a `#chr` column and one cohort column `protocol_example`). The referenced sumstats carry `chr, pos, A1, A2, beta, se, z, p`, so effect/SE/p-value/alleles are mapped to `beta`/`se`/`p`/`A1` `A2`; with no frequency or sample-size column the default STDERR scheme is used. Results go to `output/metal`, no container.\n", + "\n", + "**Role in the protocol.** METAL is the meta-analysis module: it consumes the per-cohort summary statistics produced by QTL association and combines them. It produces two outputs: (1) a list of meta-analysed summary statistics for downstream MASH and mvSuSiE-RSS analysis, and (2) a list of summary statistics in VCF format that serves as the end product of the analysis." + ] }, { "cell_type": "markdown", @@ -39,11 +51,11 @@ }, { "cell_type": "markdown", + "id": "68da14d3", "metadata": {}, "source": [ "**Timing:** Runtime varies by dataset size and compute resources. For the toy chr22 MWE dataset, most steps complete in under 10 minutes on a standard HPC node." - ], - "id": "68da14d3" + ] }, { "cell_type": "code", @@ -131,11 +143,13 @@ }, { "cell_type": "markdown", + "id": "5481599f", "metadata": {}, "source": [ - "## Anticipated Results\n\nThe pipeline produces output files in the `output/` subdirectory named after the workflow step. Verify success by checking that output files exist and are non-empty. See the **Output** section above for the expected file names and formats." - ], - "id": "5481599f" + "## Anticipated Results\n", + "\n", + "The pipeline produces output files in the `output/` subdirectory named after the workflow step. Verify success by checking that output files exist and are non-empty. See the **Output** section above for the expected file names and formats." + ] }, { "cell_type": "code", @@ -301,4 +315,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/multivariate_genome/multivariate_mixture_vignette.ipynb b/code/SoS/multivariate_genome/multivariate_mixture_vignette.ipynb index f8fd97485..b40bdf345 100644 --- a/code/SoS/multivariate_genome/multivariate_mixture_vignette.ipynb +++ b/code/SoS/multivariate_genome/multivariate_mixture_vignette.ipynb @@ -67,7 +67,7 @@ "kernel": "SoS" }, "source": [ - "### Step 1. [Estimate the mixture prior](https://statfungen.github.io/xqtl-protocol/code/multivariate_genome/MASH/mixture_prior.html)\n", + "### Step 1. [Estimate the mixture prior](https://statfungen.github.io/xqtl-protocol/mixture-prior)\n", "\n", "`ed_bovy` estimates data-driven covariance components and assembles them with canonical components." ] @@ -104,7 +104,7 @@ "kernel": "SoS" }, "source": [ - "### Step 2. [Fit MASH](https://statfungen.github.io/xqtl-protocol/code/multivariate_genome/MASH/mash_fit.html)\n", + "### Step 2. [Fit MASH](https://statfungen.github.io/xqtl-protocol/mash-fit)\n", "\n", "`mash` combines the input effects, residual correlation and estimated prior to compute posterior summaries." ] @@ -324,4 +324,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/pecotmr_integration/SuSiE_enloc.ipynb b/code/SoS/pecotmr_integration/SuSiE_enloc.ipynb index e66573abc..c114e6635 100644 --- a/code/SoS/pecotmr_integration/SuSiE_enloc.ipynb +++ b/code/SoS/pecotmr_integration/SuSiE_enloc.ipynb @@ -653,4 +653,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/pecotmr_integration/gwas_integration.ipynb b/code/SoS/pecotmr_integration/gwas_integration.ipynb index b56da011f..02d8133af 100644 --- a/code/SoS/pecotmr_integration/gwas_integration.ipynb +++ b/code/SoS/pecotmr_integration/gwas_integration.ipynb @@ -6,7 +6,10 @@ "metadata": { "kernel": "SoS" }, - "source": "# GWAS and xQTL integration mini-protocol\nIntegrate molecular-QTL results with GWAS using enrichment, colocalization, TWAS, cTWAS, quantile TWAS, or INTACT." + "source": [ + "# GWAS and xQTL integration mini-protocol\n", + "Integrate molecular-QTL results with GWAS using enrichment, colocalization, TWAS, cTWAS, quantile TWAS, or INTACT." + ] }, { "cell_type": "markdown", @@ -14,7 +17,11 @@ "metadata": { "kernel": "SoS" }, - "source": "#### Miniprotocol Timing\nThis is the total duration for the selected route; module-specific timings appear on their respective pages.\nTiming: TBD" + "source": [ + "#### Miniprotocol Timing\n", + "This is the total duration for the selected route; module-specific timings appear on their respective pages.\n", + "Timing: TBD" + ] }, { "cell_type": "markdown", @@ -22,7 +29,11 @@ "metadata": { "kernel": "SoS" }, - "source": "## Overview\nThis mini-protocol presents complementary routes for connecting molecular-QTL and GWAS evidence. Steps 1–2 call [`SuSiE_enloc.ipynb`](https://statfungen.github.io/xqtl-protocol/code/pecotmr_integration/SuSiE_enloc.html), steps 3–7 call [`twas_ctwas.ipynb`](https://statfungen.github.io/xqtl-protocol/code/pecotmr_integration/twas_ctwas.html), and step 8 calls [`intact.ipynb`](https://statfungen.github.io/xqtl-protocol/code/pecotmr_integration/intact.html).\nEnrichment, colocalization, standard TWAS, cTWAS, quantile TWAS, and INTACT answer different questions and are not one mandatory chain. Only cTWAS steps 4–6 must be run sequentially." + "source": [ + "## Overview\n", + "This mini-protocol presents complementary routes for connecting molecular-QTL and GWAS evidence. Steps 1–2 call [`SuSiE_enloc.ipynb`](https://statfungen.github.io/xqtl-protocol/susie-enloc), steps 3–7 call [`twas_ctwas.ipynb`](https://statfungen.github.io/xqtl-protocol/twas-ctwas), and step 8 calls [`intact.ipynb`](https://statfungen.github.io/xqtl-protocol/intact).\n", + "Enrichment, colocalization, standard TWAS, cTWAS, quantile TWAS, and INTACT answer different questions and are not one mandatory chain. Only cTWAS steps 4–6 must be run sequentially." + ] }, { "cell_type": "markdown", @@ -30,7 +41,20 @@ "metadata": { "kernel": "SoS" }, - "source": "## Steps\nChoose a route before running commands; the commands are not one mandatory chain.\n| Analysis goal | Commands to run, in order | Inputs |\n|---|---|---|\n| Estimate global xQTL–GWAS enrichment | 1 | `tests/fixtures/susie_enloc/protocol_example.enloc.gwas_meta.tsv`; `tests/fixtures/susie_enloc/protocol_example.enloc.xqtl_meta.tsv`; fine-mapping objects referenced by those tables |\n| Test pairwise colocalization | 2 | The same GWAS and xQTL fine-mapping metadata as step 1 |\n| Run standard TWAS and MR | 3 | `tests/fixtures/twas/protocol_example.twas.gwas_meta.tsv`; `tests/fixtures/twas/protocol_example.twas.xqtl_meta.tsv`; `tests/fixtures/ld_reference/ld_meta_file.tsv`; `tests/fixtures/twas/protocol_example.twas.LD_blocks.chr22.bed`; `tests/fixtures/twas/protocol_example.twas.data_type_table.txt` |\n| Fine-map genes and SNPs jointly with cTWAS | 4 → 5 → 6 | GWAS, xQTL, LD-reference, and region metadata used by the TWAS route |\n| Run quantile TWAS | 7 | The TWAS inputs, with quantile-specific molecular-QTL weights |\n| Combine PTWAS and fastenloc evidence | 8 | `tests/fixtures/intact/protocol_example.ptwas.output`; `tests/fixtures/intact/protocol_example.fastenloc.gene.out` |\n\nRun only the commands for the selected route. Steps 4–6 are one chained cTWAS analysis; the other numbered steps are independent alternatives." + "source": [ + "## Steps\n", + "Choose a route before running commands; the commands are not one mandatory chain.\n", + "| Analysis goal | Commands to run, in order | Inputs |\n", + "|---|---|---|\n", + "| Estimate global xQTL–GWAS enrichment | 1 | `tests/fixtures/susie_enloc/protocol_example.enloc.gwas_meta.tsv`; `tests/fixtures/susie_enloc/protocol_example.enloc.xqtl_meta.tsv`; fine-mapping objects referenced by those tables |\n", + "| Test pairwise colocalization | 2 | The same GWAS and xQTL fine-mapping metadata as step 1 |\n", + "| Run standard TWAS and MR | 3 | `tests/fixtures/twas/protocol_example.twas.gwas_meta.tsv`; `tests/fixtures/twas/protocol_example.twas.xqtl_meta.tsv`; `tests/fixtures/ld_reference/ld_meta_file.tsv`; `tests/fixtures/twas/protocol_example.twas.LD_blocks.chr22.bed`; `tests/fixtures/twas/protocol_example.twas.data_type_table.txt` |\n", + "| Fine-map genes and SNPs jointly with cTWAS | 4 → 5 → 6 | GWAS, xQTL, LD-reference, and region metadata used by the TWAS route |\n", + "| Run quantile TWAS | 7 | The TWAS inputs, with quantile-specific molecular-QTL weights |\n", + "| Combine PTWAS and fastenloc evidence | 8 | `tests/fixtures/intact/protocol_example.ptwas.output`; `tests/fixtures/intact/protocol_example.fastenloc.gene.out` |\n", + "\n", + "Run only the commands for the selected route. Steps 4–6 are one chained cTWAS analysis; the other numbered steps are independent alternatives." + ] }, { "cell_type": "markdown", @@ -38,7 +62,11 @@ "metadata": { "kernel": "SoS" }, - "source": "### 1. [Estimate global xQTL–GWAS enrichment](https://statfungen.github.io/xqtl-protocol/code/pecotmr_integration/SuSiE_enloc.html)\n\n**What it does:** Estimates whether fine-mapped xQTL signals are enriched among GWAS signals across many regions and molecular contexts." + "source": [ + "### 1. [Estimate global xQTL–GWAS enrichment](https://statfungen.github.io/xqtl-protocol/susie-enloc)\n", + "\n", + "**What it does:** Estimates whether fine-mapped xQTL signals are enriched among GWAS signals across many regions and molecular contexts." + ] }, { "cell_type": "markdown", @@ -79,7 +107,11 @@ "metadata": { "kernel": "SoS" }, - "source": "### 2. [Test pairwise colocalization](https://statfungen.github.io/xqtl-protocol/code/pecotmr_integration/SuSiE_enloc.html)\n\n**What it does:** Tests whether an xQTL and GWAS association in the same region are consistent with a shared causal variant." + "source": [ + "### 2. [Test pairwise colocalization](https://statfungen.github.io/xqtl-protocol/susie-enloc)\n", + "\n", + "**What it does:** Tests whether an xQTL and GWAS association in the same region are consistent with a shared causal variant." + ] }, { "cell_type": "markdown", @@ -122,7 +154,11 @@ "metadata": { "kernel": "SoS" }, - "source": "### 3. [Run standard TWAS and MR](https://statfungen.github.io/xqtl-protocol/code/pecotmr_integration/twas_ctwas.html)\n\n**What it does:** Tests genetically predicted molecular traits for association with the GWAS trait and reports regional Mendelian-randomization results." + "source": [ + "### 3. [Run standard TWAS and MR](https://statfungen.github.io/xqtl-protocol/twas-ctwas)\n", + "\n", + "**What it does:** Tests genetically predicted molecular traits for association with the GWAS trait and reports regional Mendelian-randomization results." + ] }, { "cell_type": "markdown", @@ -161,7 +197,11 @@ "metadata": { "kernel": "SoS" }, - "source": "### 4. [Assemble cTWAS region data](https://statfungen.github.io/xqtl-protocol/code/pecotmr_integration/twas_ctwas.html)\n\n**What it does:** Harmonizes GWAS statistics, xQTL weights, and LD for the regions that will enter cTWAS fine-mapping." + "source": [ + "### 4. [Assemble cTWAS region data](https://statfungen.github.io/xqtl-protocol/twas-ctwas)\n", + "\n", + "**What it does:** Harmonizes GWAS statistics, xQTL weights, and LD for the regions that will enter cTWAS fine-mapping." + ] }, { "cell_type": "markdown", @@ -199,7 +239,11 @@ "metadata": { "kernel": "SoS" }, - "source": "### 5. [Estimate cTWAS global parameters](https://statfungen.github.io/xqtl-protocol/code/pecotmr_integration/twas_ctwas.html)\n\n**What it does:** Reuses the assembled inputs from step 4 to estimate the prior parameters required by cTWAS." + "source": [ + "### 5. [Estimate cTWAS global parameters](https://statfungen.github.io/xqtl-protocol/twas-ctwas)\n", + "\n", + "**What it does:** Reuses the assembled inputs from step 4 to estimate the prior parameters required by cTWAS." + ] }, { "cell_type": "markdown", @@ -236,7 +280,11 @@ "metadata": { "kernel": "SoS" }, - "source": "### 6. [Fine-map genes and SNPs with cTWAS](https://statfungen.github.io/xqtl-protocol/code/pecotmr_integration/twas_ctwas.html)\n\n**What it does:** Uses the assembled inputs and estimated parameters to calculate joint posterior inclusion probabilities for genes and SNPs." + "source": [ + "### 6. [Fine-map genes and SNPs with cTWAS](https://statfungen.github.io/xqtl-protocol/twas-ctwas)\n", + "\n", + "**What it does:** Uses the assembled inputs and estimated parameters to calculate joint posterior inclusion probabilities for genes and SNPs." + ] }, { "cell_type": "markdown", @@ -274,7 +322,11 @@ "metadata": { "kernel": "SoS" }, - "source": "### 7. [Run quantile TWAS](https://statfungen.github.io/xqtl-protocol/code/pecotmr_integration/twas_ctwas.html)\n\n**What it does:** Runs the quantile-TWAS route for molecular-trait effects that may differ across the phenotype distribution." + "source": [ + "### 7. [Run quantile TWAS](https://statfungen.github.io/xqtl-protocol/twas-ctwas)\n", + "\n", + "**What it does:** Runs the quantile-TWAS route for molecular-trait effects that may differ across the phenotype distribution." + ] }, { "cell_type": "markdown", @@ -312,7 +364,11 @@ "metadata": { "kernel": "SoS" }, - "source": "### 8. [Combine PTWAS and colocalization with INTACT](https://statfungen.github.io/xqtl-protocol/code/pecotmr_integration/intact.html)\n\n**What it does:** Combines PTWAS z-scores with fastenloc colocalization probabilities to prioritize genes supported by both association and colocalization evidence." + "source": [ + "### 8. [Combine PTWAS and colocalization with INTACT](https://statfungen.github.io/xqtl-protocol/intact)\n", + "\n", + "**What it does:** Combines PTWAS z-scores with fastenloc colocalization probabilities to prioritize genes supported by both association and colocalization evidence." + ] }, { "cell_type": "markdown", @@ -385,7 +441,9 @@ "metadata": { "kernel": "SoS" }, - "source": "## Command Interface" + "source": [ + "## Command Interface" + ] }, { "cell_type": "code", @@ -453,4 +511,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/pecotmr_integration/intact.ipynb b/code/SoS/pecotmr_integration/intact.ipynb index 043e5afb0..2567f8160 100644 --- a/code/SoS/pecotmr_integration/intact.ipynb +++ b/code/SoS/pecotmr_integration/intact.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "c20a7e0e", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "f461c44d", "metadata": { "kernel": "SoS" }, @@ -30,6 +32,7 @@ }, { "cell_type": "markdown", + "id": "c00d98c5", "metadata": { "kernel": "SoS" }, @@ -47,6 +50,7 @@ }, { "cell_type": "markdown", + "id": "4bae9b8f", "metadata": { "kernel": "SoS" }, @@ -69,6 +73,7 @@ }, { "cell_type": "markdown", + "id": "91c8421c", "metadata": { "kernel": "SoS" }, @@ -80,6 +85,7 @@ }, { "cell_type": "markdown", + "id": "7ffc3e0f", "metadata": { "kernel": "SoS" }, @@ -90,6 +96,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5ef5a2f5", "metadata": { "kernel": "Bash" }, @@ -104,6 +111,7 @@ }, { "cell_type": "markdown", + "id": "21cc2162", "metadata": { "kernel": "SoS" }, @@ -113,6 +121,7 @@ }, { "cell_type": "markdown", + "id": "b5f976db", "metadata": { "kernel": "SoS" }, @@ -123,6 +132,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6467dcb2", "metadata": { "kernel": "Bash" }, @@ -133,6 +143,7 @@ }, { "cell_type": "markdown", + "id": "d96d5b6a", "metadata": { "kernel": "SoS" }, @@ -175,6 +186,7 @@ }, { "cell_type": "markdown", + "id": "b0bd7cc1", "metadata": { "kernel": "SoS" }, @@ -185,6 +197,7 @@ { "cell_type": "code", "execution_count": null, + "id": "bd26be1f", "metadata": { "kernel": "SoS" }, @@ -214,6 +227,7 @@ { "cell_type": "code", "execution_count": null, + "id": "73cfde0e", "metadata": { "kernel": "SoS" }, @@ -267,5 +281,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 + "nbformat_minor": 5 } diff --git a/code/SoS/pecotmr_integration/twas_ctwas.ipynb b/code/SoS/pecotmr_integration/twas_ctwas.ipynb index 7f3ad423a..5375550fa 100644 --- a/code/SoS/pecotmr_integration/twas_ctwas.ipynb +++ b/code/SoS/pecotmr_integration/twas_ctwas.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "1eb1aaa4", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "057f97b3", "metadata": { "kernel": "SoS" }, @@ -46,6 +48,7 @@ }, { "cell_type": "markdown", + "id": "e503f923", "metadata": { "kernel": "SoS" }, @@ -124,6 +127,7 @@ }, { "cell_type": "markdown", + "id": "9fe5f332", "metadata": { "kernel": "SoS", "tags": [] @@ -162,6 +166,7 @@ }, { "cell_type": "markdown", + "id": "3f9aaba0", "metadata": { "kernel": "SoS" }, @@ -173,6 +178,7 @@ }, { "cell_type": "markdown", + "id": "413ff733", "metadata": { "kernel": "SoS" }, @@ -184,6 +190,7 @@ }, { "cell_type": "markdown", + "id": "0d0e90e8", "metadata": { "kernel": "SoS" }, @@ -194,6 +201,7 @@ { "cell_type": "code", "execution_count": null, + "id": "f2d86139", "metadata": { "kernel": "SoS" }, @@ -213,6 +221,7 @@ }, { "cell_type": "markdown", + "id": "73ff8d29", "metadata": { "kernel": "SoS" }, @@ -228,6 +237,7 @@ }, { "cell_type": "markdown", + "id": "731ec9db", "metadata": { "kernel": "SoS" }, @@ -237,6 +247,7 @@ }, { "cell_type": "markdown", + "id": "6b52b5ca", "metadata": { "kernel": "SoS" }, @@ -247,6 +258,7 @@ { "cell_type": "code", "execution_count": null, + "id": "53dfded6", "metadata": { "kernel": "SoS" }, @@ -265,6 +277,7 @@ }, { "cell_type": "markdown", + "id": "b099e0d2", "metadata": { "kernel": "SoS" }, @@ -274,6 +287,7 @@ }, { "cell_type": "markdown", + "id": "00260551", "metadata": { "kernel": "SoS" }, @@ -284,6 +298,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2632af31", "metadata": { "kernel": "SoS" }, @@ -301,6 +316,7 @@ }, { "cell_type": "markdown", + "id": "7644b78a", "metadata": { "kernel": "SoS" }, @@ -310,6 +326,7 @@ }, { "cell_type": "markdown", + "id": "accb9340", "metadata": { "kernel": "SoS" }, @@ -320,6 +337,7 @@ { "cell_type": "code", "execution_count": null, + "id": "618870aa", "metadata": { "kernel": "SoS" }, @@ -338,6 +356,7 @@ }, { "cell_type": "markdown", + "id": "92273464", "metadata": { "kernel": "SoS" }, @@ -348,6 +367,7 @@ }, { "cell_type": "markdown", + "id": "431fac0a", "metadata": { "kernel": "SoS" }, @@ -358,6 +378,7 @@ { "cell_type": "code", "execution_count": null, + "id": "2ee607ee", "metadata": { "kernel": "SoS" }, @@ -376,6 +397,7 @@ }, { "cell_type": "markdown", + "id": "8a850938", "metadata": { "kernel": "SoS" }, @@ -386,6 +408,7 @@ { "cell_type": "code", "execution_count": null, + "id": "3508e1e2", "metadata": { "kernel": "SoS" }, @@ -396,6 +419,7 @@ }, { "cell_type": "markdown", + "id": "97846bb3", "metadata": { "kernel": "SoS" }, @@ -550,6 +574,7 @@ }, { "cell_type": "markdown", + "id": "a4d62976", "metadata": { "kernel": "SoS" }, @@ -562,6 +587,7 @@ { "cell_type": "code", "execution_count": null, + "id": "7984b144", "metadata": { "kernel": "SoS" }, @@ -598,6 +624,7 @@ { "cell_type": "code", "execution_count": null, + "id": "3ac0d81e", "metadata": { "kernel": "SoS", "vscode": { @@ -634,6 +661,7 @@ { "cell_type": "code", "execution_count": null, + "id": "270a7389", "metadata": { "kernel": "SoS", "tags": [] @@ -730,6 +758,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b31b6650", "metadata": { "kernel": "SoS", "tags": [], @@ -822,6 +851,7 @@ { "cell_type": "code", "execution_count": null, + "id": "ebe6be80", "metadata": { "kernel": "SoS", "vscode": { @@ -864,6 +894,7 @@ { "cell_type": "code", "execution_count": null, + "id": "ac8f6d8b", "metadata": { "editable": true, "kernel": "SoS", @@ -920,6 +951,7 @@ { "cell_type": "code", "execution_count": null, + "id": "80c18f9c", "metadata": { "kernel": "SoS" }, @@ -1111,5 +1143,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/pecotmr_integration/twas_vignette.ipynb b/code/SoS/pecotmr_integration/twas_vignette.ipynb index b0710f320..df4f67616 100644 --- a/code/SoS/pecotmr_integration/twas_vignette.ipynb +++ b/code/SoS/pecotmr_integration/twas_vignette.ipynb @@ -59,7 +59,7 @@ "id": "2921ebb4", "metadata": {}, "source": [ - "### [Standard TWAS](https://statfungen.github.io/xqtl-protocol/code/pecotmr_integration/twas_ctwas.html#standard-twas-scan)\n", + "### [Standard TWAS](https://statfungen.github.io/xqtl-protocol/twas-ctwas#standard-twas-scan)\n", "\n", "The `twas` workflow calculates one association result per modeled molecular trait." ] @@ -102,7 +102,7 @@ "id": "aa80f390", "metadata": {}, "source": [ - "### [cTWAS](https://statfungen.github.io/xqtl-protocol/code/pecotmr_integration/twas_ctwas.html#ctwas-fine-mapping)\n", + "### [cTWAS](https://statfungen.github.io/xqtl-protocol/twas-ctwas#ctwas-fine-mapping)\n", "\n", "The `ctwas` workflow runs its three dependent stages in order: input assembly, parameter estimation and joint fine-mapping." ] @@ -258,4 +258,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/rare_xqtl/watershed.ipynb b/code/SoS/rare_xqtl/watershed.ipynb index cbba4a520..b870200f0 100644 --- a/code/SoS/rare_xqtl/watershed.ipynb +++ b/code/SoS/rare_xqtl/watershed.ipynb @@ -28,7 +28,11 @@ "metadata": { "kernel": "SoS" }, - "source": "## \n\n> **Under active development.** This module is a work in progress: its interface, parameters and outputs may still change, and it is not yet covered by the automated test suite." + "source": [ + "## \n", + "\n", + "> **Under active development.** This module is a work in progress: its interface, parameters and outputs may still change, and it is not yet covered by the automated test suite." + ] }, { "cell_type": "code", @@ -171,4 +175,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/reference_data/generalized_TADB.ipynb b/code/SoS/reference_data/generalized_TADB.ipynb index 3705d11d8..1087352a2 100644 --- a/code/SoS/reference_data/generalized_TADB.ipynb +++ b/code/SoS/reference_data/generalized_TADB.ipynb @@ -421,4 +421,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/reference_data/ld_prune_reference.ipynb b/code/SoS/reference_data/ld_prune_reference.ipynb index 1fcde941c..a2aef9a48 100644 --- a/code/SoS/reference_data/ld_prune_reference.ipynb +++ b/code/SoS/reference_data/ld_prune_reference.ipynb @@ -295,4 +295,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/reference_data/reference_data.ipynb b/code/SoS/reference_data/reference_data.ipynb index b47a856cb..39bb5d50f 100644 --- a/code/SoS/reference_data/reference_data.ipynb +++ b/code/SoS/reference_data/reference_data.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "5c330c5a", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "f3ec096d", "metadata": { "jp-MarkdownHeadingCollapsed": true, "kernel": "SoS" @@ -27,19 +29,21 @@ }, { "cell_type": "markdown", + "id": "d8b3a7ae", "metadata": { "kernel": "SoS" }, "source": [ "## Overview\n", "\n", - "This mini-protocol walks through how the reference data used throughout the xQTL Protocol are downloaded, formatted, and indexed. It is an introductory, top-to-bottom guide: each step calls a single workflow from [`reference_data_preparation.ipynb`](https://statfungen.github.io/xqtl-protocol/code/reference_data/reference_data_preparation.html), with additional workflows from [`VCF_QC.ipynb`](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/VCF_QC.html), [`generalized_TADB.ipynb`](https://statfungen.github.io/xqtl-protocol/code/reference_data/generalized_TADB.html), and [`rss_ld_sketch.ipynb`](https://statfungen.github.io/xqtl-protocol/code/reference_data/rss_ld_sketch.html).\n", + "This mini-protocol walks through how the reference data used throughout the xQTL Protocol are downloaded, formatted, and indexed. It is an introductory, top-to-bottom guide: each step calls a single workflow from [`reference_data_preparation.ipynb`](https://statfungen.github.io/xqtl-protocol/reference-data-preparation), with additional workflows from [`VCF_QC.ipynb`](https://statfungen.github.io/xqtl-protocol/vcf-qc), [`generalized_TADB.ipynb`](https://statfungen.github.io/xqtl-protocol/generalized-tadb), and [`rss_ld_sketch.ipynb`](https://statfungen.github.io/xqtl-protocol/rss-ld-sketch).\n", "\n", "The commands are organized as selectable routes rather than one mandatory 13-step chain. Steps 1–10 prepare the core genome and the RNA-seq, RSEM, or Picard resources needed by the selected route. Steps 11–13 are independent optional workflows for dbSNP annotation, TAD-based association windows, and RSS LD sketches.\n" ] }, { "cell_type": "markdown", + "id": "7c93601d", "metadata": { "kernel": "SoS" }, @@ -63,11 +67,12 @@ }, { "cell_type": "markdown", + "id": "86e26e90", "metadata": { "kernel": "SoS" }, "source": [ - "### 1. [Download the human genome](https://statfungen.github.io/xqtl-protocol/code/reference_data/reference_data_preparation.html)\n", + "### 1. [Download the human genome](https://statfungen.github.io/xqtl-protocol/reference-data-preparation)\n", "\n", "**What it does:** Download the GRCh38 reference FASTA used by sequence-aware tools.\n" ] @@ -75,6 +80,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9e6b0e8d", "metadata": { "kernel": "SoS" }, @@ -85,11 +91,12 @@ }, { "cell_type": "markdown", + "id": "7e88575c", "metadata": { "kernel": "SoS" }, "source": [ - "### 2. [Download the gene annotation](https://statfungen.github.io/xqtl-protocol/code/reference_data/reference_data_preparation.html)\n", + "### 2. [Download the gene annotation](https://statfungen.github.io/xqtl-protocol/reference-data-preparation)\n", "\n", "**What it does:** Download the Ensembl gene annotation used to define genes and transcripts.\n" ] @@ -97,6 +104,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9e7baf10", "metadata": { "kernel": "SoS" }, @@ -107,11 +115,12 @@ }, { "cell_type": "markdown", + "id": "5b419e08", "metadata": { "kernel": "SoS" }, "source": [ - "### 3. [Download the ERCC reference](https://statfungen.github.io/xqtl-protocol/code/reference_data/reference_data_preparation.html)\n", + "### 3. [Download the ERCC reference](https://statfungen.github.io/xqtl-protocol/reference-data-preparation)\n", "\n", "**What it does:** Download ERCC spike-in sequences and annotations for RNA-seq reference construction.\n" ] @@ -119,6 +128,7 @@ { "cell_type": "code", "execution_count": null, + "id": "10a97960", "metadata": { "kernel": "SoS" }, @@ -129,11 +139,12 @@ }, { "cell_type": "markdown", + "id": "40f676aa", "metadata": { "kernel": "SoS" }, "source": [ - "### 4. [Download dbSNP](https://statfungen.github.io/xqtl-protocol/code/reference_data/reference_data_preparation.html)\n", + "### 4. [Download dbSNP](https://statfungen.github.io/xqtl-protocol/reference-data-preparation)\n", "\n", "**What it does:** Download the dbSNP variant resource used when rsID annotation is required.\n" ] @@ -141,6 +152,7 @@ { "cell_type": "code", "execution_count": null, + "id": "c72c400f", "metadata": { "kernel": "SoS" }, @@ -151,11 +163,12 @@ }, { "cell_type": "markdown", + "id": "b9d1dc04", "metadata": { "kernel": "SoS" }, "source": [ - "### 5. [Format and index the genome](https://statfungen.github.io/xqtl-protocol/code/reference_data/reference_data_preparation.html)\n", + "### 5. [Format and index the genome](https://statfungen.github.io/xqtl-protocol/reference-data-preparation)\n", "\n", "**What it does:** Remove unsupported alternate sequences, append ERCC sequences and create FASTA indices.\n" ] @@ -163,6 +176,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5208bbdf", "metadata": { "kernel": "SoS" }, @@ -173,11 +187,12 @@ }, { "cell_type": "markdown", + "id": "f4021ad6", "metadata": { "kernel": "SoS" }, "source": [ - "### 6. [Format the gene annotation](https://statfungen.github.io/xqtl-protocol/code/reference_data/reference_data_preparation.html)\n", + "### 6. [Format the gene annotation](https://statfungen.github.io/xqtl-protocol/reference-data-preparation)\n", "\n", "**What it does:** Add chromosome prefixes and construct the protocol-compatible gene models.\n" ] @@ -185,6 +200,7 @@ { "cell_type": "code", "execution_count": null, + "id": "39874dff", "metadata": { "kernel": "SoS" }, @@ -195,11 +211,12 @@ }, { "cell_type": "markdown", + "id": "3bc53ec1", "metadata": { "kernel": "SoS" }, "source": [ - "### 7. [Combine gene and ERCC annotations](https://statfungen.github.io/xqtl-protocol/code/reference_data/reference_data_preparation.html)\n", + "### 7. [Combine gene and ERCC annotations](https://statfungen.github.io/xqtl-protocol/reference-data-preparation)\n", "\n", "**What it does:** Combine the processed human and ERCC annotations into the GTF used downstream.\n" ] @@ -207,6 +224,7 @@ { "cell_type": "code", "execution_count": null, + "id": "7f8b3987", "metadata": { "kernel": "SoS" }, @@ -217,11 +235,12 @@ }, { "cell_type": "markdown", + "id": "b7234328", "metadata": { "kernel": "SoS" }, "source": [ - "### 8. [Build the STAR index](https://statfungen.github.io/xqtl-protocol/code/reference_data/reference_data_preparation.html)\n", + "### 8. [Build the STAR index](https://statfungen.github.io/xqtl-protocol/reference-data-preparation)\n", "\n", "**What it does:** Build the STAR genome index used for RNA-seq alignment.\n" ] @@ -229,6 +248,7 @@ { "cell_type": "code", "execution_count": null, + "id": "08dbddf3", "metadata": { "kernel": "SoS" }, @@ -239,11 +259,12 @@ }, { "cell_type": "markdown", + "id": "99dbed2d", "metadata": { "kernel": "SoS" }, "source": [ - "### 9. [Build the RSEM index](https://statfungen.github.io/xqtl-protocol/code/reference_data/reference_data_preparation.html)\n", + "### 9. [Build the RSEM index](https://statfungen.github.io/xqtl-protocol/reference-data-preparation)\n", "\n", "**What it does:** Build the RSEM reference used for transcript-level expression quantification.\n" ] @@ -251,6 +272,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6997c10c", "metadata": { "kernel": "SoS" }, @@ -261,11 +283,12 @@ }, { "cell_type": "markdown", + "id": "67eb1065", "metadata": { "kernel": "SoS" }, "source": [ - "### 10. [Generate RefFlat annotation](https://statfungen.github.io/xqtl-protocol/code/reference_data/reference_data_preparation.html)\n", + "### 10. [Generate RefFlat annotation](https://statfungen.github.io/xqtl-protocol/reference-data-preparation)\n", "\n", "**What it does:** Convert the processed GTF into the RefFlat annotation used by Picard RNA-seq QC.\n" ] @@ -273,6 +296,7 @@ { "cell_type": "code", "execution_count": null, + "id": "8e516ce9", "metadata": { "kernel": "SoS" }, @@ -283,11 +307,12 @@ }, { "cell_type": "markdown", + "id": "5680ea39", "metadata": { "kernel": "SoS" }, "source": [ - "### 11. [Annotate a study VCF with dbSNP identifiers (optional)](https://statfungen.github.io/xqtl-protocol/code/data_preprocessing/genotype/VCF_QC.html)\n", + "### 11. [Annotate a study VCF with dbSNP identifiers (optional)](https://statfungen.github.io/xqtl-protocol/vcf-qc)\n", "\n", "**What it does:** Add rsIDs to a study VCF when downstream tools require named variants.\n" ] @@ -295,6 +320,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b5949196", "metadata": { "kernel": "SoS" }, @@ -307,11 +333,12 @@ }, { "cell_type": "markdown", + "id": "cd583d65", "metadata": { "kernel": "SoS" }, "source": [ - "### 12. [Construct generalized TAD boundaries (optional)](https://statfungen.github.io/xqtl-protocol/code/reference_data/generalized_TADB.html)\n", + "### 12. [Construct generalized TAD boundaries (optional)](https://statfungen.github.io/xqtl-protocol/generalized-tadb)\n", "\n", "**What it does:** Combine tissue-specific TAD calls into customized cis-association windows.\n" ] @@ -319,6 +346,7 @@ { "cell_type": "code", "execution_count": null, + "id": "05a28a02", "metadata": { "kernel": "SoS" }, @@ -332,11 +360,12 @@ }, { "cell_type": "markdown", + "id": "ce96b2c7", "metadata": { "kernel": "SoS" }, "source": [ - "### 13. [Construct an LD sketch for RSS analysis (optional)](https://statfungen.github.io/xqtl-protocol/code/reference_data/rss_ld_sketch.html)\n", + "### 13. [Construct an LD sketch for RSS analysis (optional)](https://statfungen.github.io/xqtl-protocol/rss-ld-sketch)\n", "\n", "**What it does:** Generate, process and merge LD-sketch products for summary-statistics regression.\n" ] @@ -344,6 +373,7 @@ { "cell_type": "code", "execution_count": null, + "id": "95834300", "metadata": { "kernel": "SoS" }, @@ -358,6 +388,7 @@ }, { "cell_type": "markdown", + "id": "cab3721c", "metadata": { "kernel": "SoS" }, @@ -379,6 +410,7 @@ }, { "cell_type": "markdown", + "id": "6c7aeaaa", "metadata": { "kernel": "SoS" }, @@ -394,6 +426,7 @@ }, { "cell_type": "markdown", + "id": "1c10a56d", "metadata": { "kernel": "SoS" }, @@ -406,6 +439,7 @@ { "cell_type": "code", "execution_count": null, + "id": "8fe3d495", "metadata": { "kernel": "Bash" }, @@ -443,5 +477,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 + "nbformat_minor": 5 } diff --git a/code/SoS/reference_data/reference_data_preparation.ipynb b/code/SoS/reference_data/reference_data_preparation.ipynb index ead8e2d24..c158dde0d 100644 --- a/code/SoS/reference_data/reference_data_preparation.ipynb +++ b/code/SoS/reference_data/reference_data_preparation.ipynb @@ -2,6 +2,7 @@ "cells": [ { "cell_type": "markdown", + "id": "bef929e6", "metadata": { "kernel": "SoS" }, @@ -13,6 +14,7 @@ }, { "cell_type": "markdown", + "id": "db09417b", "metadata": { "kernel": "SoS" }, @@ -33,6 +35,7 @@ }, { "cell_type": "markdown", + "id": "0aaa20b5", "metadata": { "kernel": "SoS" }, @@ -43,6 +46,7 @@ }, { "cell_type": "markdown", + "id": "bab10203", "metadata": { "kernel": "SoS" }, @@ -78,6 +82,7 @@ }, { "cell_type": "markdown", + "id": "732c836b", "metadata": { "kernel": "SoS" }, @@ -88,6 +93,7 @@ }, { "cell_type": "markdown", + "id": "d63c319a", "metadata": { "kernel": "SoS" }, @@ -124,6 +130,7 @@ }, { "cell_type": "markdown", + "id": "eaf57754", "metadata": { "kernel": "SoS" }, @@ -168,6 +175,7 @@ }, { "cell_type": "markdown", + "id": "cac79ec8", "metadata": { "kernel": "SoS" }, @@ -177,6 +185,7 @@ }, { "cell_type": "markdown", + "id": "47dca9d7", "metadata": { "kernel": "SoS" }, @@ -193,6 +202,7 @@ }, { "cell_type": "markdown", + "id": "4d0b35e1", "metadata": { "kernel": "SoS" }, @@ -204,6 +214,7 @@ }, { "cell_type": "markdown", + "id": "05e8af30", "metadata": { "kernel": "SoS" }, @@ -214,6 +225,7 @@ { "cell_type": "code", "execution_count": null, + "id": "3ccb1100", "metadata": { "kernel": "Bash" }, @@ -224,6 +236,7 @@ }, { "cell_type": "markdown", + "id": "841c14be", "metadata": { "kernel": "SoS" }, @@ -235,6 +248,7 @@ }, { "cell_type": "markdown", + "id": "277e2244", "metadata": { "kernel": "SoS" }, @@ -245,6 +259,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4ea69445", "metadata": { "kernel": "Bash" }, @@ -255,6 +270,7 @@ }, { "cell_type": "markdown", + "id": "023c365a", "metadata": { "kernel": "SoS" }, @@ -266,6 +282,7 @@ }, { "cell_type": "markdown", + "id": "98a7d64f", "metadata": { "kernel": "SoS" }, @@ -276,6 +293,7 @@ { "cell_type": "code", "execution_count": null, + "id": "ee6ec7b2", "metadata": { "kernel": "Bash" }, @@ -286,6 +304,7 @@ }, { "cell_type": "markdown", + "id": "f6f204f9", "metadata": { "kernel": "SoS" }, @@ -297,6 +316,7 @@ }, { "cell_type": "markdown", + "id": "81b2c630", "metadata": { "kernel": "SoS" }, @@ -307,6 +327,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4748e4dc", "metadata": { "kernel": "Bash" }, @@ -317,6 +338,7 @@ }, { "cell_type": "markdown", + "id": "c767efbc", "metadata": { "kernel": "SoS" }, @@ -328,6 +350,7 @@ }, { "cell_type": "markdown", + "id": "b31c9876", "metadata": { "kernel": "SoS" }, @@ -338,6 +361,7 @@ { "cell_type": "code", "execution_count": null, + "id": "4b818cb7", "metadata": { "kernel": "Bash" }, @@ -351,6 +375,7 @@ }, { "cell_type": "markdown", + "id": "3211e49e", "metadata": { "kernel": "SoS" }, @@ -362,6 +387,7 @@ }, { "cell_type": "markdown", + "id": "3a231700", "metadata": { "kernel": "SoS" }, @@ -372,6 +398,7 @@ { "cell_type": "code", "execution_count": null, + "id": "498a079a", "metadata": { "kernel": "Bash" }, @@ -386,6 +413,7 @@ }, { "cell_type": "markdown", + "id": "c504bbe7", "metadata": { "kernel": "SoS" }, @@ -397,6 +425,7 @@ }, { "cell_type": "markdown", + "id": "2720f3b3", "metadata": { "kernel": "SoS" }, @@ -407,6 +436,7 @@ { "cell_type": "code", "execution_count": null, + "id": "11f1ac12", "metadata": { "kernel": "Bash" }, @@ -422,6 +452,7 @@ }, { "cell_type": "markdown", + "id": "b262df91", "metadata": { "kernel": "SoS" }, @@ -433,6 +464,7 @@ }, { "cell_type": "markdown", + "id": "6b972aaa", "metadata": { "kernel": "SoS" }, @@ -443,6 +475,7 @@ { "cell_type": "code", "execution_count": null, + "id": "5693ed9b", "metadata": { "kernel": "Bash" }, @@ -457,6 +490,7 @@ }, { "cell_type": "markdown", + "id": "d6020dba", "metadata": { "kernel": "SoS" }, @@ -468,6 +502,7 @@ }, { "cell_type": "markdown", + "id": "fcdfc7e7", "metadata": { "kernel": "SoS" }, @@ -478,6 +513,7 @@ { "cell_type": "code", "execution_count": null, + "id": "d4f8a577", "metadata": { "kernel": "Bash" }, @@ -491,6 +527,7 @@ }, { "cell_type": "markdown", + "id": "f62fe6b0", "metadata": { "kernel": "SoS" }, @@ -502,6 +539,7 @@ }, { "cell_type": "markdown", + "id": "c6563e58", "metadata": { "kernel": "SoS" }, @@ -512,6 +550,7 @@ { "cell_type": "code", "execution_count": null, + "id": "baa5e35c", "metadata": { "kernel": "Bash" }, @@ -524,6 +563,7 @@ }, { "cell_type": "markdown", + "id": "705349b6", "metadata": { "kernel": "SoS" }, @@ -536,6 +576,7 @@ }, { "cell_type": "markdown", + "id": "ce7660b5", "metadata": {}, "source": [ "#### Convert GFF3 to GTF\n", @@ -545,6 +586,7 @@ }, { "cell_type": "markdown", + "id": "1c3bd60e", "metadata": {}, "source": [ "**Timing**: TBD (on toy dataset)" @@ -553,6 +595,7 @@ { "cell_type": "code", "execution_count": null, + "id": "687b9a57", "metadata": { "kernel": "Bash" }, @@ -565,6 +608,7 @@ }, { "cell_type": "markdown", + "id": "c136f682", "metadata": {}, "source": [ "#### Patch the ERCC annotation\n", @@ -574,6 +618,7 @@ }, { "cell_type": "markdown", + "id": "84ce5271", "metadata": {}, "source": [ "**Timing**: TBD (on toy dataset)" @@ -582,6 +627,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9fd43c2a", "metadata": { "kernel": "Bash" }, @@ -594,6 +640,7 @@ }, { "cell_type": "markdown", + "id": "84368757", "metadata": {}, "source": [ "#### TAD annotation\n", @@ -603,6 +650,7 @@ }, { "cell_type": "markdown", + "id": "32e9eec4", "metadata": {}, "source": [ "**Timing**: TBD (on toy dataset)" @@ -611,6 +659,7 @@ { "cell_type": "code", "execution_count": null, + "id": "58120e42", "metadata": { "kernel": "Bash" }, @@ -625,6 +674,7 @@ }, { "cell_type": "markdown", + "id": "3c4185d6", "metadata": { "kernel": "SoS" }, @@ -635,6 +685,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6391f30c", "metadata": { "kernel": "SoS" }, @@ -645,6 +696,7 @@ }, { "cell_type": "markdown", + "id": "26275f06", "metadata": { "kernel": "SoS" }, @@ -735,6 +787,7 @@ }, { "cell_type": "markdown", + "id": "41decc3e", "metadata": { "kernel": "SoS" }, @@ -747,6 +800,7 @@ { "cell_type": "code", "execution_count": null, + "id": "e20d05db", "metadata": { "kernel": "SoS" }, @@ -773,6 +827,7 @@ { "cell_type": "code", "execution_count": null, + "id": "73b8d9bf", "metadata": { "kernel": "SoS" }, @@ -787,6 +842,7 @@ { "cell_type": "code", "execution_count": null, + "id": "a7b2c088", "metadata": { "kernel": "SoS" }, @@ -801,6 +857,7 @@ { "cell_type": "code", "execution_count": 1, + "id": "dbfa142b", "metadata": { "kernel": "SoS" }, @@ -815,6 +872,7 @@ { "cell_type": "code", "execution_count": null, + "id": "1f0f77b7", "metadata": { "kernel": "SoS" }, @@ -830,6 +888,7 @@ { "cell_type": "code", "execution_count": null, + "id": "c703099b", "metadata": { "kernel": "SoS" }, @@ -849,6 +908,7 @@ { "cell_type": "code", "execution_count": null, + "id": "752ac736", "metadata": { "kernel": "SoS" }, @@ -870,6 +930,7 @@ { "cell_type": "code", "execution_count": null, + "id": "d019f4d0", "metadata": { "kernel": "SoS" }, @@ -887,6 +948,7 @@ { "cell_type": "code", "execution_count": null, + "id": "1074f77d", "metadata": { "kernel": "SoS", "tags": [] @@ -908,6 +970,7 @@ { "cell_type": "code", "execution_count": 4, + "id": "256b4688", "metadata": { "kernel": "SoS" }, @@ -930,6 +993,7 @@ { "cell_type": "code", "execution_count": 5, + "id": "a561a3f8", "metadata": { "kernel": "SoS" }, @@ -949,6 +1013,7 @@ { "cell_type": "code", "execution_count": 6, + "id": "7e61b7cd", "metadata": { "kernel": "SoS" }, @@ -969,6 +1034,7 @@ { "cell_type": "code", "execution_count": 7, + "id": "5669752e", "metadata": { "kernel": "SoS" }, @@ -986,6 +1052,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6343b3e7", "metadata": { "kernel": "SoS" }, @@ -1017,6 +1084,7 @@ { "cell_type": "code", "execution_count": null, + "id": "1b6c6a98", "metadata": { "kernel": "SoS" }, @@ -1042,6 +1110,7 @@ { "cell_type": "code", "execution_count": null, + "id": "6d73d261", "metadata": { "kernel": "SoS" }, @@ -1061,6 +1130,7 @@ { "cell_type": "code", "execution_count": null, + "id": "9d55189d", "metadata": { "kernel": "SoS" }, @@ -1119,5 +1189,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 + "nbformat_minor": 5 } diff --git a/code/SoS/reference_data/rss_ld_sketch.ipynb b/code/SoS/reference_data/rss_ld_sketch.ipynb index 9438b17d1..83aea407a 100644 --- a/code/SoS/reference_data/rss_ld_sketch.ipynb +++ b/code/SoS/reference_data/rss_ld_sketch.ipynb @@ -19,7 +19,27 @@ "kernel": "SoS" }, "source": [ - "## Overview\n\nThis pipeline generates a stochastic genotype sample **U = WᵀG** from whole-genome sequencing VCF files and stores it as a PLINK2 pgen file for use as an LD reference panel with SuSiE-RSS fine-mapping.\n\nRather than storing the full genotype matrix G (n × p), the sketch computes U = WᵀG (B × p) using a random projection matrix $W \\sim N(0, 1/\\sqrt{n})$. The approximate LD matrix $R = U^T U / B \\approx G^T G / n$ by the Johnson–Lindenstrauss lemma. G is never stored.\n\n**Matrix dimensions:**\n- G : (n × p) — n individuals × p variants\n- W : (n × B) — projection matrix, generated once per cohort\n- U : (B × p) — stochastic genotype sample = WᵀG, stored in pgen\n- $\\hat{R}$ : (p × p) — approximate LD matrix, computed on-the-fly by SuSiE-RSS from U\n\nThe workflow has three steps run in order: `generate_W` (build the projection matrix), `process_block` (read VCF per LD block and write per-block dosage sketches), and `merge_chrom` (merge per-block dosages into one per-chromosome pgen). During `merge_chrom`, allele-frequency rows are reconciled to the final sorted `.pvar`; missing or duplicate IDs stop the workflow, while extra block rows are reported and excluded. During `merge_chrom`, the step also detects indels that form an exact REF/ALT-swap (“mirror”) pair at the same position — the ambiguous cases where insertion vs. deletion anchoring can flip effect-allele orientation during association harmonization. Only these mirror-pair indels are assigned a canonical directional event ID (e.g. `chr22:pos:INS:GAT` and `chr22:pos:DEL:GAT`); the event ID replaces the variant ID in `.pvar` and `.afreq`, and the original↔event mapping is recorded in a `.event_id.tsv`. All other variants — SNPs, equal-length substitutions, and non-mirror indels — keep their standard IDs. If a panel contains no mirror pairs, no `.event_id.tsv` is written and `.pvar`/`.afreq` are left fully standard. The `.pgen` genotype data is never modified.\n\nSummary-statistic fine-mapping needs an LD matrix, and storing the full genotype matrix for\na whole-genome panel is expensive. The sketch keeps a random projection of the genotypes\ninstead: small enough to distribute, but sufficient to reconstruct the LD structure that\nSuSiE-RSS actually uses.\n\n**When to run it.** Once per reference panel, before any `rss_analysis` run that needs an\nLD reference. The output is reference data, not a per-study result." + "## Overview\n", + "\n", + "This pipeline generates a stochastic genotype sample **U = WᵀG** from whole-genome sequencing VCF files and stores it as a PLINK2 pgen file for use as an LD reference panel with SuSiE-RSS fine-mapping.\n", + "\n", + "Rather than storing the full genotype matrix G (n × p), the sketch computes U = WᵀG (B × p) using a random projection matrix $W \\sim N(0, 1/\\sqrt{n})$. The approximate LD matrix $R = U^T U / B \\approx G^T G / n$ by the Johnson–Lindenstrauss lemma. G is never stored.\n", + "\n", + "**Matrix dimensions:**\n", + "- G : (n × p) — n individuals × p variants\n", + "- W : (n × B) — projection matrix, generated once per cohort\n", + "- U : (B × p) — stochastic genotype sample = WᵀG, stored in pgen\n", + "- $\\hat{R}$ : (p × p) — approximate LD matrix, computed on-the-fly by SuSiE-RSS from U\n", + "\n", + "The workflow has three steps run in order: `generate_W` (build the projection matrix), `process_block` (read VCF per LD block and write per-block dosage sketches), and `merge_chrom` (merge per-block dosages into one per-chromosome pgen). During `merge_chrom`, allele-frequency rows are reconciled to the final sorted `.pvar`; missing or duplicate IDs stop the workflow, while extra block rows are reported and excluded. During `merge_chrom`, the step also detects indels that form an exact REF/ALT-swap (“mirror”) pair at the same position — the ambiguous cases where insertion vs. deletion anchoring can flip effect-allele orientation during association harmonization. Only these mirror-pair indels are assigned a canonical directional event ID (e.g. `chr22:pos:INS:GAT` and `chr22:pos:DEL:GAT`); the event ID replaces the variant ID in `.pvar` and `.afreq`, and the original↔event mapping is recorded in a `.event_id.tsv`. All other variants — SNPs, equal-length substitutions, and non-mirror indels — keep their standard IDs. If a panel contains no mirror pairs, no `.event_id.tsv` is written and `.pvar`/`.afreq` are left fully standard. The `.pgen` genotype data is never modified.\n", + "\n", + "Summary-statistic fine-mapping needs an LD matrix, and storing the full genotype matrix for\n", + "a whole-genome panel is expensive. The sketch keeps a random projection of the genotypes\n", + "instead: small enough to distribute, but sufficient to reconstruct the LD structure that\n", + "SuSiE-RSS actually uses.\n", + "\n", + "**When to run it.** Once per reference panel, before any `rss_analysis` run that needs an\n", + "LD reference. The output is reference data, not a per-study result." ] }, { @@ -556,7 +576,84 @@ }, "outputs": [], "source": [ - "[merge_chrom]\nparameter: chrom = 0\nparameter: output_dir = str\nparameter: cohort_id = str\nparameter: plink2_bin = \"plink2\"\n\nimport os, glob\n\nif chrom != 0:\n chroms = [f\"chr{chrom}\"]\nelse:\n chroms = sorted(set(\n os.path.basename(_d)\n for _d in glob.glob(os.path.join(output_dir, \"chr*\"))\n if os.path.isdir(_d)\n ))\n\ninput: for_each = \"chroms\"\noutput: f\"{output_dir}/{_chroms}/{cohort_id}.{_chroms}.pgen\"\ntask: trunk_workers = 1, trunk_size = 1, walltime = walltime, mem = mem, cores = numThreads\nbash: expand = \"$[ ]\"\n\n set -euo pipefail\n shopt -s nullglob\n\n chrom_dir=\"$[output_dir]/$[_chroms]\"\n final_prefix=\"${chrom_dir}/$[cohort_id].$[_chroms]\"\n merge_list=\"${chrom_dir}/$[cohort_id].$[_chroms]_pmerge_list.txt\"\n\n # Step 1: Convert each block dosage.gz -> sorted per-block pgen\n > \"${merge_list}\"\n files=(\"${chrom_dir}\"/*/*.dosage.gz)\n if [ ${#files[@]} -eq 0 ]; then\n echo \"No dosage files found in ${chrom_dir}\" >&2\n exit 1\n fi\n for dosage_gz in \"${files[@]}\"; do\n block_dir=$(dirname \"${dosage_gz}\")\n block_tag=$(basename \"${block_dir}\")\n prefix=\"${block_dir}/$[cohort_id].${block_tag}_tmp\"\n map_file=\"${block_dir}/$[cohort_id].${block_tag}.map\"\n psam_file=\"${block_dir}/$[cohort_id].${block_tag}.psam\"\n meta_file=\"${block_dir}/$[cohort_id].${block_tag}.meta\"\n\n B=$(grep \"^B=\" \"${meta_file}\" | cut -d= -f2)\n printf '#FID\\tIID\\n' > \"${psam_file}\"\n for i in $(seq 1 ${B}); do\n printf 'S%d\\tS%d\\n' ${i} ${i} >> \"${psam_file}\"\n done\n\n $[plink2_bin] \\\n --import-dosage \"${dosage_gz}\" format=1 noheader \\\n --psam \"${psam_file}\" \\\n --map \"${map_file}\" \\\n --make-pgen \\\n --out \"${prefix}_unsorted\" \\\n --silent\n\n $[plink2_bin] \\\n --pfile \"${prefix}_unsorted\" \\\n --make-pgen \\\n --sort-vars \\\n --out \"${prefix}\" \\\n --silent\n\n rm -f \"${prefix}_unsorted.pgen\" \"${prefix}_unsorted.pvar\" \"${prefix}_unsorted.psam\"\n echo \"${prefix}\" >> \"${merge_list}\"\n done\n\n # Step 2: Merge all per-block pgens -> one per-chrom pgen\n $[plink2_bin] \\\n --pmerge-list \"${merge_list}\" pfile \\\n --make-pgen \\\n --sort-vars \\\n --out \"${final_prefix}\"\n\n # Step 3: reconcile .afreq, relabel mirror-pair indels with event IDs, summarize filters, and cleanup.\n Rscript $[modular_script_dir]/reference_data/rss_ld_sketch.R --step merge_chrom --chrom-dir \"$[output_dir]/$[_chroms]\" --final-prefix \"$[output_dir]/$[_chroms]/$[cohort_id].$[_chroms]\"\n" + "[merge_chrom]\n", + "parameter: chrom = 0\n", + "parameter: output_dir = str\n", + "parameter: cohort_id = str\n", + "parameter: plink2_bin = \"plink2\"\n", + "\n", + "import os, glob\n", + "\n", + "if chrom != 0:\n", + " chroms = [f\"chr{chrom}\"]\n", + "else:\n", + " chroms = sorted(set(\n", + " os.path.basename(_d)\n", + " for _d in glob.glob(os.path.join(output_dir, \"chr*\"))\n", + " if os.path.isdir(_d)\n", + " ))\n", + "\n", + "input: for_each = \"chroms\"\n", + "output: f\"{output_dir}/{_chroms}/{cohort_id}.{_chroms}.pgen\"\n", + "task: trunk_workers = 1, trunk_size = 1, walltime = walltime, mem = mem, cores = numThreads\n", + "bash: expand = \"$[ ]\"\n", + "\n", + " set -euo pipefail\n", + " shopt -s nullglob\n", + "\n", + " chrom_dir=\"$[output_dir]/$[_chroms]\"\n", + " final_prefix=\"${chrom_dir}/$[cohort_id].$[_chroms]\"\n", + " merge_list=\"${chrom_dir}/$[cohort_id].$[_chroms]_pmerge_list.txt\"\n", + "\n", + " # Step 1: Convert each block dosage.gz -> sorted per-block pgen\n", + " > \"${merge_list}\"\n", + " files=(\"${chrom_dir}\"/*/*.dosage.gz)\n", + " if [ ${#files[@]} -eq 0 ]; then\n", + " echo \"No dosage files found in ${chrom_dir}\" >&2\n", + " exit 1\n", + " fi\n", + " for dosage_gz in \"${files[@]}\"; do\n", + " block_dir=$(dirname \"${dosage_gz}\")\n", + " block_tag=$(basename \"${block_dir}\")\n", + " prefix=\"${block_dir}/$[cohort_id].${block_tag}_tmp\"\n", + " map_file=\"${block_dir}/$[cohort_id].${block_tag}.map\"\n", + " psam_file=\"${block_dir}/$[cohort_id].${block_tag}.psam\"\n", + " meta_file=\"${block_dir}/$[cohort_id].${block_tag}.meta\"\n", + "\n", + " B=$(grep \"^B=\" \"${meta_file}\" | cut -d= -f2)\n", + " printf '#FID\\tIID\\n' > \"${psam_file}\"\n", + " for i in $(seq 1 ${B}); do\n", + " printf 'S%d\\tS%d\\n' ${i} ${i} >> \"${psam_file}\"\n", + " done\n", + "\n", + " $[plink2_bin] \\\n", + " --import-dosage \"${dosage_gz}\" format=1 noheader \\\n", + " --psam \"${psam_file}\" \\\n", + " --map \"${map_file}\" \\\n", + " --make-pgen \\\n", + " --out \"${prefix}_unsorted\" \\\n", + " --silent\n", + "\n", + " $[plink2_bin] \\\n", + " --pfile \"${prefix}_unsorted\" \\\n", + " --make-pgen \\\n", + " --sort-vars \\\n", + " --out \"${prefix}\" \\\n", + " --silent\n", + "\n", + " rm -f \"${prefix}_unsorted.pgen\" \"${prefix}_unsorted.pvar\" \"${prefix}_unsorted.psam\"\n", + " echo \"${prefix}\" >> \"${merge_list}\"\n", + " done\n", + "\n", + " # Step 2: Merge all per-block pgens -> one per-chrom pgen\n", + " $[plink2_bin] \\\n", + " --pmerge-list \"${merge_list}\" pfile \\\n", + " --make-pgen \\\n", + " --sort-vars \\\n", + " --out \"${final_prefix}\"\n", + "\n", + " # Step 3: reconcile .afreq, relabel mirror-pair indels with event IDs, summarize filters, and cleanup.\n", + " Rscript $[modular_script_dir]/reference_data/rss_ld_sketch.R --step merge_chrom --chrom-dir \"$[output_dir]/$[_chroms]\" --final-prefix \"$[output_dir]/$[_chroms]/$[cohort_id].$[_chroms]\"\n" ] }, { @@ -614,4 +711,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/xqtl_modifier_score/ems_prediction.ipynb b/code/SoS/xqtl_modifier_score/ems_prediction.ipynb index 44ae7842b..e77a8b5cf 100644 --- a/code/SoS/xqtl_modifier_score/ems_prediction.ipynb +++ b/code/SoS/xqtl_modifier_score/ems_prediction.ipynb @@ -4,13 +4,61 @@ "cell_type": "markdown", "id": "a1b2c3d4-e5f6-7890-1234-567890abcdef", "metadata": {}, - "source": "# scEEMS Prediction\n\n> **Under active development.** This module is a work in progress: its interface, parameters and outputs may still change, and it is not yet covered by the automated test suite." + "source": [ + "# scEEMS Prediction\n", + "\n", + "> **Under active development.** This module is a work in progress: its interface, parameters and outputs may still change, and it is not yet covered by the automated test suite." + ] }, { "cell_type": "markdown", + "id": "449b26d5", "metadata": {}, "source": [ - "## Description\n\n## Pipeline Execution Workflow\n\n### Step 1: Input Data Preparation\n\nCreate a tab-separated file with variant identifiers:\n```\nvariant_id\n2:12345:A:T\n2:67890:G:C\n2:11111:T:A\n```\n\n**Format requirements:**\n- Chromosome notation: Numeric (e.g., `2`) or with prefix (e.g., `chr2`)\n- Position: 1-based genomic coordinate (GRCh38/hg38 reference)\n- Alleles: Reference and alternate alleles (ACGT notation)\n\n### Step 2: Execute Prediction Pipeline\n\n```bash\ncd ~/xqtl-protocol/code/SoS/xqtl_modifier_score/\npython model_training_model5_only.py Mic_mega_eQTL 2 \\\n --data-config data_config.yaml \\\n --model-config model_config.yaml\n```\n\n**What happens during execution:**\n1. Variant parsing and coordinate validation\n2. Feature annotation (distance, regulatory, population genetics, conservation, deep learning predictions)\n3. Gene constraint and MAF integration with imputation using training statistics\n4. Feature subsetting and absolute value transformations\n5. Model inference with 10x weighting for regulatory features\n6. Probability calibration and output generation\n\n### Step 3: Output Files Generated\n\n| File | Description | Use Case |\n|------|-------------|----------|\n| `model_standard_subset_weighted_chr_chr2_NPR_10.cbm` | Serialized CatBoost model | Future predictions, model inspection |\n| `predictions_weighted_model_chr2.tsv` | Per-variant predictions with scores | Primary analysis, variant prioritization |\n| `summary_dict_catboost_weighted_model_chr_chr2_NPR_10.pkl` | AP/AUC metrics, class distributions | Model validation, quality control |\n| `features_importance_model5_chr_chr2_NPR_10.csv` | Feature importance rankings | Biological interpretation |" + "## Description\n", + "\n", + "## Pipeline Execution Workflow\n", + "\n", + "### Step 1: Input Data Preparation\n", + "\n", + "Create a tab-separated file with variant identifiers:\n", + "```\n", + "variant_id\n", + "2:12345:A:T\n", + "2:67890:G:C\n", + "2:11111:T:A\n", + "```\n", + "\n", + "**Format requirements:**\n", + "- Chromosome notation: Numeric (e.g., `2`) or with prefix (e.g., `chr2`)\n", + "- Position: 1-based genomic coordinate (GRCh38/hg38 reference)\n", + "- Alleles: Reference and alternate alleles (ACGT notation)\n", + "\n", + "### Step 2: Execute Prediction Pipeline\n", + "\n", + "```bash\n", + "cd ~/xqtl-protocol/code/SoS/xqtl_modifier_score/\n", + "python model_training_model5_only.py Mic_mega_eQTL 2 \\\n", + " --data-config data_config.yaml \\\n", + " --model-config model_config.yaml\n", + "```\n", + "\n", + "**What happens during execution:**\n", + "1. Variant parsing and coordinate validation\n", + "2. Feature annotation (distance, regulatory, population genetics, conservation, deep learning predictions)\n", + "3. Gene constraint and MAF integration with imputation using training statistics\n", + "4. Feature subsetting and absolute value transformations\n", + "5. Model inference with 10x weighting for regulatory features\n", + "6. Probability calibration and output generation\n", + "\n", + "### Step 3: Output Files Generated\n", + "\n", + "| File | Description | Use Case |\n", + "|------|-------------|----------|\n", + "| `model_standard_subset_weighted_chr_chr2_NPR_10.cbm` | Serialized CatBoost model | Future predictions, model inspection |\n", + "| `predictions_weighted_model_chr2.tsv` | Per-variant predictions with scores | Primary analysis, variant prioritization |\n", + "| `summary_dict_catboost_weighted_model_chr_chr2_NPR_10.pkl` | AP/AUC metrics, class distributions | Model validation, quality control |\n", + "| `features_importance_model5_chr_chr2_NPR_10.csv` | Feature importance rankings | Biological interpretation |" ] }, { @@ -28,7 +76,15 @@ "metadata": {}, "outputs": [], "source": [ - "# Inspect the EMS training predictions (functional-score table written by [train]).\nR:\n suppressPackageStartupMessages(library(vroom))\n results_dir <- \"output/xqtl_modifier_score/protocol_example/model_results\"\n results <- vroom(file.path(results_dir, \"predictions_parquet_catboost\",\n \"predictions_weighted_model_chr2.tsv\"), show_col_types = FALSE)\n cat(\"Loaded predictions for\", nrow(results), \"variants\\n\")\n cat(\"Columns: variant_id, standard_subset_weighted_pred_prob (EMS score 0-1),\\n\")\n cat(\" standard_subset_weighted_pred_label (0/1), actual_label\\n\")\n" + "# Inspect the EMS training predictions (functional-score table written by [train]).\n", + "R:\n", + " suppressPackageStartupMessages(library(vroom))\n", + " results_dir <- \"output/xqtl_modifier_score/protocol_example/model_results\"\n", + " results <- vroom(file.path(results_dir, \"predictions_parquet_catboost\",\n", + " \"predictions_weighted_model_chr2.tsv\"), show_col_types = FALSE)\n", + " cat(\"Loaded predictions for\", nrow(results), \"variants\\n\")\n", + " cat(\"Columns: variant_id, standard_subset_weighted_pred_prob (EMS score 0-1),\\n\")\n", + " cat(\" standard_subset_weighted_pred_label (0/1), actual_label\\n\")\n" ] }, { @@ -36,7 +92,23 @@ "id": "example_output_1", "metadata": {}, "source": [ - "**Example output:**\n```\nModel Configuration:\n Algorithm: CatBoostClassifier\n Features: 4839\n Classes: [0, 1]\n Tree depth: 6\n Iterations: 1000\n\nLoaded predictions for 761 variants\n\nAvailable columns:\n - variant_id: Genomic coordinates\n - standard_subset_weighted_pred_prob: EMS functional score (0-1)\n - standard_subset_weighted_pred_label: Binary classification (0/1)\n - actual_label: True label from test set (if available)\n```" + "**Example output:**\n", + "```\n", + "Model Configuration:\n", + " Algorithm: CatBoostClassifier\n", + " Features: 4839\n", + " Classes: [0, 1]\n", + " Tree depth: 6\n", + " Iterations: 1000\n", + "\n", + "Loaded predictions for 761 variants\n", + "\n", + "Available columns:\n", + " - variant_id: Genomic coordinates\n", + " - standard_subset_weighted_pred_prob: EMS functional score (0-1)\n", + " - standard_subset_weighted_pred_label: Binary classification (0/1)\n", + " - actual_label: True label from test set (if available)\n", + "```" ] }, { @@ -44,7 +116,9 @@ "id": "d1e2f3g4-f5a6-7890-1234-567890abcdef", "metadata": {}, "source": [ - "## Statistical Summary of Predictions\n\nThe `standard_subset_weighted_pred_prob` column contains continuous EMS scores representing the probability that each variant has functional regulatory impact on gene expression." + "## Statistical Summary of Predictions\n", + "\n", + "The `standard_subset_weighted_pred_prob` column contains continuous EMS scores representing the probability that each variant has functional regulatory impact on gene expression." ] }, { @@ -54,7 +128,19 @@ "metadata": {}, "outputs": [], "source": [ - "R:\n suppressPackageStartupMessages(library(vroom))\n results <- vroom(\"tests/fixtures/ems_training/expected/predictions_weighted_model_chr2.tsv\", show_col_types = FALSE)\n s <- results$standard_subset_weighted_pred_prob\n cat(\"SCORE DISTRIBUTION ANALYSIS\\n\"); cat(strrep(\"=\", 50), \"\\n\", sep = \"\")\n cat(\"Total variants analyzed:\", length(s), \"\\n\\n\")\n cat(\"Priority Classification:\\n\")\n cat(sprintf(\" High priority (>0.8): %5d (%5.1f%%)\\n\", sum(s > 0.8), 100*mean(s > 0.8)))\n cat(sprintf(\" Medium priority (0.5-0.8): %5d (%5.1f%%)\\n\", sum(s > 0.5 & s <= 0.8), 100*mean(s > 0.5 & s <= 0.8)))\n cat(sprintf(\" Low priority (<=0.5): %5d (%5.1f%%)\\n\", sum(s <= 0.5), 100*mean(s <= 0.5)))\n cat(sprintf(\"\\nScore Statistics: mean=%.4f median=%.4f sd=%.4f min=%.4f max=%.4f\\n\",\n mean(s), median(s), sd(s), min(s), max(s)))\n for (q in c(90, 95, 99)) cat(sprintf(\" %dth percentile: %.4f\\n\", q, quantile(s, q/100)))\n" + "R:\n", + " suppressPackageStartupMessages(library(vroom))\n", + " results <- vroom(\"tests/fixtures/ems_training/expected/predictions_weighted_model_chr2.tsv\", show_col_types = FALSE)\n", + " s <- results$standard_subset_weighted_pred_prob\n", + " cat(\"SCORE DISTRIBUTION ANALYSIS\\n\"); cat(strrep(\"=\", 50), \"\\n\", sep = \"\")\n", + " cat(\"Total variants analyzed:\", length(s), \"\\n\\n\")\n", + " cat(\"Priority Classification:\\n\")\n", + " cat(sprintf(\" High priority (>0.8): %5d (%5.1f%%)\\n\", sum(s > 0.8), 100*mean(s > 0.8)))\n", + " cat(sprintf(\" Medium priority (0.5-0.8): %5d (%5.1f%%)\\n\", sum(s > 0.5 & s <= 0.8), 100*mean(s > 0.5 & s <= 0.8)))\n", + " cat(sprintf(\" Low priority (<=0.5): %5d (%5.1f%%)\\n\", sum(s <= 0.5), 100*mean(s <= 0.5)))\n", + " cat(sprintf(\"\\nScore Statistics: mean=%.4f median=%.4f sd=%.4f min=%.4f max=%.4f\\n\",\n", + " mean(s), median(s), sd(s), min(s), max(s)))\n", + " for (q in c(90, 95, 99)) cat(sprintf(\" %dth percentile: %.4f\\n\", q, quantile(s, q/100)))\n" ] }, { @@ -62,7 +148,29 @@ "id": "example_output_2", "metadata": {}, "source": [ - "**Example output:**\n```\nSCORE DISTRIBUTION ANALYSIS\n==================================================\nTotal variants analyzed: 761\n\nPriority Classification:\n High priority (>0.8): 12 ( 1.6%)\n Medium priority (0.5-0.8): 45 ( 5.9%)\n Low priority (<0.5): 704 ( 92.5%)\n\nScore Statistics:\n Mean: 0.1245\n Median: 0.0823\n Std: 0.1567\n Min: 0.0012\n Max: 0.9234\n\nPercentile Distribution:\n 90th percentile: 0.3421 (76 variants)\n 95th percentile: 0.5012 (38 variants)\n 99th percentile: 0.7834 (8 variants)\n```" + "**Example output:**\n", + "```\n", + "SCORE DISTRIBUTION ANALYSIS\n", + "==================================================\n", + "Total variants analyzed: 761\n", + "\n", + "Priority Classification:\n", + " High priority (>0.8): 12 ( 1.6%)\n", + " Medium priority (0.5-0.8): 45 ( 5.9%)\n", + " Low priority (<0.5): 704 ( 92.5%)\n", + "\n", + "Score Statistics:\n", + " Mean: 0.1245\n", + " Median: 0.0823\n", + " Std: 0.1567\n", + " Min: 0.0012\n", + " Max: 0.9234\n", + "\n", + "Percentile Distribution:\n", + " 90th percentile: 0.3421 (76 variants)\n", + " 95th percentile: 0.5012 (38 variants)\n", + " 99th percentile: 0.7834 (8 variants)\n", + "```" ] }, { @@ -70,7 +178,9 @@ "id": "g1h2i3j4-k5l6-7890-1234-567890abcdef", "metadata": {}, "source": [ - "## Variant Prioritization and Export\n\nExtract and rank high-confidence functional predictions for downstream experimental validation." + "## Variant Prioritization and Export\n", + "\n", + "Extract and rank high-confidence functional predictions for downstream experimental validation." ] }, { @@ -80,7 +190,19 @@ "metadata": {}, "outputs": [], "source": [ - "R:\n suppressPackageStartupMessages(library(vroom))\n results <- vroom(\"tests/fixtures/ems_training/expected/predictions_weighted_model_chr2.tsv\", show_col_types = FALSE)\n hp <- results[results$standard_subset_weighted_pred_prob > 0.8, ]\n hp <- hp[order(-hp$standard_subset_weighted_pred_prob), ]\n cat(\"HIGH-PRIORITY VARIANTS (Score >0.8)\\n\"); cat(strrep(\"=\", 50), \"\\n\", sep = \"\")\n cat(\"Total:\", nrow(hp), \"variants\\n\")\n if (nrow(hp) > 0) {\n cols <- intersect(c(\"variant_id\", \"standard_subset_weighted_pred_prob\", \"actual_label\"), names(hp))\n print(utils::head(hp[, cols], 10))\n vroom::vroom_write(hp, \"high_priority_variants_validation.tsv\", delim = \"\\t\")\n cat(\"\\nExported\", nrow(hp), \"high-priority variants to high_priority_variants_validation.tsv\\n\")\n } else cat(\"\\nNo variants exceed 0.8 threshold\\n\")\n" + "R:\n", + " suppressPackageStartupMessages(library(vroom))\n", + " results <- vroom(\"tests/fixtures/ems_training/expected/predictions_weighted_model_chr2.tsv\", show_col_types = FALSE)\n", + " hp <- results[results$standard_subset_weighted_pred_prob > 0.8, ]\n", + " hp <- hp[order(-hp$standard_subset_weighted_pred_prob), ]\n", + " cat(\"HIGH-PRIORITY VARIANTS (Score >0.8)\\n\"); cat(strrep(\"=\", 50), \"\\n\", sep = \"\")\n", + " cat(\"Total:\", nrow(hp), \"variants\\n\")\n", + " if (nrow(hp) > 0) {\n", + " cols <- intersect(c(\"variant_id\", \"standard_subset_weighted_pred_prob\", \"actual_label\"), names(hp))\n", + " print(utils::head(hp[, cols], 10))\n", + " vroom::vroom_write(hp, \"high_priority_variants_validation.tsv\", delim = \"\\t\")\n", + " cat(\"\\nExported\", nrow(hp), \"high-priority variants to high_priority_variants_validation.tsv\\n\")\n", + " } else cat(\"\\nNo variants exceed 0.8 threshold\\n\")\n" ] }, { @@ -88,7 +210,29 @@ "id": "example_output_3", "metadata": {}, "source": [ - "**Example output:**\n```\nHIGH-PRIORITY VARIANTS (Score >0.8)\n==================================================\nTotal: 12 variants\n\nTop 10 by EMS score:\n variant_id standard_subset_weighted_pred_prob actual_label\n 2:54321:A:G 0.9234 1\n 2:12345:T:C 0.9102 1\n 2:98765:C:A 0.8956 1\n 2:44444:G:T 0.8723 0\n 2:77777:A:C 0.8612 1\n 2:33333:T:G 0.8501 1\n 2:66666:C:T 0.8398 1\n 2:11111:G:A 0.8267 0\n 2:55555:A:T 0.8145 1\n 2:99999:T:A 0.8034 1\n\n Exported to: high_priority_variants_validation.tsv\n Contains all 12 high-priority variants\n Ready for: CRISPR screens, luciferase assays, functional validation\n```" + "**Example output:**\n", + "```\n", + "HIGH-PRIORITY VARIANTS (Score >0.8)\n", + "==================================================\n", + "Total: 12 variants\n", + "\n", + "Top 10 by EMS score:\n", + " variant_id standard_subset_weighted_pred_prob actual_label\n", + " 2:54321:A:G 0.9234 1\n", + " 2:12345:T:C 0.9102 1\n", + " 2:98765:C:A 0.8956 1\n", + " 2:44444:G:T 0.8723 0\n", + " 2:77777:A:C 0.8612 1\n", + " 2:33333:T:G 0.8501 1\n", + " 2:66666:C:T 0.8398 1\n", + " 2:11111:G:A 0.8267 0\n", + " 2:55555:A:T 0.8145 1\n", + " 2:99999:T:A 0.8034 1\n", + "\n", + " Exported to: high_priority_variants_validation.tsv\n", + " Contains all 12 high-priority variants\n", + " Ready for: CRISPR screens, luciferase assays, functional validation\n", + "```" ] }, { @@ -96,7 +240,9 @@ "id": "i1j2k3l4-m5n6-7890-1234-567890abcdef", "metadata": {}, "source": [ - "## Verify Model Performance\n\nReview metrics from the held-out test set to understand model reliability." + "## Verify Model Performance\n", + "\n", + "Review metrics from the held-out test set to understand model reliability." ] }, { @@ -106,7 +252,17 @@ "metadata": {}, "outputs": [], "source": [ - "# Model performance on the held-out test set (written by [train] as summary JSON).\nR:\n suppressPackageStartupMessages(library(jsonlite))\n summ <- fromJSON(\"tests/fixtures/ems_training/expected/model_5_summary_chr_chr2_NPR_1.json\")\n m <- summ$CatBoost$standard_subset_weighted\n cat(\"MODEL PERFORMANCE ON TEST SET\\n\"); cat(strrep(\"=\", 50), \"\\n\", sep = \"\")\n cat(sprintf(\"Average Precision (AP): %.4f\\n\", m$AP_test))\n cat(sprintf(\"AUC-ROC: %.4f\\n\", m$AUC_test))\n npos <- summ$CatBoost$test_num_positive_labels; nneg <- summ$CatBoost$test_num_negative_labels\n cat(sprintf(\"\\nTest set: %d positive, %d negative (%.1f%% positive)\\n\",\n npos, nneg, 100*npos/(npos+nneg)))\n" + "# Model performance on the held-out test set (written by [train] as summary JSON).\n", + "R:\n", + " suppressPackageStartupMessages(library(jsonlite))\n", + " summ <- fromJSON(\"tests/fixtures/ems_training/expected/model_5_summary_chr_chr2_NPR_1.json\")\n", + " m <- summ$CatBoost$standard_subset_weighted\n", + " cat(\"MODEL PERFORMANCE ON TEST SET\\n\"); cat(strrep(\"=\", 50), \"\\n\", sep = \"\")\n", + " cat(sprintf(\"Average Precision (AP): %.4f\\n\", m$AP_test))\n", + " cat(sprintf(\"AUC-ROC: %.4f\\n\", m$AUC_test))\n", + " npos <- summ$CatBoost$test_num_positive_labels; nneg <- summ$CatBoost$test_num_negative_labels\n", + " cat(sprintf(\"\\nTest set: %d positive, %d negative (%.1f%% positive)\\n\",\n", + " npos, nneg, 100*npos/(npos+nneg)))\n" ] }, { @@ -114,7 +270,21 @@ "id": "example_output_4", "metadata": {}, "source": [ - "**Example output:**\n```\nMODEL PERFORMANCE ON TEST SET\n==================================================\nAverage Precision (AP): 0.5050\nAUC-ROC: 0.8978\n\nTest Set Composition:\n Positive labels (functional eQTLs): 68\n Negative labels (non-functional): 693\n Positive rate: 8.9%\n\n These metrics reflect performance on 761 held-out chromosome 2 variants\n with stricter selection criteria (PIP >0.9 for positives) than training data.\n```" + "**Example output:**\n", + "```\n", + "MODEL PERFORMANCE ON TEST SET\n", + "==================================================\n", + "Average Precision (AP): 0.5050\n", + "AUC-ROC: 0.8978\n", + "\n", + "Test Set Composition:\n", + " Positive labels (functional eQTLs): 68\n", + " Negative labels (non-functional): 693\n", + " Positive rate: 8.9%\n", + "\n", + " These metrics reflect performance on 761 held-out chromosome 2 variants\n", + " with stricter selection criteria (PIP >0.9 for positives) than training data.\n", + "```" ] }, { @@ -122,20 +292,50 @@ "id": "k1l2m3n4-m5n6-7890-1234-567890abcdef", "metadata": {}, "source": [ - "## Application to New Variant Lists\n\n### Workflow for Novel Variants\n\nTo score additional variants not included in the original training/test sets:\n\n**1. Prepare input file** matching the format above (`chr:pos:ref:alt`)\n\n**2. Update configuration** (`data_config.yaml`):\n- Point `training_data.base_dir` to directory containing your variant annotations\n- Ensure gene constraint file (GeneBayes scores) is accessible\n- Verify MAF file matches your chromosome\n\n**3. Execute pipeline** (same command, different input data):\n```bash\npython model_training_model5_only.py [cohort] [chromosome] \\\n --data-config data_config.yaml \\\n --model-config model_config.yaml\n```\n\nThe pipeline will:\n- Generate all 4,839 genomic features for your variants\n- Apply the same preprocessing (subsetting, absolute values, imputation)\n- Use the trained model for inference\n- Output predictions in identical format\n\n**4. Analyze results** using the code blocks above" + "## Application to New Variant Lists\n", + "\n", + "### Workflow for Novel Variants\n", + "\n", + "To score additional variants not included in the original training/test sets:\n", + "\n", + "**1. Prepare input file** matching the format above (`chr:pos:ref:alt`)\n", + "\n", + "**2. Update configuration** (`data_config.yaml`):\n", + "- Point `training_data.base_dir` to directory containing your variant annotations\n", + "- Ensure gene constraint file (GeneBayes scores) is accessible\n", + "- Verify MAF file matches your chromosome\n", + "\n", + "**3. Execute pipeline** (same command, different input data):\n", + "```bash\n", + "python model_training_model5_only.py [cohort] [chromosome] \\\n", + " --data-config data_config.yaml \\\n", + " --model-config model_config.yaml\n", + "```\n", + "\n", + "The pipeline will:\n", + "- Generate all 4,839 genomic features for your variants\n", + "- Apply the same preprocessing (subsetting, absolute values, imputation)\n", + "- Use the trained model for inference\n", + "- Output predictions in identical format\n", + "\n", + "**4. Analyze results** using the code blocks above" ] }, { "cell_type": "markdown", + "id": "ca1f268e", "metadata": { "kernel": "SoS" }, "source": [ - "## Steps\n\n**Step 1.** Score variants with a trained feature-weighted CatBoost scEEMS model for one cohort / chromosome. The toy command below scores the microglia cohort (`protocol_example`) on chromosome 2 using a model produced by the training workflow." + "## Steps\n", + "\n", + "**Step 1.** Score variants with a trained feature-weighted CatBoost scEEMS model for one cohort / chromosome. The toy command below scores the microglia cohort (`protocol_example`) on chromosome 2 using a model produced by the training workflow." ] }, { "cell_type": "markdown", + "id": "7f47d7ab", "metadata": {}, "source": [ "**Timing:** ~varies on typical compute infrastructure." @@ -144,6 +344,7 @@ { "cell_type": "code", "execution_count": null, + "id": "49705ca8", "metadata": { "kernel": "SoS" }, @@ -159,6 +360,7 @@ }, { "cell_type": "markdown", + "id": "418e7764", "metadata": { "kernel": "SoS" }, @@ -169,6 +371,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b6cfb4bb", "metadata": { "kernel": "SoS" }, @@ -179,40 +382,73 @@ }, { "cell_type": "markdown", + "id": "cb784c85", "metadata": { "kernel": "SoS" }, "source": [ - "## Workflow implementation\n\nThe `predict` step wraps the `gems_pipeline.R predict` engine. The pipeline loads the trained model, annotates the input variants with the full feature set, applies the 10x deep-learning feature weighting used at training time, and writes per-variant EMS scores. New datasets / cell types are scored by editing the `data_config` YAML only - no code changes are required." + "## Workflow implementation\n", + "\n", + "The `predict` step wraps the `gems_pipeline.R predict` engine. The pipeline loads the trained model, annotates the input variants with the full feature set, applies the 10x deep-learning feature weighting used at training time, and writes per-variant EMS scores. New datasets / cell types are scored by editing the `data_config` YAML only - no code changes are required." ] }, { "cell_type": "code", "execution_count": null, + "id": "522c9aa9", "metadata": { "kernel": "SoS" }, "outputs": [], "source": [ - "[global]\nimport os\n# Work directory & output directory\nparameter: cwd = path('output/ems_prediction')\n# Cohort / cell type to score (must match an entry in the data config)\nparameter: cohort = 'protocol_example'\n# Chromosome to score\nparameter: chromosome = '2'\n# Trained CatBoost model (.cbm) from the EMS Training workflow\nparameter: model_path = path('output/xqtl_modifier_score/protocol_example/model_results/model_standard_subset_weighted_chr_chr2_NPR_1.cbm')\n# Data configuration YAML (cohort, variant list, feature set)\nparameter: data_config = path('code/SoS/xqtl_modifier_score/data_config.yaml')\n# Directory whose relative config paths resolve against (holds the config YAMLs)\nparameter: pipeline_dir = path('code/SoS/xqtl_modifier_score')\n# Directory holding the modular analysis scripts (code/script)\nparameter: modular_script_dir = path('code/script')\nparameter: job_size = 1\nparameter: mem = '60G'\nparameter: walltime = '24h'\n" + "[global]\n", + "import os\n", + "# Work directory & output directory\n", + "parameter: cwd = path('output/ems_prediction')\n", + "# Cohort / cell type to score (must match an entry in the data config)\n", + "parameter: cohort = 'protocol_example'\n", + "# Chromosome to score\n", + "parameter: chromosome = '2'\n", + "# Trained CatBoost model (.cbm) from the EMS Training workflow\n", + "parameter: model_path = path('output/xqtl_modifier_score/protocol_example/model_results/model_standard_subset_weighted_chr_chr2_NPR_1.cbm')\n", + "# Data configuration YAML (cohort, variant list, feature set)\n", + "parameter: data_config = path('code/SoS/xqtl_modifier_score/data_config.yaml')\n", + "# Directory whose relative config paths resolve against (holds the config YAMLs)\n", + "parameter: pipeline_dir = path('code/SoS/xqtl_modifier_score')\n", + "# Directory holding the modular analysis scripts (code/script)\n", + "parameter: modular_script_dir = path('code/script')\n", + "parameter: job_size = 1\n", + "parameter: mem = '60G'\n", + "parameter: walltime = '24h'\n" ] }, { "cell_type": "code", "execution_count": null, + "id": "ef992346", "metadata": { "kernel": "SoS" }, "outputs": [], "source": [ - "[predict]\nbash: expand = \"$[ ]\", workdir = pipeline_dir\n Rscript $[modular_script_dir:a]/xqtl_modifier_score/gems_pipeline.R \\\n --step predict \\\n --cohort $[cohort] \\\n --chromosome $[chromosome] \\\n --model-path $[model_path:a] \\\n --data-config $[data_config:a]\n" + "[predict]\n", + "bash: expand = \"$[ ]\", workdir = pipeline_dir\n", + " Rscript $[modular_script_dir:a]/xqtl_modifier_score/gems_pipeline.R \\\n", + " --step predict \\\n", + " --cohort $[cohort] \\\n", + " --chromosome $[chromosome] \\\n", + " --model-path $[model_path:a] \\\n", + " --data-config $[data_config:a]\n" ] }, { "cell_type": "markdown", + "id": "698afc38", "metadata": {}, "source": [ - "## Anticipated Results\n\nRunning the `predict` step produces per-variant scEEMS prediction scores for the supplied variant list. See the *Pipeline Execution Workflow* walkthrough above for the statistical summary, variant prioritization, and model-performance verification of these outputs." + "## Anticipated Results\n", + "\n", + "Running the `predict` step produces per-variant scEEMS prediction scores for the supplied variant list. See the *Pipeline Execution Workflow* walkthrough above for the statistical summary, variant prioritization, and model-performance verification of these outputs." ] } ], @@ -248,5 +484,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/xqtl_modifier_score/ems_training.ipynb b/code/SoS/xqtl_modifier_score/ems_training.ipynb index 3f94e2497..0425d2879 100644 --- a/code/SoS/xqtl_modifier_score/ems_training.ipynb +++ b/code/SoS/xqtl_modifier_score/ems_training.ipynb @@ -4,21 +4,56 @@ "cell_type": "markdown", "id": "ce5dca4b-66c5-4592-a2e1-2e1138264050", "metadata": {}, - "source": "# scEEMS Model Training\n\n> **Under active development.** This module is a work in progress: its interface, parameters and outputs may still change, and it is not yet covered by the automated test suite." + "source": [ + "# scEEMS Model Training\n", + "\n", + "> **Under active development.** This module is a work in progress: its interface, parameters and outputs may still change, and it is not yet covered by the automated test suite." + ] }, { "cell_type": "markdown", "id": "b465926f", "metadata": {}, "source": [ - "# Code Repo \nThe scEEMS code repository can be found at https://github.com/daklab/scEEMS. This repository contains instructions for downloading the training and test data in order to train the scEEMS models. It also contains code for how the SHAP analysis, LD Score regression, eMAGMA analysis, and functionally-informed eQTL fine-mapping were conducted." + "# Code Repo \n", + "The scEEMS code repository can be found at https://github.com/daklab/scEEMS. This repository contains instructions for downloading the training and test data in order to train the scEEMS models. It also contains code for how the SHAP analysis, LD Score regression, eMAGMA analysis, and functionally-informed eQTL fine-mapping were conducted." ] }, { "cell_type": "markdown", + "id": "d0736993", "metadata": {}, "source": [ - "## Description\n\n## Motivation\n\n### The \"Missing Regulation\" Problem\n\nMost disease-associated GWAS variants lie in non-coding regions of the genome, where they likely modulate gene expression. However, bulk-tissue eQTL studies fail to explain the majority of these variants, a phenomenon termed \"missing regulation\" [(Connally et al., 2022)](https://elifesciences.org/articles/74970v1). This gap exists because there are systematic differences between variants identified in eQTL studies versus disease GWAS [(Mostafavi et al., 2022)](https://www.nature.com/articles/s41588-023-01529-1):\n\n- **eQTLs** are enriched in promoter regions and affect genes under weaker selective constraint\n- **GWAS variants** are enriched in distal enhancer regions and affect genes under stronger selective constraint\n\nUnderstanding how non-coding GWAS variants modulate gene expression is critical for uncovering disease mechanisms, but several challenges limit our ability to make these connections:\n\n1. **Cell-type specificity**: Bulk tissue approaches cannot capture regulatory effects that occur in specific cell types, particularly rare but disease-relevant populations like microglia in Alzheimer's disease\n2. **Enhancer variants**: Disease-associated variants in distal enhancers often have weaker eQTL signals that fail to reach statistical significance, especially in underpowered single-cell studies\n3. **Limited sample sizes**: Single-cell eQTL mapping has reduced statistical power compared to bulk studies, making it difficult to detect true regulatory signals in rare cell types\n\n### scEEMS Solution\n\nscEEMS addresses these challenges by predicting causal cell-type-specific eQTLs using machine learning trained on 4,839 genomic features, including:\n- Deep learning-based variant effect predictions\n- Cell-type-specific regulatory annotations \n- Activity-by-Contact (ABC) enhancer-gene linkages\n- Distance and evolutionary constraint features\n\nBy identifying functional variants in cell-type-specific contexts—particularly in enhancer regions—scEEMS aims to bridge the gap between non-coding GWAS variants and their target genes, improving our understanding of disease mechanisms in Alzheimer's disease.\n\n### Tutorial Objective\n\nThis notebook provides a reproducible pipeline for training scEEMS models, demonstrating the complete methodology from data preparation through model evaluation. The goal is to enable the broader scientific community to apply this approach to their own cell-type-specific eQTL datasets and disease contexts." + "## Description\n", + "\n", + "## Motivation\n", + "\n", + "### The \"Missing Regulation\" Problem\n", + "\n", + "Most disease-associated GWAS variants lie in non-coding regions of the genome, where they likely modulate gene expression. However, bulk-tissue eQTL studies fail to explain the majority of these variants, a phenomenon termed \"missing regulation\" [(Connally et al., 2022)](https://elifesciences.org/articles/74970v1). This gap exists because there are systematic differences between variants identified in eQTL studies versus disease GWAS [(Mostafavi et al., 2022)](https://www.nature.com/articles/s41588-023-01529-1):\n", + "\n", + "- **eQTLs** are enriched in promoter regions and affect genes under weaker selective constraint\n", + "- **GWAS variants** are enriched in distal enhancer regions and affect genes under stronger selective constraint\n", + "\n", + "Understanding how non-coding GWAS variants modulate gene expression is critical for uncovering disease mechanisms, but several challenges limit our ability to make these connections:\n", + "\n", + "1. **Cell-type specificity**: Bulk tissue approaches cannot capture regulatory effects that occur in specific cell types, particularly rare but disease-relevant populations like microglia in Alzheimer's disease\n", + "2. **Enhancer variants**: Disease-associated variants in distal enhancers often have weaker eQTL signals that fail to reach statistical significance, especially in underpowered single-cell studies\n", + "3. **Limited sample sizes**: Single-cell eQTL mapping has reduced statistical power compared to bulk studies, making it difficult to detect true regulatory signals in rare cell types\n", + "\n", + "### scEEMS Solution\n", + "\n", + "scEEMS addresses these challenges by predicting causal cell-type-specific eQTLs using machine learning trained on 4,839 genomic features, including:\n", + "- Deep learning-based variant effect predictions\n", + "- Cell-type-specific regulatory annotations \n", + "- Activity-by-Contact (ABC) enhancer-gene linkages\n", + "- Distance and evolutionary constraint features\n", + "\n", + "By identifying functional variants in cell-type-specific contexts—particularly in enhancer regions—scEEMS aims to bridge the gap between non-coding GWAS variants and their target genes, improving our understanding of disease mechanisms in Alzheimer's disease.\n", + "\n", + "### Tutorial Objective\n", + "\n", + "This notebook provides a reproducible pipeline for training scEEMS models, demonstrating the complete methodology from data preparation through model evaluation. The goal is to enable the broader scientific community to apply this approach to their own cell-type-specific eQTL datasets and disease contexts." ] }, { @@ -26,14 +61,57 @@ "id": "ffc11f6f-5700-476f-86cb-cecc0bedac05", "metadata": {}, "source": [ - "## Methods Overview\n\n### CatBoost Algorithm\nWe use [CatBoost](https://github.com/catboost/catboost), a gradient boosting framework that builds an ensemble of decision trees sequentially. CatBoost is effective for high-dimensional biological datasets with mixed data types.\n\n### Model Training Strategy\nWe train a CatBoost model with 10x upweighting of deep learning features (feature weight = 10 for DL-VEP features vs. 1 for other features). This model was selected as optimal based on external validation and heritability analysis described in the manuscript.\n\n### Training Data Construction\n**Data Source**: Fine-mapped single-cell eQTLs from six brain cell types in the ROSMAP cohort.\n\n**Positive Class (Y=1)**:\n- Variants with PIP > 0.05 in a credible set where the maximum PIP exceeds 0.1, OR\n- Variants with PIP > 0.5 regardless of credible set membership\n\n**Negative Class (Y=0)**: \n- For each positive variant, we sample 10 negative variants from the same gene with PIP < 0.01, matched on variant type (SNP, insertion, deletion)\n\n**Test Set**:\n- Positive variants: PIP > 0.90\n- Negative variants: 10 matched variants per positive variant with PIP < 0.01\n- Restricted to MEGA genes only\n\n### Sample Weighting\n- Negative variants: weight = 1\n- Positive variants: weighted proportional to their PIP values\n- Total weight balanced between positive and negative classes\n\n### Cross-Validation: Leave-One-Chromosome-Out (LOCO)\nFor each of the 22 autosomes:\n1. Train on variants from all other 21 chromosomes\n2. Test on the held-out chromosome\n3. Aggregate predictions from all 22 held-out chromosomes for final performance metrics\n\n### Toy Dataset Note\nThis tutorial uses chromosome 2 data only for demonstration:\n- Training: 3,056 variants \n- Testing: 761 variants (non-overlapping)\n- The full study trained models across all 22 chromosomes for each of 6 cell types" + "## Methods Overview\n", + "\n", + "### CatBoost Algorithm\n", + "We use [CatBoost](https://github.com/catboost/catboost), a gradient boosting framework that builds an ensemble of decision trees sequentially. CatBoost is effective for high-dimensional biological datasets with mixed data types.\n", + "\n", + "### Model Training Strategy\n", + "We train a CatBoost model with 10x upweighting of deep learning features (feature weight = 10 for DL-VEP features vs. 1 for other features). This model was selected as optimal based on external validation and heritability analysis described in the manuscript.\n", + "\n", + "### Training Data Construction\n", + "**Data Source**: Fine-mapped single-cell eQTLs from six brain cell types in the ROSMAP cohort.\n", + "\n", + "**Positive Class (Y=1)**:\n", + "- Variants with PIP > 0.05 in a credible set where the maximum PIP exceeds 0.1, OR\n", + "- Variants with PIP > 0.5 regardless of credible set membership\n", + "\n", + "**Negative Class (Y=0)**: \n", + "- For each positive variant, we sample 10 negative variants from the same gene with PIP < 0.01, matched on variant type (SNP, insertion, deletion)\n", + "\n", + "**Test Set**:\n", + "- Positive variants: PIP > 0.90\n", + "- Negative variants: 10 matched variants per positive variant with PIP < 0.01\n", + "- Restricted to MEGA genes only\n", + "\n", + "### Sample Weighting\n", + "- Negative variants: weight = 1\n", + "- Positive variants: weighted proportional to their PIP values\n", + "- Total weight balanced between positive and negative classes\n", + "\n", + "### Cross-Validation: Leave-One-Chromosome-Out (LOCO)\n", + "For each of the 22 autosomes:\n", + "1. Train on variants from all other 21 chromosomes\n", + "2. Test on the held-out chromosome\n", + "3. Aggregate predictions from all 22 held-out chromosomes for final performance metrics\n", + "\n", + "### Toy Dataset Note\n", + "This tutorial uses chromosome 2 data only for demonstration:\n", + "- Training: 3,056 variants \n", + "- Testing: 761 variants (non-overlapping)\n", + "- The full study trained models across all 22 chromosomes for each of 6 cell types" ] }, { "cell_type": "markdown", + "id": "948a3cc7", "metadata": {}, "source": [ - "## Input\n\n## Input Data: Feature Categories\n\nBased on the manuscript (Figure 1, Table C), scEEMS uses 4,839 features across these categories:\n" + "## Input\n", + "\n", + "## Input Data: Feature Categories\n", + "\n", + "Based on the manuscript (Figure 1, Table C), scEEMS uses 4,839 features across these categories:\n" ] }, { @@ -43,7 +121,12 @@ "metadata": {}, "outputs": [], "source": [ - "# Peek at the gene-constraint table used as an EMS feature source.\nR:\n suppressPackageStartupMessages(library(readxl))\n gene_data <- read_excel(\"tests/fixtures/ems_training/protocol_example.gene_constraint.xlsx\",\n sheet = \"Supplementary Table 1\")\n print(utils::head(as.data.frame(gene_data), 3))\n" + "# Peek at the gene-constraint table used as an EMS feature source.\n", + "R:\n", + " suppressPackageStartupMessages(library(readxl))\n", + " gene_data <- read_excel(\"tests/fixtures/ems_training/protocol_example.gene_constraint.xlsx\",\n", + " sheet = \"Supplementary Table 1\")\n", + " print(utils::head(as.data.frame(gene_data), 3))\n" ] }, { @@ -51,7 +134,13 @@ "id": "dc15bca6-ab5a-4ddb-aad6-1a9ff06d4013", "metadata": {}, "source": [ - "**Sample output showing GeneBayes constraint scores:**\n```\nensg hgnc chrom obs_lof exp_lof post_mean post_lower_95 post_upper_95\nENSG00000198488 HGNC:24141 chr11 12 8.9777 6.46629E-05 7.16256E-06 0.00017805\nENSG00000164363 HGNC:26441 chr5 31 28.55 0.00016062 2.59918E-05 0.00044175\nENSG00000159337 HGNC:30038 chr15 28 41.84 0.00016978 0.000018674 0.00053317\n```" + "**Sample output showing GeneBayes constraint scores:**\n", + "```\n", + "ensg hgnc chrom obs_lof exp_lof post_mean post_lower_95 post_upper_95\n", + "ENSG00000198488 HGNC:24141 chr11 12 8.9777 6.46629E-05 7.16256E-06 0.00017805\n", + "ENSG00000164363 HGNC:26441 chr5 31 28.55 0.00016062 2.59918E-05 0.00044175\n", + "ENSG00000159337 HGNC:30038 chr15 28 41.84 0.00016978 0.000018674 0.00053317\n", + "```" ] }, { @@ -61,16 +150,41 @@ "jp-MarkdownHeadingCollapsed": true }, "source": [ - "### Feature Categories\n\n#### 1. Distance Features\n**Biological Rationale**: Physical proximity determines regulatory potential\n- `abs_distance_TSS_log`: Log-transformed distance to transcription start site\n\n#### 2. Cell-Type Regulatory Features \n**Biological Rationale**: Functional genomics assays corresponding to cell-specific regulation\n- `ABC_score_microglia`: Activity-by-Contact regulatory activity score\n\n#### 3. Population Genetics Features\n**Biological Rationale**: Minor allele frequency can impact regulatory activity in cells\n- `gnomad_MAF`: Minor allele frequency from population database\n\n#### 4. Conservation Features \n**Biological Rationale**: Evolutionary constraint of the gene can may impact whether their exists an eQTL\n- `GeneBayes Constraint Score`: Bayesian score of gene constraint\n\n#### 5. Deep Learning Predictions\n**Biological Rationale**: Sequence-to-function model predictions can determine disruption of intermediate molecular phenotypes.\n- Enformer, ChromBPNet, and BPNet predictions of various functional genomics assays (excluding gene expression predictions)\n- `diff_32_ENCFF140MBA`: Z-score normalized Enformer VEP score for H3K4me1:CD8-positive, alpha-beta T cell (using middle 32 bins corresponding 4096 bp flanking the variant).\n\n" + "### Feature Categories\n", + "\n", + "#### 1. Distance Features\n", + "**Biological Rationale**: Physical proximity determines regulatory potential\n", + "- `abs_distance_TSS_log`: Log-transformed distance to transcription start site\n", + "\n", + "#### 2. Cell-Type Regulatory Features \n", + "**Biological Rationale**: Functional genomics assays corresponding to cell-specific regulation\n", + "- `ABC_score_microglia`: Activity-by-Contact regulatory activity score\n", + "\n", + "#### 3. Population Genetics Features\n", + "**Biological Rationale**: Minor allele frequency can impact regulatory activity in cells\n", + "- `gnomad_MAF`: Minor allele frequency from population database\n", + "\n", + "#### 4. Conservation Features \n", + "**Biological Rationale**: Evolutionary constraint of the gene can may impact whether their exists an eQTL\n", + "- `GeneBayes Constraint Score`: Bayesian score of gene constraint\n", + "\n", + "#### 5. Deep Learning Predictions\n", + "**Biological Rationale**: Sequence-to-function model predictions can determine disruption of intermediate molecular phenotypes.\n", + "- Enformer, ChromBPNet, and BPNet predictions of various functional genomics assays (excluding gene expression predictions)\n", + "- `diff_32_ENCFF140MBA`: Z-score normalized Enformer VEP score for H3K4me1:CD8-positive, alpha-beta T cell (using middle 32 bins corresponding 4096 bp flanking the variant).\n", + "\n" ] }, { "cell_type": "markdown", + "id": "9cf05043", "metadata": { "kernel": "SoS" }, "source": [ - "## Steps\n\n**Step 1.** Train a feature-weighted CatBoost scEEMS model for one cohort / chromosome. The toy command below trains on the microglia cohort (`protocol_example`) for chromosome 2 using the bundled YAML configs." + "## Steps\n", + "\n", + "**Step 1.** Train a feature-weighted CatBoost scEEMS model for one cohort / chromosome. The toy command below trains on the microglia cohort (`protocol_example`) for chromosome 2 using the bundled YAML configs." ] }, { @@ -78,11 +192,22 @@ "id": "bd338657-200c-4332-b6ea-11b1e9b8b4ca", "metadata": {}, "source": [ - "## Training Workflow\n\n### Step 1: Running the GEMS Pipeline\n```bash\ncd ~/xqtl-protocol/code/SoS/xqtl_modifier_score/\nRscript gems_pipeline.R --step train --cohort Mic_mega_eQTL --chromosome 2 \\\n --data-config data_config.yaml \\\n --model-config model_config.yaml\n```\n\nThe pipeline automatically loads training data, trains the feature-weighted CatBoost model, and generates predictions." + "## Training Workflow\n", + "\n", + "### Step 1: Running the GEMS Pipeline\n", + "```bash\n", + "cd ~/xqtl-protocol/code/SoS/xqtl_modifier_score/\n", + "Rscript gems_pipeline.R --step train --cohort Mic_mega_eQTL --chromosome 2 \\\n", + " --data-config data_config.yaml \\\n", + " --model-config model_config.yaml\n", + "```\n", + "\n", + "The pipeline automatically loads training data, trains the feature-weighted CatBoost model, and generates predictions." ] }, { "cell_type": "markdown", + "id": "3eba7f23", "metadata": {}, "source": [ "**Timing:** ~varies on typical compute infrastructure." @@ -91,6 +216,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b652bba2", "metadata": { "kernel": "Bash" }, @@ -111,7 +237,13 @@ "metadata": {}, "outputs": [], "source": [ - "# Configuration file check\nR:\n files <- c(\"code/script/xqtl_modifier_score/gems_pipeline.R\",\n \"code/SoS/xqtl_modifier_score/data_config.yaml\",\n \"code/SoS/xqtl_modifier_score/model_config.yaml\")\n cat(\"Configuration File Check:\\n\"); cat(strrep(\"=\", 50), \"\\n\", sep = \"\")\n for (f in files) cat(sprintf(\" %s: %s\\n\", f, if (file.exists(f)) \"Found\" else \"Missing\"))\n" + "# Configuration file check\n", + "R:\n", + " files <- c(\"code/script/xqtl_modifier_score/gems_pipeline.R\",\n", + " \"code/SoS/xqtl_modifier_score/data_config.yaml\",\n", + " \"code/SoS/xqtl_modifier_score/model_config.yaml\")\n", + " cat(\"Configuration File Check:\\n\"); cat(strrep(\"=\", 50), \"\\n\", sep = \"\")\n", + " for (f in files) cat(sprintf(\" %s: %s\\n\", f, if (file.exists(f)) \"Found\" else \"Missing\"))\n" ] }, { @@ -168,11 +300,20 @@ "id": "fcf70085-287e-4092-aaf2-f78ff69b792a", "metadata": {}, "source": [ - "### Generated Output Files\n\n**Model File**:\n- `model_feature_weighted_chr_chr2_NPR_10.cbm` - Trained CatBoost classifier\n\n**Analysis Results**: \n- `summary_dict_catboost_1model_chr_chr2_NPR_10.pkl` - Performance metrics and validation statistics\n- `features_importance_1model_chr_chr2_NPR_10.csv` - Complete feature importance rankings\n- `predictions_1model_chr2.tsv` - Per-variant prediction probabilities\n" + "### Generated Output Files\n", + "\n", + "**Model File**:\n", + "- `model_feature_weighted_chr_chr2_NPR_10.cbm` - Trained CatBoost classifier\n", + "\n", + "**Analysis Results**: \n", + "- `summary_dict_catboost_1model_chr_chr2_NPR_10.pkl` - Performance metrics and validation statistics\n", + "- `features_importance_1model_chr_chr2_NPR_10.csv` - Complete feature importance rankings\n", + "- `predictions_1model_chr2.tsv` - Per-variant prediction probabilities\n" ] }, { "cell_type": "markdown", + "id": "e26c97e4", "metadata": { "kernel": "SoS" }, @@ -183,6 +324,7 @@ { "cell_type": "code", "execution_count": null, + "id": "b576233e", "metadata": { "kernel": "Bash" }, @@ -193,33 +335,67 @@ }, { "cell_type": "markdown", + "id": "c82eaf3b", "metadata": { "kernel": "SoS" }, "source": [ - "## Workflow implementation\n\nThe `train` step wraps the `gems_pipeline.R train` engine. The pipeline automatically loads the training data, trains the feature-weighted CatBoost model, and writes the trained model and metrics. New datasets / cell types are added by editing the YAML configs only — no code changes are required (e.g. swap `protocol_example` for `Ast_mega_eQTL` with the matching `data_config` to train on astrocytes)." + "## Workflow implementation\n", + "\n", + "The `train` step wraps the `gems_pipeline.R train` engine. The pipeline automatically loads the training data, trains the feature-weighted CatBoost model, and writes the trained model and metrics. New datasets / cell types are added by editing the YAML configs only — no code changes are required (e.g. swap `protocol_example` for `Ast_mega_eQTL` with the matching `data_config` to train on astrocytes)." ] }, { "cell_type": "code", "execution_count": null, + "id": "868b7ebc", "metadata": { "kernel": "SoS" }, "outputs": [], "source": [ - "[global]\nimport os\n# Work directory & output directory\nparameter: cwd = path('output/ems_training')\n# Cohort / cell type to train (must match an entry in the data config)\nparameter: cohort = 'protocol_example'\n# Chromosome to train on\nparameter: chromosome = '2'\n# Data configuration YAML (cohort, eQTL paths, feature set)\nparameter: data_config = path('code/SoS/xqtl_modifier_score/data_config.yaml')\n# Model configuration YAML (CatBoost hyperparameters, feature weighting)\nparameter: model_config = path('code/SoS/xqtl_modifier_score/model_config.yaml')\n# Directory whose relative config paths resolve against (holds the config YAMLs)\nparameter: pipeline_dir = path('code/SoS/xqtl_modifier_score')\n# Directory holding the modular analysis scripts (code/script)\nparameter: modular_script_dir = path('code/script')\nparameter: job_size = 1\nparameter: mem = '60G'\nparameter: walltime = '24h'\n" + "[global]\n", + "import os\n", + "# Work directory & output directory\n", + "parameter: cwd = path('output/ems_training')\n", + "# Cohort / cell type to train (must match an entry in the data config)\n", + "parameter: cohort = 'protocol_example'\n", + "# Chromosome to train on\n", + "parameter: chromosome = '2'\n", + "# Data configuration YAML (cohort, eQTL paths, feature set)\n", + "parameter: data_config = path('code/SoS/xqtl_modifier_score/data_config.yaml')\n", + "# Model configuration YAML (CatBoost hyperparameters, feature weighting)\n", + "parameter: model_config = path('code/SoS/xqtl_modifier_score/model_config.yaml')\n", + "# Directory whose relative config paths resolve against (holds the config YAMLs)\n", + "parameter: pipeline_dir = path('code/SoS/xqtl_modifier_score')\n", + "# Directory holding the modular analysis scripts (code/script)\n", + "parameter: modular_script_dir = path('code/script')\n", + "parameter: job_size = 1\n", + "parameter: mem = '60G'\n", + "parameter: walltime = '24h'\n" ] }, { "cell_type": "code", "execution_count": null, + "id": "273c389d", "metadata": { "kernel": "SoS" }, "outputs": [], "source": [ - "[train]\noutput: f'{cwd:a}/{cohort}_chr{chromosome}_scEEMS_model.done'\ntask: trunk_workers = 1, trunk_size = job_size, mem = mem, walltime = walltime, tags = f'{step_name}_{cohort}_chr{chromosome}'\nbash: expand = \"$[ ]\", workdir = pipeline_dir, stderr = f'{_output[0]}.stderr', stdout = f'{_output[0]}.stdout'\n set -e\n Rscript $[modular_script_dir:a]/xqtl_modifier_score/gems_pipeline.R \\\n --step train \\\n --cohort $[cohort] \\\n --chromosome $[chromosome] \\\n --data-config $[data_config:a] \\\n --model-config $[model_config:a]\n touch $[_output[0]:a]\n" + "[train]\n", + "output: f'{cwd:a}/{cohort}_chr{chromosome}_scEEMS_model.done'\n", + "task: trunk_workers = 1, trunk_size = job_size, mem = mem, walltime = walltime, tags = f'{step_name}_{cohort}_chr{chromosome}'\n", + "bash: expand = \"$[ ]\", workdir = pipeline_dir, stderr = f'{_output[0]}.stderr', stdout = f'{_output[0]}.stdout'\n", + " set -e\n", + " Rscript $[modular_script_dir:a]/xqtl_modifier_score/gems_pipeline.R \\\n", + " --step train \\\n", + " --cohort $[cohort] \\\n", + " --chromosome $[chromosome] \\\n", + " --data-config $[data_config:a] \\\n", + " --model-config $[model_config:a]\n", + " touch $[_output[0]:a]\n" ] }, { @@ -227,7 +403,54 @@ "id": "c5a71f8b-8a26-48ee-b953-3d11bcfd1410", "metadata": {}, "source": [ - "## GEMS Pipeline Modularity \n\n#### GEMS: Generalized Expression Modifier Scores\n\nThe pipeline uses `gems_pipeline.R` (**G**eneralized **E**xpression **M**odifier **S**cores), extending beyond single-cell data to work with any tissue or cell type.\n\n**Design Principles:**\n- New datasets added by **only modifying YAML configuration files**\n- No code changes required for different cell types\n- Subcommand architecture (train/predict) for extensible workflows\n\n#### Command Structure\n```bash\n# Training mode\nRscript gems_pipeline.R --step train --cohort --chromosome \\\n --data-config data_config.yaml \\\n --model-config model_config.yaml\n\n# Prediction mode (coming soon)\nRscript gems_pipeline.R --step predict --cohort --chromosome \\\n --model_path results/model.cbm \\\n --data-config data_config.yaml\n```\n\n### Testing Modularity with Different Cell Types\n\n**Test Case:**\n- Original development: Microglia (Mic_mega_eQTL)\n- Modularity validation: Astrocytes (Ast_mega_eQTL)\n```bash\n# Train on Microglia\nRscript gems_pipeline.R --step train --cohort Mic_mega_eQTL --chromosome 2 \\\n --data-config data_config.yaml \\\n --model-config model_config.yaml\n\n# Train on Astrocytes (same code, different YAML configuration)\nRscript gems_pipeline.R --step train --cohort Ast_mega_eQTL --chromosome 1 \\\n --data-config data_config.yaml \\\n --model-config model_config.yaml\n```\n\n**Validation Results:**\n✅ Automatic path resolution to Astrocyte data \n✅ All 4,839 genomic features loaded without code changes \n✅ Data preprocessing and imputation applied correctly \n✅ Proper train/test split maintained\n\nThis demonstrates true modularity—new cell types can be analyzed using only YAML configuration changes." + "## GEMS Pipeline Modularity \n", + "\n", + "#### GEMS: Generalized Expression Modifier Scores\n", + "\n", + "The pipeline uses `gems_pipeline.R` (**G**eneralized **E**xpression **M**odifier **S**cores), extending beyond single-cell data to work with any tissue or cell type.\n", + "\n", + "**Design Principles:**\n", + "- New datasets added by **only modifying YAML configuration files**\n", + "- No code changes required for different cell types\n", + "- Subcommand architecture (train/predict) for extensible workflows\n", + "\n", + "#### Command Structure\n", + "```bash\n", + "# Training mode\n", + "Rscript gems_pipeline.R --step train --cohort --chromosome \\\n", + " --data-config data_config.yaml \\\n", + " --model-config model_config.yaml\n", + "\n", + "# Prediction mode (coming soon)\n", + "Rscript gems_pipeline.R --step predict --cohort --chromosome \\\n", + " --model_path results/model.cbm \\\n", + " --data-config data_config.yaml\n", + "```\n", + "\n", + "### Testing Modularity with Different Cell Types\n", + "\n", + "**Test Case:**\n", + "- Original development: Microglia (Mic_mega_eQTL)\n", + "- Modularity validation: Astrocytes (Ast_mega_eQTL)\n", + "```bash\n", + "# Train on Microglia\n", + "Rscript gems_pipeline.R --step train --cohort Mic_mega_eQTL --chromosome 2 \\\n", + " --data-config data_config.yaml \\\n", + " --model-config model_config.yaml\n", + "\n", + "# Train on Astrocytes (same code, different YAML configuration)\n", + "Rscript gems_pipeline.R --step train --cohort Ast_mega_eQTL --chromosome 1 \\\n", + " --data-config data_config.yaml \\\n", + " --model-config model_config.yaml\n", + "```\n", + "\n", + "**Validation Results:**\n", + "✅ Automatic path resolution to Astrocyte data \n", + "✅ All 4,839 genomic features loaded without code changes \n", + "✅ Data preprocessing and imputation applied correctly \n", + "✅ Proper train/test split maintained\n", + "\n", + "This demonstrates true modularity—new cell types can be analyzed using only YAML configuration changes." ] }, { @@ -235,14 +458,23 @@ "id": "41dc00b8-7f01-45d3-9d2c-4bcf9ce4df7e", "metadata": {}, "source": [ - "### Using Your Trained Models\n\nOnce training is complete, load the trained models for predictions. Please refer to **[EMS Predictions](https://statfungen.github.io/xqtl-protocol/code/SoS/xqtl_modifier_score/ems_prediction.html)** for detailed prediction workflows and variant scoring." + "### Using Your Trained Models\n", + "\n", + "Once training is complete, load the trained models for predictions. Please refer to **[EMS Predictions](https://statfungen.github.io/xqtl-protocol/ems-prediction)** for detailed prediction workflows and variant scoring." ] }, { "cell_type": "markdown", + "id": "3ad2dfc2", "metadata": {}, "source": [ - "## Anticipated Results\n\n| Output Type | Description | Research Use |\n|-------------|-------------|--------------|\n| **Trained models** | Single feature-weighted CatBoost classifier | Variant effect prediction |\n| **Performance metrics** | AP/AUC scores | Model comparison |\n| **Feature importance** | Genomic feature rankings | Biological interpretation |" + "## Anticipated Results\n", + "\n", + "| Output Type | Description | Research Use |\n", + "|-------------|-------------|--------------|\n", + "| **Trained models** | Single feature-weighted CatBoost classifier | Variant effect prediction |\n", + "| **Performance metrics** | AP/AUC scores | Model comparison |\n", + "| **Feature importance** | Genomic feature rankings | Biological interpretation |" ] } ], @@ -278,5 +510,5 @@ } }, "nbformat": 4, - "nbformat_minor": 4 -} \ No newline at end of file + "nbformat_minor": 5 +} diff --git a/code/SoS/xqtl_protocol_demo.ipynb b/code/SoS/xqtl_protocol_demo.ipynb index c829cedb1..75cb7da1a 100644 --- a/code/SoS/xqtl_protocol_demo.ipynb +++ b/code/SoS/xqtl_protocol_demo.ipynb @@ -9,7 +9,7 @@ "source": [ "# Environment Setup\n", "\n", - "![Three steps: install the environment, get the repo, run a pipeline](code/images/xqtl_getting_started.gif)\n", + "![Three steps: install the environment, get the repo, run a pipeline](../images/xqtl_getting_started.gif)\n", "\n", "Three steps take you from an empty machine to your first pipeline run:\n", "\n", @@ -299,4 +299,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/xqtl_protocol_draft.ipynb b/code/SoS/xqtl_protocol_draft.ipynb index 27e019528..527956785 100644 --- a/code/SoS/xqtl_protocol_draft.ipynb +++ b/code/SoS/xqtl_protocol_draft.ipynb @@ -893,4 +893,4 @@ }, "nbformat": 4, "nbformat_minor": 5 -} \ No newline at end of file +} diff --git a/code/SoS/xqtl_protocol_workflow_builder.html b/code/SoS/xqtl_protocol_workflow_builder.html index e3cb00af9..1ec762181 100644 --- a/code/SoS/xqtl_protocol_workflow_builder.html +++ b/code/SoS/xqtl_protocol_workflow_builder.html @@ -334,7 +334,7 @@

From biological samples to interpretable xQTL discoverie

Your analysis route

-

Inputs

Reference Data

Quantification

Molecular Phenotype Quantification

Bulk RNA-seq
Single-nuclei / pseudobulk
Alternative splicing
DNA methylation
Polyadenylation

Pre-processing

Data Pre-processing

Genotype
Phenotype
Covariate

Discovery

QTL Association Testing

Multivariate modelling

Cross-cohort Meta-analysis

Multivariate Mixture (MASH)

Regression

High-dimensional Regression

Individual level
Summary statistics level

Integration

GWAS Integration

Rare-variant xQTL

Interpretation

Enrichment & Validation

xQTL Modifier Score

+

Inputs

Reference Data

Quantification

Molecular Phenotype Quantification

Bulk RNA-seq
Single-nuclei / pseudobulk
Alternative splicing
DNA methylation
Polyadenylation

Pre-processing

Data Pre-processing

Genotype
Phenotype
Covariate

Discovery

QTL Association Testing

Multivariate modelling

Cross-cohort Meta-analysis

Multivariate Mixture (MASH)

Regression

High-dimensional Regression

Individual level
Summary statistics level

Integration

GWAS Integration

Rare-variant xQTL

Interpretation

Enrichment & Validation

xQTL Modifier Score

diff --git a/code/SoS/xqtl_protocol_workflow_builder.ipynb b/code/SoS/xqtl_protocol_workflow_builder.ipynb index 827d67b66..68c683b7d 100644 --- a/code/SoS/xqtl_protocol_workflow_builder.ipynb +++ b/code/SoS/xqtl_protocol_workflow_builder.ipynb @@ -2,11 +2,12 @@ "cells": [ { "cell_type": "markdown", + "id": "bb79e353", "metadata": {}, "source": [ "# xQTL Analysis Workflow Builder\n", "\n", - "Not set up yet? Start with [Environment Setup](https://statfungen.github.io/xqtl-protocol/xqtl_protocol_demo.html) to install the software stack and run a first pipeline on the example data shipped in this repository. Each module below links to its own page, where the inputs, expected outputs and full command interface are documented.\n", + "Not set up yet? Start with [Environment Setup](https://statfungen.github.io/xqtl-protocol/xqtl-protocol-demo) to install the software stack and run a first pipeline on the example data shipped in this repository. Each module below links to its own page, where the inputs, expected outputs and full command interface are documented.\n", "\n", "\n", "````{raw} html\n", @@ -345,7 +346,7 @@ "

Your analysis route

\n", "
\n", "\n", - "

Inputs

Reference Data

Quantification

Molecular Phenotype Quantification

Bulk RNA-seq
Single-nuclei / pseudobulk
Alternative splicing
DNA methylation
Polyadenylation

Pre-processing

Data Pre-processing

Genotype
Phenotype
Covariate

Discovery

QTL Association Testing

Multivariate modelling

Cross-cohort Meta-analysis

Multivariate Mixture (MASH)

Regression

High-dimensional Regression

Individual level
Summary statistics level

Integration

GWAS Integration

Rare-variant xQTL

Interpretation

Enrichment & Validation

xQTL Modifier Score

\n", + "

Inputs

Reference Data

Quantification

Molecular Phenotype Quantification

Bulk RNA-seq
Single-nuclei / pseudobulk
Alternative splicing
DNA methylation
Polyadenylation

Pre-processing

Data Pre-processing

Genotype
Phenotype
Covariate

Discovery

QTL Association Testing

Multivariate modelling

Cross-cohort Meta-analysis

Multivariate Mixture (MASH)

Regression

High-dimensional Regression

Individual level
Summary statistics level

Integration

GWAS Integration

Rare-variant xQTL

Interpretation

Enrichment & Validation

xQTL Modifier Score

\n", "\n", "\n", "\n", @@ -761,13 +762,13 @@ "name": "sos" }, "language_info": { - "name": "sos", "file_extension": ".sos", "mimetype": "text/x-sos", - "pygments_lexer": "sos", - "nbconvert_exporter": "sos.jupyter.converter.SoS_Exporter" + "name": "sos", + "nbconvert_exporter": "sos.jupyter.converter.SoS_Exporter", + "pygments_lexer": "sos" } }, "nbformat": 4, - "nbformat_minor": 4 + "nbformat_minor": 5 } diff --git a/myst.yml b/myst.yml new file mode 100644 index 000000000..17a551c91 --- /dev/null +++ b/myst.yml @@ -0,0 +1,127 @@ +# MyST / Jupyter Book 2 configuration for the FunGen-xQTL protocol site. +# +# The TOC points at the notebooks where they actually live. MyST derives a flat +# page slug from each basename (code/SoS/reference_data/reference_data.ipynb -> +# /reference-data/) and resolves relative links and images itself, so no staging +# or link rewriting is needed. website/build_support.py handles only the two +# things MyST cannot: the workflow-builder widget and legacy-URL redirects. +version: 1 +project: + title: FunGen-xQTL Consortium + authors: + - name: The NIH/NIA Alzheimer's Disease Sequencing Project Functional Genomics xQTL Consortium + copyright: 2021+, FunGen xQTL Analysis Working Group + github: https://github.com/statfungen/xqtl-protocol + funding: + - statement: Supported by [NIH NIA](https://www.nia.nih.gov/research/ad-genetics) + bibliography: + - website/references.bib + toc: + - file: README.md + - title: Getting started + children: + - file: code/SoS/xqtl_protocol_demo.ipynb + - file: website/_generated/xqtl_protocol_workflow_builder.ipynb + - title: Reference data + children: + - file: code/SoS/reference_data/reference_data.ipynb + children: + - file: code/SoS/reference_data/reference_data_preparation.ipynb + - file: code/SoS/reference_data/generalized_TADB.ipynb + - file: code/SoS/reference_data/ld_prune_reference.ipynb + - file: code/SoS/reference_data/rss_ld_sketch.ipynb + - title: Molecular Phenotypes + children: + - file: code/SoS/molecular_phenotypes/bulk_expression.ipynb + children: + - file: code/SoS/molecular_phenotypes/calling/RNA_calling.ipynb + - file: code/SoS/molecular_phenotypes/QC/bulk_expression_QC.ipynb + - file: code/SoS/molecular_phenotypes/QC/bulk_expression_normalization.ipynb + - file: code/SoS/molecular_phenotypes/single_cell.ipynb + children: + - file: code/SoS/molecular_phenotypes/snRNAseq_preprocessing.ipynb + - file: code/SoS/molecular_phenotypes/QC/pseudobulk_preprocessing.ipynb + - file: code/SoS/molecular_phenotypes/methylation.ipynb + children: + - file: code/SoS/molecular_phenotypes/calling/methylation_calling.ipynb + - file: code/SoS/molecular_phenotypes/splicing.ipynb + children: + - file: code/SoS/molecular_phenotypes/calling/splicing_calling.ipynb + - file: code/SoS/molecular_phenotypes/QC/splicing_normalization.ipynb + - file: code/SoS/molecular_phenotypes/apa.ipynb + children: + - file: code/SoS/molecular_phenotypes/calling/apa_calling.ipynb + - file: code/SoS/molecular_phenotypes/QC/apa_impute.ipynb + - title: Data Preprocessing + children: + - file: code/SoS/data_preprocessing/genotype_preprocessing.ipynb + children: + - file: code/SoS/data_preprocessing/genotype/VCF_QC.ipynb + - file: code/SoS/data_preprocessing/genotype/genotype_formatting.ipynb + - file: code/SoS/data_preprocessing/genotype/GWAS_QC.ipynb + - file: code/SoS/data_preprocessing/genotype/PCA.ipynb + - file: code/SoS/data_preprocessing/phenotype_preprocessing.ipynb + children: + - file: code/SoS/data_preprocessing/phenotype/gene_annotation.ipynb + - file: code/SoS/data_preprocessing/phenotype/phenotype_imputation.ipynb + - file: code/SoS/data_preprocessing/phenotype/phenotype_formatting.ipynb + - file: code/SoS/data_preprocessing/covariate_preprocessing.ipynb + children: + - file: code/SoS/data_preprocessing/covariate/covariate_formatting.ipynb + - file: code/SoS/data_preprocessing/covariate/covariate_hidden_factor.ipynb + - title: QTL Association Testing + children: + - file: code/SoS/association_scan/qtl_association_testing.ipynb + children: + - file: code/SoS/association_scan/TensorQTL/TensorQTL.ipynb + - file: code/SoS/association_scan/quantile_models/qr_and_twas.ipynb + - file: code/SoS/association_scan/qtl_association_postprocessing.ipynb + - title: Multivariate Mixture Models + children: + - file: code/SoS/multivariate_genome/MASH.ipynb + children: + - file: code/SoS/multivariate_genome/multivariate_mixture_vignette.ipynb + - file: code/SoS/multivariate_genome/MASH/mash_preprocessing.ipynb + - file: code/SoS/multivariate_genome/MASH/mixture_prior.ipynb + - file: code/SoS/multivariate_genome/MASH/mash_fit.ipynb + - file: code/SoS/multivariate_genome/MASH/mash_posterior.ipynb + - title: Multiomics Regression Models + children: + - file: code/SoS/mnm_analysis/mnm_miniprotocol.ipynb + children: + - file: code/SoS/mnm_analysis/univariate_fine_mapping_twas_vignette.ipynb + - file: code/SoS/mnm_analysis/multivariate_multigene_fine_mapping_vignette.ipynb + - file: code/SoS/mnm_analysis/univariate_fine_mapping_fsusie_vignette.ipynb + - file: code/SoS/mnm_analysis/multivariate_fine_mapping_vignette.ipynb + - file: code/SoS/mnm_analysis/summary_stats_finemapping_vignette.ipynb + - file: code/SoS/mnm_analysis/mnm_methods/mnm_regression.ipynb + - file: code/SoS/mnm_analysis/mnm_methods/rss_analysis.ipynb + - file: code/SoS/mnm_analysis/mnm_methods/qtl_rss_analysis.ipynb + - file: code/SoS/mnm_analysis/mnm_postprocessing.ipynb + - title: GWAS Integration + children: + - file: code/SoS/pecotmr_integration/gwas_integration.ipynb + children: + - file: code/SoS/pecotmr_integration/SuSiE_enloc.ipynb + - file: code/SoS/pecotmr_integration/twas_ctwas.ipynb + - file: code/SoS/pecotmr_integration/intact.ipynb + - file: code/SoS/mnm_analysis/mnm_methods/colocboost.ipynb + - title: Enrichment and Validation + children: + - file: code/SoS/enrichment/enrichment_validation.ipynb + children: + - file: code/SoS/enrichment/eoo_enrichment.ipynb + - file: code/SoS/enrichment/gsea.ipynb + - file: code/SoS/enrichment/gregor.ipynb + - file: code/SoS/enrichment/sldsc_enrichment.ipynb + - title: xQTL Modifier Score + children: + - file: code/SoS/xqtl_modifier_score/ems_training.ipynb + - file: code/SoS/xqtl_modifier_score/ems_prediction.ipynb + - title: About + children: + - file: CONTRIBUTORS.md +site: + template: book-theme + options: + logo: website/xqtl_wf.png diff --git a/recipe.yaml b/recipe.yaml deleted file mode 100644 index e69de29bb..000000000 diff --git a/tests/notebooks/molecular_phenotypes/calling/test_methylation_calling.py b/tests/notebooks/molecular_phenotypes/calling/test_methylation_calling.py index cfad99d67..28e7fb086 100644 --- a/tests/notebooks/molecular_phenotypes/calling/test_methylation_calling.py +++ b/tests/notebooks/molecular_phenotypes/calling/test_methylation_calling.py @@ -14,7 +14,6 @@ from helpers.r_runner import rscript_bin NB = "pipeline/methylation_calling.ipynb" -CROSS = "data/cross_reactive_probe_Hop2020.txt" SAMPLES = [("5723646052", "R02C02", "GroupA_3"), ("5723646052", "R04C01", "GroupA_2"), ("5723646052", "R05C02", "GroupB_3"), ("5723646053", "R04C02", "GroupB_1"), ("5723646053", "R05C02", "GroupA_1"), ("5723646053", "R06C02", "GroupB_2")] diff --git a/tests/scripts/test_documented_command_paths.py b/tests/scripts/test_documented_command_paths.py index f2843a8dc..edde6b9b8 100644 --- a/tests/scripts/test_documented_command_paths.py +++ b/tests/scripts/test_documented_command_paths.py @@ -10,14 +10,32 @@ FIXTURE_PATH = re.compile(r"tests/fixtures/[A-Za-z0-9_.@+/*?\[\]{}-]+") STALE_INPUT = re.compile(r"(? 1} + assert not collisions, f"TOC entries share a published URL: {collisions}" diff --git a/website/_config.yml b/website/_config.yml deleted file mode 100644 index e891e5a37..000000000 --- a/website/_config.yml +++ /dev/null @@ -1,40 +0,0 @@ -# Book settings -# Learn more at https://jupyterbook.org/customize/config.html -# Test comment - -title: FunGen-xQTL Consortium -author: The NIH/NIA Alzheimer's Disease Sequencing Project Functional Genomics xQTL Consortium -logo: website/xqtl_wf.png -copyright: 2021+, FunGen xQTL Analysis Working Group -only_build_toc_files: true - - -# Force re-execution of notebooks on each build. -# See https://jupyterbook.org/content/execute.html -execute: - execute_notebooks: 'off' - -# Define the name of the latex output file for PDF builds -latex: - latex_documents: - targetname: website/xqtl_pipelines.tex - -# Add a bibtex file so that we can create citations -bibtex_bibfiles: - - website/references.bib - -# Information about where the book exists on the web -repository: - url: https://github.com/statfungen/xqtl-protocol # Online location of your book - path_to_book: "" # Optional path to your book, relative to the repository root - branch: gh-pages # Which branch of the repository should be used when creating links (optional) - -# Add GitHub buttons to your book -# See https://jupyterbook.org/customize/config.html#add-a-link-to-your-repository -html: - use_issues_button: true - use_edit_page_button: true - extra_navbar: Supported by NIH NIA - -launch_buttons: - binderhub_url: "" diff --git a/website/_toc.yml b/website/_toc.yml deleted file mode 100644 index e9bf8cfc1..000000000 --- a/website/_toc.yml +++ /dev/null @@ -1,110 +0,0 @@ -# Table of contents -# Learn more at https://jupyterbook.org/customize/toc.html -# Notebook files are staged flat from code/SoS by website/build_flat_book.py. - -format: jb-book -root: README -parts: - - caption: Getting started - chapters: - - file: xqtl_protocol_demo.ipynb - - file: xqtl_protocol_workflow_builder.ipynb - - caption: Reference data - chapters: - - file: reference_data.ipynb - sections: - - file: reference_data_preparation.ipynb - - file: generalized_TADB.ipynb - - file: ld_prune_reference.ipynb - - file: rss_ld_sketch.ipynb - - caption: Molecular Phenotypes - chapters: - - file: bulk_expression.ipynb - sections: - - file: RNA_calling.ipynb - - file: bulk_expression_QC.ipynb - - file: bulk_expression_normalization.ipynb - - file: single_cell.ipynb - sections: - - file: snRNAseq_preprocessing.ipynb - - file: pseudobulk_preprocessing.ipynb - - file: methylation.ipynb - sections: - - file: methylation_calling.ipynb - - file: splicing.ipynb - sections: - - file: splicing_calling.ipynb - - file: splicing_normalization.ipynb - - file: apa.ipynb - sections: - - file: apa_calling.ipynb - - file: apa_impute.ipynb - - caption: Data Preprocessing - chapters: - - file: genotype_preprocessing.ipynb - sections: - - file: VCF_QC.ipynb - - file: genotype_formatting.ipynb - - file: GWAS_QC.ipynb - - file: PCA.ipynb - - file: phenotype_preprocessing.ipynb - sections: - - file: gene_annotation.ipynb - - file: phenotype_imputation.ipynb - - file: phenotype_formatting.ipynb - - file: covariate_preprocessing.ipynb - sections: - - file: covariate_formatting.ipynb - - file: covariate_hidden_factor.ipynb - - caption: QTL Association Testing - chapters: - - file: qtl_association_testing.ipynb - sections: - - file: TensorQTL.ipynb - - file: qr_and_twas.ipynb - - file: qtl_association_postprocessing.ipynb - - caption: Multivariate Mixture Models - chapters: - - file: MASH.ipynb - sections: - - file: multivariate_mixture_vignette.ipynb - - file: mash_preprocessing.ipynb - - file: mixture_prior.ipynb - - file: mash_fit.ipynb - - file: mash_posterior.ipynb - - caption: Multiomics Regression Models - chapters: - - file: mnm_miniprotocol.ipynb - sections: - - file: univariate_fine_mapping_twas_vignette.ipynb - - file: multivariate_multigene_fine_mapping_vignette.ipynb - - file: univariate_fine_mapping_fsusie_vignette.ipynb - - file: multivariate_fine_mapping_vignette.ipynb - - file: summary_stats_finemapping_vignette.ipynb - - file: mnm_regression.ipynb - - file: rss_analysis.ipynb - - file: qtl_rss_analysis.ipynb - - file: mnm_postprocessing.ipynb - - caption: GWAS Integration - chapters: - - file: gwas_integration.ipynb - sections: - - file: SuSiE_enloc.ipynb - - file: twas_ctwas.ipynb - - file: intact.ipynb - - file: colocboost.ipynb - - caption: Enrichment and Validation - chapters: - - file: enrichment_validation.ipynb - sections: - - file: eoo_enrichment.ipynb - - file: gsea.ipynb - - file: gregor.ipynb - - file: sldsc_enrichment.ipynb - - caption: xQTL Modifier Score - chapters: - - file: ems_training.ipynb - - file: ems_prediction.ipynb - - caption: About - chapters: - - file: CONTRIBUTORS diff --git a/website/build_flat_book.py b/website/build_flat_book.py deleted file mode 100644 index 3543faafa..000000000 --- a/website/build_flat_book.py +++ /dev/null @@ -1,260 +0,0 @@ -#!/usr/bin/env python3 -"""Stage xQTL protocol notebooks as a flat Jupyter Book source tree. - -The repository keeps notebooks under code/SoS//, but the public site -uses flat page names such as reference_data.html. This script creates a -temporary build source with notebook basenames at the root, rewrites temporary -page-to-page links to those flat names, and can emit compatibility redirects for -older code/... URLs after the HTML build. -""" - -from __future__ import annotations - -import argparse -import html -import json -import os -import re -import shutil -import sys -from pathlib import Path, PurePosixPath -from urllib.parse import urlsplit, urlunsplit - -REPO_ROOT = Path(__file__).resolve().parents[1] -SOURCE_ROOT = REPO_ROOT / "code" / "SoS" -IMAGE_ROOT = REPO_ROOT / "code" / "images" -WEBSITE_ROOT = REPO_ROOT / "website" -SITE_NETLOC = "statfungen.github.io" -SITE_PREFIX = "/xqtl-protocol/" -MARKDOWN_LINK_RE = re.compile(r"(!?\[[^\]]*\]\()([^)]*)(\))") -HTML_SRC_RE = re.compile(r"(\bsrc=[\"'])([^\"']+)([\"'])", re.IGNORECASE) -TOC_FILE_RE = re.compile(r"\bfile:\s*([^\s#]+)") - - -def iter_files(root: Path, pattern: str): - for path in sorted(root.rglob(pattern)): - if ".ipynb_checkpoints" not in path.parts and path.is_file(): - yield path - - -def unique_by_basename(paths, label: str): - seen = {} - duplicates = {} - for path in paths: - name = path.name - if name in seen: - duplicates.setdefault(name, [seen[name]]).append(path) - else: - seen[name] = path - if duplicates: - lines = [f"Duplicate {label} basenames are not compatible with flat URLs:"] - for name, dupes in sorted(duplicates.items()): - rels = ", ".join(str(p.relative_to(SOURCE_ROOT)) for p in dupes) - lines.append(f" {name}: {rels}") - raise SystemExit("\n".join(lines)) - return seen - - -def notebook_index(): - notebooks = list(iter_files(SOURCE_ROOT, "*.ipynb")) - return unique_by_basename(notebooks, "notebook") - - -def asset_index(): - assets = [ - (path, path.relative_to(SOURCE_ROOT)) - for path in iter_files(SOURCE_ROOT, "*") - if path.suffix != ".ipynb" and path.name != "README.md" - ] - if IMAGE_ROOT.exists(): - assets.extend((path, Path("code/images") / path.name) for path in iter_files(IMAGE_ROOT, "*")) - - by_name = {} - for path, _ in assets: - by_name.setdefault(path.name, path) - return assets, by_name - - -def parse_toc_files(): - toc = WEBSITE_ROOT / "_toc.yml" - files = [] - for match in TOC_FILE_RE.finditer(toc.read_text(encoding="utf-8")): - value = match.group(1).strip('"\'') - if value.endswith(".ipynb"): - files.append(value) - return files - - -def copy_or_link(src: Path, dst: Path): - dst.parent.mkdir(parents=True, exist_ok=True) - if dst.exists() or dst.is_symlink(): - return - try: - dst.symlink_to(src.resolve()) - except OSError: - shutil.copy2(src, dst) - - -def flat_page_for_path(path: str, notebooks: dict[str, Path], assets: dict[str, Path]): - if not path or path.startswith("#"): - return None - name = PurePosixPath(path).name - if not name: - return None - - suffix = PurePosixPath(name).suffix - stem = PurePosixPath(name).stem - if suffix == ".ipynb" and name in notebooks: - return name - if suffix == ".html" and f"{stem}.ipynb" in notebooks: - return f"{stem}.html" - if not suffix and f"{name}.ipynb" in notebooks: - return f"{name}.html" - if suffix and name in assets: - return name - return None - - -def rewrite_url(url: str, notebooks: dict[str, Path], assets: dict[str, Path]): - if not url or url.startswith("#") or url.startswith("mailto:"): - return url - parsed = urlsplit(url) - - if parsed.scheme and not ( - parsed.scheme in {"http", "https"} - and parsed.netloc == SITE_NETLOC - and parsed.path.startswith(SITE_PREFIX) - ): - return url - - path = parsed.path - if parsed.netloc == SITE_NETLOC and path.startswith(SITE_PREFIX): - path = path.removeprefix(SITE_PREFIX) - - flat = flat_page_for_path(path, notebooks, assets) - if not flat: - return url - - if parsed.netloc == SITE_NETLOC: - return urlunsplit((parsed.scheme, parsed.netloc, SITE_PREFIX + flat, parsed.query, parsed.fragment)) - return urlunsplit(("", "", flat, parsed.query, parsed.fragment)) - - -def rewrite_markdown(text: str, notebooks: dict[str, Path], assets: dict[str, Path]): - def replace_markdown(match): - return f"{match.group(1)}{rewrite_url(match.group(2), notebooks, assets)}{match.group(3)}" - - def replace_src(match): - return f"{match.group(1)}{rewrite_url(match.group(2), notebooks, assets)}{match.group(3)}" - - text = MARKDOWN_LINK_RE.sub(replace_markdown, text) - text = HTML_SRC_RE.sub(replace_src, text) - return text - - -def write_flat_markdown(src: Path, dst: Path, notebooks: dict[str, Path], assets: dict[str, Path]): - text = src.read_text(encoding="utf-8") - dst.write_text(rewrite_markdown(text, notebooks, assets), encoding="utf-8") - - -def write_flat_notebook(src: Path, dst: Path, notebooks: dict[str, Path], assets: dict[str, Path]): - nb = json.loads(src.read_text(encoding="utf-8")) - for cell in nb.get("cells", []): - if cell.get("cell_type") != "markdown": - continue - source = cell.get("source", []) - if isinstance(source, list): - text = "".join(source) - cell["source"] = rewrite_markdown(text, notebooks, assets).splitlines(keepends=True) - elif isinstance(source, str): - cell["source"] = rewrite_markdown(source, notebooks, assets) - dst.write_text(json.dumps(nb, ensure_ascii=False, indent=1) + "\n", encoding="utf-8") - - -def stage_book(output: Path): - notebooks = notebook_index() - assets, assets_by_name = asset_index() - toc_files = parse_toc_files() - - missing = [name for name in toc_files if name not in notebooks] - nested = [name for name in toc_files if "/" in name or "\\" in name] - if nested: - raise SystemExit("TOC notebook entries should be flat basenames only:\n " + "\n ".join(nested)) - if missing: - raise SystemExit("TOC notebooks were not found under code/SoS:\n " + "\n ".join(missing)) - - if output.exists(): - shutil.rmtree(output) - output.mkdir(parents=True) - - write_flat_markdown(REPO_ROOT / "README.md", output / "README.md", notebooks, assets_by_name) - if (REPO_ROOT / "CONTRIBUTORS.md").exists(): - write_flat_markdown(REPO_ROOT / "CONTRIBUTORS.md", output / "CONTRIBUTORS.md", notebooks, assets_by_name) - website_out = output / "website" - website_out.mkdir() - for name in ["_config.yml", "_toc.yml", "references.bib", "xqtl_wf.png", "xqtl_wf.svg"]: - src = WEBSITE_ROOT / name - if src.exists(): - shutil.copy2(src, website_out / name) - - for asset, rel in assets: - copy_or_link(asset, output / rel) - copy_or_link(asset, output / asset.name) - - for name, src in notebooks.items(): - write_flat_notebook(src, output / name, notebooks, assets_by_name) - - print(f"Staged {len(notebooks)} notebooks as flat book sources in {output}") - - -def redirect_document(target: str): - escaped = html.escape(target, quote=True) - js_target = json.dumps(target) - return ( - "\n" - "\n" - "Redirecting...\n" - f"\n" - f"\n" - f"\n" - f"

Redirecting to {escaped}.

\n" - ) - - -def write_redirect(path: Path, target: Path, html_root: Path): - path.parent.mkdir(parents=True, exist_ok=True) - rel_target = os.path.relpath(target, path.parent).replace(os.sep, "/") - path.write_text(redirect_document(rel_target), encoding="utf-8") - - -def write_redirects(html_root: Path): - notebooks = notebook_index() - toc_files = parse_toc_files() - count = 0 - for name in toc_files: - src = notebooks[name] - rel_html = src.relative_to(SOURCE_ROOT).with_suffix(".html") - target = html_root / f"{Path(name).stem}.html" - for old_prefix in ("code", "code/SoS"): - redirect_path = html_root / old_prefix / rel_html - write_redirect(redirect_path, target, html_root) - count += 1 - print(f"Wrote {count} compatibility redirects under {html_root}") - - -def main(argv=None): - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--output", type=Path, help="temporary flat Jupyter Book source directory to create") - parser.add_argument("--redirects", type=Path, help="built HTML directory where old code/... redirects should be written") - args = parser.parse_args(argv) - - if not args.output and not args.redirects: - parser.error("provide --output, --redirects, or both") - if args.output: - stage_book(args.output) - if args.redirects: - write_redirects(args.redirects) - - -if __name__ == "__main__": - main() diff --git a/website/build_support.py b/website/build_support.py new file mode 100644 index 000000000..b7e9a6c19 --- /dev/null +++ b/website/build_support.py @@ -0,0 +1,222 @@ +#!/usr/bin/env python3 +"""Build-time helpers for the FunGen-xQTL Jupyter Book site. + +Jupyter Book 2 (MyST) builds straight from the repository layout: the TOC in +myst.yml points at notebooks where they live, MyST derives a flat page slug from +each basename (code/SoS/reference_data/reference_data.ipynb -> /reference-data/) +and resolves relative links and images itself. Nothing needs staging or +rewriting, so this script covers only the three things MyST does not do: + + --stage-widget The workflow builder is a self-contained HTML widget. MyST + renders a {raw} html block as escaped text and strips \n" +) +WIDGET_SUFFIX = "\n" +# Absolute: MyST would otherwise prepend BASE_URL to a root-relative path, or +# content-hash anything it can resolve as a local asset. +WIDGET_URL = f"https://statfungen.github.io/xqtl-protocol/{WIDGET_PAGE}" +WIDGET_LINK = f"**[Open the xQTL Analysis Workflow Builder]({WIDGET_URL})**\n" + + +def toc_entries(): + """Yield every (file, path) pair in the myst.yml table of contents.""" + config = yaml.safe_load(CONFIG.read_text(encoding="utf-8")) + + def walk(items): + for item in items: + if "file" in item: + yield item["file"] + yield from walk(item.get("children", [])) + + return list(walk(config["project"]["toc"])) + + +def toc_notebooks(): + return [f for f in toc_entries() if f.endswith(".ipynb")] + + +# --------------------------------------------------------------------------- # +# Workflow builder widget +# --------------------------------------------------------------------------- # + + +def _markdown_cells(notebook): + return [c for c in notebook.cells if c.cell_type == "markdown"] + + +def extract_widget_html(): + """Return the standalone builder page, built from its notebook.""" + notebook = nbformat.read(WIDGET_NOTEBOOK, as_version=4) + blocks = [] + for cell in _markdown_cells(notebook): + blocks += [m.group(2) for m in RAW_HTML_BLOCK_RE.finditer(cell.source)] + if len(blocks) != 1: + raise SystemExit( + f"Expected exactly one ```{{raw}} html block in {WIDGET_NOTEBOOK.name}, found {len(blocks)}." + ) + return WIDGET_PREFIX + blocks[0] + "\n" + WIDGET_SUFFIX + + +def stage_widget_notebook(): + """Write the book-page copy of the builder notebook: widget replaced by a link.""" + notebook = nbformat.read(WIDGET_NOTEBOOK, as_version=4) + replaced = 0 + for cell in _markdown_cells(notebook): + cell.source, count = RAW_HTML_BLOCK_RE.subn(lambda _: WIDGET_LINK, cell.source) + replaced += count + if replaced != 1: + raise SystemExit(f"Expected to replace exactly one widget block, replaced {replaced}.") + STAGED_WIDGET.parent.mkdir(parents=True, exist_ok=True) + nbformat.write(notebook, STAGED_WIDGET) + print(f"Staged book page {STAGED_WIDGET.relative_to(REPO_ROOT)}") + + +def write_widget(html_root: Path): + target = html_root / WIDGET_PAGE + target.parent.mkdir(parents=True, exist_ok=True) + target.write_text(extract_widget_html(), encoding="utf-8") + print(f"Wrote standalone {WIDGET_PAGE} ({target.stat().st_size} bytes) under {html_root}") + + +# --------------------------------------------------------------------------- # +# Legacy URL redirects +# --------------------------------------------------------------------------- # + + +def published_slugs(html_root: Path): + """Map source path -> published slug, read from MyST's own build output.""" + slugs = {} + for page in sorted(html_root.glob("*.json")): + try: + data = json.loads(page.read_text(encoding="utf-8")) + except (ValueError, OSError): + continue + if "slug" in data and "location" in data: + slugs[data["location"].lstrip("/")] = data["slug"] + if not slugs: + raise SystemExit(f"No MyST page manifests found under {html_root}; run the build first.") + return slugs + + +def redirect_document(target: str): + escaped = html.escape(target, quote=True) + js_target = json.dumps(target) + return ( + "\n" + '\n' + "Redirecting...\n" + f'\n' + f'\n' + f"\n" + f'

Redirecting to {escaped}.

\n' + ) + + +def write_redirect(path: Path, target: Path): + path.parent.mkdir(parents=True, exist_ok=True) + rel_target = os.path.relpath(target, path.parent).replace(os.sep, "/") + path.write_text(redirect_document(rel_target), encoding="utf-8") + + +def write_redirects(html_root: Path): + slugs = published_slugs(html_root) + missing = [f for f in toc_notebooks() if f not in slugs] + if missing: + raise SystemExit( + "These TOC notebooks were not published (basename collision after " + "slugification?):\n " + "\n ".join(missing) + ) + + count = 0 + for source, slug in sorted(slugs.items()): + path = Path(source) + if path.suffix != ".ipynb": + continue # README.md / CONTRIBUTORS.md have no legacy URLs + is_widget = path.name == WIDGET_NOTEBOOK.name + target = html_root / (WIDGET_PAGE if is_widget else f"{slug}/index.html") + + old_paths = [] + if not is_widget: + # Jupyter Book 1 published /.html. + old_paths.append(html_root / f"{path.stem}.html") + # The builder page is generated elsewhere; its legacy URLs follow the + # notebook it is generated from. + source = WIDGET_NOTEBOOK.relative_to(REPO_ROOT) if is_widget else path + try: + rel = source.relative_to("code/SoS").with_suffix(".html") + except ValueError: + rel = None # generated pages have no repository-relative legacy URL + if rel is not None: + old_paths += [html_root / prefix / rel for prefix in ("code", "code/SoS")] + + for old in old_paths: + write_redirect(old, target) + count += 1 + print(f"Wrote {count} compatibility redirects under {html_root}") + + +def main(argv=None): + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--stage-widget", action="store_true", + help="write the slimmed builder notebook the TOC points at") + parser.add_argument("--widget", type=Path, metavar="HTML_ROOT", + help="write the standalone builder page into a built site") + parser.add_argument("--widget-file", type=Path, + help="write the standalone builder page to this path (refreshes the committed copy)") + parser.add_argument("--redirects", type=Path, metavar="HTML_ROOT", + help="write legacy-URL redirect stubs into a built site") + args = parser.parse_args(argv) + + if not any([args.stage_widget, args.widget, args.widget_file, args.redirects]): + parser.error("provide --stage-widget, --widget, --widget-file or --redirects") + if args.stage_widget: + stage_widget_notebook() + if args.widget: + write_widget(args.widget) + if args.widget_file: + args.widget_file.write_text(extract_widget_html(), encoding="utf-8") + print(f"Wrote {args.widget_file}") + if args.redirects: + write_redirects(args.redirects) + + +if __name__ == "__main__": + main() diff --git a/website/nature_protocol/conversion_notebook.ipynb b/website/nature_protocol/conversion_notebook.ipynb index edfea87fc..b03dad3e7 100644 --- a/website/nature_protocol/conversion_notebook.ipynb +++ b/website/nature_protocol/conversion_notebook.ipynb @@ -74,28 +74,28 @@ "major_sections_keep = [\n", " 'Reference data',\n", " 'Molecular Phenotypes',\n", - " 'Data Pre-processing',\n", + " 'Data Preprocessing',\n", " 'QTL Association Testing',\n", - " 'Multivariate Mixture Model',\n", + " 'Multivariate Mixture Models',\n", " 'Multiomics Regression Models',\n", " 'GWAS Integration',\n", " 'Enrichment and Validation'\n", "]\n", "\n", "miniprotocol_keep = [\n", - " '../../code/reference_data/reference_data.ipynb',\n", - " '../../code/molecular_phenotypes/bulk_expression.ipynb',\n", - " '../../code/molecular_phenotypes/splicing.ipynb',\n", - " '../../code/data_preprocessing/genotype_preprocessing.ipynb',\n", - " '../../code/data_preprocessing/phenotype_preprocessing.ipynb',\n", - " '../../code/data_preprocessing/covariate_preprocessing.ipynb',\n", - " '../../code/association_scan/qtl_association_testing.ipynb',\n", - " '../../code/multivariate_genome/multivariate_mixture_vignette.ipynb',\n", - " '../../code/mnm_analysis/mnm_miniprotocol.ipynb',\n", - " '../../code/pecotmr_integration/SuSiE_enloc.ipynb',\n", - " '../../code/pecotmr_integration/twas_ctwas.ipynb',\n", - " '../../code/mnm_analysis/mnm_methods/colocboost.ipynb',\n", - " '../../code/enrichment/eoo_enrichment.ipynb'\n", + " '../../code/SoS/reference_data/reference_data.ipynb',\n", + " '../../code/SoS/molecular_phenotypes/bulk_expression.ipynb',\n", + " '../../code/SoS/molecular_phenotypes/splicing.ipynb',\n", + " '../../code/SoS/data_preprocessing/genotype_preprocessing.ipynb',\n", + " '../../code/SoS/data_preprocessing/phenotype_preprocessing.ipynb',\n", + " '../../code/SoS/data_preprocessing/covariate_preprocessing.ipynb',\n", + " '../../code/SoS/association_scan/qtl_association_testing.ipynb',\n", + " '../../code/SoS/multivariate_genome/multivariate_mixture_vignette.ipynb',\n", + " '../../code/SoS/mnm_analysis/mnm_miniprotocol.ipynb',\n", + " '../../code/SoS/pecotmr_integration/SuSiE_enloc.ipynb',\n", + " '../../code/SoS/pecotmr_integration/twas_ctwas.ipynb',\n", + " '../../code/SoS/mnm_analysis/mnm_methods/colocboost.ipynb',\n", + " '../../code/SoS/enrichment/eoo_enrichment.ipynb'\n", "]" ] }, @@ -107,11 +107,11 @@ "outputs": [], "source": [ "module_skip = [\n", - " '../../code/association_scan/APEX/APEX.ipynb',\n", - " #'../../code/data_preprocessing/genotype/genotype_formatting.ipynb',\n", - " '../../code/data_preprocessing/genotype/GRM.ipynb',\n", - " '../../code/data_preprocessing/genotype/GWAS_QC.ipynb',\n", - " '../../code/data_preprocessing/genotype/PCA.ipynb'\n", + " '../../code/SoS/graveyard/APEX/APEX.ipynb',\n", + " #'../../code/SoS/data_preprocessing/genotype/genotype_formatting.ipynb',\n", + " '../../code/SoS/graveyard/GRM.ipynb',\n", + " '../../code/SoS/data_preprocessing/genotype/GWAS_QC.ipynb',\n", + " '../../code/SoS/data_preprocessing/genotype/PCA.ipynb'\n", "]" ] }, @@ -141,12 +141,16 @@ "outputs": [], "source": [ "\n", - "# Specify the path to your YAML file\n", - "yaml_file_path = \"../_toc.yml\"\n", + "# Specify the path to your MyST configuration file\n", + "yaml_file_path = \"../../myst.yml\"\n", "\n", "# Load the YAML file\n", "with open(yaml_file_path, \"r\") as file:\n", - " yaml_data = yaml.safe_load(file)" + " yaml_data = yaml.safe_load(file)\n", + "\n", + "# Jupyter Book 2 nests the table of contents under project.toc as a flat list:\n", + "# top-level entries carry either a \"file\" or a \"title\" plus \"children\".\n", + "toc_parts = [item for item in yaml_data[\"project\"][\"toc\"] if \"children\" in item]\n" ] }, { @@ -207,13 +211,13 @@ "#values being lists of the miniprotocol notebooks for that major section\n", "#these values should match the keys used in the 'miniprotocol_dict' below\n", "major_section_dict = {}\n", - "for part in yaml_data['parts']:\n", - " caption = part['caption']\n", + "for part in toc_parts:\n", + " caption = part['title']\n", " #filter\n", " if caption in major_sections_keep: \n", " print(caption)\n", - " print(part['chapters'])\n", - " miniprotocols = [f\"../../{file['file']}\" for file in part['chapters'] if f\"../../{file['file']}\" in miniprotocol_keep]\n", + " print(part['children'])\n", + " miniprotocols = [f\"../../{file['file']}\" for file in part['children'] if f\"../../{file['file']}\" in miniprotocol_keep]\n", " major_section_dict[caption] = miniprotocols\n", "major_section_dict\n", "\n", @@ -284,13 +288,13 @@ "#dictionary with keys being the mininprotocol notebooks and \n", "#values being lists of the module notebooks for the miniprotocol\n", "miniprotocol_dict = {}\n", - "for part in yaml_data['parts']:\n", - " for chapter in part['chapters']:\n", + "for part in toc_parts:\n", + " for chapter in part['children']:\n", " miniprotocol = f\"../../{chapter['file']}\"\n", " #filter\n", " if miniprotocol in miniprotocol_keep:\n", - " if 'sections' in chapter:\n", - " miniprotocol_dict[miniprotocol] = [f\"../../{module['file']}\" for module in chapter['sections']]\n", + " if 'children' in chapter:\n", + " miniprotocol_dict[miniprotocol] = [f\"../../{module['file']}\" for module in chapter['children']]\n", " else:\n", " miniprotocol_dict[miniprotocol] = []\n", "\n", @@ -299,6 +303,7 @@ " miniprotocol_dict[miniprotocol]\n", " \n", "miniprotocol_dict\n", + "\n", "###used to be in this format:\n", "#miniprotocol_dict = {\n", "# f\"{WRKDIR}/bulk_expression/bulk_expression.ipynb\":[\n", @@ -314,7 +319,7 @@ "# f\"{WRKDIR}/data_preprocessing/covariate/covariate_formatting.ipynb\",\n", "# f\"{WRKDIR}/data_preprocessing/covariate/covariate_hidden_factor.ipynb\"\n", "# ]\n", - "#} " + "#}" ] }, { diff --git a/website/nature_protocol/example_manuscript.ipynb b/website/nature_protocol/example_manuscript.ipynb index dc99431f3..7fded7747 100644 --- a/website/nature_protocol/example_manuscript.ipynb +++ b/website/nature_protocol/example_manuscript.ipynb @@ -40,7 +40,7 @@ "id": "7b902786-5566-4adf-8f02-ba4f0fc7c184", "metadata": {}, "source": [ - "Hao Sun, Francis Grenn, developers and leaders on [website table](https://statfungen.github.io/xqtl-protocol/README.html#our-team)" + "Hao Sun, Francis Grenn, developers and leaders on [website table](https://statfungen.github.io/xqtl-protocol/#our-team)" ] }, { diff --git a/website/nature_protocol/example_miniprotocol.ipynb b/website/nature_protocol/example_miniprotocol.ipynb index 954ae9d48..500425dfe 100644 --- a/website/nature_protocol/example_miniprotocol.ipynb +++ b/website/nature_protocol/example_miniprotocol.ipynb @@ -7,11 +7,11 @@ "source": [ "# Example Miniprotocol\n", "\n", - "An example of a miniprotocol page on the [xqtl-protocol website](https://statfungen.github.io/xqtl-protocol/README.html) that could be used in the paper. \n", + "An example of a miniprotocol page on the [xqtl-protocol website](https://statfungen.github.io/xqtl-protocol/) that could be used in the paper. \n", "Parts marked with superscript \"hint_[name]\" would be referenced in the paper to mark places to take text from.\n", "\n", "Everything below would replace or augment what is in the page on the website. \n", - "For example, everything below would replace or augment what is on the [RNA-seq expression](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/bulk_expression.html) page under the main `MOLECULAR PHENOTYPES` section on the website." + "For example, everything below would replace or augment what is on the [RNA-seq expression](https://statfungen.github.io/xqtl-protocol/bulk-expression) page under the main `MOLECULAR PHENOTYPES` section on the website." ] }, { @@ -97,8 +97,7 @@ ] }, "metadata": {}, - "output_type": "display_data", - "source": "markdown" + "output_type": "display_data" } ], "source": [ diff --git a/website/nature_protocol/example_module.ipynb b/website/nature_protocol/example_module.ipynb index 7636109b0..8c5a842b5 100644 --- a/website/nature_protocol/example_module.ipynb +++ b/website/nature_protocol/example_module.ipynb @@ -9,11 +9,11 @@ "source": [ "# Example Module\n", "\n", - "An example of a module section on the [xqtl-protocol website](https://statfungen.github.io/xqtl-protocol/README.html) that could be used in the paper. \n", + "An example of a module section on the [xqtl-protocol website](https://statfungen.github.io/xqtl-protocol/) that could be used in the paper. \n", "Parts marked with superscript \"hint_[name]\" would be referenced in the paper to mark places to take text from.\n", "\n", "Everything below would replace or augment what is in the module page on the website. \n", - "For example, everything below would replace or augment what is on the [Quantifying expression from RNA-seq data](https://statfungen.github.io/xqtl-protocol/code/molecular_phenotypes/calling/RNA_calling.html) page under the `RNA-seq expression` miniprotocol section on the website." + "For example, everything below would replace or augment what is on the [Quantifying expression from RNA-seq data](https://statfungen.github.io/xqtl-protocol/rna-calling) page under the `RNA-seq expression` miniprotocol section on the website." ] }, { diff --git a/xqtl_flowchart.html b/website/xqtl_flowchart.html similarity index 100% rename from xqtl_flowchart.html rename to website/xqtl_flowchart.html From f85c05afb97714d155645dc3b1ed4d2f2ea6c2a3 Mon Sep 17 00:00:00 2001 From: Daniel Nachun Date: Mon, 21 Sep 2026 14:20:24 -0700 Subject: [PATCH 2/3] fix pecotmr issues --- .../pecotmr_integration/fine_mapping_vcf.R | 4 +-- code/script/pecotmr_integration/mash_common.R | 36 +++++++++++++++++++ .../pecotmr_integration/mash_covariance.R | 6 ++-- code/script/pecotmr_integration/mash_prior.R | 6 ++-- code/script/pecotmr_integration/mash_vhat.R | 7 +++- 5 files changed, 52 insertions(+), 7 deletions(-) create mode 100644 code/script/pecotmr_integration/mash_common.R diff --git a/code/script/pecotmr_integration/fine_mapping_vcf.R b/code/script/pecotmr_integration/fine_mapping_vcf.R index 79827d80f..1532fefb3 100644 --- a/code/script/pecotmr_integration/fine_mapping_vcf.R +++ b/code/script/pecotmr_integration/fine_mapping_vcf.R @@ -2,7 +2,7 @@ # fine_mapping_vcf.R # # Write a fine-mapping VCF (per-variant ES / CS / PIP in the sample column) from -# a QtlFineMappingResult / GwasFineMappingResult via pecotmr::writeSumstatsVcf. +# a QtlFineMappingResult / GwasFineMappingResult via pecotmr::writeSumStatsVcf. # Replaces the legacy inline create_vcf() + VariantAnnotation::writeVcf in the # mv_susie / uni_susie cells: the VCF assembly now lives in pecotmr's vcfWriter. # @@ -33,7 +33,7 @@ if (!methods::is(fmr, "FineMappingResultBase")) stop("--input must be a FineMappingResult RDS (got '", class(fmr)[[1L]], "').") dir.create(dirname(argv$output), showWarnings = FALSE, recursive = TRUE) -out <- writeSumstatsVcf( +out <- writeSumStatsVcf( fmr, outputPath = argv$output, sampleName = if (is.na(argv$sample_name)) NULL else argv$sample_name, splitByContext = argv$split_by_context) diff --git a/code/script/pecotmr_integration/mash_common.R b/code/script/pecotmr_integration/mash_common.R new file mode 100644 index 000000000..c7333bc74 --- /dev/null +++ b/code/script/pecotmr_integration/mash_common.R @@ -0,0 +1,36 @@ +# mash_common.R +# +# Shared helpers for the pecotmr_integration MASH wrapper scripts +# (mash_covariance / mash_vhat / mash_prior). Each script sources this from its +# own directory: +# +# .d <- dirname(sub("^--file=", "", +# grep("^--file=", commandArgs(FALSE), value = TRUE)[1L])) +# source(file.path(.d, "mash_common.R")) +# +# pecotmr 0.8.2 camelCased two MASH names: the `flash_nonneg` covariance +# component became `flashNonneg`, and the `simple_specific` Vhat method became +# `simpleSpecific`. The snake_case spellings are still the public interface on +# this side -- they are SoS step names in mixture_prior.ipynb (`[flash_nonneg]`, +# `[vhat_simple_specific]`) and they appear in the output filenames downstream +# steps consume -- so the wrappers keep accepting them and translate here. +# Both spellings are accepted, so callers already using the pecotmr names work. +# +# This is argument marshalling only; no analysis logic (that lives in pecotmr). + +MASH_COMPONENT_ALIASES <- c(flash_nonneg = "flashNonneg") +MASH_VHAT_ALIASES <- c(simple_specific = "simpleSpecific") + +# Map any aliased names to the spelling pecotmr expects, passing others through +# unchanged so pecotmr reports unknown values itself. +apply_mash_aliases <- function(values, aliases) { + hit <- match(values, names(aliases)) + ifelse(is.na(hit), values, unname(aliases[hit])) +} + +# Split a comma/space separated CLI list and normalise the names in it. +split_mash_names <- function(value, aliases) { + parts <- trimws(strsplit(value, "[ ,]+")[[1L]]) + parts <- parts[nzchar(parts)] + apply_mash_aliases(parts, aliases) +} diff --git a/code/script/pecotmr_integration/mash_covariance.R b/code/script/pecotmr_integration/mash_covariance.R index f48eedcfb..5117ab40d 100644 --- a/code/script/pecotmr_integration/mash_covariance.R +++ b/code/script/pecotmr_integration/mash_covariance.R @@ -43,14 +43,16 @@ p <- add_argument(p, "--output", type = "character", help = "output covariance-component RDS") argv <- parse_args(p) +.d <- dirname(sub("^--file=", "", grep("^--file=", commandArgs(FALSE), value = TRUE)[1L])) +source(file.path(.d, "mash_common.R")) + alpha <- if (toupper(argv$effect_model) == "EZ") 1 else 0 dat <- readRDS(argv$data) strong <- qtlSumStatsFromBetaMatrix( as.matrix(dat$strong.b), as.matrix(dat$strong.s), study = "mash") -components <- trimws(strsplit(argv$component, "[ ,]+")[[1L]]) -components <- components[nzchar(components)] +components <- split_mash_names(argv$component, MASH_COMPONENT_ALIASES) nPcs <- if (is.na(argv$npc)) NULL else argv$npc U <- mashCovarianceComponents(list(strong = strong), alpha = alpha, diff --git a/code/script/pecotmr_integration/mash_prior.R b/code/script/pecotmr_integration/mash_prior.R index bc2010b26..275b86f9a 100644 --- a/code/script/pecotmr_integration/mash_prior.R +++ b/code/script/pecotmr_integration/mash_prior.R @@ -53,6 +53,9 @@ p <- add_argument(p, "--output", type = "character", help = "output prior RDS (list(U, w, loglik))") argv <- parse_args(p) +.d <- dirname(sub("^--file=", "", grep("^--file=", commandArgs(FALSE), value = TRUE)[1L])) +source(file.path(.d, "mash_common.R")) + alpha <- if (toupper(argv$effect_model) == "EZ") 1 else 0 dat <- readRDS(argv$data) @@ -71,8 +74,7 @@ if (nzchar(argv$component_files)) { engine = argv$engine, setSeed = argv$seed) } else { # Self-contained mode: build the components here. - components <- trimws(strsplit(argv$components, "[ ,]+")[[1L]]) - components <- components[nzchar(components)] + components <- split_mash_names(argv$components, MASH_COMPONENT_ALIASES) nPcs <- if (is.na(argv$npc)) NULL else argv$npc prior <- mashPriorCovariances(list(strong = strong), alpha = alpha, vhat = vhat, components = components, engine = argv$engine, diff --git a/code/script/pecotmr_integration/mash_vhat.R b/code/script/pecotmr_integration/mash_vhat.R index 6921b4d0f..10e4e245c 100644 --- a/code/script/pecotmr_integration/mash_vhat.R +++ b/code/script/pecotmr_integration/mash_vhat.R @@ -43,6 +43,11 @@ p <- add_argument(p, "--max-iter", type = "integer", default = 6L, p <- add_argument(p, "--output", type = "character", help = "output Vhat RDS") argv <- parse_args(p) +.d <- dirname(sub("^--file=", "", grep("^--file=", commandArgs(FALSE), value = TRUE)[1L])) +source(file.path(.d, "mash_common.R")) + +method <- apply_mash_aliases(argv$method, MASH_VHAT_ALIASES) + alpha <- if (toupper(argv$effect_model) == "EZ") 1 else 0 dat <- readRDS(argv$data) @@ -62,7 +67,7 @@ U <- if (nzchar(argv$prior_data)) { if (is.list(pr) && !is.null(pr$U)) pr$U else pr } else NULL -vhat <- mashResidualCorrelation(ssl, alpha = alpha, method = argv$method, +vhat <- mashResidualCorrelation(ssl, alpha = alpha, method = method, priorCovariances = U, nSubset = argv$n_subset, maxIter = argv$max_iter) From c0de845755ed610a8fe021241d259c970050deb2 Mon Sep 17 00:00:00 2001 From: Daniel Nachun Date: Mon, 21 Sep 2026 15:05:26 -0700 Subject: [PATCH 3/3] fix pecotmr issues --- code/script/pecotmr_integration/mash_common.R | 36 ------------ .../pecotmr_integration/mash_covariance.R | 4 +- code/script/pecotmr_integration/mash_prior.R | 4 +- code/script/pecotmr_integration/mash_vhat.R | 4 +- .../pecotmr_integration/pecotmr_aliases.R | 57 +++++++++++++++++++ .../script/pecotmr_integration/twas_weights.R | 10 +++- 6 files changed, 72 insertions(+), 43 deletions(-) delete mode 100644 code/script/pecotmr_integration/mash_common.R create mode 100644 code/script/pecotmr_integration/pecotmr_aliases.R diff --git a/code/script/pecotmr_integration/mash_common.R b/code/script/pecotmr_integration/mash_common.R deleted file mode 100644 index c7333bc74..000000000 --- a/code/script/pecotmr_integration/mash_common.R +++ /dev/null @@ -1,36 +0,0 @@ -# mash_common.R -# -# Shared helpers for the pecotmr_integration MASH wrapper scripts -# (mash_covariance / mash_vhat / mash_prior). Each script sources this from its -# own directory: -# -# .d <- dirname(sub("^--file=", "", -# grep("^--file=", commandArgs(FALSE), value = TRUE)[1L])) -# source(file.path(.d, "mash_common.R")) -# -# pecotmr 0.8.2 camelCased two MASH names: the `flash_nonneg` covariance -# component became `flashNonneg`, and the `simple_specific` Vhat method became -# `simpleSpecific`. The snake_case spellings are still the public interface on -# this side -- they are SoS step names in mixture_prior.ipynb (`[flash_nonneg]`, -# `[vhat_simple_specific]`) and they appear in the output filenames downstream -# steps consume -- so the wrappers keep accepting them and translate here. -# Both spellings are accepted, so callers already using the pecotmr names work. -# -# This is argument marshalling only; no analysis logic (that lives in pecotmr). - -MASH_COMPONENT_ALIASES <- c(flash_nonneg = "flashNonneg") -MASH_VHAT_ALIASES <- c(simple_specific = "simpleSpecific") - -# Map any aliased names to the spelling pecotmr expects, passing others through -# unchanged so pecotmr reports unknown values itself. -apply_mash_aliases <- function(values, aliases) { - hit <- match(values, names(aliases)) - ifelse(is.na(hit), values, unname(aliases[hit])) -} - -# Split a comma/space separated CLI list and normalise the names in it. -split_mash_names <- function(value, aliases) { - parts <- trimws(strsplit(value, "[ ,]+")[[1L]]) - parts <- parts[nzchar(parts)] - apply_mash_aliases(parts, aliases) -} diff --git a/code/script/pecotmr_integration/mash_covariance.R b/code/script/pecotmr_integration/mash_covariance.R index 5117ab40d..d9ff2d72f 100644 --- a/code/script/pecotmr_integration/mash_covariance.R +++ b/code/script/pecotmr_integration/mash_covariance.R @@ -44,7 +44,7 @@ p <- add_argument(p, "--output", type = "character", argv <- parse_args(p) .d <- dirname(sub("^--file=", "", grep("^--file=", commandArgs(FALSE), value = TRUE)[1L])) -source(file.path(.d, "mash_common.R")) +source(file.path(.d, "pecotmr_aliases.R")) alpha <- if (toupper(argv$effect_model) == "EZ") 1 else 0 dat <- readRDS(argv$data) @@ -52,7 +52,7 @@ dat <- readRDS(argv$data) strong <- qtlSumStatsFromBetaMatrix( as.matrix(dat$strong.b), as.matrix(dat$strong.s), study = "mash") -components <- split_mash_names(argv$component, MASH_COMPONENT_ALIASES) +components <- split_pecotmr_names(argv$component, MASH_COMPONENT_ALIASES) nPcs <- if (is.na(argv$npc)) NULL else argv$npc U <- mashCovarianceComponents(list(strong = strong), alpha = alpha, diff --git a/code/script/pecotmr_integration/mash_prior.R b/code/script/pecotmr_integration/mash_prior.R index 275b86f9a..88753bdb2 100644 --- a/code/script/pecotmr_integration/mash_prior.R +++ b/code/script/pecotmr_integration/mash_prior.R @@ -54,7 +54,7 @@ p <- add_argument(p, "--output", type = "character", argv <- parse_args(p) .d <- dirname(sub("^--file=", "", grep("^--file=", commandArgs(FALSE), value = TRUE)[1L])) -source(file.path(.d, "mash_common.R")) +source(file.path(.d, "pecotmr_aliases.R")) alpha <- if (toupper(argv$effect_model) == "EZ") 1 else 0 dat <- readRDS(argv$data) @@ -74,7 +74,7 @@ if (nzchar(argv$component_files)) { engine = argv$engine, setSeed = argv$seed) } else { # Self-contained mode: build the components here. - components <- split_mash_names(argv$components, MASH_COMPONENT_ALIASES) + components <- split_pecotmr_names(argv$components, MASH_COMPONENT_ALIASES) nPcs <- if (is.na(argv$npc)) NULL else argv$npc prior <- mashPriorCovariances(list(strong = strong), alpha = alpha, vhat = vhat, components = components, engine = argv$engine, diff --git a/code/script/pecotmr_integration/mash_vhat.R b/code/script/pecotmr_integration/mash_vhat.R index 10e4e245c..f5fee7e57 100644 --- a/code/script/pecotmr_integration/mash_vhat.R +++ b/code/script/pecotmr_integration/mash_vhat.R @@ -44,9 +44,9 @@ p <- add_argument(p, "--output", type = "character", help = "output Vhat RDS") argv <- parse_args(p) .d <- dirname(sub("^--file=", "", grep("^--file=", commandArgs(FALSE), value = TRUE)[1L])) -source(file.path(.d, "mash_common.R")) +source(file.path(.d, "pecotmr_aliases.R")) -method <- apply_mash_aliases(argv$method, MASH_VHAT_ALIASES) +method <- apply_pecotmr_aliases(argv$method, MASH_VHAT_ALIASES) alpha <- if (toupper(argv$effect_model) == "EZ") 1 else 0 dat <- readRDS(argv$data) diff --git a/code/script/pecotmr_integration/pecotmr_aliases.R b/code/script/pecotmr_integration/pecotmr_aliases.R new file mode 100644 index 000000000..f6f45f9a6 --- /dev/null +++ b/code/script/pecotmr_integration/pecotmr_aliases.R @@ -0,0 +1,57 @@ +# pecotmr_aliases.R +# +# Shared name aliases for the pecotmr_integration wrapper scripts. Each script +# sources this from its own directory: +# +# .d <- dirname(sub("^--file=", "", +# grep("^--file=", commandArgs(FALSE), value = TRUE)[1L])) +# source(file.path(.d, "pecotmr_aliases.R")) +# +# pecotmr 0.8.2 camelCased its MASH component/Vhat names and its TWAS method +# tokens. The snake_case spellings remain the public interface on this side -- +# they are SoS step names in mixture_prior.ipynb (`[flash_nonneg]`, +# `[vhat_simple_specific]`), they appear in output filenames that downstream +# steps consume, and they are the documented `--twas-methods` values in +# mnm_regression.ipynb -- so the wrappers keep accepting them and translate +# here. Both spellings are accepted, so callers already using the pecotmr +# names work unchanged. +# +# This is argument marshalling only; no analysis logic (that lives in pecotmr). + +# mashCovarianceComponents(components=) / mashPriorCovariances(components=) +MASH_COMPONENT_ALIASES <- c(flash_nonneg = "flashNonneg") + +# mashResidualCorrelation(method=) +MASH_VHAT_ALIASES <- c(simple_specific = "simpleSpecific") + +# twasWeightsPipeline(methods=). Mirrors pecotmr's +# .twasKnownMethodLookupNames(); tokens that are already single words +# (susie, mrash, enet, lasso, scad, mcp, l0learn, mvsusie, mrmash) need no alias. +TWAS_METHOD_ALIASES <- c( + susie_ash = "susieAsh", + susie_inf = "susieInf", + bayes_r = "bayesR", + bayes_l = "bayesL", + bayes_a = "bayesA", + bayes_b = "bayesB", + bayes_c = "bayesC", + bayes_n = "bayesN", + b_lasso = "bLasso", + dpr_vb = "dprVb", + dpr_gibbs = "dprGibbs", + dpr_adaptive_gibbs = "dprAdaptiveGibbs" +) + +# Map any aliased names to the spelling pecotmr expects, passing others through +# unchanged so pecotmr reports unknown values itself. +apply_pecotmr_aliases <- function(values, aliases) { + hit <- match(values, names(aliases)) + ifelse(is.na(hit), values, unname(aliases[hit])) +} + +# Split a comma/space separated CLI list and normalise the names in it. +split_pecotmr_names <- function(value, aliases, split = "[ ,]+") { + parts <- trimws(strsplit(value, split)[[1L]]) + parts <- parts[nzchar(parts)] + apply_pecotmr_aliases(parts, aliases) +} diff --git a/code/script/pecotmr_integration/twas_weights.R b/code/script/pecotmr_integration/twas_weights.R index 32b4d8aab..3c039dcee 100644 --- a/code/script/pecotmr_integration/twas_weights.R +++ b/code/script/pecotmr_integration/twas_weights.R @@ -127,6 +127,9 @@ parser <- add_argument(parser, "--output", help = "Output RDS path", type = "character") argv <- parse_args(parser) +.d <- dirname(sub("^--file=", "", grep("^--file=", commandArgs(FALSE), value = TRUE)[1L])) +source(file.path(.d, "pecotmr_aliases.R")) + # Read a whitespace-delimited ID file into a unique character vector; NULL when # the path is empty / "." / missing. read_ids <- function(p) if (nzchar(p) && p != "." && file.exists(p)) @@ -187,7 +190,12 @@ parsed_method_args <- if (nzchar(argv$method_args) && argv$method_args != "." && contexts_arg <- if (nzchar(argv$contexts) && argv$contexts != ".") trimws(strsplit(argv$contexts, ",", fixed = TRUE)[[1L]]) else NULL -methods <- trimws(strsplit(argv$methods, ",", fixed = TRUE)[[1L]]) +methods <- split_pecotmr_names(argv$methods, TWAS_METHOD_ALIASES, split = ",") +# --method-args is keyed by method token, so normalise those keys too or the +# subset check below would reject an alias the user spelled consistently. +if (!is.null(parsed_method_args)) + names(parsed_method_args) <- apply_pecotmr_aliases(names(parsed_method_args), + TWAS_METHOD_ALIASES) methods_arg <- if (is.null(parsed_method_args)) { if (length(methods) == 1L && methods == "default") "default" else methods } else {