From dfc024ec980e3eb80b289635d06eef3e4dd863ec Mon Sep 17 00:00:00 2001 From: Jammy2211 Date: Fri, 17 Jul 2026 09:41:00 +0100 Subject: [PATCH] feat: read-only results-inspector MCP server (autoassistant/mcp) MCP stdio server with 7 read-only tools over PyAutoFit output directories (list fits, model/posterior/search summaries, inline images), as thin wrappers on the existing aggregator API, for chat harnesses without code execution (Claude Desktop first). Includes the af_inspect_results_mcp skill, .mcp.json registration, audit-sweep wiring and fixture tests. Phase 1 of #12. Co-Authored-By: Claude Fable 5 --- .claude/skills/af_inspect_results_mcp.md | 1 + .mcp.json | 8 ++ autoassistant/audit_skill_apis.py | 4 + autoassistant/mcp/__init__.py | 8 ++ autoassistant/mcp/__main__.py | 3 + autoassistant/mcp/server.py | 133 ++++++++++++++++++++ autoassistant/mcp/tools.py | 148 +++++++++++++++++++++++ autoassistant/tests/test_mcp_tools.py | 126 +++++++++++++++++++ skills/README.md | 3 + skills/af_inspect_results_mcp.md | 92 ++++++++++++++ skills/af_setup_environment.md | 4 +- 11 files changed, 529 insertions(+), 1 deletion(-) create mode 120000 .claude/skills/af_inspect_results_mcp.md create mode 100644 .mcp.json create mode 100644 autoassistant/mcp/__init__.py create mode 100644 autoassistant/mcp/__main__.py create mode 100644 autoassistant/mcp/server.py create mode 100644 autoassistant/mcp/tools.py create mode 100644 autoassistant/tests/test_mcp_tools.py create mode 100644 skills/af_inspect_results_mcp.md diff --git a/.claude/skills/af_inspect_results_mcp.md b/.claude/skills/af_inspect_results_mcp.md new file mode 120000 index 0000000..6dd84cd --- /dev/null +++ b/.claude/skills/af_inspect_results_mcp.md @@ -0,0 +1 @@ +../../skills/af_inspect_results_mcp.md \ No newline at end of file diff --git a/.mcp.json b/.mcp.json new file mode 100644 index 0000000..970b19a --- /dev/null +++ b/.mcp.json @@ -0,0 +1,8 @@ +{ + "mcpServers": { + "pyauto-results-inspector": { + "command": "python", + "args": ["-m", "autoassistant.mcp"] + } + } +} diff --git a/autoassistant/audit_skill_apis.py b/autoassistant/audit_skill_apis.py index 1233ac7..53a354f 100644 --- a/autoassistant/audit_skill_apis.py +++ b/autoassistant/audit_skill_apis.py @@ -416,6 +416,9 @@ def select_files(root: Path, scope: str) -> list[Path]: scripts = [ p for p in sorted((root / "scripts").rglob("*.py")) if p.name not in tooling ] + # The MCP server (autoassistant/mcp/) is the one part of autoassistant/ with + # real API usage rather than alias-pattern text, so it joins the scan. + scripts += sorted((root / "autoassistant" / "mcp").glob("*.py")) if scope == "skills": return skills if scope == "wiki": @@ -438,6 +441,7 @@ def select_idiom_files(root: Path) -> list[Path]: tooling = {"audit_skill_apis.py", "refresh_api_docs.py", "test_api_gate.py"} md = sorted((root / "skills").glob("*.md")) + sorted((root / "wiki").rglob("*.md")) py = [p for p in sorted((root / "scripts").rglob("*.py")) if p.name not in tooling] + py += sorted((root / "autoassistant" / "mcp").glob("*.py")) return md + py diff --git a/autoassistant/mcp/__init__.py b/autoassistant/mcp/__init__.py new file mode 100644 index 0000000..485b14e --- /dev/null +++ b/autoassistant/mcp/__init__.py @@ -0,0 +1,8 @@ +""" +The read-only results-inspector MCP server. + +`tools.py` holds the plain tool functions (no MCP dependency — the test suite +exercises them directly); `server.py` registers them with an MCP stdio server; +`python -m autoassistant.mcp` runs it. Documentation, client configuration and +the design rules live in `skills/af_inspect_results_mcp.md`. +""" diff --git a/autoassistant/mcp/__main__.py b/autoassistant/mcp/__main__.py new file mode 100644 index 0000000..b38a6ee --- /dev/null +++ b/autoassistant/mcp/__main__.py @@ -0,0 +1,3 @@ +from autoassistant.mcp.server import mcp + +mcp.run() diff --git a/autoassistant/mcp/server.py b/autoassistant/mcp/server.py new file mode 100644 index 0000000..1fa4891 --- /dev/null +++ b/autoassistant/mcp/server.py @@ -0,0 +1,133 @@ +""" +The read-only results-inspector MCP stdio server. + +Registers the `tools` functions with an MCP server so chat harnesses without +code execution (Claude Desktop, Claude Code) can inspect PyAutoFit output +directories. Run with `python -m autoassistant.mcp`; client configuration and +the design rules live in `skills/af_inspect_results_mcp.md`. + +Every tool is read-only: nothing here composes models, runs fits, or writes +into `output/`. +""" + +import io +import logging +import sys + +from mcp.server.fastmcp import FastMCP, Image + +from autoassistant.mcp import tools + + +def _route_logging_to_stderr(): + """ + stdout carries the JSON-RPC channel, but autofit's logging config (loaded + on import) attaches stdout stream handlers — one stray log line corrupts + the protocol, so every stdout handler is rebound to stderr. + """ + loggers = [logging.getLogger()] + [ + logging.getLogger(name) for name in logging.root.manager.loggerDict + ] + for logger in loggers: + for handler in getattr(logger, "handlers", []): + if ( + isinstance(handler, logging.StreamHandler) + and getattr(handler, "stream", None) is sys.stdout + ): + handler.setStream(sys.stderr) + + +_route_logging_to_stderr() + +mcp = FastMCP("pyauto-results-inspector") + + +def _png(image) -> Image: + buffer = io.BytesIO() + image.save(buffer, format="PNG") + return Image(data=buffer.getvalue(), format="png") + + +@mcp.tool() +def list_searches( + directory: str, + sort_by: str = "log_evidence", + limit: int = 20, + completed_only: bool = False, +) -> list: + """ + List every model-fit found under `directory` (searched recursively): one + row per fit with its name, unique tag, output directory, completion state, + log evidence, maximum log likelihood and free-parameter count. + + Rows are sorted by `sort_by` (descending; fits without that value last) — + use "log_evidence" for nested samplers or "max_log_likelihood" generally — + and truncated to `limit` (pass 0 for all). The returned `directory` of a + row is what the other tools take as their `directory` argument. + """ + return tools.list_searches( + directory, sort_by=sort_by, limit=limit, completed_only=completed_only + ) + + +@mcp.tool() +def get_model(directory: str) -> dict: + """ + The model that was fitted in one search-output directory: a human-readable + `info` block (component classes, priors) and the full model as a dict. + """ + return tools.get_model(directory) + + +@mcp.tool() +def get_result_summary(directory: str) -> str: + """ + The `model.results` text for one search-output directory: the fit's own + summary of the maximum-likelihood model and (when the search produces + them) parameter estimates with errors. + """ + return tools.get_result_summary(directory) + + +@mcp.tool() +def get_samples_summary(directory: str) -> dict: + """ + Posterior summary for one search-output directory: log evidence (None for + MLE/MCMC searches without one), maximum log likelihood, the model's + parameter paths, and the maximum-likelihood and median-PDF parameter + vectors (`median_pdf_parameters` is None for MLE searches, which have no + PDF). + """ + return tools.get_samples_summary(directory) + + +@mcp.tool() +def get_search_info(directory: str) -> dict: + """ + The non-linear search used in one search-output directory: name, unique + tag, completion state, and the search's serialized settings. + """ + return tools.get_search_info(directory) + + +@mcp.tool() +def list_images(directory: str) -> list: + """ + Names of the visualization images (`image/*.png`) available in one + search-output directory — pass a name (without `.png`) to `fetch_image`. + """ + return tools.list_images(directory) + + +@mcp.tool() +def fetch_image(directory: str, name: str = "subplot_fit") -> Image: + """ + One visualization image from a search-output directory (e.g. + "subplot_fit"), returned inline so it renders directly in chat. Use + `list_images` to see what is available. + """ + return _png(tools.fetch_image(directory, name=name)) + + +if __name__ == "__main__": + mcp.run() diff --git a/autoassistant/mcp/tools.py b/autoassistant/mcp/tools.py new file mode 100644 index 0000000..2ca9181 --- /dev/null +++ b/autoassistant/mcp/tools.py @@ -0,0 +1,148 @@ +""" +Read-only results-inspector tools over PyAutoFit output directories. + +Each function is a thin wrapper over an existing public PyAutoFit aggregator +API (`autofit.aggregator.Aggregator`, `af.SearchOutput`): argument parsing, one call, and +JSON-friendly serialization — nothing more. Any behaviour beyond that belongs +in PyAutoFit itself, not here (`skills/af_inspect_results_mcp.md`, +"glue, not code"). + +This module deliberately has no MCP dependency: `server.py` registers these +functions as MCP tools, and the test suite exercises them without the `mcp` +package installed. +""" + +import contextlib +import json +import sys +from pathlib import Path + +from autoconf.dictable import to_dict + +import autofit as af + +# The directory-backed aggregator — af.Aggregator is the database-backed one, +# and the alias also keeps the API audit from resolving it there. +from autofit.aggregator import Aggregator as DirectoryAggregator + + +@contextlib.contextmanager +def _stdout_to_stderr(): + """ + An MCP stdio server must keep stdout clean — it carries the JSON-RPC + channel — but the directory aggregator prints progress to stdout, + so every autofit call runs with stdout redirected to stderr. + """ + with contextlib.redirect_stdout(sys.stderr): + yield + + +def _float_or_none(value): + try: + return None if value is None else float(value) + except (TypeError, ValueError): + return None + + +def _search_row(search) -> dict: + summary = search.samples_summary + max_lh_sample = getattr(summary, "max_log_likelihood_sample", None) + return dict( + name=search.name, + unique_tag=search.unique_tag, + directory=str(search.directory), + is_complete=search.is_complete, + log_evidence=_float_or_none(getattr(summary, "log_evidence", None)), + max_log_likelihood=_float_or_none( + getattr(max_lh_sample, "log_likelihood", None) + ), + model_free_parameters=getattr(search.model, "prior_count", None), + ) + + +def list_searches( + directory: str, + sort_by: str = "log_evidence", + limit: int = 20, + completed_only: bool = False, +) -> list: + with _stdout_to_stderr(): + aggregator = DirectoryAggregator.from_directory( + directory, completed_only=completed_only + ) + rows = [_search_row(search) for search in aggregator] + rows.sort( + key=lambda row: (row.get(sort_by) is not None, row.get(sort_by)), + reverse=True, + ) + return rows[:limit] if limit else rows + + +def get_model(directory: str) -> dict: + with _stdout_to_stderr(): + model = af.SearchOutput(Path(directory)).model + return dict(info=model.info, model=to_dict(model)) + + +def get_result_summary(directory: str) -> str: + with _stdout_to_stderr(): + return af.SearchOutput(Path(directory)).model_results + + +def get_samples_summary(directory: str) -> dict: + with _stdout_to_stderr(): + search_output = af.SearchOutput(Path(directory)) + summary = search_output.samples_summary + if summary is None: + raise FileNotFoundError( + f"No samples summary found under {directory}/files/ — " + "is this a completed search output directory?" + ) + model = search_output.model + return dict( + log_evidence=_float_or_none(summary.log_evidence), + max_log_likelihood=_float_or_none( + summary.max_log_likelihood_sample.log_likelihood + ), + parameter_paths=[".".join(path) for path in model.paths], + max_log_likelihood_parameters=[ + float(value) + for value in summary.max_log_likelihood_sample.parameter_lists_for_model( + model + ) + ], + # MLE searches (e.g. LBFGS) have no PDF, so no median-PDF sample. + median_pdf_parameters=None + if summary.median_pdf_sample is None + else [ + float(value) + for value in summary.median_pdf_sample.parameter_lists_for_model( + model + ) + ], + ) + + +def get_search_info(directory: str) -> dict: + search_json = Path(directory) / "files" / "search.json" + with _stdout_to_stderr(): + search_output = af.SearchOutput(Path(directory)) + return dict( + name=search_output.name, + unique_tag=search_output.unique_tag, + is_complete=search_output.is_complete, + search=json.loads(search_json.read_text()) + if search_json.exists() + else None, + ) + + +def list_images(directory: str) -> list: + # Visualization outputs live under /image/ (SearchOutput.image + # reads there, despite its docstring saying `files/`). + return sorted(path.name for path in (Path(directory) / "image").glob("*.png")) + + +def fetch_image(directory: str, name: str = "subplot_fit"): + with _stdout_to_stderr(): + return af.SearchOutput(Path(directory)).image(name) diff --git a/autoassistant/tests/test_mcp_tools.py b/autoassistant/tests/test_mcp_tools.py new file mode 100644 index 0000000..75cad2b --- /dev/null +++ b/autoassistant/tests/test_mcp_tools.py @@ -0,0 +1,126 @@ +""" +Tests for the results-inspector MCP tool functions (`autoassistant/mcp/tools.py`). + +The fixture runs a real (tiny) `af.LBFGS` fit of the `af.ex.Gaussian` toy so the +output directory always matches the installed PyAutoFit's on-disk format — +a frozen fixture directory would silently drift. LBFGS is an MLE search, so +`log_evidence` is legitimately absent; tests assert that shape rather than +skipping it. +""" + +from pathlib import Path + +import numpy as np +import pytest +from PIL import Image + +import autofit as af +from autoconf import conf + +from autoassistant.mcp import tools + +ROOT = Path(__file__).resolve().parents[2] + + +@pytest.fixture(scope="module") +def output_root(tmp_path_factory): + root = tmp_path_factory.mktemp("output") + conf.instance.push(new_path=str(ROOT / "config"), output_path=str(root)) + + xvalues = np.arange(100.0) + gaussian = af.ex.Gaussian(centre=50.0, normalization=25.0, sigma=10.0) + data = gaussian.model_data_from(xvalues=xvalues) + noise_map = np.full(fill_value=1.0, shape=data.shape) + analysis = af.ex.Analysis(data=data, noise_map=noise_map) + + for name in ("fit_0", "fit_1"): + search = af.LBFGS(name=name, path_prefix="mcp_fixture") + search.fit(model=af.Model(af.ex.Gaussian), analysis=analysis) + + return root + + +@pytest.fixture(scope="module") +def fit_directory(output_root): + directory = sorted( + marker.parent for marker in output_root.rglob(".completed") + )[0] + (directory / "image").mkdir(exist_ok=True) + Image.new("RGB", (32, 16), color=(200, 40, 40)).save( + directory / "image" / "subplot_fit.png" + ) + return directory + + +def test_list_searches(output_root, fit_directory): + rows = tools.list_searches(str(output_root), sort_by="max_log_likelihood") + + assert len(rows) == 2 + assert {row["name"] for row in rows} == {"fit_0", "fit_1"} + for row in rows: + assert row["is_complete"] is True + assert isinstance(row["max_log_likelihood"], float) + assert row["log_evidence"] is None + assert row["model_free_parameters"] == 3 + + +def test_list_searches_limit(output_root): + assert len(tools.list_searches(str(output_root), limit=1)) == 1 + + +def test_list_searches_sorts_none_last(output_root): + rows = tools.list_searches(str(output_root), sort_by="log_evidence") + assert len(rows) == 2 + + +def test_get_model(fit_directory): + result = tools.get_model(str(fit_directory)) + + assert "Gaussian" in result["info"] + assert result["model"]["class_path"].endswith("Gaussian") + + +def test_get_result_summary(fit_directory): + text = tools.get_result_summary(str(fit_directory)) + + assert "Maximum Log Likelihood" in text + + +def test_get_samples_summary(fit_directory): + summary = tools.get_samples_summary(str(fit_directory)) + + assert summary["log_evidence"] is None + assert isinstance(summary["max_log_likelihood"], float) + assert summary["parameter_paths"] == [ + "centre", + "normalization", + "sigma", + ] + assert len(summary["max_log_likelihood_parameters"]) == 3 + assert summary["median_pdf_parameters"] is None + assert summary["max_log_likelihood_parameters"][0] == pytest.approx( + 50.0, abs=1.0 + ) + + +def test_get_search_info(fit_directory): + info = tools.get_search_info(str(fit_directory)) + + assert info["name"] == "fit_0" + assert info["is_complete"] is True + assert info["search"] is not None + + +def test_list_images(fit_directory): + assert "subplot_fit.png" in tools.list_images(str(fit_directory)) + + +def test_fetch_image(fit_directory): + image = tools.fetch_image(str(fit_directory), name="subplot_fit") + + assert image.size == (32, 16) + + +def test_fetch_image_missing(fit_directory): + with pytest.raises(Exception): + tools.fetch_image(str(fit_directory), name="no_such_image") diff --git a/skills/README.md b/skills/README.md index 71fd255..01f6f58 100644 --- a/skills/README.md +++ b/skills/README.md @@ -55,6 +55,9 @@ configured) via symlinks; the canonical files live here. inspection of the Result; runtime triage. - [`af_load_results.md`](./af_load_results.md) — posterior summaries, errors, evidence, and bulk result loading via the aggregator. +- [`af_inspect_results_mcp.md`](./af_inspect_results_mcp.md) — the read-only + results-inspector MCP server: browse fits, summaries and result images from chat + harnesses without code execution (Claude Desktop first). - [`af_chain_searches.md`](./af_chain_searches.md) — pass one fit's result into the next fit's priors or start points; staged pipelines. - [`af_custom_analysis.md`](./af_custom_analysis.md) — extra Analysis inputs, custom diff --git a/skills/af_inspect_results_mcp.md b/skills/af_inspect_results_mcp.md new file mode 100644 index 0000000..684a1a0 --- /dev/null +++ b/skills/af_inspect_results_mcp.md @@ -0,0 +1,92 @@ +--- +name: af_inspect_results_mcp +description: Run and configure the read-only results-inspector MCP server, which lets chat harnesses without code execution (Claude Desktop, Claude Code) inspect PyAutoFit output directories — list fits ranked by evidence, read model and posterior summaries, and view result images inline in chat. Use when the user wants to "browse/inspect my results from chat", asks about the MCP server, or wants Claude Desktop wired to their output folder. Not for loading results in Python (that is `af_load_results`) and not for running fits. +user-invocable: true +--- + +# The results-inspector MCP server + +`autoassistant/mcp/` is a read-only MCP (Model Context Protocol) stdio server over +PyAutoFit output directories. It exists for harnesses that cannot execute code — a +Claude Desktop chat gets tools to list fits, read summaries and display result images +inline, against the same `output/` folder a script-based session works with. + +It is deliberately **not** a fitting interface: composing models and running searches +stay python-first through the other skills. Exposing `search.fit` through a JSON tool +schema flattens the compositional API and is out of scope by design. + +## Orient + +- Server: `autoassistant/mcp/server.py`, run as `python -m autoassistant.mcp` from the + repo root (stdio; nothing listens on a port). +- Requires the `mcp` package in the PyAutoFit environment: `pip install mcp` + (assistant-environment dependency only — never add it to library requirements). +- All tools are read-only; exceptions surface as MCP tool errors, nothing is written. + +## Tools + +| Tool | Returns | +|------|---------| +| `list_searches(directory, sort_by="log_evidence", limit=20, completed_only=False)` | One row per fit found recursively: name, unique tag, directory, completion, log evidence, max log likelihood, free-parameter count. Row `directory` values feed the other tools. | +| `get_model(directory)` | Human-readable model `info` + full model dict. | +| `get_result_summary(directory)` | The `model.results` text. | +| `get_samples_summary(directory)` | Log evidence, max log likelihood, parameter paths, max-LH and median-PDF vectors (`None` fields where a search legitimately lacks them, e.g. MLE). | +| `get_search_info(directory)` | Search name, unique tag, completion, serialized settings. | +| `list_images(directory)` | Names of `image/*.png` visualizations. | +| `fetch_image(directory, name="subplot_fit")` | The image itself, rendered inline in chat. | + +Wrappers over `PyAutoFit:autofit/aggregator/aggregator.py` (`Aggregator.from_directory`) +and `PyAutoFit:autofit/aggregator/search_output.py` (`SearchOutput`). + +## Configure a client + +**Claude Desktop** (`claude_desktop_config.json`, Settings → Developer → Edit Config): + +```json +{ + "mcpServers": { + "pyauto-results-inspector": { + "command": "/absolute/path/to/venv/bin/python", + "args": ["-m", "autoassistant.mcp"], + "env": { "PYTHONPATH": "/absolute/path/to/autofit_assistant" } + } + } +} +``` + +Use the interpreter that has PyAutoFit + `mcp` installed. Ask about fits by absolute +path ("list the fits under /home/me/project/output ranked by evidence"). + +MCP clients spawn the server with a **minimal environment** — nothing from your shell +propagates. Anything the stack needs (`PYTHONPATH` for editable/source checkouts, +`NUMBA_CACHE_DIR`/`MPLCONFIGDIR` in restricted setups) must be declared in the +config's `env` block. + +**Claude Code**: the repo-root `.mcp.json` registers the same server automatically for +sessions opened in this repo (marginal there — a Claude Code session can already run +the aggregator in Python — but it costs nothing and exercises the same wiring). + +## Deployment tiers + +1. **Local stdio (built, above)** — Claude Desktop / Claude Code on the machine that + holds `output/`. +2. **Remote (documented only)** — claude.ai web/mobile custom connectors and ChatGPT + developer mode speak MCP but only to servers reachable over the public internet + (no stdio): expose the server via `mcp.run(transport="streamable-http")` behind an + ngrok/cloudflared tunnel. Not built or hardened here; do not tunnel a machine you + care about without thinking about auth. +3. **Hosted (future)** — a collaboration-scale deployment next to shared outputs + (e.g. sample-wide triage). Same tools; hosting, auth and scale are its own task. + +## Design rules (maintainers) + +- **Glue, not code.** Every tool is argument parsing + one existing public PyAutoFit + call + serialization. If a tool needs more, add the method to PyAutoFit first. +- **Read-only.** No fit-running, no compute, no writes into `output/`. +- **stdout is the protocol.** Autofit calls run under `tools._stdout_to_stderr()` and + `server._route_logging_to_stderr()` rebinds stdout log handlers — keep both when + adding tools; a single stray print corrupts the JSON-RPC channel. +- **Anti-drift.** `autoassistant/mcp/*.py` is scanned by `al`/`af` symbol audits via + `autoassistant/audit_skill_apis.py` (`--scope scripts`); tests + (`autoassistant/tests/test_mcp_tools.py`) build their fixture by running a real + tiny fit so format drift fails loudly. diff --git a/skills/af_setup_environment.md b/skills/af_setup_environment.md index fd62b4e..b3e7217 100644 --- a/skills/af_setup_environment.md +++ b/skills/af_setup_environment.md @@ -72,7 +72,9 @@ pip install autofit Python ≥ 3.9 works in principle (the repos declare `requires-python = ">=3.9"`); 3.11 is the recommended baseline — it's what the workspace tooling targets. If the user's likelihood is JAX-able and they want gradient-based tooling, `pip install "autofit[jax]"` -adds the JAX extra. +adds the JAX extra. If they want the results-inspector MCP server +([`af_inspect_results_mcp`](./af_inspect_results_mcp.md)), `pip install mcp` adds the +protocol SDK (assistant-environment only — never a library requirement). Verify: