diff --git a/.github/workflows/codeql.yml b/.github/workflows/codeql.yml index edb6d6d..203c863 100644 --- a/.github/workflows/codeql.yml +++ b/.github/workflows/codeql.yml @@ -24,17 +24,17 @@ jobs: steps: - name: Checkout repository - uses: actions/checkout@v4 + uses: actions/checkout@11bd71901bbe5b1630ceea73d27597364c9af683 # v4.2.2 - name: Initialize CodeQL - uses: github/codeql-action/init@v3 + uses: github/codeql-action/init@2d92b76c45b91eb80fc44c74ce3fce0ee94e8f9d # v3.30.0 with: languages: ${{ matrix.language }} - name: Autobuild - uses: github/codeql-action/autobuild@v3 + uses: github/codeql-action/autobuild@2d92b76c45b91eb80fc44c74ce3fce0ee94e8f9d # v3.30.0 - name: Perform CodeQL Analysis - uses: github/codeql-action/analyze@v3 + uses: github/codeql-action/analyze@2d92b76c45b91eb80fc44c74ce3fce0ee94e8f9d # v3.30.0 with: category: "/language:${{ matrix.language }}" diff --git a/README.md b/README.md index 6e58ba3..03e25cd 100644 --- a/README.md +++ b/README.md @@ -218,10 +218,12 @@ Errors never raise across the tool boundary: every handler catches its exception structured `ErrorResponse` dict instead, so a calling agent always gets a parseable result. > [!WARNING] -> NeuronScope puts no size cap or timeout on model loading or forward passes. If you expose -> this MCP server somewhere an untrusted agent can call it, put a resource limit around the -> process (a cgroup, `ulimit`, or a container memory/CPU cap) rather than relying on -> NeuronScope to refuse an oversized request on its own. +> NeuronScope caps model size (2B parameters by default, `NEURONSCOPE_MAX_MODEL_PARAMS`) +> and how many models can load at once (1 by default, `NEURONSCOPE_MAX_CONCURRENT_LOADS`), +> but puts no timeout on model loading or forward passes. If you expose this MCP server +> somewhere an untrusted agent can call it, still put a resource limit around the process +> (a cgroup, `ulimit`, or a container memory/CPU cap) as defense in depth rather than +> relying on these in-process caps alone. Transport is stdio, so there is nothing to host: the MCP client spawns the server as a local subprocess. Source: [`neuronscope/mcp_server.py`](neuronscope/mcp_server.py). @@ -324,12 +326,19 @@ separately if you're redistributing a bundled product rather than just calling does not do full path-patching with clean/corrupted prompt pairs, and it will not catch interaction effects between components. The `--json` output states this in its `method` field so a caller doesn't have to trust prose to know the caveat. -- **No size cap or timeout on model loading or forward passes.** NeuronScope loads whatever - model weights the caller asks for and runs the forward pass to completion, with no built-in - limit on model size or wall-clock time. If you run the MCP server somewhere an untrusted - agent can call it, put a resource limit around the process (a cgroup, `ulimit`, or a - container memory/CPU cap) rather than relying on NeuronScope to refuse an oversized - request on its own. +- **No timeout on model loading or forward passes.** Once a request passes the resource + caps below, NeuronScope runs the load and the forward pass to completion with no + built-in wall-clock limit. If you run the MCP server somewhere an untrusted agent can + call it, put a resource limit around the process (a cgroup, `ulimit`, or a container + memory/CPU cap) as defense in depth. +- **Model size and load concurrency are capped, but only in-process.** `neuronscope/core/limits.py` + rejects a model over `NEURONSCOPE_MAX_MODEL_PARAMS` (2B parameters by default) before any + weights are downloaded, and rejects a load once `NEURONSCOPE_MAX_CONCURRENT_LOADS` (1 by + default) other loads are already in flight, both with a structured error rather than a + hang or a crash. The size check is best-effort: if a model's parameter count can't be + determined (for example, fully offline with nothing cached yet), it fails open rather + than blocking a legitimate request, so it's not a hard guarantee on its own -- pair it + with a process-level resource limit for untrusted deployments. - **`HookedTransformer.from_pretrained` is deprecated upstream.** TransformerLens 3.6.0 emits a `DeprecationWarning` pointing at `TransformerBridge.boot_transformers` as the replacement. It still works today, and every command shown in this README ran on it, but diff --git a/neuronscope/cli.py b/neuronscope/cli.py index 7eee8ac..15182bc 100644 --- a/neuronscope/cli.py +++ b/neuronscope/cli.py @@ -19,6 +19,7 @@ from neuronscope import __version__ from neuronscope.backends.transformer_lens import COMPONENT_HOOK_TEMPLATES +from neuronscope.core.limits import ModelTooLargeError, TooManyConcurrentModelLoadsError from neuronscope.core.registry import UnsupportedModelError from neuronscope.core.trace import ( LayerOutOfRangeError, @@ -58,6 +59,12 @@ def _handle_error(operation: str, exc: Exception, as_json: bool) -> int: elif isinstance(exc, LayerOutOfRangeError): error_type = "LayerOutOfRangeError" exit_code = EXIT_ERROR + elif isinstance(exc, ModelTooLargeError): + error_type = "ModelTooLargeError" + exit_code = EXIT_ERROR + elif isinstance(exc, TooManyConcurrentModelLoadsError): + error_type = "TooManyConcurrentModelLoadsError" + exit_code = EXIT_ERROR else: error_type = type(exc).__name__ exit_code = EXIT_ERROR diff --git a/neuronscope/core/limits.py b/neuronscope/core/limits.py new file mode 100644 index 0000000..7d7b846 --- /dev/null +++ b/neuronscope/core/limits.py @@ -0,0 +1,164 @@ +"""Resource caps on MCP/CLI model loading. + +NeuronScope loads a fresh copy of whatever model a caller names on every single +`trace`/`activations`/`patch`/`circuit` call -- nothing here caches a loaded model +across calls (see `core/trace.py`'s module docstring). That means two things are +otherwise completely unbounded: + +- **Model size.** A caller can name an arbitrarily large checkpoint and NeuronScope + will happily try to load the whole thing into memory. +- **Concurrency.** Nothing stops several tool calls in flight at once from each + bringing a full model into memory at the same time, multiplying the memory hit. + +Either one alone is enough for a malicious or merely careless MCP client to exhaust +the host's memory. This module adds two independent, configurable caps that close +both: a parameter-count ceiling checked *before* any weights are downloaded/loaded, +and a concurrency ceiling on in-flight loads. Both fail fast with a clear, structured +error (surfaced the same way as `UnsupportedModelError` etc.) instead of blocking or +crashing -- this is a resource *cap*, not a queue. + +This is defense in depth, not a replacement for the process-level resource limit +(cgroup/ulimit/container memory cap) the README still recommends for untrusted +deployments -- it catches the common case cheaply, in-process, with no operator setup +required. +""" + +from __future__ import annotations + +import os +import threading + +# ~2B parameters is comfortably loadable in float32 on a typical 16GB-RAM machine +# (roughly 8GB resident for the weights alone, leaving headroom for activations and +# the activation cache) and covers every model named as a first-class example in the +# README and tool descriptions (GPT-2, Pythia up to ~1.4B, small Llama/Gemma/Qwen +# checkpoints). Bigger checkpoints still work fine -- raise the cap via +# NEURONSCOPE_MAX_MODEL_PARAMS on hardware that can actually hold them. +DEFAULT_MAX_MODEL_PARAMS = 2_000_000_000 + +# Model loading -- and the forward pass that follows it -- is memory-hungry and not +# meaningfully parallelizable on the CPU-by-default, single-process deployment this +# ships as, so the default only allows one load in flight at a time. Raise +# NEURONSCOPE_MAX_CONCURRENT_LOADS on a machine with enough memory to genuinely hold +# several models at once. +DEFAULT_MAX_CONCURRENT_LOADS = 1 + +MAX_MODEL_PARAMS_ENV = "NEURONSCOPE_MAX_MODEL_PARAMS" +MAX_CONCURRENT_LOADS_ENV = "NEURONSCOPE_MAX_CONCURRENT_LOADS" + + +class ModelTooLargeError(Exception): + """Raised when a requested model's parameter count exceeds the configured cap.""" + + def __init__(self, model_name: str, n_params: int, max_params: int): + self.model_name = model_name + self.n_params = n_params + self.max_params = max_params + super().__init__( + f"Model '{model_name}' has {n_params:,} parameters, which exceeds the " + f"configured cap of {max_params:,} (set via {MAX_MODEL_PARAMS_ENV}). Raise " + f"{MAX_MODEL_PARAMS_ENV} if your hardware can hold this model, or request a " + f"smaller one." + ) + + +class TooManyConcurrentModelLoadsError(Exception): + """Raised when a load request arrives while the concurrency cap is already saturated.""" + + def __init__(self, max_concurrent: int): + self.max_concurrent = max_concurrent + super().__init__( + f"Already {max_concurrent} model load(s) in flight, which is the configured " + f"cap (set via {MAX_CONCURRENT_LOADS_ENV}). Retry once the in-flight request " + f"completes, or raise {MAX_CONCURRENT_LOADS_ENV} if your hardware can hold " + f"more models loaded at once." + ) + + +def _positive_int_env(name: str, default: int) -> int: + raw = os.environ.get(name) + if raw is None: + return default + try: + value = int(raw) + except ValueError: + return default + return value if value > 0 else default + + +def max_model_params() -> int: + """The configured parameter-count cap (NEURONSCOPE_MAX_MODEL_PARAMS, or the default).""" + return _positive_int_env(MAX_MODEL_PARAMS_ENV, DEFAULT_MAX_MODEL_PARAMS) + + +def max_concurrent_loads() -> int: + """The configured concurrency cap (NEURONSCOPE_MAX_CONCURRENT_LOADS, or the default).""" + return _positive_int_env(MAX_CONCURRENT_LOADS_ENV, DEFAULT_MAX_CONCURRENT_LOADS) + + +def check_model_size(model_name: str) -> None: + """Raise ModelTooLargeError if ``model_name``'s parameter count exceeds the cap. + + Uses TransformerLens's own config lookup (``get_num_params_of_pretrained``), which + reads only the model's config, not its weights, so an oversized request is rejected + before any multi-gigabyte download happens. If the parameter count can't be + determined at all (offline with nothing cached yet, or a model type TransformerLens + doesn't report ``n_params`` for), this fails open and lets the load proceed rather + than blocking a legitimate request on a best-effort check -- the concurrency cap + below still bounds how many such loads can be in flight at once. + """ + from transformer_lens.loading_from_pretrained import get_num_params_of_pretrained + + try: + n_params = get_num_params_of_pretrained(model_name) + except Exception: + return + limit = max_model_params() + if n_params > limit: + raise ModelTooLargeError(model_name, n_params, limit) + + +class _LoadSlotLimiter: + """Thread-safe counting limiter. + + ``try_acquire`` never blocks or queues: it returns False immediately once the limit + is reached, so a caller over the cap gets an immediate, clear error instead of a + hang -- callers of ``model_load_slot`` turn that into ``TooManyConcurrentModelLoadsError``. + """ + + def __init__(self) -> None: + self._lock = threading.Lock() + self._active = 0 + + def try_acquire(self, limit: int) -> bool: + with self._lock: + if self._active >= limit: + return False + self._active += 1 + return True + + def release(self) -> None: + with self._lock: + self._active = max(0, self._active - 1) + + +# Process-wide: the whole point is to bound concurrent loads across every call into this +# process, whichever thread (MCP tool dispatch, CLI invocation, or a test) makes it. +_limiter = _LoadSlotLimiter() + + +class model_load_slot: + """Context manager guarding one model load against the concurrency cap. + + Raises ``TooManyConcurrentModelLoadsError`` on ``__enter__`` instead of blocking if + ``max_concurrent_loads()`` in-flight loads are already active. + """ + + def __enter__(self) -> "model_load_slot": + limit = max_concurrent_loads() + if not _limiter.try_acquire(limit): + raise TooManyConcurrentModelLoadsError(limit) + return self + + def __exit__(self, *exc_info: object) -> None: + _limiter.release() diff --git a/neuronscope/core/trace.py b/neuronscope/core/trace.py index 8fc36f4..7b5c684 100644 --- a/neuronscope/core/trace.py +++ b/neuronscope/core/trace.py @@ -17,6 +17,7 @@ from neuronscope.backends.base import Backend from neuronscope.backends.transformer_lens import TransformerLensBackend +from neuronscope.core.limits import check_model_size, model_load_slot from neuronscope.core.registry import resolve_backend from neuronscope.schema import ( ActivationsResponse, @@ -67,11 +68,18 @@ def __init__(self, model_name: str, layer: int, n_layers: int): def _load_and_validate(backend: Backend, model_name: str, prompt: str): """Load the model and make sure the prompt fits before any heavy backend work runs.""" - # TransformerLens prints a "Loaded pretrained model ..." line straight to stdout on - # every load, with no way to disable it via from_pretrained's own arguments. Silence - # it here so it can never end up mixed into --json output or an MCP tool result. - with contextlib.redirect_stdout(io.StringIO()): - model = backend.load_model(model_name) + # Two resource caps, both bypassable only by raising the env vars that configure + # them (see core/limits.py): reject an oversized model before paying for the + # download, and reject a load that would exceed the configured concurrency cap + # instead of piling more resident models into memory. + check_model_size(model_name) + with model_load_slot(): + # TransformerLens prints a "Loaded pretrained model ..." line straight to stdout + # on every load, with no way to disable it via from_pretrained's own arguments. + # Silence it here so it can never end up mixed into --json output or an MCP tool + # result. + with contextlib.redirect_stdout(io.StringIO()): + model = backend.load_model(model_name) # model.to_tokens() silently truncates to n_ctx by default (truncate=True), which # would make this check never fire. Count untruncated tokens explicitly so an # over-length prompt is caught here instead of quietly analyzing a truncated prompt. diff --git a/neuronscope/mcp_server.py b/neuronscope/mcp_server.py index 66c5285..985135b 100644 --- a/neuronscope/mcp_server.py +++ b/neuronscope/mcp_server.py @@ -15,6 +15,7 @@ from pydantic import Field from neuronscope.backends.transformer_lens import COMPONENT_HOOK_TEMPLATES +from neuronscope.core.limits import ModelTooLargeError, TooManyConcurrentModelLoadsError from neuronscope.core.registry import UnsupportedModelError from neuronscope.core.trace import ( DEFAULT_TOP_K, @@ -46,6 +47,10 @@ def _error_dict(operation: str, exc: Exception) -> dict[str, Any]: error_type = "PromptTooLongError" elif isinstance(exc, LayerOutOfRangeError): error_type = "LayerOutOfRangeError" + elif isinstance(exc, ModelTooLargeError): + error_type = "ModelTooLargeError" + elif isinstance(exc, TooManyConcurrentModelLoadsError): + error_type = "TooManyConcurrentModelLoadsError" else: error_type = type(exc).__name__ return ErrorResponse(operation=operation, error_type=error_type, message=str(exc)).model_dump() diff --git a/tests/test_limits.py b/tests/test_limits.py new file mode 100644 index 0000000..8698db0 --- /dev/null +++ b/tests/test_limits.py @@ -0,0 +1,237 @@ +"""Tests for the resource caps on model loading: parameter-count ceiling and the +concurrency ceiling on in-flight loads (neuronscope/core/limits.py). + +These caps close a self-disclosed gap: NeuronScope loads a fresh model on every call +with nothing caching a loaded model across calls, so an uncapped caller could either +name an arbitrarily large checkpoint or fire off several concurrent tool calls and +exhaust the host's memory either way. Both caps must fail fast with a clear error +rather than blocking or crashing -- that's what these tests assert. +""" + +from __future__ import annotations + +import threading + +import pytest +import torch + +from neuronscope.core.limits import ( + DEFAULT_MAX_CONCURRENT_LOADS, + DEFAULT_MAX_MODEL_PARAMS, + ModelTooLargeError, + TooManyConcurrentModelLoadsError, + check_model_size, + max_concurrent_loads, + max_model_params, + model_load_slot, +) +from neuronscope.core.trace import _load_and_validate + + +# --------------------------------------------------------------------------- +# Config parsing +# --------------------------------------------------------------------------- + + +def test_max_model_params_default(monkeypatch): + monkeypatch.delenv("NEURONSCOPE_MAX_MODEL_PARAMS", raising=False) + assert max_model_params() == DEFAULT_MAX_MODEL_PARAMS + + +def test_max_model_params_respects_env(monkeypatch): + monkeypatch.setenv("NEURONSCOPE_MAX_MODEL_PARAMS", "123456") + assert max_model_params() == 123456 + + +def test_max_model_params_falls_back_on_invalid_env(monkeypatch): + monkeypatch.setenv("NEURONSCOPE_MAX_MODEL_PARAMS", "not-a-number") + assert max_model_params() == DEFAULT_MAX_MODEL_PARAMS + + +def test_max_model_params_falls_back_on_nonpositive_env(monkeypatch): + monkeypatch.setenv("NEURONSCOPE_MAX_MODEL_PARAMS", "0") + assert max_model_params() == DEFAULT_MAX_MODEL_PARAMS + + +def test_max_concurrent_loads_default(monkeypatch): + monkeypatch.delenv("NEURONSCOPE_MAX_CONCURRENT_LOADS", raising=False) + assert max_concurrent_loads() == DEFAULT_MAX_CONCURRENT_LOADS + + +def test_max_concurrent_loads_respects_env(monkeypatch): + monkeypatch.setenv("NEURONSCOPE_MAX_CONCURRENT_LOADS", "4") + assert max_concurrent_loads() == 4 + + +# --------------------------------------------------------------------------- +# Parameter-count cap +# --------------------------------------------------------------------------- + + +def test_check_model_size_raises_when_over_cap(monkeypatch): + import transformer_lens.loading_from_pretrained as tlp + + monkeypatch.setattr(tlp, "get_num_params_of_pretrained", lambda name: 5_000_000_000) + monkeypatch.setenv("NEURONSCOPE_MAX_MODEL_PARAMS", "2000000000") + + with pytest.raises(ModelTooLargeError) as exc_info: + check_model_size("huge-model") + + error = exc_info.value + assert error.model_name == "huge-model" + assert error.n_params == 5_000_000_000 + assert error.max_params == 2_000_000_000 + assert "huge-model" in str(error) + assert "NEURONSCOPE_MAX_MODEL_PARAMS" in str(error) + + +def test_check_model_size_allows_when_under_cap(monkeypatch): + import transformer_lens.loading_from_pretrained as tlp + + monkeypatch.setattr(tlp, "get_num_params_of_pretrained", lambda name: 124_000_000) + monkeypatch.setenv("NEURONSCOPE_MAX_MODEL_PARAMS", "2000000000") + + check_model_size("small-model") # must not raise + + +def test_check_model_size_fails_open_when_lookup_errors(monkeypatch): + import transformer_lens.loading_from_pretrained as tlp + + def _raise(model_name: str) -> int: + raise ValueError("unknown model, can't resolve a config") + + monkeypatch.setattr(tlp, "get_num_params_of_pretrained", _raise) + + # Can't determine the size (e.g. an unresolvable name) -> best-effort check fails + # open rather than blocking the caller; UnsupportedModelError is what actually + # rejects a bad model name, raised later in resolve_backend. + check_model_size("not-a-real-model-xyz") + + +# --------------------------------------------------------------------------- +# Concurrency cap +# --------------------------------------------------------------------------- + + +def test_model_load_slot_allows_a_single_load(monkeypatch): + monkeypatch.setenv("NEURONSCOPE_MAX_CONCURRENT_LOADS", "1") + with model_load_slot(): + pass # must not raise + + +def test_model_load_slot_rejects_when_cap_is_already_saturated(monkeypatch): + monkeypatch.setenv("NEURONSCOPE_MAX_CONCURRENT_LOADS", "1") + + with model_load_slot(): + with pytest.raises(TooManyConcurrentModelLoadsError) as exc_info: + with model_load_slot(): + pass # never reached + assert exc_info.value.max_concurrent == 1 + assert "NEURONSCOPE_MAX_CONCURRENT_LOADS" in str(exc_info.value) + + +def test_model_load_slot_frees_its_slot_on_exit(monkeypatch): + monkeypatch.setenv("NEURONSCOPE_MAX_CONCURRENT_LOADS", "1") + + with model_load_slot(): + pass + # The first slot was released on exit, so a second, sequential load must succeed. + with model_load_slot(): + pass + + +def test_model_load_slot_respects_a_higher_configured_cap(monkeypatch): + monkeypatch.setenv("NEURONSCOPE_MAX_CONCURRENT_LOADS", "2") + + with model_load_slot(): + with model_load_slot(): + pass # two concurrent loads are fine under a cap of 2 + + +def test_model_load_slot_rejects_true_concurrent_contention(monkeypatch): + # Same as the nested-context test above, but across real threads, to prove the + # limiter is actually thread-safe rather than just correct in a single-threaded + # nested-call shape. + monkeypatch.setenv("NEURONSCOPE_MAX_CONCURRENT_LOADS", "1") + + first_slot_held = threading.Event() + release_first_slot = threading.Event() + results: list[bool] = [] + + def hold_first_slot(): + with model_load_slot(): + first_slot_held.set() + release_first_slot.wait(timeout=5) + + thread = threading.Thread(target=hold_first_slot) + thread.start() + assert first_slot_held.wait(timeout=5) + + try: + with pytest.raises(TooManyConcurrentModelLoadsError): + with model_load_slot(): + results.append(True) # never reached + finally: + release_first_slot.set() + thread.join(timeout=5) + + assert results == [] + # The slot is free again once the first thread's context exits. + with model_load_slot(): + pass + + +# --------------------------------------------------------------------------- +# Wired into the actual model-loading path (core/trace.py's _load_and_validate) +# --------------------------------------------------------------------------- + + +class _StubCfg: + n_ctx = 1024 + model_name = "stub-model" + + +class _StubModel: + cfg = _StubCfg() + + def to_tokens(self, prompt: str, truncate: bool = True) -> torch.Tensor: + return torch.zeros((1, 3)) + + +class _StubBackend: + name = "stub" + + def load_model(self, model_name: str, device: str | None = None) -> _StubModel: + return _StubModel() + + +def test_load_and_validate_enforces_the_concurrency_cap(monkeypatch): + monkeypatch.setenv("NEURONSCOPE_MAX_CONCURRENT_LOADS", "1") + backend = _StubBackend() + + with model_load_slot(): + with pytest.raises(TooManyConcurrentModelLoadsError): + _load_and_validate(backend, "not-a-real-model-xyz", "hello") + + +def test_load_and_validate_enforces_the_size_cap(monkeypatch): + import transformer_lens.loading_from_pretrained as tlp + + monkeypatch.setattr(tlp, "get_num_params_of_pretrained", lambda name: 5_000_000_000) + monkeypatch.setenv("NEURONSCOPE_MAX_MODEL_PARAMS", "2000000000") + backend = _StubBackend() + + with pytest.raises(ModelTooLargeError): + _load_and_validate(backend, "huge-model", "hello") + + +def test_load_and_validate_succeeds_when_under_both_caps(monkeypatch): + monkeypatch.setenv("NEURONSCOPE_MAX_CONCURRENT_LOADS", "1") + monkeypatch.setenv("NEURONSCOPE_MAX_MODEL_PARAMS", "2000000000") + import transformer_lens.loading_from_pretrained as tlp + + monkeypatch.setattr(tlp, "get_num_params_of_pretrained", lambda name: 124_000_000) + backend = _StubBackend() + + model = _load_and_validate(backend, "stub-model", "hello") + assert model.cfg.model_name == "stub-model"