Skip to content

Commit 85fbcba

Browse files
oschwaldclaude
andcommitted
Bound decoder string and bytes payload to stop amplification
A crafted database could aim many data-section pointers at one large string or bytes value. The value count stayed low, but the pure Python decoder copied each target, so a small file could materialize gigabytes. Add a call-local 2 MiB budget for the total string and bytes payload a single decode produces. Each value is charged its length wherever it is decoded, so re-decoding a shared target through another pointer recharges the budget, which stops the amplification. Also reject a variable-length integer whose declared size exceeds its type before the bytes are copied. The metadata read when a database is opened uses the same decoder, so the limit covers it too. Bump the test-data submodule to the fixtures for these cases. See GHSA-hj94-g986-h9r7. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
1 parent d456df0 commit 85fbcba

3 files changed

Lines changed: 181 additions & 24 deletions

File tree

‎HISTORY.rst‎

Lines changed: 13 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -6,12 +6,19 @@ History
66
3.2.0
77
+++++
88

9-
* Fixed a denial-of-service issue in the pure Python decoder. A crafted database
10-
could nest data-section pointers to shared targets so that decoding one record
11-
could cost exponential time and memory from a small file. The decoder now
12-
limits the number of values it decodes for a single record and rejects a
13-
database that exceeds it, along with pointer cycles and over-deep data, with an
14-
``InvalidDatabaseError``. See GHSA-hj94-g986-h9r7.
9+
* Fixed two denial-of-service issues in the pure Python decoder. A crafted
10+
database could nest data-section pointers to shared targets so that decoding
11+
one record cost exponential time and memory from a small file, or point many
12+
times at one large string or bytes value so that a record with few values
13+
materialized far more data than the file holds. The decoder now applies the
14+
limits recommended by the MaxMind DB specification to each record it decodes
15+
and to the metadata read when a database is opened. A database that exceeds
16+
a limit raises ``InvalidDatabaseError``. The limits are:
17+
18+
* 65,536 decoded values.
19+
* 2 MiB of string and bytes payload. An oversized variable-length integer is
20+
also rejected.
21+
* 512 levels of nesting, which also stops pointer cycles.
1522

1623
3.1.1 (2026-03-05)
1724
++++++++++++++++++

‎maxminddb/decoder.py‎

Lines changed: 42 additions & 17 deletions
Original file line numberDiff line numberDiff line change
@@ -32,10 +32,25 @@
3232
# call-local depth limit (see ``decode``).
3333
_MAX_VALUES = 1 << 16
3434
_MAX_DEPTH = 512
35+
# Per-lookup limit on the total string and bytes payload materialized, matching
36+
# libmaxminddb and the Go reader. It stops a payload amplification, where many
37+
# pointers to one large value would otherwise materialize N * size bytes from a
38+
# small file. Each string or bytes value is charged its length wherever it is
39+
# decoded, so re-decoding a shared target through another pointer recharges.
40+
_MAX_PAYLOAD_BYTES = 1 << 21
41+
# The widest fixed-width integer the format defines is the 16-byte uint128; a
42+
# declared size past that is malformed and could copy attacker-controlled bytes.
43+
_MAX_UINT_BYTES = 16
44+
_MAX_INT32_BYTES = 4
3545
_TOO_MANY_VALUES = (
3646
"The MaxMind DB file's data section exceeds the maximum number of values"
3747
)
3848
_TOO_DEEP = "The MaxMind DB file's data section exceeds the maximum depth"
49+
_TOO_LARGE = "The MaxMind DB file's data section exceeds the maximum payload size"
50+
_BAD_DATA = (
51+
"The MaxMind DB file's data section contains bad data "
52+
"(unknown data type or corrupt data)"
53+
)
3954

4055

4156
class Decoder:
@@ -90,8 +105,13 @@ def _decode_bytes(
90105
self,
91106
size: int,
92107
offset: int,
93-
_budget: list[int],
108+
budget: list[int],
94109
) -> tuple[bytes, int]:
110+
# Charge the payload before copying so a crafted size cannot force a
111+
# large allocation, and so pointers reusing one target recharge.
112+
budget[2] -= size
113+
if budget[2] < 0:
114+
raise InvalidDatabaseError(_TOO_LARGE)
95115
new_offset = offset + size
96116
return self._buffer[offset:new_offset], new_offset
97117

@@ -125,6 +145,8 @@ def _decode_int32(
125145
offset: int,
126146
_budget: list[int],
127147
) -> tuple[int, int]:
148+
if size > _MAX_INT32_BYTES:
149+
raise InvalidDatabaseError(_BAD_DATA)
128150
if size == 0:
129151
return 0, offset
130152
new_offset = offset + size
@@ -197,6 +219,10 @@ def _decode_uint(
197219
offset: int,
198220
_budget: list[int],
199221
) -> tuple[int, int]:
222+
# Reject a declared size past the widest defined unsigned integer before
223+
# copying, so a crafted size cannot force a large allocation.
224+
if size > _MAX_UINT_BYTES:
225+
raise InvalidDatabaseError(_BAD_DATA)
200226
new_offset = offset + size
201227
uint_bytes = self._buffer[offset:new_offset]
202228
return int.from_bytes(uint_bytes, "big"), new_offset
@@ -205,8 +231,13 @@ def _decode_utf8_string(
205231
self,
206232
size: int,
207233
offset: int,
208-
_budget: list[int],
234+
budget: list[int],
209235
) -> tuple[str, int]:
236+
# Charge the payload before copying so a crafted size cannot force a
237+
# large allocation, and so pointers reusing one target recharge.
238+
budget[2] -= size
239+
if budget[2] < 0:
240+
raise InvalidDatabaseError(_TOO_LARGE)
210241
new_offset = offset + size
211242
return self._buffer[offset:new_offset].decode("utf-8"), new_offset
212243

@@ -234,15 +265,15 @@ def decode(self, offset: int) -> tuple[Record, int]:
234265
235266
"""
236267
# Bound the work per lookup so a crafted database cannot exhaust CPU or
237-
# memory. ``budget`` carries the remaining value count and current
238-
# nested decode depth so both are shared across the recursion. It is
239-
# call-local, which keeps the decoder safe for concurrent reads. The
240-
# root value is charged here; containers charge their children. The
241-
# explicit depth limit is independent of Python's process-wide recursion
242-
# limit; RecursionError remains a fallback on interpreters whose stack
243-
# limit is reached first.
268+
# memory. ``budget`` carries the remaining value count, the current
269+
# nested decode depth, and the remaining string and bytes payload, so
270+
# all three are shared across the recursion. It is call-local, which
271+
# keeps the decoder safe for concurrent reads. The root value is charged
272+
# here; containers charge their children. The explicit depth limit
273+
# is independent of Python's process-wide recursion limit; RecursionError
274+
# remains a fallback on interpreters whose stack limit is reached first.
244275
try:
245-
return self._decode(offset, [_MAX_VALUES - 1, 0])
276+
return self._decode(offset, [_MAX_VALUES - 1, 0, _MAX_PAYLOAD_BYTES])
246277
except RecursionError as ex:
247278
raise InvalidDatabaseError(_TOO_DEEP) from ex
248279

@@ -281,13 +312,7 @@ def _read_extended(self, offset: int) -> tuple[int, int]:
281312
@staticmethod
282313
def _verify_size(expected: int, actual: int) -> None:
283314
if expected != actual:
284-
msg = (
285-
"The MaxMind DB file's data section contains bad data "
286-
"(unknown data type or corrupt data)"
287-
)
288-
raise InvalidDatabaseError(
289-
msg,
290-
)
315+
raise InvalidDatabaseError(_BAD_DATA)
291316

292317
def _size_from_ctrl_byte(
293318
self,

‎tests/decoder_test.py‎

Lines changed: 126 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -3,8 +3,9 @@
33
import mmap
44
import sys
55
import unittest
6-
from typing import TYPE_CHECKING, Any, ClassVar
6+
from typing import TYPE_CHECKING, Any, ClassVar, cast
77

8+
from maxminddb import MODE_MEMORY, open_database
89
from maxminddb.decoder import Decoder
910
from maxminddb.errors import InvalidDatabaseError
1011

@@ -15,6 +16,12 @@
1516
# cases reach the decoder's explicit limit with ample test-harness headroom.
1617
_DEPTH_TEST_RECURSION_LIMIT = 2_000
1718

19+
# Directory holding the shared MaxMind DB test fixtures.
20+
_TEST_DATA_DIR = "tests/data/test-data"
21+
_PAYLOAD_TOO_LARGE = (
22+
"^The MaxMind DB file's data section exceeds the maximum payload size$"
23+
)
24+
1825

1926
class TestDecoder(unittest.TestCase):
2027
def test_arrays(self) -> None:
@@ -324,3 +331,121 @@ def test_oversized_map_is_bounded(self) -> None:
324331
oversized_map = bytes([0xFE, 0x7E, 0xE4])
325332
with self.assertRaises(InvalidDatabaseError):
326333
Decoder(oversized_map, pointer_base=0).decode(0)
334+
335+
def test_oversized_string_payload_is_bounded(self) -> None:
336+
# A single string that declares one byte more than the 2 MiB payload
337+
# limit is rejected before its bytes are copied. This also covers the
338+
# wrapped-scalar variant: the charge is applied wherever a string is
339+
# decoded, not only for a direct pointer target. 0x5f: string with size
340+
# code 31; 0x1efee4: 2,097,153 - 65,821, one byte over 2 MiB.
341+
oversized_string = bytes([0x5F, 0x1E, 0xFE, 0xE4])
342+
with self.assertRaisesRegex(InvalidDatabaseError, _PAYLOAD_TOO_LARGE):
343+
Decoder(oversized_string, pointer_base=0).decode(0)
344+
345+
def test_oversized_bytes_payload_is_bounded(self) -> None:
346+
# As above for the bytes type. 0x9f: bytes with size code 31.
347+
oversized_bytes = bytes([0x9F, 0x1E, 0xFE, 0xE4])
348+
with self.assertRaisesRegex(InvalidDatabaseError, _PAYLOAD_TOO_LARGE):
349+
Decoder(oversized_bytes, pointer_base=0).decode(0)
350+
351+
def test_oversized_uint_is_bounded(self) -> None:
352+
# A uint128 that declares 17 bytes exceeds the 16-byte format maximum
353+
# and is rejected before the declared bytes are copied. 0x11: extended
354+
# type, size 17; 0x03: extended type number 10 (uint128).
355+
oversized_uint = bytes([0x11, 0x03])
356+
with self.assertRaises(InvalidDatabaseError):
357+
Decoder(oversized_uint, pointer_base=0).decode(0)
358+
359+
def test_oversized_int32_is_bounded(self) -> None:
360+
# An int32 that declares 5 bytes exceeds its 4-byte maximum and is
361+
# rejected before the declared bytes are copied. 0x05: extended type,
362+
# size 5; 0x01: extended type number 8 (int32).
363+
oversized_int32 = bytes([0x05, 0x01])
364+
with self.assertRaises(InvalidDatabaseError):
365+
Decoder(oversized_int32, pointer_base=0).decode(0)
366+
367+
368+
class TestDecoderResourceLimits(unittest.TestCase):
369+
"""Fixture-backed checks for the pure-Python decoder resource limits."""
370+
371+
@staticmethod
372+
def _lookup(filename: str, ip: str = "0.0.0.1") -> object:
373+
# MODE_MEMORY forces the pure-Python decoder. Each DoS fixture resolves
374+
# any IPv4 address to its single crafted record.
375+
with open_database(f"{_TEST_DATA_DIR}/{filename}", mode=MODE_MEMORY) as reader:
376+
return reader.get(ip)
377+
378+
def test_payload_amplification_is_rejected(self) -> None:
379+
# An array of 8,192 pointers to one 65,535-byte value. The value count
380+
# stays low, but copying each target would materialize about 512 MiB.
381+
with self.assertRaisesRegex(InvalidDatabaseError, _PAYLOAD_TOO_LARGE):
382+
self._lookup("MaxMind-DB-test-payload-amplification-dos.mmdb")
383+
384+
def test_payload_amplification_string_is_rejected(self) -> None:
385+
# The UTF-8 string variant, so the decode path for strings is exercised.
386+
with self.assertRaisesRegex(InvalidDatabaseError, _PAYLOAD_TOO_LARGE):
387+
self._lookup("MaxMind-DB-test-payload-amplification-dos-string.mmdb")
388+
389+
def test_payload_amplification_worst_case_is_rejected(self) -> None:
390+
# 65,535 pointers to one 65,535-byte value. The record is exactly
391+
# 65,536 values under the flat rule, so only the payload budget can
392+
# reject it.
393+
with self.assertRaisesRegex(InvalidDatabaseError, _PAYLOAD_TOO_LARGE):
394+
self._lookup("MaxMind-DB-test-payload-amplification-dos-worst-case.mmdb")
395+
396+
def test_value_count_boundary(self) -> None:
397+
# The at-limit fixture decodes to exactly 65,536 values and must decode.
398+
# The pointer-heavy fixture reaches 65,535 values through pointers,
399+
# which cost nothing beyond the values they resolve to. One value more
400+
# than the limit is rejected.
401+
self.assertIsInstance(
402+
self._lookup("MaxMind-DB-test-decoder-value-limit.mmdb"),
403+
list,
404+
)
405+
self.assertIsInstance(
406+
self._lookup("MaxMind-DB-test-decoder-value-limit-pointer-heavy.mmdb"),
407+
list,
408+
)
409+
with self.assertRaisesRegex(
410+
InvalidDatabaseError,
411+
"^The MaxMind DB file's data section exceeds the maximum number of values$",
412+
):
413+
self._lookup("MaxMind-DB-test-decoder-value-limit-over.mmdb")
414+
415+
def test_pointer_fan_out_fixture_is_rejected(self) -> None:
416+
# A full database whose record nests arrays of pointers to the level
417+
# below, the classic 2**depth fan-out.
418+
with self.assertRaises(InvalidDatabaseError):
419+
self._lookup("MaxMind-DB-test-pointer-decoder-dos.mmdb")
420+
421+
def test_payload_at_limit_is_accepted(self) -> None:
422+
# References totaling exactly 2 MiB of payload decode successfully, so
423+
# the limit does not reject a record at the boundary.
424+
self.assertIsInstance(
425+
self._lookup("MaxMind-DB-test-decoder-payload-limit.mmdb"),
426+
list,
427+
)
428+
429+
def test_payload_one_over_limit_is_rejected(self) -> None:
430+
# One byte more than 2 MiB is rejected, catching an off-by-one.
431+
with self.assertRaisesRegex(InvalidDatabaseError, _PAYLOAD_TOO_LARGE):
432+
self._lookup("MaxMind-DB-test-decoder-payload-limit-over.mmdb")
433+
434+
def test_metadata_payload_limit_is_enforced_on_open(self) -> None:
435+
# The same decoder reads metadata on open, so an over-limit metadata
436+
# structure is rejected there too.
437+
with self.assertRaisesRegex(InvalidDatabaseError, _PAYLOAD_TOO_LARGE):
438+
open_database(
439+
f"{_TEST_DATA_DIR}/MaxMind-DB-test-metadata-payload-limit.mmdb",
440+
mode=MODE_MEMORY,
441+
)
442+
443+
def test_normal_record_still_decodes(self) -> None:
444+
# A record with ordinary string and bytes values, which the payload
445+
# budget also charges, decodes unchanged.
446+
record = cast(
447+
"dict",
448+
self._lookup("MaxMind-DB-test-decoder.mmdb", "::1.1.1.0"),
449+
)
450+
self.assertEqual(record["utf8_string"], "unicode! ☯ - ♫")
451+
self.assertEqual(record["bytes"], b"\x00\x00\x00*")

0 commit comments

Comments
 (0)