Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
16 changes: 16 additions & 0 deletions HISTORY.rst
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,9 @@ History
C extension. Before, the iterator walked the new database with node
numbers from the old one. A failed ``__init__`` keeps the old database.
After ``close()``, an iterator raises ``ValueError`` in every mode.
* For a record map with a key that cannot be hashed, such as a list, the
pure Python reader raises ``InvalidDatabaseError`` instead of ``TypeError``,
as the C extension does.
* C extension:

* Fixed segmentation faults from invalid use of ``Metadata``, ``Reader`` and
Expand All @@ -29,6 +32,19 @@ History
during iteration, from another thread or from a signal handler.
* Fixed a crash on free-threaded Python when two threads advanced the same
iterator.
* Added the ``node_byte_size`` and ``search_tree_size`` properties to
``Metadata``, as the pure Python ``Metadata`` has.

* Metadata:

* The pure Python reader ignores unknown keys, which a new minor version of
the format can add. It raises ``InvalidDatabaseError`` for a missing key, a
value that is not of the expected Python type or is out of range, an
invalid ``ip_version`` or format version, a ``build_epoch`` of 0, or a
string that is not UTF-8. Previously, the reader opened most of these
files, and some raised ``TypeError`` or ``UnicodeDecodeError``.
* The C extension ignores unknown keys. Previously, ``Reader.metadata()``
crashed on them.

3.2.0 (2026-09-10)
++++++++++++++++++
Expand Down
131 changes: 113 additions & 18 deletions extension/maxminddb.c
Original file line number Diff line number Diff line change
Expand Up @@ -135,6 +135,7 @@ static inline maxminddb_state *get_maxminddb_state_from_self(PyObject *self) {
static void reader_close_database(Reader_obj *reader);
static bool can_read(const char *path);
static int get_record(PyObject *self, PyObject *args, PyObject **record);
static PyObject *metadata_value(PyObject *map, const char *key);
static PyObject *reader_iter_next(PyObject *self);
static bool format_sockaddr(struct sockaddr *addr, char *dst);
static PyObject *from_entry_data_list(maxminddb_state *state,
Expand Down Expand Up @@ -680,43 +681,97 @@ static PyObject *Reader_metadata(PyObject *self, PyObject *UNUSED(args)) {
return NULL;
}

MMDB_entry_data_list_s *entry_data_list;
int status =
// libmaxminddb checked the metadata when it opened the database, so take
// the numbers from its copy. Its strings end at the first NUL, so take
// the strings from the decoded metadata map, which keeps their lengths.
// For a repeated key, the strings come from the last entry, but
// libmaxminddb uses the first. This reader accepts the difference. Keys
// that Metadata does not know are ignored.
const MMDB_metadata_s *m = &mmdb_obj->mmdb->metadata;
uint16_t const binary_format_major_version = m->binary_format_major_version;
uint16_t const binary_format_minor_version = m->binary_format_minor_version;
uint64_t const build_epoch = m->build_epoch;
uint16_t const ip_version = m->ip_version;
uint32_t const node_count = m->node_count;
uint16_t const record_size = m->record_size;
MMDB_entry_data_list_s *entry_data_list = NULL;
int const status =
MMDB_get_metadata_as_entry_data_list(mmdb_obj->mmdb, &entry_data_list);
if (status != MMDB_SUCCESS) {
reader_release_read_lock(mmdb_obj);
MMDB_free_entry_data_list(entry_data_list);
PyErr_Format(state->MaxMindDB_error,
"Error decoding metadata. %s",
MMDB_strerror(status));
return NULL;
}
MMDB_entry_data_list_s *original_entry_data_list = entry_data_list;

PyObject *metadata_dict = from_entry_data_list(state, &entry_data_list);
MMDB_free_entry_data_list(original_entry_data_list);
if (metadata_dict == NULL || !PyDict_Check(metadata_dict)) {
reader_release_read_lock(mmdb_obj);
PyObject *map = NULL;
if (entry_data_list != NULL &&
entry_data_list->entry_data.type == MMDB_DATA_TYPE_MAP) {
map = from_map(state, &entry_data_list);
} else {
PyErr_SetString(state->MaxMindDB_error, "Error decoding metadata.");
Py_XDECREF(metadata_dict);
return NULL;
}
MMDB_free_entry_data_list(original_entry_data_list);

// Creating a Metadata can run Python code, such as an __init__, and no
// Python code may run under the read lock. The values above are copies.
reader_release_read_lock(mmdb_obj);

PyObject *args = PyTuple_New(0);
if (args == NULL) {
Py_DECREF(metadata_dict);
return NULL;
PyObject *metadata = NULL;
if (map != NULL) {
PyObject *description = NULL;
PyObject *languages = NULL;
PyObject *database_type = metadata_value(map, "database_type");
if (database_type != NULL) {
description = metadata_value(map, "description");
}
if (description != NULL) {
languages = metadata_value(map, "languages");
}
if (languages == NULL) {
// MMDB_open requires these keys, so a missing key is a bug.
if (!PyErr_Occurred()) {
PyErr_SetString(state->MaxMindDB_error,
"Error decoding metadata.");
}
} else {
// The order of the values must match kwlist in Metadata_new.
metadata = PyObject_CallFunction(state->Metadata_Type,
"HHKOOHOIH",
binary_format_major_version,
binary_format_minor_version,
(unsigned long long)build_epoch,
database_type,
description,
ip_version,
languages,
(unsigned int)node_count,
record_size);
}
Py_DECREF(map);
}

PyObject *metadata =
PyObject_Call(state->Metadata_Type, args, metadata_dict);

Py_DECREF(metadata_dict);
Py_DECREF(args);
// libmaxminddb does not check that the metadata strings are UTF-8.
if (metadata == NULL && PyErr_ExceptionMatches(PyExc_UnicodeDecodeError)) {
PyErr_SetString(state->MaxMindDB_error, "Error decoding metadata.");
}
return metadata;
}

// Return a borrowed reference to the value of key in map. NULL with no
// exception set means that the key is missing.
static PyObject *metadata_value(PyObject *map, const char *key) {
PyObject *name = PyUnicode_FromString(key);
if (name == NULL) {
return NULL;
}
PyObject *value = PyDict_GetItemWithError(map, name);
Py_DECREF(name);
return value;
}

static PyObject *Reader_close(PyObject *self, PyObject *UNUSED(args)) {
Reader_obj *mmdb_obj = (Reader_obj *)self;

Expand Down Expand Up @@ -1040,6 +1095,7 @@ Metadata_new(PyTypeObject *type, PyObject *args, PyObject *kwds) {
*build_epoch, *database_type, *description, *ip_version, *languages,
*node_count, *record_size;

// Reader_metadata passes the values in this order.
static char *kwlist[] = {"binary_format_major_version",
"binary_format_minor_version",
"build_epoch",
Expand Down Expand Up @@ -1350,6 +1406,44 @@ static PyMemberDef Metadata_members[] = {
NULL},
{NULL, 0, 0, 0, NULL}};

static PyObject *Metadata_node_byte_size(PyObject *self,
void *UNUSED(closure)) {
Metadata_obj *obj = (Metadata_obj *)self;
PyObject *four = PyLong_FromLong(4);
if (four == NULL) {
return NULL;
}
PyObject *node_byte_size = PyNumber_FloorDivide(obj->record_size, four);
Py_DECREF(four);
return node_byte_size;
}

static PyObject *Metadata_search_tree_size(PyObject *self,
void *UNUSED(closure)) {
Metadata_obj *obj = (Metadata_obj *)self;
PyObject *node_byte_size = Metadata_node_byte_size(self, NULL);
if (node_byte_size == NULL) {
return NULL;
}
PyObject *search_tree_size =
PyNumber_Multiply(obj->node_count, node_byte_size);
Py_DECREF(node_byte_size);
return search_tree_size;
}

// These match the properties of the pure Python Metadata class.
static PyGetSetDef Metadata_getset[] = {{"node_byte_size",
Metadata_node_byte_size,
NULL,
"The size of a node in bytes.",
NULL},
{"search_tree_size",
Metadata_search_tree_size,
NULL,
"The size of the search tree.",
NULL},
{NULL, NULL, NULL, NULL, NULL}};

// =============================================================================
// Type specs for heap type conversion (PEP 489)
// =============================================================================
Expand Down Expand Up @@ -1378,6 +1472,7 @@ static PyType_Slot Metadata_Type_slots[] = {
{Py_tp_new, Metadata_new},
{Py_tp_methods, Metadata_methods},
{Py_tp_members, Metadata_members},
{Py_tp_getset, Metadata_getset},
{0, NULL},
};

Expand Down
5 changes: 3 additions & 2 deletions maxminddb/decoder.py
Original file line number Diff line number Diff line change
Expand Up @@ -255,8 +255,9 @@ def decode(self, offset: int) -> tuple[Record, int]:
)
except RecursionError as ex:
raise InvalidDatabaseError(_TOO_DEEP) from ex
except (IndexError, struct.error) as ex:
# Convert failed buffer indexing and fixed-width unpacking.
except (IndexError, struct.error, TypeError) as ex:
# Convert failed buffer indexing, fixed-width unpacking, and a map
# key that cannot be hashed, such as a list.
raise InvalidDatabaseError(_BAD_DATA) from ex

# Keep type dispatch inline to avoid another call for every decoded value.
Expand Down
8 changes: 8 additions & 0 deletions maxminddb/extension.pyi
Original file line number Diff line number Diff line change
Expand Up @@ -127,3 +127,11 @@ class Metadata:
record_size: int,
) -> None:
"""Create new Metadata object from the metadata fields in the spec."""

@property
def node_byte_size(self) -> int:
"""The size of a node in bytes."""

@property
def search_tree_size(self) -> int:
"""The size of the search tree."""
98 changes: 87 additions & 11 deletions maxminddb/reader.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,7 @@

from typing_extensions import Self

from maxminddb.types import Record
from maxminddb.types import Record, RecordDict

_IPV4_MAX_NUM = 2**32
_REOPENED = "Attempt to iterate over a reopened MaxMind DB. Create a new iterator."
Expand Down Expand Up @@ -113,24 +113,25 @@ def _load(

metadata_start += len(self._METADATA_START_MARKER)
metadata_decoder = Decoder(self._buffer, metadata_start)
(metadata, _) = metadata_decoder.decode(metadata_start)
# For a repeated key, the decoder keeps the last value, but
# libmaxminddb uses the first. This reader accepts the difference.
try:
(metadata, _) = metadata_decoder.decode(metadata_start)
except (InvalidDatabaseError, UnicodeDecodeError) as e:
# Add the file name. For a string that is not UTF-8, the C
# extension raises InvalidDatabaseError too. Lookups keep
# UnicodeDecodeError.
msg = f"Error reading metadata in database file ({filename}). {e}"
raise InvalidDatabaseError(msg) from e

if not isinstance(metadata, dict):
msg = f"Error reading metadata in database file ({filename})."
raise InvalidDatabaseError( # noqa: TRY301
msg,
)

# The MaxMind DB spec fixes these keys and their value types.
fields: dict[str, Any] = metadata
self._metadata = Metadata(**fields)
self._metadata = Metadata(**_metadata_fields(metadata, filename))
self._record_size = self._metadata.record_size
if self._record_size not in (24, 28, 32):
msg = f"Unknown record size: {self._record_size}"
raise InvalidDatabaseError(msg) # noqa: TRY301
if self._metadata.node_count < 0:
msg = f"Invalid node count: {self._metadata.node_count}"
raise InvalidDatabaseError(msg) # noqa: TRY301

# Traversal reads nodes below node_count. Once the tree fits, those
# reads need no length checks of their own.
Expand Down Expand Up @@ -376,6 +377,81 @@ def __enter__(self) -> Self:
return self


# The type of each metadata value. libmaxminddb also rejects a database with a
# missing key or a value of another type. It also checks the width and sign of
# each integer, which the decoder does not report.
_METADATA_TYPES: dict[str, type] = {
"binary_format_major_version": int,
"binary_format_minor_version": int,
"build_epoch": int,
"database_type": str,
"description": dict,
"ip_version": int,
"languages": list,
"node_count": int,
"record_size": int,
}


# The size in bits of each unsigned integer metadata value in libmaxminddb that
# needs a range check. The other integers must have exact values.
_METADATA_UINT_BITS: dict[str, int] = {
"binary_format_minor_version": 16,
"build_epoch": 64,
"node_count": 32,
}


def _metadata_fields(metadata: RecordDict, filename: object) -> dict[str, Any]:
"""Return the known metadata fields after a check of their types.

A new minor version of the format can add keys. This ignores them.
"""
prefix = f"Error reading metadata in database file ({filename})."
fields: dict[str, Any] = {}
for key, value_type in _METADATA_TYPES.items():
value = metadata.get(key)
# The exact type check rejects bool, a subclass of int.
valid = type(value) is value_type
if valid and isinstance(value, list):
valid = all(type(v) is str for v in value)
elif valid and isinstance(value, dict):
valid = all(type(k) is str and type(v) is str for k, v in value.items())
if not valid:
msg = f"{prefix} The {key} value is missing or has the wrong type."
raise InvalidDatabaseError(msg)
fields[key] = value

_check_metadata_ranges(fields, prefix)
return fields


def _check_metadata_ranges(fields: dict[str, Any], prefix: str) -> None:
"""Raise InvalidDatabaseError for a value that libmaxminddb rejects."""
# libmaxminddb stores each integer as an unsigned value. Check the size of
# those in _METADATA_UINT_BITS. The reader decodes only the version 2 format,
# ip_version drives the tree walk, and record_size picks the node layout.
# These exact values need no range check. libmaxminddb also rejects
# node_count 0, but this reader accepts an empty search tree.
if fields["record_size"] not in (24, 28, 32):
msg = f"{prefix} Unknown record size: {fields['record_size']}."
raise InvalidDatabaseError(msg)
for key, bits in _METADATA_UINT_BITS.items():
if not 0 <= fields[key] < 1 << bits:
msg = f"{prefix} The {key} value {fields[key]} is out of range."
raise InvalidDatabaseError(msg)
if fields["binary_format_major_version"] != 2:
version = fields["binary_format_major_version"]
msg = f"{prefix} Unsupported binary format version {version}."
raise InvalidDatabaseError(msg)
if fields["ip_version"] not in (4, 6):
msg = f"{prefix} The ip_version is {fields['ip_version']}, not 4 or 6."
raise InvalidDatabaseError(msg)
if fields["build_epoch"] == 0:
msg = f"{prefix} The build_epoch is 0."
raise InvalidDatabaseError(msg)


@dataclass(kw_only=True, frozen=True)
class Metadata:
"""Metadata for the MaxMind DB reader."""
Expand Down
6 changes: 6 additions & 0 deletions tests/decoder_test.py
Original file line number Diff line number Diff line change
Expand Up @@ -316,6 +316,12 @@ def test_value_limit_follows_the_flat_rule(self) -> None:
with self.assertRaisesRegex(InvalidDatabaseError, _TOO_MANY_VALUES):
Decoder(self._scalar_pointer_array(65_536), pointer_base=0).decode(1)

def test_map_key_that_cannot_be_hashed_is_rejected(self) -> None:
# A map with one entry, whose key is the array [1].
decoder = Decoder(bytes.fromhex("e10104a1014178"))
with self.assertRaisesRegex(InvalidDatabaseError, "contains bad data"):
decoder.decode(0)

def test_pointer_to_pointer_is_rejected(self) -> None:
# The root array shares a pointer chain that would bypass value counting.
buf = b"\xa0" + self._pointer(0) + b"\x02\x04" + self._pointer(1) * 2
Expand Down
Loading
Loading