From 0923ccfbeaffb064d1bc40cad3780b2cea237853 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 22:30:28 +0000 Subject: [PATCH 01/38] Import InvalidDatabaseError from maxminddb.errors maxminddb.decoder does not export InvalidDatabaseError. It only imports it. mypy --strict reports the import as an implicit re-export. Co-Authored-By: Claude Opus 5.5 --- maxminddb/__init__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/maxminddb/__init__.py b/maxminddb/__init__.py index b3ef0ef6..576dc1d0 100644 --- a/maxminddb/__init__.py +++ b/maxminddb/__init__.py @@ -13,7 +13,7 @@ MODE_MMAP, MODE_MMAP_EXT, ) -from .decoder import InvalidDatabaseError +from .errors import InvalidDatabaseError from .reader import Reader if TYPE_CHECKING: From 6eb3d0b6a7cad8bd194aa74368c765b83fb5c574 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 22:31:47 +0000 Subject: [PATCH 02/38] Fix segmentation faults in the C extension Metadata_dealloc called Py_DECREF on each field. A Metadata object that init did not fill has NULL fields, so freeing it crashed. That happened after Metadata.__new__ and after a failed init. Reader.metadata() passes every metadata key to init, so a database with an unknown key crashed the process when code called metadata(). The spec allows new keys in a minor version of the format. Use Py_XDECREF. Metadata_init made every argument optional but increfed all nine locals. A missing argument increfed an uninitialized pointer. Make the arguments required. Reader_iter and ReaderIter_next checked closed == Py_True. A Reader that init did not open has closed == NULL and mmdb == NULL, so iteration dereferenced NULL. Check mmdb, as get() and metadata() do. The ReaderIter type allowed direct instantiation, and its dealloc then decrefed a NULL reader. Disallow instantiation. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 8 ++++++ extension/maxminddb.c | 56 +++++++++++++++++----------------------- maxminddb/extension.pyi | 17 +++++++++--- tests/reader_test.py | 57 +++++++++++++++++++++++++++++++++++++++++ 4 files changed, 102 insertions(+), 36 deletions(-) diff --git a/HISTORY.rst b/HISTORY.rst index 50839c9a..ddd53d2f 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -3,6 +3,14 @@ History ------- +3.3.0 +++++++++++++++++++ + +* C extension: + + * Fixed segmentation faults from invalid use of ``Metadata``, ``Reader`` and + the internal iterator type. + 3.2.0 (2026-09-10) ++++++++++++++++++ diff --git a/extension/maxminddb.c b/extension/maxminddb.c index c2dd73e7..f5219bff 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -686,7 +686,7 @@ static PyObject *Reader__enter__(PyObject *self, PyObject *UNUSED(args)) { return NULL; } - if (mmdb_obj->closed == Py_True) { + if (mmdb_obj->mmdb == NULL) { reader_release_read_lock(mmdb_obj); PyErr_SetString(PyExc_ValueError, "Attempt to reopen a closed MaxMind DB."); @@ -727,7 +727,7 @@ static PyObject *Reader_iter(PyObject *obj) { return NULL; } - if (reader->closed == Py_True) { + if (reader->mmdb == NULL) { reader_release_read_lock(reader); PyErr_SetString(PyExc_ValueError, "Attempt to iterate over a closed MaxMind DB."); @@ -777,7 +777,7 @@ static PyObject *ReaderIter_next(PyObject *self) { return NULL; } - if (ri->reader->closed == Py_True) { + if (ri->reader->mmdb == NULL) { reader_release_read_lock(ri->reader); PyErr_SetString(PyExc_ValueError, "Attempt to iterate over a closed MaxMind DB."); @@ -972,7 +972,7 @@ static int Metadata_init(PyObject *self, PyObject *args, PyObject *kwds) { if (!PyArg_ParseTupleAndKeywords(args, kwds, - "|OOOOOOOOO", + "OOOOOOOOO", kwlist, &binary_format_major_version, &binary_format_minor_version, @@ -988,40 +988,30 @@ static int Metadata_init(PyObject *self, PyObject *args, PyObject *kwds) { Metadata_obj *obj = (Metadata_obj *)self; - obj->binary_format_major_version = binary_format_major_version; - obj->binary_format_minor_version = binary_format_minor_version; - obj->build_epoch = build_epoch; - obj->database_type = database_type; - obj->description = description; - obj->ip_version = ip_version; - obj->languages = languages; - obj->node_count = node_count; - obj->record_size = record_size; - - Py_INCREF(obj->binary_format_major_version); - Py_INCREF(obj->binary_format_minor_version); - Py_INCREF(obj->build_epoch); - Py_INCREF(obj->database_type); - Py_INCREF(obj->description); - Py_INCREF(obj->ip_version); - Py_INCREF(obj->languages); - Py_INCREF(obj->node_count); - Py_INCREF(obj->record_size); + obj->binary_format_major_version = Py_NewRef(binary_format_major_version); + obj->binary_format_minor_version = Py_NewRef(binary_format_minor_version); + obj->build_epoch = Py_NewRef(build_epoch); + obj->database_type = Py_NewRef(database_type); + obj->description = Py_NewRef(description); + obj->ip_version = Py_NewRef(ip_version); + obj->languages = Py_NewRef(languages); + obj->node_count = Py_NewRef(node_count); + obj->record_size = Py_NewRef(record_size); return 0; } static void Metadata_dealloc(PyObject *self) { Metadata_obj *obj = (Metadata_obj *)self; - Py_DECREF(obj->binary_format_major_version); - Py_DECREF(obj->binary_format_minor_version); - Py_DECREF(obj->build_epoch); - Py_DECREF(obj->database_type); - Py_DECREF(obj->description); - Py_DECREF(obj->ip_version); - Py_DECREF(obj->languages); - Py_DECREF(obj->node_count); - Py_DECREF(obj->record_size); + Py_XDECREF(obj->binary_format_major_version); + Py_XDECREF(obj->binary_format_minor_version); + Py_XDECREF(obj->build_epoch); + Py_XDECREF(obj->database_type); + Py_XDECREF(obj->description); + Py_XDECREF(obj->ip_version); + Py_XDECREF(obj->languages); + Py_XDECREF(obj->node_count); + Py_XDECREF(obj->record_size); PyObject_Del(self); } @@ -1305,7 +1295,7 @@ static PyType_Slot ReaderIter_Type_slots[] = { static PyType_Spec ReaderIter_Type_spec = { .name = "maxminddb.extension.ReaderIter", .basicsize = sizeof(ReaderIter_obj), - .flags = Py_TPFLAGS_DEFAULT, + .flags = Py_TPFLAGS_DEFAULT | Py_TPFLAGS_DISALLOW_INSTANTIATION, .slots = ReaderIter_Type_slots, }; diff --git a/maxminddb/extension.pyi b/maxminddb/extension.pyi index 1a5e88b0..a4d4d707 100644 --- a/maxminddb/extension.pyi +++ b/maxminddb/extension.pyi @@ -3,7 +3,7 @@ from collections.abc import Iterator from ipaddress import IPv4Address, IPv4Network, IPv6Address, IPv6Network from os import PathLike -from typing import IO, Any +from typing import IO from typing_extensions import Self @@ -113,5 +113,16 @@ class Metadata: The bit size of a record in the search tree. """ - def __init__(self, **kwargs: Any) -> None: # noqa: ANN401 - """Create new Metadata object. kwargs are key/value pairs from spec.""" + def __init__( + self, + binary_format_major_version: int, + binary_format_minor_version: int, + build_epoch: int, + database_type: str, + description: dict[str, str], + ip_version: int, + languages: list[str], + node_count: int, + record_size: int, + ) -> None: + """Create new Metadata object from the metadata fields in the spec.""" diff --git a/tests/reader_test.py b/tests/reader_test.py index ce95714e..e616f851 100644 --- a/tests/reader_test.py +++ b/tests/reader_test.py @@ -942,6 +942,63 @@ class TestExtensionReaderWithIPObjects(BaseTestReader): reader_class = maxminddb.extension.Reader +@unittest.skipIf( + not has_maxminddb_extension() and not os.environ.get("MM_FORCE_EXT_TESTS"), + "No C extension module found. Skipping tests", +) +class TestExtensionObjects(unittest.TestCase): + """Objects in states that crashed the extension.""" + + def test_uninitialized_metadata(self) -> None: + metadata_class = maxminddb.extension.Metadata + metadata = metadata_class.__new__(metadata_class) + self.assertIsNone(metadata.languages) + del metadata + + def test_metadata_missing_argument(self) -> None: + with self.assertRaisesRegex(TypeError, "missing required argument"): + maxminddb.extension.Metadata(binary_format_major_version=2) # type: ignore[call-arg] + + def test_metadata_unknown_argument(self) -> None: + with self.assertRaisesRegex(TypeError, "keyword argument"): + maxminddb.extension.Metadata( # type: ignore[call-arg] + binary_format_major_version=2, + binary_format_minor_version=0, + build_epoch=1, + database_type="db", + description={}, + ip_version=4, + languages=[], + node_count=1, + record_size=24, + unknown=1, + ) + + def test_iterate_uninitialized_reader(self) -> None: + reader_class = maxminddb.extension.Reader + reader = reader_class.__new__(reader_class) + with self.assertRaisesRegex(ValueError, "closed MaxMind DB"): + iter(reader) + + def test_enter_uninitialized_reader(self) -> None: + reader_class = maxminddb.extension.Reader + reader = reader_class.__new__(reader_class) + with self.assertRaisesRegex(ValueError, "closed MaxMind DB"): + reader.__enter__() + + def test_iterator_type_is_not_instantiable(self) -> None: + reader = maxminddb.extension.Reader( + f"{_TEST_DATA_DIR}/MaxMind-DB-test-decoder.mmdb", + ) + iterator_class = type(iter(reader)) + # The message differs across Python versions, so check only the type. + with self.assertRaises(TypeError): + iterator_class() + with self.assertRaises(TypeError): + iterator_class.__new__(iterator_class) + reader.close() + + class TestAutoReader(BaseTestReader): mode = MODE_AUTO From 621a218aaf35115347c7e9bb7fb2f91c5e539c5d Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 22:54:11 +0000 Subject: [PATCH 03/38] Fix memory leaks and a use-after-free in the C extension The Reader, Metadata and iterator types are heap types, so each instance holds a reference to its type. The dealloc functions did not release that reference, so each object leaked one reference to its type, and the types were never freed. Reader_init did not check whether the reader was already initialized. A second __init__ leaked the open database and reinitialized a lock that other threads could hold. On a reader closed after an iterator was made, it also freed the database under that iterator, so the next step of the iterator read freed memory. Refuse any reinitialization with a ValueError, as __enter__ does for a closed reader. Metadata_init also accepted a second init, which leaked the old field values. Refuse it with the same ValueError. Co-Authored-By: Claude Opus 4.8 Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 2 ++ extension/maxminddb.c | 56 +++++++++++++++++++++++++++++++++++-------- tests/reader_test.py | 51 ++++++++++++++++++++++++++++++++++++++- 3 files changed, 98 insertions(+), 11 deletions(-) diff --git a/HISTORY.rst b/HISTORY.rst index ddd53d2f..2a1358df 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -10,6 +10,8 @@ History * Fixed segmentation faults from invalid use of ``Metadata``, ``Reader`` and the internal iterator type. + * Fixed memory leaks and a use-after-free. Reinitializing a ``Reader`` or a + ``Metadata`` now raises ``ValueError``. 3.2.0 (2026-09-10) ++++++++++++++++++ diff --git a/extension/maxminddb.c b/extension/maxminddb.c index f5219bff..974561ac 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -314,6 +314,17 @@ static int Reader_init(PyObject *self, PyObject *args, PyObject *kwds) { return -1; } + // Refuse a second init. The closed field is NULL only before the first + // init, so this covers an open reader and a closed one. A second init + // would leak the open database and reinitialize the lock. On a closed + // reader it would also leave an existing iterator pointing at freed + // memory. + if (((Reader_obj *)self)->closed != NULL) { + PyErr_SetString(PyExc_ValueError, + "Attempt to reinitialize a MaxMind DB reader."); + return -1; + } + PyObject *filepath = NULL; int mode = 0; @@ -712,7 +723,10 @@ static void Reader_dealloc(PyObject *self) { reader_lock_destroy(&obj->rwlock); + // An instance of a heap type holds a reference to its type. + PyTypeObject *type = Py_TYPE(self); PyObject_Del(self); + Py_DECREF(type); } static PyObject *Reader_iter(PyObject *obj) { @@ -950,7 +964,9 @@ static void ReaderIter_dealloc(PyObject *self) { next = cur->next; free(cur); } + PyTypeObject *type = Py_TYPE(self); PyObject_Del(self); + Py_DECREF(type); } static int Metadata_init(PyObject *self, PyObject *args, PyObject *kwds) { @@ -988,17 +1004,35 @@ static int Metadata_init(PyObject *self, PyObject *args, PyObject *kwds) { Metadata_obj *obj = (Metadata_obj *)self; - obj->binary_format_major_version = Py_NewRef(binary_format_major_version); - obj->binary_format_minor_version = Py_NewRef(binary_format_minor_version); - obj->build_epoch = Py_NewRef(build_epoch); - obj->database_type = Py_NewRef(database_type); - obj->description = Py_NewRef(description); - obj->ip_version = Py_NewRef(ip_version); - obj->languages = Py_NewRef(languages); - obj->node_count = Py_NewRef(node_count); - obj->record_size = Py_NewRef(record_size); + // Refuse a second init, as Reader_init does. Replacing a field would leak + // the old value or free it while a getter uses it. On free-threaded + // builds, the critical section makes the check and the stores atomic. + int status = 0; +#ifdef Py_GIL_DISABLED + Py_BEGIN_CRITICAL_SECTION(self); +#endif + if (obj->binary_format_major_version != NULL) { + PyErr_SetString(PyExc_ValueError, + "Attempt to reinitialize a MaxMind DB Metadata."); + status = -1; + } else { + obj->binary_format_major_version = + Py_NewRef(binary_format_major_version); + obj->binary_format_minor_version = + Py_NewRef(binary_format_minor_version); + obj->build_epoch = Py_NewRef(build_epoch); + obj->database_type = Py_NewRef(database_type); + obj->description = Py_NewRef(description); + obj->ip_version = Py_NewRef(ip_version); + obj->languages = Py_NewRef(languages); + obj->node_count = Py_NewRef(node_count); + obj->record_size = Py_NewRef(record_size); + } +#ifdef Py_GIL_DISABLED + Py_END_CRITICAL_SECTION(); +#endif - return 0; + return status; } static void Metadata_dealloc(PyObject *self) { @@ -1012,7 +1046,9 @@ static void Metadata_dealloc(PyObject *self) { Py_XDECREF(obj->languages); Py_XDECREF(obj->node_count); Py_XDECREF(obj->record_size); + PyTypeObject *type = Py_TYPE(self); PyObject_Del(self); + Py_DECREF(type); } static PyObject * diff --git a/tests/reader_test.py b/tests/reader_test.py index e616f851..0504e2e1 100644 --- a/tests/reader_test.py +++ b/tests/reader_test.py @@ -7,10 +7,11 @@ import os import pathlib import sys +import sysconfig import tempfile import threading import unittest -from typing import TYPE_CHECKING, cast +from typing import TYPE_CHECKING, Any, cast from unittest import mock import maxminddb @@ -986,6 +987,54 @@ def test_enter_uninitialized_reader(self) -> None: with self.assertRaisesRegex(ValueError, "closed MaxMind DB"): reader.__enter__() + def test_reinitialize_reader_is_refused(self) -> None: + path = f"{_TEST_DATA_DIR}/MaxMind-DB-test-decoder.mmdb" + with maxminddb.extension.Reader(path) as reader: + with self.assertRaisesRegex(ValueError, "reinitialize"): + reader.__init__(path) # type: ignore[misc] + self.assertIsNotNone(reader.get("::1.1.1.0")) + + # Re-init on a closed reader would leave this iterator pointing at a + # freed database. + closed = maxminddb.extension.Reader(path) + iterator = iter(closed) + next(iterator) + closed.close() + with self.assertRaisesRegex(ValueError, "reinitialize"): + closed.__init__(path) # type: ignore[misc] + + def test_reinitialize_metadata_is_refused(self) -> None: + fields: dict[str, Any] = { + "binary_format_major_version": 2, + "binary_format_minor_version": 0, + "build_epoch": 1, + "database_type": "db", + "description": {}, + "ip_version": 4, + "languages": [], + "node_count": 1, + "record_size": 24, + } + metadata = maxminddb.extension.Metadata(**fields) + with self.assertRaisesRegex(ValueError, "reinitialize"): + metadata.__init__(**{**fields, "record_size": 28}) # type: ignore[misc] + self.assertEqual(metadata.record_size, 24) + + @unittest.skipUnless( + hasattr(sys, "getrefcount") and not sysconfig.get_config_var("Py_GIL_DISABLED"), + "needs CPython reference counts on a build with the GIL", + ) + def test_freed_objects_release_their_type(self) -> None: + path = f"{_TEST_DATA_DIR}/MaxMind-DB-test-decoder.mmdb" + with maxminddb.extension.Reader(path) as reader: + classes = [type(reader), type(reader.metadata()), type(iter(reader))] + before = [sys.getrefcount(c) for c in classes] + for _ in range(10): + with maxminddb.extension.Reader(path) as reader: + reader.metadata() + iter(reader) + self.assertEqual([sys.getrefcount(c) for c in classes], before) + def test_iterator_type_is_not_instantiable(self) -> None: reader = maxminddb.extension.Reader( f"{_TEST_DATA_DIR}/MaxMind-DB-test-decoder.mmdb", From 86964ec14cd86ec8e969ca042da4966431b2e104 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 23:23:55 +0000 Subject: [PATCH 04/38] Initialize the reader lock in tp_new On free-threaded builds, the read-write lock did not follow the object lifetime. Reader_init created the lock, destroyed it again when the open failed, and Reader_dealloc destroyed it once more, so a failed open destroyed the lock twice. A reader made with __new__ alone, with no init, used and then destroyed a lock that was never created. glibc treats an all-zero pthread_rwlock_t as a valid unlocked lock and ignores a second destroy, so this was harmless on Linux. Other platforms, such as macOS, reject both. Create the lock once in tp_new and destroy it once in Reader_dealloc. Reader_init no longer creates or destroys the lock. Every allocated reader, including one from a bare __new__ or a failed init, then has exactly one valid lock for its whole lifetime. Reader_init now holds the write lock while it checks for a second init and stores the opened database, so two threads that call __init__ on one reader cannot both open a database. If the lock fails to initialize, tp_new frees the object directly. Reader_dealloc would destroy the failed lock. Co-Authored-By: Claude Opus 4.8 Co-Authored-By: Claude Opus 5.5 --- extension/maxminddb.c | 60 ++++++++++++++++++++++++++++--------------- 1 file changed, 39 insertions(+), 21 deletions(-) diff --git a/extension/maxminddb.c b/extension/maxminddb.c index 974561ac..d2c1f33b 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -308,23 +308,34 @@ static void reader_release_write_lock(Reader_obj *reader) { // Reader implementation // ============================================================================= +static PyObject * +Reader_new(PyTypeObject *type, PyObject *UNUSED(args), PyObject *UNUSED(kwds)) { + PyObject *self = type->tp_alloc(type, 0); + if (self == NULL) { + return NULL; + } + + // Initialize the lock once, for the whole lifetime of the object. + // Reader_dealloc destroys it. A bare __new__ or a failed Reader_init then + // still leaves a valid, unlocked lock, so no path uses or destroys an + // uninitialized lock. + if (reader_lock_init(&((Reader_obj *)self)->rwlock) != 0) { + // Skip Reader_dealloc, which would destroy the failed lock. An + // instance of a heap type holds a reference to its type. + PyObject_Del(self); + Py_DECREF(type); + return NULL; + } + + return self; +} + static int Reader_init(PyObject *self, PyObject *args, PyObject *kwds) { maxminddb_state *state = get_maxminddb_state_from_self(self); if (state == NULL) { return -1; } - // Refuse a second init. The closed field is NULL only before the first - // init, so this covers an open reader and a closed one. A second init - // would leak the open database and reinitialize the lock. On a closed - // reader it would also leave an existing iterator pointing at freed - // memory. - if (((Reader_obj *)self)->closed != NULL) { - PyErr_SetString(PyExc_ValueError, - "Attempt to reinitialize a MaxMind DB reader."); - return -1; - } - PyObject *filepath = NULL; int mode = 0; @@ -360,31 +371,36 @@ static int Reader_init(PyObject *self, PyObject *args, PyObject *kwds) { return -1; } - MMDB_s *mmdb = (MMDB_s *)malloc(sizeof(MMDB_s)); - if (mmdb == NULL) { + Reader_obj *mmdb_obj = (Reader_obj *)self; + if (reader_acquire_write_lock(mmdb_obj) != 0) { Py_XDECREF(filepath); - PyErr_NoMemory(); return -1; } - Reader_obj *mmdb_obj = (Reader_obj *)self; - if (!mmdb_obj) { + // Refuse a second init. closed is NULL until the first init or close(), + // and the write lock stops two threads from both passing this check. A + // second init would leak the open database. After close(), it would leave + // an existing iterator pointing at freed memory. + if (mmdb_obj->closed != NULL) { + reader_release_write_lock(mmdb_obj); Py_XDECREF(filepath); - free(mmdb); - PyErr_NoMemory(); + PyErr_SetString(PyExc_ValueError, + "Attempt to reinitialize a MaxMind DB reader."); return -1; } - if (reader_lock_init(&mmdb_obj->rwlock) != 0) { - free(mmdb); + MMDB_s *mmdb = (MMDB_s *)malloc(sizeof(MMDB_s)); + if (mmdb == NULL) { + reader_release_write_lock(mmdb_obj); Py_XDECREF(filepath); + PyErr_NoMemory(); return -1; } int const status = MMDB_open(filename, MMDB_MODE_MMAP, mmdb); if (status != MMDB_SUCCESS) { - reader_lock_destroy(&mmdb_obj->rwlock); + reader_release_write_lock(mmdb_obj); free(mmdb); PyErr_Format(state->MaxMindDB_error, "Error opening database file (%s). Is this a valid " @@ -398,6 +414,7 @@ static int Reader_init(PyObject *self, PyObject *args, PyObject *kwds) { mmdb_obj->mmdb = mmdb; mmdb_obj->closed = Py_False; + reader_release_write_lock(mmdb_obj); return 0; } @@ -1289,6 +1306,7 @@ static PyMemberDef Metadata_members[] = { static PyType_Slot Reader_Type_slots[] = { {Py_tp_doc, "Reader object"}, {Py_tp_dealloc, Reader_dealloc}, + {Py_tp_new, Reader_new}, {Py_tp_init, Reader_init}, {Py_tp_iter, Reader_iter}, {Py_tp_methods, Reader_methods}, From c9a12b47ae9b0455ac76ebb32f5bd361a175d49f Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Sat, 3 Oct 2026 05:12:39 +0000 Subject: [PATCH 05/38] Release the read lock before building the network ReaderIter_next held the read lock while it called ipaddress.ip_network, which runs Python code. That caused two hangs on free-threaded builds: - If the code closed the reader on the same thread, for example from a signal handler, close() waited for the write lock that the thread's own read lock blocked. - If the code ran the garbage collector while another thread waited in close() for the write lock, the collector stopped the world and waited for that thread, which waited for the read lock. The network uses only the iterator's own record, so release the lock after the record is decoded. Now no thread runs Python code while it holds the lock. A SIGALRM handler that closes the reader during iteration, and a wrapped ip_network that calls gc.collect() while another thread calls close(), both hung before this change and work after it. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 2 ++ extension/maxminddb.c | 22 ++++++++++------------ 2 files changed, 12 insertions(+), 12 deletions(-) diff --git a/HISTORY.rst b/HISTORY.rst index 2a1358df..4fedf749 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -12,6 +12,8 @@ History the internal iterator type. * Fixed memory leaks and a use-after-free. Reinitializing a ``Reader`` or a ``Metadata`` now raises ``ValueError``. + * Fixed a deadlock on free-threaded Python when a ``Reader`` was closed + during iteration, from another thread or from a signal handler. 3.2.0 (2026-09-10) ++++++++++++++++++ diff --git a/extension/maxminddb.c b/extension/maxminddb.c index d2c1f33b..d5482ca6 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -882,6 +882,9 @@ static PyObject *ReaderIter_next(PyObject *self) { case MMDB_RECORD_TYPE_EMPTY: break; case MMDB_RECORD_TYPE_DATA: { + // Read this before any Python code runs, which could close + // the reader. + uint16_t const depth = ri->reader->mmdb->depth; MMDB_entry_data_list_s *entry_data_list = NULL; int status = MMDB_get_entry_data_list(&cur->entry, &entry_data_list); @@ -901,15 +904,19 @@ static PyObject *ReaderIter_next(PyObject *self) { PyObject *record = from_entry_data_list(state, &entry_data_list); MMDB_free_entry_data_list(original_entry_data_list); + + // The rest uses only cur, which this call owns. Release the + // lock before ip_network runs Python code, which could close + // the reader on this thread. + reader_release_read_lock(ri->reader); if (record == NULL) { - reader_release_read_lock(ri->reader); free(cur); return NULL; } int ip_start = 0; int ip_length = 4; - if (ri->reader->mmdb->depth == 128) { + if (depth == 128) { if (is_ipv6(cur->ip_packed)) { // IPv6 address ip_length = 16; @@ -923,37 +930,28 @@ static PyObject *ReaderIter_next(PyObject *self) { &(cur->ip_packed[ip_start]), ip_length, cur->depth - ip_start * 8); + free(cur); if (network_tuple == NULL) { - reader_release_read_lock(ri->reader); Py_DECREF(record); - free(cur); return NULL; } PyObject *args = PyTuple_Pack(1, network_tuple); Py_DECREF(network_tuple); if (args == NULL) { - reader_release_read_lock(ri->reader); Py_DECREF(record); - free(cur); return NULL; } PyObject *network = PyObject_CallObject(state->ipaddress_ip_network, args); Py_DECREF(args); if (network == NULL) { - reader_release_read_lock(ri->reader); Py_DECREF(record); - free(cur); return NULL; } PyObject *rv = PyTuple_Pack(2, network, record); Py_DECREF(network); Py_DECREF(record); - - reader_release_read_lock(ri->reader); - - free(cur); return rv; } default: From 08c1718688d8b41194e65914a1e6e426b43c1b6c Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Sat, 3 Oct 2026 05:14:02 +0000 Subject: [PATCH 06/38] Let one thread at a time advance an iterator ReaderIter_next takes records off the iterator's pending list and adds their children while it holds only the shared read lock. On free-threaded builds, two threads that called next() on the same iterator could take the same record and both free it. The process aborted with heap corruption. Hold a critical section on the iterator for each next() call. With the GIL, the list changes already run without interruption. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 2 ++ extension/maxminddb.c | 16 ++++++++++++++++ tests/reader_test.py | 28 ++++++++++++++++++++++++++++ 3 files changed, 46 insertions(+) diff --git a/HISTORY.rst b/HISTORY.rst index 4fedf749..9a56de5b 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -14,6 +14,8 @@ History ``Metadata`` now raises ``ValueError``. * Fixed a deadlock on free-threaded Python when a ``Reader`` was closed during iteration, from another thread or from a signal handler. + * Fixed a crash on free-threaded Python when two threads advanced the same + iterator. 3.2.0 (2026-09-10) ++++++++++++++++++ diff --git a/extension/maxminddb.c b/extension/maxminddb.c index d5482ca6..136d876f 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -129,6 +129,7 @@ static inline maxminddb_state *get_maxminddb_state_from_self(PyObject *self) { static bool can_read(const char *path); static int get_record(PyObject *self, PyObject *args, PyObject **record); +static PyObject *reader_iter_next(PyObject *self); static bool format_sockaddr(struct sockaddr *addr, char *dst); static PyObject *from_entry_data_list(maxminddb_state *state, MMDB_entry_data_list_s **entry_data_list); @@ -797,6 +798,21 @@ static bool is_ipv6(char ip[16]) { } static PyObject *ReaderIter_next(PyObject *self) { +#ifdef Py_GIL_DISABLED + // The iterator's list of pending records is not thread-safe, so let only + // one thread at a time advance an iterator. The read lock is shared, so + // it does not do this. + PyObject *result; + Py_BEGIN_CRITICAL_SECTION(self); + result = reader_iter_next(self); + Py_END_CRITICAL_SECTION(); + return result; +#else + return reader_iter_next(self); +#endif +} + +static PyObject *reader_iter_next(PyObject *self) { maxminddb_state *state = get_maxminddb_state_from_self((PyObject *)self); if (state == NULL) { return NULL; diff --git a/tests/reader_test.py b/tests/reader_test.py index 0504e2e1..b2ff9d9f 100644 --- a/tests/reader_test.py +++ b/tests/reader_test.py @@ -1035,6 +1035,34 @@ def test_freed_objects_release_their_type(self) -> None: iter(reader) self.assertEqual([sys.getrefcount(c) for c in classes], before) + @unittest.skipIf( + getattr(sys, "_is_gil_enabled", lambda: True)(), + "needs free-threaded Python", + ) + def test_threads_can_share_an_iterator(self) -> None: + path = f"{_TEST_DATA_DIR}/GeoIP2-City-Test.mmdb" + with maxminddb.extension.Reader(path) as reader: + expected = sum(1 for _ in reader) + + def count( + iterator: Iterator[object], + barrier: threading.Barrier, + counts: list[int], + ) -> None: + barrier.wait() + counts.append(sum(1 for _ in iterator)) + + # A race corrupts the heap only some of the time, so repeat. + for _ in range(5): + counts: list[int] = [] + args = (iter(reader), threading.Barrier(8), counts) + threads = [threading.Thread(target=count, args=args) for _ in range(8)] + for thread in threads: + thread.start() + for thread in threads: + thread.join() + self.assertEqual(sum(counts), expected) + def test_iterator_type_is_not_instantiable(self) -> None: reader = maxminddb.extension.Reader( f"{_TEST_DATA_DIR}/MaxMind-DB-test-decoder.mmdb", From efdfae67dc40234cea300e799f1789ca0722b936 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Sat, 3 Oct 2026 05:35:14 +0000 Subject: [PATCH 07/38] Pass a Py_ssize_t length to Py_BuildValue The module defines PY_SSIZE_T_CLEAN, so the y# format reads the length as a Py_ssize_t. ReaderIter_next passed an int, which is undefined behavior on 64-bit platforms: Py_BuildValue reads 8 bytes from a 4-byte argument. Co-Authored-By: Claude Opus 5.5 --- extension/maxminddb.c | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/extension/maxminddb.c b/extension/maxminddb.c index 136d876f..bad1c809 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -931,7 +931,7 @@ static PyObject *reader_iter_next(PyObject *self) { } int ip_start = 0; - int ip_length = 4; + Py_ssize_t ip_length = 4; if (depth == 128) { if (is_ipv6(cur->ip_packed)) { // IPv6 address From 2c2ed421444b57f1a38ba6667d2513efbacac1d0 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Sat, 3 Oct 2026 06:02:35 +0000 Subject: [PATCH 08/38] Keep an exhausted iterator exhausted after close reader_iter_next checked for a closed reader before it checked for an empty list. After close(), an exhausted iterator raised ValueError instead of StopIteration, which breaks the iterator protocol. The pure Python iterator, a generator, stops as expected. Check for an empty list first. It needs no lock, because the list belongs to the iterator. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 2 ++ extension/maxminddb.c | 5 +++++ tests/reader_test.py | 10 ++++++++++ 3 files changed, 17 insertions(+) diff --git a/HISTORY.rst b/HISTORY.rst index 9a56de5b..68077aec 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -16,6 +16,8 @@ History during iteration, from another thread or from a signal handler. * Fixed a crash on free-threaded Python when two threads advanced the same iterator. + * An exhausted iterator now raises ``StopIteration`` after its ``Reader`` + closes, not ``ValueError``. 3.2.0 (2026-09-10) ++++++++++++++++++ diff --git a/extension/maxminddb.c b/extension/maxminddb.c index bad1c809..25c99926 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -820,6 +820,11 @@ static PyObject *reader_iter_next(PyObject *self) { ReaderIter_obj *ri = (ReaderIter_obj *)self; + // An exhausted iterator stays exhausted, even after the reader closes. + if (ri->next == NULL) { + return NULL; + } + if (reader_acquire_read_lock(ri->reader) != 0) { return NULL; } diff --git a/tests/reader_test.py b/tests/reader_test.py index b2ff9d9f..7f00f75d 100644 --- a/tests/reader_test.py +++ b/tests/reader_test.py @@ -600,6 +600,16 @@ def test_search_tree_past_end_of_file(self) -> None: ): reader.get(self.ipf("1.1.1.1")) + def test_exhausted_iterator_stops_after_close(self) -> None: + reader = open_database( + f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb", + self.mode, + ) + iterator = iter(reader) + list(iterator) + reader.close() + self.assertEqual(next(iterator, "done"), "done") + def test_ip_validation(self) -> None: reader = open_database( "tests/data/test-data/MaxMind-DB-test-decoder.mmdb", From 54592fe99589bc14485b5c1ddd510966b3415f09 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 23:21:47 +0000 Subject: [PATCH 09/38] Fix a reference leak when a map key fails to decode from_map built the dict, then returned NULL without releasing it when a key failed to decode, for example a map key that is not valid UTF-8. The sibling path for a failed value already released the dict. The cyclic garbage collector reclaims the dict later, so the leak does not grow without bound, but the refcount handling was still wrong. Co-Authored-By: Claude Opus 4.8 --- extension/maxminddb.c | 1 + 1 file changed, 1 insertion(+) diff --git a/extension/maxminddb.c b/extension/maxminddb.c index 25c99926..8536b30f 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -1158,6 +1158,7 @@ static PyObject *from_map(maxminddb_state *state, if (!key) { // PyUnicode_FromStringAndSize will set an appropriate exception // in this case. + Py_DECREF(py_obj); return NULL; } From e7284e76814479f91274ffb22167fbfae8211577 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Sat, 3 Oct 2026 03:24:41 +0000 Subject: [PATCH 10/38] Check the PyDict_SetItem result in from_map from_map ignored a failure from PyDict_SetItem, such as a MemoryError while the dict resizes. It then returned the dict with the exception still set, and the caller raised SystemError. Now it releases the dict and returns NULL, so the original exception reaches the caller. Co-Authored-By: Claude Opus 5.5 --- extension/maxminddb.c | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/extension/maxminddb.c b/extension/maxminddb.c index 8536b30f..81be6ffe 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -1170,9 +1170,13 @@ static PyObject *from_map(maxminddb_state *state, Py_DECREF(py_obj); return NULL; } - PyDict_SetItem(py_obj, key, value); + int const status = PyDict_SetItem(py_obj, key, value); Py_DECREF(value); Py_DECREF(key); + if (status < 0) { + Py_DECREF(py_obj); + return NULL; + } } return py_obj; From 89bc193f28f3cca9c617538e95b8410bf00c8c11 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Sat, 3 Oct 2026 06:45:10 +0000 Subject: [PATCH 11/38] Reject a map key that is not a string from_map read every map key from the utf8_string member of the entry union. libmaxminddb does not check the key type, so a database with a key of another type, such as a uint16, made from_map read an integer as a pointer. The process crashed with a segmentation fault. Raise InvalidDatabaseError for a key that is not a UTF-8 string. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 2 ++ extension/maxminddb.c | 8 ++++++++ tests/reader_test.py | 15 +++++++++++++++ 3 files changed, 25 insertions(+) diff --git a/HISTORY.rst b/HISTORY.rst index 68077aec..691d7153 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -10,6 +10,8 @@ History * Fixed segmentation faults from invalid use of ``Metadata``, ``Reader`` and the internal iterator type. + * Fixed a segmentation fault on a database with a map key that is not a + string. Such a database now raises ``InvalidDatabaseError``. * Fixed memory leaks and a use-after-free. Reinitializing a ``Reader`` or a ``Metadata`` now raises ``ValueError``. * Fixed a deadlock on free-threaded Python when a ``Reader`` was closed diff --git a/extension/maxminddb.c b/extension/maxminddb.c index 81be6ffe..d234f53d 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -1152,6 +1152,14 @@ static PyObject *from_map(maxminddb_state *state, for (i = 0; i < map_size && *entry_data_list; i++) { *entry_data_list = (*entry_data_list)->next; + // libmaxminddb does not check the key type, and the union holds a + // string only for a string key. + if ((*entry_data_list)->entry_data.type != MMDB_DATA_TYPE_UTF8_STRING) { + PyErr_SetString(state->MaxMindDB_error, + "Invalid map key: the key is not a string."); + Py_DECREF(py_obj); + return NULL; + } PyObject *key = PyUnicode_FromStringAndSize( (*entry_data_list)->entry_data.utf8_string, (*entry_data_list)->entry_data.data_size); diff --git a/tests/reader_test.py b/tests/reader_test.py index 7f00f75d..6c14c9bb 100644 --- a/tests/reader_test.py +++ b/tests/reader_test.py @@ -940,6 +940,21 @@ class TestExtensionReader(BaseTestReader): if has_maxminddb_extension(): reader_class = maxminddb.extension.Reader + def test_map_key_that_is_not_a_string_is_rejected(self) -> None: + data = bytearray( + pathlib.Path(f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb").read_bytes(), + ) + # Change the type of the "ip" key from a string to a uint16. + data[data.index(b"\x42ip")] = 0xA2 + with tempfile.TemporaryDirectory() as directory: + path = pathlib.Path(directory) / "int-key.mmdb" + path.write_bytes(data) + with ( + maxminddb.extension.Reader(path) as reader, + self.assertRaisesRegex(InvalidDatabaseError, "not a string"), + ): + reader.get("1.1.1.1") + @unittest.skipIf( not has_maxminddb_extension() and not os.environ.get("MM_FORCE_EXT_TESTS"), From 050fa0c2d933917da5a56ed5b33d5f0a7aa383a8 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Sat, 3 Oct 2026 06:45:27 +0000 Subject: [PATCH 12/38] Decode uint32 values with PyLong_FromUnsignedLong from_entry_data_list passed uint32 values to PyLong_FromLong. On platforms where a C long has 32 bits, such as Windows and 32-bit Linux, a value of 2**31 or more became negative, for example an ASN of 4200000000. The pure Python reader returns the correct value. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 1 + extension/maxminddb.c | 3 ++- 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/HISTORY.rst b/HISTORY.rst index 691d7153..c6793e96 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -12,6 +12,7 @@ History the internal iterator type. * Fixed a segmentation fault on a database with a map key that is not a string. Such a database now raises ``InvalidDatabaseError``. + * Fixed large ``uint32`` values, which came back negative on Windows. * Fixed memory leaks and a use-after-free. Reinitializing a ``Reader`` or a ``Metadata`` now raises ``ValueError``. * Fixed a deadlock on free-threaded Python when a ``Reader`` was closed diff --git a/extension/maxminddb.c b/extension/maxminddb.c index d234f53d..0a8adb06 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -1119,7 +1119,8 @@ from_entry_data_list(maxminddb_state *state, case MMDB_DATA_TYPE_UINT16: return PyLong_FromLong((*entry_data_list)->entry_data.uint16); case MMDB_DATA_TYPE_UINT32: - return PyLong_FromLong((*entry_data_list)->entry_data.uint32); + return PyLong_FromUnsignedLong( + (*entry_data_list)->entry_data.uint32); case MMDB_DATA_TYPE_BOOLEAN: return PyBool_FromLong((*entry_data_list)->entry_data.boolean); case MMDB_DATA_TYPE_UINT64: From 50ca8efe00932714d8e723afaf98737973acd93d Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 23:21:59 +0000 Subject: [PATCH 13/38] Fix a reference leak on a corrupt metadata type Reader.metadata() returned NULL without releasing the decoded object when it was not a dict. libmaxminddb validates the metadata when it opens the database, so an opened database reaches this path only through a bug, but the refcount handling was still wrong. Co-Authored-By: Claude Opus 4.8 --- extension/maxminddb.c | 1 + 1 file changed, 1 insertion(+) diff --git a/extension/maxminddb.c b/extension/maxminddb.c index 0a8adb06..a101627b 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -669,6 +669,7 @@ static PyObject *Reader_metadata(PyObject *self, PyObject *UNUSED(args)) { if (metadata_dict == NULL || !PyDict_Check(metadata_dict)) { reader_release_read_lock(mmdb_obj); PyErr_SetString(state->MaxMindDB_error, "Error decoding metadata."); + Py_XDECREF(metadata_dict); return NULL; } From b1c2646318fa89350817f8adfe0c8de984fca2d0 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Sat, 3 Oct 2026 04:10:37 +0000 Subject: [PATCH 14/38] Keep the error from decoding the metadata Reader.metadata() replaced any exception from from_entry_data_list with a generic "Error decoding metadata." That hid the cause, such as a MemoryError or the invalid data type that the decoder reported. Return the original exception, and keep the generic error for metadata that decodes to something other than a dict. Co-Authored-By: Claude Opus 5.5 --- extension/maxminddb.c | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/extension/maxminddb.c b/extension/maxminddb.c index a101627b..0ae61b87 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -666,10 +666,14 @@ static PyObject *Reader_metadata(PyObject *self, PyObject *UNUSED(args)) { PyObject *metadata_dict = from_entry_data_list(state, &entry_data_list); MMDB_free_entry_data_list(original_entry_data_list); - if (metadata_dict == NULL || !PyDict_Check(metadata_dict)) { + if (metadata_dict == NULL) { + reader_release_read_lock(mmdb_obj); + return NULL; + } + if (!PyDict_Check(metadata_dict)) { reader_release_read_lock(mmdb_obj); PyErr_SetString(state->MaxMindDB_error, "Error decoding metadata."); - Py_XDECREF(metadata_dict); + Py_DECREF(metadata_dict); return NULL; } From 383093a565c0b49afc41bd34bd723ba5c1789515 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Sat, 3 Oct 2026 04:30:08 +0000 Subject: [PATCH 15/38] Check the result of Reader_close Reader__exit__ and Reader_dealloc called Reader_close and ignored the result. Each successful close leaked a reference to None, which matters before Python 3.12, where None is not immortal. On free-threaded builds, a failure to take the write lock left an exception set: __exit__ hid it behind a successful return, and dealloc left it pending for unrelated code to find. __exit__ now returns the result of Reader_close. dealloc releases the result, or reports the error as unraisable, because dealloc cannot raise. Co-Authored-By: Claude Opus 5.5 --- extension/maxminddb.c | 13 ++++++++++--- 1 file changed, 10 insertions(+), 3 deletions(-) diff --git a/extension/maxminddb.c b/extension/maxminddb.c index 0ae61b87..fcfb8743 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -734,14 +734,21 @@ static PyObject *Reader__enter__(PyObject *self, PyObject *UNUSED(args)) { } static PyObject *Reader__exit__(PyObject *self, PyObject *UNUSED(args)) { - Reader_close(self, NULL); - Py_RETURN_NONE; + return Reader_close(self, NULL); } static void Reader_dealloc(PyObject *self) { Reader_obj *obj = (Reader_obj *)self; if (obj->mmdb != NULL) { - Reader_close(self, NULL); + PyObject *result = Reader_close(self, NULL); + if (result == NULL) { + // dealloc cannot raise, so report the error and continue. Pass + // NULL, because the hook would take a reference to self, whose + // count is already 0. + PyErr_WriteUnraisable(NULL); + } else { + Py_DECREF(result); + } } reader_lock_destroy(&obj->rwlock); From 810f52b009f4c87d0933b615fab7befd2012209c Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Sat, 3 Oct 2026 05:35:02 +0000 Subject: [PATCH 16/38] Reject a search tree deeper than the address size Neither reader limited the depth of the tree walk during iteration. A corrupt tree, such as one where a node points back to itself, made the C extension set bits past the end of the 16-byte ip_packed array, and then past its heap allocation. The process aborted with heap corruption. The pure Python reader recursed until it raised RecursionError, which a caller that catches InvalidDatabaseError does not catch. A node at the full address depth has no valid children, so both readers now raise InvalidDatabaseError when they reach one. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 5 +++++ extension/maxminddb.c | 11 +++++++++++ maxminddb/reader.py | 7 ++++++- tests/reader_test.py | 15 +++++++++++++++ 4 files changed, 37 insertions(+), 1 deletion(-) diff --git a/HISTORY.rst b/HISTORY.rst index c6793e96..9a78a56a 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -6,6 +6,11 @@ History 3.3.0 ++++++++++++++++++ +* Iterating over a database with a corrupt search tree, such as one with a + cycle, now raises ``InvalidDatabaseError``. Previously, the C extension + could corrupt memory, and the pure Python reader raised ``RecursionError`` + or returned part of the networks. + * C extension: * Fixed segmentation faults from invalid use of ``Metadata``, ``Reader`` and diff --git a/extension/maxminddb.c b/extension/maxminddb.c index fcfb8743..0450413f 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -866,6 +866,17 @@ static PyObject *reader_iter_next(PyObject *self) { // These are aliased networks. Skip them. break; } + // A node at the full address depth would write its children + // past the end of ip_packed. Only a corrupt tree, such as one + // with a cycle, has one. + if (cur->depth >= ri->reader->mmdb->depth) { + reader_release_read_lock(ri->reader); + PyErr_SetString(state->MaxMindDB_error, + "The MaxMind DB file's search tree is " + "corrupt"); + free(cur); + return NULL; + } MMDB_search_node_s node; int status = MMDB_read_node( ri->reader->mmdb, (uint32_t)cur->record, &node); diff --git a/maxminddb/reader.py b/maxminddb/reader.py index b4def914..3d140c9c 100644 --- a/maxminddb/reader.py +++ b/maxminddb/reader.py @@ -199,8 +199,8 @@ def _generate_children(self, node: int, depth: int, ip_acc: int) -> Iterator: return node_count = self._metadata.node_count + bits = 128 if self._metadata.ip_version == 6 else 32 if node > node_count: - bits = 128 if self._metadata.ip_version == 6 else 32 ip_acc <<= bits - depth if ip_acc <= _IPV4_MAX_NUM and bits == 128: depth -= 96 @@ -211,6 +211,11 @@ def _generate_children(self, node: int, depth: int, ip_acc: int) -> Iterator: ), ) elif node < node_count: + # A node at the full address depth has no valid children. Only a + # corrupt tree, such as one with a cycle, has one. + if depth >= bits: + msg = "The MaxMind DB file's search tree is corrupt" + raise InvalidDatabaseError(msg) left = self._read_node(node, 0) ip_acc <<= 1 depth += 1 diff --git a/tests/reader_test.py b/tests/reader_test.py index 6c14c9bb..198b3cf5 100644 --- a/tests/reader_test.py +++ b/tests/reader_test.py @@ -610,6 +610,21 @@ def test_exhausted_iterator_stops_after_close(self) -> None: reader.close() self.assertEqual(next(iterator, "done"), "done") + def test_cyclic_search_tree_is_rejected(self) -> None: + data = bytearray( + pathlib.Path(f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb").read_bytes(), + ) + # Point the left record of node 1 back at node 1. + data[6:9] = b"\x00\x00\x01" + with tempfile.TemporaryDirectory() as directory: + path = pathlib.Path(directory) / "cyclic.mmdb" + path.write_bytes(data) + with ( + open_database(str(path), self.mode) as reader, + self.assertRaisesRegex(InvalidDatabaseError, "search tree is corrupt"), + ): + list(reader) + def test_ip_validation(self) -> None: reader = open_database( "tests/data/test-data/MaxMind-DB-test-decoder.mmdb", From 1f47d49f2d7bfb399ee0dd463662d306aeee0e8e Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Sat, 3 Oct 2026 06:23:44 +0000 Subject: [PATCH 17/38] Stop an iterator after an error After an error, such as a corrupt search tree, the C iterator kept its pending records, and the next call continued from them. With a cycle in the tree, a caller that skipped bad records could get many errors before StopIteration. The pure Python iterator is a generator, so it stops after its first error. Free the pending records when next() fails, so the C iterator stops too. Reuse the loop from ReaderIter_dealloc as free_records. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 2 ++ extension/maxminddb.c | 30 ++++++++++++++++++++---------- tests/reader_test.py | 14 +++++++++----- 3 files changed, 31 insertions(+), 15 deletions(-) diff --git a/HISTORY.rst b/HISTORY.rst index 9a78a56a..6d940a91 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -26,6 +26,8 @@ History iterator. * An exhausted iterator now raises ``StopIteration`` after its ``Reader`` closes, not ``ValueError``. + * The iterator now stops after it raises an error, as the pure Python + iterator does. 3.2.0 (2026-09-10) ++++++++++++++++++ diff --git a/extension/maxminddb.c b/extension/maxminddb.c index 0450413f..2624c412 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -130,6 +130,7 @@ static inline maxminddb_state *get_maxminddb_state_from_self(PyObject *self) { static bool can_read(const char *path); static int get_record(PyObject *self, PyObject *args, PyObject **record); static PyObject *reader_iter_next(PyObject *self); +static void free_records(struct record *next); static bool format_sockaddr(struct sockaddr *addr, char *dst); static PyObject *from_entry_data_list(maxminddb_state *state, MMDB_entry_data_list_s **entry_data_list); @@ -810,18 +811,24 @@ static bool is_ipv6(char ip[16]) { } static PyObject *ReaderIter_next(PyObject *self) { + PyObject *result; #ifdef Py_GIL_DISABLED // The iterator's list of pending records is not thread-safe, so let only // one thread at a time advance an iterator. The read lock is shared, so // it does not do this. - PyObject *result; Py_BEGIN_CRITICAL_SECTION(self); +#endif result = reader_iter_next(self); + // Stop after an error, as a generator does. + if (result == NULL && PyErr_Occurred()) { + ReaderIter_obj *ri = (ReaderIter_obj *)self; + free_records(ri->next); + ri->next = NULL; + } +#ifdef Py_GIL_DISABLED Py_END_CRITICAL_SECTION(); - return result; -#else - return reader_iter_next(self); #endif + return result; } static PyObject *reader_iter_next(PyObject *self) { @@ -1012,17 +1019,20 @@ static PyObject *reader_iter_next(PyObject *self) { return NULL; } -static void ReaderIter_dealloc(PyObject *self) { - ReaderIter_obj *ri = (ReaderIter_obj *)self; - - Py_DECREF(ri->reader); - - struct record *next = ri->next; +static void free_records(struct record *next) { while (next != NULL) { struct record *cur = next; next = cur->next; free(cur); } +} + +static void ReaderIter_dealloc(PyObject *self) { + ReaderIter_obj *ri = (ReaderIter_obj *)self; + + Py_DECREF(ri->reader); + + free_records(ri->next); PyTypeObject *type = Py_TYPE(self); PyObject_Del(self); Py_DECREF(type); diff --git a/tests/reader_test.py b/tests/reader_test.py index 198b3cf5..2fe89931 100644 --- a/tests/reader_test.py +++ b/tests/reader_test.py @@ -619,11 +619,15 @@ def test_cyclic_search_tree_is_rejected(self) -> None: with tempfile.TemporaryDirectory() as directory: path = pathlib.Path(directory) / "cyclic.mmdb" path.write_bytes(data) - with ( - open_database(str(path), self.mode) as reader, - self.assertRaisesRegex(InvalidDatabaseError, "search tree is corrupt"), - ): - list(reader) + with open_database(str(path), self.mode) as reader: + iterator = iter(reader) + with self.assertRaisesRegex( + InvalidDatabaseError, + "search tree is corrupt", + ): + list(iterator) + # The iterator stops after an error, as a generator does. + self.assertEqual(next(iterator, "done"), "done") def test_ip_validation(self) -> None: reader = open_database( From 9103464d27f25738b080285b8fc6193df8966775 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 22:54:40 +0000 Subject: [PATCH 18/38] Validate metadata in the pure Python reader The reader passed every decoded metadata key to Metadata, with no check. An unknown key or a missing key raised a bare TypeError. The spec says that a new key is a minor version change, so a reader must accept keys that it does not know. A value of the wrong type passed, so Metadata held values that did not match its annotations. For example, a string node_count failed later with TypeError, and a string languages value opened with no error. libmaxminddb rejects a missing key or a wrong type with InvalidDatabaseError. Pass only the known keys to Metadata, after a check that each one is present and has the expected type. This also removes the annotated local that widened the unchecked metadata to dict[str, Any], and its comment, which said that the spec fixes the metadata keys. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 7 +++ maxminddb/reader.py | 58 +++++++++++++++++++++-- tests/reader_test.py | 110 +++++++++++++++++++++++++++++++++++++++++-- 3 files changed, 168 insertions(+), 7 deletions(-) diff --git a/HISTORY.rst b/HISTORY.rst index 6d940a91..0e3301f8 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -29,6 +29,13 @@ History * The iterator now stops after it raises an error, as the pure Python iterator does. +* Metadata: + + * The pure Python reader ignores unknown keys, which a new minor version of + the format can add. It raises ``InvalidDatabaseError`` for a missing key, a + value of the wrong type, an invalid ``ip_version`` or format version, or a + ``build_epoch`` of 0. + 3.2.0 (2026-09-10) ++++++++++++++++++ diff --git a/maxminddb/reader.py b/maxminddb/reader.py index 3d140c9c..9151ee5e 100644 --- a/maxminddb/reader.py +++ b/maxminddb/reader.py @@ -24,7 +24,7 @@ from typing_extensions import Self - from maxminddb.types import Record + from maxminddb.types import Record, RecordDict _IPV4_MAX_NUM = 2**32 @@ -94,9 +94,7 @@ def __init__( msg, ) - # The MaxMind DB spec fixes these keys and their value types. - fields: dict[str, Any] = metadata - self._metadata = Metadata(**fields) + self._metadata = Metadata(**_metadata_fields(metadata, filename)) self._record_size = self._metadata.record_size if self._record_size not in (24, 28, 32): msg = f"Unknown record size: {self._record_size}" @@ -340,6 +338,58 @@ def __enter__(self) -> Self: return self +# The type of each metadata value. libmaxminddb also rejects a database with a +# missing key or a value of another type. It also checks the width and sign of +# each integer, which the decoder does not report. +_METADATA_TYPES: dict[str, type] = { + "binary_format_major_version": int, + "binary_format_minor_version": int, + "build_epoch": int, + "database_type": str, + "description": dict, + "ip_version": int, + "languages": list, + "node_count": int, + "record_size": int, +} + + +def _metadata_fields(metadata: RecordDict, filename: object) -> dict[str, Any]: + """Return the known metadata fields after a check of their types. + + A new minor version of the format can add keys. This ignores them. + """ + prefix = f"Error reading metadata in database file ({filename})." + fields: dict[str, Any] = {} + for key, value_type in _METADATA_TYPES.items(): + value = metadata.get(key) + # The exact type check rejects bool, a subclass of int. + valid = type(value) is value_type + if valid and isinstance(value, list): + valid = all(type(v) is str for v in value) + elif valid and isinstance(value, dict): + valid = all(type(k) is str and type(v) is str for k, v in value.items()) + if not valid: + msg = f"{prefix} The {key} value is missing or has the wrong type." + raise InvalidDatabaseError(msg) + fields[key] = value + + # Range checks that libmaxminddb also makes. The reader decodes only the + # version 2 format, and ip_version drives the tree walk. libmaxminddb also + # rejects node_count 0, but this reader accepts an empty search tree. + if fields["binary_format_major_version"] != 2: + version = fields["binary_format_major_version"] + msg = f"{prefix} Unsupported binary format version {version}." + raise InvalidDatabaseError(msg) + if fields["ip_version"] not in (4, 6): + msg = f"{prefix} The ip_version is {fields['ip_version']}, not 4 or 6." + raise InvalidDatabaseError(msg) + if fields["build_epoch"] == 0: + msg = f"{prefix} The build_epoch is 0." + raise InvalidDatabaseError(msg) + return fields + + @dataclass(kw_only=True, frozen=True) class Metadata: """Metadata for the MaxMind DB reader.""" diff --git a/tests/reader_test.py b/tests/reader_test.py index 2fe89931..3a1aa7dd 100644 --- a/tests/reader_test.py +++ b/tests/reader_test.py @@ -1,11 +1,13 @@ from __future__ import annotations import contextlib +import dataclasses import io import ipaddress import multiprocessing import os import pathlib +import struct import sys import sysconfig import tempfile @@ -30,6 +32,7 @@ MODE_MMAP, MODE_MMAP_EXT, ) +from maxminddb.decoder import Decoder if TYPE_CHECKING: from collections.abc import Iterator @@ -95,6 +98,60 @@ def address_space_in_use() -> int: resource.setrlimit(resource.RLIMIT_AS, (soft, hard)) +_METADATA_START_MARKER = b"\xab\xcd\xefMaxMind.com" +# libmaxminddb requires these unsigned integer types for metadata values. +_METADATA_UINT_TYPES = {"build_epoch": 9, "node_count": 6} +_UINT16_TYPE = 5 + + +def _database_with_metadata(**changes: object) -> bytes: + """Return the decoder test database with changed metadata. + + A value of None removes the key. + """ + data = pathlib.Path(f"{_TEST_DATA_DIR}/MaxMind-DB-test-decoder.mmdb").read_bytes() + start = data.rfind(_METADATA_START_MARKER) + len(_METADATA_START_MARKER) + (metadata, _) = Decoder(data, start).decode(start) + merged = {**cast("dict[str, object]", metadata), **changes} + edited = {k: v for k, v in merged.items() if v is not None} + return data[:start] + _encode_value(edited) + + +def _encode_value(value: object, key: str = "") -> bytes: + if isinstance(value, str): + encoded = value.encode() + return _encode_control(2, len(encoded)) + encoded + if isinstance(value, bool): + return _encode_control(14, int(value)) + if isinstance(value, float): + return _encode_control(3, 8) + struct.pack(">d", value) + if isinstance(value, int): + encoded = value.to_bytes((value.bit_length() + 7) // 8, "big") + type_num = _METADATA_UINT_TYPES.get(key, _UINT16_TYPE) + return _encode_control(type_num, len(encoded)) + encoded + if isinstance(value, list): + items = b"".join(_encode_value(v) for v in value) + return _encode_control(11, len(value)) + items + if isinstance(value, dict): + items = b"".join( + _encode_value(k) + _encode_value(v, k) for k, v in value.items() + ) + return _encode_control(7, len(value)) + items + msg = f"cannot encode {value!r}" + raise TypeError(msg) + + +def _encode_control(type_num: int, size: int) -> bytes: + # Sizes from 29 to 284 use one extra size byte. These tests need no more. + extended = b"" + if type_num > 7: + extended = bytes([type_num - 7]) + type_num = 0 + if size < 29: + return bytes([type_num << 5 | size]) + extended + return bytes([type_num << 5 | 29]) + extended + bytes([size - 29]) + + def get_reader_from_file_descriptor(filepath: str, mode: int) -> Reader: """Patches open_database() for class TestFDReader().""" if mode == MODE_FD: @@ -629,6 +686,33 @@ def test_cyclic_search_tree_is_rejected(self) -> None: # The iterator stops after an error, as a generator does. self.assertEqual(next(iterator, "done"), "done") + def test_invalid_metadata_is_rejected(self) -> None: + cases: dict[str, dict[str, object]] = { + "missing languages": {"languages": None}, + "missing description": {"description": None}, + "string node_count": {"node_count": "1"}, + "double node_count": {"node_count": 1.5}, + "boolean record_size": {"record_size": True}, + "string languages": {"languages": "en"}, + "integer in languages": {"languages": [1]}, + "integer in description": {"description": {"en": 1}}, + "integer key in description": {"description": {1: "en"}}, + "integer database_type": {"database_type": 5}, + "ip_version 5": {"ip_version": 5}, + "binary_format_major_version 3": {"binary_format_major_version": 3}, + "build_epoch 0": {"build_epoch": 0}, + } + with tempfile.TemporaryDirectory() as directory: + path = pathlib.Path(directory) / "invalid-metadata.mmdb" + for name, changes in cases.items(): + with self.subTest(name): + path.write_bytes(_database_with_metadata(**changes)) + with ( + self.assertRaises(InvalidDatabaseError), + open_database(str(path), self.mode), + ): + pass + def test_ip_validation(self) -> None: reader = open_database( "tests/data/test-data/MaxMind-DB-test-decoder.mmdb", @@ -1165,6 +1249,19 @@ def setUp(self) -> None: class TestReaderInitialization(unittest.TestCase): + def test_metadata_types_match_metadata_fields(self) -> None: + self.assertEqual( + list(maxminddb.reader._METADATA_TYPES), # noqa: SLF001 + [field.name for field in dataclasses.fields(maxminddb.reader.Metadata)], + ) + + def test_unknown_metadata_key_is_ignored(self) -> None: + data = _database_with_metadata(unknown_key="value") + with maxminddb.reader.Reader(io.BytesIO(data), MODE_FD) as reader: + metadata = reader.metadata() + self.assertEqual(metadata.database_type, "MaxMind DB Decoder Test") + self.assertFalse(hasattr(metadata, "unknown_key")) + def test_empty_search_tree_is_accepted(self) -> None: data = pathlib.Path( f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb" @@ -1202,11 +1299,18 @@ def test_invalid_tree_metadata_is_rejected_on_open(self) -> None: def test_failed_initialization_closes_buffer(self) -> None: reader_class = maxminddb.reader.Reader - marker = b"\xab\xcd\xefMaxMind.com" cases = ( (b"not a database", InvalidDatabaseError, "Is this a valid MaxMind DB"), - (marker + b"\x40", InvalidDatabaseError, "Error reading metadata"), - (marker + b"\xe0", TypeError, "required keyword-only arguments"), + ( + _METADATA_START_MARKER + b"\x40", + InvalidDatabaseError, + "Error reading metadata", + ), + ( + _METADATA_START_MARKER + b"\xe0", + InvalidDatabaseError, + "missing or has the wrong type", + ), ( pathlib.Path( f"{_TEST_DATA_DIR}/MaxMind-DB-test-metadata-payload-limit.mmdb" From bcbdd3d9211b7ee9c65c227ec19069ef2af378c8 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 22:55:00 +0000 Subject: [PATCH 19/38] Ignore unknown metadata keys in the C extension The spec says that a new metadata key is a minor version change, so a reader must accept keys that it does not know. Reader.metadata() passed every key to the Metadata constructor, which accepts only the nine known keys. A database with an unknown key made metadata() raise TypeError. Before the segmentation fault fix, it crashed the process. Pass only the known keys to Metadata. Metadata_init and Reader_metadata now share one key list. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 2 ++ extension/maxminddb.c | 57 ++++++++++++++++++++++++++++++------------- tests/reader_test.py | 9 +++++++ 3 files changed, 51 insertions(+), 17 deletions(-) diff --git a/HISTORY.rst b/HISTORY.rst index 0e3301f8..9f97580c 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -35,6 +35,8 @@ History the format can add. It raises ``InvalidDatabaseError`` for a missing key, a value of the wrong type, an invalid ``ip_version`` or format version, or a ``build_epoch`` of 0. + * The C extension ignores unknown keys. Previously, ``Reader.metadata()`` + crashed on them. 3.2.0 (2026-09-10) ++++++++++++++++++ diff --git a/extension/maxminddb.c b/extension/maxminddb.c index 2624c412..5270ef22 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -634,6 +634,18 @@ static bool format_sockaddr(struct sockaddr *sa, char *dst) { return false; } +// The keys that Metadata accepts, in argument order. +static char *metadata_keys[] = {"binary_format_major_version", + "binary_format_minor_version", + "build_epoch", + "database_type", + "description", + "ip_version", + "languages", + "node_count", + "record_size", + NULL}; + static PyObject *Reader_metadata(PyObject *self, PyObject *UNUSED(args)) { maxminddb_state *state = get_maxminddb_state_from_self(self); if (state == NULL) { @@ -680,16 +692,38 @@ static PyObject *Reader_metadata(PyObject *self, PyObject *UNUSED(args)) { reader_release_read_lock(mmdb_obj); - PyObject *args = PyTuple_New(0); + // A newer minor version of the format can add metadata keys. Pass only + // the keys that Metadata accepts. + PyObject *args = PyTuple_New(Py_ARRAY_LENGTH(metadata_keys) - 1); if (args == NULL) { Py_DECREF(metadata_dict); return NULL; } - - PyObject *metadata = - PyObject_Call(state->Metadata_Type, args, metadata_dict); - + for (Py_ssize_t i = 0; metadata_keys[i] != NULL; i++) { + PyObject *key = PyUnicode_FromString(metadata_keys[i]); + if (key == NULL) { + Py_DECREF(args); + Py_DECREF(metadata_dict); + return NULL; + } + PyObject *value = PyDict_GetItemWithError(metadata_dict, key); + Py_DECREF(key); + if (value == NULL) { + if (!PyErr_Occurred()) { + PyErr_Format(state->MaxMindDB_error, + "Error decoding metadata. The %s value is " + "missing.", + metadata_keys[i]); + } + Py_DECREF(args); + Py_DECREF(metadata_dict); + return NULL; + } + PyTuple_SET_ITEM(args, i, Py_NewRef(value)); + } Py_DECREF(metadata_dict); + + PyObject *metadata = PyObject_CallObject(state->Metadata_Type, args); Py_DECREF(args); return metadata; } @@ -1044,21 +1078,10 @@ static int Metadata_init(PyObject *self, PyObject *args, PyObject *kwds) { *build_epoch, *database_type, *description, *ip_version, *languages, *node_count, *record_size; - static char *kwlist[] = {"binary_format_major_version", - "binary_format_minor_version", - "build_epoch", - "database_type", - "description", - "ip_version", - "languages", - "node_count", - "record_size", - NULL}; - if (!PyArg_ParseTupleAndKeywords(args, kwds, "OOOOOOOOO", - kwlist, + metadata_keys, &binary_format_major_version, &binary_format_minor_version, &build_epoch, diff --git a/tests/reader_test.py b/tests/reader_test.py index 3a1aa7dd..166da7ec 100644 --- a/tests/reader_test.py +++ b/tests/reader_test.py @@ -1058,6 +1058,15 @@ def test_map_key_that_is_not_a_string_is_rejected(self) -> None: ): reader.get("1.1.1.1") + def test_unknown_metadata_key_is_ignored(self) -> None: + with tempfile.TemporaryDirectory() as directory: + path = pathlib.Path(directory) / "unknown-key.mmdb" + path.write_bytes(_database_with_metadata(unknown_key="value")) + with maxminddb.extension.Reader(path) as reader: + metadata = reader.metadata() + self.assertEqual(metadata.database_type, "MaxMind DB Decoder Test") + self.assertFalse(hasattr(metadata, "unknown_key")) + @unittest.skipIf( not has_maxminddb_extension() and not os.environ.get("MM_FORCE_EXT_TESTS"), From 1d617496078fa55a9885b10d1d5b954528522f15 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 22:40:44 +0000 Subject: [PATCH 20/38] Add node_byte_size and search_tree_size to the C Metadata open_database() declares the pure Python Reader as its return type, even when it returns the extension Reader. Type checkers therefore accepted metadata().node_byte_size and metadata().search_tree_size, but the extension Metadata did not have them. In MODE_AUTO with the extension, the code raised AttributeError. Add both properties to the C type and to the stub. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 2 ++ extension/maxminddb.c | 47 +++++++++++++++++++++++++++++++++++++++++ maxminddb/extension.pyi | 8 +++++++ tests/reader_test.py | 9 ++++++++ 4 files changed, 66 insertions(+) diff --git a/HISTORY.rst b/HISTORY.rst index 9f97580c..d96a7031 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -28,6 +28,8 @@ History closes, not ``ValueError``. * The iterator now stops after it raises an error, as the pure Python iterator does. + * Added the ``node_byte_size`` and ``search_tree_size`` properties to + ``Metadata``, as the pure Python ``Metadata`` has. * Metadata: diff --git a/extension/maxminddb.c b/extension/maxminddb.c index 5270ef22..13a6d982 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -1388,6 +1388,52 @@ static PyMemberDef Metadata_members[] = { NULL}, {NULL, 0, 0, 0, NULL}}; +static PyObject *Metadata_node_byte_size(PyObject *self, + void *UNUSED(closure)) { + Metadata_obj *obj = (Metadata_obj *)self; + if (obj->record_size == NULL) { + PyErr_SetString(PyExc_AttributeError, "record_size is not set"); + return NULL; + } + PyObject *four = PyLong_FromLong(4); + if (four == NULL) { + return NULL; + } + PyObject *node_byte_size = PyNumber_FloorDivide(obj->record_size, four); + Py_DECREF(four); + return node_byte_size; +} + +static PyObject *Metadata_search_tree_size(PyObject *self, + void *UNUSED(closure)) { + Metadata_obj *obj = (Metadata_obj *)self; + if (obj->node_count == NULL) { + PyErr_SetString(PyExc_AttributeError, "node_count is not set"); + return NULL; + } + PyObject *node_byte_size = Metadata_node_byte_size(self, NULL); + if (node_byte_size == NULL) { + return NULL; + } + PyObject *search_tree_size = + PyNumber_Multiply(obj->node_count, node_byte_size); + Py_DECREF(node_byte_size); + return search_tree_size; +} + +// These match the properties of the pure Python Metadata class. +static PyGetSetDef Metadata_getset[] = {{"node_byte_size", + Metadata_node_byte_size, + NULL, + "The size of a node in bytes.", + NULL}, + {"search_tree_size", + Metadata_search_tree_size, + NULL, + "The size of the search tree.", + NULL}, + {NULL, NULL, NULL, NULL, NULL}}; + // ============================================================================= // Type specs for heap type conversion (PEP 489) // ============================================================================= @@ -1416,6 +1462,7 @@ static PyType_Slot Metadata_Type_slots[] = { {Py_tp_init, Metadata_init}, {Py_tp_methods, Metadata_methods}, {Py_tp_members, Metadata_members}, + {Py_tp_getset, Metadata_getset}, {0, NULL}, }; diff --git a/maxminddb/extension.pyi b/maxminddb/extension.pyi index a4d4d707..358431d8 100644 --- a/maxminddb/extension.pyi +++ b/maxminddb/extension.pyi @@ -126,3 +126,11 @@ class Metadata: record_size: int, ) -> None: """Create new Metadata object from the metadata fields in the spec.""" + + @property + def node_byte_size(self) -> int: + """The size of a node in bytes.""" + + @property + def search_tree_size(self) -> int: + """The size of the search tree.""" diff --git a/tests/reader_test.py b/tests/reader_test.py index 166da7ec..f781265c 100644 --- a/tests/reader_test.py +++ b/tests/reader_test.py @@ -963,6 +963,11 @@ def _check_metadata( self.assertGreater(metadata.node_count, 36) self.assertEqual(metadata.record_size, record_size) + self.assertEqual(metadata.node_byte_size, record_size // 4) + self.assertEqual( + metadata.search_tree_size, + metadata.node_count * record_size // 4, + ) def _check_ip_v4(self, reader: Reader, file_name: str) -> None: for i in range(6): @@ -1091,6 +1096,10 @@ def test_uninitialized_metadata(self) -> None: metadata_class = maxminddb.extension.Metadata metadata = metadata_class.__new__(metadata_class) self.assertIsNone(metadata.languages) + with self.assertRaisesRegex(AttributeError, "record_size is not set"): + _ = metadata.node_byte_size + with self.assertRaisesRegex(AttributeError, "node_count is not set"): + _ = metadata.search_tree_size del metadata def test_metadata_missing_argument(self) -> None: From 646a1549da16a44847e4f6bb793b044237de3701 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 22:38:44 +0000 Subject: [PATCH 21/38] Export Mode from maxminddb Since 3.0.0, the README has said that the modes are available from maxminddb.Mode. maxminddb did not import Mode, so that attribute raised AttributeError. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 2 ++ maxminddb/__init__.py | 2 ++ tests/reader_test.py | 5 +++++ 3 files changed, 9 insertions(+) diff --git a/HISTORY.rst b/HISTORY.rst index d96a7031..81554209 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -40,6 +40,8 @@ History * The C extension ignores unknown keys. Previously, ``Reader.metadata()`` crashed on them. +* Added ``maxminddb.Mode``, which the README already described. + 3.2.0 (2026-09-10) ++++++++++++++++++ diff --git a/maxminddb/__init__.py b/maxminddb/__init__.py index 576dc1d0..c17b230a 100644 --- a/maxminddb/__init__.py +++ b/maxminddb/__init__.py @@ -12,6 +12,7 @@ MODE_MEMORY, MODE_MMAP, MODE_MMAP_EXT, + Mode, ) from .errors import InvalidDatabaseError from .reader import Reader @@ -33,6 +34,7 @@ "MODE_MMAP", "MODE_MMAP_EXT", "InvalidDatabaseError", + "Mode", "Reader", "open_database", ] diff --git a/tests/reader_test.py b/tests/reader_test.py index f781265c..7f9519bc 100644 --- a/tests/reader_test.py +++ b/tests/reader_test.py @@ -1085,6 +1085,11 @@ class TestExtensionReaderWithIPObjects(BaseTestReader): reader_class = maxminddb.extension.Reader +class TestModule(unittest.TestCase): + def test_mode_is_exported(self) -> None: + self.assertIs(maxminddb.Mode, maxminddb.const.Mode) + + @unittest.skipIf( not has_maxminddb_extension() and not os.environ.get("MM_FORCE_EXT_TESTS"), "No C extension module found. Skipping tests", From 0c2afa5bb4c8274f86e3fecb26a1a599dfff0102 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 22:55:06 +0000 Subject: [PATCH 22/38] Add a release note for the concrete Record types GitHub #464 replaced the AnyStr TypeVar in Primitive with str | bytes. Primitive and Record are no longer generic aliases, so a subscript such as Record[str] now raises TypeError. HISTORY.rst did not mention the change. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 3 +++ 1 file changed, 3 insertions(+) diff --git a/HISTORY.rst b/HISTORY.rst index 81554209..dbb25922 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -41,6 +41,9 @@ History crashed on them. * Added ``maxminddb.Mode``, which the README already described. +* ``maxminddb.types.Record`` and ``Primitive`` are no longer generic type + aliases. Remove any subscript, such as ``Record[str]``. Pull request by Adam + Hitchcock. GitHub #464. 3.2.0 (2026-09-10) ++++++++++++++++++ From 920a294f43dfc6a85fc555c139b190851623a0d6 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 22:36:18 +0000 Subject: [PATCH 23/38] Add bytearray to Primitive The C extension decodes the MaxMind DB bytes type to bytearray. The pure Python reader decodes it to bytes. MODE_AUTO uses the extension when it is available, so most callers get bytearray. mypy does not treat bytearray as bytes, so it reported an isinstance(value, bytearray) check as unreachable. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 5 +++++ maxminddb/types.py | 2 +- 2 files changed, 6 insertions(+), 1 deletion(-) diff --git a/HISTORY.rst b/HISTORY.rst index dbb25922..8b3074f2 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -45,6 +45,11 @@ History aliases. Remove any subscript, such as ``Record[str]``. Pull request by Adam Hitchcock. GitHub #464. +* Type hints: + + * ``Primitive`` includes ``bytearray``, which the C extension returns for + the ``bytes`` type. + 3.2.0 (2026-09-10) ++++++++++++++++++ diff --git a/maxminddb/types.py b/maxminddb/types.py index dcb7b61e..eafe1655 100644 --- a/maxminddb/types.py +++ b/maxminddb/types.py @@ -4,7 +4,7 @@ from typing import TypeAlias -Primitive: TypeAlias = str | bytes | bool | float | int +Primitive: TypeAlias = str | bytes | bytearray | bool | float | int RecordList: TypeAlias = list["Record"] """RecordList is a type for lists in a database record.""" From caa6b67a4a8052f867c4d4ceaf7e5be8f15d6268 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 22:36:32 +0000 Subject: [PATCH 24/38] Add type arguments to Reader.__iter__ __iter__ and _generate_children returned a bare Iterator, so strict type checkers inferred Unknown for the network and record of each item. The extension stub already declares the item type. Use the same type. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 1 + maxminddb/reader.py | 11 ++++++++--- 2 files changed, 9 insertions(+), 3 deletions(-) diff --git a/HISTORY.rst b/HISTORY.rst index 8b3074f2..227381ae 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -49,6 +49,7 @@ History * ``Primitive`` includes ``bytearray``, which the C extension returns for the ``bytes`` type. + * ``Reader.__iter__`` declares its item type. 3.2.0 (2026-09-10) ++++++++++++++++++ diff --git a/maxminddb/reader.py b/maxminddb/reader.py index 9151ee5e..84f32d6b 100644 --- a/maxminddb/reader.py +++ b/maxminddb/reader.py @@ -10,7 +10,7 @@ import contextlib import ipaddress from dataclasses import dataclass -from ipaddress import IPv4Address, IPv6Address +from ipaddress import IPv4Address, IPv4Network, IPv6Address, IPv6Network from typing import IO, TYPE_CHECKING, Any from maxminddb.const import MODE_AUTO, MODE_FD, MODE_FILE, MODE_MEMORY, MODE_MMAP @@ -188,10 +188,15 @@ def get_with_prefix_len( return self._resolve_data_pointer(pointer), prefix_len return None, prefix_len - def __iter__(self) -> Iterator: + def __iter__(self) -> Iterator[tuple[IPv4Network | IPv6Network, Record]]: return self._generate_children(0, 0, 0) - def _generate_children(self, node: int, depth: int, ip_acc: int) -> Iterator: + def _generate_children( + self, + node: int, + depth: int, + ip_acc: int, + ) -> Iterator[tuple[IPv4Network | IPv6Network, Record]]: if ip_acc != 0 and node == self._ipv4_start: # Skip nodes aliased to IPv4 return From a78cf165ccd1356bda4e7af1f6deae699b753bca Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Sat, 3 Oct 2026 06:47:31 +0000 Subject: [PATCH 25/38] Convert only IPv4 networks when iterating an IPv6 tree Both iterators turned any network whose first 96 bits were zero into an IPv4 network and subtracted 96 from its prefix length. For a network shorter than /96, such as ::/1, the prefix length became negative, and iteration raised ValueError. An IPv4 network in an IPv6 tree is at least /96, so convert only those. The pure Python iterator had two more bugs here. It compared the address with 2**32 using <=, so ::1:0:0/96 got a prefix length of 0 and raised ValueError. It also skipped every data record equal to the IPv4 start node, although only a search node can be the IPv4 subtree. Use <, and skip only a search node, as the C iterator does. Build the network with IPv4Network or IPv6Network, because ip_network() picks IPv4 for any small integer. Its alias rule also differed from the C iterator. It skipped the IPv4 start node in an IPv4 tree, where a record that points back to the root is a cycle, and inside the IPv4 subtree of an IPv6 tree. Both hid a corrupt tree behind partial results. Skip the subtree only when an address with a set bit in its first 96 bits leads to it, as the C iterator does, and raise InvalidDatabaseError for a record that points to the root, which libmaxminddb treats as invalid. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 3 ++ extension/maxminddb.c | 9 ++++- maxminddb/reader.py | 40 +++++++++++-------- tests/reader_test.py | 90 +++++++++++++++++++++++++++++++++++++++++++ 4 files changed, 125 insertions(+), 17 deletions(-) diff --git a/HISTORY.rst b/HISTORY.rst index 227381ae..53e6c41c 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -6,6 +6,9 @@ History 3.3.0 ++++++++++++++++++ +* Fixed iteration over an IPv6 database with a network shorter than /96 + whose first bits are zero, such as ``::/1``. The readers raised + ``ValueError`` or skipped networks. * Iterating over a database with a corrupt search tree, such as one with a cycle, now raises ``InvalidDatabaseError``. Previously, the C extension could corrupt memory, and the pure Python reader raised ``RecursionError`` diff --git a/extension/maxminddb.c b/extension/maxminddb.c index 13a6d982..9fb43cda 100644 --- a/extension/maxminddb.c +++ b/extension/maxminddb.c @@ -896,8 +896,11 @@ static PyObject *reader_iter_next(PyObject *self) { switch (cur->type) { case MMDB_RECORD_TYPE_INVALID: reader_release_read_lock(ri->reader); + // libmaxminddb before 1.14 returns this type for a record + // that points to the root. Later versions fail in + // MMDB_read_node instead. PyErr_SetString(state->MaxMindDB_error, - "Invalid record when reading node"); + "The MaxMind DB file's search tree is corrupt"); free(cur); return NULL; case MMDB_RECORD_TYPE_SEARCH_NODE: { @@ -1002,7 +1005,9 @@ static PyObject *reader_iter_next(PyObject *self) { int ip_start = 0; Py_ssize_t ip_length = 4; if (depth == 128) { - if (is_ipv6(cur->ip_packed)) { + // A network shorter than /96 is IPv6, even if its first + // 96 bits are zero. + if (is_ipv6(cur->ip_packed) || cur->depth < 96) { // IPv6 address ip_length = 16; } else { diff --git a/maxminddb/reader.py b/maxminddb/reader.py index 84f32d6b..db6d2575 100644 --- a/maxminddb/reader.py +++ b/maxminddb/reader.py @@ -197,26 +197,36 @@ def _generate_children( depth: int, ip_acc: int, ) -> Iterator[tuple[IPv4Network | IPv6Network, Record]]: - if ip_acc != 0 and node == self._ipv4_start: - # Skip nodes aliased to IPv4 - return - node_count = self._metadata.node_count bits = 128 if self._metadata.ip_version == 6 else 32 + # Skip the IPv4 subtree when an IPv6 address other than ::/96 leads + # to it, as the C extension does. Inside the IPv4 subtree, or in an + # IPv4 tree, a record that points back to it is a cycle. + if ( + node == self._ipv4_start + and bits == 128 + and node < node_count + and ip_acc >> max(depth - 96, 0) != 0 + ): + return + if node > node_count: ip_acc <<= bits - depth - if ip_acc <= _IPV4_MAX_NUM and bits == 128: - depth -= 96 - yield ( - ipaddress.ip_network((ip_acc, depth)), - self._resolve_data_pointer( - node, - ), - ) + network: IPv4Network | IPv6Network + if bits == 32: + network = IPv4Network((ip_acc, depth)) + elif depth >= 96 and ip_acc < _IPV4_MAX_NUM: + # An IPv4 network in an IPv6 tree is at least /96, and its + # first 96 bits are zero. + network = IPv4Network((ip_acc, depth - 96)) + else: + network = IPv6Network((ip_acc, depth)) + yield (network, self._resolve_data_pointer(node)) elif node < node_count: - # A node at the full address depth has no valid children. Only a - # corrupt tree, such as one with a cycle, has one. - if depth >= bits: + # A node at the full address depth has no valid children, and no + # record can point to the root. Only a corrupt tree, such as one + # with a cycle, has either. + if depth >= bits or (node == 0 and depth > 0): msg = "The MaxMind DB file's search tree is corrupt" raise InvalidDatabaseError(msg) left = self._read_node(node, 0) diff --git a/tests/reader_test.py b/tests/reader_test.py index 7f9519bc..bbd77816 100644 --- a/tests/reader_test.py +++ b/tests/reader_test.py @@ -104,6 +104,33 @@ def address_space_in_use() -> int: _UINT16_TYPE = 5 +def _database(records: tuple[int, ...], *, ip_version: int) -> bytes: + """Return a database with a 24-bit search tree and one data record. + + The tree has two records per node, and a record of node_count + 16 points + at the data record, the string "net". + """ + metadata = { + "binary_format_major_version": 2, + "binary_format_minor_version": 0, + "build_epoch": 1, + "database_type": "Test", + "description": {"en": "Test"}, + "ip_version": ip_version, + "languages": ["en"], + "node_count": len(records) // 2, + "record_size": 24, + } + tree = b"".join(record.to_bytes(3, "big") for record in records) + return ( + tree + + bytes(16) + + _encode_value("net") + + _METADATA_START_MARKER + + _encode_value(metadata) + ) + + def _database_with_metadata(**changes: object) -> bytes: """Return the decoder test database with changed metadata. @@ -667,6 +694,44 @@ def test_exhausted_iterator_stops_after_close(self) -> None: reader.close() self.assertEqual(next(iterator, "done"), "done") + def test_record_that_points_to_the_root_is_rejected(self) -> None: + # Node 1's right record points back to the root. The left records point + # at data, so a walk through the root again would yield networks that + # the tree does not have, such as 192.0.0.0/3. + records = (18, 1, 18, 0) + with tempfile.TemporaryDirectory() as directory: + path = pathlib.Path(directory) / "root-record.mmdb" + path.write_bytes(_database(records, ip_version=4)) + with open_database(str(path), self.mode) as reader: + seen: list[ipaddress.IPv4Network | ipaddress.IPv6Network] = [] + with self.assertRaisesRegex( + InvalidDatabaseError, + "search tree is corrupt", + ): + for network, _ in reader: + seen.append(network) + valid = { + ipaddress.ip_network("0.0.0.0/1"), + ipaddress.ip_network("128.0.0.0/2"), + } + self.assertLessEqual(set(seen), valid) + + def test_iterate_ipv6_networks_shorter_than_96_bits(self) -> None: + # One node whose two records point at the same data record, so the + # tree holds ::/1 and 8000::/1. + records = (17, 17) + with tempfile.TemporaryDirectory() as directory: + path = pathlib.Path(directory) / "short-ipv6.mmdb" + path.write_bytes(_database(records, ip_version=6)) + with open_database(str(path), self.mode) as reader: + self.assertEqual( + list(reader), + [ + (ipaddress.ip_network("::/1"), "net"), + (ipaddress.ip_network("8000::/1"), "net"), + ], + ) + def test_cyclic_search_tree_is_rejected(self) -> None: data = bytearray( pathlib.Path(f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb").read_bytes(), @@ -686,6 +751,31 @@ def test_cyclic_search_tree_is_rejected(self) -> None: # The iterator stops after an error, as a generator does. self.assertEqual(next(iterator, "done"), "done") + # A record that points back to the root of an IPv4 tree. + broken = f"{_TEST_DATA_DIR}/MaxMind-DB-test-broken-search-tree-24.mmdb" + with ( + open_database(broken, self.mode) as reader, + self.assertRaisesRegex(InvalidDatabaseError, "search tree is corrupt"), + ): + list(reader) + + # A record in the IPv4 subtree of an IPv6 tree that points back to the + # IPv4 start node, 96. + mixed = bytearray( + pathlib.Path( + f"{_TEST_DATA_DIR}/MaxMind-DB-test-mixed-24.mmdb" + ).read_bytes(), + ) + mixed[240 * 6 : 240 * 6 + 3] = (96).to_bytes(3, "big") + with tempfile.TemporaryDirectory() as directory: + path = pathlib.Path(directory) / "ipv4-cycle.mmdb" + path.write_bytes(mixed) + with ( + open_database(str(path), self.mode) as reader, + self.assertRaisesRegex(InvalidDatabaseError, "search tree is corrupt"), + ): + list(reader) + def test_invalid_metadata_is_rejected(self) -> None: cases: dict[str, dict[str, object]] = { "missing languages": {"languages": None}, From a6d8628175d9ecdb57381ad7d7062aaf2dfee361 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Sat, 3 Oct 2026 12:30:35 +0000 Subject: [PATCH 26/38] Reject a data pointer into the separator _resolve_data_pointer checked only that a data pointer was inside the buffer. A search tree record that pointed into the 16-byte separator between the search tree and the data section decoded the zero bytes there, so get() returned {} and iteration yielded the network with {}. libmaxminddb rejects such a record as a corrupt search tree. Reject a pointer before the start of the data section too. A lookup benchmark showed no measurable cost. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 3 +++ maxminddb/reader.py | 13 +++++++++++-- tests/reader_test.py | 16 ++++++++++++++++ 3 files changed, 30 insertions(+), 2 deletions(-) diff --git a/HISTORY.rst b/HISTORY.rst index 53e6c41c..60751ce8 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -9,6 +9,9 @@ History * Fixed iteration over an IPv6 database with a network shorter than /96 whose first bits are zero, such as ``::/1``. The readers raised ``ValueError`` or skipped networks. +* The pure Python reader now raises ``InvalidDatabaseError`` for a search tree + record that points before the data section. Previously, it returned an + empty map. * Iterating over a database with a corrupt search tree, such as one with a cycle, now raises ``InvalidDatabaseError``. Previously, the C extension could corrupt memory, and the pure Python reader raised ``RecursionError`` diff --git a/maxminddb/reader.py b/maxminddb/reader.py index db6d2575..c3a34dfb 100644 --- a/maxminddb/reader.py +++ b/maxminddb/reader.py @@ -45,6 +45,8 @@ class Reader: _metadata: Metadata _record_size: int _ipv4_start: int + _search_tree_size: int + _data_start: int def __init__( self, @@ -119,6 +121,11 @@ def __init__( self._buffer, self._metadata.search_tree_size + self._DATA_SECTION_SEPARATOR_SIZE, ) + # _resolve_data_pointer uses these on every lookup. + self._search_tree_size = self._metadata.search_tree_size + self._data_start = ( + self._search_tree_size + self._DATA_SECTION_SEPARATOR_SIZE + ) self.closed = False ipv4_start = 0 @@ -283,9 +290,11 @@ def _read_node(self, node_number: int, index: int) -> int: raise InvalidDatabaseError(msg) def _resolve_data_pointer(self, pointer: int) -> Record: - resolved = pointer - self._metadata.node_count + self._metadata.search_tree_size + resolved = pointer - self._metadata.node_count + self._search_tree_size - if resolved >= self._buffer_size: + # A pointer into the separator between the tree and the data section + # is as corrupt as one past the end, as libmaxminddb checks. + if resolved < self._data_start or resolved >= self._buffer_size: msg = "The MaxMind DB file's search tree is corrupt" raise InvalidDatabaseError(msg) diff --git a/tests/reader_test.py b/tests/reader_test.py index bbd77816..e5a3f054 100644 --- a/tests/reader_test.py +++ b/tests/reader_test.py @@ -732,6 +732,22 @@ def test_iterate_ipv6_networks_shorter_than_96_bits(self) -> None: ], ) + def test_record_that_points_into_the_separator_is_rejected(self) -> None: + # The left record, node_count + 1, points into the 16-byte separator + # between the search tree and the data section. libmaxminddb before + # 1.14 reports it as bad data. + with tempfile.TemporaryDirectory() as directory: + path = pathlib.Path(directory) / "separator.mmdb" + path.write_bytes(_database((2, 17), ip_version=4)) + with ( + open_database(str(path), self.mode) as reader, + self.assertRaisesRegex( + InvalidDatabaseError, + "search tree is corrupt|contains bad data", + ), + ): + reader.get(self.ipf("1.1.1.1")) + def test_cyclic_search_tree_is_rejected(self) -> None: data = bytearray( pathlib.Path(f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb").read_bytes(), From 82ff2b47779e47178be7ab21e7701994af75a08a Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 22:36:34 +0000 Subject: [PATCH 27/38] Annotate the __exit__ arguments Both __exit__ methods left their arguments untyped and suppressed the ruff warning. Strict type checkers report the missing annotation. Co-Authored-By: Claude Opus 5.5 --- maxminddb/extension.pyi | 2 +- maxminddb/reader.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/maxminddb/extension.pyi b/maxminddb/extension.pyi index 358431d8..cda98d74 100644 --- a/maxminddb/extension.pyi +++ b/maxminddb/extension.pyi @@ -59,7 +59,7 @@ class Reader: def __iter__(self) -> Iterator[tuple[IPv4Network | IPv6Network, Record]]: ... def __enter__(self) -> Self: ... - def __exit__(self, *args) -> None: ... # noqa: ANN002 + def __exit__(self, *args: object) -> None: ... class Metadata: """Metadata for the MaxMind DB reader.""" diff --git a/maxminddb/reader.py b/maxminddb/reader.py index c3a34dfb..522e586a 100644 --- a/maxminddb/reader.py +++ b/maxminddb/reader.py @@ -352,7 +352,7 @@ def close(self) -> None: self.closed = True - def __exit__(self, *_) -> None: # noqa: ANN002 + def __exit__(self, *_: object) -> None: self.close() def __enter__(self) -> Self: From 9fdb8889daa768f1211fdbe255bc633555835c27 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 22:36:42 +0000 Subject: [PATCH 28/38] Remove an unneeded type: ignore hasattr() never causes an attr-defined error. mypy --strict reports the ignore as unused. The ignore on the os.pread() call stays because Windows has no os.pread. Co-Authored-By: Claude Opus 5.5 --- maxminddb/file.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/maxminddb/file.py b/maxminddb/file.py index 1901d9ba..1d7945ea 100644 --- a/maxminddb/file.py +++ b/maxminddb/file.py @@ -51,7 +51,7 @@ def close(self) -> None: """Close file.""" self._handle.close() - if hasattr(os, "pread"): # type: ignore[attr-defined] + if hasattr(os, "pread"): def _read(self, buffersize: int, offset: int) -> bytes: """Read that uses pread.""" From 84dedfd036a67459bd457ec4956e1bda7bcc8210 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 22:37:12 +0000 Subject: [PATCH 29/38] Add type aliases for the database argument open_database(), Reader.__init__ and Reader._load_buffer each spelled out the database argument union. A change to the accepted types had to edit every copy, and nothing caught a missed one. Define StrOrBytesPath and DatabaseSource in maxminddb.types and use them. The extension stub accepts different types, so it keeps its own annotation. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 2 ++ maxminddb/__init__.py | 6 +++--- maxminddb/reader.py | 9 ++++----- maxminddb/types.py | 11 +++++++++-- 4 files changed, 18 insertions(+), 10 deletions(-) diff --git a/HISTORY.rst b/HISTORY.rst index 60751ce8..ba13d727 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -56,6 +56,8 @@ History * ``Primitive`` includes ``bytearray``, which the C extension returns for the ``bytes`` type. * ``Reader.__iter__`` declares its item type. + * Added the ``StrOrBytesPath`` and ``DatabaseSource`` aliases to + ``maxminddb.types``. 3.2.0 (2026-09-10) ++++++++++++++++++ diff --git a/maxminddb/__init__.py b/maxminddb/__init__.py index c17b230a..2d96923e 100644 --- a/maxminddb/__init__.py +++ b/maxminddb/__init__.py @@ -3,7 +3,7 @@ from __future__ import annotations from importlib.metadata import version -from typing import IO, TYPE_CHECKING, cast +from typing import TYPE_CHECKING, cast from .const import ( MODE_AUTO, @@ -18,7 +18,7 @@ from .reader import Reader if TYPE_CHECKING: - import os + from .types import DatabaseSource try: from . import extension as _extension @@ -41,7 +41,7 @@ def open_database( - database: str | bytes | int | os.PathLike[str] | os.PathLike[bytes] | IO[bytes], + database: DatabaseSource, mode: int = MODE_AUTO, ) -> Reader: """Open a MaxMind DB database. diff --git a/maxminddb/reader.py b/maxminddb/reader.py index 522e586a..f8dc83e4 100644 --- a/maxminddb/reader.py +++ b/maxminddb/reader.py @@ -11,7 +11,7 @@ import ipaddress from dataclasses import dataclass from ipaddress import IPv4Address, IPv4Network, IPv6Address, IPv6Network -from typing import IO, TYPE_CHECKING, Any +from typing import TYPE_CHECKING, Any from maxminddb.const import MODE_AUTO, MODE_FD, MODE_FILE, MODE_MEMORY, MODE_MMAP from maxminddb.decoder import Decoder @@ -20,11 +20,10 @@ if TYPE_CHECKING: from collections.abc import Iterator - from os import PathLike from typing_extensions import Self - from maxminddb.types import Record, RecordDict + from maxminddb.types import DatabaseSource, Record, RecordDict _IPV4_MAX_NUM = 2**32 @@ -50,7 +49,7 @@ class Reader: def __init__( self, - database: str | bytes | int | PathLike[str] | PathLike[bytes] | IO[bytes], + database: DatabaseSource, mode: int = MODE_AUTO, ) -> None: """Reader for the MaxMind DB file format. @@ -303,7 +302,7 @@ def _resolve_data_pointer(self, pointer: int) -> Record: def _load_buffer( self, - database: str | bytes | int | PathLike[str] | PathLike[bytes] | IO[bytes], + database: DatabaseSource, mode: int = MODE_AUTO, ) -> str: filename: Any diff --git a/maxminddb/types.py b/maxminddb/types.py index eafe1655..bd731da3 100644 --- a/maxminddb/types.py +++ b/maxminddb/types.py @@ -1,8 +1,9 @@ -"""Types representing database records.""" +"""Types for database records and database arguments.""" from __future__ import annotations -from typing import TypeAlias +import os +from typing import IO, TypeAlias Primitive: TypeAlias = str | bytes | bytearray | bool | float | int @@ -13,3 +14,9 @@ """RecordDict is a type for dicts in a database record.""" Record: TypeAlias = Primitive | RecordList | RecordDict + +StrOrBytesPath: TypeAlias = str | bytes | os.PathLike[str] | os.PathLike[bytes] +"""StrOrBytesPath is a type for a path to a database file.""" + +DatabaseSource: TypeAlias = StrOrBytesPath | int | IO[bytes] +"""DatabaseSource is a type for the database argument of a reader.""" From 9d2fe7b7b3d4e39f217f13fac176365924efc80f Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 22:59:42 +0000 Subject: [PATCH 30/38] Accept only paths in the extension Reader stub The extension parses the database argument with PyUnicode_FSConverter, so it accepts only str, bytes and os.PathLike. The stub also accepted int and IO[bytes], which raise TypeError. The extension does not support MODE_FD, so the docstring was wrong too. open_database() passes any database argument to the extension in MODE_AUTO and MODE_MMAP_EXT. The narrower stub shows that mismatch. Keep the runtime behavior and explain the type: ignore. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 2 ++ maxminddb/__init__.py | 5 +++-- maxminddb/extension.pyi | 8 +++----- 3 files changed, 8 insertions(+), 7 deletions(-) diff --git a/HISTORY.rst b/HISTORY.rst index ba13d727..a62deb52 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -58,6 +58,8 @@ History * ``Reader.__iter__`` declares its item type. * Added the ``StrOrBytesPath`` and ``DatabaseSource`` aliases to ``maxminddb.types``. + * The ``maxminddb.extension.Reader`` stub accepts only a path, as the + extension does. 3.2.0 (2026-09-10) ++++++++++++++++++ diff --git a/maxminddb/__init__.py b/maxminddb/__init__.py index 2d96923e..49b8576c 100644 --- a/maxminddb/__init__.py +++ b/maxminddb/__init__.py @@ -86,8 +86,9 @@ def open_database( # The C type exposes the same API as the Python Reader, so for type # checking purposes, pretend it is one. (Ideally this would be a subclass # of, or share a common parent class with, the Python Reader - # implementation.) - return cast("Reader", _extension.Reader(database, mode)) + # implementation.) The extension accepts only a path. It raises TypeError + # for a file descriptor or a file object. + return cast("Reader", _extension.Reader(database, mode)) # type: ignore[arg-type] __version__ = version("maxminddb") diff --git a/maxminddb/extension.pyi b/maxminddb/extension.pyi index cda98d74..15d3a4c4 100644 --- a/maxminddb/extension.pyi +++ b/maxminddb/extension.pyi @@ -2,12 +2,10 @@ from collections.abc import Iterator from ipaddress import IPv4Address, IPv4Network, IPv6Address, IPv6Network -from os import PathLike -from typing import IO from typing_extensions import Self -from maxminddb.types import Record +from maxminddb.types import Record, StrOrBytesPath class Reader: """A C extension implementation of a reader for the MaxMind DB format. @@ -19,14 +17,14 @@ class Reader: def __init__( self, - database: str | bytes | int | PathLike[str] | PathLike[bytes] | IO[bytes], + database: StrOrBytesPath, mode: int = ..., ) -> None: """Reader for the MaxMind DB file format. Arguments: database: A path to a valid MaxMind DB file such as a GeoIP database - file, or a file descriptor in the case of MODE_FD. + file. mode: mode to open the database with. The only supported modes are MODE_AUTO and MODE_MMAP_EXT. From 483ccde227ab62bab933c7674dbc32f91cfd816a Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 22:37:57 +0000 Subject: [PATCH 31/38] Accept any binary reader with MODE_FD MODE_FD calls only database.read() and reads database.name when it exists. IO[bytes] requires much more, so a GzipFile and other binary readers did not type-check, although they work at runtime. Accept any object with a read() method that returns bytes. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 3 +++ maxminddb/types.py | 12 ++++++++++-- 2 files changed, 13 insertions(+), 2 deletions(-) diff --git a/HISTORY.rst b/HISTORY.rst index a62deb52..fb56f36d 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -60,6 +60,9 @@ History ``maxminddb.types``. * The ``maxminddb.extension.Reader`` stub accepts only a path, as the extension does. + * ``MODE_FD`` accepts any object whose ``read()`` method returns ``bytes``, + such as a ``gzip.GzipFile``. ``maxminddb.types.SupportsRead`` describes + this type. 3.2.0 (2026-09-10) ++++++++++++++++++ diff --git a/maxminddb/types.py b/maxminddb/types.py index bd731da3..b14eabb3 100644 --- a/maxminddb/types.py +++ b/maxminddb/types.py @@ -3,7 +3,7 @@ from __future__ import annotations import os -from typing import IO, TypeAlias +from typing import Protocol, TypeAlias Primitive: TypeAlias = str | bytes | bytearray | bool | float | int @@ -18,5 +18,13 @@ StrOrBytesPath: TypeAlias = str | bytes | os.PathLike[str] | os.PathLike[bytes] """StrOrBytesPath is a type for a path to a database file.""" -DatabaseSource: TypeAlias = StrOrBytesPath | int | IO[bytes] + +class SupportsRead(Protocol): + """SupportsRead is a type for a binary file object for MODE_FD.""" + + def read(self) -> bytes: + """Return the remaining bytes.""" + + +DatabaseSource: TypeAlias = StrOrBytesPath | int | SupportsRead """DatabaseSource is a type for the database argument of a reader.""" From 3ddcafd3bdbe679851b2617bf630b9ac9ca1f42c Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Sat, 3 Oct 2026 04:30:57 +0000 Subject: [PATCH 32/38] Document maxminddb.types The release notes point users to the type aliases in maxminddb.types, such as Record, DatabaseSource and SupportsRead, but the API docs did not include that module. Add it. Co-Authored-By: Claude Opus 5.5 --- docs/index.rst | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/docs/index.rst b/docs/index.rst index b062a0d9..5eca7cb3 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -35,6 +35,14 @@ Database Reader :undoc-members: :show-inheritance: +===== +Types +===== + +.. automodule:: maxminddb.types + :members: + :undoc-members: + ================== Indices and tables ================== From 749fc07b35ddcd157a6ba349ea99216fcfe7db83 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 23:00:02 +0000 Subject: [PATCH 33/38] Choose the reader from the database type in MODE_AUTO MODE_AUTO passed every database argument to the C extension when it was installed, but the extension accepts only a path. A file object raised TypeError, although the pure Python reader can read it. Without the extension, a file object also raised TypeError. In MODE_AUTO, read a file object as MODE_FD does. A path still opens as a path, even if it also has read(), because some path objects, such as py.path.local, have a text read(). A file descriptor still raises TypeError with the extension, as before. The pure Python path modes close a descriptor, and a caller of the default mode might not expect that. Without the extension, MODE_AUTO still opens a descriptor, as before. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 3 +++ README.rst | 2 +- maxminddb/__init__.py | 31 +++++++++++++++++++------------ maxminddb/const.py | 5 ++++- maxminddb/reader.py | 10 +++++++++- tests/reader_test.py | 43 +++++++++++++++++++++++++++++++++++++++++++ 6 files changed, 79 insertions(+), 15 deletions(-) diff --git a/HISTORY.rst b/HISTORY.rst index fb56f36d..e4abff0a 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -64,6 +64,9 @@ History such as a ``gzip.GzipFile``. ``maxminddb.types.SupportsRead`` describes this type. +* ``MODE_AUTO`` now accepts a binary file object. Previously, this could + raise ``TypeError``. + 3.2.0 (2026-09-10) ++++++++++++++++++ diff --git a/README.rst b/README.rst index ccdc4ebc..9ed4f83d 100644 --- a/README.rst +++ b/README.rst @@ -49,7 +49,7 @@ second argument. The modes are available from ``maxminddb.Mode``. Valid modes ar * ``Mode.MEMORY`` - load database into memory. Pure Python. * ``Mode.FD`` - load database into memory from a file descriptor. Pure Python. * ``Mode.AUTO`` - try ``Mode.MMAP_EXT``, ``Mode.MMAP``, ``Mode.FILE`` in that - order. Default. + order. A file object uses ``Mode.FD``. Default. **NOTE**: When using ``Mode.FD``, it is the *caller's* responsibility to be sure that the file descriptor gets closed properly. The caller may close the diff --git a/maxminddb/__init__.py b/maxminddb/__init__.py index 49b8576c..c75fa9e0 100644 --- a/maxminddb/__init__.py +++ b/maxminddb/__init__.py @@ -2,6 +2,7 @@ from __future__ import annotations +import os from importlib.metadata import version from typing import TYPE_CHECKING, cast @@ -57,7 +58,7 @@ def open_database( * MODE_FD - the param passed via database is a file descriptor, not a path. This mode implies MODE_MEMORY. * MODE_AUTO - tries MODE_MMAP_EXT, MODE_MMAP, MODE_FILE in that - order. Default mode. + order. Uses MODE_FD for a file object. Default mode. """ if mode not in ( @@ -72,23 +73,29 @@ def open_database( raise ValueError(msg) has_extension = _extension and hasattr(_extension, "Reader") - use_extension = has_extension if mode == MODE_AUTO else mode == MODE_MMAP_EXT - if not use_extension: - return Reader(database, mode) - - if not has_extension: + if mode == MODE_MMAP_EXT and not has_extension: msg = "MODE_MMAP_EXT requires the maxminddb.extension module to be available" raise ValueError( msg, ) - # The C type exposes the same API as the Python Reader, so for type - # checking purposes, pretend it is one. (Ideally this would be a subclass - # of, or share a common parent class with, the Python Reader - # implementation.) The extension accepts only a path. It raises TypeError - # for a file descriptor or a file object. - return cast("Reader", _extension.Reader(database, mode)) # type: ignore[arg-type] + # The extension accepts only a path, so MODE_AUTO gives a file object to + # the pure Python reader. It still refuses a file descriptor, as before. + # The cast pretends the C reader is the pure Python Reader, which has the + # same API. + if mode in (MODE_AUTO, MODE_MMAP_EXT) and has_extension: + if isinstance(database, (str, bytes, os.PathLike)): + return cast("Reader", _extension.Reader(database, mode)) + if mode == MODE_MMAP_EXT or isinstance(database, int): + msg = ( + f"The C extension requires a path ({type(database).__name__} " + "given). Use MODE_FD for a file object, or MODE_MMAP for a " + "file descriptor." + ) + raise TypeError(msg) + + return Reader(database, mode) __version__ = version("maxminddb") diff --git a/maxminddb/const.py b/maxminddb/const.py index 0f7e9826..3aa541ec 100644 --- a/maxminddb/const.py +++ b/maxminddb/const.py @@ -10,7 +10,10 @@ class Mode(IntEnum): """ AUTO = 0 - """Try MODE_MMAP_EXT, MODE_MMAP, MODE_FILE in that order. Default mode.""" + """Try MODE_MMAP_EXT, MODE_MMAP, MODE_FILE in that order. Default mode. + + A file object uses MODE_FD. + """ MMAP_EXT = 1 """Use the C extension with memory map.""" diff --git a/maxminddb/reader.py b/maxminddb/reader.py index f8dc83e4..a6b80492 100644 --- a/maxminddb/reader.py +++ b/maxminddb/reader.py @@ -9,6 +9,7 @@ import contextlib import ipaddress +import os from dataclasses import dataclass from ipaddress import IPv4Address, IPv4Network, IPv6Address, IPv6Network from typing import TYPE_CHECKING, Any @@ -61,7 +62,8 @@ def __init__( * MODE_MMAP - read from memory map. * MODE_FILE - read database as standard file. * MODE_MEMORY - load database into memory. - * MODE_AUTO - tries MODE_MMAP and then MODE_FILE. Default. + * MODE_AUTO - tries MODE_MMAP and then MODE_FILE. Uses + MODE_FD for a file object. Default. * MODE_FD - the param passed via database is a file descriptor, not a path. This mode implies MODE_MEMORY. @@ -305,6 +307,12 @@ def _load_buffer( database: DatabaseSource, mode: int = MODE_AUTO, ) -> str: + # MODE_AUTO reads a file object as MODE_FD does. A path wins over + # read(), because some path objects also have a text read(). + if mode == MODE_AUTO and not isinstance( + database, (str, bytes, int, os.PathLike) + ): + mode = MODE_FD filename: Any if (mode == MODE_AUTO and mmap) or mode == MODE_MMAP: with open(database, "rb") as db_file: # type: ignore[arg-type] diff --git a/tests/reader_test.py b/tests/reader_test.py index e5a3f054..b446d83c 100644 --- a/tests/reader_test.py +++ b/tests/reader_test.py @@ -38,6 +38,7 @@ from collections.abc import Iterator from maxminddb.reader import Reader + from maxminddb.types import DatabaseSource # Directory holding the shared MaxMind DB test fixtures. @@ -1178,6 +1179,10 @@ def test_unknown_metadata_key_is_ignored(self) -> None: self.assertEqual(metadata.database_type, "MaxMind DB Decoder Test") self.assertFalse(hasattr(metadata, "unknown_key")) + def test_file_object_is_refused(self) -> None: + with self.assertRaisesRegex(TypeError, "requires a path"): + open_database(io.BytesIO(b""), MODE_MMAP_EXT) + @unittest.skipIf( not has_maxminddb_extension() and not os.environ.get("MM_FORCE_EXT_TESTS"), @@ -1384,6 +1389,44 @@ def test_metadata_types_match_metadata_fields(self) -> None: [field.name for field in dataclasses.fields(maxminddb.reader.Metadata)], ) + def test_auto_mode_accepts_any_database_type(self) -> None: + path = f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb" + data = pathlib.Path(path).read_bytes() + + class PathWithTextRead: + """A path object with a text read(), as py.path.local has.""" + + def __fspath__(self) -> str: + return path + + def read(self) -> str: + return "not the database" + + with open(path, "rb") as file_object: + sources: list[tuple[str, DatabaseSource]] = [ + ("path", path), + ("file object", file_object), + ("BytesIO", io.BytesIO(data)), + # A path wins over read(). + ("path with a text read()", PathWithTextRead()), + ] + for name, database in sources: + with ( + self.subTest(name), + maxminddb.open_database(database, MODE_AUTO) as reader, + ): + self.assertEqual(reader.get("1.1.1.1"), {"ip": "1.1.1.1"}) + + # The pure Python reader takes ownership of a descriptor and closes it. + with maxminddb.reader.Reader(os.open(path, os.O_RDONLY), MODE_AUTO) as reader: + self.assertEqual(reader.get("1.1.1.1"), {"ip": "1.1.1.1"}) + if has_maxminddb_extension(): + # open_database refuses a descriptor with the extension, as before. + descriptor = os.open(path, os.O_RDONLY) + self.addCleanup(os.close, descriptor) + with self.assertRaisesRegex(TypeError, r"\(int given\)"): + maxminddb.open_database(descriptor, MODE_AUTO) + def test_unknown_metadata_key_is_ignored(self) -> None: data = _database_with_metadata(unknown_key="value") with maxminddb.reader.Reader(io.BytesIO(data), MODE_FD) as reader: From e5a8ea4cef94fdab8f91f3a0cbf90f8c95c4b6d2 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 23:00:12 +0000 Subject: [PATCH 34/38] Describe the MODE_FD argument as a binary file object MODE_FD calls database.read(), so it takes a binary file object, and MODE_AUTO now reads one the same way. The docstrings and the README said that MODE_FD takes a file descriptor. An int file descriptor fails with AttributeError in MODE_FD. Co-Authored-By: Claude Opus 5.5 --- README.rst | 8 ++++---- maxminddb/__init__.py | 10 +++++++--- maxminddb/const.py | 2 +- maxminddb/reader.py | 11 ++++++++--- maxminddb/types.py | 2 +- 5 files changed, 21 insertions(+), 12 deletions(-) diff --git a/README.rst b/README.rst index 9ed4f83d..c546415b 100644 --- a/README.rst +++ b/README.rst @@ -39,7 +39,7 @@ provide `free GeoLite databases files must be decompressed with ``gunzip``. After you have obtained a database and imported the module, call -``open_database`` with a path, or file descriptor (in the case of ``Mode.FD``), +``open_database`` with a path, or binary file object (with ``Mode.FD`` or ``Mode.AUTO``), to the database as the first argument. Optionally, you may pass a mode as the second argument. The modes are available from ``maxminddb.Mode``. Valid modes are: @@ -47,13 +47,13 @@ second argument. The modes are available from ``maxminddb.Mode``. Valid modes ar * ``Mode.MMAP`` - read from memory map. Pure Python. * ``Mode.FILE`` - read database as standard file. Pure Python. * ``Mode.MEMORY`` - load database into memory. Pure Python. -* ``Mode.FD`` - load database into memory from a file descriptor. Pure Python. +* ``Mode.FD`` - load database into memory from a binary file object. Pure Python. * ``Mode.AUTO`` - try ``Mode.MMAP_EXT``, ``Mode.MMAP``, ``Mode.FILE`` in that order. A file object uses ``Mode.FD``. Default. **NOTE**: When using ``Mode.FD``, it is the *caller's* responsibility to be -sure that the file descriptor gets closed properly. The caller may close the -file descriptor immediately after the ``Reader`` object is created. +sure that the file object gets closed properly. The caller may close the +file object immediately after the ``Reader`` object is created. The ``open_database`` function returns a ``Reader`` object. To look up an IP address, use the ``get`` method on this object. The method will return the diff --git a/maxminddb/__init__.py b/maxminddb/__init__.py index c75fa9e0..967a132f 100644 --- a/maxminddb/__init__.py +++ b/maxminddb/__init__.py @@ -49,14 +49,18 @@ def open_database( Arguments: database: A path to a valid MaxMind DB file such as a GeoIP database - file, or a file descriptor in the case of MODE_FD. + file, or a binary file object for MODE_FD or MODE_AUTO. + MODE_MMAP, MODE_FILE and MODE_MEMORY also accept the file + descriptor of a regular file. MODE_MEMORY reads it from its + current offset, the others from the start. The reader + closes it, even when the file is not a valid database. mode: mode to open the database with. Valid mode are: * MODE_MMAP_EXT - use the C extension with memory map. * MODE_MMAP - read from memory map. Pure Python. * MODE_FILE - read database as standard file. Pure Python. * MODE_MEMORY - load database into memory. Pure Python. - * MODE_FD - the param passed via database is a file descriptor, not - a path. This mode implies MODE_MEMORY. + * MODE_FD - the param passed via database is a binary file + object, not a path. This mode implies MODE_MEMORY. * MODE_AUTO - tries MODE_MMAP_EXT, MODE_MMAP, MODE_FILE in that order. Uses MODE_FD for a file object. Default mode. diff --git a/maxminddb/const.py b/maxminddb/const.py index 3aa541ec..2b1669d7 100644 --- a/maxminddb/const.py +++ b/maxminddb/const.py @@ -28,7 +28,7 @@ class Mode(IntEnum): """Load database into memory. Pure Python.""" FD = 16 - """Database is a file descriptor, not a path. This mode implies MODE_MEMORY.""" + """Database is a binary file object, not a path. This mode implies MODE_MEMORY.""" # Backward compatibility: export both enum members and old-style constants diff --git a/maxminddb/reader.py b/maxminddb/reader.py index a6b80492..d46b336f 100644 --- a/maxminddb/reader.py +++ b/maxminddb/reader.py @@ -57,15 +57,20 @@ def __init__( Arguments: database: A path to a valid MaxMind DB file such as a GeoIP database - file, or a file descriptor in the case of MODE_FD. + file, or a binary file object for MODE_FD or MODE_AUTO. + MODE_AUTO, MODE_MMAP, MODE_FILE and MODE_MEMORY also + accept the file descriptor of a regular file. MODE_MEMORY + reads it from its current offset, the others from the + start. The reader closes it, even when the file is not a + valid database. mode: mode to open the database with. Valid mode are: * MODE_MMAP - read from memory map. * MODE_FILE - read database as standard file. * MODE_MEMORY - load database into memory. * MODE_AUTO - tries MODE_MMAP and then MODE_FILE. Uses MODE_FD for a file object. Default. - * MODE_FD - the param passed via database is a file descriptor, not - a path. This mode implies MODE_MEMORY. + * MODE_FD - the param passed via database is a binary file + object, not a path. This mode implies MODE_MEMORY. """ filename = self._load_buffer(database, mode) diff --git a/maxminddb/types.py b/maxminddb/types.py index b14eabb3..4f06a5b0 100644 --- a/maxminddb/types.py +++ b/maxminddb/types.py @@ -20,7 +20,7 @@ class SupportsRead(Protocol): - """SupportsRead is a type for a binary file object for MODE_FD.""" + """SupportsRead is a type for a binary file object for MODE_FD or MODE_AUTO.""" def read(self) -> bytes: """Return the remaining bytes.""" From 3a80f71329184cbf9fb5e95eae5f5ea05a7a40a9 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 22:56:53 +0000 Subject: [PATCH 35/38] Narrow the database type before opening it _load_buffer passed the database argument to open(), FileBuffer and .read() behind 5 type: ignore comments, and it declared -> str although it returns the database argument or a file object name. An Any local hid that mismatch. The ignores hid real mismatches too. MODE_FD with a path raised AttributeError, and a path mode with a file object raised a TypeError from open() that did not name the mode. Check the argument type for the mode and raise TypeError with the fix. The checks narrow the type, so the ignores go away. FileBuffer now takes the path types that open() accepts. The unsupported mode check still comes first. The function returns object, because the caller only formats the name into error messages. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 5 +- README.rst | 5 +- maxminddb/__init__.py | 15 +++-- maxminddb/file.py | 7 ++- maxminddb/reader.py | 118 ++++++++++++++++++++++++++++------------ tests/reader_test.py | 124 ++++++++++++++++++++++++++++++++++++++++++ 6 files changed, 230 insertions(+), 44 deletions(-) diff --git a/HISTORY.rst b/HISTORY.rst index e4abff0a..8f08019a 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -65,7 +65,10 @@ History this type. * ``MODE_AUTO`` now accepts a binary file object. Previously, this could - raise ``TypeError``. + raise ``TypeError``. It reads the file object into memory with the pure + Python reader, as ``MODE_FD`` does, so lookups are slower than with a path. +* The pure Python reader raises ``TypeError`` when the database argument does + not suit the mode. 3.2.0 (2026-09-10) ++++++++++++++++++ diff --git a/README.rst b/README.rst index c546415b..e4df5544 100644 --- a/README.rst +++ b/README.rst @@ -49,9 +49,10 @@ second argument. The modes are available from ``maxminddb.Mode``. Valid modes ar * ``Mode.MEMORY`` - load database into memory. Pure Python. * ``Mode.FD`` - load database into memory from a binary file object. Pure Python. * ``Mode.AUTO`` - try ``Mode.MMAP_EXT``, ``Mode.MMAP``, ``Mode.FILE`` in that - order. A file object uses ``Mode.FD``. Default. + order. A file object is read into memory with the pure Python reader, as + with ``Mode.FD``. Pass a path to use the faster C extension. Default. -**NOTE**: When using ``Mode.FD``, it is the *caller's* responsibility to be +**NOTE**: When using a file object, it is the *caller's* responsibility to be sure that the file object gets closed properly. The caller may close the file object immediately after the ``Reader`` object is created. diff --git a/maxminddb/__init__.py b/maxminddb/__init__.py index 967a132f..54a83249 100644 --- a/maxminddb/__init__.py +++ b/maxminddb/__init__.py @@ -2,7 +2,6 @@ from __future__ import annotations -import os from importlib.metadata import version from typing import TYPE_CHECKING, cast @@ -16,7 +15,7 @@ Mode, ) from .errors import InvalidDatabaseError -from .reader import Reader +from .reader import _PATH_TYPES, Reader if TYPE_CHECKING: from .types import DatabaseSource @@ -62,7 +61,8 @@ def open_database( * MODE_FD - the param passed via database is a binary file object, not a path. This mode implies MODE_MEMORY. * MODE_AUTO - tries MODE_MMAP_EXT, MODE_MMAP, MODE_FILE in that - order. Uses MODE_FD for a file object. Default mode. + order. Reads a file object into memory, as MODE_FD + does. Default mode. """ if mode not in ( @@ -89,9 +89,14 @@ def open_database( # The cast pretends the C reader is the pure Python Reader, which has the # same API. if mode in (MODE_AUTO, MODE_MMAP_EXT) and has_extension: - if isinstance(database, (str, bytes, os.PathLike)): + if isinstance(database, _PATH_TYPES): return cast("Reader", _extension.Reader(database, mode)) - if mode == MODE_MMAP_EXT or isinstance(database, int): + # An object with __index__, such as an int, is a file descriptor. + # Leave a bool to the pure Python reader, which refuses it. + is_descriptor = not isinstance(database, bool) and hasattr( + type(database), "__index__" + ) + if mode == MODE_MMAP_EXT or is_descriptor: msg = ( f"The C extension requires a path ({type(database).__name__} " "given). Use MODE_FD for a file object, or MODE_MMAP for a " diff --git a/maxminddb/file.py b/maxminddb/file.py index 1d7945ea..62ddb62c 100644 --- a/maxminddb/file.py +++ b/maxminddb/file.py @@ -3,18 +3,21 @@ from __future__ import annotations import os -from typing import overload +from typing import TYPE_CHECKING, overload try: from multiprocessing import Lock except ImportError: from threading import Lock # type: ignore[assignment] +if TYPE_CHECKING: + from maxminddb.types import StrOrBytesPath + class FileBuffer: """A slice-able file reader.""" - def __init__(self, database: str) -> None: + def __init__(self, database: StrOrBytesPath | int) -> None: """Create FileBuffer.""" self._handle = open(database, "rb") # noqa: SIM115 self._size = os.fstat(self._handle.fileno()).st_size diff --git a/maxminddb/reader.py b/maxminddb/reader.py index d46b336f..4deeb9db 100644 --- a/maxminddb/reader.py +++ b/maxminddb/reader.py @@ -9,6 +9,7 @@ import contextlib import ipaddress +import operator import os from dataclasses import dataclass from ipaddress import IPv4Address, IPv4Network, IPv6Address, IPv6Network @@ -24,9 +25,18 @@ from typing_extensions import Self - from maxminddb.types import DatabaseSource, Record, RecordDict + from maxminddb.types import ( + DatabaseSource, + Record, + RecordDict, + StrOrBytesPath, + ) _IPV4_MAX_NUM = 2**32 +# The path types, which the C extension also accepts. +_PATH_TYPES = (str, bytes, os.PathLike) +# The database types that the path modes pass to open(). +_PATH_OR_FD_TYPES = (*_PATH_TYPES, int) class Reader: @@ -67,8 +77,9 @@ def __init__( * MODE_MMAP - read from memory map. * MODE_FILE - read database as standard file. * MODE_MEMORY - load database into memory. - * MODE_AUTO - tries MODE_MMAP and then MODE_FILE. Uses - MODE_FD for a file object. Default. + * MODE_AUTO - tries MODE_MMAP and then MODE_FILE. Reads a + file object into memory, as MODE_FD does. + Default. * MODE_FD - the param passed via database is a binary file object, not a path. This mode implies MODE_MEMORY. @@ -311,48 +322,87 @@ def _load_buffer( self, database: DatabaseSource, mode: int = MODE_AUTO, - ) -> str: - # MODE_AUTO reads a file object as MODE_FD does. A path wins over - # read(), because some path objects also have a text read(). - if mode == MODE_AUTO and not isinstance( - database, (str, bytes, int, os.PathLike) + ) -> object: + """Load the database and return a name for it in error messages.""" + if mode not in (MODE_AUTO, MODE_FD, MODE_FILE, MODE_MEMORY, MODE_MMAP): + msg = ( + f"Unsupported open mode ({mode}). Only MODE_AUTO, MODE_FILE, " + "MODE_MEMORY and MODE_FD are supported by the pure Python " + "Reader" + ) + raise ValueError( + msg, + ) + + # bool is an int, but it is never a file descriptor. MODE_FD refuses + # it below, as any object without read(). + if isinstance(database, bool) and mode != MODE_FD: + msg = "Unsupported database type (bool). Pass a path or a file object." + raise TypeError(msg) + # open() also takes an object with __index__, such as numpy.int64, as + # a file descriptor. + if ( + mode != MODE_FD + and not isinstance(database, _PATH_OR_FD_TYPES) + and hasattr(type(database), "__index__") + ): + database = operator.index(database) # type: ignore[arg-type] + # A path wins over read() in the path modes, because some path objects + # also have a text read(). MODE_FD reads any object with read(). + if mode != MODE_FD and isinstance(database, _PATH_OR_FD_TYPES): + return self._load_path(database, mode) + # MODE_AUTO reads a file object into memory, as MODE_FD does. + if mode == MODE_FD or ( + mode == MODE_AUTO and callable(getattr(database, "read", None)) ): - mode = MODE_FD - filename: Any + return self._load_file_object(database) + if mode == MODE_AUTO: + hint = "Pass a path or a binary file object." + else: + hint = "Use MODE_FD for a file object." + msg = f"Unsupported database type ({type(database).__name__}). {hint}" + raise TypeError(msg) + + def _load_path(self, database: StrOrBytesPath | int, mode: int) -> object: + """Memory-map or read a path or file descriptor.""" if (mode == MODE_AUTO and mmap) or mode == MODE_MMAP: - with open(database, "rb") as db_file: # type: ignore[arg-type] + with open(database, "rb") as db_file: self._buffer = mmap.mmap(db_file.fileno(), 0, access=mmap.ACCESS_READ) self._buffer_size = self._buffer.size() - filename = database elif mode in (MODE_AUTO, MODE_FILE): - self._buffer = FileBuffer(database) # type: ignore[arg-type] + self._buffer = FileBuffer(database) self._buffer_size = self._buffer.size() - filename = database - elif mode == MODE_MEMORY: - with open(database, "rb") as db_file: # type: ignore[arg-type] + else: + with open(database, "rb") as db_file: buf = db_file.read() self._buffer = buf self._buffer_size = len(buf) - filename = database - elif mode == MODE_FD: - self._buffer = database.read() # type: ignore[union-attr] - self._buffer_size = len(self._buffer) # type: ignore[arg-type] - # io buffers are not guaranteed to have a name attribute - if hasattr(database, "name"): - filename = database.name # type: ignore[union-attr] - else: - filename = f"<{type(database)}>" - else: + return database + + def _load_file_object(self, database: DatabaseSource) -> object: + """Read a binary file object into memory.""" + read = getattr(database, "read", None) + if not callable(read): msg = ( - f"Unsupported open mode ({mode}). Only MODE_AUTO, MODE_FILE, " - "MODE_MEMORY and MODE_FD are supported by the pure Python " - "Reader" - ) - raise ValueError( - msg, + f"Unsupported database type for MODE_FD " + f"({type(database).__name__}). Pass a binary file object." ) - - return filename + raise TypeError(msg) + buf = read() + if isinstance(buf, bytearray): + # The decoder takes bytes. This copies the database once. + buf = bytes(buf) + if not isinstance(buf, bytes): + msg = f"The database file object returned {type(buf).__name__}, not bytes." + if isinstance(buf, str): + msg += " Open it in binary mode." + raise TypeError(msg) + self._buffer = buf + self._buffer_size = len(buf) + # io buffers are not guaranteed to have a name attribute + if hasattr(database, "name"): + return database.name + return f"<{type(database)}>" def close(self) -> None: """Close the MaxMind DB file and returns the resources to the system. diff --git a/tests/reader_test.py b/tests/reader_test.py index b446d83c..7c1902aa 100644 --- a/tests/reader_test.py +++ b/tests/reader_test.py @@ -2,6 +2,7 @@ import contextlib import dataclasses +import gzip import io import ipaddress import multiprocessing @@ -1389,6 +1390,45 @@ def test_metadata_types_match_metadata_fields(self) -> None: [field.name for field in dataclasses.fields(maxminddb.reader.Metadata)], ) + def test_auto_mode_reads_file_object_from_its_position(self) -> None: + data = pathlib.Path( + f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb", + ).read_bytes() + with tempfile.TemporaryFile() as file_object: + file_object.write(b"header" + data) + file_object.seek(len(b"header")) + with maxminddb.open_database(file_object, MODE_AUTO) as reader: + self.assertEqual(reader.get("1.1.1.1"), {"ip": "1.1.1.1"}) + + def test_auto_mode_reads_wrapped_and_piped_streams(self) -> None: + path = f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb" + data = pathlib.Path(path).read_bytes() + + with tempfile.TemporaryDirectory() as directory: + gz_path = pathlib.Path(directory) / "db.mmdb.gz" + with gzip.open(gz_path, "wb") as compressed: + compressed.write(data) + with open(gz_path, "rb") as backing: + stream = io.BufferedReader(gzip.GzipFile(fileobj=backing)) + with maxminddb.open_database(stream, MODE_AUTO) as reader: + self.assertEqual(reader.get("1.1.1.1"), {"ip": "1.1.1.1"}) + # The reader does not close the caller's stream. + self.assertFalse(stream.closed) + stream.close() + + # The fixture is far smaller than the pipe buffer, so the write to the + # pipe does not block. + read_fd, write_fd = os.pipe() + os.write(write_fd, data) + os.close(write_fd) + pipe = os.fdopen(read_fd, "rb") + try: + with maxminddb.open_database(pipe, MODE_AUTO) as reader: + self.assertEqual(reader.get("1.1.1.1"), {"ip": "1.1.1.1"}) + self.assertFalse(pipe.closed) + finally: + pipe.close() + def test_auto_mode_accepts_any_database_type(self) -> None: path = f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb" data = pathlib.Path(path).read_bytes() @@ -1427,6 +1467,90 @@ def read(self) -> str: with self.assertRaisesRegex(TypeError, r"\(int given\)"): maxminddb.open_database(descriptor, MODE_AUTO) + def test_database_type_must_match_mode(self) -> None: + path = f"{_TEST_DATA_DIR}/MaxMind-DB-test-decoder.mmdb" + reader_class = maxminddb.reader.Reader + with self.assertRaisesRegex(TypeError, r"MODE_FD \(str\)"): + reader_class(path, MODE_FD) + with open(path, "rb") as database: + for mode in (MODE_FILE, MODE_MEMORY, MODE_MMAP): + with ( + self.subTest(mode=mode), + self.assertRaisesRegex(TypeError, "Use MODE_FD"), + ): + reader_class(database, mode) + with self.assertRaisesRegex(ValueError, "Unsupported open mode"): + reader_class(database, MODE_MMAP_EXT) + + for bad in (None, object()): + with self.assertRaisesRegex(TypeError, "Unsupported database type"): + reader_class(bad, MODE_AUTO) # type: ignore[arg-type] + + for mode in (MODE_AUTO, MODE_FD): + with ( + self.subTest(mode=mode), + open(path, encoding="latin-1") as text_file, + self.assertRaisesRegex(TypeError, "binary mode"), + ): + reader_class(text_file, mode) # type: ignore[arg-type] + + # A bool is an int, but not a file descriptor. + with self.assertRaisesRegex(TypeError, r"\(bool\)"): + reader_class(False, MODE_AUTO) # noqa: FBT003 + + def test_path_modes_accept_a_descriptor_with_index(self) -> None: + class Descriptor: + """A file descriptor object, as numpy.int64 is.""" + + def __init__(self, fd: int) -> None: + self.fd = fd + + def __index__(self) -> int: + return self.fd + + path = f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb" + # The reader takes ownership of the descriptor and closes it. + descriptor = Descriptor(os.open(path, os.O_RDONLY)) + with maxminddb.reader.Reader(descriptor, MODE_MMAP) as reader: # type: ignore[arg-type] + self.assertEqual(reader.get("1.1.1.1"), {"ip": "1.1.1.1"}) + + if has_maxminddb_extension(): + # With the extension, MODE_AUTO refuses it, as it refuses an int. + fd = os.open(path, os.O_RDONLY) + self.addCleanup(os.close, fd) + with self.assertRaisesRegex(TypeError, r"\(Descriptor given\)"): + maxminddb.open_database(Descriptor(fd)) # type: ignore[arg-type] + # A bool gets the bool error, not advice to use MODE_MMAP. + with self.assertRaisesRegex(TypeError, r"\(bool\)\. Pass a path"): + maxminddb.open_database(False) # noqa: FBT003 + + def test_fd_mode_reads_any_binary_reader(self) -> None: + data = pathlib.Path( + f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb", + ).read_bytes() + + class FileWithPath(io.BytesIO): + def __fspath__(self) -> str: + return "does-not-exist.mmdb" + + class BytearrayReader: + def read(self) -> bytes: + return bytearray(data) # type: ignore[return-value] + + for database in (FileWithPath(data), BytearrayReader()): + with ( + self.subTest(type(database).__name__), + maxminddb.reader.Reader(database, MODE_FD) as reader, + ): + self.assertEqual(reader.get("1.1.1.1"), {"ip": "1.1.1.1"}) + + class NoneReader: + def read(self) -> None: + return None + + with self.assertRaisesRegex(TypeError, r"returned NoneType, not bytes\.$"): + maxminddb.reader.Reader(NoneReader(), MODE_FD) # type: ignore[arg-type] + def test_unknown_metadata_key_is_ignored(self) -> None: data = _database_with_metadata(unknown_key="value") with maxminddb.reader.Reader(io.BytesIO(data), MODE_FD) as reader: From 8b5f39f2fbe87ab84f1bed3c85d7864a88d05f66 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Sat, 3 Oct 2026 04:29:45 +0000 Subject: [PATCH 36/38] List MODE_MMAP in the unsupported mode error The pure Python Reader accepts MODE_MMAP, but the error for an unsupported mode listed only MODE_AUTO, MODE_FILE, MODE_MEMORY and MODE_FD. A caller who passed MODE_MMAP_EXT was told that MODE_MMAP was not supported. Co-Authored-By: Claude Opus 5.5 --- maxminddb/reader.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/maxminddb/reader.py b/maxminddb/reader.py index 4deeb9db..de911bb9 100644 --- a/maxminddb/reader.py +++ b/maxminddb/reader.py @@ -326,9 +326,9 @@ def _load_buffer( """Load the database and return a name for it in error messages.""" if mode not in (MODE_AUTO, MODE_FD, MODE_FILE, MODE_MEMORY, MODE_MMAP): msg = ( - f"Unsupported open mode ({mode}). Only MODE_AUTO, MODE_FILE, " - "MODE_MEMORY and MODE_FD are supported by the pure Python " - "Reader" + f"Unsupported open mode ({mode}). Only MODE_AUTO, MODE_MMAP, " + "MODE_FILE, MODE_MEMORY and MODE_FD are supported by the pure " + "Python Reader" ) raise ValueError( msg, From 54dbd9831fed902ea86072d02dcef793f922cbca Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Sat, 3 Oct 2026 11:17:20 +0000 Subject: [PATCH 37/38] Name the file object type plainly in error messages The name for a file object without a name attribute was f"<{type(database)}>", so error messages showed "<>". MODE_AUTO now reads file objects too, so these messages are more common. Use the type name, as in "". Co-Authored-By: Claude Opus 5.5 --- maxminddb/reader.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/maxminddb/reader.py b/maxminddb/reader.py index de911bb9..ed48ef89 100644 --- a/maxminddb/reader.py +++ b/maxminddb/reader.py @@ -402,7 +402,7 @@ def _load_file_object(self, database: DatabaseSource) -> object: # io buffers are not guaranteed to have a name attribute if hasattr(database, "name"): return database.name - return f"<{type(database)}>" + return f"<{type(database).__name__}>" def close(self) -> None: """Close the MaxMind DB file and returns the resources to the system. From 8c2e838ae91553c3c148859bf4764da013459c1e Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 22:42:11 +0000 Subject: [PATCH 38/38] Add static type checks for the public API No test checked the public types, and non-strict mypy accepts a bare generic alias with no error. A TypeVar back in Primitive, a bare Iterator, or a wider stub would pass CI. mypy already checks tests/ in the lint environment. Add assert_type checks under TYPE_CHECKING, so the file costs nothing at runtime. The file enables warn-unused-ignores, so each type: ignore asserts that a call fails the type check. Co-Authored-By: Claude Opus 5.5 --- tests/typing_test.py | 51 ++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 51 insertions(+) create mode 100644 tests/typing_test.py diff --git a/tests/typing_test.py b/tests/typing_test.py new file mode 100644 index 00000000..353d52c0 --- /dev/null +++ b/tests/typing_test.py @@ -0,0 +1,51 @@ +# mypy: warn-unused-ignores +"""Static type checks for the public API. + +mypy checks this file in the lint environment. The code does not run. Each +type: ignore marks a call that must fail the type check. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +if TYPE_CHECKING: + import gzip + import io + from ipaddress import IPv4Network, IPv6Network + + from typing_extensions import assert_type + + import maxminddb + import maxminddb.extension + from maxminddb.types import Primitive, Record + + reader = maxminddb.open_database("GeoIP2-City.mmdb") + assert_type(reader.get("1.1.1.1"), Record | None) + assert_type(reader.get_with_prefix_len("1.1.1.1"), tuple[Record | None, int]) + for network, record in reader: + assert_type(network, IPv4Network | IPv6Network) + assert_type(record, Record) + assert_type(reader.metadata().search_tree_size, int) + + # Record includes the bytearray that the C extension returns for the + # bytes type. + value: Record = bytearray(b"\x00") + + # A TypeVar in either alias would give Record members of type Any. + def check_not_generic( + primitive: Primitive[str], # type: ignore[type-arg] + record: Record[str], # type: ignore[type-arg] + ) -> None: + pass + + def check_mode_fd(gzip_file: gzip.GzipFile, text_file: io.TextIOWrapper) -> None: + maxminddb.open_database(gzip_file, maxminddb.Mode.FD) + maxminddb.open_database(text_file, maxminddb.Mode.FD) # type: ignore[arg-type] + + extension_reader = maxminddb.extension.Reader("GeoIP2-City.mmdb") + for network, record in extension_reader: + assert_type(network, IPv4Network | IPv6Network) + assert_type(record, Record) + assert_type(extension_reader.metadata().node_byte_size, int) + maxminddb.extension.Reader(3) # type: ignore[arg-type]