From 90a5414e25717344cdda051eca5f5bd6fbe83e0c Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 23:00:02 +0000 Subject: [PATCH 1/5] Choose the reader from the database type in MODE_AUTO MODE_AUTO passed every database argument to the C extension when it was installed, but the extension accepts only a path. A file object raised TypeError, although the pure Python reader can read it. Without the extension, a file object also raised TypeError. In MODE_AUTO, read a file object as MODE_FD does. A path still opens as a path, even if it also has read(), because some path objects, such as py.path.local, have a text read(). A file descriptor still raises TypeError with the extension, as before. The pure Python path modes close a descriptor, and a caller of the default mode might not expect that. Without the extension, MODE_AUTO still opens a descriptor, as before. Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 3 +++ README.rst | 2 +- maxminddb/__init__.py | 31 +++++++++++++++++++------------ maxminddb/const.py | 5 ++++- maxminddb/reader.py | 10 +++++++++- tests/reader_test.py | 43 +++++++++++++++++++++++++++++++++++++++++++ 6 files changed, 79 insertions(+), 15 deletions(-) diff --git a/HISTORY.rst b/HISTORY.rst index 5d49a971..4b2c9134 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -74,6 +74,9 @@ History such as a ``gzip.GzipFile``. ``maxminddb.types.SupportsRead`` describes this type. +* ``MODE_AUTO`` now accepts a binary file object. Previously, this could + raise ``TypeError``. + 3.2.0 (2026-09-10) ++++++++++++++++++ diff --git a/README.rst b/README.rst index ccdc4ebc..9ed4f83d 100644 --- a/README.rst +++ b/README.rst @@ -49,7 +49,7 @@ second argument. The modes are available from ``maxminddb.Mode``. Valid modes ar * ``Mode.MEMORY`` - load database into memory. Pure Python. * ``Mode.FD`` - load database into memory from a file descriptor. Pure Python. * ``Mode.AUTO`` - try ``Mode.MMAP_EXT``, ``Mode.MMAP``, ``Mode.FILE`` in that - order. Default. + order. A file object uses ``Mode.FD``. Default. **NOTE**: When using ``Mode.FD``, it is the *caller's* responsibility to be sure that the file descriptor gets closed properly. The caller may close the diff --git a/maxminddb/__init__.py b/maxminddb/__init__.py index 49b8576c..c75fa9e0 100644 --- a/maxminddb/__init__.py +++ b/maxminddb/__init__.py @@ -2,6 +2,7 @@ from __future__ import annotations +import os from importlib.metadata import version from typing import TYPE_CHECKING, cast @@ -57,7 +58,7 @@ def open_database( * MODE_FD - the param passed via database is a file descriptor, not a path. This mode implies MODE_MEMORY. * MODE_AUTO - tries MODE_MMAP_EXT, MODE_MMAP, MODE_FILE in that - order. Default mode. + order. Uses MODE_FD for a file object. Default mode. """ if mode not in ( @@ -72,23 +73,29 @@ def open_database( raise ValueError(msg) has_extension = _extension and hasattr(_extension, "Reader") - use_extension = has_extension if mode == MODE_AUTO else mode == MODE_MMAP_EXT - if not use_extension: - return Reader(database, mode) - - if not has_extension: + if mode == MODE_MMAP_EXT and not has_extension: msg = "MODE_MMAP_EXT requires the maxminddb.extension module to be available" raise ValueError( msg, ) - # The C type exposes the same API as the Python Reader, so for type - # checking purposes, pretend it is one. (Ideally this would be a subclass - # of, or share a common parent class with, the Python Reader - # implementation.) The extension accepts only a path. It raises TypeError - # for a file descriptor or a file object. - return cast("Reader", _extension.Reader(database, mode)) # type: ignore[arg-type] + # The extension accepts only a path, so MODE_AUTO gives a file object to + # the pure Python reader. It still refuses a file descriptor, as before. + # The cast pretends the C reader is the pure Python Reader, which has the + # same API. + if mode in (MODE_AUTO, MODE_MMAP_EXT) and has_extension: + if isinstance(database, (str, bytes, os.PathLike)): + return cast("Reader", _extension.Reader(database, mode)) + if mode == MODE_MMAP_EXT or isinstance(database, int): + msg = ( + f"The C extension requires a path ({type(database).__name__} " + "given). Use MODE_FD for a file object, or MODE_MMAP for a " + "file descriptor." + ) + raise TypeError(msg) + + return Reader(database, mode) __version__ = version("maxminddb") diff --git a/maxminddb/const.py b/maxminddb/const.py index 0f7e9826..3aa541ec 100644 --- a/maxminddb/const.py +++ b/maxminddb/const.py @@ -10,7 +10,10 @@ class Mode(IntEnum): """ AUTO = 0 - """Try MODE_MMAP_EXT, MODE_MMAP, MODE_FILE in that order. Default mode.""" + """Try MODE_MMAP_EXT, MODE_MMAP, MODE_FILE in that order. Default mode. + + A file object uses MODE_FD. + """ MMAP_EXT = 1 """Use the C extension with memory map.""" diff --git a/maxminddb/reader.py b/maxminddb/reader.py index ce1d2b05..77f1e017 100644 --- a/maxminddb/reader.py +++ b/maxminddb/reader.py @@ -9,6 +9,7 @@ import contextlib import ipaddress +import os from dataclasses import dataclass from ipaddress import IPv4Address, IPv4Network, IPv6Address, IPv6Network from typing import TYPE_CHECKING, Any @@ -66,7 +67,8 @@ def __init__( * MODE_MMAP - read from memory map. * MODE_FILE - read database as standard file. * MODE_MEMORY - load database into memory. - * MODE_AUTO - tries MODE_MMAP and then MODE_FILE. Default. + * MODE_AUTO - tries MODE_MMAP and then MODE_FILE. Uses + MODE_FD for a file object. Default. * MODE_FD - the param passed via database is a file descriptor, not a path. This mode implies MODE_MEMORY. @@ -343,6 +345,12 @@ def _load_buffer( database: DatabaseSource, mode: int = MODE_AUTO, ) -> str: + # MODE_AUTO reads a file object as MODE_FD does. A path wins over + # read(), because some path objects also have a text read(). + if mode == MODE_AUTO and not isinstance( + database, (str, bytes, int, os.PathLike) + ): + mode = MODE_FD filename: Any if (mode == MODE_AUTO and mmap) or mode == MODE_MMAP: with open(database, "rb") as db_file: # type: ignore[arg-type] diff --git a/tests/reader_test.py b/tests/reader_test.py index 0529f1fc..5d92e889 100644 --- a/tests/reader_test.py +++ b/tests/reader_test.py @@ -44,6 +44,7 @@ from typing import IO from maxminddb.reader import Reader + from maxminddb.types import DatabaseSource # Directory holding the shared MaxMind DB test fixtures. @@ -1361,6 +1362,10 @@ def fail_to_decode(count: int) -> None: # A leaked dict on each failure keeps about 128 KB. self.assertLess(after - before, 16_000) + def test_file_object_is_refused(self) -> None: + with self.assertRaisesRegex(TypeError, "requires a path"): + open_database(io.BytesIO(b""), MODE_MMAP_EXT) + @unittest.skipIf( not has_maxminddb_extension() and not os.environ.get("MM_FORCE_EXT_TESTS"), @@ -1872,6 +1877,44 @@ def test_metadata_types_match_metadata_fields(self) -> None: [field.name for field in dataclasses.fields(maxminddb.reader.Metadata)], ) + def test_auto_mode_accepts_any_database_type(self) -> None: + path = f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb" + data = pathlib.Path(path).read_bytes() + + class PathWithTextRead: + """A path object with a text read(), as py.path.local has.""" + + def __fspath__(self) -> str: + return path + + def read(self) -> str: + return "not the database" + + with open(path, "rb") as file_object: + sources: list[tuple[str, DatabaseSource]] = [ + ("path", path), + ("file object", file_object), + ("BytesIO", io.BytesIO(data)), + # A path wins over read(). + ("path with a text read()", PathWithTextRead()), + ] + for name, database in sources: + with ( + self.subTest(name), + maxminddb.open_database(database, MODE_AUTO) as reader, + ): + self.assertEqual(reader.get("1.1.1.1"), {"ip": "1.1.1.1"}) + + # The pure Python reader takes ownership of a descriptor and closes it. + with maxminddb.reader.Reader(os.open(path, os.O_RDONLY), MODE_AUTO) as reader: + self.assertEqual(reader.get("1.1.1.1"), {"ip": "1.1.1.1"}) + if has_maxminddb_extension(): + # open_database refuses a descriptor with the extension, as before. + descriptor = os.open(path, os.O_RDONLY) + self.addCleanup(os.close, descriptor) + with self.assertRaisesRegex(TypeError, r"\(int given\)"): + maxminddb.open_database(descriptor, MODE_AUTO) + def test_empty_search_tree_is_accepted(self) -> None: data = pathlib.Path( f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb" From 6818636d8916c997199633439fcfd3cd17a2586b Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 23:00:12 +0000 Subject: [PATCH 2/5] Describe the MODE_FD argument as a binary file object MODE_FD calls database.read(), so it takes a binary file object, and MODE_AUTO now reads one the same way. The docstrings and the README said that MODE_FD takes a file descriptor. An int file descriptor fails with AttributeError in MODE_FD. Co-Authored-By: Claude Opus 5.5 --- README.rst | 8 ++++---- maxminddb/__init__.py | 11 ++++++++--- maxminddb/const.py | 2 +- maxminddb/reader.py | 11 ++++++++--- maxminddb/types.py | 2 +- 5 files changed, 22 insertions(+), 12 deletions(-) diff --git a/README.rst b/README.rst index 9ed4f83d..c546415b 100644 --- a/README.rst +++ b/README.rst @@ -39,7 +39,7 @@ provide `free GeoLite databases files must be decompressed with ``gunzip``. After you have obtained a database and imported the module, call -``open_database`` with a path, or file descriptor (in the case of ``Mode.FD``), +``open_database`` with a path, or binary file object (with ``Mode.FD`` or ``Mode.AUTO``), to the database as the first argument. Optionally, you may pass a mode as the second argument. The modes are available from ``maxminddb.Mode``. Valid modes are: @@ -47,13 +47,13 @@ second argument. The modes are available from ``maxminddb.Mode``. Valid modes ar * ``Mode.MMAP`` - read from memory map. Pure Python. * ``Mode.FILE`` - read database as standard file. Pure Python. * ``Mode.MEMORY`` - load database into memory. Pure Python. -* ``Mode.FD`` - load database into memory from a file descriptor. Pure Python. +* ``Mode.FD`` - load database into memory from a binary file object. Pure Python. * ``Mode.AUTO`` - try ``Mode.MMAP_EXT``, ``Mode.MMAP``, ``Mode.FILE`` in that order. A file object uses ``Mode.FD``. Default. **NOTE**: When using ``Mode.FD``, it is the *caller's* responsibility to be -sure that the file descriptor gets closed properly. The caller may close the -file descriptor immediately after the ``Reader`` object is created. +sure that the file object gets closed properly. The caller may close the +file object immediately after the ``Reader`` object is created. The ``open_database`` function returns a ``Reader`` object. To look up an IP address, use the ``get`` method on this object. The method will return the diff --git a/maxminddb/__init__.py b/maxminddb/__init__.py index c75fa9e0..d7542fe0 100644 --- a/maxminddb/__init__.py +++ b/maxminddb/__init__.py @@ -49,14 +49,19 @@ def open_database( Arguments: database: A path to a valid MaxMind DB file such as a GeoIP database - file, or a file descriptor in the case of MODE_FD. + file, or a binary file object for MODE_FD or MODE_AUTO. + MODE_MMAP, MODE_FILE and MODE_MEMORY also accept the file + descriptor of a regular file. MODE_MEMORY reads it from its + current offset, the others from the start. Without the C + extension, MODE_AUTO accepts one too. The reader closes it, + even when the file is not a valid database. mode: mode to open the database with. Valid mode are: * MODE_MMAP_EXT - use the C extension with memory map. * MODE_MMAP - read from memory map. Pure Python. * MODE_FILE - read database as standard file. Pure Python. * MODE_MEMORY - load database into memory. Pure Python. - * MODE_FD - the param passed via database is a file descriptor, not - a path. This mode implies MODE_MEMORY. + * MODE_FD - the param passed via database is a binary file + object, not a path. This mode implies MODE_MEMORY. * MODE_AUTO - tries MODE_MMAP_EXT, MODE_MMAP, MODE_FILE in that order. Uses MODE_FD for a file object. Default mode. diff --git a/maxminddb/const.py b/maxminddb/const.py index 3aa541ec..2b1669d7 100644 --- a/maxminddb/const.py +++ b/maxminddb/const.py @@ -28,7 +28,7 @@ class Mode(IntEnum): """Load database into memory. Pure Python.""" FD = 16 - """Database is a file descriptor, not a path. This mode implies MODE_MEMORY.""" + """Database is a binary file object, not a path. This mode implies MODE_MEMORY.""" # Backward compatibility: export both enum members and old-style constants diff --git a/maxminddb/reader.py b/maxminddb/reader.py index 77f1e017..2147454a 100644 --- a/maxminddb/reader.py +++ b/maxminddb/reader.py @@ -62,15 +62,20 @@ def __init__( Arguments: database: A path to a valid MaxMind DB file such as a GeoIP database - file, or a file descriptor in the case of MODE_FD. + file, or a binary file object for MODE_FD or MODE_AUTO. + MODE_AUTO, MODE_MMAP, MODE_FILE and MODE_MEMORY also + accept the file descriptor of a regular file. MODE_MEMORY + reads it from its current offset, the others from the + start. The reader closes it, even when the file is not a + valid database. mode: mode to open the database with. Valid mode are: * MODE_MMAP - read from memory map. * MODE_FILE - read database as standard file. * MODE_MEMORY - load database into memory. * MODE_AUTO - tries MODE_MMAP and then MODE_FILE. Uses MODE_FD for a file object. Default. - * MODE_FD - the param passed via database is a file descriptor, not - a path. This mode implies MODE_MEMORY. + * MODE_FD - the param passed via database is a binary file + object, not a path. This mode implies MODE_MEMORY. A second call reopens the reader with the new database. A failed call keeps the old one. Like close(), a second call can make reads in diff --git a/maxminddb/types.py b/maxminddb/types.py index eee642e1..32885cf8 100644 --- a/maxminddb/types.py +++ b/maxminddb/types.py @@ -22,7 +22,7 @@ class SupportsRead(Protocol): - """SupportsRead is a type for a binary file object for MODE_FD.""" + """SupportsRead is a type for a binary file object for MODE_FD or MODE_AUTO.""" def read(self) -> bytes: """Return the remaining bytes.""" From 4cf522cbe0a941fb0e8875a568fa22fbaa3eb233 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Fri, 2 Oct 2026 22:56:53 +0000 Subject: [PATCH 3/5] Narrow the database type before opening it _load_buffer passed the database argument to open(), FileBuffer and .read() behind 5 type: ignore comments, and it declared -> str although it returns the database argument or a file object name. An Any local hid that mismatch. The ignores hid real mismatches too. MODE_FD with a path raised AttributeError, and a path mode with a file object raised a TypeError from open() that did not name the mode. Check the argument type for the mode and raise TypeError with the fix. The checks narrow the type, so the ignores go away. FileBuffer now takes the path types that open() accepts. The unsupported mode check still comes first. The function returns object, because the caller only formats the name into error messages. A text file is refused before read(), which could fail to decode it. A bytearray from read() stays a bytearray, as before, so the buffer and decoder types now include it. MODE_FD now rejects an mmap from read(), so a source can no longer return a closable buffer. Remove the test of a source that returns the same mmap on each read(). Co-Authored-By: Claude Opus 5.5 --- HISTORY.rst | 8 ++- README.rst | 5 +- maxminddb/__init__.py | 25 ++++--- maxminddb/decoder.py | 4 +- maxminddb/file.py | 7 +- maxminddb/reader.py | 121 +++++++++++++++++++++++---------- tests/reader_test.py | 154 ++++++++++++++++++++++++++++++++++++++---- 7 files changed, 254 insertions(+), 70 deletions(-) diff --git a/HISTORY.rst b/HISTORY.rst index 4b2c9134..32332112 100644 --- a/HISTORY.rst +++ b/HISTORY.rst @@ -74,8 +74,12 @@ History such as a ``gzip.GzipFile``. ``maxminddb.types.SupportsRead`` describes this type. -* ``MODE_AUTO`` now accepts a binary file object. Previously, this could - raise ``TypeError``. +* ``MODE_AUTO`` now accepts a binary file object. Previously, it raised + ``TypeError``. It reads the file object into memory with the pure Python + reader, as ``MODE_FD`` does. The C extension needs a path, so lookups are + slower than with a path when the extension is installed. +* The pure Python reader raises ``TypeError`` when the database argument does + not suit the mode. 3.2.0 (2026-09-10) ++++++++++++++++++ diff --git a/README.rst b/README.rst index c546415b..e4df5544 100644 --- a/README.rst +++ b/README.rst @@ -49,9 +49,10 @@ second argument. The modes are available from ``maxminddb.Mode``. Valid modes ar * ``Mode.MEMORY`` - load database into memory. Pure Python. * ``Mode.FD`` - load database into memory from a binary file object. Pure Python. * ``Mode.AUTO`` - try ``Mode.MMAP_EXT``, ``Mode.MMAP``, ``Mode.FILE`` in that - order. A file object uses ``Mode.FD``. Default. + order. A file object is read into memory with the pure Python reader, as + with ``Mode.FD``. Pass a path to use the faster C extension. Default. -**NOTE**: When using ``Mode.FD``, it is the *caller's* responsibility to be +**NOTE**: When using a file object, it is the *caller's* responsibility to be sure that the file object gets closed properly. The caller may close the file object immediately after the ``Reader`` object is created. diff --git a/maxminddb/__init__.py b/maxminddb/__init__.py index d7542fe0..8b4491be 100644 --- a/maxminddb/__init__.py +++ b/maxminddb/__init__.py @@ -2,7 +2,6 @@ from __future__ import annotations -import os from importlib.metadata import version from typing import TYPE_CHECKING, cast @@ -16,7 +15,7 @@ Mode, ) from .errors import InvalidDatabaseError -from .reader import Reader +from .reader import _PATH_TYPES, Reader if TYPE_CHECKING: from .types import DatabaseSource @@ -63,7 +62,8 @@ def open_database( * MODE_FD - the param passed via database is a binary file object, not a path. This mode implies MODE_MEMORY. * MODE_AUTO - tries MODE_MMAP_EXT, MODE_MMAP, MODE_FILE in that - order. Uses MODE_FD for a file object. Default mode. + order. Reads a file object into memory, as MODE_FD + does. Default mode. """ if mode not in ( @@ -90,14 +90,19 @@ def open_database( # The cast pretends the C reader is the pure Python Reader, which has the # same API. if mode in (MODE_AUTO, MODE_MMAP_EXT) and has_extension: - if isinstance(database, (str, bytes, os.PathLike)): + if isinstance(database, _PATH_TYPES): return cast("Reader", _extension.Reader(database, mode)) - if mode == MODE_MMAP_EXT or isinstance(database, int): - msg = ( - f"The C extension requires a path ({type(database).__name__} " - "given). Use MODE_FD for a file object, or MODE_MMAP for a " - "file descriptor." - ) + # An object with __index__, such as an int, is a file descriptor. + # Leave a bool to the pure Python reader, which refuses it. + is_descriptor = not isinstance(database, bool) and hasattr( + type(database), "__index__" + ) + if mode == MODE_MMAP_EXT or is_descriptor: + msg = f"The C extension requires a path ({type(database).__name__} given)." + if is_descriptor: + msg += " Use MODE_MMAP for a file descriptor." + elif callable(getattr(database, "read", None)): + msg += " Use MODE_FD for a file object." raise TypeError(msg) return Reader(database, mode) diff --git a/maxminddb/decoder.py b/maxminddb/decoder.py index 9ef985ab..b0bb9412 100644 --- a/maxminddb/decoder.py +++ b/maxminddb/decoder.py @@ -65,7 +65,7 @@ class Decoder: def __init__( self, - database_buffer: FileBuffer | mmap.mmap | bytes, + database_buffer: FileBuffer | mmap.mmap | bytes | bytearray, pointer_base: int = 0, pointer_test: bool = False, # noqa: FBT001, FBT002 ) -> None: @@ -116,7 +116,7 @@ def _decode_bytes( size: int, offset: int, budget: _DecodeBudget, - ) -> tuple[bytes, int]: + ) -> tuple[bytes | bytearray, int]: # Charge the payload before copying so a crafted size cannot force a # large allocation, and so pointers reusing one target recharge. remaining = budget.payload_left - size diff --git a/maxminddb/file.py b/maxminddb/file.py index 1d7945ea..62ddb62c 100644 --- a/maxminddb/file.py +++ b/maxminddb/file.py @@ -3,18 +3,21 @@ from __future__ import annotations import os -from typing import overload +from typing import TYPE_CHECKING, overload try: from multiprocessing import Lock except ImportError: from threading import Lock # type: ignore[assignment] +if TYPE_CHECKING: + from maxminddb.types import StrOrBytesPath + class FileBuffer: """A slice-able file reader.""" - def __init__(self, database: str) -> None: + def __init__(self, database: StrOrBytesPath | int) -> None: """Create FileBuffer.""" self._handle = open(database, "rb") # noqa: SIM115 self._size = os.fstat(self._handle.fileno()).st_size diff --git a/maxminddb/reader.py b/maxminddb/reader.py index 2147454a..bc7abb55 100644 --- a/maxminddb/reader.py +++ b/maxminddb/reader.py @@ -8,7 +8,9 @@ mmap = None # type: ignore[assignment] import contextlib +import io import ipaddress +import operator import os from dataclasses import dataclass from ipaddress import IPv4Address, IPv4Network, IPv6Address, IPv6Network @@ -20,15 +22,24 @@ from maxminddb.file import FileBuffer if TYPE_CHECKING: - from collections.abc import Iterator + from collections.abc import Callable, Iterator from typing_extensions import Self - from maxminddb.types import DatabaseSource, Record, RecordDict + from maxminddb.types import ( + DatabaseSource, + Record, + RecordDict, + StrOrBytesPath, + ) _IPV4_MAX_NUM = 2**32 _REOPENED = "Attempt to iterate over a reopened MaxMind DB. Create a new iterator." _CLOSED = "Attempt to iterate over a closed MaxMind DB." +# The path types, which the C extension also accepts. +_PATH_TYPES = (str, bytes, os.PathLike) +# The database types that the path modes pass to open(). +_PATH_OR_FD_TYPES = (*_PATH_TYPES, int) class Reader: @@ -40,7 +51,7 @@ class Reader: _DATA_SECTION_SEPARATOR_SIZE = 16 _METADATA_START_MARKER = b"\xab\xcd\xefMaxMind.com" - _buffer: bytes | FileBuffer | "mmap.mmap" # noqa: UP037 + _buffer: bytes | bytearray | FileBuffer | "mmap.mmap" # noqa: UP037 _buffer_size: int # No database is open until __init__ succeeds. closed: bool = True @@ -72,8 +83,9 @@ def __init__( * MODE_MMAP - read from memory map. * MODE_FILE - read database as standard file. * MODE_MEMORY - load database into memory. - * MODE_AUTO - tries MODE_MMAP and then MODE_FILE. Uses - MODE_FD for a file object. Default. + * MODE_AUTO - tries MODE_MMAP and then MODE_FILE. Reads a + file object into memory, as MODE_FD does. + Default. * MODE_FD - the param passed via database is a binary file object, not a path. This mode implies MODE_MEMORY. @@ -349,48 +361,83 @@ def _load_buffer( self, database: DatabaseSource, mode: int = MODE_AUTO, - ) -> str: - # MODE_AUTO reads a file object as MODE_FD does. A path wins over - # read(), because some path objects also have a text read(). - if mode == MODE_AUTO and not isinstance( - database, (str, bytes, int, os.PathLike) - ): - mode = MODE_FD - filename: Any + ) -> object: + """Load the database and return a name for it in error messages.""" + if mode not in (MODE_AUTO, MODE_FD, MODE_FILE, MODE_MEMORY, MODE_MMAP): + msg = ( + f"Unsupported open mode ({mode}). Only MODE_AUTO, MODE_FILE, " + "MODE_MEMORY and MODE_FD are supported by the pure Python " + "Reader" + ) + raise ValueError( + msg, + ) + + # bool is an int, but it is never a file descriptor. + if mode != MODE_FD and not isinstance(database, bool): + # open() also takes an object with __index__, such as numpy.int64, + # as a file descriptor. + if not isinstance(database, _PATH_OR_FD_TYPES) and hasattr( + type(database), "__index__" + ): + database = operator.index(database) # type: ignore[arg-type] + # A path wins over read(), because some path objects also have a + # text read(). MODE_FD reads any object with read(). + if isinstance(database, _PATH_OR_FD_TYPES): + return self._load_path(database, mode) + # MODE_AUTO reads a file object into memory, as MODE_FD does. + read = getattr(database, "read", None) + if mode in (MODE_AUTO, MODE_FD) and callable(read): + return self._load_file_object(database, read) + if mode == MODE_AUTO: + hint = "Pass a path or a binary file object." + elif mode == MODE_FD: + hint = ( + "MODE_FD takes a binary file object. Use MODE_MMAP, MODE_FILE " + "or MODE_MEMORY for a path or a file descriptor." + ) + else: + hint = "Pass a path or a file descriptor. Use MODE_FD for a file object." + msg = f"Unsupported database type ({type(database).__name__}). {hint}" + raise TypeError(msg) + + def _load_path(self, database: StrOrBytesPath | int, mode: int) -> object: + """Memory-map or read a path or file descriptor.""" if (mode == MODE_AUTO and mmap) or mode == MODE_MMAP: - with open(database, "rb") as db_file: # type: ignore[arg-type] + with open(database, "rb") as db_file: self._buffer = mmap.mmap(db_file.fileno(), 0, access=mmap.ACCESS_READ) self._buffer_size = self._buffer.size() - filename = database elif mode in (MODE_AUTO, MODE_FILE): - self._buffer = FileBuffer(database) # type: ignore[arg-type] + self._buffer = FileBuffer(database) self._buffer_size = self._buffer.size() - filename = database - elif mode == MODE_MEMORY: - with open(database, "rb") as db_file: # type: ignore[arg-type] + else: + with open(database, "rb") as db_file: buf = db_file.read() self._buffer = buf self._buffer_size = len(buf) - filename = database - elif mode == MODE_FD: - self._buffer = database.read() # type: ignore[union-attr] - self._buffer_size = len(self._buffer) # type: ignore[arg-type] - # io buffers are not guaranteed to have a name attribute - if hasattr(database, "name"): - filename = database.name # type: ignore[union-attr] - else: - filename = f"<{type(database)}>" - else: + return database + + def _load_file_object(self, database: object, read: Callable[[], object]) -> object: + """Read a binary file object into memory.""" + # A text file can fail to decode inside read(), so check it first. + if isinstance(database, io.TextIOBase): msg = ( - f"Unsupported open mode ({mode}). Only MODE_AUTO, MODE_FILE, " - "MODE_MEMORY and MODE_FD are supported by the pure Python " - "Reader" - ) - raise ValueError( - msg, + f"The database file object is a text file " + f"({type(database).__name__}). Open it in binary mode." ) - - return filename + raise TypeError(msg) + buf = read() + if not isinstance(buf, (bytes, bytearray)): + msg = f"The database file object returned {type(buf).__name__}, not bytes." + if isinstance(buf, str): + msg += " Open it in binary mode." + raise TypeError(msg) + self._buffer = buf + self._buffer_size = len(buf) + # io buffers are not guaranteed to have a name attribute + if hasattr(database, "name"): + return database.name + return f"<{type(database)}>" def close(self) -> None: """Close the MaxMind DB file and returns the resources to the system. diff --git a/tests/reader_test.py b/tests/reader_test.py index 5d92e889..3e3b50f5 100644 --- a/tests/reader_test.py +++ b/tests/reader_test.py @@ -3,9 +3,9 @@ import contextlib import dataclasses import gc +import gzip import io import ipaddress -import mmap import multiprocessing import os import pathlib @@ -1820,20 +1820,6 @@ def load_and_read(new: Reader, database: str, mode: int) -> object: self.assertEqual(records_during_load, [{"ip": "1.1.1.1"}]) self.assertEqual(reader.metadata().database_type, "MaxMind DB Decoder Test") - def test_reinitialize_from_a_source_that_returns_the_same_mmap(self) -> None: - with open(_DECODER_DB, "rb") as database: - buffer = mmap.mmap(database.fileno(), 0, access=mmap.ACCESS_READ) - self.addCleanup(buffer.close) - - class Source: - def read(self) -> mmap.mmap: - return buffer - - reader = maxminddb.reader.Reader(Source(), MODE_FD) # type: ignore[arg-type] - # A reinit must not close the buffer that it then uses. - reader.__init__(Source(), MODE_FD) # type: ignore[misc] - self.assertIsNotNone(reader.get("::1.1.1.0")) - def test_reinitialize_from_the_same_source(self) -> None: ipv4 = pathlib.Path(f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb") ipv6 = pathlib.Path(f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv6-24.mmdb") @@ -1877,6 +1863,45 @@ def test_metadata_types_match_metadata_fields(self) -> None: [field.name for field in dataclasses.fields(maxminddb.reader.Metadata)], ) + def test_auto_mode_reads_file_object_from_its_position(self) -> None: + data = pathlib.Path( + f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb", + ).read_bytes() + with tempfile.TemporaryFile() as file_object: + file_object.write(b"header" + data) + file_object.seek(len(b"header")) + with maxminddb.open_database(file_object, MODE_AUTO) as reader: + self.assertEqual(reader.get("1.1.1.1"), {"ip": "1.1.1.1"}) + + def test_auto_mode_reads_wrapped_and_piped_streams(self) -> None: + path = f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb" + data = pathlib.Path(path).read_bytes() + + with tempfile.TemporaryDirectory() as directory: + gz_path = pathlib.Path(directory) / "db.mmdb.gz" + with gzip.open(gz_path, "wb") as compressed: + compressed.write(data) + with open(gz_path, "rb") as backing: + stream = io.BufferedReader(gzip.GzipFile(fileobj=backing)) + with maxminddb.open_database(stream, MODE_AUTO) as reader: + self.assertEqual(reader.get("1.1.1.1"), {"ip": "1.1.1.1"}) + # The reader does not close the caller's stream. + self.assertFalse(stream.closed) + stream.close() + + # The fixture is far smaller than the pipe buffer, so the write to the + # pipe does not block. + read_fd, write_fd = os.pipe() + os.write(write_fd, data) + os.close(write_fd) + pipe = os.fdopen(read_fd, "rb") + try: + with maxminddb.open_database(pipe, MODE_AUTO) as reader: + self.assertEqual(reader.get("1.1.1.1"), {"ip": "1.1.1.1"}) + self.assertFalse(pipe.closed) + finally: + pipe.close() + def test_auto_mode_accepts_any_database_type(self) -> None: path = f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb" data = pathlib.Path(path).read_bytes() @@ -1915,6 +1940,105 @@ def read(self) -> str: with self.assertRaisesRegex(TypeError, r"\(int given\)"): maxminddb.open_database(descriptor, MODE_AUTO) + def test_database_type_must_match_mode(self) -> None: + path = f"{_TEST_DATA_DIR}/MaxMind-DB-test-decoder.mmdb" + reader_class = maxminddb.reader.Reader + with self.assertRaisesRegex(TypeError, r"\(str\)\. MODE_FD takes"): + reader_class(path, MODE_FD) + with self.assertRaisesRegex(TypeError, r"\(int\)\. MODE_FD takes"): + reader_class(3, MODE_FD) + with open(path, "rb") as database: + for mode in (MODE_FILE, MODE_MEMORY, MODE_MMAP): + with ( + self.subTest(mode=mode), + self.assertRaisesRegex(TypeError, "Use MODE_FD"), + ): + reader_class(database, mode) + with self.assertRaisesRegex(ValueError, "Unsupported open mode"): + reader_class(database, MODE_MMAP_EXT) + + for bad in (None, object()): + with self.assertRaisesRegex(TypeError, "Unsupported database type"): + reader_class(bad, MODE_AUTO) # type: ignore[arg-type] + + for mode in (MODE_AUTO, MODE_FD): + with ( + self.subTest(mode=mode), + # The check comes before read(), which fails to decode. + open(path, encoding="utf-8") as text_file, + self.assertRaisesRegex(TypeError, "binary mode"), + ): + reader_class(text_file, mode) # type: ignore[arg-type] + + # A bool is an int, but not a file descriptor. + with self.assertRaisesRegex(TypeError, r"\(bool\)"): + reader_class(False, MODE_AUTO) # noqa: FBT003 + with self.assertRaisesRegex(TypeError, r"\(bool\)\. Pass a path or a file"): + reader_class(False, MODE_FILE) # noqa: FBT003 + + def test_path_modes_accept_a_descriptor_with_index(self) -> None: + class Descriptor: + """A file descriptor object, as numpy.int64 is.""" + + def __init__(self, fd: int) -> None: + self.fd = fd + + def __index__(self) -> int: + return self.fd + + path = f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb" + # The reader takes ownership of the descriptor and closes it. + descriptor = Descriptor(os.open(path, os.O_RDONLY)) + with maxminddb.reader.Reader(descriptor, MODE_MMAP) as reader: # type: ignore[arg-type] + self.assertEqual(reader.get("1.1.1.1"), {"ip": "1.1.1.1"}) + + if has_maxminddb_extension(): + # With the extension, MODE_AUTO refuses it, as it refuses an int. + fd = os.open(path, os.O_RDONLY) + self.addCleanup(os.close, fd) + with self.assertRaisesRegex(TypeError, r"\(Descriptor given\)"): + maxminddb.open_database(Descriptor(fd)) # type: ignore[arg-type] + # A bool gets the bool error, not advice to use MODE_MMAP. + with self.assertRaisesRegex(TypeError, r"\(bool\)\. Pass a path"): + maxminddb.open_database(False) # noqa: FBT003 + if has_maxminddb_extension(): + with self.assertRaisesRegex(TypeError, r"\(bool given\)\.$"): + maxminddb.open_database(False, MODE_MMAP_EXT) # noqa: FBT003 + + def test_fd_mode_reads_any_binary_reader(self) -> None: + data = pathlib.Path( + f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb", + ).read_bytes() + + class FileWithPath(io.BytesIO): + def __fspath__(self) -> str: + return "does-not-exist.mmdb" + + class BytearrayReader: + def read(self) -> bytes: + return bytearray(data) # type: ignore[return-value] + + for database in (FileWithPath(data), BytearrayReader()): + with ( + self.subTest(type(database).__name__), + maxminddb.reader.Reader(database, MODE_FD) as reader, + ): + self.assertEqual(reader.get("1.1.1.1"), {"ip": "1.1.1.1"}) + + class NoneReader: + def read(self) -> None: + return None + + with self.assertRaisesRegex(TypeError, r"returned NoneType, not bytes\.$"): + maxminddb.reader.Reader(NoneReader(), MODE_FD) # type: ignore[arg-type] + + class StrReader: + def read(self) -> str: + return "text" + + with self.assertRaisesRegex(TypeError, r"returned str, not bytes\. Open"): + maxminddb.reader.Reader(StrReader(), MODE_FD) # type: ignore[arg-type] + def test_empty_search_tree_is_accepted(self) -> None: data = pathlib.Path( f"{_TEST_DATA_DIR}/MaxMind-DB-test-ipv4-24.mmdb" From 91d0880e804688f4240d09f38b21914dff887df3 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Sat, 3 Oct 2026 04:29:45 +0000 Subject: [PATCH 4/5] List MODE_MMAP in the unsupported mode error The pure Python Reader accepts MODE_MMAP, but the error for an unsupported mode listed only MODE_AUTO, MODE_FILE, MODE_MEMORY and MODE_FD. A caller who passed MODE_MMAP_EXT was told that MODE_MMAP was not supported. Co-Authored-By: Claude Opus 5.5 --- maxminddb/reader.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/maxminddb/reader.py b/maxminddb/reader.py index bc7abb55..00604d59 100644 --- a/maxminddb/reader.py +++ b/maxminddb/reader.py @@ -365,9 +365,9 @@ def _load_buffer( """Load the database and return a name for it in error messages.""" if mode not in (MODE_AUTO, MODE_FD, MODE_FILE, MODE_MEMORY, MODE_MMAP): msg = ( - f"Unsupported open mode ({mode}). Only MODE_AUTO, MODE_FILE, " - "MODE_MEMORY and MODE_FD are supported by the pure Python " - "Reader" + f"Unsupported open mode ({mode}). Only MODE_AUTO, MODE_MMAP, " + "MODE_FILE, MODE_MEMORY and MODE_FD are supported by the pure " + "Python Reader" ) raise ValueError( msg, From bc8b99932ab85347bb6a8821d20a8356129e0a85 Mon Sep 17 00:00:00 2001 From: Gregory Oschwald Date: Sat, 3 Oct 2026 11:17:20 +0000 Subject: [PATCH 5/5] Name the file object type plainly in error messages The name for a file object without a name attribute was f"<{type(database)}>", so error messages showed "<>". MODE_AUTO now reads file objects too, so these messages are more common. Use the type name, as in "". Co-Authored-By: Claude Opus 5.5 --- maxminddb/reader.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/maxminddb/reader.py b/maxminddb/reader.py index 00604d59..f8a39397 100644 --- a/maxminddb/reader.py +++ b/maxminddb/reader.py @@ -437,7 +437,7 @@ def _load_file_object(self, database: object, read: Callable[[], object]) -> obj # io buffers are not guaranteed to have a name attribute if hasattr(database, "name"): return database.name - return f"<{type(database)}>" + return f"<{type(database).__name__}>" def close(self) -> None: """Close the MaxMind DB file and returns the resources to the system.