From cf1def2ffa68f80007bf8ae6028d811dd6e88e15 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sat, 3 Oct 2026 13:40:43 +0200 Subject: [PATCH 01/35] Recognize Caterva2 dataset URLs in the public opener --- plans/caterva2-access-improvements.md | 416 ++++++++++++++++++++++++++ src/blosc2/caterva2_url.py | 68 +++++ src/blosc2/schunk.py | 29 ++ tests/test_caterva2_access.py | 77 +++++ 4 files changed, 590 insertions(+) create mode 100644 plans/caterva2-access-improvements.md create mode 100644 src/blosc2/caterva2_url.py create mode 100644 tests/test_caterva2_access.py diff --git a/plans/caterva2-access-improvements.md b/plans/caterva2-access-improvements.md new file mode 100644 index 000000000..4b7e24486 --- /dev/null +++ b/plans/caterva2-access-improvements.md @@ -0,0 +1,416 @@ +# Caterva2 access improvements for Python-Blosc2 and b2view + +Status: implementation proposal; no implementation implied by this document. + +Implementation record (updated as milestones land): + +- M0/M1: selected `remote_service="auto" | "caterva2" | "fsspec"` and + implemented root-marker recognition before format dispatch. String service + URLs default to lazy access; explicit URLPath defaults remain unchanged. + Inspection confirms `open` currently defaults to `mode="r"` (the proposal's + concern about changing that default requires no code change). +- Repository design: use a public RemoteRepository subclass of RemoteStore as + a browsing facade with independently owned roots and per-root cache budgets. + Repository persistence is explicitly unsupported initially. Single-root + servers flatten to their actual root object; empty/multi-root servers return + the repository facade. Root owners stay alive through normal returned child + handles after the facade closes. +- Roots discovery uses the mapping response implemented by cat2lite and + Caterva2, not an assumed list. No global discovery cache is introduced. +- M1 validation: 20 offline URL/dispatch tests passed in the blosc2 environment. + +## 1. Goal + +Make Caterva2-compatible servers, including cat2lite, first-class URL-addressable +sources in Python-Blosc2. Users should be able to browse a curated remote +repository through a shared caching gateway with the same viewer and opening +API used for individual files. + +Target usage: + +```python +import blosc2 + +repository = blosc2.open("http://localhost:8000", mode="r") +public = blosc2.open("http://localhost:8000/@public", mode="r") +group = blosc2.open("http://localhost:8000/@public/hdf5", mode="r") +array = blosc2.open("http://localhost:8000/@public/hdf5/d0/d1/a2", mode="r") +``` + +```sh +b2view http://localhost:8000 +b2view http://localhost:8000/@public +b2view https://cat2.cloud/demo/@public/example +``` + +The examples above describe proposed behavior. Preserve existing direct-file +opening, including HTTP B2ND/B2Z, HDF5, Zarr, and Parquet sources. + +Architecture: + +```text +b2view / Python applications + | + | Caterva2 metadata, listings, bounded slices + v +cat2lite + shared server-side disk cache + | + v +remote repositories exposed through .catl mounts +``` + +Do not create a separate cat2lite-view application. Source recognition belongs +in Python-Blosc2; the existing b2view UI should consume the resulting objects. + +## 2. Current implementation and gaps + +Relevant code to inspect before implementation: + +- `src/blosc2/schunk.py`: `open`, `_open_c2_urlpath`, remote option validation, + lazy dispatch, and existing format-specific openers. +- `src/blosc2/c2array.py`: `URLPath` integration, API URL construction, shared + HTTP clients, authentication, info/list/fetch helpers. +- `src/blosc2/remote_store.py`: Caterva2 source descriptors, `_open_caterva2`, + `resolve`, `kind`, `list_children`, owner/cache lifecycle, and table fetching. +- `src/blosc2/b2view/model.py`: `StoreBrowser._open_store`, tree traversal, + remote leaf ownership, previews, and table capabilities. +- `src/blosc2/b2view/app.py`: background opening, tree expansion, failure display. +- `src/blosc2/b2view/cli.py`: source arguments, cache options, initial path. + +Existing foundations: + +- Explicit `blosc2.URLPath(path, urlbase=...)` already selects Caterva2 handling. +- The lazy Caterva2 opener discovers groups, arrays, and CTables; groups can + return `RemoteStore`, arrays use existing C2Array/RemoteArray semantics, and + tables can return `RemoteCTable`. +- b2view already displays RemoteStore hierarchies and remote leaves. +- cat2lite already implements roots, info, list, and bounded data access. + +Gaps: + +1. A plain HTTP string is not converted into a Caterva2 URLPath. +2. Bare server URLs are treated as data files, without roots discovery. +3. `_open_caterva2` currently assumes an initial recursive listing describes the + hierarchy and initializes discovered groups' child lists from that snapshot. + cat2lite catalog listings deliberately stop at mounted-source boundaries. + Thus a discovered mounted group can appear empty instead of being expanded + through another API list request. +4. b2view has format-specific URL dispatch that must not override the new + Caterva2 interpretation when a published dataset name ends in `.h5`, etc. +5. A server with multiple roots has no agreed repository-level browsing object. + +## 3. URL and opening contract + +### 3.1 Explicit dataset/group shorthand + +Recognize an HTTP(S) URL containing an `@`-prefixed path component as Caterva2 +shorthand, consistently with cat2lite's `.catl` convention. + +Example: + +```text +https://host/demo/@public/group/array + urlbase = https://host/demo + path = @public/group/array +``` + +Rules: + +- Parse with a real URL parser, preserving bracketed IPv6 and port numbers. +- Inspect path components only; ignore `@` in hostname, userinfo, or query. +- Use the first root-marker component; later `@` components are dataset names. +- Preserve the encoded deployment prefix; decode logical dataset components + exactly once and use existing Caterva2 path validation/endpoint encoding. +- Reject empty root markers, traversal, encoded separators, malformed encoding, + control characters, and unsupported query/fragment/selector combinations. +- Do not reinterpret fsspec `::/member` selectors as Caterva2 dataset selectors. +- Authentication uses existing Caterva2 facilities, not URL userinfo. +- Explicit URLPath inputs retain precedence and behavior. +- Return the object's actual type; do not wrap every dataset in RemoteStore. +- Preserve existing `mode`, `offset`, `lazy`, and cache-option validation. + Remote sources remain read-only. Specify `mode="r"` in documentation rather + than changing the global default mode as part of this work. +- `lazy=False` must retain its existing meaning or explicitly reject group/ + repository opening; do not silently ignore it to force browser semantics. + +### 3.2 Ordinary HTTP override + +An ordinary file server can legitimately contain `@` path components. Provide +an explicit bypass for Caterva2 recognition and repository probing. + +Proposed API decision for M0: a narrowly scoped selector such as +`remote_service="auto" | "caterva2" | "fsspec"` on `blosc2.open`. +This name is provisional. Audit existing options before adding it; do not +overload `source_format`, which currently describes data formats. + +- `auto`: root-marker shorthand, known-file dispatch, then eligible base probe. +- `caterva2`: explicitly interpret a URL as a repository base or dataset URL. +- `fsspec`: use the ordinary fsspec-backed source handling, even with `@` + components, bypassing Caterva2 recognition and probing. This identifies the + access backend, not a local file or a single-file-only source; remote + containers and stores remain supported. +- Explicit URLPath plus a conflicting override is an error. +- Reject this option for inputs where it has no meaning rather than ignoring it. +- If needed, expose the same choice in b2view through a small CLI flag. + +### 3.3 Bare server and deployment-base URLs + +For an eligible ambiguous HTTP(S) URL, probe `/api/roots` with a bounded +timeout and validate the response against the actual Caterva2 roots contract. +Inspect both Caterva2 and cat2lite responses during M0; do not assume a list +of root names if the API returns a mapping with metadata. + +Recommended dispatch order: + +1. Explicit URLPath or explicit service override. +2. Root-marker URL shorthand. +3. Known file/container URL, including selectors and trailing-slash Zarr. +4. Discovery for ambiguous base candidates, including `/demo` prefixes. +5. Existing generic remote-file path if discovery conclusively says this is + not a Caterva2-compatible server. + +Discovery must not add roots requests to recognized direct-file reads or turn +an existing permission/format error into an unrelated service-detection error. + +Probe handling: + +- Valid roots response: bind the server base and authentication context. +- Missing endpoint, HTML, malformed JSON, or invalid schema: classify as + non-Caterva2 in auto mode; provide a clear discovery error in explicit mode. +- 401/403: report authentication/authorization failure; no anonymous fallback. +- Connection/TLS failure or timeout: report an actionable connection error, + preserving the original cause; avoid serial retry/fallback delays. +- Server errors: preserve the server error rather than saying “not Caterva2”. +- Preserve deployment prefixes and follow existing redirect policy. Never + forward authentication credentials to an unrelated redirect origin. +- Close response resources, reuse the existing transport, and avoid duplicate + roots/info requests when passing discovery results into an owner. +- Do not introduce a process-global negative discovery cache initially. + Any later positive cache must include authentication identity and lifetime. + +## 4. Repository root behavior + +Recommended user-facing behavior: + +- Exactly one root: open it directly, so cat2lite's `@public` server feels like + a directly opened hierarchy. +- Multiple roots: expose a synthetic repository group with roots as children. +- Zero visible roots: return an empty repository group, with b2view showing + “No accessible roots”; this is not necessarily a server error. +- An explicit `/@public` URL always opens that root, regardless of other roots. + +Implementation decision for M0: prefer a repository-backed RemoteStore mode +over a viewer-only wrapper. Confirm it fits ownership and persistence semantics +before choosing a new public class. All applications should get the same +`keys`, lookup, `kind`, attrs, context-manager, and close behavior. + +The synthetic group: + +- Is not sent to `/api/info` as a fabricated empty dataset path. +- Discovers each root only when accessed; it does not open all roots at startup. +- Isolates inaccessible/broken roots without hiding healthy siblings. +- Has its own identity distinct from a dataset-root descriptor. +- Keeps independent root owners under a repository lifetime; closing the + repository must follow existing live-child ownership conventions. +- Makes cache allowance scope explicit. Prefer a repository-wide bound where + the owner architecture supports it; otherwise document per-root allowances + before shipping rather than implying a total bound. +- Either defines safe persistence/reopening explicitly or rejects repository + persistence clearly in the first implementation. In-memory browsing must + not accidentally write an invalid existing dataset descriptor. + +Single-root flattening means relative paths can change if server roots change +between opens. Document this and recommend explicit root URLs for scripts. + +## 5. Lazy Caterva2 hierarchy discovery + +Refactor Caterva2 discovery into separate root metadata, node metadata, and +group-listing operations. Track whether a group has actually been listed; +“known group with no discovered children” is not “known empty group”. + +Required behavior: + +1. Open a known root with bounded metadata work, without recursive remote + source expansion. +2. Expand a group through `/api/list/` only when needed. +3. Accept existing recursive relative-path lists and catalog lists that stop + at mounts. Synthesize structural parent groups where necessary. +4. Mark only the queried group's listing as complete. A returned group may + need its own request, even if no descendants were returned initially. +5. Fetch and memoize metadata needed to classify actual children. Do not + assume every listed name is an array or eagerly inspect every deep leaf. +6. Resolve a directly requested descendant through info requests even when its + parents have not been expanded. Update the same node registry afterward. +7. Keep listings deterministic and deduplicated; empty groups remain visible. +8. Bound node growth and discovery work; reject malformed or escaping paths. +9. Coalesce concurrent discovery for the same group and publish registry + updates atomically. Failed discovery must remain retryable. +10. Preserve attrs and separate `catalog_attrs` annotations without overwriting + source attrs. Decide how annotations are displayed in b2view metadata. + +There is a protocol constraint: an existing server may return a large recursive +list in a single response. This client change cannot promise paginated or +constant-size listings without a server API extension. Avoid eagerly fetching +info for every returned descendant; measure remaining list costs separately. + +## 6. Cache and data-access semantics + +- A cat2lite server cache is shared across its HTTP clients. Python-Blosc2's + client cache is a separate layer, with independently configured policies. +- Reuse existing none/memory/disk options; do not invent a special server-cache + policy or reinterpret `shared_cache=True` as “use cat2lite”. That existing + option has its own local-cache semantics and locking requirements. +- Normalize shorthand strings and equivalent URLPath sources to the same + dataset cache identity, including base prefix, path, and authentication scope. +- Repository synthetic identities must not collide with dataset identities. +- Do not persist credentials or allow a cached authenticated view to leak into + a different identity. Follow the current authenticated-cache restrictions. +- Preserve exclusive disk-owner locking, bounded retention, restart behavior, + and reference-counted leaf lifetimes. +- Browsing uses metadata only; previews fetch bounded slices/pages rather than + downloading whole containers. An upstream backend may still prefetch source + bytes according to its own semantics; distinguish this from client downloads. +- Preserve immutable-source assumptions and explicit invalidation requirements. + No mutable-source TTL or automatic refresh protocol is introduced here. +- Surface server slice-limit errors with guidance to request a smaller preview. +- Do not imply all server-side table operations are available: cat2lite's + unsupported filters/indices must not be silently ignored or trigger an + unbounded local materialization. + +## 7. b2view integration + +- Route service recognition through shared Python-Blosc2 opening helpers. +- Retain special direct-format handling only where needed for existing array/ + table/group ownership behavior. Caterva2 interpretation takes precedence over + extensions in published dataset names. +- Run discovery, group expansion, and previews in background work, keeping the + UI responsive during slow upstream metadata reads. +- Show loading, empty-group, and retryable error states distinctly. +- Support initial paths within a root or synthetic repository. +- Check remote arrays, table paging/projection, scalar and multidimensional + previews, attrs, and plotting through existing data adapters. +- Gate unsupported table actions consistently; do not offer operations that + require full remote downloads merely because a local CTable supports them. +- Cancellation/closing the UI must release owners and transports cleanly. +- Keep CLI cache options controlling the client cache, and label this clearly + in help. The user configures the shared server cache on cat2lite separately. + +## 8. Implementation milestones + +### M0 — Freeze contract and establish fixtures + +- Audit roots schemas, opener defaults/lazy behavior, authentication, and + existing descriptor/persistence constraints. +- Finalize the explicit HTTP override and multi-root object design. +- Add deterministic local HTTP fixtures for both conventional recursive + Caterva2 listings and catalog mount-boundary listings. +- Record request counters and configurable failures/delays in those fixtures. + +Gate: a documented dispatch/return-type matrix and fixtures representing both +protocol profiles; no network dependency for ordinary tests. + +### M1 — Root-marker URL shorthand + +- Implement reusable normalization and integrate it before file-format dispatch. +- Preserve all existing URLPath and option semantics. +- Add override support and documentation for literal `@` file URLs. + +Gate: equivalent strings and URLPath objects open the same groups/arrays/tables, +including deployment prefixes and IPv6, without roots probing. + +### M2 — Lazy mount-aware RemoteStore traversal + +- Split metadata discovery from listing completion. +- Implement direct descendant lookup, lazy expansion, memoization, and bounded + concurrent discovery. +- Cover nested mounts and conventional recursively listed stores. + +Gate: a group omitted below the initial mount boundary can be expanded and +read; healthy siblings remain usable after a failed expansion. + +### M3 — Base discovery and repository roots + +- Add bounded roots discovery and transport/error handling. +- Implement single-root, empty-root, and multiple-root behavior. +- Define repository cache and lifecycle rules; cover auth-separated identities. + +Gate: bare and prefixed server URLs browse correctly, while direct-file sources +retain their dispatch and do not acquire extra roots requests. + +### M4 — b2view integration + +- Connect the shared opener, lazy tree expansion, and repository root object. +- Update title/source display, capabilities, errors, and CLI help. +- Add headless Textual tests against local servers. + +Gate: bare-server and explicit-root command forms browse groups and display +bounded array/table previews without blocking the event loop. + +### M5 — Cross-project acceptance, documentation, and measurements + +- Run against a real debug cat2lite server serving a mixed `.catl` fixture. +- Verify shared server-cache reuse across two independent clients and after + restart where the backend guarantees persistent reuse. +- Update API docs, b2view guide, examples, and release notes. +- Record startup/expansion request counts and representative cache evidence. + +Gate: all focused tests and relevant offline regressions pass; the use cases in +section 1 work with documented limits and no new viewer executable. + +## 9. Regression and acceptance matrix + +### URL dispatch + +- HTTP/HTTPS, IPv4/IPv6, ports, trailing slash, deployment prefixes. +- Root-marker group, array, table, and extension-bearing published names. +- Multiple `@` components, encoded spaces, Unicode, encoded `@` root markers. +- Query/userinfo/fragment ambiguity, traversal and double-decoding attempts. +- Ordinary direct-file `@` paths via override; signed direct-file URLs. +- Existing fsspec selectors, explicit URLPath, local paths, and non-HTTP inputs. +- Cache-option forwarding and rejection of contradictory inputs. + +### Discovery and hierarchy + +- Valid zero/one/multiple roots; unexpected JSON/HTML; 404/401/403/5xx; timeout. +- Prefix-preserving redirects and credential behavior. +- Recursive listings, mount-boundary listings, empty groups, duplicate entries. +- Direct access before parent expansion, repeated and concurrent expansion. +- Node limits and per-group failures with subsequent successful retry. +- No eager contact with unrelated remote mounts merely to open a repository. + +### Data and caching + +- Scalar/ND arrays, CTable row windows and projections, attrs/annotations. +- No implicit full fetch for previews; correct handling of server byte limits. +- none/memory/disk, identity equivalence, auth separation, close/reopen, restart. +- Two processes with independent client caches reuse one server-side cache. + Use upstream byte/request counters and fixtures larger than backend prefetch + thresholds; do not confuse client-cache hits with server-cache hits. +- Concurrent owner lifetimes and expected exclusive-lock failures where local + disk caches are deliberately shared without the supported sharing mode. + +### Viewer and compatibility + +- Headless tree expansion, starting path, empty/error states, preview paging, + cancellation, and clean shutdown under warnings-as-errors. +- Existing direct HDF5/Zarr/B2Z/Parquet viewing remains functional. +- Explicit URLPath callers retain return types and lazy=False behavior. +- Existing Caterva2 and remote-store tests remain green. + +Run Python/build commands in the `blosc2` conda environment. Use targeted pytest +runs during implementation, then the relevant offline opener/remote-store/ +remote-array/remote-table/b2view suites, Ruff on touched files, and the full +offline suite for final validation. Follow actual test filenames discovered in +the repository rather than assuming names in this plan are commands. + +## 10. Completion criteria and boundaries + +Complete when all proposed CLI forms work, equivalent Python opening works, +catalog-mounted hierarchies expand lazily, multi-root behavior is documented, +and cache reuse is demonstrated with measured upstream traffic. + +This work does not add a new viewer, a new remote protocol, write access, +catalog parsing in Python-Blosc2, automatic cache invalidation, a server +deployment/authentication system, or a guarantee that arbitrary remote formats +support cheap random access. It reuses Caterva2's protocol and existing object +adapters to make the shared-cache-server workflow convenient and predictable. diff --git a/src/blosc2/caterva2_url.py b/src/blosc2/caterva2_url.py new file mode 100644 index 000000000..fcaadc793 --- /dev/null +++ b/src/blosc2/caterva2_url.py @@ -0,0 +1,68 @@ +"""Recognition of Caterva2 service URLs, separate from source format detection.""" + +from urllib.parse import unquote, urlsplit, urlunsplit + +import blosc2 + + +def validate_service_url(value): + """Validate a service URL without silently normalizing unsafe components.""" + if not isinstance(value, str) or not value.startswith(("http://", "https://")): + raise ValueError("Caterva2 service URLs require http:// or https://") + if any(char.isspace() or ord(char) < 32 or ord(char) == 127 for char in value) or "\\" in value: + raise ValueError("Unsafe Caterva2 service URL") + parsed = urlsplit(value) + if ( + not parsed.netloc + or not parsed.hostname + or parsed.username is not None + or parsed.password is not None + ): + raise ValueError("Caterva2 service URL requires an authority without userinfo") + _ = parsed.port + if parsed.query or parsed.fragment or "::" in parsed.path: + raise ValueError("Caterva2 service URLs do not support queries, fragments, or selectors") + for component in parsed.path.split("/"): + decoded = decode_service_component(component) + if ( + decoded in {".", ".."} + or any(c in decoded for c in "/\\%?#") + or any(ord(c) < 32 or ord(c) == 127 for c in decoded) + ): + raise ValueError("Unsafe Caterva2 URL path component") + return parsed + + +def decode_service_component(component): + """Decode exactly once, rejecting malformed escapes and invalid UTF-8.""" + import re + + if re.search(r"%(?![0-9a-fA-F]{2})", component): + raise ValueError("Malformed URL escape") + return unquote(component, errors="strict") + + +def caterva2_urlpath(value): + """Return a URLPath for an @-root service URL, or None for an ordinary URL. + + This function never makes a network request. Use an explicit fsspec override + when an ordinary HTTP source happens to have an @-prefixed path component. + """ + if not isinstance(value, str) or not value.startswith(("http://", "https://")): + return None + parsed = urlsplit(value) + components = parsed.path.split("/") + marker = next( + (i for i, component in enumerate(components) if decode_service_component(component).startswith("@")), + None, + ) + if marker is None: + return None + validate_service_url(value) + logical = [decode_service_component(component) for component in components[marker:]] + if logical[-1] == "": + logical.pop() + if not logical or logical[0] == "@" or any(not component for component in logical): + raise ValueError("Caterva2 URL requires a root and nonempty dataset components") + base = urlunsplit((parsed.scheme, parsed.netloc, "/".join(components[:marker]), "", "")) + return blosc2.URLPath("/".join(logical), urlbase=base) diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index e241bd6ea..78f98a1af 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2787,6 +2787,7 @@ def open( # noqa: C901 path: str | None = None, shared_cache: bool = False, deserialize: str = "safe", + remote_service: str = "auto", **kwargs: dict, ) -> ( blosc2.SChunk @@ -2959,6 +2960,12 @@ def open( # noqa: C901 a ``.b2z`` path selects B2Z automatically. An explicit value supports suffix-free paths. Zarr, HDF5, B2Z and Parquet sources automatically enable ``lazy=True``. + remote_service: {"auto", "caterva2", "fsspec"}, optional + Select remote access independently of the source format. In auto mode, + HTTP(S) URLs with an @-prefixed root component are Caterva2 dataset + references and default to lazy access. Use ``"fsspec"`` to open an + ordinary remote source containing such a component without service + recognition. Explicit :ref:`URLPath` inputs retain their lazy defaults. parquet_options: dict, optional PyArrow ``ParquetFile`` reader options for a Parquet source. Conversion options such as ``columns`` and ``max_rows`` are passed separately. @@ -3064,6 +3071,28 @@ def open( # noqa: C901 """ deserialize = normalize_deserialize(deserialize) dataset = blosc2.core.resolve_dataset_path(dataset, path) + if remote_service not in {"auto", "caterva2", "fsspec"}: + raise ValueError("remote_service must be 'auto', 'caterva2', or 'fsspec'") + if isinstance(urlpath, blosc2.URLPath): + if remote_service == "fsspec": + raise ValueError("remote_service='fsspec' conflicts with URLPath") + else: + from blosc2.caterva2_url import caterva2_urlpath + + service_path = None if remote_service == "fsspec" else caterva2_urlpath(urlpath) + if service_path is not None: + urlpath = service_path + kwargs.setdefault("lazy", True) + elif remote_service == "caterva2": + raise ValueError("A Caterva2 dataset URL requires an @-prefixed root") + elif remote_service == "fsspec" and not is_fsspec_url(urlpath): + raise ValueError("remote_service='fsspec' requires a remote URL") + if isinstance(urlpath, blosc2.URLPath): + if dataset is not None or hdf5_index is not None: + raise ValueError("dataset/path and hdf5_index are unsupported for Caterva2 sources") + _reject_table_buffer_options(kwargs) + _validate_shared_cache_request(urlpath, shared_cache, kwargs) + return _open_c2_urlpath(urlpath, mode, offset, kwargs) if kwargs.get("source_format") == "parquet" or ( isinstance(urlpath, (str, pathlib.Path)) and str(urlpath).split("?", 1)[0].lower().endswith(".parquet") diff --git a/tests/test_caterva2_access.py b/tests/test_caterva2_access.py new file mode 100644 index 000000000..582c65184 --- /dev/null +++ b/tests/test_caterva2_access.py @@ -0,0 +1,77 @@ +"""Service URL recognition and repository discovery (offline loopback fixtures).""" + +import pytest +from test_remote_caterva2 import caterva2_source # noqa: F401 + +import blosc2 +from blosc2.caterva2_url import caterva2_urlpath + + +@pytest.mark.parametrize( + ("url", "base", "path"), + [ + ("http://localhost:8000/@public", "http://localhost:8000", "@public"), + ("http://[::1]:8000/demo/@public/a/", "http://[::1]:8000/demo", "@public/a"), + ( + "https://host/deploy%20here/%40public/my%20array/@nested", + "https://host/deploy%20here", + "@public/my array/@nested", + ), + ], +) +def test_root_marker(url, base, path): + target = caterva2_urlpath(url) + assert target.urlbase == base + assert target.path == path + + +@pytest.mark.parametrize( + "url", ["https://host/file.h5", "https://host/data?x=@public", "s3://bucket/@public/a"] +) +def test_ordinary_url(url): + assert caterva2_urlpath(url) is None + + +@pytest.mark.parametrize( + "url", + [ + "https://host/@", + "https://host/@public//a", + "https://host/../@public", + "https://host/@public/%2F", + "https://host/@public/%252F", + "https://host/@public/a%GG", + "https://host/@public/%FF", + "https://host/@public/a?x=1", + "https://host/@public/a#x", + "https://user:pass@host/@public", + "https://host/@public/a::/member", + "https://host/@public/%00", + ], +) +def test_unsafe_shorthand(url): + with pytest.raises((ValueError, UnicodeError)): + caterva2_urlpath(url) + + +def test_shorthand_open(caterva2_source): # noqa: F811 + base, array, _, _ = caterva2_source + with blosc2.open(base + "@public/group") as group: + assert isinstance(group, blosc2.RemoteStore) + assert group.keys() == ["array", "table"] + with group["array"] as remote: + assert (remote[1:2, 2:4] == array[1:2, 2:4]).all() + with blosc2.open(base + "@public/table") as table: + assert isinstance(table, blosc2.RemoteCTable) + assert table.slice(0, 2).nrows == 2 + assert isinstance(blosc2.open(base + "@public/group/array", lazy=False), blosc2.C2Array) + + +def test_override_and_option_validation(monkeypatch): + monkeypatch.setattr(blosc2.schunk, "_open_fsspec_url", lambda *args: "ordinary-file") + assert blosc2.open("https://host/@public/data.b2nd", remote_service="fsspec") == "ordinary-file" + for options in ({"remote_service": "wrong"}, {"dataset": "a"}, {"hdf5_index": {}}): + with pytest.raises(ValueError): + blosc2.open("https://host/@public/a", **options) + with pytest.raises(ValueError): + blosc2.open(blosc2.URLPath("@public", urlbase="https://host"), remote_service="fsspec") From 8b562deb3edb3528141c02fea7ab4ce97c1eef08 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sat, 3 Oct 2026 13:43:04 +0200 Subject: [PATCH 02/35] Discover Caterva2 mounted hierarchies lazily --- plans/caterva2-access-improvements.md | 6 ++ src/blosc2/remote_store.py | 99 ++++++++++++++++++--------- src/blosc2/schunk.py | 1 + tests/test_caterva2_access.py | 61 +++++++++++++++++ tests/test_remote_caterva2.py | 44 +++++++++--- 5 files changed, 171 insertions(+), 40 deletions(-) diff --git a/plans/caterva2-access-improvements.md b/plans/caterva2-access-improvements.md index 4b7e24486..0cbf065f5 100644 --- a/plans/caterva2-access-improvements.md +++ b/plans/caterva2-access-improvements.md @@ -18,6 +18,12 @@ Implementation record (updated as milestones land): - Roots discovery uses the mapping response implemented by cat2lite and Caterva2, not an assumed list. No global discovery cache is introduced. - M1 validation: 20 offline URL/dispatch tests passed in the blosc2 environment. +- M2: Caterva2 groups now discover immediate children on demand, support direct + descendant lookup, memoize completed listings under the owner lock, and roll + back node-limit failures. Opening a string root uses the opener's metadata + seed to avoid duplicate info calls. Catalog annotations are exposed separately + on RemoteNode and retained in discovery metadata. Focused URL/Caterva2 tests: + 37 passed, including existing reference/table/cache regressions. ## 1. Goal diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index c178e392d..57b4d529e 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -332,6 +332,7 @@ class RemoteNode: kind: str attrs: RemoteMetadataMapping | None diagnostic: str | None = None + catalog_attrs: RemoteMetadataMapping | None = None def _resolve_hdf5_options(hdf5_index, private_index, source_format): @@ -368,6 +369,7 @@ def __init__( # noqa: C901 _b2z_blob=None, _local_source=False, _parquet_conversion=None, + _root_info=None, ): self.caterva2 = _caterva2_bound_urlpath(urlpath) if isinstance(urlpath, blosc2.URLPath) else None if self.caterva2 is not None: @@ -416,6 +418,7 @@ def __init__( # noqa: C901 manifest = None # Rebind legacy metadata/payload to the source digest once. self.traffic = _traffic if _traffic is not None else Traffic() self.transport = _transport + self.root_info = _root_info self.nodes = {} self.attrs = {} self.listed = {} @@ -982,39 +985,63 @@ def _caterva2_get(self, endpoint, path): return response.json() def _open_caterva2(self): - root_info = self._caterva2_get("info", self.root) - root_kind = self._caterva2_kind(root_info) - if root_kind == "ctable": - self._add(self.root, "ctable", self._caterva2_table_metadata(self.root, root_info)) - self.attrs[self.root] = dict(root_info.get("attrs") or {}) - return - if root_kind == "ndarray": - self._add(self.root, "ndarray") - self.attrs[self.root] = dict(root_info.get("attrs") or {}) - return - if root_kind != "group": + self._discover_caterva2_node(self.root) + if self.nodes[self.root][0] not in {"group", "ctable", "ndarray"}: raise ValueError("Caterva2 source root is not an array, group, or CTable") - self._add(self.root, "group") - self.attrs[self.root] = dict(root_info.get("attrs") or {}) - leaves = self._caterva2_get("list", self.root) + + def _discover_caterva2_node(self, full): + # attrs membership distinguishes actual metadata from structural group + # placeholders. An unlisted mount must not be mistaken for an empty group. + if full in self.attrs: + return + info = ( + self.root_info + if full == self.root and self.root_info is not None + else self._caterva2_get("info", full) + ) + if not isinstance(info, dict): + raise ValueError("Invalid Caterva2 info response") + kind = self._caterva2_kind(info) + value = self._caterva2_table_metadata(full, info) if kind == "ctable" else None + if kind == "unsupported": + value = "Caterva2 object kind is unsupported" + attrs = dict(info.get("attrs") or {}) + annotations = dict(info.get("catalog_attrs") or {}) + # Validate all potentially failing work before changing the registry. + previous = self.nodes.copy() + try: + self.nodes.pop(full, None) + self._add(full, kind, value) + except BaseException: + self.nodes = previous + raise + self.attrs[full] = attrs + if annotations: + self.metadata.setdefault("catalog_attrs", {})[full] = annotations + + def _list_caterva2(self, full): + leaves = self._caterva2_get("list", full) if not isinstance(leaves, list) or any(not isinstance(path, str) for path in leaves): raise ValueError("Invalid Caterva2 list response") - for relative in sorted(set(leaves)): + children = set() + for relative in leaves: + if not relative: + raise ValueError("Invalid empty Caterva2 list entry") self._validate(relative) - full = "/".join((self.root, relative)) - info = self._caterva2_get("info", full) - kind = self._caterva2_kind(info) - if kind == "ctable": - self._add(full, kind, self._caterva2_table_metadata(full, info)) - elif kind in {"ndarray", "group"}: - self._add(full, kind) - else: - self._add(full, "unsupported", "Caterva2 object kind is unsupported") - self.attrs[full] = dict(info.get("attrs") or {}) - for parent in (path for path, (kind, _) in self.nodes.items() if kind == "group"): - self.listed[parent] = sorted( - path for path in self.nodes if path != parent and path.rpartition("/")[0] == parent - ) + if any(c in relative for c in "%?#"): + raise ValueError("Unsafe Caterva2 list path") + children.add(full + "/" + relative.split("/", 1)[0]) + previous = self.nodes.copy() + try: + for child in sorted(children): + if child not in self.nodes: + self._add(child, "group") + except BaseException: + self.nodes = previous + raise + # Only the queried group is complete. Recursive server listings do not + # imply that descendant groups/mounts have themselves been listed. + self.listed[full] = sorted(children) def _path(self, path): if not isinstance(path, str): @@ -1026,6 +1053,8 @@ def _path(self, path): def resolve(self, path): """Resolve a relative path without listing a Zarr parent.""" full = self._path(path) + if self.format == "caterva2": + self._discover_caterva2_node(full) if self.format == "zarr" and ( full not in self.nodes or (self.nodes[full][0] != "unsupported" and self.nodes[full][1] is None) ): @@ -1046,14 +1075,16 @@ def resolve(self, path): return full def kind(self, path): - return self.nodes[self._path(path)][0] + return self.nodes[self.resolve(path)][0] def list_children(self, path): # noqa: C901 full = self._path(path) if self.nodes[full][0] != "group": return [] if full not in self.listed: - if self.format == "zarr": + if self.format == "caterva2": + self._list_caterva2(full) + elif self.format == "zarr": import zarr group = self.nodes[full][1] @@ -2094,6 +2125,7 @@ def __init__( # noqa: C901 _b2z_blob=None, _allow_local_source=False, _parquet_conversion=None, + _root_info=None, ): dataset = blosc2.core.resolve_dataset_path(dataset, path) caterva2_input = isinstance(urlpath, blosc2.URLPath) @@ -2209,6 +2241,7 @@ def __init__( # noqa: C901 _b2z_blob=_b2z_blob, _local_source=local_source, _parquet_conversion=_parquet_conversion, + _root_info=_root_info, ) break except (KeyError, TypeError, ValueError): @@ -2907,6 +2940,7 @@ def get_info(self, path=""): kind, None if attrs is None else RemoteMetadataMapping(attrs), diagnostic, + RemoteMetadataMapping(self._owner.metadata.get("catalog_attrs", {}).get(full, {})), ) def kind(self, path=""): @@ -2941,7 +2975,8 @@ def info_items(self) -> list[tuple[str, object]]: continue for child in children: relative = child[len(root) + 1 :] if root else child - kind = self._owner.nodes[child][0] + owner_path = child[len(self._owner.root) + 1 :] if self._owner.root else child + kind = self._owner.nodes[self._owner.resolve(owner_path)][0] entries[relative] = f" [{kind}]" if kind == "group": pending.append(relative) diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index 78f98a1af..9b7a5eaa8 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2213,6 +2213,7 @@ def _open_c2_urlpath(urlpath: blosc2.URLPath, mode: str, offset: int, kwargs: di return blosc2.RemoteStore( urlpath, cache_dir=cache_dir, + _root_info=metadata, **store_options, ) if kind == "ctable": diff --git a/tests/test_caterva2_access.py b/tests/test_caterva2_access.py index 582c65184..bc42db901 100644 --- a/tests/test_caterva2_access.py +++ b/tests/test_caterva2_access.py @@ -75,3 +75,64 @@ def test_override_and_option_validation(monkeypatch): blosc2.open("https://host/@public/a", **options) with pytest.raises(ValueError): blosc2.open(blosc2.URLPath("@public", urlbase="https://host"), remote_service="fsspec") + + +def test_mount_expansion_is_lazy_and_memoized(caterva2_source): # noqa: F811 + base, array, _, stats = caterva2_source + with blosc2.open(base + "@public") as store: + assert stats["requests"] == ["/api/info/@public"] + assert dict(store.get_info().catalog_attrs) == {"note": "curated"} + assert dict(store.attrs) == {"name": "fixture"} + assert store.keys() == ["mount"] + assert "/api/list/@public/mount" not in stats["requests"] + with store["mount"] as mount: + assert mount.keys() == ["array", "empty", "table"] + assert mount.kind("array") == "ndarray" + with mount["empty"] as empty: + assert empty.keys() == [] + before = stats["requests"].copy() + assert mount.keys() == ["array", "empty", "table"] + assert stats["requests"] == before + with store["mount/array"] as remote: + assert (remote[0:1, 0:2] == array[0:1, 0:2]).all() + + +def test_direct_lookup_and_recursive_list(caterva2_source): # noqa: F811 + base, _, _, stats = caterva2_source + stats["groups"]["@public"] = ["mount/array", "mount/table", "mount/empty", "mount/array"] + with blosc2.open(base + "@public") as store: + assert store.kind("mount/array") == "ndarray" + assert not any("/api/list/" in request for request in stats["requests"]) + assert store.keys() == ["mount"] + assert "/api/info/@public/mount/table" not in stats["requests"] + with store["mount"] as mount: + assert mount.keys() == ["array", "empty", "table"] + + +def test_listing_failure_can_retry_and_limits_are_atomic(caterva2_source): # noqa: F811 + import httpx + + base, _, _, stats = caterva2_source + with blosc2.RemoteStore(blosc2.URLPath("@public", urlbase=base), _max_nodes=5) as store: + assert store.keys() == ["mount"] + stats["fail_list"] = "@public/mount" + with store["mount"] as mount: + with pytest.raises(httpx.HTTPStatusError): + mount.keys() + stats["fail_list"] = None + previous = store._owner.nodes.copy() + with pytest.raises(ValueError, match="node limit"): + mount.keys() + assert store._owner.nodes == previous + stats["groups"]["@public/mount"] = ["empty"] + assert mount.keys() == ["empty"] + + +def test_concurrent_expansion_is_coalesced(caterva2_source): # noqa: F811 + from concurrent.futures import ThreadPoolExecutor + + base, _, _, stats = caterva2_source + with blosc2.open(base + "@public") as store, ThreadPoolExecutor(4) as executor: + results = list(executor.map(lambda _: store.keys(), range(12))) + assert results == [["mount"]] * 12 + assert stats["requests"].count("/api/list/@public") == 1 diff --git a/tests/test_remote_caterva2.py b/tests/test_remote_caterva2.py index e6efe6eb8..3f72530dd 100644 --- a/tests/test_remote_caterva2.py +++ b/tests/test_remote_caterva2.py @@ -84,6 +84,16 @@ def caterva2_source(request): "limit_rows": None, "fail_start": None, "short_start": None, + "requests": [], + "roots": {"@public": {"name": "@public"}}, + "roots_status": 200, + "groups": { + "@public": ["mount"], + "@public/mount": ["array", "table", "empty"], + "@public/mount/empty": [], + "@public/group": ["array", "table"], + }, + "fail_list": None, } def array_info(): @@ -130,35 +140,53 @@ def do_GET(self): stats["cookies"].append(self.headers.get("Cookie")) parsed = urllib.parse.urlsplit(self.path) path = urllib.parse.unquote(parsed.path) + path = path.removeprefix("/demo") + stats["requests"].append(path) + if path == "/api/roots": + if stats["roots_status"] != 200: + self.send_error(stats["roots_status"]) + else: + self.send(json.dumps(stats["roots"]).encode(), "application/json") + return if path.startswith("/api/info/"): key = path.removeprefix("/api/info/") - if key == "@public/group": - info = {"kind": "group", "attrs": {"name": "fixture"}} - elif key == "@public/group/array": + if key in stats["groups"]: + info = { + "kind": "group", + "attrs": {"name": "fixture"}, + "catalog_attrs": {"note": "curated"}, + } + elif key in {"@public/group/array", "@public/mount/array"}: info = array_info() - elif key in {"@public/group/table", "@public/table"}: + elif key in {"@public/group/table", "@public/table", "@public/mount/table"}: info = table_info() else: self.send_error(404) return self.send(json.dumps(info).encode(), "application/json") return - if path == "/api/list/@public/group": - self.send(json.dumps(["array", "table"]).encode(), "application/json") + if path.startswith("/api/list/"): + key = path.removeprefix("/api/list/") + if key == stats["fail_list"]: + self.send_error(503) + elif key in stats["groups"]: + self.send(json.dumps(stats["groups"][key]).encode(), "application/json") + else: + self.send_error(404) return if path.startswith("/api/fetch/"): stats["fetches"] += 1 key = path.removeprefix("/api/fetch/") query = urllib.parse.parse_qs(parsed.query) selection = query.get("slice_", [""])[0] - if key == "@public/group/array": + if key in {"@public/group/array", "@public/mount/array"}: axes = tuple( slice(*(int(part) if part else None for part in axis.split(":"))) for axis in selection.split(",") ) self.send(array.slice(axes).to_cframe(), "application/octet-stream") return - if key in {"@public/group/table", "@public/table"}: + if key in {"@public/group/table", "@public/table", "@public/mount/table"}: start, stop = (int(part) for part in selection.split(":")) stats["ranges"].append((start, stop)) if stats["fail_start"] == start: From c2111dc1c186c5b158d7625027c7670f3e129e2b Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sat, 3 Oct 2026 13:46:55 +0200 Subject: [PATCH 03/35] Discover Caterva2 services and browse independent repository roots --- plans/caterva2-access-improvements.md | 7 + src/blosc2/__init__.py | 2 + src/blosc2/caterva2_url.py | 102 +++++++++++++ src/blosc2/remote_repository.py | 205 ++++++++++++++++++++++++++ src/blosc2/schunk.py | 62 +++++++- tests/test_caterva2_access.py | 145 ++++++++++++++++++ 6 files changed, 520 insertions(+), 3 deletions(-) create mode 100644 src/blosc2/remote_repository.py diff --git a/plans/caterva2-access-improvements.md b/plans/caterva2-access-improvements.md index 0cbf065f5..c3067a529 100644 --- a/plans/caterva2-access-improvements.md +++ b/plans/caterva2-access-improvements.md @@ -24,6 +24,13 @@ Implementation record (updated as milestones land): seed to avoid duplicate info calls. Catalog annotations are exposed separately on RemoteNode and retained in discovery metadata. Focused URL/Caterva2 tests: 37 passed, including existing reference/table/cache regressions. +- M3: added bounded roots discovery (3-second deadline, 1 MiB response bound, + same-origin redirects only), direct-source bypass, strict error/fallback + classification, and the public RemoteRepository browsing facade. Roots open + independently and lazily; facade aliases share owners, returned children + outlive the facade, and authentication is frozen at opening. Per-root cache + budgets and unsupported repository persistence are explicit. Validation: + 38 access tests passed, including zero/one/multiple roots and error handling. ## 1. Goal diff --git a/src/blosc2/__init__.py b/src/blosc2/__init__.py index 75e61e82a..a41e13861 100644 --- a/src/blosc2/__init__.py +++ b/src/blosc2/__init__.py @@ -627,6 +627,7 @@ def _raise(exc): from .remote_object import RemoteObject from .remote_array import RemoteMetadataMapping, RemoteArray from .remote_store import RemoteNode, RemoteStore +from .remote_repository import RemoteRepository from . import linalg from .linalg import tensordot, vecdot, permute_dims, matrix_transpose, matmul, transpose, diagonal, outer from .utils import linalg_funcs as linalg_funcs_list @@ -931,6 +932,7 @@ def _raise(exc): "RemoteCTable", "RemoteNode", "RemoteStore", + "RemoteRepository", "SChunk", "SimpleProxy", "SpecialValue", diff --git a/src/blosc2/caterva2_url.py b/src/blosc2/caterva2_url.py index fcaadc793..761e61541 100644 --- a/src/blosc2/caterva2_url.py +++ b/src/blosc2/caterva2_url.py @@ -66,3 +66,105 @@ def caterva2_urlpath(value): raise ValueError("Caterva2 URL requires a root and nonempty dataset components") base = urlunsplit((parsed.scheme, parsed.netloc, "/".join(components[:marker]), "", "")) return blosc2.URLPath("/".join(logical), urlbase=base) + + +def service_probe_candidate(value): + """Whether an HTTP URL is ambiguous rather than a recognized data source.""" + if not isinstance(value, str) or not value.startswith(("http://", "https://")): + return False + parsed = urlsplit(value) + if parsed.query or parsed.fragment or "::" in parsed.path: + return False + suffixes = ( + ".b2nd", + ".b2z", + ".b2d", + ".b2f", + ".b2frame", + ".b2b", + ".b2e", + ".h5", + ".hdf5", + ".zarr", + ".parquet", + ) + return not any( + decode_service_component(part).lower().endswith(suffixes) for part in parsed.path.split("/") + ) + + +def validate_roots(roots): + """Validate the Caterva2 roots mapping without assuming a root name prefix.""" + if not isinstance(roots, dict) or len(roots) > 1000: + raise ValueError("Invalid Caterva2 roots response") + for name, metadata in roots.items(): + if ( + not isinstance(name, str) + or not name + or name in {".", ".."} + or any(c in name for c in "/\\:%?#") + or any(ord(c) < 32 or ord(c) == 127 for c in name) + or not isinstance(metadata, dict) + or metadata.get("name", name) != name + ): + raise ValueError("Invalid Caterva2 root entry") + return roots + + +def discover_service(value, *, required=False, auth_token=None): + """Probe api/roots with bounded bytes, time, and same-origin redirects. + + Return a validated roots mapping or None for a conclusively non-service + response. Authentication, connectivity, and server failures are preserved. + """ + import time + + from blosc2.c2array import _auth_headers, _server_url, _sync_client + + parsed = validate_service_url(value) + url = _server_url(value.rstrip("/"), "api/roots") + client = _sync_client() + deadline = time.monotonic() + 3 + for _ in range(4): + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("Caterva2 service discovery exceeded 3 seconds") + with client.stream( + "GET", url, headers=_auth_headers(auth_token), timeout=remaining, follow_redirects=False + ) as response: + if response.status_code in {301, 302, 303, 307, 308}: + from urllib.parse import urljoin + + target = urljoin(url, response.headers.get("location", "")) + redirect = urlsplit(target) + if (redirect.scheme, redirect.hostname, redirect.port) != ( + parsed.scheme, + parsed.hostname, + parsed.port, + ): + raise ValueError("Caterva2 discovery cannot redirect credentials to another origin") + if redirect.username is not None or redirect.password is not None: + raise ValueError("Unsafe Caterva2 discovery redirect") + url = target + continue + if response.status_code in {404, 405}: + if required: + response.raise_for_status() + return None + response.raise_for_status() + payload = bytearray() + for chunk in response.iter_bytes(): + if time.monotonic() > deadline: + raise TimeoutError("Caterva2 service discovery exceeded 3 seconds") + payload.extend(chunk) + if len(payload) > 1 << 20: + raise ValueError("Caterva2 roots response exceeds 1 MiB") + import json + + try: + return validate_roots(json.loads(payload)) + except (ValueError, TypeError) as error: + if required: + raise ValueError("Invalid Caterva2 roots response") from error + return None + raise ValueError("Too many Caterva2 discovery redirects") diff --git a/src/blosc2/remote_repository.py b/src/blosc2/remote_repository.py new file mode 100644 index 000000000..986730fdb --- /dev/null +++ b/src/blosc2/remote_repository.py @@ -0,0 +1,205 @@ +"""A browsing facade for the independent roots of a Caterva2 service.""" + +import threading +import weakref +from pathlib import Path + +import blosc2 +from blosc2.caterva2_url import validate_roots, validate_service_url +from blosc2.info import InfoReporter +from blosc2.proxy_source import Traffic +from blosc2.remote_array import CACHE_POLICY_DEFAULT, RemoteMetadataMapping +from blosc2.remote_store import RemoteNode, RemoteStore + + +def _close_roots(owners, lock, state): + with lock: + state["users"] -= 1 + if state["users"]: + return + for store in owners.values(): + store.close() + owners.clear() + + +class RemoteRepository(RemoteStore): + """Read-only, nonpersistent group of Caterva2 roots. + + Returned by ``blosc2.open`` for empty or multi-root services. Root owners + open lazily. Cache allowances are **per root**, not a repository-wide bound. + Closing this facade leaves previously returned child handles usable. + Open a specific root to persist or materialize a selection. + """ + + def __init__(self, urlbase, roots, *, auth_token=None, **options): + validate_service_url(urlbase) + self.urlbase = urlbase.rstrip("/") + self.roots = dict(validate_roots(roots)) + self.auth_token = blosc2.c2array._server_data["auth_token"] if auth_token is None else auth_token + self.options = dict(options) + policy, limit = self._validate_cache_config( + self.options.pop("cache_policy", CACHE_POLICY_DEFAULT), + self.options.pop("max_cache_bytes", CACHE_POLICY_DEFAULT), + self.options.get("cache_dir"), + ) + if self.options.keys() - {"cache_dir"}: + raise TypeError("RemoteRepository accepts only cache_dir, cache_policy, and max_cache_bytes") + self.options.update(cache_policy=policy, max_cache_bytes=limit) + self._lock = threading.RLock() + self._roots = {} + self._state = {"users": 1} + self._traffic = Traffic() + self._finalizer = weakref.finalize(self, _close_roots, self._roots, self._lock, self._state) + self.is_tree = True + + def _ensure_open(self): + if not self._finalizer.alive: + raise RuntimeError("RemoteRepository is closed") + + def _resolve(self, path): + raise NotImplementedError("Open a repository root before using this RemoteStore operation") + + def _check_open(self): + self._ensure_open() + + def _root(self, name): + self._ensure_open() + if name not in self.roots: + raise KeyError(name) + if name not in self._roots: + options = self.options.copy() + if options.get("cache_dir") is not None: + # The root owner's existing identity hashing includes the base, + # dataset, and authentication scope. Each root gets its own budget. + options["cache_dir"] = Path(options["cache_dir"]) + self._roots[name] = RemoteStore( + blosc2.URLPath(name, urlbase=self.urlbase, auth_token=self.auth_token), + _traffic=self._traffic, + **options, + ) + return self._roots[name] + + def keys(self): + with self._lock: + self._ensure_open() + return sorted(self.roots) + + def __getitem__(self, path): + with self._lock: + self._ensure_open() + if not isinstance(path, str): + raise TypeError("RemoteRepository paths must be strings") + relative = path.strip("/") + if not relative: + alias = object.__new__(type(self)) + alias.__dict__.update(self.__dict__) + self._state["users"] += 1 + alias._finalizer = weakref.finalize( + alias, _close_roots, self._roots, self._lock, self._state + ) + return alias + from blosc2.remote_store import RemoteDiscovery + + RemoteDiscovery._validate(relative) + name, _, suffix = relative.partition("/") + return self._root(name)[suffix] + + def get_info(self, path=""): + with self._lock: + self._ensure_open() + if not isinstance(path, str): + raise TypeError("RemoteRepository paths must be strings") + relative = path.strip("/") + from blosc2.remote_store import RemoteDiscovery + + RemoteDiscovery._validate(relative) + if not relative: + return RemoteNode("", "group", RemoteMetadataMapping({})) + name, _, suffix = relative.partition("/") + if name not in self.roots: + raise KeyError(name) + if not suffix: + # Listing the service must not contact all its roots. + return RemoteNode(relative, "group", RemoteMetadataMapping(self.roots[name])) + info = self._root(name).get_info(suffix) + return RemoteNode(relative, info.kind, info.attrs, info.diagnostic, info.catalog_attrs) + + @property + def source(self): + self._ensure_open() + return {"kind": "caterva2_repository", "version": 1, "urlbase": self.urlbase + "/"} + + @property + def info(self): + return InfoReporter(self) + + @property + def info_items(self): + return [ + ("type", type(self).__name__), + ("source", self.source), + ("roots", self.keys()), + ("cache allowance", "per root"), + ] + + @property + def attrs(self): + return self.get_info().attrs + + @property + def cache_bytes(self): + with self._lock: + self._ensure_open() + return sum(store.cache_bytes for store in self._roots.values()) + + @property + def metadata_bytes(self): + with self._lock: + self._ensure_open() + return sum(store.metadata_bytes for store in self._roots.values()) + + @property + def max_cache_bytes(self): + """Configured allowance per root (not a total repository allowance).""" + self._ensure_open() + return self.options["max_cache_bytes"] + + @property + def cache_policy(self): + self._ensure_open() + return self.options["cache_policy"] + + @property + def traffic(self): + """Shared counter for opened roots; the initial roots probe is excluded.""" + self._ensure_open() + return self._traffic + + @property + def mutable(self): + return False + + @property + def is_cache_mutable(self): + self._ensure_open() + return True + + def read_cached(self, path, item=(), *, nchunk=None): + with self._lock: + self._ensure_open() + name, _, suffix = path.strip("/").partition("/") + return self._root(name).read_cached(suffix, item, nchunk=nchunk) + + def save(self, *args, **kwargs): + raise NotImplementedError("Repository persistence is unsupported; save a specific root") + + def materialize(self, *args, **kwargs): + raise NotImplementedError("Repository materialization is unsupported; select a specific root") + + def refresh(self): + raise NotImplementedError( + "Reopen the repository to rediscover roots; refresh a specific root separately" + ) + + def close(self): + self._finalizer() diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index 9b7a5eaa8..14e7305bf 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2778,6 +2778,44 @@ def _validate_shared_cache_request(urlpath, shared_cache, kwargs): kwargs["lazy"] = True +def _open_service_base(urlpath, roots, mode, offset, shared_cache, kwargs): + """Open one visible root or a lazy repository facade, without a second probe.""" + if mode != "r" or offset != 0: + raise ValueError("Caterva2 services require mode='r' and offset=0") + if kwargs.get("lazy") is False: + raise NotImplementedError( + "Repository/group access requires lazy=True; select an array for lazy=False" + ) + if len(roots) == 1: + target = blosc2.URLPath(next(iter(roots)), urlbase=urlpath.rstrip("/")) + kwargs.setdefault("lazy", True) + _validate_shared_cache_request(target, shared_cache, kwargs) + return _open_c2_urlpath(target, mode, offset, kwargs) + if shared_cache: + raise NotImplementedError("shared_cache requires a specific repository root") + cache_dir, cache_path = _remote_cache_options(kwargs) + if cache_path is not None: + raise ValueError("Repositories use cache_dir, not cache_path") + _validate_c2_urlpath_options(kwargs) + lazy = kwargs.pop("lazy", None) + if lazy is not None and not isinstance(lazy, bool): + raise TypeError("lazy must be a bool") + if kwargs.pop("assume_immutable", True) is not True: + raise ValueError("Repositories currently require assume_immutable=True") + from blosc2.remote_array import CACHE_POLICY_DEFAULT + + policy, limit = blosc2.RemoteStore._validate_cache_config( + kwargs.pop("cache_policy", CACHE_POLICY_DEFAULT), + kwargs.pop("max_cache_bytes", CACHE_POLICY_DEFAULT), + cache_dir, + ) + if kwargs: + raise NotImplementedError(f"{', '.join(sorted(kwargs))} is unsupported for repositories") + return blosc2.RemoteRepository( + urlpath, roots, cache_dir=cache_dir, cache_policy=policy, max_cache_bytes=limit + ) + + def open( # noqa: C901 urlpath: str | pathlib.Path | blosc2.URLPath, mode: str = "r", @@ -2967,6 +3005,11 @@ def open( # noqa: C901 references and default to lazy access. Use ``"fsspec"`` to open an ordinary remote source containing such a component without service recognition. Explicit :ref:`URLPath` inputs retain their lazy defaults. + Ambiguous HTTP(S) base URLs are probed through ``api/roots`` with a + short timeout. One root opens directly; zero or multiple roots return + :class:`RemoteRepository`. ``"caterva2"`` requires service access and + disables ordinary-file fallback. Client cache budgets on a repository + are per root, independent of the server's shared cache. parquet_options: dict, optional PyArrow ``ParquetFile`` reader options for a Parquet source. Conversion options such as ``columns`` and ``max_rows`` are passed separately. @@ -3078,14 +3121,27 @@ def open( # noqa: C901 if remote_service == "fsspec": raise ValueError("remote_service='fsspec' conflicts with URLPath") else: - from blosc2.caterva2_url import caterva2_urlpath + from blosc2.caterva2_url import caterva2_urlpath, discover_service, service_probe_candidate service_path = None if remote_service == "fsspec" else caterva2_urlpath(urlpath) if service_path is not None: urlpath = service_path kwargs.setdefault("lazy", True) - elif remote_service == "caterva2": - raise ValueError("A Caterva2 dataset URL requires an @-prefixed root") + elif remote_service == "caterva2" or ( + remote_service == "auto" + and service_probe_candidate(urlpath) + and dataset is None + and hdf5_index is None + and kwargs.get("source_format") is None + and kwargs.get("storage_options") is None + ): + if mode != "r" or offset: + raise ValueError("Remote service discovery requires mode='r' and offset=0") + if dataset is not None or hdf5_index is not None: + raise ValueError("dataset/path and hdf5_index are unsupported for Caterva2 services") + roots = discover_service(urlpath, required=remote_service == "caterva2") + if roots is not None: + return _open_service_base(urlpath, roots, mode, offset, shared_cache, kwargs) elif remote_service == "fsspec" and not is_fsspec_url(urlpath): raise ValueError("remote_service='fsspec' requires a remote URL") if isinstance(urlpath, blosc2.URLPath): diff --git a/tests/test_caterva2_access.py b/tests/test_caterva2_access.py index bc42db901..8e4c6e6bd 100644 --- a/tests/test_caterva2_access.py +++ b/tests/test_caterva2_access.py @@ -136,3 +136,148 @@ def test_concurrent_expansion_is_coalesced(caterva2_source): # noqa: F811 results = list(executor.map(lambda _: store.keys(), range(12))) assert results == [["mount"]] * 12 assert stats["requests"].count("/api/list/@public") == 1 + + +def test_bare_and_prefixed_service(caterva2_source): # noqa: F811 + base, _, _, stats = caterva2_source + for url in (base.rstrip("/"), base + "demo"): + with blosc2.open(url) as store: + assert isinstance(store, blosc2.RemoteStore) + assert not isinstance(store, blosc2.RemoteRepository) + assert store.keys() == ["mount"] + assert stats["requests"].count("/api/roots") == 2 + + +def test_repository_is_lazy_and_children_outlive_it(caterva2_source, tmp_path): # noqa: F811 + base, array, _, stats = caterva2_source + stats["roots"]["@broken"] = {"name": "@broken"} + repo = blosc2.open(base, cache_dir=tmp_path / "cache") + try: + assert isinstance(repo, blosc2.RemoteRepository) + assert repo.keys() == ["@broken", "@public"] + assert stats["requests"] == ["/api/roots"] + assert repo.kind("@broken") == "group" + import httpx + + with pytest.raises(httpx.HTTPStatusError): + repo["@broken"] + with repo[""] as alias: + child = alias["@public/mount/array"] + assert child is not None + with pytest.raises(NotImplementedError, match="persistence"): + repo.save(tmp_path / "repository.b2z") + assert repo.cache_policy is blosc2.CachePolicy.DISK + assert repo.cache_bytes == 0 + assert "per root" in str(repo.info) + finally: + repo.close() + with child: + assert (child[0:1, 0:2] == array[0:1, 0:2]).all() + with pytest.raises(RuntimeError, match="closed"): + repo.keys() + + +def test_empty_service_and_repository_option_validation(caterva2_source): # noqa: F811 + base, _, _, stats = caterva2_source + stats["roots"] = {} + with blosc2.open(base) as repo: + assert repo.keys() == [] + assert repo.kind() == "group" + for options in ( + {"lazy": False}, + {"cache_path": "a.b2nd"}, + {"shared_cache": True}, + {"storage_options": {}, "remote_service": "caterva2"}, + ): + with pytest.raises((ValueError, NotImplementedError)): + blosc2.open(base, **options) + + +def test_direct_format_does_not_probe(monkeypatch): + monkeypatch.setattr(blosc2.schunk, "_open_fsspec_url", lambda *args: "ordinary-file") + monkeypatch.setattr( + blosc2.caterva2_url, "discover_service", lambda *args, **kwargs: pytest.fail("unexpected probe") + ) + for url in ( + "https://host/a.b2nd", + "https://host/a.zarr/", + "https://host/a.h5::/group", + "https://host/a.b2z?version=2", + ): + assert blosc2.open(url) == "ordinary-file" + + +@pytest.mark.parametrize("status", [401, 403, 500, 503]) +def test_discovery_preserves_http_failure(caterva2_source, status): # noqa: F811 + import httpx + + base, _, _, stats = caterva2_source + stats["roots_status"] = status + with pytest.raises(httpx.HTTPStatusError) as error: + blosc2.open(base) + assert error.value.response.status_code == status + assert stats["requests"] == ["/api/roots"] + + +@pytest.mark.parametrize("roots", [[], {"../evil": {}}, {"@public": "bad"}, {"@public": {"name": "other"}}]) +def test_invalid_discovery_falls_back_only_in_auto(caterva2_source, monkeypatch, roots): # noqa: F811 + base, _, _, stats = caterva2_source + stats["roots"] = roots + monkeypatch.setattr(blosc2.schunk, "_open_fsspec_url", lambda *args: "ordinary-file") + assert blosc2.open(base) == "ordinary-file" + with pytest.raises(ValueError, match="roots response"): + blosc2.open(base, remote_service="caterva2") + + +def test_discovery_redirects_timeout_and_response_limits(monkeypatch): + import httpx + + from blosc2.caterva2_url import discover_service + + cases = [ + ( + lambda request: httpx.Response(302, headers={"location": "https://elsewhere/api/roots"}), + ValueError, + ), + (lambda request: httpx.Response(200, content=b"x" * ((1 << 20) + 1)), ValueError), + (lambda request: httpx.Response(302, headers={"location": str(request.url)}), ValueError), + ] + for handler, error in cases: + with httpx.Client(transport=httpx.MockTransport(handler)) as client: + monkeypatch.setattr(blosc2.c2array, "_sync_client", lambda: client) + with pytest.raises(error): + discover_service("https://host/demo", auth_token="secret") + + def timeout(request): + raise httpx.ReadTimeout("slow service", request=request) + + with httpx.Client(transport=httpx.MockTransport(timeout)) as client: + monkeypatch.setattr(blosc2.c2array, "_sync_client", lambda: client) + with pytest.raises(httpx.ReadTimeout): + discover_service("https://host") + + seen = [] + + def redirect(request): + seen.append(str(request.url)) + return ( + httpx.Response(302, headers={"location": "/demo/api/roots/"}) + if len(seen) == 1 + else httpx.Response(200, json={"@public": {"name": "@public"}}) + ) + + with httpx.Client(transport=httpx.MockTransport(redirect)) as client: + monkeypatch.setattr(blosc2.c2array, "_sync_client", lambda: client) + assert discover_service("https://host/demo") == {"@public": {"name": "@public"}} + assert seen == ["https://host/demo/api/roots", "https://host/demo/api/roots/"] + + +def test_repository_freezes_auth_context(caterva2_source): # noqa: F811 + base, _, _, stats = caterva2_source + stats["roots"]["@broken"] = {"name": "@broken"} + with blosc2.c2context(auth_token="alice=secret"): + repo = blosc2.open(base) + with blosc2.c2context(auth_token="bob=secret"), repo: + with repo["@public"] as root: + assert root.keys() == ["mount"] + assert set(stats["cookies"]) == {"alice=secret"} From 1708856519835cfd81c63641ee45511b92653983 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sat, 3 Oct 2026 13:52:02 +0200 Subject: [PATCH 04/35] Browse Caterva2 repositories and mounted sources in b2view --- doc/guides/b2view.rst | 40 +++++++++++++ plans/caterva2-access-improvements.md | 7 +++ src/blosc2/b2view/app.py | 22 +++++++ src/blosc2/b2view/cli.py | 13 ++++- src/blosc2/b2view/model.py | 66 +++++++++++++++++++-- tests/b2view/test_caterva2.py | 82 +++++++++++++++++++++++++++ tests/b2view/test_cli.py | 18 +++++- 7 files changed, 237 insertions(+), 11 deletions(-) create mode 100644 tests/b2view/test_caterva2.py diff --git a/doc/guides/b2view.rst b/doc/guides/b2view.rst index c980e3f7b..d579bef62 100644 --- a/doc/guides/b2view.rst +++ b/doc/guides/b2view.rst @@ -63,6 +63,46 @@ You can also jump straight to a node and panel: Remote containers and arrays ~~~~~~~~~~~~~~~~~~~~~~~~~~~~ +Caterva2 and shared cat2lite repositories +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Open a Caterva2-compatible server, a published root, or a selected group/leaf: + +.. code-block:: console + + b2view http://localhost:8000 + b2view http://localhost:8000/@public + b2view http://localhost:8000/@public/hdf5 /d0/d1/a2 + b2view https://cat2.cloud/demo/@public/example + +A bare server URL probes ``api/roots``. A single visible root opens directly; +multiple roots appear as children of a repository group. An empty server shows +"No accessible roots". Deployment prefixes such as ``/demo`` are preserved. +For scripts and reproducible starting paths, prefer an explicit root URL. + +Group expansion discovers mounted hierarchies on demand. Arrays and tables use +bounded previews/pages through the server API; the viewer does not download the +whole source container. Catalog annotations appear separately as +``catalog_attrs`` in metadata, without replacing source attrs. + +Use ``--remote-service fsspec`` when an ordinary HTTP data URL happens to contain +an ``@``-prefixed path component. Use ``--remote-service caterva2`` to require +service discovery rather than falling back to a file opener. + +``--cache-dir`` and ``--max-cache-bytes`` configure the **client** cache, not the +server's shared cache. Multi-root repository budgets are per opened root. +The cat2lite administrator configures shared upstream caching independently. +Sources are assumed immutable; updates require deliberate cache invalidation. +Overlapping reads across clients benefit most from a shared server cache. + +Remote tables support previews and projection, but filtering, sorting, and +grouping are disabled rather than implicitly downloading the whole table. +Table plotting requires a bounded locked row window (``v``). Source formats +behind cat2lite need their dependencies on the server, not on each viewer client. + +Direct source URLs +^^^^^^^^^^^^^^^^^^ + Browse remote B2Z, Zarr, and HDF5 containers directly from their root: .. code-block:: console diff --git a/plans/caterva2-access-improvements.md b/plans/caterva2-access-improvements.md index c3067a529..006d33405 100644 --- a/plans/caterva2-access-improvements.md +++ b/plans/caterva2-access-improvements.md @@ -31,6 +31,13 @@ Implementation record (updated as milestones land): outlive the facade, and authentication is frozen at opening. Per-root cache budgets and unsupported repository persistence are explicit. Validation: 38 access tests passed, including zero/one/multiple roots and error handling. +- M4: b2view uses shared service recognition ahead of direct-format handling, + supports the backend override, displays catalog annotations and empty-root + notices, and keeps mounted-source metadata failures isolated. Remote table + metadata/paging work without table-wide downloads; transforms are disabled + and plotting requires an explicit bounded row window. Fixed metadata display + for remote sources that do not report compressed size. All viewer/model + regressions, including headless Textual and fresh-process decoders: 155 passed. ## 1. Goal diff --git a/src/blosc2/b2view/app.py b/src/blosc2/b2view/app.py index e131b8494..a34809d06 100644 --- a/src/blosc2/b2view/app.py +++ b/src/blosc2/b2view/app.py @@ -2029,6 +2029,7 @@ def __init__( storage_options: dict[str, Any] | None = None, cache_dir: str | None = None, max_cache_bytes: int | None = None, + remote_service: str = "auto", ): super().__init__() self.sub_title = f"Python-Blosc2 {blosc2.__version__}" # shown beside the title in the header @@ -2043,6 +2044,7 @@ def __init__( self.storage_options = storage_options self.cache_dir = cache_dir self.max_cache_bytes = max_cache_bytes + self.remote_service = remote_service self.download_url = download_url # when set, fetch urlpath before browsing self.info_url = info_url # optional: metadata endpoint giving the size # Header label: the path as given on the CLI, or the @public-relative @@ -2164,6 +2166,7 @@ def _start_browsing(self) -> None: browser_kwargs: dict[str, Any] = { "storage_options": self.storage_options, "cache_dir": self.cache_dir, + "remote_service": self.remote_service, } if self.max_cache_bytes is not None: browser_kwargs["max_cache_bytes"] = self.max_cache_bytes @@ -2312,6 +2315,7 @@ def _open_remote(self, session, start_path): browser_kwargs: dict[str, Any] = { "storage_options": self.storage_options, "cache_dir": self.cache_dir, + "remote_service": self.remote_service, } if self.max_cache_bytes is not None: browser_kwargs["max_cache_bytes"] = self.max_cache_bytes @@ -3524,6 +3528,12 @@ def action_filter_rows(self) -> None: if self.table_page.get("source_kind") != "ctable": self.notify("Filtering is only supported for CTable nodes", severity="warning") return + if not self.browser.supports_table_transforms(self.selected_path): + self.notify( + "Filtering remote tables is unsupported; materialize a bounded selection first", + severity="warning", + ) + return if self.browser.get_group(self.selected_path): self.notify("Ungroup (Esc) before filtering", severity="warning") return @@ -3536,6 +3546,12 @@ def action_sort_rows(self) -> None: if self.table_page.get("source_kind") != "ctable": self.notify("Sorting is only supported for CTable nodes", severity="warning") return + if not self.browser.supports_table_transforms(self.selected_path): + self.notify( + "Sorting remote tables is unsupported; materialize a bounded selection first", + severity="warning", + ) + return if self.browser.get_group(self.selected_path): # Sort the (tiny) grouped result by any of its columns — key or aggregate. columns = self.browser.column_names(self.selected_path) or [] @@ -3627,6 +3643,12 @@ def action_group_rows(self) -> None: if self.table_page.get("source_kind") != "ctable": self.notify("Grouping is only supported for CTable nodes", severity="warning") return + if not self.browser.supports_table_transforms(self.selected_path): + self.notify( + "Grouping remote tables is unsupported; materialize a bounded selection first", + severity="warning", + ) + return keys = self.browser.group_key_columns(self.selected_path) if not keys: self.notify("No dictionary/numeric columns to group by", severity="warning") diff --git a/src/blosc2/b2view/cli.py b/src/blosc2/b2view/cli.py index a12d67e36..cb7954cb2 100644 --- a/src/blosc2/b2view/cli.py +++ b/src/blosc2/b2view/cli.py @@ -40,7 +40,15 @@ def resolve_source( def build_parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser(description="Browse a Blosc2 bundle or array in the terminal.") - parser.add_argument("urlpath", nargs="?", default=None, help="Local path or remote array URL") + parser.add_argument( + "urlpath", nargs="?", default=None, help="Local path, remote data URL, or Caterva2 server/root URL" + ) + parser.add_argument( + "--remote-service", + choices=["auto", "caterva2", "fsspec"], + default="auto", + help="Remote access backend (fsspec bypasses Caterva2 recognition)", + ) parser.add_argument("path", nargs="?", default="/", help="Optional starting path inside the bundle") parser.add_argument("--profile", help="S3 credential profile") parser.add_argument("--endpoint-url", help="S3 endpoint URL") @@ -58,7 +66,7 @@ def build_parser() -> argparse.ArgumentParser: parser.add_argument( "--cache-dir", metavar="DIR", - help="Directory for persistent disk caching of remote stores and arrays", + help="Client-side disk cache directory (independent of the server cache; repository budgets are per root)", ) parser.add_argument( "--max-cache-bytes", @@ -130,6 +138,7 @@ def main(argv: list[str] | None = None) -> int: info_url=info_url, cache_dir=args.cache_dir, max_cache_bytes=args.max_cache_bytes, + remote_service=args.remote_service, storage_options={ key: value for key, value in {"profile": args.profile, "endpoint_url": args.endpoint_url}.items() diff --git a/src/blosc2/b2view/model.py b/src/blosc2/b2view/model.py index cd55ca5bd..e07634c3f 100644 --- a/src/blosc2/b2view/model.py +++ b/src/blosc2/b2view/model.py @@ -260,6 +260,7 @@ def __init__( storage_options: dict[str, Any] | None = None, cache_dir: str | None = None, max_cache_bytes: int | None = None, + remote_service: str = "auto", ): self.urlpath = urlpath self.cache_dir = cache_dir @@ -269,7 +270,11 @@ def __init__( self._remote_child_counts = {} self._remote_leaf_path = None self.store = self._open_store( - urlpath, storage_options, cache_dir=cache_dir, max_cache_bytes=max_cache_bytes + urlpath, + storage_options, + cache_dir=cache_dir, + max_cache_bytes=max_cache_bytes, + remote_service=remote_service, ) self.is_tree = isinstance(self.store, blosc2.TreeStore) or ( isinstance(self.store, blosc2.RemoteStore) and self.store.kind() == "group" @@ -304,8 +309,19 @@ def _open_store( *, cache_dir: str | None = None, max_cache_bytes: int | None = None, + remote_service: str = "auto", ): options = {} if storage_options is None else {"storage_options": storage_options} + from blosc2.caterva2_url import caterva2_urlpath + + if remote_service == "caterva2" or ( + remote_service == "auto" and caterva2_urlpath(urlpath) is not None + ): + if cache_dir is not None: + options["cache_dir"] = cache_dir + if max_cache_bytes is not None: + options["max_cache_bytes"] = max_cache_bytes + return blosc2.open(urlpath, mode="r", lazy=True, remote_service=remote_service, **options) if is_fsspec_url(urlpath): _, _, source_format = parse_container_url(urlpath) if source_format in {"b2z", "zarr", "hdf5"}: @@ -332,7 +348,7 @@ def _open_store( options["max_cache_bytes"] = 64 << 20 if max_cache_bytes is None else max_cache_bytes if cache_dir is not None: options["cache_dir"] = cache_dir - return blosc2.open(urlpath, mode="r", **options) + return blosc2.open(urlpath, mode="r", remote_service=remote_service, **options) def _release_remote_leaf(self): if self._remote_leaf is not None: @@ -371,7 +387,13 @@ def list_children(self, path: str = "/") -> list[NodeInfo]: with self.store[path] as group: children = [] for name in group: - kind = group.kind(name) + try: + kind = group.kind(name) + except Exception: + # Metadata failure in one mounted source must not hide + # its healthy siblings. Selecting it displays the error + # and can retry discovery without a cached failure. + kind = "unavailable" children.append( NodeInfo( path=self.normalize_path(path.rstrip("/") + "/" + name), @@ -447,17 +469,23 @@ def get_info(self, path: str) -> ObjectInfo: def _remote_info(self, path): node = self.store.get_info(path) metadata = {"type": f"{self.store.source['kind'].upper()} {node.kind}"} + if node.catalog_attrs: + metadata["catalog_attrs"] = dict(node.catalog_attrs) attrs = node.attrs - if node.kind == "ndarray": + if node.kind in {"ndarray", "ctable"}: obj = self._get_object(path) metadata.update(object_metadata(obj)) - attrs = self._attrs_dict(obj.vlmeta) + if node.kind == "ctable": + metadata["type"] = f"{self.store.source['kind'].upper()} {node.kind}" + attrs = self._attrs_dict(obj.attrs) else: self._release_remote_leaf() if node.kind in {"group", "remote_store"} and path in self._remote_child_counts: metadata["children"] = self._remote_child_counts[path] if node.diagnostic: metadata["preview" if node.kind == "unsupported" else "notice"] = node.diagnostic + if isinstance(self.store, blosc2.RemoteRepository) and path == "/" and not self.store.keys(): + metadata["notice"] = "No accessible roots" return ObjectInfo(path, node.kind, metadata, attrs) def preview( @@ -567,6 +595,8 @@ def plot_series( # fast-path spans the whole column in original order, so it is only # valid when nothing narrows *or reorders* the series. view = self._ordered_object(path, obj) + if isinstance(view, blosc2.RemoteCTable): + raise NotImplementedError("Plotting remote tables requires a bounded locked row window") narrowed = view is not obj n = len(view) start, stop = self._clamp_range(row_start, row_stop, n) @@ -668,6 +698,8 @@ def read_series( # Window > filter > sort, matching preview()/read_cell() so the # hi-res view tracks the visible grid (rows and order). view = self._ordered_object(path, obj) + if isinstance(view, blosc2.RemoteCTable): + raise NotImplementedError("Plotting remote tables requires a bounded locked row window") n = len(view) start, stop = self._clamp_range(row_start, row_stop, n) stride = self._series_stride(stop - start, max_points) @@ -745,6 +777,8 @@ def read_xy( # Window > filter > sort, matching read_series() so the scatter tracks # exactly the visible rows (and order). view = self._ordered_object(path, obj) + if isinstance(view, blosc2.RemoteCTable): + raise NotImplementedError("Scatter plotting remote tables requires a bounded locked row window") n = len(view) start, stop = self._clamp_range(row_start, row_stop, n) width = stop - start @@ -930,6 +964,10 @@ def column_names(self, path: str) -> list[str] | None: names = list(getattr(self._get_object(path), "col_names", []) or []) return names or None + def supports_table_transforms(self, path: str) -> bool: + """Whether filtering/sorting/grouping can run without remote materialization.""" + return not isinstance(self._get_object(self.normalize_path(path)), blosc2.RemoteCTable) + def set_filter(self, path: str, expr: str | None) -> int: """Set or clear the row filter of a CTable path; return its row count. @@ -937,6 +975,10 @@ def set_filter(self, path: str, expr: str | None) -> int: propagate to the caller and leave any previous filter untouched. """ path = self.normalize_path(path) + if expr and not self.supports_table_transforms(path): + raise NotImplementedError( + "Filtering remote tables is unsupported; materialize a bounded selection first" + ) expr = (expr or "").strip() if not expr: self._filters.pop(path, None) @@ -973,6 +1015,10 @@ def set_sort(self, path: str, column: str, reverse: bool) -> None: streams from the index, so the full table is never materialised. """ path = self.normalize_path(path) + if not self.supports_table_transforms(path): + raise NotImplementedError( + "Sorting remote tables is unsupported; materialize a bounded selection first" + ) self._window_views.pop(path, None) view = self._row_source(path).sort_by(column, ascending=not reverse, view=True) self._sorts[path] = (column, reverse) @@ -1073,6 +1119,10 @@ def set_group(self, path: str, key: str, op: str, value_col: str | None) -> int: """ path = self.normalize_path(path) # Memoize the materialized result: the store is read-only, so a given + if not self.supports_table_transforms(path): + raise NotImplementedError( + "Grouping remote tables is unsupported; materialize a bounded selection first" + ) # (path, filter, key, op, value_col) always aggregates to the same tiny # CTable. The active filter expr is part of the key so a filtered group # never collides with the unfiltered one. @@ -1334,6 +1384,10 @@ def object_metadata(obj: Any) -> dict[str, Any]: """Extract lightweight metadata from a supported object.""" kind = object_kind(obj) if kind in {"ndarray", "c2array"}: + try: + cbytes = getattr(obj, "cbytes", None) + except NotImplementedError: + cbytes = None return { "shape": getattr(obj, "shape", None), "ndim": len(getattr(obj, "shape", ()) or ()), @@ -1341,7 +1395,7 @@ def object_metadata(obj: Any) -> dict[str, Any]: "chunks": getattr(obj, "chunks", None), "blocks": getattr(obj, "blocks", None), "nbytes": getattr(obj, "nbytes", None), - "cbytes": getattr(obj, "cbytes", None), + "cbytes": cbytes, } if kind == "ctable": try: diff --git a/tests/b2view/test_caterva2.py b/tests/b2view/test_caterva2.py new file mode 100644 index 000000000..b1e538585 --- /dev/null +++ b/tests/b2view/test_caterva2.py @@ -0,0 +1,82 @@ +"""Repository and mount-boundary browsing through the existing viewer.""" + +import numpy as np +import pytest +from test_remote_caterva2 import caterva2_source # noqa: F401 +from tui_wait import wait_until + +from blosc2.b2view.model import StoreBrowser + + +def test_service_browser_metadata_and_previews(caterva2_source): # noqa: F811 + base, array, _, stats = caterva2_source + with StoreBrowser(base) as browser: + assert browser.is_tree + assert [node.name for node in browser.list_children()] == ["mount"] + assert not stats["fetches"] + assert [node.kind for node in browser.list_children("/mount")] == ["ndarray", "group", "ctable"] + assert browser.get_info("/mount").metadata["catalog_attrs"] == {"note": "curated"} + assert browser.get_info("/mount/table").metadata["nrows"] == 12 + preview = browser.preview("/mount/table", start=2, stop=4, max_cols=1) + np.testing.assert_array_equal(preview["data"]["ident"], [2, 3]) + assert not browser.supports_table_transforms("/mount/table") + for operation in ( + lambda: browser.set_filter("/mount/table", "ident > 2"), + lambda: browser.set_sort("/mount/table", "ident", False), + lambda: browser.set_group("/mount/table", "ident", "count", None), + lambda: browser.read_series("/mount/table", column="ident"), + ): + with pytest.raises(NotImplementedError): + operation() + np.testing.assert_array_equal( + browser.preview("/mount/array", slices=(slice(2), slice(3))), array[:2, :3] + ) + + +def test_repository_browser_and_broken_sibling(caterva2_source): # noqa: F811 + base, _, _, stats = caterva2_source + stats["roots"]["@broken"] = {"name": "@broken"} + with StoreBrowser(base) as browser: + assert [node.name for node in browser.list_children()] == ["@broken", "@public"] + assert [node.name for node in browser.list_children("/@public")] == ["mount"] + stats["roots"].pop("@broken") + stats["groups"]["@public/mount"].append("broken") + with StoreBrowser(base) as browser: + children = browser.list_children("/mount") + assert [(node.name, node.kind) for node in children] == [ + ("array", "ndarray"), + ("broken", "unavailable"), + ("empty", "group"), + ("table", "ctable"), + ] + + +@pytest.mark.tui +@pytest.mark.asyncio +@pytest.mark.parametrize("multiple", [False, True]) +async def test_repository_tui_expands_mounts_and_previews(caterva2_source, multiple): # noqa: F811 + from blosc2.b2view.app import B2ViewApp + + base, array, _, stats = caterva2_source + prefix = "/@public" if multiple else "" + if multiple: + stats["roots"]["@broken"] = {"name": "@broken"} + app = B2ViewApp(base, start_path=prefix + "/mount/array") + async with app.run_test(size=(120, 40)) as pilot: + await wait_until(pilot, lambda: app.table_page is not None and bool(app.table_page["columns"])) + assert app.selected_path == prefix + "/mount/array" + page = app.table_page + np.testing.assert_array_equal(page["data"]["0"], array[: page["stop"], 0]) + app.update_panels(prefix + "/mount/table") + await wait_until( + pilot, lambda: app.table_page is not None and app.table_page.get("source_kind") == "ctable" + ) + assert app.table_page["nrows"] == 12 + app.wait_for_close() + + +def test_cli_service_override(): + from blosc2.b2view.cli import build_parser + + args = build_parser().parse_args(["https://host/@data/a.h5", "--remote-service", "fsspec"]) + assert args.remote_service == "fsspec" diff --git a/tests/b2view/test_cli.py b/tests/b2view/test_cli.py index 6de04a7a0..4dd315817 100644 --- a/tests/b2view/test_cli.py +++ b/tests/b2view/test_cli.py @@ -82,7 +82,9 @@ def run(app, **kwargs): for key, value in options.items(): argv.extend(["--" + key.replace("_", "-"), value]) assert main(argv) == 0 - assert opened == [(url, {"storage_options": options or None, "cache_dir": None})] + assert opened == [ + (url, {"storage_options": options or None, "cache_dir": None, "remote_service": "auto"}) + ] def test_cache_dir_reaches_browser(monkeypatch, tmp_path): @@ -106,7 +108,7 @@ def run(app, **kwargs): monkeypatch.setattr(B2ViewApp, "run", run) cache_path = str(tmp_path / "cache") assert main([url, "--cache-dir", cache_path]) == 0 - assert opened == [(url, {"storage_options": None, "cache_dir": cache_path})] + assert opened == [(url, {"storage_options": None, "cache_dir": cache_path, "remote_service": "auto"})] def test_max_cache_bytes_reaches_browser(monkeypatch): @@ -129,4 +131,14 @@ def run(app, **kwargs): monkeypatch.setattr(B2ViewApp, "run", run) assert main([url, "--max-cache-bytes", "1048576"]) == 0 - assert opened == [(url, {"storage_options": None, "cache_dir": None, "max_cache_bytes": 1048576})] + assert opened == [ + ( + url, + { + "storage_options": None, + "cache_dir": None, + "max_cache_bytes": 1048576, + "remote_service": "auto", + }, + ) + ] From 31eda45be051561922fbe2fc89e75e5d2a8ee569 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sat, 3 Oct 2026 13:56:24 +0200 Subject: [PATCH 05/35] Validate shared cat2lite access and document repository browsing --- RELEASE_NOTES.md | 17 +++ doc/guides/remote_objects.md | 46 ++++++ doc/reference/classes.rst | 1 + doc/reference/remoterepository.rst | 29 ++++ doc/reference/remotestore.rst | 7 + plans/caterva2-access-improvements.md | 8 + tests/test_caterva2_gateway.py | 202 ++++++++++++++++++++++++++ 7 files changed, 310 insertions(+) create mode 100644 doc/reference/remoterepository.rst create mode 100644 tests/test_caterva2_gateway.py diff --git a/RELEASE_NOTES.md b/RELEASE_NOTES.md index 5d0ace25f..e6f5ba1fa 100644 --- a/RELEASE_NOTES.md +++ b/RELEASE_NOTES.md @@ -46,6 +46,23 @@ XXX version-specific blurb XXX requirements, environment precedence, caching, and diagnostics. Corrected the generated allocation declaration that caused an Apple Clang warning. +### Caterva2 repository access + +- `blosc2.open()` recognizes HTTP(S) dataset URLs such as + `http://localhost:8000/@public/group`, including deployment prefixes and IPv6. + String service URLs default to lazy access; explicit `URLPath` behavior is unchanged. +- Bare server/base URLs discover `api/roots`. One root opens directly; empty or + multiple-root services return a lazy `RemoteRepository` browsing facade. +- `remote_service="auto" | "caterva2" | "fsspec"` controls service recognition + independently of source format. The fsspec override preserves ordinary URLs + with literal `@` path components and disables service probes. +- Caterva2 `RemoteStore` groups expand lazily, including catalog mount boundaries; + catalog annotations are separate from source attributes. +- `b2view` browses these server/root URLs with the existing interface, including + bounded array/table previews. Remote table-wide transforms are disabled; + plotting requires an explicit bounded row window. Client caches are separate + from a shared cat2lite server cache; repository allowances are per root. + ## Changes from 4.14.0 to 4.14.1 Python-Blosc2 4.14.1 is a security and feature release introducing safe diff --git a/doc/guides/remote_objects.md b/doc/guides/remote_objects.md index 3f799142a..fd53702cf 100644 --- a/doc/guides/remote_objects.md +++ b/doc/guides/remote_objects.md @@ -33,6 +33,52 @@ call. See {doc}`remote_arrays` for local-source caching and selector details. ## Explore a hierarchy +### Caterva2 servers and shared caching gateways + +```python +with blosc2.open("http://localhost:8000") as repository: + print(repository.keys()) + +with blosc2.open("http://localhost:8000/@public/hdf5") as group: + print(group.keys()) + with group["d0/d1/a2"] as array: + values = array[:10, :10] +``` + +An HTTP(S) path component starting with `@` identifies a Caterva2 root. The +preceding path stays in the deployment base, e.g. `https://host/demo/@public`. +These string service references default to lazy access and return the selected +group, array, or table type. Explicit `URLPath` inputs retain their existing +lazy defaults. Dataset selectors are in the service URL, not `path=` or `::`. + +An ambiguous bare base URL probes `/api/roots` with a short timeout. A +single visible root opens directly. Zero or multiple roots return +{ref}`RemoteRepository`, with root names as children and independent, lazily +opened root owners. Prefer explicit root URLs for scripts: relative paths on +a bare service can change if its root count changes. Repository persistence and +materialization are unsupported; select a specific root for those operations. + +Use `remote_service="fsspec"` to disable recognition/probing for ordinary data +URLs containing `@` components, or for extensionless file sources. Use +`remote_service="caterva2"` to require a service without file fallback. Known +direct-format URLs are not probed in auto mode. Malformed/non-service responses +allow file fallback; authentication, connection, and server errors remain errors. +Discovery redirects are limited to the same origin. Authentication uses existing +{func}`blosc2.c2context`/`URLPath` facilities and is bound to opened owners. + +Caterva2 groups list immediate children on demand. A catalog mount not expanded +yet is not treated as empty. Direct descendant lookup also works without parent +expansion. `RemoteNode.catalog_attrs` contains catalog annotations separately +from source `attrs`. Existing servers may return recursive listings; their +response cost cannot be eliminated without a server-side pagination extension. + +A cat2lite gateway can share cached upstream chunks between independent clients. +Python-Blosc2's `cache_dir`, `cache_policy`, and `max_cache_bytes` still configure +the **client** cache. They do not configure the gateway cache, and `shared_cache` +does not mean “use the server cache.” Repository allowances apply per opened +root; retained payload totals can therefore exceed a single root's allowance. +Immutable-source assumptions and explicit invalidation requirements still apply. + - `store.keys()` or iteration lists immediate children. - `store["group/array"]` and `store["group"]["array"]` select the same leaf. - `store.attrs` exposes group attributes. diff --git a/doc/reference/classes.rst b/doc/reference/classes.rst index b4adc0b8b..d7882d697 100644 --- a/doc/reference/classes.rst +++ b/doc/reference/classes.rst @@ -156,6 +156,7 @@ container APIs above. remoteobject remotearray remotestore + remoterepository remotectable proxysource proxyndsource diff --git a/doc/reference/remoterepository.rst b/doc/reference/remoterepository.rst new file mode 100644 index 000000000..5c1cb1b29 --- /dev/null +++ b/doc/reference/remoterepository.rst @@ -0,0 +1,29 @@ +.. _RemoteRepository: + +RemoteRepository +================ + +``RemoteRepository`` is a read-only browsing facade returned by +``blosc2.open("https://host/base")`` when a Caterva2-compatible service has zero +or multiple accessible roots. A single root instead opens directly. + +The facade subclasses :ref:`RemoteStore` for hierarchy consumers such as b2view. +``keys()`` lists roots without opening them; ``repository["@public"]`` returns +an independent group handle. Direct descendant lookup is supported. Closing +the repository leaves previously returned child handles usable. Alias handles +from ``repository[""]`` share root owners and lifetime accounting. + +Cache policy and allowance are per root. ``cache_bytes`` and ``metadata_bytes`` +sum opened roots; ``traffic`` counts their reads and excludes the initial roots +probe. Source descriptors contain no authentication token. Root owners bind the +authentication context at repository creation and use existing isolated cache +identities. + +Repository persistence, materialization, refresh, and root-level shared sparse +cache opening are intentionally unsupported. Select a specific root for those +operations. ``get_info`` at a root name reports the roots registry metadata; +select/open that root for its actual source attributes. No synthetic empty path +is sent to the server's info endpoint. + +.. autoclass:: blosc2.RemoteRepository + :members: keys, get_info, close, source, attrs, cache_policy, max_cache_bytes, cache_bytes, metadata_bytes, traffic diff --git a/doc/reference/remotestore.rst b/doc/reference/remotestore.rst index b5c7884f9..2a4ed1890 100644 --- a/doc/reference/remotestore.rst +++ b/doc/reference/remotestore.rst @@ -8,6 +8,13 @@ RemoteStore source session: a B2Z archive, a native HDF5 index, or a Zarr store. Zarr listing remains lazy. +Caterva2 groups also list lazily, through explicit ``URLPath`` inputs or service +URL strings passed to ``blosc2.open``. Each group is expanded independently, +including virtual catalog mount boundaries. ``RemoteNode.catalog_attrs`` exposes +catalog annotations separately from source attrs. See +:doc:`../guides/remote_objects` for URL discovery and :ref:`RemoteRepository` +for browsing services with multiple roots. + The default ``CachePolicy.MEMORY`` shares a 256 MiB allowance across all leaves. Set ``max_cache_bytes`` to a positive integer to change it. ``CachePolicy.NONE`` retains no payload and rejects a limit. Passing ``cache_dir`` selects DISK when diff --git a/plans/caterva2-access-improvements.md b/plans/caterva2-access-improvements.md index 006d33405..7740ab149 100644 --- a/plans/caterva2-access-improvements.md +++ b/plans/caterva2-access-improvements.md @@ -38,6 +38,14 @@ Implementation record (updated as milestones land): and plotting requires an explicit bounded row window. Fixed metadata display for remote sources that do not report compressed size. All viewer/model regressions, including headless Textual and fresh-process decoders: 155 passed. +- M5: added opt-in `tests/test_caterva2_gateway.py` against a real debug + cat2lite server and deterministic loopback B2ND/HDF5/Zarr/Parquet sources, + including headless b2view and independent no-client-cache subprocesses. + Measured large-array upstream reads: 72,950 bytes / 3 GETs after warming; + second client unchanged; after server restart 81,142 bytes / 4 GETs (8 KiB + source headers, no warm payload refetch). API/viewer docs and release notes + updated. Full default offline suite: 10,640 passed, 36 skipped. Gateway + acceptance passed with CAT2LITE_SERVER set; it is skipped by default. ## 1. Goal diff --git a/tests/test_caterva2_gateway.py b/tests/test_caterva2_gateway.py new file mode 100644 index 000000000..459ef42e1 --- /dev/null +++ b/tests/test_caterva2_gateway.py @@ -0,0 +1,202 @@ +"""Opt-in cross-project acceptance against an actual cat2lite server. + +Run with CAT2LITE_SERVER=/absolute/path/to/cat2lite-server in the blosc2 env. +All upstream data is deterministic and served on loopback; no public network. +""" + +import hashlib +import json +import os +import select +import subprocess +import sys +import threading +from contextlib import contextmanager +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from urllib.parse import urlsplit + +import numpy as np +import pytest + +import blosc2 + +pytestmark = pytest.mark.skipif( + not os.environ.get("CAT2LITE_SERVER"), reason="set CAT2LITE_SERVER for gateway acceptance" +) + + +@contextmanager +def gateway(base, catalog): + config = base / "server.toml" + config.write_text( + f'[server]\nlisten = "127.0.0.1:0"\npython = {json.dumps(sys.executable)}\n' + f"[remote]\ncache_dir = {json.dumps(str(base / 'server-cache'))}\n" + ) + process = subprocess.Popen( + [os.environ["CAT2LITE_SERVER"], "--config", str(config), str(catalog)], + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + text=True, + ) + try: + assert select.select([process.stdout], [], [], 20)[0], "cat2lite startup timed out" + line = process.stdout.readline() + while line and not line.startswith("listening on "): + line = process.stdout.readline() + assert line.startswith("listening on "), line + address = line.removeprefix("listening on ").strip() + yield address if address.startswith("http") else "http://" + address + finally: + process.terminate() + process.wait(timeout=15) + for stream in (process.stdout, process.stderr): + stream.close() + + +def test_real_catalog_browsing_and_shared_server_cache(tmp_path): + h5py = pytest.importorskip("h5py") + zarr = pytest.importorskip("zarr") + pa = pytest.importorskip("pyarrow") + pq = pytest.importorskip("pyarrow.parquet") + from blosc2.b2view.model import StoreBrowser + + data = tmp_path / "upstream" + data.mkdir() + expected = np.arange(60, dtype="i4").reshape(6, 10) + with h5py.File(data / "hierarchy.h5", "w") as file: + file.create_dataset("group/array", data=expected) + file.create_dataset("scalar", data=np.int32(42)) + group = zarr.open_group(data / "hierarchy.zarr", mode="w", zarr_format=2) + group.create_array("group/array", data=expected, chunks=(3, 5)) + zarr.consolidate_metadata(data / "hierarchy.zarr") + pq.write_table( + pa.table({"ident": list(range(20)), "value": [i * 3 for i in range(20)]}), data / "readings.parquet" + ) + # Larger than small-source prefetch thresholds, with incompressible chunks. + large = np.random.default_rng(7).integers(0, 2**31, (1024, 1024), dtype="i4") + frame = blosc2.asarray(large, chunks=(128, 128), blocks=(32, 32)).to_cframe() + (data / "large.b2nd").write_bytes(frame) + files = { + "/" + path.relative_to(data).as_posix(): path.read_bytes() + for path in data.rglob("*") + if path.is_file() + } + stats = {"large_bytes": 0, "large_gets": 0} + + class Source(BaseHTTPRequestHandler): + def serve(self, body): + path = urlsplit(self.path).path + if path not in files: + self.send_error(404) + return + payload = files[path] + length = len(payload) + offset, stop = 0, length + requested = self.headers.get("Range") + if requested: + start, _, end = requested.removeprefix("bytes=").partition("-") + offset = int(start) if start else max(0, length - int(end)) + stop = min(length, int(end) + 1) if start and end else length + self.send_response(206 if requested else 200) + self.send_header("Content-Length", str(stop - offset)) + self.send_header("Accept-Ranges", "bytes") + self.send_header("ETag", '"' + hashlib.sha256(payload).hexdigest() + '"') + if requested: + self.send_header("Content-Range", f"bytes {offset}-{stop - 1}/{length}") + self.end_headers() + if body: + if path == "/large.b2nd": + stats["large_bytes"] += stop - offset + stats["large_gets"] += 1 + self.wfile.write(payload[offset:stop]) + + def do_GET(self): + self.serve(True) + + def do_HEAD(self): + self.serve(False) + + def log_message(self, *args): + pass + + server = ThreadingHTTPServer(("127.0.0.1", 0), Source) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + try: + upstream = f"http://127.0.0.1:{server.server_port}" + catalog = tmp_path / "repo.catl" + catalog.write_text( + "entries:\n" + + "".join( + f" {name}: {upstream}/{source}\n" + for name, source in [ + ("hdf5", "hierarchy.h5"), + ("zarr", "hierarchy.zarr"), + ("table", "readings.parquet"), + ("large", "large.b2nd"), + ] + ) + ) + with gateway(tmp_path, catalog) as url: + with StoreBrowser(url) as browser: + assert [node.name for node in browser.list_children()] == ["hdf5", "large", "table", "zarr"] + assert browser.list_children("/hdf5")[0].name == "group" + assert browser.get_info("/hdf5/scalar").metadata["shape"] == () + np.testing.assert_array_equal( + browser.preview("/hdf5/group/array", slices=(slice(2), slice(3))), expected[:2, :3] + ) + np.testing.assert_array_equal( + browser.preview("/zarr/group/array", slices=(slice(2), slice(3))), expected[:2, :3] + ) + assert list(browser.preview("/table", stop=3)["data"]["ident"]) == [0, 1, 2] + import asyncio + + from blosc2.b2view.app import B2ViewApp + + async def viewer_acceptance(): + app = B2ViewApp(url, start_path="/hdf5/group/array") + async with app.run_test(size=(120, 40)) as pilot: + for _ in range(200): + if app.table_page and app.table_page["columns"]: + break + await pilot.pause(0.02) + assert app.selected_path == "/hdf5/group/array" + assert app.table_page + assert app.table_page["columns"] + np.testing.assert_array_equal( + app.table_page["data"]["0"], expected[: app.table_page["stop"], 0] + ) + app.wait_for_close() + + asyncio.run(viewer_acceptance()) + code = """ +import sys +import numpy as np +import blosc2 +with blosc2.open(sys.argv[1] + '/@public/large', cache_policy=blosc2.CachePolicy.NONE) as array: + value = array[:2, :3] + assert value.shape == (2, 3) + print(value.tolist()) +""" + + def independent_read(): + result = subprocess.run( + [sys.executable, "-c", code, url], capture_output=True, text=True, timeout=30 + ) + assert result.returncode == 0, result.stderr + assert json.loads(result.stdout) == large[:2, :3].tolist() + + independent_read() + warm = stats.copy() + independent_read() + assert stats == warm, "second no-cache process should reuse server payload" + assert warm["large_bytes"] < large.nbytes // 2 + with gateway(tmp_path, catalog) as url: + independent_read() + # A restart may re-read source headers, but not the warm data chunk. + assert stats["large_bytes"] - warm["large_bytes"] < 16 << 10 + print("gateway cache evidence:", warm, "after restart:", stats) + finally: + server.shutdown() + server.server_close() + thread.join(timeout=5) From 6051acb81e65880dbc49ebb7b5d8837a080d61d3 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sat, 3 Oct 2026 14:10:05 +0200 Subject: [PATCH 06/35] Harden Caterva2 discovery and remote viewer lifecycle after review --- doc/guides/remote_objects.md | 5 ++ plans/caterva2-access-improvements.md | 28 +++++++++-- src/blosc2/b2view/app.py | 33 +++++++++++++ src/blosc2/b2view/model.py | 13 +++-- src/blosc2/caterva2_url.py | 11 ++++- src/blosc2/remote_repository.py | 12 +++++ src/blosc2/remote_store.py | 60 +++++++++++++++++----- src/blosc2/schunk.py | 7 ++- tests/b2view/test_caterva2.py | 69 ++++++++++++++++++++++++++ tests/test_caterva2_access.py | 71 +++++++++++++++++++++++++-- tests/test_remote_caterva2.py | 7 +++ 11 files changed, 291 insertions(+), 25 deletions(-) diff --git a/doc/guides/remote_objects.md b/doc/guides/remote_objects.md index fd53702cf..d4f415f6a 100644 --- a/doc/guides/remote_objects.md +++ b/doc/guides/remote_objects.md @@ -71,6 +71,11 @@ yet is not treated as empty. Direct descendant lookup also works without parent expansion. `RemoteNode.catalog_attrs` contains catalog annotations separately from source `attrs`. Existing servers may return recursive listings; their response cost cannot be eliminated without a server-side pagination extension. +Caterva2 discovery retains at most 10,000 nodes per owner by default, rejects +list responses above 100,000 entries, and rejects discovery bodies above 8 MiB +before JSON decoding. Existing transports buffer those discovery bodies before +the size check. Explicit hierarchy summaries (`store.info`) visit groups and can +contact mounted sources; failures are reported as incomplete listings. A cat2lite gateway can share cached upstream chunks between independent clients. Python-Blosc2's `cache_dir`, `cache_policy`, and `max_cache_bytes` still configure diff --git a/plans/caterva2-access-improvements.md b/plans/caterva2-access-improvements.md index 7740ab149..60fc91517 100644 --- a/plans/caterva2-access-improvements.md +++ b/plans/caterva2-access-improvements.md @@ -1,6 +1,8 @@ # Caterva2 access improvements for Python-Blosc2 and b2view -Status: implementation proposal; no implementation implied by this document. +Status: M0–M5 implemented and validated on `cat2-improvements`; final review +fixes included. The sections below retain the delivery design and acceptance +criteria; the implementation record is the authoritative completion evidence. Implementation record (updated as milestones land): @@ -46,6 +48,24 @@ Implementation record (updated as milestones land): source headers, no warm payload refetch). API/viewer docs and release notes updated. Full default offline suite: 10,640 passed, 36 skipped. Gateway acceptance passed with CAT2LITE_SERVER set; it is skipped by default. +- Final review: old eager-list cache snapshots rediscover listings without + discarding retained payload; direct paths reject query/fragment/escape + ambiguity; missing nodes raise KeyError; nonlazy groups/tables fail clearly. + Added default node/list/discovery-response limits, avoided full-registry copies + per node inspection, and isolated failed sources in hierarchy summaries. + Failed viewer nodes remain expandable for retry. Row-window reads are prepared + in background workers and published only if selection/session still match; + Caterva2 windows materialize explicitly bounded selections for local plotting. +- Final validation: 10,646 default-suite tests passed, 37 skipped (including + the opt-in gateway test); 213 focused access/Caterva2/viewer/model tests passed + with headless TUI cases enabled; real cat2lite gateway acceptance passed again + with the same cache traffic measurements. Ruff lint/format and diff checks + passed. HTML docs built in external temporary storage with notebook execution + disabled, generated/stashed doc copies excluded, and type-comment introspection + disabled; the wider documentation build still emits many autosummary/theme/ + cross-reference warnings (754), so this is not a warnings-clean documentation + build. No Python native-extension rebuild was required; cat2lite's three debug + binaries were rebuilt for cross-project acceptance. ## 1. Goal @@ -71,7 +91,7 @@ b2view http://localhost:8000/@public b2view https://cat2.cloud/demo/@public/example ``` -The examples above describe proposed behavior. Preserve existing direct-file +The examples above describe implemented behavior. Preserve existing direct-file opening, including HTTP B2ND/B2Z, HDF5, Zarr, and Parquet sources. Architecture: @@ -166,9 +186,9 @@ Rules: An ordinary file server can legitimately contain `@` path components. Provide an explicit bypass for Caterva2 recognition and repository probing. -Proposed API decision for M0: a narrowly scoped selector such as +Selected API: the narrowly scoped selector `remote_service="auto" | "caterva2" | "fsspec"` on `blosc2.open`. -This name is provisional. Audit existing options before adding it; do not +This selects the service backend independently of data format; do not overload `source_format`, which currently describes data formats. - `auto`: root-marker shorthand, known-file dispatch, then eligible base probe. diff --git a/src/blosc2/b2view/app.py b/src/blosc2/b2view/app.py index a34809d06..4973e7162 100644 --- a/src/blosc2/b2view/app.py +++ b/src/blosc2/b2view/app.py @@ -3470,6 +3470,13 @@ def _view_plot_range(self, span: tuple[int, int] | None) -> None: def _enter_row_window(self, start: int, stop: int, *, backend: str) -> None: """Replace the grid with a locked [start:stop] window (in place).""" if backend == "ctable": + if self._remote: + self._remote_request += 1 + self._prepare_remote_window( + self._remote_session, self._remote_request, self.browser, self.selected_path, start, stop + ) + self.query_one("#metadata", Static).update("Loading row window…") + return try: self.browser.set_row_window(self.selected_path, start, stop) except Exception as exc: # pragma: no cover - defensive @@ -3483,6 +3490,32 @@ def _enter_row_window(self, start: int, stop: int, *, backend: str) -> None: self._reload_row_window(0) self.notify(f"Locked to rows {start}:{stop} · esc to unlock") + @work(thread=True, exit_on_error=False) + def _prepare_remote_window(self, session, request, browser, path, start, stop): + try: + with browser.io_lock: + if session != self._remote_session or request != self._remote_request: + return + view = browser.prepare_row_window(path, start, stop) + self._deliver_remote( + session, self._finish_remote_window, request, browser, path, start, stop, view, None + ) + except Exception as exc: + self._deliver_remote( + session, self._finish_remote_window, request, browser, path, start, stop, None, exc + ) + + def _finish_remote_window(self, request, browser, path, start, stop, view, error): + if request != self._remote_request or path != self.selected_path or browser is not self.browser: + return + if error is not None: + self._remote_error(error) + return + browser.install_row_window(path, view) + self.row_window = (start, stop) + self._reload_row_window(0) + self.notify(f"Locked to rows {start}:{stop} · esc to unlock") + def _exit_row_window(self) -> None: """Unlock the row window and restore the full grid.""" if self.row_window is None: diff --git a/src/blosc2/b2view/model.py b/src/blosc2/b2view/model.py index e07634c3f..41c5a4e57 100644 --- a/src/blosc2/b2view/model.py +++ b/src/blosc2/b2view/model.py @@ -399,7 +399,7 @@ def list_children(self, path: str = "/") -> list[NodeInfo]: path=self.normalize_path(path.rstrip("/") + "/" + name), name=name, kind=kind, - has_children=kind in {"group", "remote_store"}, + has_children=kind in {"group", "remote_store", "unavailable"}, ) ) self._remote_child_counts[path] = len(children) @@ -1058,6 +1058,10 @@ def set_row_window(self, path: str, start: int, stop: int) -> int: currently visible (so it composes over any active row filter). Paging then cannot leave the range because the view reports only its own rows. """ + return self.install_row_window(path, self.prepare_row_window(path, start, stop)) + + def prepare_row_window(self, path: str, start: int, stop: int): + """Read a bounded window without publishing it (may perform remote I/O).""" path = self.normalize_path(path) # The sort view already incorporates any filter, so prefer it; fall back # to the bare filter view, then the base table. @@ -1067,8 +1071,11 @@ def set_row_window(self, path: str, start: int, stop: int) -> int: base = self._filter_views[path] else: base = self._get_object(path) - view = base.slice(start, stop, copy=False) - self._window_views[path] = view + return base.slice(start, stop, copy=isinstance(base, blosc2.RemoteCTable)) + + def install_row_window(self, path: str, view) -> int: + """Publish an already prepared window without remote I/O.""" + self._window_views[self.normalize_path(path)] = view return len(view) def clear_row_window(self, path: str) -> None: diff --git a/src/blosc2/caterva2_url.py b/src/blosc2/caterva2_url.py index 761e61541..705ca01dd 100644 --- a/src/blosc2/caterva2_url.py +++ b/src/blosc2/caterva2_url.py @@ -77,6 +77,9 @@ def service_probe_candidate(value): return False suffixes = ( ".b2nd", + ".b2", + ".b2t", + ".b2o", ".b2z", ".b2d", ".b2f", @@ -137,10 +140,14 @@ def discover_service(value, *, required=False, auth_token=None): target = urljoin(url, response.headers.get("location", "")) redirect = urlsplit(target) - if (redirect.scheme, redirect.hostname, redirect.port) != ( + if ( + redirect.scheme, + redirect.hostname, + redirect.port or (443 if redirect.scheme == "https" else 80), + ) != ( parsed.scheme, parsed.hostname, - parsed.port, + parsed.port or (443 if parsed.scheme == "https" else 80), ): raise ValueError("Caterva2 discovery cannot redirect credentials to another origin") if redirect.username is not None or redirect.password is not None: diff --git a/src/blosc2/remote_repository.py b/src/blosc2/remote_repository.py index 986730fdb..284f8a018 100644 --- a/src/blosc2/remote_repository.py +++ b/src/blosc2/remote_repository.py @@ -177,8 +177,13 @@ def traffic(self): @property def mutable(self): + self._ensure_open() return False + @mutable.setter + def mutable(self, value): + raise NotImplementedError("Repository persistence is unsupported; select a specific root") + @property def is_cache_mutable(self): self._ensure_open() @@ -190,9 +195,16 @@ def read_cached(self, path, item=(), *, nchunk=None): name, _, suffix = path.strip("/").partition("/") return self._root(name).read_cached(suffix, item, nchunk=nchunk) + def read_cached_table(self, operation): + raise NotImplementedError("Select a specific root for cached table operations") + def save(self, *args, **kwargs): raise NotImplementedError("Repository persistence is unsupported; save a specific root") + @classmethod + def with_sparse_cache(cls, *args, **kwargs): + raise NotImplementedError("Repository shared sparse caching requires selecting a specific root") + def materialize(self, *args, **kwargs): raise NotImplementedError("Repository materialization is unsupported; select a specific root") diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 57b4d529e..62389e41c 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -463,7 +463,7 @@ def __init__( # noqa: C901 self.batch_validator = _batch_validator self.filesystem_resolver = _filesystem_resolver self.manifest_validator = _manifest_validator - self.max_nodes = _max_nodes + self.max_nodes = 10000 if self.format == "caterva2" and _max_nodes is None else _max_nodes self.metadata_bytes = 0 self.restoring = False self.filesystem = None @@ -537,6 +537,12 @@ def _restore_manifest(self, manifest): # noqa: C901 raise ValueError("Invalid RemoteStore child list") self.attrs = manifest["attrs"] self.listed = manifest["listed"] + if self.format == "caterva2" and self.metadata.get("caterva2_listing_version") != 2: + # Earlier snapshots marked all discovered groups as fully listed, + # including unexpanded catalog mounts. Retain payload, rediscover + # listings rather than permanently restoring false empty groups. + self.listed = {} + self.metadata["caterva2_listing_version"] = 2 self.notice = manifest.get("notice") if self.format == "b2z" and self.archive is None: from blosc2.b2z_source import B2ZArchive @@ -982,9 +988,12 @@ def _caterva2_get(self, endpoint, path): response = client.get(url, headers=_auth_headers(self.caterva2.auth_token)) response.raise_for_status() self.traffic.charge(len(response.content)) + if len(response.content) > 8 << 20: + raise ValueError("Caterva2 discovery response exceeds 8 MiB") return response.json() def _open_caterva2(self): + self.metadata["caterva2_listing_version"] = 2 self._discover_caterva2_node(self.root) if self.nodes[self.root][0] not in {"group", "ctable", "ndarray"}: raise ValueError("Caterva2 source root is not an array, group, or CTable") @@ -994,11 +1003,16 @@ def _discover_caterva2_node(self, full): # placeholders. An unlisted mount must not be mistaken for an empty group. if full in self.attrs: return - info = ( - self.root_info - if full == self.root and self.root_info is not None - else self._caterva2_get("info", full) - ) + try: + info = ( + self.root_info + if full == self.root and self.root_info is not None + else self._caterva2_get("info", full) + ) + except Exception as error: + if getattr(getattr(error, "response", None), "status_code", None) == 404: + raise KeyError(full) from error + raise if not isinstance(info, dict): raise ValueError("Invalid Caterva2 info response") kind = self._caterva2_kind(info) @@ -1008,12 +1022,21 @@ def _discover_caterva2_node(self, full): attrs = dict(info.get("attrs") or {}) annotations = dict(info.get("catalog_attrs") or {}) # Validate all potentially failing work before changing the registry. - previous = self.nodes.copy() + # Only this node and its ancestors can change. Do not copy a complete + # large catalog for each child's metadata request. + ancestors = [full] + while ancestors[-1]: + ancestors.append(ancestors[-1].rpartition("/")[0]) + previous = {key: self.nodes.get(key) for key in ancestors} try: self.nodes.pop(full, None) self._add(full, kind, value) except BaseException: - self.nodes = previous + for key, old in previous.items(): + if old is None: + self.nodes.pop(key, None) + else: + self.nodes[key] = old raise self.attrs[full] = attrs if annotations: @@ -1023,6 +1046,8 @@ def _list_caterva2(self, full): leaves = self._caterva2_get("list", full) if not isinstance(leaves, list) or any(not isinstance(path, str) for path in leaves): raise ValueError("Invalid Caterva2 list response") + if len(leaves) > 100000: + raise ValueError("Caterva2 listing exceeds the 100000-entry discovery limit") children = set() for relative in leaves: if not relative: @@ -1048,6 +1073,8 @@ def _path(self, path): raise TypeError("RemoteStore paths must be strings") relative = path.strip("/") self._validate(relative) + if self.format == "caterva2" and any(c in relative for c in "%?#"): + raise ValueError("Unsafe Caterva2 dataset path") return "/".join(p for p in (self.root, relative) if p) def resolve(self, path): @@ -2966,17 +2993,26 @@ def info_items(self) -> list[tuple[str, object]]: entries = {} unavailable = [] pending = [""] + listing_errors = (OSError, KeyError, ValueError) + if self._owner.format == "caterva2": + listing_errors += (blosc2.c2array._httpx().HTTPError,) while pending: - path, _ = self._resolve(pending.pop()) + requested = pending.pop() try: + path, _ = self._resolve(requested) children = self._owner.list_children(path) - except OSError as exc: - unavailable.append(f"/{path}: {exc}") + except listing_errors as exc: + unavailable.append(f"/{requested}: {exc}") continue for child in children: relative = child[len(root) + 1 :] if root else child owner_path = child[len(self._owner.root) + 1 :] if self._owner.root else child - kind = self._owner.nodes[self._owner.resolve(owner_path)][0] + try: + kind = self._owner.nodes[self._owner.resolve(owner_path)][0] + except listing_errors as exc: + unavailable.append(f"/{relative}: {exc}") + entries[relative] = " [unavailable]" + continue entries[relative] = f" [{kind}]" if kind == "group": pending.append(relative) diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index 14e7305bf..9b559cde2 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2159,7 +2159,12 @@ def _open_non_lazy_c2( raise NotImplementedError("cache_dir and cache_path for a Caterva2 array require lazy=True") if max_concurrency is not None: raise NotImplementedError("max_concurrency is only supported with lazy=True") - return blosc2.C2Array(urlpath.path, urlbase=urlpath.urlbase, auth_token=urlpath.auth_token) + metadata = blosc2.c2array.info(urlpath.path, urlpath.urlbase, auth_token=urlpath.auth_token) + if "shape" not in metadata or "dtype" not in metadata: + raise NotImplementedError("Caterva2 group/table access requires lazy=True") + return blosc2.C2Array( + urlpath.path, urlbase=urlpath.urlbase, auth_token=urlpath.auth_token, _meta=metadata + ) def _open_c2_urlpath(urlpath: blosc2.URLPath, mode: str, offset: int, kwargs: dict): # noqa: C901 diff --git a/tests/b2view/test_caterva2.py b/tests/b2view/test_caterva2.py index b1e538585..a879ca03d 100644 --- a/tests/b2view/test_caterva2.py +++ b/tests/b2view/test_caterva2.py @@ -49,6 +49,21 @@ def test_repository_browser_and_broken_sibling(caterva2_source): # noqa: F811 ("empty", "group"), ("table", "ctable"), ] + assert next(node for node in children if node.name == "broken").has_children + stats["groups"]["@public/mount/broken"] = [] + assert browser.list_children("/mount/broken") == [] + + +def test_published_extensions_do_not_select_direct_file_backends(caterva2_source): # noqa: F811 + base, _, _, stats = caterva2_source + stats["aliases"]["@public/group.h5"] = "@public/group" + stats["aliases"]["@public/table.parquet"] = "@public/table" + with StoreBrowser(base + "@public/group.h5") as browser: + assert browser.is_tree + assert [node.name for node in browser.list_children()] == ["array", "table"] + with StoreBrowser(base + "@public/table.parquet") as browser: + assert browser.kind("/") == "ctable" + assert browser.preview("/", stop=2)["nrows"] == 12 @pytest.mark.tui @@ -80,3 +95,57 @@ def test_cli_service_override(): args = build_parser().parse_args(["https://host/@data/a.h5", "--remote-service", "fsspec"]) assert args.remote_service == "fsspec" + + +@pytest.mark.tui +@pytest.mark.asyncio +async def test_remote_table_window_is_background_and_stale_results_are_discarded( + caterva2_source, # noqa: F811 + monkeypatch, +): + import threading + + from blosc2.b2view.app import B2ViewApp + + base, _, _, _ = caterva2_source + entered, release = threading.Event(), threading.Event() + original = StoreBrowser.prepare_row_window + + def slow(self, path, start, stop): + entered.set() + assert release.wait(5) + return original(self, path, start, stop) + + monkeypatch.setattr(StoreBrowser, "prepare_row_window", slow) + app = B2ViewApp(base, start_path="/mount/table") + try: + async with app.run_test(size=(120, 40)) as pilot: + await wait_until( + pilot, lambda: app.table_page is not None and app.table_page.get("source_kind") == "ctable" + ) + app._enter_row_window(2, 5, backend="ctable") + await wait_until(pilot, entered.is_set) + # A slow bounded table fetch must not block event handling. + await pilot.press("tab") + app.update_panels("/mount") + release.set() + await wait_until( + pilot, lambda: app._selected_info is not None and app._selected_info.path == "/mount" + ) + assert not app.browser.get_row_window("/mount/table") + app.update_panels("/mount/table") + await wait_until( + pilot, lambda: app.table_page is not None and app.table_page.get("source_kind") == "ctable" + ) + app._enter_row_window(2, 5, backend="ctable") + await wait_until( + pilot, + lambda: app.row_window == (2, 5) and app.table_page and bool(app.table_page["columns"]), + ) + assert app.table_page["nrows"] == 3 + np.testing.assert_array_equal(app.table_page["data"]["ident"], [2, 3, 4]) + plotted = app.browser.plot_series("/mount/table", column="ident") + assert plotted["n"] == 3 + finally: + release.set() + app.wait_for_close() diff --git a/tests/test_caterva2_access.py b/tests/test_caterva2_access.py index 8e4c6e6bd..265aa1dca 100644 --- a/tests/test_caterva2_access.py +++ b/tests/test_caterva2_access.py @@ -157,9 +157,7 @@ def test_repository_is_lazy_and_children_outlive_it(caterva2_source, tmp_path): assert repo.keys() == ["@broken", "@public"] assert stats["requests"] == ["/api/roots"] assert repo.kind("@broken") == "group" - import httpx - - with pytest.raises(httpx.HTTPStatusError): + with pytest.raises(KeyError): repo["@broken"] with repo[""] as alias: child = alias["@public/mount/array"] @@ -200,6 +198,7 @@ def test_direct_format_does_not_probe(monkeypatch): ) for url in ( "https://host/a.b2nd", + "https://host/a.b2", "https://host/a.zarr/", "https://host/a.h5::/group", "https://host/a.b2z?version=2", @@ -281,3 +280,69 @@ def test_repository_freezes_auth_context(caterva2_source): # noqa: F811 with repo["@public"] as root: assert root.keys() == ["mount"] assert set(stats["cookies"]) == {"alice=secret"} + + +def test_nonlazy_groups_and_lookup_escaping_fail_clearly(caterva2_source): # noqa: F811 + base, _, _, stats = caterva2_source + with pytest.raises(NotImplementedError, match="requires lazy=True"): + blosc2.open(base + "@public", lazy=False) + with blosc2.open(base + "@public") as store: + before = stats["requests"].copy() + for path in ("mount/array?x=1", "mount/array#x", "mount/array%2F"): + with pytest.raises(ValueError, match="Unsafe"): + store.get_info(path) + assert stats["requests"] == before + with pytest.raises(KeyError): + store["nonexistent"] + + +def test_old_catalog_listing_cache_is_rediscovered(caterva2_source, tmp_path): # noqa: F811 + base, _, _, _ = caterva2_source + cache = tmp_path / "cache" + with blosc2.open(base + "@public", cache_dir=cache) as store: + with store["mount"] as mount: + assert mount.keys() == ["array", "empty", "table"] + manifest = store._owner.disk.load() + manifest["metadata"].pop("caterva2_listing_version") + manifest["listed"]["@public/mount"] = [] + store._owner.disk.publish(manifest) + with blosc2.open(base + "@public", cache_dir=cache) as reopened: + with reopened["mount"] as mount: + assert mount.keys() == ["array", "empty", "table"] + + +def test_equivalent_openers_share_cache_identity_and_auth_does_not(caterva2_source, tmp_path): # noqa: F811 + base, _, _, stats = caterva2_source + cache = tmp_path / "cache" + folders = [] + for token, source in [ + ("alice=secret", base + "@public/group"), + ("alice=secret", blosc2.URLPath("@public/group", urlbase=base)), + ("bob=secret", base + "@public/group"), + ]: + with blosc2.c2context(auth_token=token), blosc2.open(source, lazy=True, cache_dir=cache) as store: + folders.append(store._owner.disk.path) + with store["array"] as remote: + remote[:1, :2] + assert folders[0] == folders[1] + assert folders[0] != folders[2] + assert stats["fetches"] == 2 + + +def test_hierarchy_summary_isolates_failed_sources(caterva2_source): # noqa: F811 + base, _, _, stats = caterva2_source + stats["groups"]["@public/mount"].append("broken") + with blosc2.open(base + "@public") as store: + summary = str(store.info) + assert "listing incomplete" in summary + assert "[unavailable]" in summary + assert "array" in summary + + +def test_default_discovery_limits(caterva2_source): # noqa: F811 + base, _, _, stats = caterva2_source + stats["groups"]["@public"] = ["mount/array"] * 100001 + with blosc2.open(base + "@public") as store: + assert store._owner.max_nodes == 10000 + with pytest.raises(ValueError, match="100000-entry"): + store.keys() diff --git a/tests/test_remote_caterva2.py b/tests/test_remote_caterva2.py index 3f72530dd..f1c728dc7 100644 --- a/tests/test_remote_caterva2.py +++ b/tests/test_remote_caterva2.py @@ -94,6 +94,7 @@ def caterva2_source(request): "@public/group": ["array", "table"], }, "fail_list": None, + "aliases": {}, } def array_info(): @@ -142,6 +143,12 @@ def do_GET(self): path = urllib.parse.unquote(parsed.path) path = path.removeprefix("/demo") stats["requests"].append(path) + for endpoint in ("info", "list", "fetch"): + prefix = f"/api/{endpoint}/" + if path.startswith(prefix): + key = path.removeprefix(prefix) + path = prefix + stats["aliases"].get(key, key) + break if path == "/api/roots": if stats["roots_status"] != 200: self.send_error(stats["roots_status"]) From 028821cab2fcec3bf721675a1bbc01924bd0d9af Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sat, 3 Oct 2026 14:19:55 +0200 Subject: [PATCH 07/35] Display full Caterva2 paths using root naming conventions in b2view --- src/blosc2/b2view/model.py | 25 +++++++++++++++++++++++-- src/blosc2/b2view/render.py | 2 +- tests/b2view/test_caterva2.py | 34 ++++++++++++++++++++++++++++++++++ 3 files changed, 58 insertions(+), 3 deletions(-) diff --git a/src/blosc2/b2view/model.py b/src/blosc2/b2view/model.py index 41c5a4e57..f79e10245 100644 --- a/src/blosc2/b2view/model.py +++ b/src/blosc2/b2view/model.py @@ -131,6 +131,7 @@ class ObjectInfo: kind: str metadata: dict[str, Any] user_attrs: dict[str, Any] | None = None + display_path: str | None = None @dataclass @@ -464,7 +465,27 @@ def get_info(self, path: str) -> ObjectInfo: if user_attrs is None and self.is_tree: store_attrs = getattr(self.store, "attrs", getattr(self.store, "vlmeta", None)) user_attrs = self._attrs_dict(store_attrs) - return ObjectInfo(path=path, kind=kind, metadata=metadata, user_attrs=user_attrs) + return ObjectInfo( + path=path, + kind=kind, + metadata=metadata, + user_attrs=user_attrs, + display_path=self._display_path(path), + ) + + def _display_path(self, path): + """Keep service paths fully qualified without changing browser navigation.""" + if isinstance(self.store, (blosc2.RemoteStore, blosc2.RemoteArray, blosc2.RemoteCTable)): + source = self.store.source + if source["kind"] == "caterva2": + full = "/".join(part for part in (source["path"], path.strip("/")) if part) + return full if full.startswith("@") else "/" + full + if source["kind"] == "caterva2_repository" and path.startswith("/@"): + return path.lstrip("/") + elif isinstance(self.store, blosc2.C2Array): + full = self.store.path.strip("/") + return full if full.startswith("@") else "/" + full + return path def _remote_info(self, path): node = self.store.get_info(path) @@ -486,7 +507,7 @@ def _remote_info(self, path): metadata["preview" if node.kind == "unsupported" else "notice"] = node.diagnostic if isinstance(self.store, blosc2.RemoteRepository) and path == "/" and not self.store.keys(): metadata["notice"] = "No accessible roots" - return ObjectInfo(path, node.kind, metadata, attrs) + return ObjectInfo(path, node.kind, metadata, attrs, display_path=self._display_path(path)) def preview( self, diff --git a/src/blosc2/b2view/render.py b/src/blosc2/b2view/render.py index e304aa6e1..2be0b5b68 100644 --- a/src/blosc2/b2view/render.py +++ b/src/blosc2/b2view/render.py @@ -17,7 +17,7 @@ def make_metadata_renderable(info, *, show_path=True): table.add_column("key", style="bold cyan", no_wrap=True) table.add_column("value") if show_path: - table.add_row("path", info.path) + table.add_row("path", info.display_path or info.path) table.add_row("kind", info.kind) for key, value in info.metadata.items(): table.add_row(str(key), _format_metadata_value(value)) diff --git a/tests/b2view/test_caterva2.py b/tests/b2view/test_caterva2.py index a879ca03d..5990914f1 100644 --- a/tests/b2view/test_caterva2.py +++ b/tests/b2view/test_caterva2.py @@ -97,6 +97,40 @@ def test_cli_service_override(): assert args.remote_service == "fsspec" +@pytest.mark.parametrize( + ("target", "relative", "expected"), + [ + ("", "/", "@public"), + ("", "/mount/table", "@public/mount/table"), + ("@public", "/mount/table", "@public/mount/table"), + ("@public/mount", "/array", "@public/mount/array"), + ("@public/table", "/", "@public/table"), + ], +) +def test_service_metadata_displays_full_path(caterva2_source, target, relative, expected): # noqa: F811 + from rich.console import Console + + from blosc2.b2view.render import make_metadata_renderable + + base, _, _, _ = caterva2_source + with StoreBrowser(base + target) as browser: + info = browser.get_info(relative) + assert info.path == relative + assert info.display_path == expected + console = Console(record=True, width=120) + console.print(make_metadata_renderable(info)) + assert expected in console.export_text() + assert "/@public" not in console.export_text() + + +def test_multiroot_display_does_not_duplicate_prefix(caterva2_source): # noqa: F811 + base, _, _, stats = caterva2_source + stats["roots"]["@broken"] = {"name": "@broken"} + with StoreBrowser(base) as browser: + info = browser.get_info("/@public/mount/table") + assert info.display_path == "@public/mount/table" + + @pytest.mark.tui @pytest.mark.asyncio async def test_remote_table_window_is_background_and_stale_results_are_discarded( From 8d6ceb398e7a29488247a2fa9d9c45fcc7a24593 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sat, 3 Oct 2026 14:34:28 +0200 Subject: [PATCH 08/35] Add lazy Caterva2 file byte streams and atomic original-file downloads --- plans/caterva2-access-improvements2.md | 352 +++++++++++++++++++++++++ src/blosc2/__init__.py | 2 + src/blosc2/remote_file.py | 244 +++++++++++++++++ src/blosc2/remote_store.py | 31 ++- src/blosc2/schunk.py | 20 +- tests/test_remote_caterva2.py | 23 +- tests/test_remote_file.py | 93 +++++++ 7 files changed, 760 insertions(+), 5 deletions(-) create mode 100644 plans/caterva2-access-improvements2.md create mode 100644 src/blosc2/remote_file.py create mode 100644 tests/test_remote_file.py diff --git a/plans/caterva2-access-improvements2.md b/plans/caterva2-access-improvements2.md new file mode 100644 index 000000000..3a25ebd80 --- /dev/null +++ b/plans/caterva2-access-improvements2.md @@ -0,0 +1,352 @@ +# Caterva2 ordinary-file access and previews in b2view + +Status: implementation in progress following explicit user authorization. +It follows `plans/caterva2-access-improvements.md` and its completed URL discovery, +lazy hierarchy browsing, and viewer integration work. + +## Implementation record + +- M0: verified Caterva2 demo README transport: `api/chunk?nchunk=0` returns + a 552-byte compressed chunk that decompresses to the original 811 bytes; + `api/download` returns original Markdown bytes, not the `.b2` carrier. +- API selected: `RemoteFile(RemoteObject)` with bounded `read_bytes`, streamed + original-byte `download`, shared owner/cache/auth lifecycle, and unsupported + reference persistence. Fixed-chunk SChunks are accepted, including typed byte + payloads; irregular layouts and oversized chunks are refused explicitly. +- M1: recognition precedes array fallback; file handles participate in root and + direct-leaf opening. Existing exact-payload cache coordination/disk machinery + is reused with hash-separated file-chunk keys. Listing and metadata do not + fetch payload. Downloads stage on the destination filesystem and publish + atomically with no overwrite by default. Focused transport/access regression + validation: 62 tests passed in the blosc2 environment. + +## 1. Goal and delivery order + +Browse ordinary files published by Caterva2-compatible services without treating +them as unsupported array/table objects. Preserve the current hierarchy, root +conventions, lazy discovery, authentication, and client/server cache separation. + +Deliver in this order: + +1. Ordinary-file metadata, bounded byte access, explicit downloads, and text/ + Markdown previews. +2. Image previews using optional image dependencies and terminal capabilities. +3. PDF identification, download, and explicit external opening. Embedded PDF + rendering is optional follow-up work, not required for this plan's completion. + +Every file remains downloadable even if its format cannot be previewed. Do not +make downloading an entire file an implicit prerequisite for listing a directory. + +Target viewer behavior: + +| Selection | Metadata/data behavior | +| --- | --- | +| `@public/examples/README.md` | File metadata and bounded Markdown/text preview | +| `@public/examples/Wutujing-River.jpg` | File metadata and size-limited image preview, or a clear fallback | +| `@public/examples/cat2cloud-brochure.pdf` | File metadata, PDF notice, download/external-open actions | +| Unknown binary file | File metadata, preview-unavailable notice, download action | +| Generic SChunk without ordinary-file semantics | Byte-stream metadata/download; no invented original-file type | + +Paths displayed for `@` roots have no leading slash. Internal browser selection +paths remain unchanged. Local and direct fsspec regular-file support is outside +this first delivery; do not accidentally intercept existing direct-format URLs. + +## 2. Current implementation and verified evidence + +Relevant code: + +- `src/blosc2/remote_store.py`: `_caterva2_kind`, discovery metadata, node lookup, + leaf handles, owner lifecycle, cache coordination, and `RemoteNode`. +- `src/blosc2/schunk.py`: public `open` and Caterva2 leaf dispatch. +- `src/blosc2/c2array.py`: existing authentication and info/fetch/chunk transport. +- `src/blosc2/remote_object.py`: common remote-object contract. +- `src/blosc2/remote_repository.py`: independently owned repository roots. +- `src/blosc2/b2view/model.py`: object kinds, metadata, preview selection, leaf + lifetime, and display paths. +- `src/blosc2/b2view/app.py`: worker requests, stale-result rejection, download + UI, and optional `textual-image` support currently used by plot screens. +- `src/blosc2/b2view/render.py`: Rich metadata and data rendering. +- `tests/test_remote_caterva2.py`, `tests/test_caterva2_access.py`, and + `tests/b2view/test_caterva2.py`: deterministic API/viewer fixtures. + +The current `_caterva2_kind` accepts groups, NDArrays, and CTables. Metadata +without array shape/dtype or table schema consequently becomes `unsupported`. + +The following live info responses were inspected during planning at +`https://cat2.cloud/demo/api/info/@public/examples/`: + +| Name | Uncompressed bytes | Compressed bytes | Chunks | +| --- | ---: | ---: | ---: | +| `README.md` | 811 | 552 | 1 | +| `Wutujing-River.jpg` | 724,498 | 723,929 | 1 | +| `cat2cloud-brochure.pdf` | 44,155 | 36,889 | 1 | + +These responses expose SChunk fields (`nbytes`, `cbytes`, `chunksize`, `nchunks`, +`cparams`, `vlmeta`) and server-internal `.b2` backing paths, not NDArray metadata. +Do not display or construct download names from those internal filesystem paths. +These observations establish metadata shape, not full transport compatibility: +chunk and download semantics must still be verified before implementation. + +## 3. M0 — Protocol characterization and API decisions + +### Protocol matrix + +Create loopback fixtures for an ordinary-file SChunk, a generic byte SChunk, an +NDArray, a table, an unknown object, and a group. Characterize: + +- Explicit kind values where available, plus legacy SChunk-shaped info responses. +- `api/chunk` addressing and response framing for SChunks; final partial chunks, + empty streams, and variable-length/irregular layouts. +- Whether bounded `api/fetch` selection uses byte positions or typed elements. +- `api/download` semantics: original file bytes versus compressed Blosc frame, + suffix handling, response headers, streaming, and redirects. +- Metadata preservation for MIME type, logical filename, user attrs, and catalog + annotations, without assuming those fields exist on legacy servers. +- Compatibility with an actual Caterva2 deployment and cat2lite. cat2lite has + SChunk and download routes, but arbitrary native-file publication is not an + assumed existing capability. Record unsupported server combinations explicitly. + +Prefer chunk-based lazy reads if the protocol supports them reliably. Do not +fall back to whole-file fetches silently for a bounded text preview. If a server +cannot provide bounded work, expose metadata/download and explain the preview +limitation, or require explicit consent for a capped full-file fetch. + +### Proposed public abstraction + +Use a `RemoteFile` (or a clearly documented byte-stream equivalent) inheriting +`RemoteObject`, separate from array indexing and table operations. Resolve its +final public name and API in M0 before implementation; audit existing SChunk and +Proxy machinery for reusable transport/cache behavior. + +Proposed contract: + +- `source`, `attrs`, `traffic`, cache policy/allowance/usage, `close`, and context + management follow existing remote-object conventions. +- `nbytes`, logical `name`, optional media type, and chunk/compressed-size metadata. +- Explicit bounded `read_bytes(start, stop)` returning original uncompressed bytes. + Validate offsets and bounds; do not overload NDArray slicing semantics. +- Streaming `download(destination, overwrite=False)` writes the original payload, + never an undocumented `.b2` carrier, with optional progress/cancellation hooks. +- Downloading bytes is distinct from `save` of a remote reference. Reference + persistence/materialization is explicitly unsupported initially unless existing + artifact machinery can support it safely within scope. +- No writable source operations. Closing a root/repository leaves returned file + handles usable, as with existing array/table handles. + +Direct service leaf opening via `blosc2.open("https://host/@public/README.md")` +should return the same handle as hierarchy lookup. Preserve explicit URLPath +defaults: define and test the nonlazy byte-stream case rather than silently +changing all existing URLPath behavior. + +## 4. M1 — Discovery, byte access, cache, and download + +### Recognition and metadata + +- Recognize explicit file/SChunk kinds and strictly validated legacy SChunk + metadata. Classify NDArray/CTable metadata first to avoid confusing their + embedded chunk metadata with a file. +- Validate nonnegative byte counts, chunk counts, and consistent layouts. Reject + booleans masquerading as counts, invalid compression metadata, and impossible + lengths. Do not assume every SChunk is a regular fixed-chunk byte stream. +- Separate the transport kind from preview format. A `.jpg` suffix does not make + invalid metadata valid, nor guarantee that payload bytes are an image. +- Retain source attrs, catalog attrs, and file transport metadata separately. +- Preserve unknown metadata as an unsupported node with a useful diagnostic; + an invalid file must not hide healthy siblings. + +### Transport and resource ownership + +- Reuse bound auth, deployment-prefix handling, safe URL components, transport + pooling, and existing owner locks; never use a server-local `urlpath` as a URL. +- Fetch only necessary compressed chunks; validate frame/chunk headers and output + length before allocating/decompressing. Charge actual response bytes once. +- State preview transfer bounds separately from output bounds: a 64 KiB prefix + may require a much larger first chunk. Reject automatic preview work above + compressed/decompressed chunk limits instead of claiming it is bounded. +- Streaming download uses bounded buffers/chunks and checks cancellation between + reads. Enforce response and decompression limits on the streamed path, not only + after loading a response into memory. +- Never deserialize Python objects or execute embedded metadata to read a file. + +### Cache semantics + +- `MEMORY`: share the owner's existing retained-payload allowance across file, + array, and table leaves. Do not create a hidden per-file unlimited cache. +- `NONE`: retain no payload between operations. Transient preview/download + buffers still have explicit limits and are not called persistent cache. +- `DISK`: reuse source/auth identity, exclusive ownership, atomic publication, + and eviction conventions. Test reopened warm chunks and partial downloads. +- Keep client caching distinct from gateway caching. Repository budgets remain + per root. Immutable-source assumptions and explicit invalidation remain. +- Preview render buffers/decoded images have a separate, documented lifecycle + and limit; they are not silently counted as compressed chunk-cache bytes. + +### Safe downloads + +- Destination is chosen by the user. Sanitize suggested names from logical paths, + not Content-Disposition or server-local paths; reject traversal and separators. +- No overwrite without confirmation; publish only a completed download. Use a + staging file on the destination filesystem and a no-clobber publication strategy + when overwrite is false, including concurrent-creation tests. +- Failure/cancellation cleans up only this operation's staging file. Leave prior + destination contents untouched. Offer actionable permission/disk-full errors. +- Download the original `.md`, `.jpg`, or `.pdf` bytes; expose compressed-carrier + export only as a separately named future operation if needed. + +Acceptance: list/inspect without payload reads; exact arbitrary byte ranges and +streamed original-file downloads; mixed-leaf owner/cache/lifecycle tests pass. + +## 5. M2 — Text/Markdown previews and viewer actions + +Add a file node kind/icon, lightweight file metadata, and a preview result type +separate from numerical array/table previews. Do not route files through grid +paging, table filters, plotting, or scalar-array coercion. + +Suggested initial limits (finalize constants in M0 and document them): + +- Automatic text prefix: 64 KiB of original bytes, at most 1,000 displayed lines. +- Automatic chunk work: at most 8 MiB compressed and 16 MiB decompressed per + chunk; higher limits require explicit user action, not silent escalation. +- Download: streamed rather than held whole in RAM; explicit consent includes + known original size and destination. Unknown size gets a warning and progress + bytes without a fabricated percentage. + +Decode UTF-8/UTF-8 BOM first, with incremental decoding for a prefix that cuts a +multibyte character. Detect binary/NUL-heavy content and avoid garbage previews. +Invalid text gets an honest replacement/encoding notice; do not guess encodings +aggressively. Escape terminal control/ANSI sequences, including OSC hyperlinks. + +For Markdown, use Textual's existing Markdown capabilities without new mandatory +dependencies. Offer raw text fallback/toggle. Disable automatic embedded-image +fetches, local file access, and external-link launching: a preview must not issue +unrelated network requests. Mark truncated content, including unmatched fences; +rendering errors fall back to safe text rather than breaking the viewer. + +Add explicit Download and Open externally actions. Audit existing keybindings +before selecting keys; document them in help and expose enabled/disabled states. +For PDF/unknown binary files, these actions are available even without a preview. + +All remote fetch/decode/download work runs outside the UI thread. Bind results +to session, selection, and request IDs. Stale results must not change the visible +preview or launch an external application. Show progress/errors, preserve healthy +sibling browsing, and support retry. Shutdown releases all owned resources. + +Acceptance: README renders at its published path, large text is visibly truncated, +binary files remain downloadable, and delayed/cancelled reads do not freeze or +overwrite the current selection. + +## 6. M3 — Images + +Reuse Pillow and the optional `textual-image` integration; do not require +matplotlib just to display an existing JPEG/PNG. Audit packaging extras so the +installation hint accurately describes the minimum image dependencies. + +- Validate file signature/decoder result, not suffix alone. Start with JPEG/PNG; + do not promise arbitrary image formats or SVG/remote-resource rendering. +- A normal image preview may require the complete original file. Suggested + automatic original-payload cap: 16 MiB, with independent chunk/transfer bounds. +- Inspect dimensions before full decoding; suggested decoded-image budget: + 64 MiB with an explicit pixel cap. Treat Pillow decompression-bomb warnings as + refusal and reject malformed/oversized inputs. Downsampling after decoding is + not a substitute for a predecode limit. +- Correct EXIF orientation, aspect ratio, and resize-to-panel behavior. Show + dimensions and detected format. Bound animated-image handling to a first frame. +- If dependencies or terminal protocols are unavailable, show metadata and a + fallback notice with Download/Open externally. Do not leave a blank panel. +- Release image buffers/widgets on selection change and shutdown; no unbounded + decoded-thumbnail cache. Preserve worker/stale-result checks during resize. + +Acceptance: small JPEG/PNG fixtures display with image support enabled; missing +dependency/protocol and unsafe-image cases have useful, tested fallbacks. + +## 7. M4 — PDFs and external opening + +PDF support in this delivery means recognition, metadata, and convenient safe +download/opening, not a mandatory terminal PDF renderer. A PDF may be rendered +externally even when the terminal cannot show images. + +- Identify PDF candidates from extension/media metadata and verify signature when + bytes are read. Do not invoke a PDF parser just to list or inspect file size. +- Display "PDF preview unavailable; download or open externally" and actionable + controls. Missing renderer is not an unsupported-file error. +- External opening is an explicit user action, never triggered by selection, + refresh, MIME detection, or restoring a session. +- Prefer opening a completed local download, so bound service credentials never + appear in command-line URLs. Use platform launchers with argument arrays and + no shell (`open`, `xdg-open`, or the appropriate Windows API), check availability, + and use a safe absolute filename. Test launcher behavior with mocks only. +- Confirm opening untrusted content. Never mark downloads executable or dispatch + unknown scripts/binaries to a launcher; external-open initially has a narrow + document/image allowlist. Download remains available for other file types. +- Distinguish user-owned downloads from application-owned temporary copies. + External viewers can outlive b2view: document retention/cleanup and do not + delete a temporary file immediately after launch. Use a private session/temp + directory with cleanup of owned stale copies under an explicit policy. +- Report no-GUI/headless/missing-launcher failures without losing the downloaded + file; show its location and allow the user to open it manually. + +Optional later milestone: PDF first-page thumbnails/text extraction using an +opt-in backend, with page/pixel/time limits and untrusted-parser isolation where +appropriate. This is not needed for M4 acceptance and must not become a mandatory +dependency for ordinary-file access. + +## 8. M5 — Integration, documentation, and final review + +Extend offline fixtures with Markdown, invalid UTF-8, arbitrary binary data, +JPEG/PNG, and PDF bytes, all transported as deterministic SChunks. Include empty, +partial final chunk, typed/irregular SChunk, false suffix, and malformed responses. + +Acceptance matrix: + +| Area | Required checks | +| --- | --- | +| Discovery | Legacy/explicit kinds; groups/arrays/tables unchanged; no payload on listing | +| Opening | String URL, URLPath, bare single/multi-root services, nested mounts; @ display paths | +| Byte access | Empty/prefix/cross-chunk/end ranges; invalid bounds; output-length mismatch | +| Cache | NONE/MEMORY/DISK; mixed-leaf budgets; eviction; reopened warm data; auth isolation | +| Lifecycle | Children outlive parents; refresh/stale handles; cancellation and concurrency | +| Text | Markdown/raw; UTF-8 boundary/BOM; binary/control characters; truncation; no linked fetches | +| Images | JPEG/PNG; missing dependency/protocol; corruption; pixel/decode caps; resize/release | +| Download | Original byte equality; overwrite race; unsafe names; cancelled/truncated response; disk failure | +| External open | Explicit consent; no shell; launcher unavailable; no secrets; stale launch suppressed | +| PDF | Useful nonrendering fallback; download/open without PDF dependency | +| UI | Headless worker responsiveness; failed leaf isolation; actions/help and existing grids unaffected | + +Add opt-in real-service acceptance, not public-network dependency in the default +suite. Where supported, verify two independent clients can reuse gateway payload +cache; distinguish metadata rereads from payload rereads. Extend the actual +cat2lite acceptance only for supported publication/protocol behavior, not an +invented arbitrary-file server capability. + +Update public class/open documentation, remote-object guide, b2view guide, +optional-extra installation instructions, and release notes. Include limits, +download versus reference export, platform opening behavior, and fallbacks. + +Run Python/tests/build commands only in the `blosc2` conda environment. Run Ruff, +focused access/viewer tests (including headless TUI cases), the default suite, +and documentation checks. Preserve unrelated work and downloaded user files. + +Final review focuses on: + +1. Protocol assumptions validated on supported server versions. +2. Peak-memory/transfer limits, especially one-chunk large files and images. +3. Auth and path safety, unsafe content, and no implicit external side effects. +4. Shared cache accounting and returned-handle lifetime. +5. Worker cancellation, stale-result suppression, and safe atomic download. +6. Missing-dependency/headless/platform fallbacks and remaining limitations. + +## 9. Known tradeoffs and non-goals + +- Compression granularity may make a tiny byte prefix expensive. Preview refusal + is preferable to an unbounded hidden download. +- MIME/suffix hints are imperfect; content recognition is bounded and conservative. +- Budgets bound retained compressed payload, not all decoder/transient RAM; + preview/decompression limits must be enforced independently. +- External applications have their own security and lifecycle behavior; explicit + consent is not a sandbox. Credentials must never be passed to them. +- No recursive directory download, archive extraction, file editing/upload, + automatic script execution, OCR, animated playback, or embedded browser. +- No compulsory PDF renderer; download/external opening is a complete first + delivery for PDF and terminal-unrenderable supported documents. +- Ordinary-file reference archives and full fsspec/local file-browser expansion + are separate future work unless M0 proves a small, safe existing integration. diff --git a/src/blosc2/__init__.py b/src/blosc2/__init__.py index a41e13861..39008fd0b 100644 --- a/src/blosc2/__init__.py +++ b/src/blosc2/__init__.py @@ -627,6 +627,7 @@ def _raise(exc): from .remote_object import RemoteObject from .remote_array import RemoteMetadataMapping, RemoteArray from .remote_store import RemoteNode, RemoteStore +from .remote_file import RemoteFile from .remote_repository import RemoteRepository from . import linalg from .linalg import tensordot, vecdot, permute_dims, matrix_transpose, matmul, transpose, diagonal, outer @@ -932,6 +933,7 @@ def _raise(exc): "RemoteCTable", "RemoteNode", "RemoteStore", + "RemoteFile", "RemoteRepository", "SChunk", "SimpleProxy", diff --git a/src/blosc2/remote_file.py b/src/blosc2/remote_file.py new file mode 100644 index 000000000..03ddf0ebd --- /dev/null +++ b/src/blosc2/remote_file.py @@ -0,0 +1,244 @@ +"""Read-only Caterva2 SChunk byte streams, not array or object deserialization.""" + +import hashlib +import os +import struct +import tempfile +import weakref +from pathlib import Path + +import blosc2 +from blosc2.remote_array import RemoteMetadataMapping +from blosc2.remote_object import RemoteObject + +MAX_COMPRESSED_CHUNK = 8 << 20 +MAX_DECODED_CHUNK = 16 << 20 +MAX_READ_BYTES = 16 << 20 + + +def validate_file_metadata(info): + """Accept regular fixed-size SChunks; reject ambiguous/variable-length layouts.""" + for name in ("nbytes", "cbytes", "nchunks", "chunksize"): + value = info.get(name) + if type(value) is not int or value < 0: + raise ValueError(f"Invalid Caterva2 file {name}") + size, chunk = info["nbytes"], info["chunksize"] + if (chunk == 0 and size) or info["nchunks"] != ((size + chunk - 1) // chunk if chunk else 0): + raise ValueError("Caterva2 file requires a fixed-chunk byte stream") + params = info.get("cparams") + if not isinstance(params, dict) or type(params.get("typesize")) is not int or params["typesize"] <= 0: + raise ValueError("Invalid Caterva2 file compression metadata") + return {key: info[key] for key in ("nbytes", "cbytes", "nchunks", "chunksize", "cparams")} + + +class RemoteFile(RemoteObject): + """A lazy, read-only Caterva2 byte stream. + + ``read_bytes(start, stop)`` returns original bytes (at most 16 MiB per call). + ``download(destination)`` streams original bytes with atomic publication and + no overwrite by default. Chunk work is limited to 8 MiB compressed / 16 MiB + decoded; oversized or irregular streams require server-side rechunking. + Caches and lifetime are shared with the containing RemoteStore owner. + """ + + @classmethod + def _from_owner(cls, owner, path): + obj = object.__new__(cls) + obj._owner, obj.path = owner, path + obj._generation = owner.generation + obj._meta = validate_file_metadata(owner.metadata["caterva2_files"][path]) + owner.acquire() + obj._finalizer = weakref.finalize(obj, owner.release) + return obj + + def _check_open(self): + if not self._finalizer.alive: + raise RuntimeError("RemoteFile is closed") + if self._generation != self._owner.generation: + raise RuntimeError("RemoteFile is stale; reopen after refresh") + + @property + def source(self): + self._check_open() + from blosc2.remote_store import caterva2_source_descriptor + + return caterva2_source_descriptor(blosc2.URLPath(self.path, urlbase=self._owner.caterva2.urlbase)) + + @property + def name(self): + self._check_open() + return self.path.rsplit("/", 1)[-1] + + @property + def nbytes(self): + self._check_open() + return self._meta["nbytes"] + + @property + def cbytes(self): + self._check_open() + return self._meta["cbytes"] + + @property + def attrs(self): + self._check_open() + return RemoteMetadataMapping(self._owner.attrs.get(self.path, {})) + + @property + def traffic(self): + self._check_open() + return self._owner.traffic + + @property + def cache_policy(self): + self._check_open() + return self._owner.cache_policy + + @property + def max_cache_bytes(self): + self._check_open() + return self._owner.max_cache_bytes + + @property + def cache_bytes(self): + """Retained bytes for the shared owner, including other leaves.""" + self._check_open() + return self._owner.cache_coordinator.cache_bytes + + @property + def metadata_bytes(self): + self._check_open() + return self._owner.metadata_bytes + + @property + def info(self): + from blosc2.info import InfoReporter + + return InfoReporter(self) + + @property + def info_items(self): + self._check_open() + return [ + ("type", "RemoteFile"), + ("name", self.name), + ("source", self.source), + ("nbytes", self.nbytes), + ("cbytes", self.cbytes), + ("cache policy", self.cache_policy), + ("cache bytes (owner)", self.cache_bytes), + ] + + def _chunk(self, index, cancel=None): + from blosc2.c2array import _auth_headers, _server_url, _sync_client + from blosc2.remote_store import _Caterva2FrameCache + + self._check_open() + if cancel is not None and cancel(): + raise InterruptedError("File operation cancelled") + size = min(self._meta["chunksize"], self.nbytes - index * self._meta["chunksize"]) + if size > MAX_DECODED_CHUNK: + raise ValueError("File chunk exceeds 16 MiB decoded limit; rechunk on the server") + owner = self._owner + key = hashlib.sha256(f"file-chunk:{self.path}:{index}".encode()).hexdigest() + cache = None + if self.cache_policy is not blosc2.CachePolicy.NONE: + if owner.table_frame_cache is None: + owner.table_frame_cache = _Caterva2FrameCache(owner) + cache = owner.table_frame_cache + payload = None if cache is None else cache.get(key) + missing = payload is None + if payload is None: + url = _server_url(owner.caterva2.urlbase, f"api/chunk/{self.path}") + client = owner.transport or _sync_client() + data = bytearray() + with client.stream( + "GET", + url, + params={"nchunk": index}, + headers=_auth_headers(owner.caterva2.auth_token), + timeout=10, + follow_redirects=False, + ) as response: + response.raise_for_status() + if response.is_redirect: + raise ValueError("File chunk redirects are unsupported") + for part in response.iter_bytes(chunk_size=64 << 10): + owner.traffic.charge(len(part)) + if cancel is not None and cancel(): + raise InterruptedError("File operation cancelled") + if len(data) + len(part) > MAX_COMPRESSED_CHUNK: + raise ValueError("File chunk exceeds 8 MiB compressed limit") + data.extend(part) + payload = bytes(data) + if len(payload) < 16 or len(payload) > MAX_COMPRESSED_CHUNK: + raise ValueError("Invalid compressed file chunk length") + nbytes, _, cbytes = struct.unpack_from(" MAX_READ_BYTES: + raise ValueError("read_bytes exceeds 16 MiB; use streaming download") + if start == stop: + return b"" + chunk = self._meta["chunksize"] + result = bytearray() + for index in range(start // chunk, (stop - 1) // chunk + 1): + data = self._chunk(index) + result.extend(data[max(0, start - index * chunk) : min(len(data), stop - index * chunk)]) + return bytes(result) + + def download(self, destination, *, overwrite=False, progress=None, cancel=None): + """Stream original bytes, publishing only on success. + + ``progress(done, total)`` and ``cancel()`` run on the caller's thread. + A false overwrite uses an atomic hard-link publish to avoid races. + """ + self._check_open() + destination = Path(destination).absolute() + if not overwrite and destination.exists(): + raise FileExistsError(destination) + fd, temporary = tempfile.mkstemp(prefix=".b2view-download-", dir=destination.parent) + try: + with os.fdopen(fd, "wb") as file: + done = 0 + for index in range(self._meta["nchunks"]): + with self._owner.lock: + data = self._chunk(index, cancel) + file.write(data) + done += len(data) + if progress is not None: + progress(done, self.nbytes) + file.flush() + os.fsync(file.fileno()) + self._check_open() + if cancel is not None and cancel(): + raise InterruptedError("File operation cancelled") + if overwrite: + os.replace(temporary, destination) + else: + os.link(temporary, destination) + return str(destination) + finally: + Path(temporary).unlink(missing_ok=True) + + def save(self, *args, **kwargs): + raise NotImplementedError( + "RemoteFile reference persistence is unsupported; use download for original bytes" + ) + + def close(self): + self._finalizer() diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 62389e41c..b97eac44c 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -520,7 +520,7 @@ def _restore_manifest(self, manifest): # noqa: C901 if ( not isinstance(entry, (list, tuple)) or len(entry) != 2 - or entry[0] not in {"group", "ndarray", "ctable", "remote_store", "unsupported"} + or entry[0] not in {"group", "ndarray", "ctable", "remote_store", "unsupported", "file"} ): raise ValueError("Invalid RemoteStore node") self.nodes[path] = tuple(entry) @@ -536,6 +536,10 @@ def _restore_manifest(self, manifest): # noqa: C901 ): raise ValueError("Invalid RemoteStore child list") self.attrs = manifest["attrs"] + if self.format == "caterva2": + for key, (kind, _) in self.nodes.items(): + if kind == "unsupported": + self.attrs.pop(key, None) self.listed = manifest["listed"] if self.format == "caterva2" and self.metadata.get("caterva2_listing_version") != 2: # Earlier snapshots marked all discovered groups as fully listed, @@ -953,6 +957,13 @@ def _caterva2_kind(info): return "ctable" if "shape" in info and "dtype" in info: return "ndarray" + if kind in {"file", "schunk"} or all( + key in info for key in ("nbytes", "nchunks", "chunksize", "cparams") + ): + from blosc2.remote_file import validate_file_metadata + + validate_file_metadata(info) + return "file" return "unsupported" def _caterva2_table_metadata(self, path, info): @@ -995,7 +1006,7 @@ def _caterva2_get(self, endpoint, path): def _open_caterva2(self): self.metadata["caterva2_listing_version"] = 2 self._discover_caterva2_node(self.root) - if self.nodes[self.root][0] not in {"group", "ctable", "ndarray"}: + if self.nodes[self.root][0] not in {"group", "ctable", "ndarray", "file"}: raise ValueError("Caterva2 source root is not an array, group, or CTable") def _discover_caterva2_node(self, full): @@ -1019,7 +1030,7 @@ def _discover_caterva2_node(self, full): value = self._caterva2_table_metadata(full, info) if kind == "ctable" else None if kind == "unsupported": value = "Caterva2 object kind is unsupported" - attrs = dict(info.get("attrs") or {}) + attrs = dict(info.get("attrs") or info.get("vlmeta") or {}) annotations = dict(info.get("catalog_attrs") or {}) # Validate all potentially failing work before changing the registry. # Only this node and its ancestors can change. Do not copy a complete @@ -1039,6 +1050,10 @@ def _discover_caterva2_node(self, full): self.nodes[key] = old raise self.attrs[full] = attrs + if kind == "file": + from blosc2.remote_file import validate_file_metadata + + self.metadata.setdefault("caterva2_files", {})[full] = validate_file_metadata(info) if annotations: self.metadata.setdefault("catalog_attrs", {})[full] = annotations @@ -1773,6 +1788,10 @@ def save_selection( # noqa: C901 metadata = self._export_metadata() src_desc, nodes, attrs, listed, candidates = self._collect_export_nodes(full_path, include_cache) + if any(kind == "file" for kind, _ in nodes.values()): + raise NotImplementedError( + "Hierarchy export containing files is unsupported; download files individually" + ) staging_dir = tempfile.mkdtemp(prefix="b2z-export-", dir=dest_dir) fd, tmp_zip = tempfile.mkstemp(prefix="export-", suffix=".b2z.tmp", dir=dest_dir) @@ -2941,6 +2960,8 @@ def __getitem__(self, path): return group if kind == "ctable": return blosc2.RemoteCTable._from_owner(self._owner, full) + if kind == "file": + return blosc2.RemoteFile._from_owner(self._owner, full) if kind == "unsupported": raise NotImplementedError(f"{path!r}: {value}") self._owner.open_source(relative) @@ -3176,6 +3197,10 @@ def save( ) -> str: """Export the current store or subtree to a portable .b2z reference archive.""" self._ensure_open() + if self._owner.metadata.get("caterva2_files"): + raise NotImplementedError( + "Caterva2 hierarchy export with ordinary files is unsupported; download files individually" + ) with self._owner.lock: _, full = self._resolve("") return self._owner.save_selection( diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index 9b559cde2..8818b3c34 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2202,7 +2202,25 @@ def _open_c2_urlpath(urlpath: blosc2.URLPath, mode: str, offset: int, kwargs: di if remote_array_options["assume_immutable"] is not True: raise ValueError("shared_cache=True requires assume_immutable=True") metadata = blosc2.c2array.info(urlpath.path, urlpath.urlbase, auth_token=urlpath.auth_token) - kind = metadata.get("kind") + from blosc2.remote_store import RemoteDiscovery + + kind = RemoteDiscovery._caterva2_kind(metadata) + if kind == "file": + if remote_array_options.get("assume_immutable") is not True: + raise ValueError("RemoteFile requires assume_immutable=True") + if cache_path is not None or shared_cache or max_concurrency is not None: + raise NotImplementedError( + "RemoteFile uses cache_dir; shared_cache and max_concurrency are unsupported" + ) + options = { + key: remote_array_options[key] + for key in ("cache_policy", "max_cache_bytes") + if key in remote_array_options + } + with blosc2.RemoteStore( + urlpath, cache_dir=cache_dir, _allow_array_root=True, _root_info=metadata, **options + ) as store: + return store[""] if kind == "group": if cache_path is not None: raise ValueError("Caterva2 groups use cache_dir, not cache_path") diff --git a/tests/test_remote_caterva2.py b/tests/test_remote_caterva2.py index f1c728dc7..966a53340 100644 --- a/tests/test_remote_caterva2.py +++ b/tests/test_remote_caterva2.py @@ -95,6 +95,9 @@ def caterva2_source(request): }, "fail_list": None, "aliases": {}, + "files": {}, + "file_chunks": [], + "file_metadata": {}, } def array_info(): @@ -157,7 +160,18 @@ def do_GET(self): return if path.startswith("/api/info/"): key = path.removeprefix("/api/info/") - if key in stats["groups"]: + if key in stats["files"]: + stream = stats["files"][key] + info = { + "nbytes": stream.nbytes, + "cbytes": stream.cbytes, + "nchunks": stream.nchunks, + "chunksize": stream.chunksize, + "cparams": safe(stream.cparams), + "vlmeta": {"title": "ordinary file"}, + } + info.update(stats["file_metadata"].get(key, {})) + elif key in stats["groups"]: info = { "kind": "group", "attrs": {"name": "fixture"}, @@ -181,6 +195,13 @@ def do_GET(self): else: self.send_error(404) return + if path.startswith("/api/chunk/"): + key = path.removeprefix("/api/chunk/") + if key in stats["files"]: + index = int(urllib.parse.parse_qs(parsed.query)["nchunk"][0]) + stats["file_chunks"].append((key, index)) + self.send(stats["files"][key].get_chunk(index), "application/octet-stream") + return if path.startswith("/api/fetch/"): stats["fetches"] += 1 key = path.removeprefix("/api/fetch/") diff --git a/tests/test_remote_file.py b/tests/test_remote_file.py new file mode 100644 index 000000000..e83b6e819 --- /dev/null +++ b/tests/test_remote_file.py @@ -0,0 +1,93 @@ +"""Deterministic Caterva2 ordinary-file byte transport and lifetime tests.""" + +import pytest +from test_remote_caterva2 import caterva2_source # noqa: F401 + +import blosc2 + + +@pytest.fixture +def file_source(caterva2_source): # noqa: F811 + base, _, _, stats = caterva2_source + payload = b"# Hello\n\nAn ordinary **Markdown** file.\n" * 20 + stream = blosc2.SChunk(chunksize=64, cparams={"typesize": 1}) + for offset in range(0, len(payload), 64): + stream.append_data(payload[offset : offset + 64]) + stats["files"]["@public/README.md"] = stream + stats["groups"]["@public"].append("README.md") + return base, payload, stats + + +def test_discovery_ranges_and_download(file_source, tmp_path): + base, payload, stats = file_source + with blosc2.open(base + "@public") as store: + assert store.kind("README.md") == "file" + assert not stats["file_chunks"] + with store["README.md"] as file: + assert isinstance(file, blosc2.RemoteFile) + assert file.name == "README.md" + assert file.attrs["title"] == "ordinary file" + assert file.read_bytes(59, 135) == payload[59:135] + assert file.read_bytes(1, 1) == b"" + for start, stop in [(-1, 3), (4, 3), (True, 4), (0, len(payload) + 1)]: + with pytest.raises(ValueError): + file.read_bytes(start, stop) + dest = tmp_path / "README.md" + progress = [] + file.download(dest, progress=lambda done, total: progress.append((done, total))) + assert dest.read_bytes() == payload + assert progress[-1] == (len(payload), len(payload)) + with pytest.raises(FileExistsError): + file.download(dest) + with pytest.raises(NotImplementedError): + file.save(tmp_path / "file.b2z") + + +@pytest.mark.parametrize( + "policy", [blosc2.CachePolicy.NONE, blosc2.CachePolicy.MEMORY, blosc2.CachePolicy.DISK] +) +def test_cache_and_child_lifetime(file_source, tmp_path, policy): + base, payload, stats = file_source + options = {"cache_policy": policy} + if policy is blosc2.CachePolicy.DISK: + options["cache_dir"] = tmp_path / "cache" + store = blosc2.open(base + "@public", **options) + file = store["README.md"] + store.close() + with file: + assert file.read_bytes(0, 10) == payload[:10] + before = len(stats["file_chunks"]) + assert file.read_bytes(0, 10) == payload[:10] + assert len(stats["file_chunks"]) == before + (policy is blosc2.CachePolicy.NONE) + with pytest.raises(RuntimeError, match="closed"): + file.read_bytes(0, 1) + if policy is blosc2.CachePolicy.DISK: + with blosc2.open(base + "@public/README.md", **options) as direct: + assert direct.read_bytes(0, 10) == payload[:10] + # A different selected source has a distinct cache identity. + with blosc2.open(base + "@public", **options) as root, root["README.md"] as warm: + before = len(stats["file_chunks"]) + warm.read_bytes(0, 10) + assert len(stats["file_chunks"]) == before + + +def test_direct_file_and_cancelled_atomic_download(file_source, tmp_path): + base, payload, _ = file_source + dest = tmp_path / "test.md" + dest.write_bytes(b"old data") + with blosc2.open(base + "@public/README.md") as file: + assert file.read_bytes() == payload + with pytest.raises(InterruptedError): + file.download(dest, overwrite=True, cancel=lambda: True) + assert dest.read_bytes() == b"old data" + assert not list(tmp_path.glob(".b2view-download-*")) + with pytest.raises(NotImplementedError, match="lazy=True"): + blosc2.open(blosc2.URLPath("@public/README.md", urlbase=base)) + + +def test_metadata_validation(file_source): + base, _, stats = file_source + for overrides in ({"nbytes": True}, {"chunksize": 0}, {"nchunks": 999}, {"cparams": {"typesize": 0}}): + stats["file_metadata"]["@public/README.md"] = overrides + with pytest.raises(ValueError): + blosc2.open(base + "@public/README.md") From c739e381de1c607e46fd75b1cd766ea96dca27e7 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sat, 3 Oct 2026 14:37:15 +0200 Subject: [PATCH 09/35] Preview Caterva2 text and images and offer document download and opening --- plans/caterva2-access-improvements2.md | 10 ++ src/blosc2/b2view/app.py | 190 +++++++++++++++++++++++- src/blosc2/b2view/file_preview.py | 110 ++++++++++++++ src/blosc2/b2view/model.py | 21 ++- src/blosc2/b2view/render.py | 14 ++ tests/b2view/test_files.py | 191 +++++++++++++++++++++++++ tests/test_remote_file.py | 75 ++++++++++ 7 files changed, 608 insertions(+), 3 deletions(-) create mode 100644 src/blosc2/b2view/file_preview.py create mode 100644 tests/b2view/test_files.py diff --git a/plans/caterva2-access-improvements2.md b/plans/caterva2-access-improvements2.md index 3a25ebd80..45fd4bcc3 100644 --- a/plans/caterva2-access-improvements2.md +++ b/plans/caterva2-access-improvements2.md @@ -19,6 +19,16 @@ lazy hierarchy browsing, and viewer integration work. fetch payload. Downloads stage on the destination filesystem and publish atomically with no overwrite by default. Focused transport/access regression validation: 62 tests passed in the blosc2 environment. +- M2–M4: b2view recognizes file leaves, renders passive bounded text/Markdown + (64 KiB / 1,000 lines), toggles raw text, and decodes JPEG/PNG under independent + file/chunk/pixel limits. Image widgets reuse textual-image without matplotlib; + absent dependencies have download/open fallbacks. PDFs require no renderer: + D downloads original bytes and O prompts for destination and explicit trust + consent before a shell-free platform launch. All fetch/decode/download work is + backgrounded; downloads use independent handles and cancellation. Viewer/model + plus file transport regressions: 175 passed with TUI cases enabled. Live demo + checks: README renders; the JPEG decodes as 2034 × 1144 (bounded preview 1600 × + 900); PDF offers download/external opening without fetching content on selection. ## 1. Goal and delivery order diff --git a/src/blosc2/b2view/app.py b/src/blosc2/b2view/app.py index 4973e7162..2b0ce5a70 100644 --- a/src/blosc2/b2view/app.py +++ b/src/blosc2/b2view/app.py @@ -67,6 +67,7 @@ "c2array": "▦", "ctable": "▤", "schunk": "▣", + "file": "📄", "unknown": "?", } @@ -292,6 +293,15 @@ class HelpScreen(ModalScreen[None]): ("escape", "close the plot (q quits b2view)"), ], ), + ( + "Ordinary files", + [ + ("D", "download original bytes to a chosen destination (no overwrite)"), + ("O", "download and open a document externally (requires explicit trust consent)"), + ("T", "toggle raw text / Markdown rendering"), + ("escape", "cancel a file download or close its dialog"), + ], + ), ( "Dim mode (N-D arrays)", [ @@ -1884,6 +1894,102 @@ def on_progress(downloaded: int, content_total: int | None) -> None: self.app.call_from_thread(self.dismiss, True) +class FileTransferScreen(ModalScreen): + """Explicit destination/consent and cancellable background original-byte download.""" + + CSS = """ + FileTransferScreen { align: center middle; } + #file-transfer { width: 75; height: auto; border: thick $accent; padding: 1 2; background: $surface; } + """ + BINDINGS: ClassVar = [("escape", "cancel", "Cancel")] + + def __init__(self, file, *, external=False): + super().__init__() + self.file = file + self.external = external + self.cancelled = threading.Event() + self.started = False + + def compose(self): + from pathlib import Path + + from blosc2.b2view.file_preview import safe_text + + name = safe_text(self.file.name).replace("\\", "_").replace("/", "_") + with Vertical(id="file-transfer"): + yield Static( + f"Download original file: {name} ({self.file.nbytes:,} bytes)\n" + "Choose a destination; Enter starts. Escape cancels. Existing files are not overwritten.", + markup=False, + ) + yield Input(value=str(Path.cwd() / name), id="file-destination") + if self.external: + yield Checkbox( + "I trust this file and want to open it in an external application", id="file-consent" + ) + yield Static("", id="file-status", markup=False) + yield ProgressBar(id="file-progress") + + def on_mount(self): + self.origin = (self.app._remote_session, self.app._remote_request, self.app.selected_path) + self.query_one(Input).focus() + + def on_input_submitted(self): + if self.started: + return + if self.external and not self.query_one("#file-consent", Checkbox).value: + self.query_one("#file-status", Static).update( + "Explicit consent is required for external opening." + ) + return + destination = self.query_one(Input).value + if not destination.strip(): + return + self.started = True + self.query_one(Input).disabled = True + self._transfer(destination) + + @work(thread=True, exit_on_error=False) + def _transfer(self, destination): + def progress(done, total): + self.app.call_from_thread( + self.query_one("#file-progress", ProgressBar).update, total=total, progress=done + ) + + try: + path = self.file.download(destination, progress=progress, cancel=self.cancelled.is_set) + # Navigation/refresh/shutdown must never launch an obsolete request. + current = (self.app._remote_session, self.app._remote_request, self.app.selected_path) + if ( + not self.cancelled.is_set() + and self.external + and current == self.origin + and not self.app._closing + ): + from blosc2.b2view.file_preview import open_external + + open_external(path) + if not self.cancelled.is_set(): + self.app.call_from_thread(self._finished, f"Saved: {path}") + except Exception as error: + if not self.cancelled.is_set(): + self.app.call_from_thread(self._finished, f"{error}\nDestination: {destination}") + finally: + self.file.close() + + def _finished(self, message): + self.query_one("#file-status", Static).update(message + "\nEscape closes this dialog.") + + def action_cancel(self): + self.cancelled.set() + self.dismiss() + + def on_unmount(self): + self.cancelled.set() + if not self.started: + self.file.close() + + class B2ViewHeader(Header): """App header that also shows the open bundle's filename, left of the title. @@ -1980,6 +2086,7 @@ class B2ViewApp(App): #data-header { height: auto; padding: 0 1; } #data-table-row { height: 1fr; } #data-table { width: 1fr; height: 1fr; } + #file-image { height: auto; } #row-scrollbar { width: 1; height: 1fr; color: $primary; } #col-scrollbar { height: 1; width: 1fr; color: $primary; } #meta-scroll, #attrs-scroll, #data-scroll { height: 1fr; padding: 0 1; } @@ -2013,6 +2120,9 @@ class B2ViewApp(App): Binding("d", "dim_cycle", "Dim mode", show=False), Binding("enter", "dim_toggle_nav", "Toggle nav", show=False), Binding("escape", "dim_exit", "Exit dim mode", show=False), + Binding("D", "download_file", "Download file", show=False), + Binding("O", "open_file", "Open externally", show=False), + Binding("T", "raw_file", "Raw/Markdown", show=False), ] def __init__( @@ -2063,6 +2173,7 @@ def __init__( self._remote = is_fsspec_url(urlpath) self._remote_session = 0 self._remote_request = 0 + self._file_raw = False self._remote_page_request = 0 self._remote_page_pending = False self._remote_col_end = None @@ -2119,6 +2230,7 @@ def compose(self) -> ComposeResult: yield Static("", id="col-scrollbar") with VerticalScroll(id="data-scroll", can_focus=True): yield Static("", id="preview") + yield Vertical(id="file-image") yield Footer() def on_mount(self) -> None: @@ -2439,7 +2551,9 @@ def _read_remote_info(self, session, request, browser, path): if info.kind == "unsupported": data = {"message": info.metadata.get("preview", "Preview unavailable")} elif info.kind not in {"group", "remote_store"} and not self._uses_grid_preview(info): - data = browser.preview(path, max_rows=self.preview_rows, max_cols=self.preview_cols) + data = browser.preview( + path, max_rows=self.preview_rows, max_cols=self.preview_cols, raw_text=self._file_raw + ) self._deliver_remote(session, self._finish_remote_info, request, path, info, data, None) except Exception as exc: self._deliver_remote(session, self._finish_remote_info, request, path, None, None, exc) @@ -2459,6 +2573,7 @@ def _render_panels(self, path, info=None, remote_data=None): data_table_row = self.query_one("#data-table-row", Horizontal) data_scroll = self.query_one("#data-scroll", VerticalScroll) preview = self.query_one("#preview", Static) + self.run_worker(self._show_file_image(path, remote_data), exclusive=True, group="file-image") attrs_pane = self.query_one("#attrs-pane", B2ViewPanel) attrs_widget = self.query_one("#attrs-data", Static) try: @@ -2532,6 +2647,79 @@ def _render_panels(self, path, info=None, remote_data=None): self._apply_focus_on_next_update = False self.call_after_refresh(self._apply_start_focus) + async def _show_file_image(self, path, data): + body = self.query_one("#file-image", Vertical) + await body.remove_children() + if path != self.selected_path or not isinstance(data, dict) or "file_image" not in data: + return + if TextualImage is None: + self.query_one("#preview", Static).update( + data["message"] + "\nTerminal image preview needs textual-image; use D or O." + ) + return + try: + await body.mount(TextualImage(data["file_image"])) + except Exception as error: + self.query_one("#preview", Static).update(f"Image display unavailable: {error}; use D or O.") + + def _file_action(self, external=False): + if self.browser is None or self._selected_info is None or self._selected_info.kind != "file": + self.notify("Select an ordinary file first", severity="warning") + return + self._prepare_file_transfer( + self._remote_session, self._remote_request, self.browser, self.selected_path, external + ) + + @work(thread=True, exit_on_error=False) + def _prepare_file_transfer(self, session, request, browser, path, external): + from pathlib import PurePosixPath + + from blosc2.b2view.file_preview import EXTERNAL_SUFFIXES + + alias = None + try: + with browser.io_lock: + if session != self._remote_session or request != self._remote_request: + return + obj = browser._get_object(path) + if not isinstance(obj, blosc2.RemoteFile): + return + if external and PurePosixPath(obj.name).suffix.lower() not in EXTERNAL_SUFFIXES: + raise ValueError( + "External opening is restricted to document/image files; use D to download" + ) + alias = blosc2.RemoteFile._from_owner(obj._owner, obj.path) + delivered = self._deliver_remote( + session, self._finish_file_transfer, request, path, alias, external, None + ) + if not delivered: + alias.close() + except Exception as error: + if alias is not None: + alias.close() + self._deliver_remote(session, self._finish_file_transfer, request, path, None, external, error) + + def _finish_file_transfer(self, request, path, file, external, error): + if request != self._remote_request or path != self.selected_path: + if file is not None: + file.close() + return + if error is not None: + self.notify(str(error), severity="warning") + return + self.push_screen(FileTransferScreen(file, external=external)) + + def action_download_file(self): + self._file_action() + + def action_open_file(self): + self._file_action(external=True) + + def action_raw_file(self): + if self._selected_info is not None and self._selected_info.kind == "file": + self._file_raw = not self._file_raw + self.update_panels(self.selected_path) + @staticmethod def _format_attr_value(value: Any) -> str: """Format an attribute value for display.""" diff --git a/src/blosc2/b2view/file_preview.py b/src/blosc2/b2view/file_preview.py new file mode 100644 index 000000000..23b94e058 --- /dev/null +++ b/src/blosc2/b2view/file_preview.py @@ -0,0 +1,110 @@ +"""Bounded, passive previews of untrusted ordinary-file content.""" + +import codecs +import io +import re +import warnings +from pathlib import PurePosixPath + +TEXT_BYTES = 64 << 10 +TEXT_LINES = 1000 +IMAGE_BYTES = 16 << 20 +IMAGE_PIXELS = (64 << 20) // 4 +TEXT_SUFFIXES = {".md", ".txt", ".rst", ".csv", ".json", ".yaml", ".yml", ".log", ".py"} +IMAGE_SUFFIXES = {".jpg", ".jpeg", ".png"} +EXTERNAL_SUFFIXES = IMAGE_SUFFIXES | {".pdf", ".md", ".txt"} + + +def safe_text(text): + """Remove ANSI/OSC sequences and unsafe terminal controls from decoded text.""" + text = re.sub(r"\x1b\][^\x07\x1b]*(?:\x07|\x1b\\)", "", text) + text = re.sub(r"\x1b\[[0-?]*[ -/]*[@-~]", "", text) + return "".join(c for c in text if c in "\n\t" or (ord(c) >= 32 and not 127 <= ord(c) < 160)) + + +def preview_file(file, *, raw=False): + """Return metadata/render content without executing links or decoding objects.""" + suffix = PurePosixPath(file.name).suffix.lower() + try: + if suffix == ".pdf": + return {"message": "PDF preview unavailable; use D to download or O to open externally."} + if suffix in IMAGE_SUFFIXES: + return preview_image(file) + if suffix not in TEXT_SUFFIXES: + return {"message": "Binary/unknown file; use D to download original bytes."} + data = file.read_bytes(0, min(file.nbytes, TEXT_BYTES)) + if b"\0" in data: + return {"message": "Binary content despite text suffix; use D to download."} + truncated = file.nbytes > len(data) + decoder = codecs.getincrementaldecoder("utf-8-sig")("replace") + text = safe_text(decoder.decode(data, final=not truncated)) + lines = text.splitlines(keepends=True) + truncated |= len(lines) > TEXT_LINES + text = "".join(lines[:TEXT_LINES]) + notice = "Preview truncated (64 KiB / 1,000 line limit)." if truncated else "" + if "\ufffd" in text: + notice += " Invalid UTF-8 replaced." + return { + "file_text": text, + "markdown": suffix == ".md" and not raw, + "notice": notice, + "message": "D: download · O: open externally · T: raw/Markdown", + } + except Exception as error: + return {"message": f"Preview unavailable: {error}. Use D to download (chunk limits still apply)."} + + +def preview_image(file): + try: + from PIL import Image, ImageOps + except ImportError: + return {"message": "Image preview needs Pillow; download (D) or open externally (O)."} + if file.nbytes > IMAGE_BYTES: + return {"message": "Automatic image preview exceeds 16 MiB; download (D) or open externally (O)."} + data = file.read_bytes(0, file.nbytes) + if not (data.startswith(b"\xff\xd8\xff") or data.startswith(b"\x89PNG\r\n\x1a\n")): + return {"message": "Invalid JPEG/PNG signature; download available (D)."} + with warnings.catch_warnings(): + warnings.simplefilter("error", Image.DecompressionBombWarning) + with Image.open(io.BytesIO(data)) as original: + if original.format not in {"JPEG", "PNG"} or original.width * original.height > IMAGE_PIXELS: + raise ValueError("Image exceeds decoded pixel budget (64 MiB RGBA)") + size, format_name = original.size, original.format + original.seek(0) + image = ImageOps.exif_transpose(original) + image.thumbnail((1600, 1200)) + image = image.convert("RGB") + return { + "file_image": image, + "message": f"{format_name} · {size[0]} × {size[1]} · D: download · O: open externally", + } + + +def open_external(path): + """Launch a user-confirmed document with no shell or credential-bearing URL.""" + import os + import shutil + import subprocess + import sys + from pathlib import Path + + path = Path(path).absolute() + if path.suffix.lower() not in EXTERNAL_SUFFIXES: + raise ValueError("External opening is restricted to PDF, JPEG/PNG, Markdown and text") + with path.open("rb") as file: + signature = file.read(1024) + suffix = path.suffix.lower() + if ( + (suffix == ".pdf" and not signature.startswith(b"%PDF-")) + or (suffix in {".jpg", ".jpeg"} and not signature.startswith(b"\xff\xd8\xff")) + or (suffix == ".png" and not signature.startswith(b"\x89PNG\r\n\x1a\n")) + or (suffix in {".txt", ".md"} and b"\0" in signature) + ): + raise ValueError("File content does not match the external-open document type") + if sys.platform == "win32": + os.startfile(str(path)) + return + launcher = "open" if sys.platform == "darwin" else "xdg-open" + if shutil.which(launcher) is None: + raise RuntimeError(f"{launcher} is unavailable; open the downloaded file manually: {path}") + subprocess.run([launcher, str(path)], check=True, timeout=10, capture_output=True) diff --git a/src/blosc2/b2view/model.py b/src/blosc2/b2view/model.py index f79e10245..67f312e64 100644 --- a/src/blosc2/b2view/model.py +++ b/src/blosc2/b2view/model.py @@ -475,7 +475,9 @@ def get_info(self, path: str) -> ObjectInfo: def _display_path(self, path): """Keep service paths fully qualified without changing browser navigation.""" - if isinstance(self.store, (blosc2.RemoteStore, blosc2.RemoteArray, blosc2.RemoteCTable)): + if isinstance( + self.store, (blosc2.RemoteStore, blosc2.RemoteArray, blosc2.RemoteCTable, blosc2.RemoteFile) + ): source = self.store.source if source["kind"] == "caterva2": full = "/".join(part for part in (source["path"], path.strip("/")) if part) @@ -493,7 +495,7 @@ def _remote_info(self, path): if node.catalog_attrs: metadata["catalog_attrs"] = dict(node.catalog_attrs) attrs = node.attrs - if node.kind in {"ndarray", "ctable"}: + if node.kind in {"ndarray", "ctable", "file"}: obj = self._get_object(path) metadata.update(object_metadata(obj)) if node.kind == "ctable": @@ -522,6 +524,7 @@ def preview( col_start: int = 0, slice_indices: list[int] | None = None, layout: DataSliceLayout | None = None, + raw_text: bool = False, ) -> Any: """Return a bounded data preview for *path*. @@ -531,6 +534,10 @@ def preview( path = self.normalize_path(path) obj = self._get_object(path) kind = object_kind(obj) + if kind == "file": + from blosc2.b2view.file_preview import preview_file + + return preview_file(obj, raw=raw_text) if kind in {"ndarray", "c2array"}: shape = tuple(getattr(obj, "shape", ()) or ()) if slices is None: @@ -1397,6 +1404,8 @@ def object_kind(obj: Any) -> str: """Return a stable b2view kind string for *obj*.""" if isinstance(obj, blosc2.TreeStore): return "group" + if isinstance(obj, blosc2.RemoteFile): + return "file" if isinstance(obj, (blosc2.NDArray, blosc2.RemoteArray, blosc2.Proxy)): return "ndarray" if isinstance(obj, blosc2.CTable): @@ -1411,6 +1420,14 @@ def object_kind(obj: Any) -> str: def object_metadata(obj: Any) -> dict[str, Any]: """Extract lightweight metadata from a supported object.""" kind = object_kind(obj) + if kind == "file": + return { + "type": "Caterva2 file", + "name": obj.name, + "nbytes": obj.nbytes, + "cbytes": obj.cbytes, + "actions": "D: download original · O: open externally · T: raw/Markdown", + } if kind in {"ndarray", "c2array"}: try: cbytes = getattr(obj, "cbytes", None) diff --git a/src/blosc2/b2view/render.py b/src/blosc2/b2view/render.py index 2be0b5b68..1728ba997 100644 --- a/src/blosc2/b2view/render.py +++ b/src/blosc2/b2view/render.py @@ -41,6 +41,20 @@ def make_preview_renderables(preview: Any): from rich.table import Table from rich.text import Text + if isinstance(preview, dict) and "file_text" in preview: + from rich.markdown import Markdown + + body = ( + Markdown(preview["file_text"], hyperlinks=False) + if preview.get("markdown") + else Text(preview["file_text"]) + ) + return Text( + " · ".join(part for part in (preview.get("notice"), preview.get("message")) if part) + ), body + if isinstance(preview, dict) and "file_image" in preview: + return None, Text(preview["message"]) + if isinstance(preview, np.ndarray): return None, Text(np.array2string(preview, threshold=200, edgeitems=5), no_wrap=False) diff --git a/tests/b2view/test_files.py b/tests/b2view/test_files.py new file mode 100644 index 000000000..3d4a950ea --- /dev/null +++ b/tests/b2view/test_files.py @@ -0,0 +1,191 @@ +"""Ordinary-file preview and explicit download/external-open UI tests.""" + +import io + +import pytest +from test_remote_caterva2 import caterva2_source # noqa: F401 +from test_remote_file import file_source # noqa: F401 +from tui_wait import wait_until + +from blosc2.b2view.file_preview import IMAGE_PIXELS, TEXT_BYTES, open_external, preview_file, safe_text +from blosc2.b2view.model import StoreBrowser + + +class BytesFile: + def __init__(self, name, data): + self.name, self.data = name, data + self.nbytes = len(data) + self.reads = [] + + def read_bytes(self, start, stop): + self.reads.append((start, stop)) + return self.data[start:stop] + + +def test_text_limits_controls_and_binary_fallback(): + file = BytesFile("README.md", b"a" * (TEXT_BYTES - 1) + "é".encode() + b"long content") + result = preview_file(file) + assert result["markdown"] + assert "truncated" in result["notice"] + assert "\ufffd" not in result["file_text"] + assert file.reads == [(0, TEXT_BYTES)] + assert safe_text("hello\x1b[31mred\x1b[0m\x1b]8;;https://evil\x07link\x1b]8;;\x07\0") == "helloredlink" + assert "Binary" in preview_file(BytesFile("bad.md", b"a\0b"))["message"] + assert "Invalid UTF-8" in preview_file(BytesFile("bad.txt", b"a\xffb"))["notice"] + assert "truncated" in preview_file(BytesFile("many.txt", b"x\n" * 1001))["notice"] + assert not preview_file(BytesFile("README.md", b"# Title"), raw=True)["markdown"] + + +def test_pdf_and_binary_do_not_fetch(): + for name in ("doc.pdf", "unknown.bin"): + file = BytesFile(name, b"unknown payload") + assert "download" in preview_file(file)["message"] + assert file.reads == [] + + +def test_image_preview_and_limits(monkeypatch): + pil = pytest.importorskip("PIL.Image") + stream = io.BytesIO() + pil.new("RGB", (20, 10), "blue").save(stream, format="PNG") + result = preview_file(BytesFile("image.png", stream.getvalue())) + assert result["file_image"].size == (20, 10) + assert "20 × 10" in result["message"] + assert "Invalid" in preview_file(BytesFile("image.jpg", b"not a JPEG"))["message"] + monkeypatch.setattr("blosc2.b2view.file_preview.IMAGE_PIXELS", 10) + assert "budget" in preview_file(BytesFile("image.png", stream.getvalue()))["message"] + assert IMAGE_PIXELS <= (64 << 20) // 4 + + +def test_external_open_is_shell_free_and_restricted(tmp_path, monkeypatch): + import sys + + document = tmp_path / "--unsafe name.pdf" + document.write_bytes(b"%PDF-1.4\n") + seen = [] + monkeypatch.setattr("shutil.which", lambda name: "/usr/bin/" + name) + monkeypatch.setattr("subprocess.run", lambda args, **kw: seen.append((args, kw))) + if sys.platform == "win32": + pytest.skip("POSIX launcher contract") + open_external(document) + assert seen[0][0][1] == str(document.absolute()) + assert "shell" not in seen[0][1] + script = tmp_path / "run.py" + script.write_text("print('danger')") + with pytest.raises(ValueError, match="restricted"): + open_external(script) + document.write_bytes(b"not pdf") + with pytest.raises(ValueError, match="content"): + open_external(document) + + +def test_browser_file_metadata_and_preview(file_source): # noqa: F811 + base, _, stats = file_source + with StoreBrowser(base) as browser: + assert browser.kind("/README.md") == "file" + info = browser.get_info("/README.md") + assert info.display_path == "@public/README.md" + assert info.metadata["name"] == "README.md" + assert not stats["file_chunks"] + assert browser.preview("/README.md")["markdown"] + + +@pytest.mark.tui +@pytest.mark.asyncio +async def test_tui_file_preview_and_download(file_source, tmp_path): # noqa: F811 + from textual.widgets import Input, Static + + from blosc2.b2view.app import B2ViewApp, FileTransferScreen + + base, payload, _ = file_source + app = B2ViewApp(base, start_path="/README.md") + async with app.run_test(size=(120, 40)) as pilot: + await wait_until(pilot, lambda: app._selected_info is not None and app._selected_info.kind == "file") + assert app._selected_info.display_path == "@public/README.md" + await pilot.press("T") + assert app._file_raw + await wait_until(pilot, lambda: not app._remote_page_pending) + app.action_download_file() + await wait_until(pilot, lambda: isinstance(app.screen, FileTransferScreen)) + destination = tmp_path / "README.md" + app.screen.query_one(Input).value = str(destination) + await pilot.press("enter") + await wait_until(pilot, destination.exists) + assert destination.read_bytes() == payload + await wait_until( + pilot, lambda: "Saved" in str(app.screen.query_one("#file-status", Static).render()) + ) + await pilot.press("escape") + app.wait_for_close() + + +@pytest.mark.tui +@pytest.mark.asyncio +async def test_pdf_external_open_requires_consent(file_source, tmp_path, monkeypatch): # noqa: F811 + from textual.widgets import Checkbox, Input, Static + + import blosc2 + from blosc2.b2view.app import B2ViewApp, FileTransferScreen + + base, _, stats = file_source + data = b"%PDF-1.4\nDocument fixture" + stream = blosc2.SChunk(chunksize=64, cparams={"typesize": 1}) + stream.append_data(data) + stats["files"]["@public/doc.pdf"] = stream + stats["groups"]["@public"].append("doc.pdf") + launched = [] + monkeypatch.setattr("blosc2.b2view.file_preview.open_external", launched.append) + app = B2ViewApp(base, start_path="/doc.pdf") + async with app.run_test(size=(120, 40)) as pilot: + await wait_until(pilot, lambda: app._selected_info is not None and app._selected_info.kind == "file") + assert not stats["file_chunks"] + app.action_open_file() + await wait_until(pilot, lambda: isinstance(app.screen, FileTransferScreen)) + destination = tmp_path / "doc.pdf" + app.screen.query_one(Input).value = str(destination) + await pilot.press("enter") + assert not destination.exists() + assert not launched + assert "consent" in str(app.screen.query_one("#file-status", Static).render()) + app.screen.query_one(Checkbox).value = True + await pilot.press("enter") + await wait_until(pilot, lambda: bool(launched)) + assert destination.read_bytes() == data + await pilot.press("escape") + app.wait_for_close() + + +@pytest.mark.tui +@pytest.mark.asyncio +@pytest.mark.parametrize("image_widget", [True, False]) +async def test_image_widget_and_dependency_fallback(file_source, monkeypatch, image_widget): # noqa: F811 + from textual.widgets import Static + + import blosc2 + from blosc2.b2view.app import B2ViewApp + + pil = pytest.importorskip("PIL.Image") + base, _, stats = file_source + encoded = io.BytesIO() + pil.new("RGB", (20, 10), "red").save(encoded, format="PNG") + stream = blosc2.SChunk(chunksize=1024, cparams={"typesize": 1}) + stream.append_data(encoded.getvalue()) + stats["files"]["@public/image.png"] = stream + stats["groups"]["@public"].append("image.png") + monkeypatch.setattr( + "blosc2.b2view.app.TextualImage", + (lambda image: Static(f"Image {image.size}")) if image_widget else None, + ) + app = B2ViewApp(base, start_path="/image.png") + async with app.run_test(size=(120, 40)) as pilot: + if image_widget: + await wait_until(pilot, lambda: bool(app.query_one("#file-image").children)) + else: + await wait_until( + pilot, lambda: "needs textual-image" in str(app.query_one("#preview", Static).render()) + ) + app.update_panels("/mount") + await wait_until( + pilot, lambda: app._selected_info is not None and app._selected_info.kind == "group" + ) + await wait_until(pilot, lambda: not app.query_one("#file-image").children) + app.wait_for_close() diff --git a/tests/test_remote_file.py b/tests/test_remote_file.py index e83b6e819..91b6b1868 100644 --- a/tests/test_remote_file.py +++ b/tests/test_remote_file.py @@ -91,3 +91,78 @@ def test_metadata_validation(file_source): stats["file_metadata"]["@public/README.md"] = overrides with pytest.raises(ValueError): blosc2.open(base + "@public/README.md") + + +def test_empty_file_and_metadata_only(file_source, tmp_path): + base, _, stats = file_source + stats["files"]["@public/empty.txt"] = blosc2.SChunk(chunksize=64, cparams={"typesize": 1}) + with blosc2.open(base + "@public/empty.txt") as file: + assert file.read_bytes() == b"" + file.download(tmp_path / "empty.txt") + assert (tmp_path / "empty.txt").read_bytes() == b"" + assert not stats["file_chunks"] + + +def test_download_publish_race_and_cancellation(file_source, tmp_path): + base, _, _ = file_source + dest = tmp_path / "race.md" + with blosc2.open(base + "@public/README.md") as file: + with pytest.raises(FileExistsError): + file.download(dest, progress=lambda *_: dest.write_bytes(b"concurrent creator")) + assert dest.read_bytes() == b"concurrent creator" + assert not list(tmp_path.glob(".b2view-download-*")) + cancelled = [] + with pytest.raises(InterruptedError): + file.download( + dest, + overwrite=True, + cancel=lambda: bool(cancelled), + progress=lambda *_: cancelled.append(True), + ) + assert dest.read_bytes() == b"concurrent creator" + + +def test_corrupt_chunk_rejected_before_decompression(file_source, monkeypatch): + import struct + + import httpx + + base, _, _ = file_source + with blosc2.open(base + "@public/README.md", cache_policy=blosc2.CachePolicy.NONE) as file: + payload = b"\0" * 4 + struct.pack(" Date: Sat, 3 Oct 2026 14:49:19 +0200 Subject: [PATCH 10/35] Harden ordinary-file transfers and validate and document file browsing --- RELEASE_NOTES.md | 7 +++ doc/getting_started/installation.rst | 8 ++- doc/guides/b2view.rst | 34 +++++++++++ doc/guides/remote_objects.md | 10 +++- doc/reference/classes.rst | 1 + doc/reference/remotefile.rst | 45 ++++++++++++++ plans/caterva2-access-improvements2.md | 35 ++++++++++- pyproject.toml | 3 +- src/blosc2/b2view/model.py | 3 + src/blosc2/remote_file.py | 81 ++++++++++++++++++-------- src/blosc2/remote_store.py | 10 +++- src/blosc2/schunk.py | 5 ++ tests/b2view/test_files.py | 56 ++++++++++++++++++ tests/test_caterva2_gateway.py | 25 +++++++- tests/test_remote_file.py | 42 +++++++++++++ 15 files changed, 332 insertions(+), 33 deletions(-) create mode 100644 doc/reference/remotefile.rst diff --git a/RELEASE_NOTES.md b/RELEASE_NOTES.md index e6f5ba1fa..ddc2b11fa 100644 --- a/RELEASE_NOTES.md +++ b/RELEASE_NOTES.md @@ -48,6 +48,13 @@ XXX version-specific blurb XXX ### Caterva2 repository access +- Ordinary Caterva2 files now open lazily as `RemoteFile`: bounded original-byte + reads, shared compressed-chunk caching, and atomic streaming downloads. +- b2view previews text/Markdown and optional JPEG/PNG images (`blosc2[images]`). + PDF and other unrenderable content remain downloadable; supported documents + can open externally only after explicit trust consent. No PDF dependency is + required. `D` downloads, `O` opens externally, and `T` toggles raw Markdown. + - `blosc2.open()` recognizes HTTP(S) dataset URLs such as `http://localhost:8000/@public/group`, including deployment prefixes and IPv6. String service URLs default to lazy access; explicit `URLPath` behavior is unchanged. diff --git a/doc/getting_started/installation.rst b/doc/getting_started/installation.rst index 8d51e063c..fd0e689de 100644 --- a/doc/getting_started/installation.rst +++ b/doc/getting_started/installation.rst @@ -33,10 +33,13 @@ grouped into *extras* that you opt into with the ``blosc2[extra]`` syntax: - The :doc:`b2view <../guides/b2view>` terminal browser (``textual``, ``textual-plotext``), including its in-terminal braille plot (the ``p`` key). Required by the ``b2view`` command. - * - ``hires`` + * - ``hires`` - The high-resolution image view in b2view (the ``h`` key), which renders a real ``matplotlib`` image in the terminal - (``textual-image``, ``matplotlib``). Includes ``tui``. + (``textual-image``, ``matplotlib``). Includes ``tui``. + * - ``images`` + - JPEG/PNG ordinary-file previews in b2view (``Pillow``, ``textual-image``). + Includes ``tui`` without requiring matplotlib. ``hires`` includes this extra. * - ``parquet`` - The ``parquet-to-blosc2`` converter (``pyarrow``); see :doc:`../guides/parquet_to_blosc2`. @@ -59,6 +62,7 @@ argument in shells like ``zsh`` that treat brackets specially): .. code-block:: console pip install "blosc2[tui]" # the b2view terminal browser + pip install "blosc2[images]" # b2view + JPEG/PNG file previews pip install "blosc2[hires]" # b2view + its high-res view (h key) pip install "blosc2[parquet]" # the Parquet converter pip install "blosc2[fsspec]" # fsspec URLs, including HTTP(S) diff --git a/doc/guides/b2view.rst b/doc/guides/b2view.rst index d579bef62..a928e88c2 100644 --- a/doc/guides/b2view.rst +++ b/doc/guides/b2view.rst @@ -100,6 +100,40 @@ grouping are disabled rather than implicitly downloading the whole table. Table plotting requires a bounded locked row window (``v``). Source formats behind cat2lite need their dependencies on the server, not on each viewer client. +Ordinary Caterva2 files +^^^^^^^^^^^^^^^^^^^^^^ + +Published ordinary files (compressed SChunk byte streams) appear as file nodes. +Metadata inspection does not fetch their payload. Text and Markdown preview a +UTF-8 prefix, bounded to 64 KiB and 1,000 lines; truncated/invalid text is labelled. +Markdown links/images are passive and never fetch other resources. ``T`` toggles +raw text and Markdown; terminal controls are stripped from file content. + +Install ``blosc2[images]`` for JPEG/PNG previews using Pillow and textual-image, +without requiring matplotlib. Terminal protocol support may fall back to colored +half-cells. Missing dependencies/display support show a download/open notice. +Automatic image input is limited to 16 MiB; original pixel count is limited to +16,777,216 pixels (64 MiB RGBA). Orientation is corrected, the first frame is used, +and the displayed image is reduced to at most 1600 × 1200 pixels. These limits +bound individual buffers, not total process RAM including decoder copies. + +``D`` prompts for a destination and streams the **original file**, not its Blosc +carrier. ``O`` additionally asks for explicit trust consent before launching the +completed local download in the platform's external viewer. Both actions work for +PDFs, which deliberately need no terminal PDF renderer. External opening is +restricted to PDF, JPEG/PNG, Markdown, and text; other binary files remain +downloadable. No shell or credential-bearing URL is passed to the launcher. +Downloads are user-owned: they remain after closing the dialog or b2view, including +when external opening fails. Existing destinations are never overwritten by the +viewer; choose another name. Escape cancels an active transfer. + +File reads/downloads have per-chunk limits of 8 MiB compressed and 16 MiB decoded. +A small preview may require a much larger chunk. Oversized or irregular sources +require server-side rechunking; the viewer never silently fetches a whole file +to work around these limits. Local/direct-fsspec regular files and file reference +archives are not supported by this feature. Incompatible files do not hide healthy +siblings. + Direct source URLs ^^^^^^^^^^^^^^^^^^ diff --git a/doc/guides/remote_objects.md b/doc/guides/remote_objects.md index d4f415f6a..657265d16 100644 --- a/doc/guides/remote_objects.md +++ b/doc/guides/remote_objects.md @@ -8,12 +8,20 @@ one API. Data is read on demand and cached locally; remote sources are read-only | Standalone `.b2nd`, or a B2Z/Zarr/HDF5 array | {ref}`RemoteArray` | | Parquet file, B2Z CTable, PyTables table, or Caterva2 table | {ref}`RemoteCTable` | | B2Z, Zarr, HDF5, or Caterva2 group | {ref}`RemoteStore` | +| Caterva2 ordinary-file / fixed-chunk SChunk byte stream | {ref}`RemoteFile` | -All three inherit {ref}`RemoteObject` and expose source metadata, cache controls, +These types inherit {ref}`RemoteObject` and expose source metadata, cache controls, traffic counters, reference saving, and context-manager support. See {doc}`remote_arrays` for array computations and {doc}`remote_tables` for table queries and format-specific behavior. +For ordinary Caterva2 files, `file.read_bytes(start, stop)` reads original byte +ranges and `file.download(destination)` streams original content. Download is +atomic and refuses overwrite by default. File reference persistence is currently +unsupported; use download rather than `save`. Fixed-chunk byte streams share +cache budgets and owner lifetime with array/table leaves, but oversized chunks +and irregular layouts are refused. See {ref}`RemoteFile` for bounds and callbacks. + ```python import blosc2 diff --git a/doc/reference/classes.rst b/doc/reference/classes.rst index d7882d697..f783c4919 100644 --- a/doc/reference/classes.rst +++ b/doc/reference/classes.rst @@ -157,6 +157,7 @@ container APIs above. remotearray remotestore remoterepository + remotefile remotectable proxysource proxyndsource diff --git a/doc/reference/remotefile.rst b/doc/reference/remotefile.rst new file mode 100644 index 000000000..f0f0657eb --- /dev/null +++ b/doc/reference/remotefile.rst @@ -0,0 +1,45 @@ +.. _RemoteFile: + +RemoteFile +========== + +``RemoteFile`` is a read-only Caterva2 fixed-chunk SChunk byte stream. It is +returned by lazy service leaf opening or ``RemoteStore`` file lookup. Discovery +and file metadata do not fetch payload. Ordinary files such as Markdown, JPEG, +and PDF published as compressed byte streams are supported; there is no Python +object deserialization or implicit interpretation as an array. + +.. code-block:: python + + import blosc2 + + with blosc2.open("https://cat2.cloud/demo/@public/examples/README.md") as file: + prefix = file.read_bytes(0, min(file.nbytes, 4096)) + file.download("README.md") + +``read_bytes(start, stop)`` returns original bytes, with at most 16 MiB per call. +The omitted stop means the file end, not an unbounded streaming read. Use +``download`` for larger files. A chunk can contain more data than the requested +range: transfers are chunk-granular, bounded at 8 MiB compressed and 16 MiB +decoded per chunk. Oversized/irregular streams require rechunking on the server. +Generic fixed-chunk typed SChunks expose their raw byte representation; they do +not imply a text/image/document format. + +Downloads stage beside the destination, then publish completed original bytes. +Existing destinations are not overwritten unless ``overwrite=True`` is explicit. +No-overwrite publication uses a hard link to prevent concurrent-creation races; +the destination filesystem must support hard links. ``progress(done, total)`` +and ``cancel()`` callbacks run on the calling thread. Cancellation or failure +removes only the operation's staging file and preserves any previous destination. + +Cache policy/allowance, traffic, and lifetime are shared with the source owner. +``cache_bytes`` includes other leaves in that owner. MEMORY/DISK retain compressed +chunks; NONE retains no payload between calls. Transient decoded buffers are +separately bounded, not included in retained cache bytes. Sources must remain +immutable until explicit refresh; returned children outlive their parent handle. +``save`` of a file reference is intentionally unsupported; ``download`` is an +original-byte export, not a reference archive. Explicit nonlazy ``URLPath`` input +still requires ``lazy=True`` for byte-stream access. + +.. autoclass:: blosc2.RemoteFile + :members: read_bytes, download, close, source, name, media_type, nbytes, cbytes, nchunks, chunksize, attrs, traffic, info, cache_policy, max_cache_bytes, cache_bytes, metadata_bytes diff --git a/plans/caterva2-access-improvements2.md b/plans/caterva2-access-improvements2.md index 45fd4bcc3..10f713d39 100644 --- a/plans/caterva2-access-improvements2.md +++ b/plans/caterva2-access-improvements2.md @@ -1,6 +1,7 @@ # Caterva2 ordinary-file access and previews in b2view -Status: implementation in progress following explicit user authorization. +Status: M0–M5 implemented and validated on `cat2-improvements` following explicit +user authorization. The design below is retained as the delivery/acceptance record. It follows `plans/caterva2-access-improvements.md` and its completed URL discovery, lazy hierarchy browsing, and viewer integration work. @@ -29,6 +30,38 @@ lazy hierarchy browsing, and viewer integration work. plus file transport regressions: 175 passed with TUI cases enabled. Live demo checks: README renders; the JPEG decodes as 2034 × 1144 (bounded preview 1600 × 900); PDF offers download/external opening without fetching content on selection. +- M5/final review: added API/guide/installation/release documentation and the + `images` extra (Pillow/textual-image, no matplotlib; included by hires). Real + cat2lite native `.b2frame` file transport/download passes alongside existing + remote-format/shared-cache acceptance. Native cat2lite exposes this carrier + name; arbitrary ordinary-file publication/aliasing is not added to that server. +- Review hardening: bounded network and cached chunk reads before decompression; + identity HTTP encoding, content-length/body caps, 10-second I/O timeouts and a + checked 20-second chunk deadline; independent transfer aliases prepared on a + background worker; no launcher after stale selection/session or shutdown. + Old unsupported file snapshots rediscover metadata. Added download publication + race, cancellation, empty-stream, corruption, cache eviction/refresh, auth, + missing image dependency, consent, and responsive slow-transfer regressions. +- Final validation: default suite 10,670 passed, 38 skipped; focused offline + access/file/viewer/model suite 230 passed with headless TUI cases enabled; + opt-in live Caterva2 demo file downloads passed; both actual cat2lite acceptance + tests passed. Ruff/diff checks passed. HTML docs built using the existing + type-comment/notebook/generated-copy workarounds, with 764 wider autosummary/ + theme/cross-reference warnings; the build is not warnings-clean. +- Remaining constraints: fixed-chunk byte streams only; 8 MiB compressed / 16 MiB + decoded chunk caps apply to downloads too, so large single-chunk files need + server-side rechunking. No-overwrite downloads require hard-link-capable + destination filesystems. File reference persistence/hierarchy export is deferred. + MIME is a filename hint; binary/unknown files have no automatic preview. Image + decoder copies are not a whole-process memory limit. External launchers depend + on platform/GUI associations and are not a sandbox; tests mock launchers, never + open documents automatically. User-selected downloads are retained (no hidden + temporary external-viewer copies to manage). Local/fsspec regular-file support + and in-terminal PDF rendering remain explicit non-goals. +- Live headless image integration also verified the real demo JPEG mounts an + `AutoImage` widget with the installed textual-image package (not a mocked + renderer). Explicit downloads remain bounded even when previews are unavailable; + no automatic raw-download endpoint fallback bypasses chunk safety limits. ## 1. Goal and delivery order diff --git a/pyproject.toml b/pyproject.toml index 66e29ab23..07664b095 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -58,9 +58,10 @@ hdf5 = ["h5py", "hdf5plugin"] # wasm32 (no TTY). Install with `pip install "blosc2[tui]"`. This also pulls # textual-plotext for the in-terminal braille plot (the 'p' key). tui = ["textual", "textual-plotext"] +images = ["blosc2[tui]", "textual-image", "pillow"] # Adds the high-res 'h' view on top of [tui], rendering a real matplotlib image # (kitty/iTerm2/sixel, or half-cells elsewhere) — matplotlib is the heavy part. -hires = ["blosc2[tui]", "textual-image", "matplotlib"] +hires = ["blosc2[images]", "matplotlib"] # Read/write single-file containers through any fsspec URL (https://, s3://, # gs://, zip://, memory://...). HTTP support is included; the other protocol # backends (s3fs, gcsfs, adlfs...) are the caller's install: diff --git a/src/blosc2/b2view/model.py b/src/blosc2/b2view/model.py index 67f312e64..4a3485cdc 100644 --- a/src/blosc2/b2view/model.py +++ b/src/blosc2/b2view/model.py @@ -1424,8 +1424,11 @@ def object_metadata(obj: Any) -> dict[str, Any]: return { "type": "Caterva2 file", "name": obj.name, + "media type (hint)": obj.media_type, "nbytes": obj.nbytes, "cbytes": obj.cbytes, + "chunksize": obj.chunksize, + "nchunks": obj.nchunks, "actions": "D: download original · O: open externally · T: raw/Markdown", } if kind in {"ndarray", "c2array"}: diff --git a/src/blosc2/remote_file.py b/src/blosc2/remote_file.py index 03ddf0ebd..4102db9f2 100644 --- a/src/blosc2/remote_file.py +++ b/src/blosc2/remote_file.py @@ -4,6 +4,7 @@ import os import struct import tempfile +import time import weakref from pathlib import Path @@ -69,6 +70,13 @@ def name(self): self._check_open() return self.path.rsplit("/", 1)[-1] + @property + def media_type(self): + """A filename-based MIME hint, not a guarantee about the payload.""" + import mimetypes + + return mimetypes.guess_type(self.name)[0] + @property def nbytes(self): self._check_open() @@ -79,6 +87,16 @@ def cbytes(self): self._check_open() return self._meta["cbytes"] + @property + def chunksize(self): + self._check_open() + return self._meta["chunksize"] + + @property + def nchunks(self): + self._check_open() + return self._meta["nchunks"] + @property def attrs(self): self._check_open() @@ -125,12 +143,42 @@ def info_items(self): ("source", self.source), ("nbytes", self.nbytes), ("cbytes", self.cbytes), + ("chunksize", self._meta["chunksize"]), + ("nchunks", self._meta["nchunks"]), ("cache policy", self.cache_policy), ("cache bytes (owner)", self.cache_bytes), ] - def _chunk(self, index, cancel=None): + def _fetch_chunk(self, index, cancel): from blosc2.c2array import _auth_headers, _server_url, _sync_client + + owner = self._owner + url = _server_url(owner.caterva2.urlbase, f"api/chunk/{self.path}") + client = owner.transport or _sync_client() + data = bytearray() + deadline = time.monotonic() + 20 + headers = dict(_auth_headers(owner.caterva2.auth_token) or {}) + headers["Accept-Encoding"] = "identity" + with client.stream( + "GET", url, params={"nchunk": index}, headers=headers, timeout=10, follow_redirects=False + ) as response: + response.raise_for_status() + if response.headers.get("content-encoding", "identity") != "identity": + raise ValueError("Encoded file chunk responses are unsupported") + if int(response.headers.get("content-length", 0)) > MAX_COMPRESSED_CHUNK: + raise ValueError("File chunk exceeds 8 MiB compressed limit") + for part in response.iter_bytes(): + owner.traffic.charge(len(part)) + if time.monotonic() > deadline: + raise TimeoutError("File chunk transfer exceeded 20 seconds") + if cancel is not None and cancel(): + raise InterruptedError("File operation cancelled") + if len(data) + len(part) > MAX_COMPRESSED_CHUNK: + raise ValueError("File chunk exceeds 8 MiB compressed limit") + data.extend(part) + return bytes(data) + + def _chunk(self, index, cancel=None): from blosc2.remote_store import _Caterva2FrameCache self._check_open() @@ -146,31 +194,10 @@ def _chunk(self, index, cancel=None): if owner.table_frame_cache is None: owner.table_frame_cache = _Caterva2FrameCache(owner) cache = owner.table_frame_cache - payload = None if cache is None else cache.get(key) + payload = None if cache is None else cache.get(key, max_bytes=MAX_COMPRESSED_CHUNK) missing = payload is None if payload is None: - url = _server_url(owner.caterva2.urlbase, f"api/chunk/{self.path}") - client = owner.transport or _sync_client() - data = bytearray() - with client.stream( - "GET", - url, - params={"nchunk": index}, - headers=_auth_headers(owner.caterva2.auth_token), - timeout=10, - follow_redirects=False, - ) as response: - response.raise_for_status() - if response.is_redirect: - raise ValueError("File chunk redirects are unsupported") - for part in response.iter_bytes(chunk_size=64 << 10): - owner.traffic.charge(len(part)) - if cancel is not None and cancel(): - raise InterruptedError("File operation cancelled") - if len(data) + len(part) > MAX_COMPRESSED_CHUNK: - raise ValueError("File chunk exceeds 8 MiB compressed limit") - data.extend(part) - payload = bytes(data) + payload = self._fetch_chunk(index, cancel) if len(payload) < 16 or len(payload) > MAX_COMPRESSED_CHUNK: raise ValueError("Invalid compressed file chunk length") nbytes, _, cbytes = struct.unpack_from(" max_bytes: + raise ValueError("Cached file chunk exceeds compressed limit") self._cache_lru.pop(key, None) self._cache_lru[key] = None self.owner.cache_coordinator.touch(self, key) diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index 8818b3c34..76d690585 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2861,6 +2861,7 @@ def open( # noqa: C901 | blosc2.CTable | blosc2.RemoteCTable | blosc2.RemoteStore + | blosc2.RemoteFile | blosc2.LazyArray | blosc2.Proxy | blosc2.DictStore @@ -3033,6 +3034,10 @@ def open( # noqa: C901 :class:`RemoteRepository`. ``"caterva2"`` requires service access and disables ordinary-file fallback. Client cache budgets on a repository are per root, independent of the server's shared cache. + Regular fixed-chunk Caterva2 SChunk leaves return :class:`RemoteFile` + with lazy access. ``read_bytes`` reads bounded original-byte ranges; + ``download`` streams the original file without exporting its Blosc + carrier. Irregular streams and oversized chunks are refused explicitly. parquet_options: dict, optional PyArrow ``ParquetFile`` reader options for a Parquet source. Conversion options such as ``columns`` and ``max_rows`` are passed separately. diff --git a/tests/b2view/test_files.py b/tests/b2view/test_files.py index 3d4a950ea..9cdef52b2 100644 --- a/tests/b2view/test_files.py +++ b/tests/b2view/test_files.py @@ -43,6 +43,13 @@ def test_pdf_and_binary_do_not_fetch(): assert file.reads == [] +def test_missing_image_dependency_does_not_read(monkeypatch): + monkeypatch.setitem(__import__("sys").modules, "PIL", None) + file = BytesFile("image.png", b"not needed") + assert "needs Pillow" in preview_file(file)["message"] + assert file.reads == [] + + def test_image_preview_and_limits(monkeypatch): pil = pytest.importorskip("PIL.Image") stream = io.BytesIO() @@ -189,3 +196,52 @@ async def test_image_widget_and_dependency_fallback(file_source, monkeypatch, im ) await wait_until(pilot, lambda: not app.query_one("#file-image").children) app.wait_for_close() + + +@pytest.mark.tui +@pytest.mark.asyncio +async def test_slow_transfer_can_cancel_without_blocking_ui(file_source, tmp_path, monkeypatch): # noqa: F811 + import threading + + from textual.widgets import Input + + import blosc2 + from blosc2.b2view.app import B2ViewApp, FileTransferScreen + + entered, release, finished = threading.Event(), threading.Event(), threading.Event() + original = blosc2.RemoteFile._chunk + + def slow(file, index, cancel=None): + if cancel is not None: + entered.set() + assert release.wait(5) + try: + return original(file, index, cancel) + finally: + if cancel is not None: + finished.set() + + monkeypatch.setattr(blosc2.RemoteFile, "_chunk", slow) + base, _, _ = file_source + app = B2ViewApp(base, start_path="/README.md") + destination = tmp_path / "cancelled.md" + try: + async with app.run_test(size=(120, 40)) as pilot: + await wait_until( + pilot, lambda: app._selected_info is not None and app._selected_info.kind == "file" + ) + app.action_download_file() + await wait_until(pilot, lambda: isinstance(app.screen, FileTransferScreen)) + app.screen.query_one(Input).value = str(destination) + await pilot.press("enter") + await wait_until(pilot, entered.is_set) + await pilot.press("escape") + assert not isinstance(app.screen, FileTransferScreen) + await pilot.press("tab") + release.set() + await wait_until(pilot, finished.is_set) + await wait_until(pilot, lambda: not list(tmp_path.glob(".b2view-download-*"))) + assert not destination.exists() + finally: + release.set() + app.wait_for_close() diff --git a/tests/test_caterva2_gateway.py b/tests/test_caterva2_gateway.py index 459ef42e1..5342f36b4 100644 --- a/tests/test_caterva2_gateway.py +++ b/tests/test_caterva2_gateway.py @@ -32,8 +32,9 @@ def gateway(base, catalog): f'[server]\nlisten = "127.0.0.1:0"\npython = {json.dumps(sys.executable)}\n' f"[remote]\ncache_dir = {json.dumps(str(base / 'server-cache'))}\n" ) + source_args = ["--data-dir", str(catalog)] if catalog.is_dir() else [str(catalog)] process = subprocess.Popen( - [os.environ["CAT2LITE_SERVER"], "--config", str(config), str(catalog)], + [os.environ["CAT2LITE_SERVER"], "--config", str(config), *source_args], stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True, @@ -43,7 +44,7 @@ def gateway(base, catalog): line = process.stdout.readline() while line and not line.startswith("listening on "): line = process.stdout.readline() - assert line.startswith("listening on "), line + assert line.startswith("listening on "), line or process.stderr.read() address = line.removeprefix("listening on ").strip() yield address if address.startswith("http") else "http://" + address finally: @@ -53,6 +54,26 @@ def gateway(base, catalog): stream.close() +def test_real_cat2lite_schunk_file_download(tmp_path): + from blosc2.b2view.model import StoreBrowser + + data = tmp_path / "files" + data.mkdir() + payload = b"ordinary file bytes\n" * 10 + stream = blosc2.SChunk(chunksize=64, cparams={"typesize": 1}) + for offset in range(0, len(payload), 64): + stream.append_data(payload[offset : offset + 64]) + (data / "bytes.b2frame").write_bytes(stream.to_cframe()) + with gateway(tmp_path, data) as url: + with StoreBrowser(url) as browser: + assert browser.kind("/bytes.b2frame") == "file" + assert browser.get_info("/bytes.b2frame").metadata["nbytes"] == len(payload) + with blosc2.open(url + "/@public/bytes.b2frame") as file: + assert file.read_bytes(61, 135) == payload[61:135] + file.download(tmp_path / "original.bin") + assert (tmp_path / "original.bin").read_bytes() == payload + + def test_real_catalog_browsing_and_shared_server_cache(tmp_path): h5py = pytest.importorskip("h5py") zarr = pytest.importorskip("zarr") diff --git a/tests/test_remote_file.py b/tests/test_remote_file.py index 91b6b1868..69a0c0ed5 100644 --- a/tests/test_remote_file.py +++ b/tests/test_remote_file.py @@ -39,6 +39,8 @@ def test_discovery_ranges_and_download(file_source, tmp_path): assert progress[-1] == (len(payload), len(payload)) with pytest.raises(FileExistsError): file.download(dest) + with pytest.raises(TypeError, match="bool"): + file.download(dest, overwrite="false") with pytest.raises(NotImplementedError): file.save(tmp_path / "file.b2z") @@ -154,6 +156,46 @@ def test_budget_eviction_and_refresh(file_source): assert stats["file_chunks"] +def test_response_bounds_and_auth_isolation(file_source, monkeypatch): + import httpx + + base, _, stats = file_source + calls = [] + for token in ("alice=secret", "bob=secret"): + with blosc2.c2context(auth_token=token), blosc2.open(base + "@public/README.md") as file: + file.read_bytes(0, 1) + assert set(stats["cookies"]) >= {"alice=secret", "bob=secret"} + assert len(stats["file_chunks"]) == 2 + with blosc2.open(base + "@public/README.md", cache_policy=blosc2.CachePolicy.NONE) as file: + + def handler(request): + calls.append(request) + return httpx.Response(200, headers={"content-length": str(9 << 20)}, content=b"small") + + with httpx.Client(transport=httpx.MockTransport(handler)) as client: + file._owner.transport = client + with pytest.raises(ValueError, match="8 MiB"): + file.read_bytes(0, 1) + assert calls[-1].headers["Accept-Encoding"] == "identity" + file._meta.update(nbytes=17 << 20, chunksize=17 << 20, nchunks=1) + with pytest.raises(ValueError, match="16 MiB decoded"): + file.read_bytes(0, 1) + + +def test_old_unsupported_file_snapshot_is_rediscovered(file_source, tmp_path): + base, _, _ = file_source + options = {"cache_dir": tmp_path / "cache"} + with blosc2.open(base + "@public", **options) as store: + assert store.kind("README.md") == "file" + store._owner.save_manifest() + manifest = store._owner.disk.load() + manifest["nodes"]["@public/README.md"] = ["unsupported", "legacy"] + manifest["metadata"].pop("caterva2_files") + store._owner.disk.publish(manifest) + with blosc2.open(base + "@public", **options) as store: + assert store.kind("README.md") == "file" + + @pytest.mark.network def test_live_caterva2_demo_original_files(tmp_path): base = "https://cat2.cloud/demo/@public/examples/" From 1a480f553da5e45ad2ef56e88534e1762025a0ce Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sat, 3 Oct 2026 15:00:27 +0200 Subject: [PATCH 11/35] Fix terminal image layout and simplify external opening --- RELEASE_NOTES.md | 2 +- doc/guides/b2view.rst | 5 +++-- plans/caterva2-access-improvements2.md | 6 ++++++ src/blosc2/b2view/app.py | 25 +++++++++++------------- tests/b2view/test_files.py | 27 ++++++++++++++++---------- 5 files changed, 38 insertions(+), 27 deletions(-) diff --git a/RELEASE_NOTES.md b/RELEASE_NOTES.md index ddc2b11fa..e6fa879ae 100644 --- a/RELEASE_NOTES.md +++ b/RELEASE_NOTES.md @@ -52,7 +52,7 @@ XXX version-specific blurb XXX reads, shared compressed-chunk caching, and atomic streaming downloads. - b2view previews text/Markdown and optional JPEG/PNG images (`blosc2[images]`). PDF and other unrenderable content remain downloadable; supported documents - can open externally only after explicit trust consent. No PDF dependency is + can open externally through the explicit `O` action and destination dialog. No PDF dependency is required. `D` downloads, `O` opens externally, and `T` toggles raw Markdown. - `blosc2.open()` recognizes HTTP(S) dataset URLs such as diff --git a/doc/guides/b2view.rst b/doc/guides/b2view.rst index a928e88c2..460ac119d 100644 --- a/doc/guides/b2view.rst +++ b/doc/guides/b2view.rst @@ -118,8 +118,9 @@ and the displayed image is reduced to at most 1600 × 1200 pixels. These limits bound individual buffers, not total process RAM including decoder copies. ``D`` prompts for a destination and streams the **original file**, not its Blosc -carrier. ``O`` additionally asks for explicit trust consent before launching the -completed local download in the platform's external viewer. Both actions work for +carrier. ``O`` downloads and opens the completed local file in the platform's +external viewer after you submit the destination dialog, without an extra trust +checkbox. Both actions work for PDFs, which deliberately need no terminal PDF renderer. External opening is restricted to PDF, JPEG/PNG, Markdown, and text; other binary files remain downloadable. No shell or credential-bearing URL is passed to the launcher. diff --git a/plans/caterva2-access-improvements2.md b/plans/caterva2-access-improvements2.md index 10f713d39..1f75869da 100644 --- a/plans/caterva2-access-improvements2.md +++ b/plans/caterva2-access-improvements2.md @@ -7,6 +7,12 @@ lazy hierarchy browsing, and viewer integration work. ## Implementation record +- Follow-up UX decision: the extra trust checkbox is removed for all supported + documents, including images and PDFs. The explicit O action and destination + submission initiate external opening; signature/type restrictions and stale + request/shutdown guards remain. Earlier checkbox requirements below are + superseded by this user-requested decision. + - M0: verified Caterva2 demo README transport: `api/chunk?nchunk=0` returns a 552-byte compressed chunk that decompresses to the original 811 bytes; `api/download` returns original Markdown bytes, not the `.b2` carrier. diff --git a/src/blosc2/b2view/app.py b/src/blosc2/b2view/app.py index 2b0ce5a70..1f83e1b20 100644 --- a/src/blosc2/b2view/app.py +++ b/src/blosc2/b2view/app.py @@ -42,7 +42,7 @@ PlotextPlot = None try: - # Auto-selects the best terminal image protocol (kitty/iTerm2/sixel), + # Auto-selects a supported terminal image protocol (kitty/sixel), # degrading to colored half-cells; used by the high-res 'h' plot view. from textual_image.widget import Image as TextualImage except ImportError: # high-res view is optional @@ -297,7 +297,7 @@ class HelpScreen(ModalScreen[None]): "Ordinary files", [ ("D", "download original bytes to a chosen destination (no overwrite)"), - ("O", "download and open a document externally (requires explicit trust consent)"), + ("O", "download and open a document externally"), ("T", "toggle raw text / Markdown rendering"), ("escape", "cancel a file download or close its dialog"), ], @@ -1895,7 +1895,7 @@ def on_progress(downloaded: int, content_total: int | None) -> None: class FileTransferScreen(ModalScreen): - """Explicit destination/consent and cancellable background original-byte download.""" + """Explicit destination and cancellable background original-byte download/open.""" CSS = """ FileTransferScreen { align: center middle; } @@ -1916,17 +1916,14 @@ def compose(self): from blosc2.b2view.file_preview import safe_text name = safe_text(self.file.name).replace("\\", "_").replace("/", "_") + action = "Download and open externally" if self.external else "Download original file" with Vertical(id="file-transfer"): yield Static( - f"Download original file: {name} ({self.file.nbytes:,} bytes)\n" + f"{action}: {name} ({self.file.nbytes:,} bytes)\n" "Choose a destination; Enter starts. Escape cancels. Existing files are not overwritten.", markup=False, ) yield Input(value=str(Path.cwd() / name), id="file-destination") - if self.external: - yield Checkbox( - "I trust this file and want to open it in an external application", id="file-consent" - ) yield Static("", id="file-status", markup=False) yield ProgressBar(id="file-progress") @@ -1937,11 +1934,6 @@ def on_mount(self): def on_input_submitted(self): if self.started: return - if self.external and not self.query_one("#file-consent", Checkbox).value: - self.query_one("#file-status", Static).update( - "Explicit consent is required for external opening." - ) - return destination = self.query_one(Input).value if not destination.strip(): return @@ -2658,7 +2650,12 @@ async def _show_file_image(self, path, data): ) return try: - await body.mount(TextualImage(data["file_image"])) + image = TextualImage(data["file_image"]) + # The default fractional height collapses inside an auto-height + # container. Let the widget derive its height from the image ratio. + image.styles.width = "auto" + image.styles.height = "auto" + await body.mount(image) except Exception as error: self.query_one("#preview", Static).update(f"Image display unavailable: {error}; use D or O.") diff --git a/tests/b2view/test_files.py b/tests/b2view/test_files.py index 9cdef52b2..55e30eb7e 100644 --- a/tests/b2view/test_files.py +++ b/tests/b2view/test_files.py @@ -127,8 +127,8 @@ async def test_tui_file_preview_and_download(file_source, tmp_path): # noqa: F8 @pytest.mark.tui @pytest.mark.asyncio -async def test_pdf_external_open_requires_consent(file_source, tmp_path, monkeypatch): # noqa: F811 - from textual.widgets import Checkbox, Input, Static +async def test_pdf_external_open_requires_only_explicit_action(file_source, tmp_path, monkeypatch): # noqa: F811 + from textual.widgets import Input import blosc2 from blosc2.b2view.app import B2ViewApp, FileTransferScreen @@ -149,11 +149,9 @@ async def test_pdf_external_open_requires_consent(file_source, tmp_path, monkeyp await wait_until(pilot, lambda: isinstance(app.screen, FileTransferScreen)) destination = tmp_path / "doc.pdf" app.screen.query_one(Input).value = str(destination) - await pilot.press("enter") + assert not app.screen.query("#file-consent") assert not destination.exists() assert not launched - assert "consent" in str(app.screen.query_one("#file-status", Static).render()) - app.screen.query_one(Checkbox).value = True await pilot.press("enter") await wait_until(pilot, lambda: bool(launched)) assert destination.read_bytes() == data @@ -163,7 +161,7 @@ async def test_pdf_external_open_requires_consent(file_source, tmp_path, monkeyp @pytest.mark.tui @pytest.mark.asyncio -@pytest.mark.parametrize("image_widget", [True, False]) +@pytest.mark.parametrize("image_widget", [True, False, "real"]) async def test_image_widget_and_dependency_fallback(file_source, monkeypatch, image_widget): # noqa: F811 from textual.widgets import Static @@ -178,14 +176,23 @@ async def test_image_widget_and_dependency_fallback(file_source, monkeypatch, im stream.append_data(encoded.getvalue()) stats["files"]["@public/image.png"] = stream stats["groups"]["@public"].append("image.png") - monkeypatch.setattr( - "blosc2.b2view.app.TextualImage", - (lambda image: Static(f"Image {image.size}")) if image_widget else None, - ) + if image_widget == "real": + pytest.importorskip("textual_image.widget") + else: + monkeypatch.setattr( + "blosc2.b2view.app.TextualImage", + (lambda image: Static(f"Image {image.size}")) if image_widget else None, + ) app = B2ViewApp(base, start_path="/image.png") async with app.run_test(size=(120, 40)) as pilot: if image_widget: await wait_until(pilot, lambda: bool(app.query_one("#file-image").children)) + await wait_until(pilot, lambda: app.query_one("#file-image").children[0].region.height > 0) + image = app.query_one("#file-image").children[0] + assert image.region.overlaps(app.query_one("#data-scroll").region) + await pilot.resize_terminal(80, 30) + await wait_until(pilot, lambda: image.region.height > 0) + assert image.region.overlaps(app.query_one("#data-scroll").region) else: await wait_until( pilot, lambda: "needs textual-image" in str(app.query_one("#preview", Static).render()) From 9f0760a7becba464acef317377020f60d2a8f3cc Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sat, 3 Oct 2026 15:15:54 +0200 Subject: [PATCH 12/35] Fix Caterva2 structured and subarray dtypes and audit demo datasets --- RELEASE_NOTES.md | 3 + plans/caterva2-demo-dataset-audit.md | 140 +++++++++++++++++++++++++++ src/blosc2/c2array.py | 9 +- src/blosc2/remote_store.py | 11 ++- tests/test_caterva2_dtype.py | 124 ++++++++++++++++++++++++ tests/test_remote_caterva2.py | 18 +++- 6 files changed, 299 insertions(+), 6 deletions(-) create mode 100644 plans/caterva2-demo-dataset-audit.md create mode 100644 tests/test_caterva2_dtype.py diff --git a/RELEASE_NOTES.md b/RELEASE_NOTES.md index e6fa879ae..b60fa786f 100644 --- a/RELEASE_NOTES.md +++ b/RELEASE_NOTES.md @@ -48,6 +48,9 @@ XXX version-specific blurb XXX ### Caterva2 repository access +- Fix structured and subarray dtype decoding in `C2Array` and synthesized + RemoteStore chunks, including compound/subarray HDF5 leaves. + - Ordinary Caterva2 files now open lazily as `RemoteFile`: bounded original-byte reads, shared compressed-chunk caching, and atomic streaming downloads. - b2view previews text/Markdown and optional JPEG/PNG images (`blosc2[images]`). diff --git a/plans/caterva2-demo-dataset-audit.md b/plans/caterva2-demo-dataset-audit.md new file mode 100644 index 000000000..5fdfae937 --- /dev/null +++ b/plans/caterva2-demo-dataset-audit.md @@ -0,0 +1,140 @@ +# Caterva2 demo dataset audit + +Audit of `b2view https://cat2.cloud/demo`, 2026-10-03, in the `blosc2` +environment. The live catalog has one root, `@public`. All paths below are +relative to that root. This is a snapshot, not a guarantee about future server +contents or installed codec versions. + +## Method and fixes + +- Enumerated `api/roots`, `api/list/@public`, and the HDF5 mount's own listing. +- Inspected every listed leaf's metadata through `StoreBrowser.get_info` and + attempted a bounded preview (`max_rows=2`, `max_cols=2`). The legacy N-D + preview path uses up to 20 rows of one plane. Array reads remain chunk-granular; + no whole large array, notebook, PDF, or Parquet file was downloaded. +- There are 32 top-level listing entries: 31 leaves and one HDF5 group. That + group exposes 11 more leaves: **42 leaves audited individually**. +- Fixed `C2Array.dtype` to accept literal structured/subarray descriptors, as + native b2nd readers already do. Both NumPy `TypeError` and `ValueError` paths + matter. Parsing uses `ast.literal_eval`, not executable `eval`. +- Fixed synthesized Caterva2 chunks for top-level subarray dtypes: NumPy expands + these into trailing dimensions, so chunk/block geometry must expand too when + repacking. The public source shape/dtype remain unchanged. Tests include + nonzero data, multiple chunks, partial edge chunks, block padding, and both + one- and two-dimensional outer shapes. +- Five previously failing dtype cases now preview successfully: three native + structured arrays, HDF5 compound dtype, and HDF5 subarray dtype. The literal + directory name `unsupported` is not authoritative about current capabilities. + +## Every top-level entry + +| Path | Result after fixes | Analysis | +| --- | --- | --- | +| `examples/README.md` | Text/Markdown preview works | Ordinary SChunk byte stream. | +| `examples/Wutujing-River.jpg` | Image preview works | JPEG; Pillow decodes and reduces the image. Terminal image layout was fixed separately. | +| `examples/cat2cloud-brochure.pdf` | No inline preview, intentional | PDF download/external opening supported; no embedded PDF renderer. | +| `examples/cube-1k-1k-1k.b2nd` | Numeric preview works | `int32`, shape `(1000,1000,1000)`; preview of one plane, not the entire volume. | +| `examples/cubeA.b2nd` | Numeric preview works | `float64`, shape `(1,1000,1000)`. | +| `examples/cubeB.b2nd` | Numeric preview works | `float64`, shape `(1,1000,1000)`. | +| `examples/dir1/ds-2d.b2nd` | Numeric preview works | `uint16`, shape `(10,20)`; chunks need edge/block padding. | +| `examples/dir1/ds-3d.b2nd` | Numeric preview works | `float32`, shape `(3,4,5)`. | +| `examples/dir2/ds-4d.b2nd` | Complex preview works | `complex128`, shape `(2,3,4,5)`; no dtype parsing failure. | +| `examples/ds-1d-b.b2nd` | Byte-string preview works | `S6`, shape `(1000,)`; values contain `foobar`. | +| `examples/ds-1d-fields.b2nd` | **Fixed** | Structured integer/float/byte-string/bool dtype was a stringified field list. | +| `examples/ds-1d.b2nd` | Numeric preview works | `int64`, shape `(1000,)`. | +| `examples/ds-2d-fields.b2nd` | **Fixed** | Structured float32/float64 dtype; shape `(100,200)`. | +| `examples/ds-hello.b2frame` | Downloadable, no automatic preview | Plain SChunk has no text suffix. A bounded read confirms repeated `Hello world!` bytes. It is not corrupt; content sniffing/raw SChunk previews are not implemented. | +| `examples/ds-sc-attr.b2nd` | Scalar string preview works | Zero-dimensional ``: +each returns HTTP 500, including `company`, so this is not just dotted names. +Without `field`, `slice_=0:2` returns 5,658 bytes that decode successfully. +The viewer/RemoteCTable uses column projection and therefore encounters this +server defect. A blanket retry without projection was not introduced: it would +hide server errors and could multiply transfer sizes. The deployment needs a +projection fix, or a deliberately bounded/capability-aware compatibility path. + +### Passive preview limitations (eight leaves) + +The PDF, plain `.b2frame`, five notebooks, and opaque Parquet file are recognized +file handles, not unrecognized nodes. Unsupported preview does not mean broken +discovery. PDF has explicit external opening; notebook/SChunk content can be +downloaded and examined separately. Parquet is the exception: this particular +server's chunk geometry exceeds byte-stream download limits too. Direct Python +Blosc2 Parquet-source support does not imply a Caterva2 opaque file is a table. + +## Summary + +After the dtype/chunk fixes: **30 leaves successfully preview**, **8 have no +automatic preview by current policy**, and **4 fail for external reasons** +(3 local GROK-plugin loads, 1 server table-projection error). The HDF5 group's +`unsupported` directory now has three readable server representations. These +findings describe sampled reads, not full-data integrity or equality to original +HDF5 sources. + +## Validation + +- Default suite: **10,685 passed, 38 skipped** in the `blosc2` environment. +- Focused offline dtype/access/viewer/model tests: **234 passed**. +- Opt-in live regression tests for all five repaired dtype leaves: **5 passed**. +- Actual headless `B2ViewApp` sessions populated the data grid for each of those + five live leaves (including HDF5 subarray cells), not just model-only reads. diff --git a/src/blosc2/c2array.py b/src/blosc2/c2array.py index 59c12d1c2..4567e00f7 100644 --- a/src/blosc2/c2array.py +++ b/src/blosc2/c2array.py @@ -7,6 +7,7 @@ from __future__ import annotations +import ast import asyncio import atexit import json @@ -1607,7 +1608,13 @@ def blocks(self) -> tuple[int]: @property def dtype(self) -> np.dtype: """The dtype of the remote array""" - return np.dtype(self.meta["dtype"]) + dtype = self.meta["dtype"] + try: + return np.dtype(dtype) + except (TypeError, ValueError): + # Caterva2 sends structured/subarray descriptors as their Python repr, + # just like the b2nd metalayer. Parse literals, never executable code. + return np.dtype(ast.literal_eval(dtype)) @property def cparams(self) -> blosc2.CParams: diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index fcfb76b81..58a52a25d 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -64,7 +64,16 @@ def get_chunk(self, nchunk): full = np.zeros(self.chunks, dtype=self.dtype) region = tuple(slice(0, item.stop - item.start) for item in selection) full[region] = data[...] - packed = blosc2.asarray(full, chunks=self.chunks, blocks=self.blocks, cparams=self.cparams) + # NumPy expands a top-level subarray dtype into trailing dimensions. + # Keep the synthesized chunk's block ordering consistent with that + # expanded buffer, without changing the source's logical geometry. + inner_shape = self.dtype.subdtype[1] if self.dtype.subdtype is not None else () + packed = blosc2.asarray( + full, + chunks=(*self.chunks, *inner_shape), + blocks=(*self.blocks, *inner_shape), + cparams=self.cparams, + ) return packed.schunk.get_chunk(0) diff --git a/tests/test_caterva2_dtype.py b/tests/test_caterva2_dtype.py new file mode 100644 index 000000000..cd3a7ee7d --- /dev/null +++ b/tests/test_caterva2_dtype.py @@ -0,0 +1,124 @@ +"""Caterva2 dtype representations must agree with native b2nd decoding.""" + +import numpy as np +import pytest +from test_remote_caterva2 import caterva2_source # noqa: F401 + +import blosc2 +from blosc2.b2view.model import StoreBrowser + + +@pytest.mark.parametrize( + "dtype", + [ + np.dtype("int32"), + np.dtype("S6"), + np.dtype("U6"), + np.dtype("complex128"), + np.dtype([("a", " Date: Mon, 5 Oct 2026 05:58:46 +0200 Subject: [PATCH 13/35] Exit b2view gracefully when its initial source cannot open --- RELEASE_NOTES.md | 3 + src/blosc2/b2view/app.py | 47 +++++++-- src/blosc2/b2view/cli.py | 2 +- tests/b2view/test_startup_errors.py | 153 ++++++++++++++++++++++++++++ 4 files changed, 198 insertions(+), 7 deletions(-) create mode 100644 tests/b2view/test_startup_errors.py diff --git a/RELEASE_NOTES.md b/RELEASE_NOTES.md index b60fa786f..d36df17ee 100644 --- a/RELEASE_NOTES.md +++ b/RELEASE_NOTES.md @@ -48,6 +48,9 @@ XXX version-specific blurb XXX ### Caterva2 repository access +- b2view now exits cleanly with a clear diagnostic and nonzero status when + its initial source cannot be opened, rather than leaving an unusable TUI. + - Fix structured and subarray dtype decoding in `C2Array` and synthesized RemoteStore chunks, including compound/subarray HDF5 leaves. diff --git a/src/blosc2/b2view/app.py b/src/blosc2/b2view/app.py index 1f83e1b20..d1842c7c8 100644 --- a/src/blosc2/b2view/app.py +++ b/src/blosc2/b2view/app.py @@ -2158,6 +2158,8 @@ def __init__( self.preview_rows = preview_rows self.preview_cols = preview_cols self.browser: StoreBrowser | None = None + self._source_opened = False + self.startup_error: str | None = None # Set when a remote browser is closed on its own thread (on_unmount); # lets teardown wait for the cache-dir lock to be released. self._browser_close_thread: threading.Thread | None = None @@ -2259,7 +2261,7 @@ def _after_download(self, result: bool | str) -> None: if result is True: self._start_browsing() else: - self.exit(message=f"Download failed: {result}") + self._source_open_error(RuntimeError(f"Download failed: {result}")) def _start_browsing(self) -> None: """Open the bundle and populate the tree (the normal startup path).""" @@ -2274,7 +2276,12 @@ def _start_browsing(self) -> None: } if self.max_cache_bytes is not None: browser_kwargs["max_cache_bytes"] = self.max_cache_bytes - self.browser = StoreBrowser(self.urlpath, **browser_kwargs) + try: + self.browser = StoreBrowser(self.urlpath, **browser_kwargs) + except Exception as exc: + self._source_open_error(exc) + return + self._source_opened = True self._populate_browser() def _populate_browser(self) -> None: @@ -2439,15 +2446,18 @@ def _open_remote(self, session, start_path): browser.close() except Exception as exc: if browser is not None: - browser.close() - self._deliver_remote(session, self._remote_error, exc) + # Cleanup must not mask the opening failure and strand the UI. + with contextlib.suppress(Exception): + browser.close() + self._deliver_remote(session, self._source_open_error, exc) def _finish_remote_open(self, browser, children): self.browser = browser + self._source_opened = True self._remote_children = children self._populate_browser() - def _remote_error(self, exc): + def _error_message(self, exc): # Transport exceptions can include signed URLs or credentials. Keep # source-specific limitations, but remove runtime URLs and option values. import re @@ -2463,7 +2473,32 @@ def redact(options): message = message.replace(value, "