diff --git a/RELEASE_NOTES.md b/RELEASE_NOTES.md index 5d0ace25f..1d7bd308c 100644 --- a/RELEASE_NOTES.md +++ b/RELEASE_NOTES.md @@ -46,6 +46,87 @@ XXX version-specific blurb XXX requirements, environment precedence, caching, and diagnostics. Corrected the generated allocation declaration that caused an Apple Clang warning. +### Caterva2 repository access + +- Show unobtrusive animated loading dots in b2view's data-pane border while + background metadata/previews/pages are pending, without hiding current data. + +- Browse fsspec `.b2` document carriers with bounded range-based metadata and + chunk reads, original-name previews and streamed decoded copies. HTTP requires + validated byte-range responses; native dataset behavior remains unchanged. + +- Raise document chunk limits to 32 MiB compressed / 256 MiB decoded for + Caterva2 files and local `.b2` carriers. Preview and per-read limits remain + unchanged; even a small preview may decode a complete 256 MiB chunk. + +- Display saved PNG/JPEG notebook outputs beneath their cells in b2view, using + passive terminal-image rendering with shared image-count/byte/pixel limits. + Invalid images retain readable cells; HTML, SVG, scripts and widgets stay disabled. + +- Reduce b2view startup and remote-page flicker by sizing requests after layout, + retaining visible metadata/data during reloads, and reusing buffered rows for + height-only resizes instead of fetching them again. + +- Restore tagged non-finite floating-point schema values from strict-JSON + Caterva2 metadata, preserving NaN/infinity null sentinels and defaults. + +- Empty lazy Parquet table slices no longer read row data or scan dictionaries, + fixing slow Caterva2 schema fetches. Small slices/takes build dictionaries from + selected rows instead of scanning every row group; native table dictionaries + retain their existing copy behavior. + +- b2view now exits cleanly with a clear diagnostic and nonzero status when + its initial source cannot be opened, rather than leaving an unusable TUI. + +- b2view file fallback panels distinguish unavailable previews, missing optional + dependencies and read/display failures, with consistent capability-aware + action hints. Preview errors retain file metadata and redact transport URLs. + +- Open local and direct-fsspec ordinary files in b2view with the same bounded + text/image previews, PDF fallbacks and explicit copy/external-open actions. + Transfers are streamed, cancellable and atomic, with no overwrite by default; + existing container/dataset opening remains unchanged. + +- Open ordinary local/fsspec directories as lazy b2view tree roots. Preview + files and expand dataset containers in place; local symlinks are not followed. + +- Use Ctrl+F to filter discovered tree paths without network access, or explicitly + search local/Caterva2/fsspec hierarchies recursively with progress, cancellation + and discovery limits. Selecting a result reveals it in the existing tree. + +- Preview local compressed documents such as `README.md.b2` and `photo.png.b2` + using bounded, lazy SChunk decoding. Copies stream original bytes under the + original filename; oversized chunks and non-document container types are refused. + +- Preview nbformat-4 Jupyter notebooks passively in b2view: Markdown, highlighted + code and bounded saved plain-text outputs, with a raw-JSON toggle. No kernels + or active HTML/media rendering; supported across existing file backends. + +- Fix structured and subarray dtype decoding in `C2Array` and synthesized + RemoteStore chunks, including compound/subarray HDF5 leaves. + +- Ordinary Caterva2 files now open lazily as `RemoteFile`: bounded original-byte + reads, shared compressed-chunk caching, and atomic streaming downloads. +- b2view previews text/Markdown and optional JPEG/PNG images (`blosc2[images]`). + PDF and other unrenderable content remain downloadable; supported documents + can open externally through the explicit `O` action and destination dialog. No PDF dependency is + required. `D` downloads, `O` opens externally, and `T` toggles raw Markdown. + +- `blosc2.open()` recognizes HTTP(S) dataset URLs such as + `http://localhost:8000/@public/group`, including deployment prefixes and IPv6. + String service URLs default to lazy access; explicit `URLPath` behavior is unchanged. +- Bare server/base URLs discover `api/roots`. One root opens directly; empty or + multiple-root services return a lazy `RemoteRepository` browsing facade. +- `remote_service="auto" | "caterva2" | "fsspec"` controls service recognition + independently of source format. The fsspec override preserves ordinary URLs + with literal `@` path components and disables service probes. +- Caterva2 `RemoteStore` groups expand lazily, including catalog mount boundaries; + catalog annotations are separate from source attributes. +- `b2view` browses these server/root URLs with the existing interface, including + bounded array/table previews. Remote table-wide transforms are disabled; + plotting requires an explicit bounded row window. Client caches are separate + from a shared cat2lite server cache; repository allowances are per root. + ## Changes from 4.14.0 to 4.14.1 Python-Blosc2 4.14.1 is a security and feature release introducing safe diff --git a/doc/getting_started/installation.rst b/doc/getting_started/installation.rst index 8d51e063c..fd0e689de 100644 --- a/doc/getting_started/installation.rst +++ b/doc/getting_started/installation.rst @@ -33,10 +33,13 @@ grouped into *extras* that you opt into with the ``blosc2[extra]`` syntax: - The :doc:`b2view <../guides/b2view>` terminal browser (``textual``, ``textual-plotext``), including its in-terminal braille plot (the ``p`` key). Required by the ``b2view`` command. - * - ``hires`` + * - ``hires`` - The high-resolution image view in b2view (the ``h`` key), which renders a real ``matplotlib`` image in the terminal - (``textual-image``, ``matplotlib``). Includes ``tui``. + (``textual-image``, ``matplotlib``). Includes ``tui``. + * - ``images`` + - JPEG/PNG ordinary-file previews in b2view (``Pillow``, ``textual-image``). + Includes ``tui`` without requiring matplotlib. ``hires`` includes this extra. * - ``parquet`` - The ``parquet-to-blosc2`` converter (``pyarrow``); see :doc:`../guides/parquet_to_blosc2`. @@ -59,6 +62,7 @@ argument in shells like ``zsh`` that treat brackets specially): .. code-block:: console pip install "blosc2[tui]" # the b2view terminal browser + pip install "blosc2[images]" # b2view + JPEG/PNG file previews pip install "blosc2[hires]" # b2view + its high-res view (h key) pip install "blosc2[parquet]" # the Parquet converter pip install "blosc2[fsspec]" # fsspec URLs, including HTTP(S) diff --git a/doc/guides/b2view.rst b/doc/guides/b2view.rst index c980e3f7b..d40ac56db 100644 --- a/doc/guides/b2view.rst +++ b/doc/guides/b2view.rst @@ -63,6 +63,258 @@ You can also jump straight to a node and panel: Remote containers and arrays ~~~~~~~~~~~~~~~~~~~~~~~~~~~~ +Caterva2 and shared cat2lite repositories +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Open a Caterva2-compatible server, a published root, or a selected group/leaf: + +.. code-block:: console + + b2view http://localhost:8000 + b2view http://localhost:8000/@public + b2view http://localhost:8000/@public/hdf5 /d0/d1/a2 + b2view https://cat2.cloud/demo/@public/example + +A bare server URL probes ``api/roots``. A single visible root opens directly; +multiple roots appear as children of a repository group. An empty server shows +"No accessible roots". Deployment prefixes such as ``/demo`` are preserved. +For scripts and reproducible starting paths, prefer an explicit root URL. + +Group expansion discovers mounted hierarchies on demand. Arrays and tables use +bounded previews/pages through the server API; the viewer does not download the +whole source container. Catalog annotations appear separately as +``catalog_attrs`` in metadata, without replacing source attrs. + +Use ``--remote-service fsspec`` when an ordinary HTTP data URL happens to contain +an ``@``-prefixed path component. Use ``--remote-service caterva2`` to require +service discovery rather than falling back to a file opener. + +``--cache-dir`` and ``--max-cache-bytes`` configure the **client** cache, not the +server's shared cache. Multi-root repository budgets are per opened root. +The cat2lite administrator configures shared upstream caching independently. +Sources are assumed immutable; updates require deliberate cache invalidation. +Overlapping reads across clients benefit most from a shared server cache. + +Remote tables support previews and projection, but filtering, sorting, and +grouping are disabled rather than implicitly downloading the whole table. +Table plotting requires a bounded locked row window (``v``). Source formats +behind cat2lite need their dependencies on the server, not on each viewer client. + +Ordinary Caterva2 files +^^^^^^^^^^^^^^^^^^^^^^ + +Published ordinary files (compressed SChunk byte streams) appear as file nodes. +Metadata inspection does not fetch their payload. Text and Markdown preview a +UTF-8 prefix, bounded to 64 KiB and 1,000 lines; truncated/invalid text is labelled. +Markdown links/images are passive and never fetch other resources. ``T`` toggles +raw text and Markdown; terminal controls are stripped from file content. + +Install ``blosc2[images]`` for JPEG/PNG previews using Pillow and textual-image, +without requiring matplotlib. Terminal protocol support may fall back to colored +half-cells. Missing dependencies/display support show a download/open notice. +Automatic image input is limited to 16 MiB; original pixel count is limited to +16,777,216 pixels (64 MiB RGBA). Orientation is corrected, the first frame is used, +and the displayed image is reduced to at most 1600 × 1200 pixels. These limits +bound individual buffers, not total process RAM including decoder copies. + +``D`` prompts for a destination and streams the **original file**, not its Blosc +carrier. ``O`` downloads and opens the completed local file in the platform's +external viewer after you submit the destination dialog, without an extra trust +checkbox. Both actions work for +PDFs, which deliberately need no terminal PDF renderer. External opening is +restricted to PDF, JPEG/PNG, Markdown, and text; other binary files remain +downloadable. No shell or credential-bearing URL is passed to the launcher. +Downloads are user-owned: they remain after closing the dialog or b2view, including +when external opening fails. Existing destinations are never overwritten by the +viewer; choose another name. Escape cancels an active transfer. + +File reads/downloads have per-chunk limits of 32 MiB compressed and 256 MiB decoded. +Animated ``loading`` dots in the data-pane border indicate pending background +loads without hiding the current page or changing the layout. The indicator +disappears when the load completes or fails; it is activity, not byte progress. +Even a small preview can decode one complete chunk; temporary buffers can bring +peak memory above the decoded-chunk limit. Preview and per-call read limits +remain independent of these chunk limits. +A small preview may require a much larger chunk. Oversized or irregular sources +require server-side rechunking; the viewer never silently fetches a whole file +to work around these limits. File reference archives are not supported by this +feature. Incompatible files do not hide healthy +siblings. + +Local and direct-fsspec ordinary files +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Open a regular file directly with the same previews, fallback panels and actions: + +.. code-block:: console + + b2view README.md + b2view photo.png + b2view brochure.pdf + b2view https://example.org/README.md + b2view --profile blosc2 s3://my-bucket/notes/README.md + +Local ``file://`` URLs and fsspec protocol chains are supported too. Install +``blosc2[fsspec]`` and the relevant filesystem dependency for remote files. +Authentication/storage options use the same CLI flags as direct dataset URLs. +For an HTTP file without a filename extension, or a file URL containing an +``@`` path component, pass ``--remote-service fsspec`` to bypass service discovery. + +Direct files open without a tree panel. Reads and image decoding run in the +background; preview text/image limits are unchanged. ``D`` copies original bytes +to a chosen local destination; ``O`` copies and opens the supported document +after explicit destination submission. Local sources are never modified, and +existing destinations are not overwritten. Choose a different destination when +the dialog defaults to the source's own name. Transfers stream in 1 MiB spans +and support cancellation and atomic publication, rather than loading the whole +file into memory. Ordinary files must have a known size; a short read or a size +change during copying fails without publishing a partial destination. + +These viewer-only handles do not extend ``blosc2.open`` or Caterva2's caches. +``--cache-dir``/``--max-cache-bytes`` do not configure direct ordinary-file caching; +filesystem backends may have their own buffering (notably chained archives). +HTTP operations have a default 10-second timeout. Existing Blosc2/Zarr/HDF5 +hierarchies keep their dataset semantics. +Recognized dataset extensions retain their normal opener and corruption errors; +unrecognized filenames receive a small native-frame signature check, not object +deserialization or automatic content execution. + +Ordinary directories +^^^^^^^^^^^^^^^^^^^^ + +Local directories (including ``file://`` URLs) and directories on fsspec +filesystems with directory-listing support open as lazy tree roots: + +.. code-block:: console + + b2view ./documents/ + b2view ../cat2lite/tests/fixtures/data-cat2-demo/root-example/ + b2view --profile blosc2 s3://my-bucket/documents/ + +Only immediate children are listed; expanding a directory discovers the next +level. Selecting a file uses the same previews and copy/open actions as a direct +file input. Dataset containers such as B2Z, HDF5 and Zarr can be expanded as +mounted subtrees without first downloading the whole container. A corrupt leaf +does not hide its siblings. Refresh reopens the directory and its selected node. +Local symlinks are shown but not followed, and parent-path navigation cannot +escape the opened root. Opening a dataset directory such as ``.b2d`` or ``.zarr`` +directly continues to use the dataset opener, not the ordinary-directory browser. +Client cache directories are separated per mounted dataset; cache-byte budgets +apply per mounted source, not as a global limit for the whole directory tree. + +Compressed local and fsspec documents +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Document carriers such as ``README.md.b2``, ``photo.png.b2`` and +``brochure.pdf.b2`` use the same file previews/actions as their originals: + +.. code-block:: console + + b2view README.md.b2 + b2view ./documents/ /README.md.b2 + b2view s3://bucket/documents/README.md.b2 + b2view https://example.org/documents/analysis.ipynb.b2 + +Supported text, JPEG/PNG and PDF suffixes followed by ``.b2`` identify this +viewer-only convention. A carrier must be a plain fixed-chunk SChunk byte stream, +not an NDArray, serialized object or remote reference. Local carriers are mapped +read-only; fsspec carriers use byte ranges on contiguous frames. Metadata +inspection does not decompress document chunks. Plain ``data.b2`` and +``.b2frame`` datasets keep their normal native behavior. + +Remote reads fetch the frame header, then a bounded chunk index when values are +first requested, and each needed chunk's header before its payload. HTTP servers +must return valid ``206`` byte-range responses; ignored ranges, incorrect +lengths and encoded responses are refused without reading the complete body. +Frame headers are capped at 1 MiB and chunk indexes at 8 MiB compressed/decoded. +Sparse frames and nonzero special-value chunks are not supported remotely. +Filesystem storage options (including authentication) apply to previews and +independent transfer handles. Other fsspec backends may have their own buffering. + +Text previews decode only chunks covering the first 64 KiB, displaying at most +1,000 lines. Images decode complete original bytes only within the existing +16 MiB input and pixel limits. PDF selection needs no decompression. Chunk +headers are checked before copying/decompressing payload: each chunk must fit +32 MiB compressed and 256 MiB decoded. Oversized/irregular carriers are refused; +rechunk them rather than increasing automatic preview memory usage. + +``D`` streams decoded original bytes to the destination, keeping at most one +decoded chunk between reads. ``O`` uses the same explicit copy/external-open +workflow. Both suggest the original name (``README.md``, not ``README.md.b2``); +carrier metadata retains the on-disk filename. No entire document is assembled +for downloading, and no original-file payload is deserialized or executed. +Sources are assumed immutable while open; refresh after replacing a carrier. + +Passive Jupyter notebook previews +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +``.ipynb`` files open as passive nbformat-4 documents, not executable notebooks: + +.. code-block:: console + + b2view analysis.ipynb + b2view analysis.ipynb.b2 + b2view https://cat2.cloud/demo/@public/examples/slice-time.ipynb + +Markdown cells render without active links/images; code cells are syntax +highlighted using the declared supported language (otherwise plain text). +Raw cells, saved stdout/stderr, plain-text result representations and tracebacks +are shown as text with terminal controls removed. No kernel starts, cells never +execute. Saved PNG/JPEG outputs are decoded from the notebook and displayed +beneath their cells using the existing terminal-image support +(``blosc2[images]``). HTML/JavaScript, SVG, widget outputs and other media remain +omitted with a notice. There is no requirement for Jupyter or nbformat to be +installed. Markdown image links/attachments are not fetched. + +At most eight saved images are displayed. They share a 16 MiB encoded-image-byte +budget and a 16,777,216 decoded-pixel budget (64 MiB at RGBA), in addition to the +2 MiB notebook input cap below. Images are validated before pixel decoding and +reduced to at most 1600 by 1200 pixels for display. Invalid or oversized images +are skipped with a notice, keeping cell text available. Only saved static +images are shown; interactive plots/widgets are not executed, and animated +images show only their first frame. + +Parsing requires the complete JSON document, so input is capped at 2 MiB. +Displayed source/output text shares a 64 KiB / 1,000-line budget, with at most +100 cells. Truncation is labelled; oversized or unsupported notebooks show the +normal fallback panel rather than being fetched without bounds. ``T`` switches +between the cell view and a bounded raw-JSON prefix; ``D`` copies the original +notebook. Notebook external opening is not enabled. These previews work with +local, fsspec and Caterva2 files, and local/fsspec ``.ipynb.b2`` document carriers, under +their existing transport/chunk limits. + +Find nodes in local or remote trees +^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + +Press ``f`` with the tree focused, or ``ctrl+f`` from any panel, to open tree +search. The tree frame includes a compact key hint. Typing filters **already discovered** paths +by case-insensitive substring, without network requests. Matching nodes retain +their ancestor context. An empty filter shows discovered nodes again; Escape +closes the dialog without changing the main tree's selection or expansion state. +The dialog displays at most 500 matches at once; refine the filter for more. + +Choose **Search recursively** to discover unopened directories/groups in the +background. This works with local trees, Caterva2 repositories and fsspec +hierarchies using their existing listing APIs. It requests listings and necessary +metadata, not file previews, array values or table rows. For example, Caterva2 +table classification can require an empty frame containing its schema. + +Progress, listing errors and incomplete/cancelled status are shown explicitly. +Search stops after 200 directories, 10,000 discovered nodes or depth 12; the final +indivisible backend listing can exceed the node budget. **Cancel search** retains +completed results; Escape cancels and closes. An in-flight backend operation may +finish before cancellation takes effect, but the dialog can close immediately. +Discovered listings are reused in subsequent searches; refresh the tree to +invalidate them. No recursive discovery happens merely by typing a filter. + +Press Enter in the input to focus the result tree, then select a matching node +with Enter (or the mouse) to reveal its ancestors in the main tree and open its +normal preview. Row/column filtering shortcuts remain unchanged. Standalone +files and arrays have no tree to search. + +Direct source URLs +^^^^^^^^^^^^^^^^^^ + Browse remote B2Z, Zarr, and HDF5 containers directly from their root: .. code-block:: console diff --git a/doc/guides/remote_arrays.md b/doc/guides/remote_arrays.md index 1af2e7d80..39e23f055 100644 --- a/doc/guides/remote_arrays.md +++ b/doc/guides/remote_arrays.md @@ -216,6 +216,10 @@ Shared caching selects lazy access for every remote source when `lazy` is omitted or `None`; explicit `lazy=False` is rejected. Authentication from `URLPath` or `c2context` remains in process memory. Authenticated users must use separate cache directories. +Automatic service discovery inherits credentials only when the destination has +the same scheme, host, and port as the configured `c2context(urlbase=...)`. +Other origins are probed and browsed anonymously; explicitly select a Caterva2 +service or provide an authenticated `URLPath` to authorize another destination. All handles using a shared cache must enable sharing. For advanced attachment with seed carriers or authorized sources, `RemoteArray.with_sparse_cache()` remains available. diff --git a/doc/guides/remote_objects.md b/doc/guides/remote_objects.md index 3f799142a..657265d16 100644 --- a/doc/guides/remote_objects.md +++ b/doc/guides/remote_objects.md @@ -8,12 +8,20 @@ one API. Data is read on demand and cached locally; remote sources are read-only | Standalone `.b2nd`, or a B2Z/Zarr/HDF5 array | {ref}`RemoteArray` | | Parquet file, B2Z CTable, PyTables table, or Caterva2 table | {ref}`RemoteCTable` | | B2Z, Zarr, HDF5, or Caterva2 group | {ref}`RemoteStore` | +| Caterva2 ordinary-file / fixed-chunk SChunk byte stream | {ref}`RemoteFile` | -All three inherit {ref}`RemoteObject` and expose source metadata, cache controls, +These types inherit {ref}`RemoteObject` and expose source metadata, cache controls, traffic counters, reference saving, and context-manager support. See {doc}`remote_arrays` for array computations and {doc}`remote_tables` for table queries and format-specific behavior. +For ordinary Caterva2 files, `file.read_bytes(start, stop)` reads original byte +ranges and `file.download(destination)` streams original content. Download is +atomic and refuses overwrite by default. File reference persistence is currently +unsupported; use download rather than `save`. Fixed-chunk byte streams share +cache budgets and owner lifetime with array/table leaves, but oversized chunks +and irregular layouts are refused. See {ref}`RemoteFile` for bounds and callbacks. + ```python import blosc2 @@ -33,6 +41,57 @@ call. See {doc}`remote_arrays` for local-source caching and selector details. ## Explore a hierarchy +### Caterva2 servers and shared caching gateways + +```python +with blosc2.open("http://localhost:8000") as repository: + print(repository.keys()) + +with blosc2.open("http://localhost:8000/@public/hdf5") as group: + print(group.keys()) + with group["d0/d1/a2"] as array: + values = array[:10, :10] +``` + +An HTTP(S) path component starting with `@` identifies a Caterva2 root. The +preceding path stays in the deployment base, e.g. `https://host/demo/@public`. +These string service references default to lazy access and return the selected +group, array, or table type. Explicit `URLPath` inputs retain their existing +lazy defaults. Dataset selectors are in the service URL, not `path=` or `::`. + +An ambiguous bare base URL probes `/api/roots` with a short timeout. A +single visible root opens directly. Zero or multiple roots return +{ref}`RemoteRepository`, with root names as children and independent, lazily +opened root owners. Prefer explicit root URLs for scripts: relative paths on +a bare service can change if its root count changes. Repository persistence and +materialization are unsupported; select a specific root for those operations. + +Use `remote_service="fsspec"` to disable recognition/probing for ordinary data +URLs containing `@` components, or for extensionless file sources. Use +`remote_service="caterva2"` to require a service without file fallback. Known +direct-format URLs are not probed in auto mode. Malformed/non-service responses +allow file fallback; authentication, connection, and server errors remain errors. +Discovery redirects are limited to the same origin. Authentication uses existing +{func}`blosc2.c2context`/`URLPath` facilities and is bound to opened owners. + +Caterva2 groups list immediate children on demand. A catalog mount not expanded +yet is not treated as empty. Direct descendant lookup also works without parent +expansion. `RemoteNode.catalog_attrs` contains catalog annotations separately +from source `attrs`. Existing servers may return recursive listings; their +response cost cannot be eliminated without a server-side pagination extension. +Caterva2 discovery retains at most 10,000 nodes per owner by default, rejects +list responses above 100,000 entries, and rejects discovery bodies above 8 MiB +before JSON decoding. Existing transports buffer those discovery bodies before +the size check. Explicit hierarchy summaries (`store.info`) visit groups and can +contact mounted sources; failures are reported as incomplete listings. + +A cat2lite gateway can share cached upstream chunks between independent clients. +Python-Blosc2's `cache_dir`, `cache_policy`, and `max_cache_bytes` still configure +the **client** cache. They do not configure the gateway cache, and `shared_cache` +does not mean “use the server cache.” Repository allowances apply per opened +root; retained payload totals can therefore exceed a single root's allowance. +Immutable-source assumptions and explicit invalidation requirements still apply. + - `store.keys()` or iteration lists immediate children. - `store["group/array"]` and `store["group"]["array"]` select the same leaf. - `store.attrs` exposes group attributes. diff --git a/doc/reference/classes.rst b/doc/reference/classes.rst index b4adc0b8b..f783c4919 100644 --- a/doc/reference/classes.rst +++ b/doc/reference/classes.rst @@ -156,6 +156,8 @@ container APIs above. remoteobject remotearray remotestore + remoterepository + remotefile remotectable proxysource proxyndsource diff --git a/doc/reference/remotefile.rst b/doc/reference/remotefile.rst new file mode 100644 index 000000000..05f51f1b2 --- /dev/null +++ b/doc/reference/remotefile.rst @@ -0,0 +1,45 @@ +.. _RemoteFile: + +RemoteFile +========== + +``RemoteFile`` is a read-only Caterva2 fixed-chunk SChunk byte stream. It is +returned by lazy service leaf opening or ``RemoteStore`` file lookup. Discovery +and file metadata do not fetch payload. Ordinary files such as Markdown, JPEG, +and PDF published as compressed byte streams are supported; there is no Python +object deserialization or implicit interpretation as an array. + +.. code-block:: python + + import blosc2 + + with blosc2.open("https://cat2.cloud/demo/@public/examples/README.md") as file: + prefix = file.read_bytes(0, min(file.nbytes, 4096)) + file.download("README.md") + +``read_bytes(start, stop)`` returns original bytes, with at most 16 MiB per call. +The omitted stop means the file end, not an unbounded streaming read. Use +``download`` for larger files. A chunk can contain more data than the requested +range: transfers are chunk-granular, bounded at 32 MiB compressed and 256 MiB +decoded per chunk. Oversized/irregular streams require rechunking on the server. +Generic fixed-chunk typed SChunks expose their raw byte representation; they do +not imply a text/image/document format. + +Downloads stage beside the destination, then publish completed original bytes. +Existing destinations are not overwritten unless ``overwrite=True`` is explicit. +No-overwrite publication uses a hard link to prevent concurrent-creation races; +the destination filesystem must support hard links. ``progress(done, total)`` +and ``cancel()`` callbacks run on the calling thread. Cancellation or failure +removes only the operation's staging file and preserves any previous destination. + +Cache policy/allowance, traffic, and lifetime are shared with the source owner. +``cache_bytes`` includes other leaves in that owner. MEMORY/DISK retain compressed +chunks; NONE retains no payload between calls. Transient decoded buffers are +separately bounded, not included in retained cache bytes. Sources must remain +immutable until explicit refresh; returned children outlive their parent handle. +``save`` of a file reference is intentionally unsupported; ``download`` is an +original-byte export, not a reference archive. Explicit nonlazy ``URLPath`` input +still requires ``lazy=True`` for byte-stream access. + +.. autoclass:: blosc2.RemoteFile + :members: read_bytes, download, close, source, name, media_type, nbytes, cbytes, nchunks, chunksize, attrs, traffic, info, cache_policy, max_cache_bytes, cache_bytes, metadata_bytes diff --git a/doc/reference/remoterepository.rst b/doc/reference/remoterepository.rst new file mode 100644 index 000000000..5c1cb1b29 --- /dev/null +++ b/doc/reference/remoterepository.rst @@ -0,0 +1,29 @@ +.. _RemoteRepository: + +RemoteRepository +================ + +``RemoteRepository`` is a read-only browsing facade returned by +``blosc2.open("https://host/base")`` when a Caterva2-compatible service has zero +or multiple accessible roots. A single root instead opens directly. + +The facade subclasses :ref:`RemoteStore` for hierarchy consumers such as b2view. +``keys()`` lists roots without opening them; ``repository["@public"]`` returns +an independent group handle. Direct descendant lookup is supported. Closing +the repository leaves previously returned child handles usable. Alias handles +from ``repository[""]`` share root owners and lifetime accounting. + +Cache policy and allowance are per root. ``cache_bytes`` and ``metadata_bytes`` +sum opened roots; ``traffic`` counts their reads and excludes the initial roots +probe. Source descriptors contain no authentication token. Root owners bind the +authentication context at repository creation and use existing isolated cache +identities. + +Repository persistence, materialization, refresh, and root-level shared sparse +cache opening are intentionally unsupported. Select a specific root for those +operations. ``get_info`` at a root name reports the roots registry metadata; +select/open that root for its actual source attributes. No synthetic empty path +is sent to the server's info endpoint. + +.. autoclass:: blosc2.RemoteRepository + :members: keys, get_info, close, source, attrs, cache_policy, max_cache_bytes, cache_bytes, metadata_bytes, traffic diff --git a/doc/reference/remotestore.rst b/doc/reference/remotestore.rst index b5c7884f9..2a4ed1890 100644 --- a/doc/reference/remotestore.rst +++ b/doc/reference/remotestore.rst @@ -8,6 +8,13 @@ RemoteStore source session: a B2Z archive, a native HDF5 index, or a Zarr store. Zarr listing remains lazy. +Caterva2 groups also list lazily, through explicit ``URLPath`` inputs or service +URL strings passed to ``blosc2.open``. Each group is expanded independently, +including virtual catalog mount boundaries. ``RemoteNode.catalog_attrs`` exposes +catalog annotations separately from source attrs. See +:doc:`../guides/remote_objects` for URL discovery and :ref:`RemoteRepository` +for browsing services with multiple roots. + The default ``CachePolicy.MEMORY`` shares a 256 MiB allowance across all leaves. Set ``max_cache_bytes`` to a positive integer to change it. ``CachePolicy.NONE`` retains no payload and rejects a limit. Passing ``cache_dir`` selects DISK when diff --git a/plans/b2view-tree-search.md b/plans/b2view-tree-search.md new file mode 100644 index 000000000..16641b1e5 --- /dev/null +++ b/plans/b2view-tree-search.md @@ -0,0 +1,22 @@ +# b2view tree search + +Implemented following user authorization. F with tree focus (or global Ctrl+F) +opens a filtered mirror of +discovered nodes; case-insensitive path filtering performs no I/O. Ancestors +remain visible, and closing restores the unchanged main tree selection/expansion. + +Explicit recursive search traverses existing StoreBrowser listings breadth-first +on a worker, reusing known listings. Local, Caterva2 and fsspec roots use the same +implementation. It does not call preview/data access: necessary discovery +metadata (including empty Caterva2 table schema frames) is allowed. + +Limits: 200 directories, 10,000 nodes, depth 12; the final atomic listing may +overshoot the node count. Only 500 matches are rendered at once. Progress is +throttled to keep result-tree rebuilding responsive. Cancellation and session +guards reject obsolete callbacks/results; in-flight provider calls cannot be +forcibly interrupted. Errors and partial-search status are reported explicitly. + +Selecting a result merges discovered listings and reveals the node using normal +navigation. Closing retains delivered discoveries without expanding the main +tree; refresh invalidates extra discovery snapshots. No backend-specific global +catalog optimization or filename-content/full-text search is introduced. diff --git a/plans/caterva2-access-improvements.md b/plans/caterva2-access-improvements.md new file mode 100644 index 000000000..60fc91517 --- /dev/null +++ b/plans/caterva2-access-improvements.md @@ -0,0 +1,464 @@ +# Caterva2 access improvements for Python-Blosc2 and b2view + +Status: M0–M5 implemented and validated on `cat2-improvements`; final review +fixes included. The sections below retain the delivery design and acceptance +criteria; the implementation record is the authoritative completion evidence. + +Implementation record (updated as milestones land): + +- M0/M1: selected `remote_service="auto" | "caterva2" | "fsspec"` and + implemented root-marker recognition before format dispatch. String service + URLs default to lazy access; explicit URLPath defaults remain unchanged. + Inspection confirms `open` currently defaults to `mode="r"` (the proposal's + concern about changing that default requires no code change). +- Repository design: use a public RemoteRepository subclass of RemoteStore as + a browsing facade with independently owned roots and per-root cache budgets. + Repository persistence is explicitly unsupported initially. Single-root + servers flatten to their actual root object; empty/multi-root servers return + the repository facade. Root owners stay alive through normal returned child + handles after the facade closes. +- Roots discovery uses the mapping response implemented by cat2lite and + Caterva2, not an assumed list. No global discovery cache is introduced. +- M1 validation: 20 offline URL/dispatch tests passed in the blosc2 environment. +- M2: Caterva2 groups now discover immediate children on demand, support direct + descendant lookup, memoize completed listings under the owner lock, and roll + back node-limit failures. Opening a string root uses the opener's metadata + seed to avoid duplicate info calls. Catalog annotations are exposed separately + on RemoteNode and retained in discovery metadata. Focused URL/Caterva2 tests: + 37 passed, including existing reference/table/cache regressions. +- M3: added bounded roots discovery (3-second deadline, 1 MiB response bound, + same-origin redirects only), direct-source bypass, strict error/fallback + classification, and the public RemoteRepository browsing facade. Roots open + independently and lazily; facade aliases share owners, returned children + outlive the facade, and authentication is frozen at opening. Per-root cache + budgets and unsupported repository persistence are explicit. Validation: + 38 access tests passed, including zero/one/multiple roots and error handling. +- M4: b2view uses shared service recognition ahead of direct-format handling, + supports the backend override, displays catalog annotations and empty-root + notices, and keeps mounted-source metadata failures isolated. Remote table + metadata/paging work without table-wide downloads; transforms are disabled + and plotting requires an explicit bounded row window. Fixed metadata display + for remote sources that do not report compressed size. All viewer/model + regressions, including headless Textual and fresh-process decoders: 155 passed. +- M5: added opt-in `tests/test_caterva2_gateway.py` against a real debug + cat2lite server and deterministic loopback B2ND/HDF5/Zarr/Parquet sources, + including headless b2view and independent no-client-cache subprocesses. + Measured large-array upstream reads: 72,950 bytes / 3 GETs after warming; + second client unchanged; after server restart 81,142 bytes / 4 GETs (8 KiB + source headers, no warm payload refetch). API/viewer docs and release notes + updated. Full default offline suite: 10,640 passed, 36 skipped. Gateway + acceptance passed with CAT2LITE_SERVER set; it is skipped by default. +- Final review: old eager-list cache snapshots rediscover listings without + discarding retained payload; direct paths reject query/fragment/escape + ambiguity; missing nodes raise KeyError; nonlazy groups/tables fail clearly. + Added default node/list/discovery-response limits, avoided full-registry copies + per node inspection, and isolated failed sources in hierarchy summaries. + Failed viewer nodes remain expandable for retry. Row-window reads are prepared + in background workers and published only if selection/session still match; + Caterva2 windows materialize explicitly bounded selections for local plotting. +- Final validation: 10,646 default-suite tests passed, 37 skipped (including + the opt-in gateway test); 213 focused access/Caterva2/viewer/model tests passed + with headless TUI cases enabled; real cat2lite gateway acceptance passed again + with the same cache traffic measurements. Ruff lint/format and diff checks + passed. HTML docs built in external temporary storage with notebook execution + disabled, generated/stashed doc copies excluded, and type-comment introspection + disabled; the wider documentation build still emits many autosummary/theme/ + cross-reference warnings (754), so this is not a warnings-clean documentation + build. No Python native-extension rebuild was required; cat2lite's three debug + binaries were rebuilt for cross-project acceptance. + +## 1. Goal + +Make Caterva2-compatible servers, including cat2lite, first-class URL-addressable +sources in Python-Blosc2. Users should be able to browse a curated remote +repository through a shared caching gateway with the same viewer and opening +API used for individual files. + +Target usage: + +```python +import blosc2 + +repository = blosc2.open("http://localhost:8000", mode="r") +public = blosc2.open("http://localhost:8000/@public", mode="r") +group = blosc2.open("http://localhost:8000/@public/hdf5", mode="r") +array = blosc2.open("http://localhost:8000/@public/hdf5/d0/d1/a2", mode="r") +``` + +```sh +b2view http://localhost:8000 +b2view http://localhost:8000/@public +b2view https://cat2.cloud/demo/@public/example +``` + +The examples above describe implemented behavior. Preserve existing direct-file +opening, including HTTP B2ND/B2Z, HDF5, Zarr, and Parquet sources. + +Architecture: + +```text +b2view / Python applications + | + | Caterva2 metadata, listings, bounded slices + v +cat2lite + shared server-side disk cache + | + v +remote repositories exposed through .catl mounts +``` + +Do not create a separate cat2lite-view application. Source recognition belongs +in Python-Blosc2; the existing b2view UI should consume the resulting objects. + +## 2. Current implementation and gaps + +Relevant code to inspect before implementation: + +- `src/blosc2/schunk.py`: `open`, `_open_c2_urlpath`, remote option validation, + lazy dispatch, and existing format-specific openers. +- `src/blosc2/c2array.py`: `URLPath` integration, API URL construction, shared + HTTP clients, authentication, info/list/fetch helpers. +- `src/blosc2/remote_store.py`: Caterva2 source descriptors, `_open_caterva2`, + `resolve`, `kind`, `list_children`, owner/cache lifecycle, and table fetching. +- `src/blosc2/b2view/model.py`: `StoreBrowser._open_store`, tree traversal, + remote leaf ownership, previews, and table capabilities. +- `src/blosc2/b2view/app.py`: background opening, tree expansion, failure display. +- `src/blosc2/b2view/cli.py`: source arguments, cache options, initial path. + +Existing foundations: + +- Explicit `blosc2.URLPath(path, urlbase=...)` already selects Caterva2 handling. +- The lazy Caterva2 opener discovers groups, arrays, and CTables; groups can + return `RemoteStore`, arrays use existing C2Array/RemoteArray semantics, and + tables can return `RemoteCTable`. +- b2view already displays RemoteStore hierarchies and remote leaves. +- cat2lite already implements roots, info, list, and bounded data access. + +Gaps: + +1. A plain HTTP string is not converted into a Caterva2 URLPath. +2. Bare server URLs are treated as data files, without roots discovery. +3. `_open_caterva2` currently assumes an initial recursive listing describes the + hierarchy and initializes discovered groups' child lists from that snapshot. + cat2lite catalog listings deliberately stop at mounted-source boundaries. + Thus a discovered mounted group can appear empty instead of being expanded + through another API list request. +4. b2view has format-specific URL dispatch that must not override the new + Caterva2 interpretation when a published dataset name ends in `.h5`, etc. +5. A server with multiple roots has no agreed repository-level browsing object. + +## 3. URL and opening contract + +### 3.1 Explicit dataset/group shorthand + +Recognize an HTTP(S) URL containing an `@`-prefixed path component as Caterva2 +shorthand, consistently with cat2lite's `.catl` convention. + +Example: + +```text +https://host/demo/@public/group/array + urlbase = https://host/demo + path = @public/group/array +``` + +Rules: + +- Parse with a real URL parser, preserving bracketed IPv6 and port numbers. +- Inspect path components only; ignore `@` in hostname, userinfo, or query. +- Use the first root-marker component; later `@` components are dataset names. +- Preserve the encoded deployment prefix; decode logical dataset components + exactly once and use existing Caterva2 path validation/endpoint encoding. +- Reject empty root markers, traversal, encoded separators, malformed encoding, + control characters, and unsupported query/fragment/selector combinations. +- Do not reinterpret fsspec `::/member` selectors as Caterva2 dataset selectors. +- Authentication uses existing Caterva2 facilities, not URL userinfo. +- Explicit URLPath inputs retain precedence and behavior. +- Return the object's actual type; do not wrap every dataset in RemoteStore. +- Preserve existing `mode`, `offset`, `lazy`, and cache-option validation. + Remote sources remain read-only. Specify `mode="r"` in documentation rather + than changing the global default mode as part of this work. +- `lazy=False` must retain its existing meaning or explicitly reject group/ + repository opening; do not silently ignore it to force browser semantics. + +### 3.2 Ordinary HTTP override + +An ordinary file server can legitimately contain `@` path components. Provide +an explicit bypass for Caterva2 recognition and repository probing. + +Selected API: the narrowly scoped selector +`remote_service="auto" | "caterva2" | "fsspec"` on `blosc2.open`. +This selects the service backend independently of data format; do not +overload `source_format`, which currently describes data formats. + +- `auto`: root-marker shorthand, known-file dispatch, then eligible base probe. +- `caterva2`: explicitly interpret a URL as a repository base or dataset URL. +- `fsspec`: use the ordinary fsspec-backed source handling, even with `@` + components, bypassing Caterva2 recognition and probing. This identifies the + access backend, not a local file or a single-file-only source; remote + containers and stores remain supported. +- Explicit URLPath plus a conflicting override is an error. +- Reject this option for inputs where it has no meaning rather than ignoring it. +- If needed, expose the same choice in b2view through a small CLI flag. + +### 3.3 Bare server and deployment-base URLs + +For an eligible ambiguous HTTP(S) URL, probe `/api/roots` with a bounded +timeout and validate the response against the actual Caterva2 roots contract. +Inspect both Caterva2 and cat2lite responses during M0; do not assume a list +of root names if the API returns a mapping with metadata. + +Recommended dispatch order: + +1. Explicit URLPath or explicit service override. +2. Root-marker URL shorthand. +3. Known file/container URL, including selectors and trailing-slash Zarr. +4. Discovery for ambiguous base candidates, including `/demo` prefixes. +5. Existing generic remote-file path if discovery conclusively says this is + not a Caterva2-compatible server. + +Discovery must not add roots requests to recognized direct-file reads or turn +an existing permission/format error into an unrelated service-detection error. + +Probe handling: + +- Valid roots response: bind the server base and authentication context. +- Missing endpoint, HTML, malformed JSON, or invalid schema: classify as + non-Caterva2 in auto mode; provide a clear discovery error in explicit mode. +- 401/403: report authentication/authorization failure; no anonymous fallback. +- Connection/TLS failure or timeout: report an actionable connection error, + preserving the original cause; avoid serial retry/fallback delays. +- Server errors: preserve the server error rather than saying “not Caterva2”. +- Preserve deployment prefixes and follow existing redirect policy. Never + forward authentication credentials to an unrelated redirect origin. +- Close response resources, reuse the existing transport, and avoid duplicate + roots/info requests when passing discovery results into an owner. +- Do not introduce a process-global negative discovery cache initially. + Any later positive cache must include authentication identity and lifetime. + +## 4. Repository root behavior + +Recommended user-facing behavior: + +- Exactly one root: open it directly, so cat2lite's `@public` server feels like + a directly opened hierarchy. +- Multiple roots: expose a synthetic repository group with roots as children. +- Zero visible roots: return an empty repository group, with b2view showing + “No accessible roots”; this is not necessarily a server error. +- An explicit `/@public` URL always opens that root, regardless of other roots. + +Implementation decision for M0: prefer a repository-backed RemoteStore mode +over a viewer-only wrapper. Confirm it fits ownership and persistence semantics +before choosing a new public class. All applications should get the same +`keys`, lookup, `kind`, attrs, context-manager, and close behavior. + +The synthetic group: + +- Is not sent to `/api/info` as a fabricated empty dataset path. +- Discovers each root only when accessed; it does not open all roots at startup. +- Isolates inaccessible/broken roots without hiding healthy siblings. +- Has its own identity distinct from a dataset-root descriptor. +- Keeps independent root owners under a repository lifetime; closing the + repository must follow existing live-child ownership conventions. +- Makes cache allowance scope explicit. Prefer a repository-wide bound where + the owner architecture supports it; otherwise document per-root allowances + before shipping rather than implying a total bound. +- Either defines safe persistence/reopening explicitly or rejects repository + persistence clearly in the first implementation. In-memory browsing must + not accidentally write an invalid existing dataset descriptor. + +Single-root flattening means relative paths can change if server roots change +between opens. Document this and recommend explicit root URLs for scripts. + +## 5. Lazy Caterva2 hierarchy discovery + +Refactor Caterva2 discovery into separate root metadata, node metadata, and +group-listing operations. Track whether a group has actually been listed; +“known group with no discovered children” is not “known empty group”. + +Required behavior: + +1. Open a known root with bounded metadata work, without recursive remote + source expansion. +2. Expand a group through `/api/list/` only when needed. +3. Accept existing recursive relative-path lists and catalog lists that stop + at mounts. Synthesize structural parent groups where necessary. +4. Mark only the queried group's listing as complete. A returned group may + need its own request, even if no descendants were returned initially. +5. Fetch and memoize metadata needed to classify actual children. Do not + assume every listed name is an array or eagerly inspect every deep leaf. +6. Resolve a directly requested descendant through info requests even when its + parents have not been expanded. Update the same node registry afterward. +7. Keep listings deterministic and deduplicated; empty groups remain visible. +8. Bound node growth and discovery work; reject malformed or escaping paths. +9. Coalesce concurrent discovery for the same group and publish registry + updates atomically. Failed discovery must remain retryable. +10. Preserve attrs and separate `catalog_attrs` annotations without overwriting + source attrs. Decide how annotations are displayed in b2view metadata. + +There is a protocol constraint: an existing server may return a large recursive +list in a single response. This client change cannot promise paginated or +constant-size listings without a server API extension. Avoid eagerly fetching +info for every returned descendant; measure remaining list costs separately. + +## 6. Cache and data-access semantics + +- A cat2lite server cache is shared across its HTTP clients. Python-Blosc2's + client cache is a separate layer, with independently configured policies. +- Reuse existing none/memory/disk options; do not invent a special server-cache + policy or reinterpret `shared_cache=True` as “use cat2lite”. That existing + option has its own local-cache semantics and locking requirements. +- Normalize shorthand strings and equivalent URLPath sources to the same + dataset cache identity, including base prefix, path, and authentication scope. +- Repository synthetic identities must not collide with dataset identities. +- Do not persist credentials or allow a cached authenticated view to leak into + a different identity. Follow the current authenticated-cache restrictions. +- Preserve exclusive disk-owner locking, bounded retention, restart behavior, + and reference-counted leaf lifetimes. +- Browsing uses metadata only; previews fetch bounded slices/pages rather than + downloading whole containers. An upstream backend may still prefetch source + bytes according to its own semantics; distinguish this from client downloads. +- Preserve immutable-source assumptions and explicit invalidation requirements. + No mutable-source TTL or automatic refresh protocol is introduced here. +- Surface server slice-limit errors with guidance to request a smaller preview. +- Do not imply all server-side table operations are available: cat2lite's + unsupported filters/indices must not be silently ignored or trigger an + unbounded local materialization. + +## 7. b2view integration + +- Route service recognition through shared Python-Blosc2 opening helpers. +- Retain special direct-format handling only where needed for existing array/ + table/group ownership behavior. Caterva2 interpretation takes precedence over + extensions in published dataset names. +- Run discovery, group expansion, and previews in background work, keeping the + UI responsive during slow upstream metadata reads. +- Show loading, empty-group, and retryable error states distinctly. +- Support initial paths within a root or synthetic repository. +- Check remote arrays, table paging/projection, scalar and multidimensional + previews, attrs, and plotting through existing data adapters. +- Gate unsupported table actions consistently; do not offer operations that + require full remote downloads merely because a local CTable supports them. +- Cancellation/closing the UI must release owners and transports cleanly. +- Keep CLI cache options controlling the client cache, and label this clearly + in help. The user configures the shared server cache on cat2lite separately. + +## 8. Implementation milestones + +### M0 — Freeze contract and establish fixtures + +- Audit roots schemas, opener defaults/lazy behavior, authentication, and + existing descriptor/persistence constraints. +- Finalize the explicit HTTP override and multi-root object design. +- Add deterministic local HTTP fixtures for both conventional recursive + Caterva2 listings and catalog mount-boundary listings. +- Record request counters and configurable failures/delays in those fixtures. + +Gate: a documented dispatch/return-type matrix and fixtures representing both +protocol profiles; no network dependency for ordinary tests. + +### M1 — Root-marker URL shorthand + +- Implement reusable normalization and integrate it before file-format dispatch. +- Preserve all existing URLPath and option semantics. +- Add override support and documentation for literal `@` file URLs. + +Gate: equivalent strings and URLPath objects open the same groups/arrays/tables, +including deployment prefixes and IPv6, without roots probing. + +### M2 — Lazy mount-aware RemoteStore traversal + +- Split metadata discovery from listing completion. +- Implement direct descendant lookup, lazy expansion, memoization, and bounded + concurrent discovery. +- Cover nested mounts and conventional recursively listed stores. + +Gate: a group omitted below the initial mount boundary can be expanded and +read; healthy siblings remain usable after a failed expansion. + +### M3 — Base discovery and repository roots + +- Add bounded roots discovery and transport/error handling. +- Implement single-root, empty-root, and multiple-root behavior. +- Define repository cache and lifecycle rules; cover auth-separated identities. + +Gate: bare and prefixed server URLs browse correctly, while direct-file sources +retain their dispatch and do not acquire extra roots requests. + +### M4 — b2view integration + +- Connect the shared opener, lazy tree expansion, and repository root object. +- Update title/source display, capabilities, errors, and CLI help. +- Add headless Textual tests against local servers. + +Gate: bare-server and explicit-root command forms browse groups and display +bounded array/table previews without blocking the event loop. + +### M5 — Cross-project acceptance, documentation, and measurements + +- Run against a real debug cat2lite server serving a mixed `.catl` fixture. +- Verify shared server-cache reuse across two independent clients and after + restart where the backend guarantees persistent reuse. +- Update API docs, b2view guide, examples, and release notes. +- Record startup/expansion request counts and representative cache evidence. + +Gate: all focused tests and relevant offline regressions pass; the use cases in +section 1 work with documented limits and no new viewer executable. + +## 9. Regression and acceptance matrix + +### URL dispatch + +- HTTP/HTTPS, IPv4/IPv6, ports, trailing slash, deployment prefixes. +- Root-marker group, array, table, and extension-bearing published names. +- Multiple `@` components, encoded spaces, Unicode, encoded `@` root markers. +- Query/userinfo/fragment ambiguity, traversal and double-decoding attempts. +- Ordinary direct-file `@` paths via override; signed direct-file URLs. +- Existing fsspec selectors, explicit URLPath, local paths, and non-HTTP inputs. +- Cache-option forwarding and rejection of contradictory inputs. + +### Discovery and hierarchy + +- Valid zero/one/multiple roots; unexpected JSON/HTML; 404/401/403/5xx; timeout. +- Prefix-preserving redirects and credential behavior. +- Recursive listings, mount-boundary listings, empty groups, duplicate entries. +- Direct access before parent expansion, repeated and concurrent expansion. +- Node limits and per-group failures with subsequent successful retry. +- No eager contact with unrelated remote mounts merely to open a repository. + +### Data and caching + +- Scalar/ND arrays, CTable row windows and projections, attrs/annotations. +- No implicit full fetch for previews; correct handling of server byte limits. +- none/memory/disk, identity equivalence, auth separation, close/reopen, restart. +- Two processes with independent client caches reuse one server-side cache. + Use upstream byte/request counters and fixtures larger than backend prefetch + thresholds; do not confuse client-cache hits with server-cache hits. +- Concurrent owner lifetimes and expected exclusive-lock failures where local + disk caches are deliberately shared without the supported sharing mode. + +### Viewer and compatibility + +- Headless tree expansion, starting path, empty/error states, preview paging, + cancellation, and clean shutdown under warnings-as-errors. +- Existing direct HDF5/Zarr/B2Z/Parquet viewing remains functional. +- Explicit URLPath callers retain return types and lazy=False behavior. +- Existing Caterva2 and remote-store tests remain green. + +Run Python/build commands in the `blosc2` conda environment. Use targeted pytest +runs during implementation, then the relevant offline opener/remote-store/ +remote-array/remote-table/b2view suites, Ruff on touched files, and the full +offline suite for final validation. Follow actual test filenames discovered in +the repository rather than assuming names in this plan are commands. + +## 10. Completion criteria and boundaries + +Complete when all proposed CLI forms work, equivalent Python opening works, +catalog-mounted hierarchies expand lazily, multi-root behavior is documented, +and cache reuse is demonstrated with measured upstream traffic. + +This work does not add a new viewer, a new remote protocol, write access, +catalog parsing in Python-Blosc2, automatic cache invalidation, a server +deployment/authentication system, or a guarantee that arbitrary remote formats +support cheap random access. It reuses Caterva2's protocol and existing object +adapters to make the shared-cache-server workflow convenient and predictable. diff --git a/plans/caterva2-access-improvements2.md b/plans/caterva2-access-improvements2.md new file mode 100644 index 000000000..6e8f1b929 --- /dev/null +++ b/plans/caterva2-access-improvements2.md @@ -0,0 +1,436 @@ +# Caterva2 ordinary-file access and previews in b2view + +Status: M0–M5 implemented and validated on `cat2-improvements` following explicit +user authorization. The design below is retained as the delivery/acceptance record. +It follows `plans/caterva2-access-improvements.md` and its completed URL discovery, +lazy hierarchy browsing, and viewer integration work. + +## Implementation record + +- Notebook follow-up: passive nbformat-4 cell views for local/fsspec/Caterva2 + files and local `.ipynb.b2` carriers. Markdown/code/raw cells and saved text + outputs share 64 KiB / 1,000 displayed lines and 100-cell limits; complete JSON + input is capped at 2 MiB. T toggles a bounded raw prefix. No Jupyter/kernel + dependency, execution, HTML/JavaScript, widget or embedded-media rendering. + +- Compressed local document follow-up: supported text/image/PDF names ending in + `.b2` are read-only mapped SChunk byte streams. Metadata stays payload-free; + text prefixes decode only intersecting chunks, images retain input/pixel caps, + and copies decode/write incrementally under the original filename. Lazy chunk + headers enforce 8 MiB compressed / 16 MiB decoded limits before payload copies. + Native arrays/serialized objects/references are not document carriers; plain + `.b2`/`.b2frame` dataset behavior is unchanged. Direct-fsspec carriers remain + separate work. No original file content is deserialized or executed. + +- Follow-up interface consistency: fallback panels have separate unavailable, + missing-dependency and failed-preview states, with a common status header, + reason and capability-aware action hints. Successful text previews share the + same hints (O only for supported external document types; T only for Markdown). + Read failures preserve file metadata and redact transport URLs/option values. + Local/fsspec ordinary-file support is delivered as a separate viewer-only + follow-up: direct regular-file inputs, shared passive previews/actions, + background I/O, known-size bounded reads and streamed atomic copies. Native + container formats retain their openers; unknown names get a small frame-magic + check. Generic directories, hierarchy export and new public RemoteFile + backends remain outside this follow-up. Backend caches are not Caterva2 caches. + Earlier local/fsspec exclusions in this delivery record are superseded for + direct ordinary-file inputs only. + +- User-requested directory follow-up: ordinary local/file-URL/fsspec directories + now expose immediate children lazily, with file previews/actions and dataset + containers mounted as browsable subtrees. Native dataset directories preserve + their opener. Parent traversal and local symlink following are refused. + Earlier generic-directory exclusions are superseded by this follow-up. + +- Follow-up UX decision: the extra trust checkbox is removed for all supported + documents, including images and PDFs. The explicit O action and destination + submission initiate external opening; signature/type restrictions and stale + request/shutdown guards remain. Earlier checkbox requirements below are + superseded by this user-requested decision. + +- M0: verified Caterva2 demo README transport: `api/chunk?nchunk=0` returns + a 552-byte compressed chunk that decompresses to the original 811 bytes; + `api/download` returns original Markdown bytes, not the `.b2` carrier. +- API selected: `RemoteFile(RemoteObject)` with bounded `read_bytes`, streamed + original-byte `download`, shared owner/cache/auth lifecycle, and unsupported + reference persistence. Fixed-chunk SChunks are accepted, including typed byte + payloads; irregular layouts and oversized chunks are refused explicitly. +- M1: recognition precedes array fallback; file handles participate in root and + direct-leaf opening. Existing exact-payload cache coordination/disk machinery + is reused with hash-separated file-chunk keys. Listing and metadata do not + fetch payload. Downloads stage on the destination filesystem and publish + atomically with no overwrite by default. Focused transport/access regression + validation: 62 tests passed in the blosc2 environment. +- M2–M4: b2view recognizes file leaves, renders passive bounded text/Markdown + (64 KiB / 1,000 lines), toggles raw text, and decodes JPEG/PNG under independent + file/chunk/pixel limits. Image widgets reuse textual-image without matplotlib; + absent dependencies have download/open fallbacks. PDFs require no renderer: + D downloads original bytes and O prompts for destination and explicit trust + consent before a shell-free platform launch. All fetch/decode/download work is + backgrounded; downloads use independent handles and cancellation. Viewer/model + plus file transport regressions: 175 passed with TUI cases enabled. Live demo + checks: README renders; the JPEG decodes as 2034 × 1144 (bounded preview 1600 × + 900); PDF offers download/external opening without fetching content on selection. +- M5/final review: added API/guide/installation/release documentation and the + `images` extra (Pillow/textual-image, no matplotlib; included by hires). Real + cat2lite native `.b2frame` file transport/download passes alongside existing + remote-format/shared-cache acceptance. Native cat2lite exposes this carrier + name; arbitrary ordinary-file publication/aliasing is not added to that server. +- Review hardening: bounded network and cached chunk reads before decompression; + identity HTTP encoding, content-length/body caps, 10-second I/O timeouts and a + checked 20-second chunk deadline; independent transfer aliases prepared on a + background worker; no launcher after stale selection/session or shutdown. + Old unsupported file snapshots rediscover metadata. Added download publication + race, cancellation, empty-stream, corruption, cache eviction/refresh, auth, + missing image dependency, consent, and responsive slow-transfer regressions. +- Final validation: default suite 10,670 passed, 38 skipped; focused offline + access/file/viewer/model suite 230 passed with headless TUI cases enabled; + opt-in live Caterva2 demo file downloads passed; both actual cat2lite acceptance + tests passed. Ruff/diff checks passed. HTML docs built using the existing + type-comment/notebook/generated-copy workarounds, with 764 wider autosummary/ + theme/cross-reference warnings; the build is not warnings-clean. +- Remaining constraints: fixed-chunk byte streams only; 8 MiB compressed / 16 MiB + decoded chunk caps apply to downloads too, so large single-chunk files need + server-side rechunking. No-overwrite downloads require hard-link-capable + destination filesystems. File reference persistence/hierarchy export is deferred. + MIME is a filename hint; binary/unknown files have no automatic preview. Image + decoder copies are not a whole-process memory limit. External launchers depend + on platform/GUI associations and are not a sandbox; tests mock launchers, never + open documents automatically. User-selected downloads are retained (no hidden + temporary external-viewer copies to manage). Local/fsspec regular-file support + and in-terminal PDF rendering remain explicit non-goals. +- Live headless image integration also verified the real demo JPEG mounts an + `AutoImage` widget with the installed textual-image package (not a mocked + renderer). Explicit downloads remain bounded even when previews are unavailable; + no automatic raw-download endpoint fallback bypasses chunk safety limits. + +## 1. Goal and delivery order + +Browse ordinary files published by Caterva2-compatible services without treating +them as unsupported array/table objects. Preserve the current hierarchy, root +conventions, lazy discovery, authentication, and client/server cache separation. + +Deliver in this order: + +1. Ordinary-file metadata, bounded byte access, explicit downloads, and text/ + Markdown previews. +2. Image previews using optional image dependencies and terminal capabilities. +3. PDF identification, download, and explicit external opening. Embedded PDF + rendering is optional follow-up work, not required for this plan's completion. + +Every file remains downloadable even if its format cannot be previewed. Do not +make downloading an entire file an implicit prerequisite for listing a directory. + +Target viewer behavior: + +| Selection | Metadata/data behavior | +| --- | --- | +| `@public/examples/README.md` | File metadata and bounded Markdown/text preview | +| `@public/examples/Wutujing-River.jpg` | File metadata and size-limited image preview, or a clear fallback | +| `@public/examples/cat2cloud-brochure.pdf` | File metadata, PDF notice, download/external-open actions | +| Unknown binary file | File metadata, preview-unavailable notice, download action | +| Generic SChunk without ordinary-file semantics | Byte-stream metadata/download; no invented original-file type | + +Paths displayed for `@` roots have no leading slash. Internal browser selection +paths remain unchanged. Local and direct fsspec regular-file support is outside +this first delivery; do not accidentally intercept existing direct-format URLs. + +## 2. Current implementation and verified evidence + +Relevant code: + +- `src/blosc2/remote_store.py`: `_caterva2_kind`, discovery metadata, node lookup, + leaf handles, owner lifecycle, cache coordination, and `RemoteNode`. +- `src/blosc2/schunk.py`: public `open` and Caterva2 leaf dispatch. +- `src/blosc2/c2array.py`: existing authentication and info/fetch/chunk transport. +- `src/blosc2/remote_object.py`: common remote-object contract. +- `src/blosc2/remote_repository.py`: independently owned repository roots. +- `src/blosc2/b2view/model.py`: object kinds, metadata, preview selection, leaf + lifetime, and display paths. +- `src/blosc2/b2view/app.py`: worker requests, stale-result rejection, download + UI, and optional `textual-image` support currently used by plot screens. +- `src/blosc2/b2view/render.py`: Rich metadata and data rendering. +- `tests/test_remote_caterva2.py`, `tests/test_caterva2_access.py`, and + `tests/b2view/test_caterva2.py`: deterministic API/viewer fixtures. + +The current `_caterva2_kind` accepts groups, NDArrays, and CTables. Metadata +without array shape/dtype or table schema consequently becomes `unsupported`. + +The following live info responses were inspected during planning at +`https://cat2.cloud/demo/api/info/@public/examples/`: + +| Name | Uncompressed bytes | Compressed bytes | Chunks | +| --- | ---: | ---: | ---: | +| `README.md` | 811 | 552 | 1 | +| `Wutujing-River.jpg` | 724,498 | 723,929 | 1 | +| `cat2cloud-brochure.pdf` | 44,155 | 36,889 | 1 | + +These responses expose SChunk fields (`nbytes`, `cbytes`, `chunksize`, `nchunks`, +`cparams`, `vlmeta`) and server-internal `.b2` backing paths, not NDArray metadata. +Do not display or construct download names from those internal filesystem paths. +These observations establish metadata shape, not full transport compatibility: +chunk and download semantics must still be verified before implementation. + +## 3. M0 — Protocol characterization and API decisions + +### Protocol matrix + +Create loopback fixtures for an ordinary-file SChunk, a generic byte SChunk, an +NDArray, a table, an unknown object, and a group. Characterize: + +- Explicit kind values where available, plus legacy SChunk-shaped info responses. +- `api/chunk` addressing and response framing for SChunks; final partial chunks, + empty streams, and variable-length/irregular layouts. +- Whether bounded `api/fetch` selection uses byte positions or typed elements. +- `api/download` semantics: original file bytes versus compressed Blosc frame, + suffix handling, response headers, streaming, and redirects. +- Metadata preservation for MIME type, logical filename, user attrs, and catalog + annotations, without assuming those fields exist on legacy servers. +- Compatibility with an actual Caterva2 deployment and cat2lite. cat2lite has + SChunk and download routes, but arbitrary native-file publication is not an + assumed existing capability. Record unsupported server combinations explicitly. + +Prefer chunk-based lazy reads if the protocol supports them reliably. Do not +fall back to whole-file fetches silently for a bounded text preview. If a server +cannot provide bounded work, expose metadata/download and explain the preview +limitation, or require explicit consent for a capped full-file fetch. + +### Proposed public abstraction + +Use a `RemoteFile` (or a clearly documented byte-stream equivalent) inheriting +`RemoteObject`, separate from array indexing and table operations. Resolve its +final public name and API in M0 before implementation; audit existing SChunk and +Proxy machinery for reusable transport/cache behavior. + +Proposed contract: + +- `source`, `attrs`, `traffic`, cache policy/allowance/usage, `close`, and context + management follow existing remote-object conventions. +- `nbytes`, logical `name`, optional media type, and chunk/compressed-size metadata. +- Explicit bounded `read_bytes(start, stop)` returning original uncompressed bytes. + Validate offsets and bounds; do not overload NDArray slicing semantics. +- Streaming `download(destination, overwrite=False)` writes the original payload, + never an undocumented `.b2` carrier, with optional progress/cancellation hooks. +- Downloading bytes is distinct from `save` of a remote reference. Reference + persistence/materialization is explicitly unsupported initially unless existing + artifact machinery can support it safely within scope. +- No writable source operations. Closing a root/repository leaves returned file + handles usable, as with existing array/table handles. + +Direct service leaf opening via `blosc2.open("https://host/@public/README.md")` +should return the same handle as hierarchy lookup. Preserve explicit URLPath +defaults: define and test the nonlazy byte-stream case rather than silently +changing all existing URLPath behavior. + +## 4. M1 — Discovery, byte access, cache, and download + +### Recognition and metadata + +- Recognize explicit file/SChunk kinds and strictly validated legacy SChunk + metadata. Classify NDArray/CTable metadata first to avoid confusing their + embedded chunk metadata with a file. +- Validate nonnegative byte counts, chunk counts, and consistent layouts. Reject + booleans masquerading as counts, invalid compression metadata, and impossible + lengths. Do not assume every SChunk is a regular fixed-chunk byte stream. +- Separate the transport kind from preview format. A `.jpg` suffix does not make + invalid metadata valid, nor guarantee that payload bytes are an image. +- Retain source attrs, catalog attrs, and file transport metadata separately. +- Preserve unknown metadata as an unsupported node with a useful diagnostic; + an invalid file must not hide healthy siblings. + +### Transport and resource ownership + +- Reuse bound auth, deployment-prefix handling, safe URL components, transport + pooling, and existing owner locks; never use a server-local `urlpath` as a URL. +- Fetch only necessary compressed chunks; validate frame/chunk headers and output + length before allocating/decompressing. Charge actual response bytes once. +- State preview transfer bounds separately from output bounds: a 64 KiB prefix + may require a much larger first chunk. Reject automatic preview work above + compressed/decompressed chunk limits instead of claiming it is bounded. +- Streaming download uses bounded buffers/chunks and checks cancellation between + reads. Enforce response and decompression limits on the streamed path, not only + after loading a response into memory. +- Never deserialize Python objects or execute embedded metadata to read a file. + +### Cache semantics + +- `MEMORY`: share the owner's existing retained-payload allowance across file, + array, and table leaves. Do not create a hidden per-file unlimited cache. +- `NONE`: retain no payload between operations. Transient preview/download + buffers still have explicit limits and are not called persistent cache. +- `DISK`: reuse source/auth identity, exclusive ownership, atomic publication, + and eviction conventions. Test reopened warm chunks and partial downloads. +- Keep client caching distinct from gateway caching. Repository budgets remain + per root. Immutable-source assumptions and explicit invalidation remain. +- Preview render buffers/decoded images have a separate, documented lifecycle + and limit; they are not silently counted as compressed chunk-cache bytes. + +### Safe downloads + +- Destination is chosen by the user. Sanitize suggested names from logical paths, + not Content-Disposition or server-local paths; reject traversal and separators. +- No overwrite without confirmation; publish only a completed download. Use a + staging file on the destination filesystem and a no-clobber publication strategy + when overwrite is false, including concurrent-creation tests. +- Failure/cancellation cleans up only this operation's staging file. Leave prior + destination contents untouched. Offer actionable permission/disk-full errors. +- Download the original `.md`, `.jpg`, or `.pdf` bytes; expose compressed-carrier + export only as a separately named future operation if needed. + +Acceptance: list/inspect without payload reads; exact arbitrary byte ranges and +streamed original-file downloads; mixed-leaf owner/cache/lifecycle tests pass. + +## 5. M2 — Text/Markdown previews and viewer actions + +Add a file node kind/icon, lightweight file metadata, and a preview result type +separate from numerical array/table previews. Do not route files through grid +paging, table filters, plotting, or scalar-array coercion. + +Suggested initial limits (finalize constants in M0 and document them): + +- Automatic text prefix: 64 KiB of original bytes, at most 1,000 displayed lines. +- Automatic chunk work: at most 8 MiB compressed and 16 MiB decompressed per + chunk; higher limits require explicit user action, not silent escalation. +- Download: streamed rather than held whole in RAM; explicit consent includes + known original size and destination. Unknown size gets a warning and progress + bytes without a fabricated percentage. + +Decode UTF-8/UTF-8 BOM first, with incremental decoding for a prefix that cuts a +multibyte character. Detect binary/NUL-heavy content and avoid garbage previews. +Invalid text gets an honest replacement/encoding notice; do not guess encodings +aggressively. Escape terminal control/ANSI sequences, including OSC hyperlinks. + +For Markdown, use Textual's existing Markdown capabilities without new mandatory +dependencies. Offer raw text fallback/toggle. Disable automatic embedded-image +fetches, local file access, and external-link launching: a preview must not issue +unrelated network requests. Mark truncated content, including unmatched fences; +rendering errors fall back to safe text rather than breaking the viewer. + +Add explicit Download and Open externally actions. Audit existing keybindings +before selecting keys; document them in help and expose enabled/disabled states. +For PDF/unknown binary files, these actions are available even without a preview. + +All remote fetch/decode/download work runs outside the UI thread. Bind results +to session, selection, and request IDs. Stale results must not change the visible +preview or launch an external application. Show progress/errors, preserve healthy +sibling browsing, and support retry. Shutdown releases all owned resources. + +Acceptance: README renders at its published path, large text is visibly truncated, +binary files remain downloadable, and delayed/cancelled reads do not freeze or +overwrite the current selection. + +## 6. M3 — Images + +Reuse Pillow and the optional `textual-image` integration; do not require +matplotlib just to display an existing JPEG/PNG. Audit packaging extras so the +installation hint accurately describes the minimum image dependencies. + +- Validate file signature/decoder result, not suffix alone. Start with JPEG/PNG; + do not promise arbitrary image formats or SVG/remote-resource rendering. +- A normal image preview may require the complete original file. Suggested + automatic original-payload cap: 16 MiB, with independent chunk/transfer bounds. +- Inspect dimensions before full decoding; suggested decoded-image budget: + 64 MiB with an explicit pixel cap. Treat Pillow decompression-bomb warnings as + refusal and reject malformed/oversized inputs. Downsampling after decoding is + not a substitute for a predecode limit. +- Correct EXIF orientation, aspect ratio, and resize-to-panel behavior. Show + dimensions and detected format. Bound animated-image handling to a first frame. +- If dependencies or terminal protocols are unavailable, show metadata and a + fallback notice with Download/Open externally. Do not leave a blank panel. +- Release image buffers/widgets on selection change and shutdown; no unbounded + decoded-thumbnail cache. Preserve worker/stale-result checks during resize. + +Acceptance: small JPEG/PNG fixtures display with image support enabled; missing +dependency/protocol and unsafe-image cases have useful, tested fallbacks. + +## 7. M4 — PDFs and external opening + +PDF support in this delivery means recognition, metadata, and convenient safe +download/opening, not a mandatory terminal PDF renderer. A PDF may be rendered +externally even when the terminal cannot show images. + +- Identify PDF candidates from extension/media metadata and verify signature when + bytes are read. Do not invoke a PDF parser just to list or inspect file size. +- Display "PDF preview unavailable; download or open externally" and actionable + controls. Missing renderer is not an unsupported-file error. +- External opening is an explicit user action, never triggered by selection, + refresh, MIME detection, or restoring a session. +- Prefer opening a completed local download, so bound service credentials never + appear in command-line URLs. Use platform launchers with argument arrays and + no shell (`open`, `xdg-open`, or the appropriate Windows API), check availability, + and use a safe absolute filename. Test launcher behavior with mocks only. +- Confirm opening untrusted content. Never mark downloads executable or dispatch + unknown scripts/binaries to a launcher; external-open initially has a narrow + document/image allowlist. Download remains available for other file types. +- Distinguish user-owned downloads from application-owned temporary copies. + External viewers can outlive b2view: document retention/cleanup and do not + delete a temporary file immediately after launch. Use a private session/temp + directory with cleanup of owned stale copies under an explicit policy. +- Report no-GUI/headless/missing-launcher failures without losing the downloaded + file; show its location and allow the user to open it manually. + +Optional later milestone: PDF first-page thumbnails/text extraction using an +opt-in backend, with page/pixel/time limits and untrusted-parser isolation where +appropriate. This is not needed for M4 acceptance and must not become a mandatory +dependency for ordinary-file access. + +## 8. M5 — Integration, documentation, and final review + +Extend offline fixtures with Markdown, invalid UTF-8, arbitrary binary data, +JPEG/PNG, and PDF bytes, all transported as deterministic SChunks. Include empty, +partial final chunk, typed/irregular SChunk, false suffix, and malformed responses. + +Acceptance matrix: + +| Area | Required checks | +| --- | --- | +| Discovery | Legacy/explicit kinds; groups/arrays/tables unchanged; no payload on listing | +| Opening | String URL, URLPath, bare single/multi-root services, nested mounts; @ display paths | +| Byte access | Empty/prefix/cross-chunk/end ranges; invalid bounds; output-length mismatch | +| Cache | NONE/MEMORY/DISK; mixed-leaf budgets; eviction; reopened warm data; auth isolation | +| Lifecycle | Children outlive parents; refresh/stale handles; cancellation and concurrency | +| Text | Markdown/raw; UTF-8 boundary/BOM; binary/control characters; truncation; no linked fetches | +| Images | JPEG/PNG; missing dependency/protocol; corruption; pixel/decode caps; resize/release | +| Download | Original byte equality; overwrite race; unsafe names; cancelled/truncated response; disk failure | +| External open | Explicit consent; no shell; launcher unavailable; no secrets; stale launch suppressed | +| PDF | Useful nonrendering fallback; download/open without PDF dependency | +| UI | Headless worker responsiveness; failed leaf isolation; actions/help and existing grids unaffected | + +Add opt-in real-service acceptance, not public-network dependency in the default +suite. Where supported, verify two independent clients can reuse gateway payload +cache; distinguish metadata rereads from payload rereads. Extend the actual +cat2lite acceptance only for supported publication/protocol behavior, not an +invented arbitrary-file server capability. + +Update public class/open documentation, remote-object guide, b2view guide, +optional-extra installation instructions, and release notes. Include limits, +download versus reference export, platform opening behavior, and fallbacks. + +Run Python/tests/build commands only in the `blosc2` conda environment. Run Ruff, +focused access/viewer tests (including headless TUI cases), the default suite, +and documentation checks. Preserve unrelated work and downloaded user files. + +Final review focuses on: + +1. Protocol assumptions validated on supported server versions. +2. Peak-memory/transfer limits, especially one-chunk large files and images. +3. Auth and path safety, unsafe content, and no implicit external side effects. +4. Shared cache accounting and returned-handle lifetime. +5. Worker cancellation, stale-result suppression, and safe atomic download. +6. Missing-dependency/headless/platform fallbacks and remaining limitations. + +## 9. Known tradeoffs and non-goals + +- Compression granularity may make a tiny byte prefix expensive. Preview refusal + is preferable to an unbounded hidden download. +- MIME/suffix hints are imperfect; content recognition is bounded and conservative. +- Budgets bound retained compressed payload, not all decoder/transient RAM; + preview/decompression limits must be enforced independently. +- External applications have their own security and lifecycle behavior; explicit + consent is not a sandbox. Credentials must never be passed to them. +- No recursive directory download, archive extraction, file editing/upload, + automatic script execution, OCR, animated playback, or embedded browser. +- No compulsory PDF renderer; download/external opening is a complete first + delivery for PDF and terminal-unrenderable supported documents. +- Ordinary-file reference archives and full fsspec/local file-browser expansion + are separate future work unless M0 proves a small, safe existing integration. diff --git a/plans/caterva2-demo-dataset-audit.md b/plans/caterva2-demo-dataset-audit.md new file mode 100644 index 000000000..5fdfae937 --- /dev/null +++ b/plans/caterva2-demo-dataset-audit.md @@ -0,0 +1,140 @@ +# Caterva2 demo dataset audit + +Audit of `b2view https://cat2.cloud/demo`, 2026-10-03, in the `blosc2` +environment. The live catalog has one root, `@public`. All paths below are +relative to that root. This is a snapshot, not a guarantee about future server +contents or installed codec versions. + +## Method and fixes + +- Enumerated `api/roots`, `api/list/@public`, and the HDF5 mount's own listing. +- Inspected every listed leaf's metadata through `StoreBrowser.get_info` and + attempted a bounded preview (`max_rows=2`, `max_cols=2`). The legacy N-D + preview path uses up to 20 rows of one plane. Array reads remain chunk-granular; + no whole large array, notebook, PDF, or Parquet file was downloaded. +- There are 32 top-level listing entries: 31 leaves and one HDF5 group. That + group exposes 11 more leaves: **42 leaves audited individually**. +- Fixed `C2Array.dtype` to accept literal structured/subarray descriptors, as + native b2nd readers already do. Both NumPy `TypeError` and `ValueError` paths + matter. Parsing uses `ast.literal_eval`, not executable `eval`. +- Fixed synthesized Caterva2 chunks for top-level subarray dtypes: NumPy expands + these into trailing dimensions, so chunk/block geometry must expand too when + repacking. The public source shape/dtype remain unchanged. Tests include + nonzero data, multiple chunks, partial edge chunks, block padding, and both + one- and two-dimensional outer shapes. +- Five previously failing dtype cases now preview successfully: three native + structured arrays, HDF5 compound dtype, and HDF5 subarray dtype. The literal + directory name `unsupported` is not authoritative about current capabilities. + +## Every top-level entry + +| Path | Result after fixes | Analysis | +| --- | --- | --- | +| `examples/README.md` | Text/Markdown preview works | Ordinary SChunk byte stream. | +| `examples/Wutujing-River.jpg` | Image preview works | JPEG; Pillow decodes and reduces the image. Terminal image layout was fixed separately. | +| `examples/cat2cloud-brochure.pdf` | No inline preview, intentional | PDF download/external opening supported; no embedded PDF renderer. | +| `examples/cube-1k-1k-1k.b2nd` | Numeric preview works | `int32`, shape `(1000,1000,1000)`; preview of one plane, not the entire volume. | +| `examples/cubeA.b2nd` | Numeric preview works | `float64`, shape `(1,1000,1000)`. | +| `examples/cubeB.b2nd` | Numeric preview works | `float64`, shape `(1,1000,1000)`. | +| `examples/dir1/ds-2d.b2nd` | Numeric preview works | `uint16`, shape `(10,20)`; chunks need edge/block padding. | +| `examples/dir1/ds-3d.b2nd` | Numeric preview works | `float32`, shape `(3,4,5)`. | +| `examples/dir2/ds-4d.b2nd` | Complex preview works | `complex128`, shape `(2,3,4,5)`; no dtype parsing failure. | +| `examples/ds-1d-b.b2nd` | Byte-string preview works | `S6`, shape `(1000,)`; values contain `foobar`. | +| `examples/ds-1d-fields.b2nd` | **Fixed** | Structured integer/float/byte-string/bool dtype was a stringified field list. | +| `examples/ds-1d.b2nd` | Numeric preview works | `int64`, shape `(1000,)`. | +| `examples/ds-2d-fields.b2nd` | **Fixed** | Structured float32/float64 dtype; shape `(100,200)`. | +| `examples/ds-hello.b2frame` | Downloadable, no automatic preview | Plain SChunk has no text suffix. A bounded read confirms repeated `Hello world!` bytes. It is not corrupt; content sniffing/raw SChunk previews are not implemented. | +| `examples/ds-sc-attr.b2nd` | Scalar string preview works | Zero-dimensional ``: +each returns HTTP 500, including `company`, so this is not just dotted names. +Without `field`, `slice_=0:2` returns 5,658 bytes that decode successfully. +The viewer/RemoteCTable uses column projection and therefore encounters this +server defect. A blanket retry without projection was not introduced: it would +hide server errors and could multiply transfer sizes. The deployment needs a +projection fix, or a deliberately bounded/capability-aware compatibility path. + +### Passive preview limitations (eight leaves) + +The PDF, plain `.b2frame`, five notebooks, and opaque Parquet file are recognized +file handles, not unrecognized nodes. Unsupported preview does not mean broken +discovery. PDF has explicit external opening; notebook/SChunk content can be +downloaded and examined separately. Parquet is the exception: this particular +server's chunk geometry exceeds byte-stream download limits too. Direct Python +Blosc2 Parquet-source support does not imply a Caterva2 opaque file is a table. + +## Summary + +After the dtype/chunk fixes: **30 leaves successfully preview**, **8 have no +automatic preview by current policy**, and **4 fail for external reasons** +(3 local GROK-plugin loads, 1 server table-projection error). The HDF5 group's +`unsupported` directory now has three readable server representations. These +findings describe sampled reads, not full-data integrity or equality to original +HDF5 sources. + +## Validation + +- Default suite: **10,685 passed, 38 skipped** in the `blosc2` environment. +- Focused offline dtype/access/viewer/model tests: **234 passed**. +- Opt-in live regression tests for all five repaired dtype leaves: **5 passed**. +- Actual headless `B2ViewApp` sessions populated the data grid for each of those + five live leaves (including HDF5 subarray cells), not just model-only reads. diff --git a/pyproject.toml b/pyproject.toml index 66e29ab23..07664b095 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -58,9 +58,10 @@ hdf5 = ["h5py", "hdf5plugin"] # wasm32 (no TTY). Install with `pip install "blosc2[tui]"`. This also pulls # textual-plotext for the in-terminal braille plot (the 'p' key). tui = ["textual", "textual-plotext"] +images = ["blosc2[tui]", "textual-image", "pillow"] # Adds the high-res 'h' view on top of [tui], rendering a real matplotlib image # (kitty/iTerm2/sixel, or half-cells elsewhere) — matplotlib is the heavy part. -hires = ["blosc2[tui]", "textual-image", "matplotlib"] +hires = ["blosc2[images]", "matplotlib"] # Read/write single-file containers through any fsspec URL (https://, s3://, # gs://, zip://, memory://...). HTTP support is included; the other protocol # backends (s3fs, gcsfs, adlfs...) are the caller's install: diff --git a/src/blosc2/__init__.py b/src/blosc2/__init__.py index 75e61e82a..39008fd0b 100644 --- a/src/blosc2/__init__.py +++ b/src/blosc2/__init__.py @@ -627,6 +627,8 @@ def _raise(exc): from .remote_object import RemoteObject from .remote_array import RemoteMetadataMapping, RemoteArray from .remote_store import RemoteNode, RemoteStore +from .remote_file import RemoteFile +from .remote_repository import RemoteRepository from . import linalg from .linalg import tensordot, vecdot, permute_dims, matrix_transpose, matmul, transpose, diagonal, outer from .utils import linalg_funcs as linalg_funcs_list @@ -931,6 +933,8 @@ def _raise(exc): "RemoteCTable", "RemoteNode", "RemoteStore", + "RemoteFile", + "RemoteRepository", "SChunk", "SimpleProxy", "SpecialValue", diff --git a/src/blosc2/b2view/app.py b/src/blosc2/b2view/app.py index e131b8494..47a2974ff 100644 --- a/src/blosc2/b2view/app.py +++ b/src/blosc2/b2view/app.py @@ -42,7 +42,7 @@ PlotextPlot = None try: - # Auto-selects the best terminal image protocol (kitty/iTerm2/sixel), + # Auto-selects a supported terminal image protocol (kitty/sixel), # degrading to colored half-cells; used by the high-res 'h' plot view. from textual_image.widget import Image as TextualImage except ImportError: # high-res view is optional @@ -67,6 +67,7 @@ "c2array": "▦", "ctable": "▤", "schunk": "▣", + "file": "📄", "unknown": "?", } @@ -253,6 +254,7 @@ class HelpScreen(ModalScreen[None]): [ ("up / down", "move between nodes"), ("enter", "select node (and expand groups)"), + ("f / ctrl+f", "filter discovered paths; optionally search recursively"), ], ), ( @@ -292,6 +294,15 @@ class HelpScreen(ModalScreen[None]): ("escape", "close the plot (q quits b2view)"), ], ), + ( + "Ordinary files", + [ + ("D", "download original bytes to a chosen destination (no overwrite)"), + ("O", "download and open a document externally"), + ("T", "toggle raw text / Markdown or notebook rendering"), + ("escape", "cancel a file download or close its dialog"), + ], + ), ( "Dim mode (N-D arrays)", [ @@ -1884,6 +1895,96 @@ def on_progress(downloaded: int, content_total: int | None) -> None: self.app.call_from_thread(self.dismiss, True) +class FileTransferScreen(ModalScreen): + """Explicit destination and cancellable background original-byte download/open.""" + + CSS = """ + FileTransferScreen { align: center middle; } + #file-transfer { width: 75; height: auto; border: thick $accent; padding: 1 2; background: $surface; } + """ + BINDINGS: ClassVar = [("escape", "cancel", "Cancel")] + + def __init__(self, file, *, external=False): + super().__init__() + self.file = file + self.external = external + self.cancelled = threading.Event() + self.started = False + + def compose(self): + from pathlib import Path + + from blosc2.b2view.file_preview import safe_text + + name = safe_text(self.file.name).replace("\\", "_").replace("/", "_") + action = "Download and open externally" if self.external else "Download original file" + with Vertical(id="file-transfer"): + yield Static( + f"{action}: {name} ({self.file.nbytes:,} bytes)\n" + "Choose a destination; Enter starts. Escape cancels. Existing files are not overwritten.", + markup=False, + ) + yield Input(value=str(Path.cwd() / name), id="file-destination") + yield Static("", id="file-status", markup=False) + yield ProgressBar(id="file-progress") + + def on_mount(self): + self.origin = (self.app._remote_session, self.app._remote_request, self.app.selected_path) + self.query_one(Input).focus() + + def on_input_submitted(self): + if self.started: + return + destination = self.query_one(Input).value + if not destination.strip(): + return + self.started = True + self.query_one(Input).disabled = True + self._transfer(destination) + + @work(thread=True, exit_on_error=False) + def _transfer(self, destination): + def progress(done, total): + self.app.call_from_thread( + self.query_one("#file-progress", ProgressBar).update, total=total, progress=done + ) + + try: + path = self.file.download(destination, progress=progress, cancel=self.cancelled.is_set) + # Navigation/refresh/shutdown must never launch an obsolete request. + current = (self.app._remote_session, self.app._remote_request, self.app.selected_path) + if ( + not self.cancelled.is_set() + and self.external + and current == self.origin + and not self.app._closing + ): + from blosc2.b2view.file_preview import open_external + + open_external(path) + if not self.cancelled.is_set(): + self.app.call_from_thread(self._finished, f"Saved: {path}") + except Exception as error: + if not self.cancelled.is_set(): + self.app.call_from_thread( + self._finished, f"{self.app._error_message(error)}\nDestination: {destination}" + ) + finally: + self.file.close() + + def _finished(self, message): + self.query_one("#file-status", Static).update(message + "\nEscape closes this dialog.") + + def action_cancel(self): + self.cancelled.set() + self.dismiss() + + def on_unmount(self): + self.cancelled.set() + if not self.started: + self.file.close() + + class B2ViewHeader(Header): """App header that also shows the open bundle's filename, left of the title. @@ -1980,6 +2081,7 @@ class B2ViewApp(App): #data-header { height: auto; padding: 0 1; } #data-table-row { height: 1fr; } #data-table { width: 1fr; height: 1fr; } + #file-image { height: auto; } #row-scrollbar { width: 1; height: 1fr; color: $primary; } #col-scrollbar { height: 1; width: 1fr; color: $primary; } #meta-scroll, #attrs-scroll, #data-scroll { height: 1fr; padding: 0 1; } @@ -1999,12 +2101,13 @@ class B2ViewApp(App): Binding("g", "go_to_row", "Go to row", show=False), ("m", "maximize_panel", "Maximize"), ("r", "restore_or_refresh", "Restore/Refresh"), + ("ctrl+f", "tree_search", "Find"), Binding("t", "grid_row_top", "Top", show=False), Binding("b", "grid_row_bottom", "Bottom", show=False), Binding("s", "grid_col_start", "Row start", show=False), Binding("e", "grid_col_end", "Row end", show=False), Binding("c", "go_to_column", "Go to column", show=False), - Binding("f", "filter_rows", "Filter rows", show=False), + Binding("f", "find_or_filter", "Find / Filter rows", show=False), Binding("S", "sort_rows", "Sort by", show=False), Binding("R", "reverse_sort", "Reverse sort", show=False), Binding("G", "group_rows", "Group by", show=False), @@ -2013,6 +2116,9 @@ class B2ViewApp(App): Binding("d", "dim_cycle", "Dim mode", show=False), Binding("enter", "dim_toggle_nav", "Toggle nav", show=False), Binding("escape", "dim_exit", "Exit dim mode", show=False), + Binding("D", "download_file", "Download file", show=False), + Binding("O", "open_file", "Open externally", show=False), + Binding("T", "raw_file", "Raw/Rendered", show=False), ] def __init__( @@ -2029,8 +2135,11 @@ def __init__( storage_options: dict[str, Any] | None = None, cache_dir: str | None = None, max_cache_bytes: int | None = None, + remote_service: str = "auto", ): super().__init__() + self.register_theme(BLOSC2_THEME) + self.theme = "blosc2" self.sub_title = f"Python-Blosc2 {blosc2.__version__}" # shown beside the title in the header if parse_container_url(urlpath)[2] in {"zarr", "hdf5"}: # Initialize before Textual captures stderr (fileno=-1), which @@ -2043,6 +2152,7 @@ def __init__( self.storage_options = storage_options self.cache_dir = cache_dir self.max_cache_bytes = max_cache_bytes + self.remote_service = remote_service self.download_url = download_url # when set, fetch urlpath before browsing self.info_url = info_url # optional: metadata endpoint giving the size # Header label: the path as given on the CLI, or the @public-relative @@ -2054,13 +2164,19 @@ def __init__( self.preview_rows = preview_rows self.preview_cols = preview_cols self.browser: StoreBrowser | None = None + self._source_opened = False + self.startup_error: str | None = None # Set when a remote browser is closed on its own thread (on_unmount); # lets teardown wait for the cache-dir lock to be released. self._browser_close_thread: threading.Thread | None = None self.loaded_paths: set[str] = set() - self._remote = is_fsspec_url(urlpath) + from blosc2.b2view.ordinary_file import ordinary_source + + # Keep ordinary-file disk I/O and image decoding off the UI thread too. + self._remote = is_fsspec_url(urlpath) or ordinary_source(urlpath, remote_service) self._remote_session = 0 self._remote_request = 0 + self._file_raw = False self._remote_page_request = 0 self._remote_page_pending = False self._remote_col_end = None @@ -2079,6 +2195,9 @@ def __init__( self._active_dim = 0 self._dim_mode = False self.loading_table_page = False + self._data_busy = False + self._data_busy_frame = 0 + self._data_busy_timer = None # One-shot: apply the --panel start focus after the first update_panels, # once the data panel's display/contents have settled (see update_panels). self._apply_focus_on_next_update = False @@ -2093,6 +2212,7 @@ def compose(self) -> ComposeResult: with Horizontal(id="main"): with B2ViewPanel(id="tree-pane") as tree_pane: tree_pane.border_title = "tree" + tree_pane.border_subtitle = "?(help) | f(ind) | r(efresh)" yield Tree("/", id="tree") with Vertical(id="right-pane"): with Horizontal(id="top-row"): @@ -2117,11 +2237,11 @@ def compose(self) -> ComposeResult: yield Static("", id="col-scrollbar") with VerticalScroll(id="data-scroll", can_focus=True): yield Static("", id="preview") + yield Vertical(id="file-image") yield Footer() def on_mount(self) -> None: - self.register_theme(BLOSC2_THEME) - self.theme = "blosc2" + self._data_busy_timer = self.set_interval(0.2, self._animate_data_busy, pause=True) if self.download_url: # Fetch the bundle first, then open it from _after_download. The # message shows the @public-relative path (e.g. "large/foo.b2z"), @@ -2153,21 +2273,28 @@ def _after_download(self, result: bool | str) -> None: if result is True: self._start_browsing() else: - self.exit(message=f"Download failed: {result}") + self._source_open_error(RuntimeError(f"Download failed: {result}")) def _start_browsing(self) -> None: """Open the bundle and populate the tree (the normal startup path).""" if self._remote: + self._set_data_busy(True) self.query_one("#metadata", Static).update("Loading remote container…") self._open_remote(self._remote_session, self.start_path) return browser_kwargs: dict[str, Any] = { "storage_options": self.storage_options, "cache_dir": self.cache_dir, + "remote_service": self.remote_service, } if self.max_cache_bytes is not None: browser_kwargs["max_cache_bytes"] = self.max_cache_bytes - self.browser = StoreBrowser(self.urlpath, **browser_kwargs) + try: + self.browser = StoreBrowser(self.urlpath, **browser_kwargs) + except Exception as exc: + self._source_open_error(exc) + return + self._source_opened = True self._populate_browser() def _populate_browser(self) -> None: @@ -2217,7 +2344,7 @@ def _focus_panel_by_name(self, name: str) -> None: if getter is not None: getter().focus() - def _navigate_to_path(self, path: str) -> None: + def _navigate_to_path(self, path: str, *, focus_tree: bool = False) -> None: """Expand the tree and select the node at *path*.""" tree = self.query_one("#tree", Tree) parts = [p for p in path.split("/") if p] @@ -2246,6 +2373,8 @@ def _navigate_to_path(self, path: str) -> None: def _do_select(): tree.select_node(node) tree.scroll_to_node(node) + if focus_tree: + tree.focus() self.call_after_refresh(_do_select) @@ -2312,6 +2441,7 @@ def _open_remote(self, session, start_path): browser_kwargs: dict[str, Any] = { "storage_options": self.storage_options, "cache_dir": self.cache_dir, + "remote_service": self.remote_service, } if self.max_cache_bytes is not None: browser_kwargs["max_cache_bytes"] = self.max_cache_bytes @@ -2331,15 +2461,18 @@ def _open_remote(self, session, start_path): browser.close() except Exception as exc: if browser is not None: - browser.close() - self._deliver_remote(session, self._remote_error, exc) + # Cleanup must not mask the opening failure and strand the UI. + with contextlib.suppress(Exception): + browser.close() + self._deliver_remote(session, self._source_open_error, exc) def _finish_remote_open(self, browser, children): self.browser = browser + self._source_opened = True self._remote_children = children self._populate_browser() - def _remote_error(self, exc): + def _error_message(self, exc): # Transport exceptions can include signed URLs or credentials. Keep # source-specific limitations, but remove runtime URLs and option values. import re @@ -2355,7 +2488,58 @@ def redact(options): message = message.replace(value, "