From a1b3baeb3d097b042eecc5e5bd7053f5166f4f46 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Thu, 17 Sep 2026 19:26:24 +0200 Subject: [PATCH 01/82] Add remote CTable access --- examples/ctable/remote_handling.py | 191 +++++++++++++++++++ examples/dict-store.py | 2 +- examples/embed-store.py | 2 +- examples/embedded-expr-udf-b2z.py | 4 +- examples/ndarray/proxy-ndarray.py | 8 +- examples/ndarray/swmr-enlarge-bars.py | 6 +- examples/ndarray/swmr-enlarge.py | 12 +- examples/ref-object.py | 12 +- examples/remote/s3-access.py | 6 +- examples/tree-store-blog.py | 22 +-- examples/tree-store.py | 14 +- examples/vlmeta.py | 22 +-- plans/remote-ctable.md | 261 ++++++++++++++++++++++++++ src/blosc2/__init__.py | 2 + src/blosc2/ctable.py | 92 ++++----- src/blosc2/ctable_storage.py | 157 +++++++++++++++- src/blosc2/remote_ctable.py | 116 ++++++++++++ src/blosc2/remote_store.py | 138 ++++++++++---- tests/b2view/test_hierarchy.py | 5 +- tests/ctable/test_remote_ctable.py | 132 +++++++++++++ tests/test_remote_store.py | 32 ++++ 21 files changed, 1083 insertions(+), 153 deletions(-) create mode 100644 examples/ctable/remote_handling.py create mode 100644 plans/remote-ctable.md create mode 100644 src/blosc2/remote_ctable.py create mode 100644 tests/ctable/test_remote_ctable.py diff --git a/examples/ctable/remote_handling.py b/examples/ctable/remote_handling.py new file mode 100644 index 000000000..cf1ada097 --- /dev/null +++ b/examples/ctable/remote_handling.py @@ -0,0 +1,191 @@ +#!/usr/bin/env python3 +####################################################################### +# Copyright (c) 2019-present, Blosc Development Team +# All rights reserved. +# +# SPDX-License-Identifier: BSD-3-Clause +####################################################################### + +"""Create or access a fixed-width, mask-nullable remote CTable.""" + +import argparse +import pprint +import sys +import time +from dataclasses import dataclass +from pathlib import Path + +import numpy as np + +import blosc2 + +DEFAULT_PROFILE = "blosc2" +DEFAULT_ENDPOINT_URL = "https://s3.us-west-001.backblazeb2.com" + + +@dataclass +class Reading: + id: int = blosc2.field(blosc2.int64()) + station_id: int = blosc2.field(blosc2.int32()) + temperature: float = blosc2.field(blosc2.float32(null_storage="mask")) + humidity: int = blosc2.field(blosc2.int16(null_storage="mask")) + status: str = blosc2.field(blosc2.string(max_length=8, null_storage="mask")) + active: bool = blosc2.field(blosc2.bool()) + + +def write_table(args) -> None: + output = args.write + if output.suffix != ".b2z": + raise ValueError("output must end in .b2z") + if args.rows < 1 or args.batch_size < 1: + raise ValueError("--rows and --batch-size must be positive") + if output.exists() and not args.overwrite: + raise FileExistsError(f"{output} already exists; pass --overwrite to replace it") + + output.parent.mkdir(parents=True, exist_ok=True) + rng = np.random.default_rng(42) + statuses = np.array(["ok", "warning", "offline", ""], dtype=object) + attrs = { + "version": 1, + "sampling_interval": 0.5, + "description": "Synthetic weather-station readings", + "example_station_ids": [0, 1, 2, 3], + } + + with blosc2.CTable( + Reading, + urlpath=str(output), + mode="w", + expected_size=args.rows, + validate=False, + create_summary_index=False, + ) as table: + for name, value in attrs.items(): + table.attrs[name] = value + for start in range(0, args.rows, args.batch_size): + stop = min(start + args.batch_size, args.rows) + ids = np.arange(start, stop, dtype=np.int64) + temperature = rng.normal(18, 10, len(ids)).astype(np.float32).astype(object) + humidity = rng.integers(0, 101, len(ids), dtype=np.int16).astype(object) + status = statuses[(ids // 7) % len(statuses)].copy() + temperature[ids % 17 == 0] = None + humidity[ids % 29 == 0] = None + status[ids % 41 == 0] = None + table.extend( + { + "id": ids, + "station_id": (ids % 100).astype(np.int32), + "temperature": temperature, + "humidity": humidity, + "status": status, + "active": ids % 5 != 0, + }, + validate=False, + ) + + with blosc2.CTable.open(str(output)) as table: + assert len(table) == args.rows + assert table.attrs[:] == attrs + null_counts = {name: table[name].null_count() for name in ("temperature", "humidity", "status")} + + print(f"Created {output} ({output.stat().st_size / 1_000_000:.1f} MB, {args.rows:,} rows)") + print(f"Mask-backed null counts: {null_counts}") + print(f"Now upload {output} to your cloud object storage.") + + +def access_table(args) -> None: + storage_options = {} + if args.url.startswith("s3://"): + storage_options = { + "profile": args.profile, + "client_kwargs": {"endpoint_url": args.endpoint_url}, + } + + print(f"Accessing: {args.url}") + started = time.perf_counter() + with blosc2.RemoteCTable(args.url, storage_options=storage_options) as table: + nbytes = table.nbytes + metadata = { + "type": type(table).__name__, + "source": table.source, + "rows": table.nrows, + "columns": table.col_names, + "chunks": table.chunks, + "blocks": table.blocks, + "nbytes": f"{nbytes} ({nbytes / 1024**2:.2f} MiB)", + "cache_policy": table.cache_policy.name, + "cache_bytes": f"{table.cache_bytes} ({table.cache_bytes / 1024:.2f} KiB)", + "schema": table.schema_dict(), + "attrs": dict(table.attrs), + } + metadata_time = time.perf_counter() - started + metadata_bytes = table.traffic.nbytes + metadata_requests = table.traffic.requests + + print("\n[Format: Blosc2 B2Z (Lazy RemoteCTable)]") + for name, value in metadata.items(): + rendered = pprint.pformat(value) if isinstance(value, (dict, list)) else value + if name == "attrs" and value: + rendered = "{\n " + rendered[1:] + print(f"{name:<13}: {rendered}") + + print("\nSample rows (1st fetch):") + started = time.perf_counter() + sample = str(table[:5]) + first_time = time.perf_counter() - started + first_bytes = table.traffic.nbytes - metadata_bytes + first_requests = table.traffic.requests - metadata_requests + print(sample) + + started = time.perf_counter() + _ = str(table[:5]) + second_time = time.perf_counter() - started + second_bytes = table.traffic.nbytes - metadata_bytes - first_bytes + second_requests = table.traffic.requests - metadata_requests - first_requests + + print("\nTiming & Network Traffic:") + print( + f" - Metadata open : {metadata_time * 1000:7.1f} ms " + f"({metadata_requests} requests, {metadata_bytes / 1024:8.2f} KB transferred)" + ) + print( + f" - 1st row fetch : {first_time * 1000:7.1f} ms " + f"({first_requests} requests, {first_bytes / 1024:8.2f} KB transferred)" + ) + cache_hit = " (cache hit!)" if second_requests == second_bytes == 0 else "" + print( + f" - 2nd row fetch : {second_time * 1000:7.1f} ms " + f"({second_requests} requests, {second_bytes / 1024:8.2f} KB transferred){cache_hit}" + ) + print( + f" - Total network : {table.traffic.requests} requests, " + f"{table.traffic.nbytes / 1024:8.2f} KB transferred" + ) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("url", nargs="?", help="Remote .b2z CTable URL") + parser.add_argument("--write", type=Path, metavar="FILE", help="Create a local .b2z CTable instead") + parser.add_argument("--rows", type=int, default=1_000_000) + parser.add_argument("--batch-size", type=int, default=100_000) + parser.add_argument("--overwrite", action="store_true") + parser.add_argument("--profile", default=DEFAULT_PROFILE) + parser.add_argument("--endpoint-url", default=DEFAULT_ENDPOINT_URL) + args = parser.parse_args() + + if args.write is not None and args.url is not None: + parser.error("URL cannot be combined with --write") + if args.write is None and args.url is None: + parser.error("provide a remote URL or use --write FILE.b2z") + + try: + write_table(args) if args.write is not None else access_table(args) + except Exception as exc: + print(f"Error: {exc}", file=sys.stderr) + return 1 + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/examples/dict-store.py b/examples/dict-store.py index 8895c6063..ae2367aeb 100644 --- a/examples/dict-store.py +++ b/examples/dict-store.py @@ -17,7 +17,7 @@ arr_remote = blosc2.open(urlpath, mode="r") dstore["/dir1/node3"] = arr_remote arr_external = blosc2.arange(3, urlpath="external_node3.b2nd", mode="w") - arr_external.vlmeta["description"] = "This is vlmeta for /dir1/node3" + arr_external.attrs["description"] = "This is metadata for /dir1/node3" dstore["/dir2/node4"] = arr_external print("DictStore keys:", list(dstore.keys())) diff --git a/examples/embed-store.py b/examples/embed-store.py index 7a2208211..bd6364f19 100644 --- a/examples/embed-store.py +++ b/examples/embed-store.py @@ -21,7 +21,7 @@ arr_remote = blosc2.open(urlpath, mode="r") estore["/dir1/node3"] = arr_remote arr_external = blosc2.arange(3, urlpath="external_node3.b2nd", mode="w") -arr_external.vlmeta["description"] = "This is vlmeta for /dir1/node4" +arr_external.attrs["description"] = "This is metadata for /dir1/node4" estore["/dir2/node4"] = arr_external print("EmbedStore keys:", list(estore.keys())) diff --git a/examples/embedded-expr-udf-b2z.py b/examples/embedded-expr-udf-b2z.py index 8111dd5ae..230f7b66a 100644 --- a/examples/embedded-expr-udf-b2z.py +++ b/examples/embedded-expr-udf-b2z.py @@ -58,8 +58,8 @@ def masked_energy(a, b, mask): show("Reopened expr type", type(expr).__name__) show("Reopened udf type", type(udf).__name__) - show("Expr operand refs", expr.array.schunk.vlmeta["b2o"]["operands"]) - show("UDF operand refs", udf.array.schunk.vlmeta["b2o"]["operands"]) + show("Expr operand refs", expr.array.schunk.attrs["b2o"]["operands"]) + show("UDF operand refs", udf.array.schunk.attrs["b2o"]["operands"]) show("Expr values", np.round(expr_result[:], 4)) show("UDF values", udf_result[:]) diff --git a/examples/ndarray/proxy-ndarray.py b/examples/ndarray/proxy-ndarray.py index 1f8b35809..eef63e7e9 100644 --- a/examples/ndarray/proxy-ndarray.py +++ b/examples/ndarray/proxy-ndarray.py @@ -49,7 +49,7 @@ print(f"Size 'a' (disk): {os.stat(a.urlpath).st_size}") print(f"Size 'b' (disk): {os.stat(b.urlpath).st_size}") -# Check vlmeta -print("*** VLmeta ***") -print(f"VLmeta in 'a': {list(a.vlmeta)}") -print(f"VLmeta in 'b': {list(b.vlmeta)}") +# Check attrs +print("*** Attrs ***") +print(f"Attrs in 'a': {list(a.attrs)}") +print(f"Attrs in 'b': {list(b.attrs)}") diff --git a/examples/ndarray/swmr-enlarge-bars.py b/examples/ndarray/swmr-enlarge-bars.py index fdd33bc6f..82a60e961 100644 --- a/examples/ndarray/swmr-enlarge-bars.py +++ b/examples/ndarray/swmr-enlarge-bars.py @@ -17,7 +17,7 @@ # # The mechanics are identical to `swmr-enlarge.py` -- read that one # first for the annotated contract (staleness detection, the -# "settled vs. just-resized" gap, the vlmeta "done" signal, and why +# "settled vs. just-resized" gap, the attrs "done" signal, and why # reads are retried). This file adds a `multiprocessing.Queue` purely # to ship progress numbers from the writer/reader processes back to the # one process that owns the terminal and draws the bars; it plays no @@ -82,7 +82,7 @@ def writer(queue): time.sleep(WRITER_DELAY) # Signal readers it is safe to trust and verify the whole array, tail # included -- see swmr-enlarge.py for why this is needed. - arr.schunk.vlmeta["done"] = True + arr.schunk.attrs["done"] = True def reader(rank, queue): @@ -110,7 +110,7 @@ def reader(rank, queue): ) verified = settled queue.put((f"reader{rank}", verified * NCOLS * ITEMSIZE)) - done = "done" in arr.schunk.vlmeta + done = "done" in arr.schunk.attrs except RuntimeError: # A reader can race the writer mid-mutation and hit a transient # read error -- retry on the next poll. diff --git a/examples/ndarray/swmr-enlarge.py b/examples/ndarray/swmr-enlarge.py index d1909c6e3..7ef02da42 100644 --- a/examples/ndarray/swmr-enlarge.py +++ b/examples/ndarray/swmr-enlarge.py @@ -13,7 +13,7 @@ # "SWMR without locking", for the contract this relies on: # # - A reader's cached shape re-syncs the next time it *touches* the -# container -- a data read, a vlmeta lookup, or an explicit +# container -- a data read, an attrs lookup, or an explicit # `NDArray.refresh()` poll that doesn't read any data at all. # - Growth changes the on-disk length, which is what staleness # detection keys off, so it is reliably picked up. @@ -24,7 +24,7 @@ # This example sidesteps that by only treating a batch as safe to # verify once the *next* batch's resize has started (the same # technique used in tests/test_swmr.py::test_cross_process_growth_hammer), -# and by using a vlmeta flag -- itself re-synced on every access -- as +# and by using an attrs flag -- itself re-synced on every access -- as # an explicit "the writer is done" signal for the final tail check. # - A reader racing the writer without locking can still occasionally hit # a transient read error mid-mutation, even in a region considered @@ -63,7 +63,7 @@ def writer(): time.sleep(0.005) # let readers observe growth in more than one step # Set once everything is written: readers use this as the signal that # it is safe to read and verify the whole array, tail included. - arr.schunk.vlmeta["done"] = True + arr.schunk.attrs["done"] = True def reader(rank): @@ -97,9 +97,9 @@ def reader(rank): ) verified = settled - # "done" in vlmeta re-syncs the container's metadata cache too, - # so this also picks up the writer's final vlmeta write. - done = "done" in arr.schunk.vlmeta + # "done" in attrs re-syncs the container's metadata cache too, + # so this also picks up the writer's final attrs write. + done = "done" in arr.schunk.attrs except RuntimeError: # A reader can race the writer mid-mutation and hit a transient # read error even here -- retry on the next poll. diff --git a/examples/ref-object.py b/examples/ref-object.py index ae754b281..b7cf4431f 100644 --- a/examples/ref-object.py +++ b/examples/ref-object.py @@ -56,13 +56,13 @@ def show(label, value): show("ObjectArray round-trip types", [type(ref0).__name__, type(ref1).__name__]) show("ObjectArray round-trip values", [ref0.open()[:], ref1.open()[:]]) -# Refs can also be stored in vlmeta and recovered both individually and in bulk. +# Refs can also be stored in attrs and recovered both individually and in bulk. meta_holder = blosc2.SChunk() -meta_holder.vlmeta["array_ref"] = array_ref -show("vlmeta single-key Ref", meta_holder.vlmeta["array_ref"]) -show("vlmeta single-key values", meta_holder.vlmeta["array_ref"].open()[:]) -show("vlmeta bulk Ref", meta_holder.vlmeta[:]["array_ref"]) -show("vlmeta bulk values", meta_holder.vlmeta[:]["array_ref"].open()[:]) +meta_holder.attrs["array_ref"] = array_ref +show("attrs single-key Ref", meta_holder.attrs["array_ref"]) +show("attrs single-key values", meta_holder.attrs["array_ref"].open()[:]) +show("attrs bulk Ref", meta_holder.attrs[:]["array_ref"]) +show("attrs bulk values", meta_holder.attrs[:]["array_ref"].open()[:]) for path in (array_path, store_path, store_src_path, refs_path): blosc2.remove_urlpath(path) diff --git a/examples/remote/s3-access.py b/examples/remote/s3-access.py index 4894f7dba..0bd871943 100755 --- a/examples/remote/s3-access.py +++ b/examples/remote/s3-access.py @@ -194,9 +194,9 @@ def get_traffic_bytes() -> int | None: meta = getattr(arr, "meta", None) if meta: print(f"{'meta':<12} : {dict(meta)}") - vlmeta = getattr(arr, "vlmeta", None) - if vlmeta is not None: - print(f"{'vlmeta':<12} : {dict(vlmeta) if vlmeta else {}}") + attrs = getattr(arr, "attrs", None) + if attrs is not None: + print(f"{'attrs':<12} : {dict(attrs) if attrs else {}}") print("\nSample slice data (1st fetch):") t0 = time.perf_counter() diff --git a/examples/tree-store-blog.py b/examples/tree-store-blog.py index 85b90787c..1298b35cc 100644 --- a/examples/tree-store-blog.py +++ b/examples/tree-store-blog.py @@ -21,9 +21,9 @@ # Create a group with a dataset that can be a blosc2 NDArray ts["/group1/dataset1"] = blosc2.zeros((10,)) - # You can also store blosc2 arrays directly (vlmeta included) + # You can also store blosc2 arrays directly (attrs included) ext = blosc2.linspace(0, 1, 10_000, dtype=np.float32) - ext.vlmeta["desc"] = "dataset2 metadata" + ext.attrs["desc"] = "dataset2 metadata" ts["/group1/dataset2"] = ext print("Created 'my_experiment.b2z' with initial data.\n") @@ -39,27 +39,27 @@ # Access the external array that has been stored internally dataset2 = ts["/group1/dataset2"] print("Dataset 2", dataset2[:]) - print("Dataset 2 metadata:", dataset2.vlmeta[:]) + print("Dataset 2 metadata:", dataset2.attrs[:]) # List all paths in the store print("Paths in TreeStore:", list(ts)) print() -# --- 3. Storing Metadata with `vlmeta` --- -print("--- 3. Storing Metadata with `vlmeta` ---") +# --- 3. Storing Metadata with `attrs` --- +print("--- 3. Storing Metadata with `attrs` ---") with blosc2.TreeStore("my_experiment.b2z", mode="a") as ts: # 'a' for append/modify # Add metadata to the root - ts.vlmeta["author"] = "The Blosc Team" - ts.vlmeta["date"] = "2025-08-17" + ts.attrs["author"] = "The Blosc Team" + ts.attrs["date"] = "2025-08-17" # Add metadata to a group - ts["/group1"].vlmeta["description"] = "Data from the first run" + ts["/group1"].attrs["description"] = "Data from the first run" # Reading metadata with blosc2.TreeStore("my_experiment.b2z", mode="r") as ts: - print("Root metadata:", ts.vlmeta[:]) - print("Group 1 metadata:", ts["/group1"].vlmeta[:]) + print("Root metadata:", ts.attrs[:]) + print("Group 1 metadata:", ts["/group1"].attrs[:]) print() @@ -85,7 +85,7 @@ if isinstance(node, blosc2.NDArray): print(f"Found dataset at '{path}' with shape {node.shape}") else: # It's a group - print(f"Found group at '{path}' with metadata: {node.vlmeta[:]}") + print(f"Found group at '{path}' with metadata: {node.attrs[:]}") print() # --- Cleanup --- diff --git a/examples/tree-store.py b/examples/tree-store.py index e65aac3d1..c6b6f3a8d 100644 --- a/examples/tree-store.py +++ b/examples/tree-store.py @@ -5,7 +5,7 @@ # SPDX-License-Identifier: BSD-3-Clause ####################################################################### -# Example usage of TreeStore with hierarchical navigation, vlmeta, and CTables +# Example usage of TreeStore with hierarchical navigation, attrs, and CTables from dataclasses import dataclass @@ -22,7 +22,7 @@ # External arrays can also be included ext = blosc2.linspace(0, 1, 5, urlpath="external_leaf.b2nd", mode="w") - ext.vlmeta["desc"] = "external /dir1/node3 metadata" # NDArray-level metadata + ext.attrs["desc"] = "external /dir1/node3 metadata" # NDArray-level metadata tstore["/dir1/node3"] = ext # Remote array (read-only), referenced via URLPath @@ -31,17 +31,17 @@ tstore["/dir2/remote"] = arr_remote # TreeStore-level metadata (persists with the store) - tstore.vlmeta["author"] = "blosc2" - tstore.vlmeta["version"] = 1 - tstore.vlmeta[:] = {"purpose": "TreeStore example", "scale": 2.5} + tstore.attrs["author"] = "blosc2" + tstore.attrs["version"] = 1 + tstore.attrs[:] = {"purpose": "TreeStore example", "scale": 2.5} print("TreeStore keys:", sorted(tstore.keys())) print("/child0/data:", tstore["/child0/data"][:]) print("/dir1/node3 (external) first 3:", tstore["/dir1/node3"][:3]) print("/dir2/remote first 3:", tstore["/dir2/remote"][:3]) - print("Stored vlmeta:", tstore.vlmeta[:]) + print("Stored attrs:", tstore.attrs[:]) node3 = tstore["/dir1/node3"] - print("Node '/dir1/node3' vlmeta.desc:", node3.vlmeta["desc"]) # NDArray metadata + print("Node '/dir1/node3' attrs.desc:", node3.attrs["desc"]) # NDArray metadata # Access a subtree view rooted at /child0 root = tstore["/child0"] # or tstore["/child0"] diff --git a/examples/vlmeta.py b/examples/vlmeta.py index 91eb2d5c4..9bd188928 100644 --- a/examples/vlmeta.py +++ b/examples/vlmeta.py @@ -16,19 +16,19 @@ nchunks_ = schunk.append_data(buffer) assert nchunks_ == (i + 1) -# Initially the vlmeta is empty +# Initially attrs is empty print(len(schunk.attrs)) -# Add a vlmeta -schunk.attrs["meta1"] = "first vlmetalayer" +# Add an attribute +schunk.attrs["meta1"] = "first metadata value" print(schunk.attrs.getall()) -# Update the vlmeta -schunk.attrs["meta1"] = "new vlmetalayer" +# Update the attribute +schunk.attrs["meta1"] = "new metadata value" print(schunk.attrs.getall()) -# Add another vlmeta -schunk.attrs["vlmeta2"] = "second vlmeta" +# Add another attribute +schunk.attrs["meta2"] = "second metadata value" # Check that it has been added -assert "vlmeta2" in schunk.attrs +assert "meta2" in schunk.attrs -# Delete a vlmeta -del schunk.attrs["vlmeta2"] -assert "vlmeta2" not in schunk.attrs +# Delete an attribute +del schunk.attrs["meta2"] +assert "meta2" not in schunk.attrs diff --git a/plans/remote-ctable.md b/plans/remote-ctable.md new file mode 100644 index 000000000..a0369ce33 --- /dev/null +++ b/plans/remote-ctable.md @@ -0,0 +1,261 @@ +# RemoteCTable implementation plan + +Status: initial fixed-width, read-only implementation completed on 2026-09-17; +UTF-8, batch-backed columns, persisted indexes and portable references remain follow-ups. + +## Objective and architecture + +Add `blosc2.RemoteCTable` for read-only, on-demand access to CTable objects in +immutable remote `.b2z` archives through fsspec. Keep RemoteArray, RemoteStore and +RemoteCTable as distinct public objects: arrays, hierarchies and tables have +different interfaces. Do not introduce a public RemoteSource or automatic URL +factory in this work. + +Use a thin CTable subclass and a read-only `RemoteTableStorage` backend. Reuse +`CTable._open_from_storage()`, schema reconstruction, lazy column opening and +existing table query logic. Share the existing remote discovery owner, B2Z range +reader, RemoteArray leaves, cache coordinator and traffic accounting. No new +dependency or on-disk table format is planned. + +The ponytail principle for this implementation is to adapt the existing storage +boundary and repair shared read paths, without copying table algorithms into a +second implementation or building a general remote-object framework. + +## Implemented in this branch + +- Added the public `blosc2.RemoteCTable` class and exported it from the package. + It accepts `dataset`, `storage_options`, `cache_policy`, `max_cache_bytes` and + `cache_dir`, and exposes the shared source, traffic and cache metrics. +- Added `RemoteTableStorage`, backed by the existing remote discovery owner and + `RemoteArray` leaves. It opens table schema, fixed-width columns, row validity, + nullable-column masks and user `attrs` lazily and rejects mutation. +- Extended B2Z discovery and `RemoteStore` so CTable roots and nested CTable + nodes are recognized, remain opaque during hierarchy traversal, and open as + `RemoteCTable` objects. +- Added slice-based shared CTable fallbacks needed by remote operands, preserving + local optimized paths. Scalar and sliced rows, iteration, filtering, + reductions, fixed-width strings and mask-backed nulls work remotely. +- Kept persisted indexes disabled for remote tables; queries scan the required + columns rather than entering local sidecar paths. +- Fixed owned `s3fs` cleanup so its registered finalizer closes the aiobotocore + session exactly once, while HTTP fsspec sessions retain deterministic closing. +- Added `examples/ctable/remote_handling.py`. `--write FILE.b2z` generates a + one-million-row fixed-width table with three null masks and representative + integer, float, string and list-valued attributes. Passing a remote URL opens + it, reports schema and attributes, reads sample rows twice, and reports time, + requests and transferred bytes for metadata, cold and warm access. +- Updated Python examples to promote the public `attrs` alias instead of + `vlmeta`; compatibility tests continue to cover the older name. + +## Evidence from the current implementation + +- `ctable.py`: `_open_from_storage()` reconstructs a table without the original + Python row class; `_LazyColumnDict` opens physical columns on demand. +- `ctable_storage.py`: fixed-width columns, `_valid_rows` and null masks are + NDArrays. The schema and table metadata live in `_meta` SChunk vlmeta. +- `b2z_source.py`: `B2ZNDSource` reads external, unencrypted ZIP_STORED `.b2nd` + members. `member_vlmeta()` and `B2ZEmbeddedMetadata` provide metadata reads. +- `remote_store.py`: discovery already recognizes CTable roots and nested table + boundaries, but deliberately marks them unsupported and hides their internals. +- Persisted index descriptors are currently resolved through local paths and + ZIP-offset registration. Batch-backed columns use local `.b2b` opening paths. + +A disposable probe in the `blosc2` environment created a numeric CTable archive, +uploaded it to fsspec `memory://`, and opened it through a minimal storage adapter +returning RemoteArray leaves. Row count, column slicing and a `where()` query +worked. Scalar access and `sum()` failed because CTable calls +`iterchunks_info()`, which RemoteArray does not provide. This is evidence of +integration feasibility, not complete correctness or network performance. + +## Decisions resolved for the first release + +### Public API + +Available usage: + +```python +with blosc2.RemoteCTable("https://example.org/measurements.b2z") as table: + values = table["temperature"][:10] + selected = table.where(table["temperature"] > 20) + values = selected["temperature"][:] + +with blosc2.RemoteCTable( + "s3://bucket/archive.b2z", + dataset="experiments/run1", + storage_options={...}, + cache_dir="table-cache", +) as table: + values = table["temperature"][:10] + +with blosc2.RemoteStore("s3://bucket/archive.b2z") as store: + with store["experiments/run1"] as table: + values = table["temperature"][:10] +``` + +- Constructor keywords: `dataset`, `storage_options`, `cache_policy`, + `max_cache_bytes`, `cache_dir`. Reuse existing container URL parsing and + validation; no separate source-format or writable-mode option. +- Match RemoteStore defaults: bounded MEMORY caching without `cache_dir`, DISK + when `cache_dir` is supplied, and explicit NONE support. Reuse its validation + and default limit rather than duplicate constants. No per-column cache budget. +- Expose `source`, `traffic`, `cache_policy`, `max_cache_bytes`, `cache_bytes` + and `metadata_bytes` using owner accounting. Document that metrics are shared + with sibling objects when obtained through a RemoteStore. +- `RemoteStore.kind()`/`get_info()` gain a `ctable` kind; indexing a supported + table node returns RemoteCTable. Preserve the table as an opaque hierarchy + node: `_cols`, `_meta` and index files are not ordinary public children. +- A standalone table archive is opened with RemoteCTable. RemoteStore retains + its group-root requirement and gives a diagnostic directing users to it. +- No automatic change to `blosc2.open()` or `CTable.open()` URL dispatch. + +### Supported data and operations + +- First-release data: external fixed-width NDArray columns, including supported + fixed-width strings, timestamps, booleans and per-row NDArray values; row + validity and nullable-column masks; nested schema paths whose leaves qualify. +- First-release behavior: metadata and schema inspection, logical row count, + scalar/slice/fancy reads, row iteration, column iteration, filtering and + reductions that local CTable supports for those types. Preserve deleted-row, + spare-capacity and null semantics, including on filtered views. +- Computed columns may reuse the existing safe expression machinery when all + dependencies are supported. Do not execute arbitrary serialized callables or + import code named by remote metadata. Give a precise error for unsupported + expression forms or dependencies. +- Unsupported columns do not prevent schema inspection or reading supported + siblings. Opening such a column raises `NotImplementedError` naming its path, + storage/type and limitation. Whole-table operations requiring it also fail + explicitly; never silently omit a column or download the whole archive. +- UTF-8 offsets/data arrays are a follow-up, despite fitting RemoteArray well; + their wrapper and query paths need their own compatibility checks. Lists, + other batch-backed variable-length values and dictionary value stores are + also deferred. Dictionary codes alone do not constitute dictionary support. +- Embedded array payloads, encrypted members and ZIP-compressed members retain + existing remote-reader limitations. Reuse supported embedded metadata access + where possible; reject unsupported manifests explicitly. + +### Queries, writes and persistence + +- Initial queries scan required columns. The remote storage backend returns no + usable index catalog, so persisted indexes cannot accidentally enter local + sidecar opening paths. Document this even when the source contains indexes. +- Preserve local optimized paths. Where CTable assumes native chunk metadata, + add a bounded slice-based fallback in the shared helper after tracing callers. + Do not fabricate special-chunk metadata or make ordinary reads depend on a + local cache SChunk: NONE must work too. +- Opening must not read all column payloads or scan the validity mask when the + saved row count is available. Missing row counts and deleted-row position + resolution may require bounded mask scans; document their cost. +- Reject data/schema mutation, index creation and remote metadata writes before + changing any state. Source read-only status is separate from cache mutability. + Audit inherited methods and raw column/metadata access, not only assignment. +- Materializing supported data into a new local CTable through existing copy or + save operations is allowed and must produce an ordinary local table. It must + not accidentally construct a RemoteCTable with missing owner state. +- Remote-reference artifact serialization, automatic b2object registration, + sparse-cache convenience APIs and remote writes are out of scope. Distinguish + local data export from saving a portable remote reference. +- RemoteStore artifact export encountering a table must fail explicitly until + table-reference serialization is implemented; never silently drop that node. + +### Lifetime and source identity + +- Acquire/release the existing shared discovery owner. Closing a RemoteStore + leaves an already returned RemoteCTable usable. Closing a table releases its + resources; independent raw RemoteArray handles keep their own owner leases. +- Table Column objects and table views are borrowed from their root table and + require it to remain open. Close on a view must not close the root. All access + after root close must fail consistently, including opening an untouched column. +- Audit CTable view/copy constructors, some of which use `cls.__new__` and others + `CTable.__new__`. Read-only views may be ordinary CTable objects, provided root + ownership and generation checks are enforced; detached copies are local CTable. +- The remote archive is immutable for the session. Reopen a directly constructed + RemoteCTable to observe replacement; no table-level `refresh()` in version one. +- Existing RemoteStore root `refresh()` invalidates previously returned tables, + their views and columns, including metadata-only operations. Preserve atomic + refresh failure behavior and check generations before opening new leaves. +- Cache identity includes the archive identity, table/leaf path and existing + storage-options fingerprint. Do not persist credentials in source descriptors. + +## Implementation sequence + +1. Extend B2Z discovery with table descriptors and internal member lookup. + Validate table kind, supported schema version, paths and required metadata. + Reuse archive/metadata readers; retain table boundaries and existing limits. + Update cached discovery manifest validation for the new node kind, with safe + rejection or rediscovery of incompatible cached metadata. + +2. Add `RemoteTableStorage` in `ctable_storage.py` and a thin public class in + `remote_ctable.py`, exported from `__init__.py`. Implement schema, validity, + null-mask and column opening, epoch reads, empty index catalog, read-only + metadata and cleanup. Construct through `_open_from_storage()`. Extract only + the owner-creation code needed to avoid duplicating RemoteStore initialization. + +3. Repair shared CTable reads for remote operands. Start with + `_find_physical_index()` and `Column.iter_chunks()`, then trace null handling, + reductions, fancy indexing, expressions and materialization. Audit NDArray + type checks, `.schunk`, `.urlpath`, native-extension calls and local sidecar + opening. Reuse slice-based traversal where it already exists. Keep local + special-chunk optimizations and add no broad array protocol refactor. + +4. Integrate `RemoteStore[table_path]`, shared cache accounting, close behavior, + generation invalidation and root diagnostics. Ensure view construction and + local copy/save behavior follow the decisions above. Guard unsupported remote + reference export paths. + +5. Add focused regression coverage and API documentation. Describe supported + columns, borrowed views, immutable sources, scan costs, cache accounting and + unsupported indexes. Add one small local/remote usage example. No C/Cython + changes are anticipated; establish a concrete need before expanding scope. + +## Verification and acceptance + +Use the `blosc2` conda environment for all Python and test commands. Add tests +under `tests/ctable/` and extend `tests/test_remote_store.py` as appropriate. + +- Compare remote results with a local read-only CTable from the exact same + archive: empty and nonempty tables, chunk edges, deleted rows, spare capacity, + masks/sentinels, scalar/fancy/strided reads, filtering and reductions. Include + fixed-width string, timestamp, nested and NDArray-valued columns. +- Cover both standalone and nested tables, unsupported-column isolation, + malformed schema/metadata and missing members. Validate read-only enforcement + and that local exports have the intended class and values. +- Exercise NONE, MEMORY and DISK using an instrumented fsspec memory filesystem. + Check lazy opening, bounded range requests, reuse of warmed chunks, aggregate + cache limits and disk reopen. Account for existing bounded small-member + prefetch; do not assert that no payload byte ever arrives with metadata. +- Use a deterministic local HTTP range server to verify actual range transport + without cloud credentials. Keep optional external-service tests network-marked. +- Test parent-store close, root-table close, borrowed views, raw-array leases, + refresh invalidation and failed refresh. Ensure cleanup on partial open failure. +- Re-run relevant existing CTable, remote array/store and B2Z tests, followed by + the default suite before declaring implementation complete. Warnings are errors. +- Record a small cold/warm read experiment on an archive much larger than the + requested slice: report transferred bytes, requests and retained payload. +No universal latency target; the acceptance criterion is bounded on-demand I/O +and correct results, not a claim that scan queries avoid reading their operands. + +### Results recorded for the initial implementation + +- The default suite passed with 10,209 tests and 36 skips after the core + implementation. Ruff passed for the changed source and test files. +- The focused RemoteStore and RemoteCTable suites passed with 157 tests and five + skips after the S3 lifecycle fix. +- The generated example archive round-tripped through local CTable and fsspec + `memory://`, preserving all three null masks and all four user attributes. +- A one-million-row `readings.b2z` uploaded to Backblaze B2 was opened from + `s3://blosc2/readings.b2z` through its S3-compatible endpoint. Metadata and + sample rows were read on demand; the repeated sample read issued zero requests + and transferred zero bytes from the warm memory cache. + +## Follow-ups, separately scoped + +1. UTF-8 columns through remote offsets and bytes, with null/query/size reporting. +2. Remote batch reads for lists, variable-length values and dictionary stores. +3. Persisted indexes through a remote-aware sidecar resolver, starting with + SUMMARY indexes and measuring query transfer savings. +4. Portable RemoteCTable references, RemoteStore artifact inclusion and sparse + runtime-cache APIs if needed by actual consumers. + +There are no blocking API questions left for the initial fixed-width scope. The +remaining items above are deliberately separate extensions rather than blockers +for the implemented read-only API. diff --git a/src/blosc2/__init__.py b/src/blosc2/__init__.py index d80b5fb8e..f299c93af 100644 --- a/src/blosc2/__init__.py +++ b/src/blosc2/__init__.py @@ -780,6 +780,7 @@ def _raise(exc): var, where, ) +from .remote_ctable import RemoteCTable from .schema import ( DictionarySpec, NDArraySpec, @@ -915,6 +916,7 @@ def _raise(exc): "Ref", "RemoteMetadataMapping", "RemoteArray", + "RemoteCTable", "RemoteNode", "RemoteStore", "SChunk", diff --git a/src/blosc2/ctable.py b/src/blosc2/ctable.py index 2ffaa67bd..7abfad227 100644 --- a/src/blosc2/ctable.py +++ b/src/blosc2/ctable.py @@ -679,6 +679,32 @@ def rows_for_ranks(rank_lo, rank_hi) -> np.ndarray: return rows_for_ranks +def _iter_true_segments(arr): + """Yield ``(start, size, mask)``; ``mask is None`` means all rows are true.""" + chunk_size = arr.chunks[0] + iter_info = getattr(arr, "iterchunks_info", None) + if iter_info is None: + for start in range(0, arr.shape[0], chunk_size): + size = min(chunk_size, arr.shape[0] - start) + mask = np.asarray(arr[start : start + size], dtype=np.bool_) + if np.any(mask): + yield start, size, None if np.all(mask) else mask + return + + for info in iter_info(): + size = min(chunk_size, arr.shape[0] - info.nchunk * chunk_size) + start = info.nchunk * chunk_size + if info.special == blosc2.SpecialValue.ZERO: + continue + if info.special == blosc2.SpecialValue.VALUE: + if np.frombuffer(info.repeated_value, dtype=arr.dtype)[0]: + yield start, size, None + continue + mask = np.asarray(arr[start : start + size], dtype=np.bool_) + if np.any(mask): + yield start, size, None if np.all(mask) else mask + + def _find_physical_index(arr: blosc2.NDArray, logical_key: int) -> int: """Translate a logical (valid-row) index into a physical array index. @@ -696,31 +722,14 @@ def _find_physical_index(arr: blosc2.NDArray, logical_key: int) -> int: If the logical index is out of range or the array is inconsistent. """ count = 0 - chunk_size = arr.chunks[0] - - for info in arr.iterchunks_info(): - actual_size = min(chunk_size, arr.shape[0] - info.nchunk * chunk_size) - chunk_start = info.nchunk * chunk_size - - if info.special == blosc2.SpecialValue.ZERO: - continue - - if info.special == blosc2.SpecialValue.VALUE: - val = np.frombuffer(info.repeated_value, dtype=arr.dtype)[0] - if not val: - continue - if count + actual_size <= logical_key: - count += actual_size - continue - return chunk_start + (logical_key - count) - - chunk_data = arr[chunk_start : chunk_start + actual_size] - n_true = int(np.count_nonzero(chunk_data)) + for chunk_start, actual_size, mask in _iter_true_segments(arr): + n_true = actual_size if mask is None else int(np.count_nonzero(mask)) if count + n_true <= logical_key: count += n_true continue - - return chunk_start + int(np.flatnonzero(chunk_data)[logical_key - count]) + if mask is None: + return chunk_start + logical_key - count + return chunk_start + int(np.flatnonzero(mask)[logical_key - count]) raise IndexError("Unexpected error finding physical index.") @@ -1915,26 +1924,9 @@ def __iter__(self): if self.is_list or self.is_varlen_scalar: yield from self._raw_col[np.where(self._valid_rows[:])[0]] return - arr = self._valid_rows - chunk_size = arr.chunks[0] - - for info in arr.iterchunks_info(): - actual_size = min(chunk_size, arr.shape[0] - info.nchunk * chunk_size) - chunk_start = info.nchunk * chunk_size - - if info.special == blosc2.SpecialValue.ZERO: - continue - - if info.special == blosc2.SpecialValue.VALUE: - val = np.frombuffer(info.repeated_value, dtype=arr.dtype)[0] - if not val: - continue - yield from self._raw_col[chunk_start : chunk_start + actual_size] - continue - - mask_chunk = arr[chunk_start : chunk_start + actual_size] + for chunk_start, actual_size, mask_chunk in _iter_true_segments(self._valid_rows): data_chunk = self._raw_col[chunk_start : chunk_start + actual_size] - yield from data_chunk[mask_chunk] + yield from data_chunk if mask_chunk is None else data_chunk[mask_chunk] @staticmethod def _format_array_value(value) -> str: @@ -3004,26 +2996,14 @@ def iter_chunks(self, size: int = 65536): raise TypeError("Column.iter_chunks() is not supported for varlen scalar columns.") valid = self._valid_rows raw = self._raw_col - arr_len = len(valid) - phys_chunk = valid.chunks[0] pending: list[np.ndarray] = [] pending_count = 0 - for info in valid.iterchunks_info(): - actual = min(phys_chunk, arr_len - info.nchunk * phys_chunk) - start = info.nchunk * phys_chunk - - if info.special == blosc2.SpecialValue.ZERO: - continue - - if info.special == blosc2.SpecialValue.VALUE: - val = np.frombuffer(info.repeated_value, dtype=valid.dtype)[0] - if not val: - continue + for start, actual, mask in _iter_true_segments(valid): + if mask is None: segment = raw[start : start + actual] else: - mask = valid[start : start + actual] data_part = raw[start : start + actual] if len(data_part) < actual: # Logically-sized storage (utf8) is shorter than the @@ -7097,7 +7077,7 @@ def _open_from_storage(cls, storage: TableStorage) -> CTable: schema = schema_from_dict(schema_dict) col_names = [c["name"] for c in schema_dict["columns"]] - obj = cls.__new__(cls) + obj = object.__new__(cls) obj._row_type = None obj._validate = True obj._table_cparams = None diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index c8813f1e0..774356541 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -8,13 +8,15 @@ """Storage backends for CTable. -Two concrete backends: +The main concrete backends are: * :class:`InMemoryTableStorage` — all arrays live in RAM (default when ``urlpath`` is not provided). * :class:`FileTableStorage` — arrays are stored inside a :class:`blosc2.TreeStore` rooted at ``urlpath``; logical object metadata lives in ``/_meta`` and table data lives under ``/_valid_rows`` and ``/_cols/``. +* :class:`RemoteTableStorage` — fixed-width arrays are opened lazily as + :class:`blosc2.RemoteArray` objects from a remote B2Z archive. """ from __future__ import annotations @@ -642,6 +644,159 @@ def index_anchor_path(self, col_name: str) -> str | None: return None +class RemoteTableStorage(TableStorage): + """Read-only fixed-width CTable storage over a shared RemoteStore owner.""" + + def __init__(self, owner, root_key: str) -> None: + self._owner = owner + self._root_key = root_key.strip("/") + self._generation = owner.generation + self._arrays: list[blosc2.RemoteArray] = [] + self._closed = False + owner.acquire() + + def _check_open(self) -> None: + if self._closed: + raise RuntimeError("RemoteCTable handle is closed") + if self._generation != self._owner.generation: + raise RuntimeError("RemoteCTable handle is stale; look it up again after refresh") + + def _full_key(self, logical_key: str) -> str: + return "/".join(part for part in (self._root_key, logical_key.strip("/")) if part) + + def _open_array(self, logical_key: str) -> blosc2.RemoteArray: + with self._owner.lock: + self._check_open() + full = self._full_key(logical_key) + self._owner.open_ctable_array(self._root_key, logical_key) + array = self._owner.remote_array(full) + self._arrays.append(array) + return array + + def _metadata(self) -> dict: + self._check_open() + try: + kind, metadata = self._owner.nodes[self._root_key] + except KeyError as exc: + raise RuntimeError("RemoteCTable source is unavailable") from exc + if kind != "ctable" or not isinstance(metadata, dict): + raise ValueError(f"Object at {self._root_key!r} is not a CTable") + return metadata + + def _has_array(self, logical_key: str) -> bool: + self._check_open() + member = self._full_key(logical_key) + ".b2nd" + return sum(info.filename == member for info in self._owner.archive.members) == 1 + + @staticmethod + def _not_supported(*args, **kwargs): + raise RuntimeError("RemoteTableStorage is read-only") + + def open_column(self, name: str) -> blosc2.RemoteArray: + return self._open_array(f"{_COLS_DIR}/{_column_name_to_relpath(name)}") + + def open_list_column(self, name: str) -> ListArray: + raise NotImplementedError(f"Remote CTable list column {name!r} is not supported") + + def open_varlen_scalar_column(self, name: str, spec) -> _ScalarVarLenArray: + raise NotImplementedError( + f"Remote CTable variable-length column {name!r} ({type(spec).__name__}) is not supported" + ) + + def open_dictionary_column(self, name: str, spec) -> DictionaryColumn: + raise NotImplementedError(f"Remote CTable dictionary column {name!r} is not supported") + + def open_valid_rows(self) -> blosc2.RemoteArray: + return self._open_array("_valid_rows") + + def open_null_mask(self, name: str) -> blosc2.RemoteArray: + return self._open_array(f"{_COLS_DIR}/{_column_name_to_relpath(name)}{_NOTNULL_SUFFIX}") + + def has_null_mask(self, name: str) -> bool: + return self._has_array(f"{_COLS_DIR}/{_column_name_to_relpath(name)}{_NOTNULL_SUFFIX}") + + def check_kind(self) -> None: + kind = self._metadata().get("kind") + if isinstance(kind, bytes): + kind = kind.decode() + if kind != "ctable": + raise ValueError(f"Object at {self._root_key!r} is not a CTable (kind={kind!r})") + + def load_schema(self) -> dict[str, Any]: + raw = self._metadata().get("schema") + if isinstance(raw, bytes): + raw = raw.decode() + if not isinstance(raw, str): + raise ValueError(f"Remote CTable at {self._root_key!r} has no schema") + return json.loads(raw) + + def load_user_attrs(self) -> dict: + self._check_open() + member = self._full_key("_vlmeta") + ".b2f" + matches = [info for info in self._owner.archive.members if info.filename == member] + if not matches: + return {} + if len(matches) != 1: + raise ValueError(f"Duplicate Remote CTable metadata member {member!r}") + from blosc2.b2z_source import member_vlmeta + + return dict(member_vlmeta(self._owner.archive, matches[0])) + + def table_exists(self) -> bool: + try: + return self._metadata().get("kind") in {"ctable", b"ctable"} + except (RuntimeError, ValueError): + return False + + def is_read_only(self) -> bool: + return True + + def open_mode(self) -> str: + return "r" + + def close(self) -> None: + if self._closed: + return + self._closed = True + for array in self._arrays: + array.close() + self._arrays.clear() + self._owner.release() + + discard = close + + create_column = _not_supported + install_column = _not_supported + create_list_column = _not_supported + install_list_column = _not_supported + create_varlen_scalar_column = _not_supported + create_dictionary_column = _not_supported + create_valid_rows = _not_supported + create_null_mask = _not_supported + install_null_mask = _not_supported + delete_null_mask = _not_supported + save_schema = _not_supported + save_vlmeta = _not_supported + delete_column = _not_supported + rename_column = _not_supported + + def load_index_catalog(self) -> dict: + self._check_open() + return {} + + save_index_catalog = _not_supported + + def get_epoch_counters(self) -> tuple[int, int]: + metadata = self._metadata() + return int(metadata.get("value_epoch", 0) or 0), int(metadata.get("visibility_epoch", 0) or 0) + + bump_value_epoch = _not_supported + bump_visibility_epoch = _not_supported + + def index_anchor_path(self, col_name: str) -> None: + return None + + class FileTableStorage(TableStorage): """Arrays stored as TreeStore leaves inside *urlpath*. diff --git a/src/blosc2/remote_ctable.py b/src/blosc2/remote_ctable.py new file mode 100644 index 000000000..e4b6edafe --- /dev/null +++ b/src/blosc2/remote_ctable.py @@ -0,0 +1,116 @@ +####################################################################### +# Copyright (c) 2019-present, Blosc Development Team +# All rights reserved. +# +# SPDX-License-Identifier: BSD-3-Clause +####################################################################### +"""Read-only remote CTable access through fsspec-backed B2Z sources.""" + +from __future__ import annotations + +from blosc2.ctable import CTable +from blosc2.ctable_storage import RemoteTableStorage +from blosc2.remote_array import CACHE_POLICY_DEFAULT, RemoteMetadataMapping + + +class RemoteCTable(CTable): + """A read-only CTable whose fixed-width columns are fetched on demand.""" + + def __new__( + cls, + urlpath=None, + *, + dataset=None, + storage_options=None, + cache_policy=CACHE_POLICY_DEFAULT, + max_cache_bytes=CACHE_POLICY_DEFAULT, + cache_dir=None, + _filesystem=None, + ): + if urlpath is None: + raise TypeError("RemoteCTable requires a remote B2Z URL") + + from blosc2.remote_store import RemoteStore + + store = RemoteStore( + urlpath, + dataset=dataset, + storage_options=storage_options, + cache_policy=cache_policy, + max_cache_bytes=max_cache_bytes, + cache_dir=cache_dir, + _allow_array_root=True, + _filesystem=_filesystem, + ) + try: + _, full = store._resolve("") + kind, diagnostic = store._owner.nodes[full] + if kind != "ctable": + if kind == "unsupported": + raise NotImplementedError(str(diagnostic)) + raise ValueError("RemoteCTable requires a CTable node") + return cls._from_owner(store._owner, full) + finally: + store.close() + + def __init__(self, *args, **kwargs): + # Construction is completed by CTable._open_from_storage() in __new__. + pass + + @classmethod + def _from_owner(cls, owner, full_path): + storage = RemoteTableStorage(owner, full_path) + try: + return cls._open_from_storage(storage) + except BaseException: + storage.close() + raise + + def _remote_storage(self) -> RemoteTableStorage: + storage = getattr(self, "_storage", None) + if not isinstance(storage, RemoteTableStorage): + raise RuntimeError("RemoteCTable handle is closed") + storage._check_open() + return storage + + def close(self) -> None: + storage = getattr(self, "_storage", None) + if isinstance(storage, RemoteTableStorage): + storage.close() + + @property + def vlmeta(self): + return RemoteMetadataMapping(self._remote_storage().load_user_attrs()) + + @property + def source(self): + storage = self._remote_storage() + return { + "kind": "b2z", + "version": 1, + "urlpath": storage._owner.urlpath, + "dataset": storage._root_key, + "assume_immutable": True, + } + + @property + def traffic(self): + return self._remote_storage()._owner.traffic + + @property + def cache_policy(self): + return self._remote_storage()._owner.cache_policy + + @property + def max_cache_bytes(self): + return self._remote_storage()._owner.max_cache_bytes + + @property + def cache_bytes(self): + return self._remote_storage()._owner.cache_coordinator.cache_bytes + + @property + def metadata_bytes(self): + storage = self._remote_storage() + storage._owner.save_manifest() + return storage._owner.metadata_bytes diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 3aca6502f..b93ff72f1 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -141,7 +141,7 @@ def _restore_manifest(self, manifest): if ( not isinstance(entry, (list, tuple)) or len(entry) != 2 - or entry[0] not in {"group", "ndarray", "unsupported"} + or entry[0] not in {"group", "ndarray", "ctable", "unsupported"} ): raise ValueError("Invalid RemoteStore node") self.nodes[path] = tuple(entry) @@ -191,7 +191,7 @@ def save_manifest(self): elif self.format == "b2z": self.metadata = self.archive.metadata nodes = { - path: (kind, value if kind == "unsupported" else None) + path: (kind, value if kind in {"ctable", "unsupported"} else None) for path, (kind, value) in self.nodes.items() } manifest = { @@ -245,23 +245,28 @@ def _check_node_limit(self): def _find_b2z_ctable_roots(self, members, embedded, registry): from blosc2.b2z_source import B2ZEmbeddedMetadata, member_vlmeta - roots = {key.strip("/") for key in registry} + roots = { + key.strip("/"): None + for key, value in registry.items() + if isinstance(value, dict) and value.get("kind") == "ctable" + } embedded_reader = None # Match TreeStore's legacy CTable manifest check using metadata only. for name, info in members.items(): - if (name == "_meta.b2f" or name.endswith("/_meta.b2f")) and member_vlmeta( - self.archive, info - ).get("kind") in {"ctable", b"ctable"}: - roots.add(name.rpartition("/")[0]) + if name == "_meta.b2f" or name.endswith("/_meta.b2f"): + metadata = member_vlmeta(self.archive, info) + if metadata.get("kind") in {"ctable", b"ctable"}: + roots[name.rpartition("/")[0]] = metadata for key, entry in embedded.items(): if key.endswith("/_meta"): try: if embedded_reader is None: embedded_reader = B2ZEmbeddedMetadata(self.archive, members["embed.b2e"]) - if embedded_reader.attrs(entry).get("kind") in {"ctable", b"ctable"}: - roots.add(key.rpartition("/")[0].strip("/")) + attrs = embedded_reader.attrs(entry) + if attrs.get("kind") in {"ctable", b"ctable"}: + roots[key.rpartition("/")[0].strip("/")] = attrs except NotImplementedError as exc: - roots.add(key.rpartition("/")[0].strip("/")) + roots.setdefault(key.rpartition("/")[0].strip("/"), None) self.notice = f"Partial B2Z metadata: object boundary cannot be verified: {exc}." return roots, embedded_reader @@ -347,12 +352,23 @@ def _open_b2z(self): roots, embedded_reader = self._find_b2z_ctable_roots(members, embedded, registry) if "" in roots: - self.nodes[""] = ("unsupported", "CTable access is unavailable for remote B2Z hierarchies") + metadata = roots[""] + self.nodes[""] = ( + ("ctable", metadata) + if metadata is not None + else ("unsupported", "CTable metadata is unavailable") + ) + self.archive._opening_ranges.clear() + self.archive.capture_metadata = False return self._add("", "group") for root in sorted(roots, key=len): if not any(root.startswith(other + "/") for other in roots if other != root): - self._add(root, "unsupported", "CTable access is unavailable for remote B2Z hierarchies") + metadata = roots[root] + if metadata is None: + self._add(root, "unsupported", "CTable metadata is unavailable") + else: + self._add(root, "ctable", metadata) self._process_b2z_members(members, roots) self._process_b2z_embedded(embedded, roots, embedded_reader, members) self.archive._opening_ranges.clear() @@ -524,6 +540,57 @@ def open_source(self, path): self.sources[full] = source return source + def open_ctable_array(self, table_path, logical_key): + """Open one external NDArray member hidden below a CTable node.""" + if self.format != "b2z": + raise NotImplementedError("Remote CTable access currently requires a B2Z source") + full = "/".join(part.strip("/") for part in (table_path, logical_key) if part.strip("/")) + self._validate(full) + if full in self.sources: + return self.sources[full] + matches = [info for info in self.archive.members if info.filename == full + ".b2nd"] + if len(matches) != 1: + raise NotImplementedError(f"Remote CTable array {full!r} is unavailable or not external") + info = matches[0] + if info.flag_bits & 1 or info.compress_type != 0: + raise NotImplementedError( + f"Remote CTable array {full!r} requires an unencrypted ZIP_STORED member" + ) + from blosc2.b2z_source import B2ZNDSource + + source = B2ZNDSource(self.urlpath, full, _archive=self.archive) + if self.source_validator is not None: + self.source_validator(source) + self.nodes[full] = ("ndarray", None) + self.sources[full] = source + return source + + def remote_array(self, full): + """Return a RemoteArray sharing this discovery owner's resources.""" + relative = full[len(self.root) + 1 :] if self.root else full + source = self.open_source(relative) if full in self.nodes else None + if source is None: + raise KeyError(full) + descriptor = { + "kind": self.format, + "version": 1, + "urlpath": source.urlpath, + "assume_immutable": True, + } + if self.format in {"b2z", "hdf5"}: + descriptor["dataset"] = full + return blosc2.RemoteArray( + source, + _source_descriptor=descriptor, + _store_owner=self, + cache_policy=self.cache_policy, + max_cache_bytes=( + self.max_cache_bytes + if self.cache_policy is not blosc2.CachePolicy.NONE + else CACHE_POLICY_DEFAULT + ), + ) + def acquire(self): with self.lock: if self._closed: @@ -646,8 +713,10 @@ def _close_resources(self): self._closed = True if self.archive is not None: self.archive.close() + self.archive = None if self.zstore is not None: self.zstore.close() + self.zstore = None for source in self.sources.values(): if isinstance(source, blosc2.HDF5NDSource): source.close() @@ -658,14 +727,14 @@ def _close_resources(self): self.listed.clear() self.hdf5_index = None if self.filesystem is not None and self._external_filesystem is None: - # fsspec's HTTP and S3 clients expose their own synchronous close hook. + # s3fs already registered this exact close through weakref.finalize, + # and its aiobotocore cleanup is not idempotent. Other fsspec + # clients (notably HTTP) still need deterministic closure here. + session = getattr(self.filesystem, "_session", None) close = getattr(self.filesystem, "close_session", None) - session = getattr(self.filesystem, "_s3creator", None) or getattr( - self.filesystem, "_session", None - ) - if close is not None and session is not None: + if close is not None and session is not None and not hasattr(self.filesystem, "_s3creator"): close(self.filesystem.loop, session) - self.filesystem = None + self.filesystem = None class RemoteStore: @@ -780,6 +849,8 @@ def __init__( owner.close() if kind == "unsupported": raise NotImplementedError(str(diagnostic)) + if kind == "ctable": + raise ValueError("RemoteStore requires a group; use RemoteCTable for a table") raise ValueError("RemoteStore requires a group; use RemoteArray for an array") owner.cache_policy = cache_policy owner.max_cache_bytes = limit @@ -967,28 +1038,12 @@ def __getitem__(self, path): group = object.__new__(type(self)) group._attach(self._owner, relative) return group + if kind == "ctable": + return blosc2.RemoteCTable._from_owner(self._owner, full) if kind == "unsupported": raise NotImplementedError(f"{path!r}: {value}") - source = self._owner.open_source(relative) - descriptor = { - "kind": self._owner.format, - "version": 1, - "urlpath": source.urlpath, - "assume_immutable": True, - } - if self._owner.format in {"b2z", "hdf5"}: - descriptor["dataset"] = full - return blosc2.RemoteArray( - source, - _source_descriptor=descriptor, - _store_owner=self._owner, - cache_policy=self.cache_policy, - max_cache_bytes=( - self.max_cache_bytes - if self.cache_policy is not blosc2.CachePolicy.NONE - else CACHE_POLICY_DEFAULT - ), - ) + self._owner.open_source(relative) + return self._owner.remote_array(full) def get_info(self, path=""): """Return node kind, known attributes and unsupported-node diagnostics.""" @@ -1005,7 +1060,7 @@ def get_info(self, path=""): ) def kind(self, path=""): - """Return 'group', 'ndarray' or 'unsupported'.""" + """Return 'group', 'ndarray', 'ctable' or 'unsupported'.""" return self.get_info(path).kind @property @@ -1257,6 +1312,11 @@ def _validate_save_destination(self, destination, overwrite): def _collect_export_nodes(self, include_cache): group_full = "/".join(p for p in (self._owner.root, self._path.strip("/")) if p) prefix = (group_full + "/") if group_full else "" + if any( + kind == "ctable" and (path == group_full or not prefix or path.startswith(prefix)) + for path, (kind, _) in self._owner.nodes.items() + ): + raise NotImplementedError("RemoteStore artifacts do not yet support RemoteCTable nodes") exported_source = { "urlpath": self._owner.urlpath, "dataset": group_full, diff --git a/tests/b2view/test_hierarchy.py b/tests/b2view/test_hierarchy.py index 6cba00331..e415f29c9 100644 --- a/tests/b2view/test_hierarchy.py +++ b/tests/b2view/test_hierarchy.py @@ -515,12 +515,13 @@ class Row: assert [(n.name, n.kind) for n in browser.list_children()] == [ ("empty", "group"), ("ordinary", "group"), - ("table", "unsupported"), + ("table", "ctable"), ] assert browser.list_children("/empty") == [] assert browser.get_info("/empty").user_attrs == {"empty": True} assert browser.list_children("/table") == [] - assert "CTable" in browser.get_info("/table").metadata["preview"] + assert browser.get_info("/table").metadata["type"] == "B2Z ctable" + np.testing.assert_array_equal(browser.preview("/table", max_rows=1)["data"]["x"], [3]) def test_b2z_large_embedded_chunk_notice(): diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py new file mode 100644 index 000000000..9c758b85b --- /dev/null +++ b/tests/ctable/test_remote_ctable.py @@ -0,0 +1,132 @@ +"""Read-only CTable access through fsspec-backed B2Z archives.""" + +from __future__ import annotations + +import dataclasses + +import numpy as np +import pytest + +import blosc2 + +fsspec = pytest.importorskip("fsspec") + + +@dataclasses.dataclass +class Row: + x: int = blosc2.field(blosc2.int64(null_storage="mask")) + vec: np.ndarray = blosc2.field( # noqa: RUF009 + blosc2.ndarray((2,), dtype=blosc2.float32()) + ) + tag: str = blosc2.field(blosc2.string(max_length=8), default="") + + +def remote_table_url(tmp_path, table, name="table"): + path = tmp_path / f"{name}.b2z" + table.to_b2z(path) + url = f"memory://{tmp_path.name}-{name}.b2z" + fsspec.filesystem("memory").pipe(url, path.read_bytes()) + return url + + +def test_remote_ctable_fixed_width_reads_and_queries(tmp_path): + rows = [ + (1, np.array([1, 2], dtype=np.float32), "one"), + (None, np.array([3, 4], dtype=np.float32), "missing"), + (3, np.array([5, 6], dtype=np.float32), "three"), + (4, np.array([7, 8], dtype=np.float32), "four"), + ] + local = blosc2.CTable(Row, rows, create_summary_index=False) + persistent = str(tmp_path / "persistent.b2d") + local.save(persistent) + with blosc2.CTable.open(persistent, mode="a") as table: + table.attrs["source"] = "test" + with blosc2.CTable.open(persistent) as table: + url = remote_table_url(tmp_path, table) + + with blosc2.RemoteCTable(url) as table: + assert len(table) == 4 + assert table.col_names == ["x", "vec", "tag"] + assert table.attrs["source"] == "test" + assert table["x"][0] == 1 + assert table["x"].is_null().tolist() == [False, True, False, False] + assert table["tag"][1:3].tolist() == ["missing", "three"] + np.testing.assert_array_equal(table["vec"][2], [5, 6]) + np.testing.assert_array_equal(table.where(table["x"] > 2)["x"][:], [3, 4]) + assert table["x"].sum() == 8 + assert table.source["urlpath"] == url + assert table.cache_policy is blosc2.CachePolicy.MEMORY + assert table.cache_bytes > 0 + with pytest.raises(ValueError, match="read-only"): + table.append((5, [9, 10], "five")) + + with pytest.raises(RuntimeError, match="closed"): + table["tag"][:] + + +def test_remote_store_returns_table_with_independent_lifetime(tmp_path): + source = tmp_path / "tree.b2z" + table = blosc2.CTable(Row, [(1, [1, 2], "one"), (2, [3, 4], "two")], create_summary_index=False) + with blosc2.TreeStore(source, mode="w", threshold=0) as root: + root["/group/table"] = table + root["/group/array"] = blosc2.arange(3) + url = f"memory://{tmp_path.name}-tree.b2z" + fsspec.filesystem("memory").pipe(url, source.read_bytes()) + + with blosc2.RemoteStore(url) as store: + assert store.kind("group/table") == "ctable" + remote = store["group/table"] + assert isinstance(remote, blosc2.RemoteCTable) + np.testing.assert_array_equal(remote["x"][:], [1, 2]) + remote.close() + + with blosc2.RemoteCTable(url, dataset="group/table") as direct: + np.testing.assert_array_equal(direct["x"][:], [1, 2]) + + +def test_remote_ctable_unsupported_column_is_lazy(tmp_path): + @dataclasses.dataclass + class Mixed: + x: int = 0 + text: str = blosc2.field(blosc2.vlstring(), default="") + + url = remote_table_url( + tmp_path, + blosc2.CTable(Mixed, [(1, "a"), (2, "bb")], create_summary_index=False), + "mixed", + ) + with blosc2.RemoteCTable(url, cache_policy=blosc2.CachePolicy.NONE) as table: + np.testing.assert_array_equal(table["x"][:], [1, 2]) + with pytest.raises(NotImplementedError, match="variable-length column 'text'"): + table["text"][:] + + +def test_remote_ctable_deleted_rows_and_disk_cache(tmp_path): + source = str(tmp_path / "deleted.b2d") + table = blosc2.CTable(Row, urlpath=source, mode="w", expected_size=8, create_summary_index=False) + table.extend([(i, [i, i + 1], str(i)) for i in range(6)]) + table.delete([1, 4]) + table.close() + with blosc2.CTable.open(source) as table: + url = remote_table_url(tmp_path, table, "deleted") + + cache = tmp_path / "cache" + with blosc2.RemoteCTable(url, cache_dir=cache) as remote: + np.testing.assert_array_equal(remote["x"][:], [0, 2, 3, 5]) + assert remote["x"][1] == 2 + assert list(remote["x"]) == [0, 2, 3, 5] + assert remote["x"].sum() == 10 + assert remote.cache_policy is blosc2.CachePolicy.DISK + + with blosc2.RemoteCTable(url, cache_dir=cache) as reopened: + np.testing.assert_array_equal(reopened["x"][:], [0, 2, 3, 5]) + + +def test_remote_store_rejects_table_root(tmp_path): + url = remote_table_url( + tmp_path, + blosc2.CTable(Row, [(1, [1, 2], "one")], create_summary_index=False), + "root", + ) + with pytest.raises(ValueError, match="use RemoteCTable"): + blosc2.RemoteStore(url) diff --git a/tests/test_remote_store.py b/tests/test_remote_store.py index 80386a217..93ca6f9fa 100644 --- a/tests/test_remote_store.py +++ b/tests/test_remote_store.py @@ -4,6 +4,7 @@ import json import sys import weakref +from types import SimpleNamespace import numpy as np import pytest @@ -623,6 +624,37 @@ def test_subgroup_and_garbage_collection(hierarchy): assert owner._closed +def test_owned_filesystem_session_is_left_to_fsspec_finalizer(): + from blosc2.remote_store import RemoteDiscovery + + closed = [] + close_calls = [] + archive = SimpleNamespace(close=lambda: closed.append(True)) + filesystem = SimpleNamespace( + loop=object(), _s3creator=object(), close_session=lambda *args: close_calls.append(args) + ) + owner = SimpleNamespace( + _closed=False, + archive=archive, + zstore=None, + sources={}, + caches={}, + nodes={}, + attrs={}, + listed={}, + hdf5_index=None, + filesystem=filesystem, + _external_filesystem=None, + ) + + RemoteDiscovery._close_resources(owner) + + assert closed == [True] + assert owner.archive is None + assert owner.filesystem is None + assert close_calls == [] + + def test_zarr_direct_lookup_without_listing(hierarchy, monkeypatch): url, data = hierarchy if not url.endswith(".zarr"): From 26c0fd98334d9b3b3ab483b6b2ae3aa0d37eaa15 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Thu, 17 Sep 2026 19:32:10 +0200 Subject: [PATCH 02/82] Advertise that https:// is supported too --- examples/ctable/remote_handling.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/examples/ctable/remote_handling.py b/examples/ctable/remote_handling.py index cf1ada097..8196a02b4 100644 --- a/examples/ctable/remote_handling.py +++ b/examples/ctable/remote_handling.py @@ -6,7 +6,7 @@ # SPDX-License-Identifier: BSD-3-Clause ####################################################################### -"""Create or access a fixed-width, mask-nullable remote CTable.""" +"""Create or access a fixed-width, mask-nullable CTable over S3 or HTTP(S).""" import argparse import pprint @@ -165,7 +165,7 @@ def access_table(args) -> None: def main() -> int: parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("url", nargs="?", help="Remote .b2z CTable URL") + parser.add_argument("url", nargs="?", help="Remote .b2z CTable URL (s3://, http://, or https://)") parser.add_argument("--write", type=Path, metavar="FILE", help="Create a local .b2z CTable instead") parser.add_argument("--rows", type=int, default=1_000_000) parser.add_argument("--batch-size", type=int, default=100_000) From 626abef975f0b2dad1eb248681cd651c3d4e618e Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Thu, 17 Sep 2026 20:19:02 +0200 Subject: [PATCH 03/82] Add remote UTF-8 table support and unified open dispatch --- doc/guides/remote_arrays.md | 41 ++++ examples/ctable/remote_handling.py | 85 +++++++-- plans/remote-ctable-v2.md | 187 ++++++++++++++++++ plans/remote-ctable.md | 15 +- src/blosc2/_utf8_array.py | 32 +++- src/blosc2/b2z_source.py | 12 +- src/blosc2/ctable_storage.py | 22 ++- src/blosc2/proxy_source.py | 10 + src/blosc2/remote_array.py | 9 + src/blosc2/remote_ctable.py | 2 +- src/blosc2/remote_store.py | 14 +- src/blosc2/schunk.py | 81 ++++++-- tests/b2view/test_hierarchy.py | 9 +- tests/ctable/test_remote_ctable.py | 296 +++++++++++++++++++++++++++++ tests/test_b2z_source.py | 41 +++- tests/test_fsspec.py | 12 +- 16 files changed, 803 insertions(+), 65 deletions(-) create mode 100644 plans/remote-ctable-v2.md diff --git a/doc/guides/remote_arrays.md b/doc/guides/remote_arrays.md index 0630d3262..df7bbbc32 100644 --- a/doc/guides/remote_arrays.md +++ b/doc/guides/remote_arrays.md @@ -140,6 +140,7 @@ What differs between the transports is the types of remote objects each can open | --------------------------------------- | -------------------------------- | ---------------------------- | | Standalone contiguous `.b2nd` | Yes (`blosc2.open` / `RemoteArray`) | Yes | | NDArray leaf inside `.b2z` | Yes (`blosc2.open` / `RemoteArray`) | Yes | +| CTable inside `.b2z` | Yes (`blosc2.open` / `RemoteCTable`) | No | | Zarr v2/v3 array | Yes (`blosc2.open` / `RemoteArray`) | No | | HDF5 dataset | Yes (`blosc2.open` / `RemoteArray`) | Yes | | Lazy or computed array | No | Yes | @@ -535,6 +536,46 @@ a[mask] For Caterva2, a bare {ref}`C2Array` can be substantially more efficient for one-off point queries: it sends coordinates to the server, which evaluates the selection and returns only the selected values. Prefer direct `C2Array` indexing for sparse, one-off point retrieval; prefer a {ref}`RemoteArray` when reuse through a local cache matters. +## Remote tables + +`RemoteCTable` opens a read-only CTable in an immutable remote `.b2z` archive. +Fixed-width and `blosc2.utf8()` columns are fetched on demand, including their +null masks. A table inside a hierarchy can also be opened through `RemoteStore`. + +`blosc2.open()` dispatches local table archives to `CTable` and remote table +archives to `RemoteCTable`. Remote `.b2z` groups return `RemoteStore` by default; +array leaves retain their `RemoteArray` behavior. Use `dataset="group/table"` +or a `::group/table` URL suffix to select a nested table. For a complete local +download instead, pass `lazy=False, cache_dir="download-cache"`. + +```python +with blosc2.open("https://example.org/readings.b2z") as table: + notes = table["note"][:5] + selected = table.where(table["note"] == "café") + ids = selected["id"][:] +``` + +UTF-8 strings use two compressed arrays: row offsets and encoded bytes. A slice +first reads its offsets, then its byte span. Both reads use the existing range +transport, fetching compressed blocks when worthwhile or whole compressed chunks +otherwise, and decompressing locally. Archive members are ZIP_STORED, as produced +by the Blosc2 writers; the arrays inside remain Blosc2-compressed. + +The backing arrays share the table's cache budget and traffic counters. MEMORY, +DISK (with `cache_dir`) and NONE policies are supported. Size reporting uses source +metadata without scanning strings. Small-member metadata prefetch may also fetch +some payload. Repeated reads can reuse cached blocks; filtering scans the required +columns because persisted indexes are not used remotely. + +Columns and views are borrowed from the root table and require it to remain open. +Closing a parent RemoteStore leaves a returned table usable; refreshing the store +invalidates previously returned tables and their columns. Copies and data exports +produce local tables. Remote writes, batch-backed `vlstring`/lists/objects, +dictionary columns and portable table-reference export remain unsupported. + +See `examples/ctable/remote_handling.py` for a batched archive writer with a nullable +multilingual UTF-8 column, plus sample row and string-slice traffic measurements. + ## Handle remote changes ### Standalone arrays and Caterva2 sources diff --git a/examples/ctable/remote_handling.py b/examples/ctable/remote_handling.py index 8196a02b4..7f2efa927 100644 --- a/examples/ctable/remote_handling.py +++ b/examples/ctable/remote_handling.py @@ -6,7 +6,7 @@ # SPDX-License-Identifier: BSD-3-Clause ####################################################################### -"""Create or access a fixed-width, mask-nullable CTable over S3 or HTTP(S).""" +"""Create or access a mask-nullable CTable with fixed-width and UTF-8 columns locally or remotely.""" import argparse import pprint @@ -31,6 +31,16 @@ class Reading: humidity: int = blosc2.field(blosc2.int16(null_storage="mask")) status: str = blosc2.field(blosc2.string(max_length=8, null_storage="mask")) active: bool = blosc2.field(blosc2.bool()) + note: str = blosc2.field(blosc2.utf8(null_storage="mask")) + + +def make_notes(ids): + notes = np.array(["", "café", "東京の観測", "🌦️ weather improving"], dtype=object)[ids % 4] + notes = np.array( + [f"{text} #{i}" if text else "" for i, text in zip(ids, notes, strict=True)], dtype=object + ) + notes[ids % 43 == 0] = None + return notes def write_table(args) -> None: @@ -79,6 +89,7 @@ def write_table(args) -> None: "humidity": humidity, "status": status, "active": ids % 5 != 0, + "note": make_notes(ids), }, validate=False, ) @@ -86,7 +97,14 @@ def write_table(args) -> None: with blosc2.CTable.open(str(output)) as table: assert len(table) == args.rows assert table.attrs[:] == attrs - null_counts = {name: table[name].null_count() for name in ("temperature", "humidity", "status")} + null_counts = { + name: table[name].null_count() for name in ("temperature", "humidity", "status", "note") + } + sample_ids = np.arange(min(args.rows, 8)) + expected = make_notes(sample_ids) + expected[sample_ids % 43 == 0] = "" + np.testing.assert_array_equal(table["note"][: len(sample_ids)], expected) + assert null_counts["note"] == (args.rows + 42) // 43 print(f"Created {output} ({output.stat().st_size / 1_000_000:.1f} MB, {args.rows:,} rows)") print(f"Mask-backed null counts: {null_counts}") @@ -103,26 +121,30 @@ def access_table(args) -> None: print(f"Accessing: {args.url}") started = time.perf_counter() - with blosc2.RemoteCTable(args.url, storage_options=storage_options) as table: + with blosc2.open(args.url, storage_options=storage_options or None) as table: + if not isinstance(table, blosc2.CTable): + raise ValueError("input must be a CTable archive") + remote = isinstance(table, blosc2.RemoteCTable) nbytes = table.nbytes metadata = { "type": type(table).__name__, - "source": table.source, + "source": table.source if remote else args.url, "rows": table.nrows, "columns": table.col_names, "chunks": table.chunks, "blocks": table.blocks, "nbytes": f"{nbytes} ({nbytes / 1024**2:.2f} MiB)", - "cache_policy": table.cache_policy.name, - "cache_bytes": f"{table.cache_bytes} ({table.cache_bytes / 1024:.2f} KiB)", "schema": table.schema_dict(), "attrs": dict(table.attrs), } + if remote: + metadata["cache_policy"] = table.cache_policy.name + metadata["cache_bytes"] = f"{table.cache_bytes} ({table.cache_bytes / 1024:.2f} KiB)" metadata_time = time.perf_counter() - started - metadata_bytes = table.traffic.nbytes - metadata_requests = table.traffic.requests + metadata_bytes = table.traffic.nbytes if remote else 0 + metadata_requests = table.traffic.requests if remote else 0 - print("\n[Format: Blosc2 B2Z (Lazy RemoteCTable)]") + print(f"\n[Format: Blosc2 B2Z ({type(table).__name__})]") for name, value in metadata.items(): rendered = pprint.pformat(value) if isinstance(value, (dict, list)) else value if name == "attrs" and value: @@ -133,17 +155,17 @@ def access_table(args) -> None: started = time.perf_counter() sample = str(table[:5]) first_time = time.perf_counter() - started - first_bytes = table.traffic.nbytes - metadata_bytes - first_requests = table.traffic.requests - metadata_requests + first_bytes = table.traffic.nbytes - metadata_bytes if remote else 0 + first_requests = table.traffic.requests - metadata_requests if remote else 0 print(sample) started = time.perf_counter() _ = str(table[:5]) second_time = time.perf_counter() - started - second_bytes = table.traffic.nbytes - metadata_bytes - first_bytes - second_requests = table.traffic.requests - metadata_requests - first_requests + second_bytes = table.traffic.nbytes - metadata_bytes - first_bytes if remote else 0 + second_requests = table.traffic.requests - metadata_requests - first_requests if remote else 0 - print("\nTiming & Network Traffic:") + print("\nTiming & Network Traffic:" if remote else "\nTiming (local reads; no network traffic):") print( f" - Metadata open : {metadata_time * 1000:7.1f} ms " f"({metadata_requests} requests, {metadata_bytes / 1024:8.2f} KB transferred)" @@ -152,20 +174,41 @@ def access_table(args) -> None: f" - 1st row fetch : {first_time * 1000:7.1f} ms " f"({first_requests} requests, {first_bytes / 1024:8.2f} KB transferred)" ) - cache_hit = " (cache hit!)" if second_requests == second_bytes == 0 else "" + cache_hit = " (cache hit!)" if remote and second_requests == second_bytes == 0 else "" print( f" - 2nd row fetch : {second_time * 1000:7.1f} ms " f"({second_requests} requests, {second_bytes / 1024:8.2f} KB transferred){cache_hit}" ) - print( - f" - Total network : {table.traffic.requests} requests, " - f"{table.traffic.nbytes / 1024:8.2f} KB transferred" - ) + if "note" in table.col_names and table["note"].is_utf8: + start = max(0, table.nrows - 5) + print(f"\nUTF-8 note slice [{start}:{table.nrows}] (may overlap warmed blocks in small tables):") + for label in ("1st", "2nd"): + before_bytes = table.traffic.nbytes if remote else 0 + before_requests = table.traffic.requests if remote else 0 + started = time.perf_counter() + notes = table["note"][start:] + elapsed = time.perf_counter() - started + transferred = table.traffic.nbytes - before_bytes if remote else 0 + requests = table.traffic.requests - before_requests if remote else 0 + print( + f" - {label} note fetch: {elapsed * 1000:7.1f} ms " + f"({requests} requests, {transferred / 1024:8.2f} KB transferred)" + ) + if label == "1st": + print(f" {notes}") + if remote: + print( + f" - Total network : {table.traffic.requests} requests, " + f"{table.traffic.nbytes / 1024:8.2f} KB transferred" + ) + print(f" - Retained cache: {table.cache_bytes / 1024:8.2f} KB") def main() -> int: parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("url", nargs="?", help="Remote .b2z CTable URL (s3://, http://, or https://)") + parser.add_argument( + "url", nargs="?", help="Local .b2z CTable path or remote URL (s3://, http://, https://)" + ) parser.add_argument("--write", type=Path, metavar="FILE", help="Create a local .b2z CTable instead") parser.add_argument("--rows", type=int, default=1_000_000) parser.add_argument("--batch-size", type=int, default=100_000) @@ -177,7 +220,7 @@ def main() -> int: if args.write is not None and args.url is not None: parser.error("URL cannot be combined with --write") if args.write is None and args.url is None: - parser.error("provide a remote URL or use --write FILE.b2z") + parser.error("provide a local path or remote URL, or use --write FILE.b2z") try: write_table(args) if args.write is not None else access_table(args) diff --git a/plans/remote-ctable-v2.md b/plans/remote-ctable-v2.md new file mode 100644 index 000000000..c1173bb8b --- /dev/null +++ b/plans/remote-ctable-v2.md @@ -0,0 +1,187 @@ +# RemoteCTable v2: UTF-8 columns + +Status: implemented on 2026-09-17; verification results recorded below. + +Subsequent integration: `blosc2.open()` now dispatches remote B2Z table/group nodes +to RemoteCTable/RemoteStore by default, while local table archives remain CTable. +The example uses this common opener and accepts local paths as well as URLs. +Explicit `lazy=False, cache_dir=...` retains whole-archive localization. +Integration verification: 10,245 default-suite tests passed, 36 skipped; Ruff +passed. The command `python examples/ctable/remote_handling.py readings.b2z` +was also verified against a local one-million-row archive. + +## Objective and scope + +Extend the existing read-only RemoteCTable to support `blosc2.utf8()` columns, +including nulls, queries, size reporting and local materialization. Update +`examples/ctable/remote_handling.py` to demonstrate a UTF-8 column alongside its +existing fixed-width columns and report cold/warm remote reads. + +Reuse `UTF8Array`, RemoteArray, the shared discovery owner and existing CTable +read paths. No new public class, dependency, archive format or constructor option +is needed. Preserve local behavior and optimizations. + +Batch-backed `vlstring`, bytes, lists, objects and dictionary stores remain +unsupported. Persisted indexes, portable references, remote writes and new cache +APIs remain separately scoped follow-ups from `remote-ctable.md`. + +## Storage and transfer model + +A UTF-8 column already consists of two external NDArray members: + +- `_cols/.b2nd`: int64 offsets, one more entry than stored rows. +- `_cols/.utf8.b2nd`: uint8 encoded string bytes. + +Mask-backed nulls use the existing `.notnull` companion convention; sentinel +nulls retain their existing semantics. Reuse the storage constants and column +path conversion rather than spelling these conventions again in code. + +For rows `[a:b]`, `UTF8Array` reads `offsets[a:b+1]`, then the byte range between +the first and last offsets, and constructs strings locally. Both backing arrays +will be RemoteArrays. Each independently uses the existing B2Z byte-range reader +to fetch compressed Blosc2 blocks when worthwhile, or whole compressed chunks +otherwise. Decompression and UTF-8 decoding happen locally. Offset reads precede +the dependent data reads; no new prefetch or concurrency scheduler is planned. + +Our archive writers already use ZIP_STORED members, while their Blosc2 payloads +remain compressed. Retain existing rejection of encrypted, ZIP-compressed or +unsupported embedded array members. Small-member metadata prefetch may include +payload bytes. Transfer granularity is blocks/chunks, not exact string lengths. + +Both arrays share the table owner's cache budget, traffic accounting and source +identity. NONE, MEMORY and DISK must all work. Full predicates can scan a column; +UTF-8 support does not imply indexed or server-side filtering. + +## Implementation sequence + +1. **Open the existing representation remotely.** In + `RemoteTableStorage.open_varlen_scalar_column()`, handle `UTF8Spec` by opening + offsets and data through `_open_array()` and constructing the existing + `UTF8Array`. Preserve explicit errors for every other variable-length spec. + Check both members through the existing discovery/validation path and ensure + partial failures do not leak handles. Keep opening lazy: schema inspection + and supported sibling reads must not require opening the UTF-8 payload. + +2. **Adapt shared UTF-8 reads only where necessary.** Trace callers in CTable, + `UTF8Array` and expression helpers before changing assumptions about native + NDArrays, `.schunk`, `.urlpath` or native-extension inputs. Slicing and sparse + gathers already read offsets/bytes in spans; reuse them. Make `nbytes`, + `cbytes` and compression-ratio reporting work from backing-array metadata + without materializing strings or requiring a cache SChunk. Verify table and + column size reporting, repr, chunks/blocks reporting and local copy/save. + Preserve the existing meaning of each size metric. + +3. **Preserve read-only and lifetime guarantees.** Audit raw UTF8Array access as + well as table mutation methods: append/extend can change pending rows before + touching backing arrays, so relying on RemoteArray write rejection alone is + insufficient. Add the smallest shared guard needed to reject mutation before + changing state. Ensure root close and RemoteStore refresh invalidate wrapper + reads and metadata access, including accesses answerable from cached wrapper + state. Reuse existing ownership/generation mechanisms; views remain borrowed + and detached copies must be ordinary writable local objects. + +4. **Verify nulls and queries.** Reuse CTable's UTF-8 comparisons, string + predicates and span-based expression evaluation. Cover mixed fixed-width and + UTF-8 predicates, nullable filtering and existing supported aggregations. + Keep persisted indexes disabled even when present in the source. Preserve + local errors for unsupported UTF-8 computations and unsupported siblings. + Repair shared paths rather than introducing remote-specific query algorithms. + +5. **Extend the demonstration and documentation.** Add a nullable + `blosc2.utf8(null_storage="mask")` column named `note` to the example's + `Reading` schema. Generate deterministic varying-length notes, including + accented text, non-Latin text, emoji, empty strings and actual nulls, within + the existing batched writer. Include note null counts and representative + values in the local round-trip checks; retain the fixed-width `status` column. + Update the description and sample output to describe mixed column storage. + Preserve the current row-fetch cold/warm report, and add a short direct note + slice read twice from a distant, previously untouched region for sufficiently + large tables. Report traffic deltas and elapsed time for both reads; do not + call a read cold merely because it is the first explicit column read after + sample rows have already warmed it. Handle small tables and older archives + without `note` gracefully, using the actual schema to select the extra demo. + Document UTF-8 support, scan costs and the remaining unsupported column types. + +## Verification and acceptance + +Use the `blosc2` conda environment for all Python, test and build commands. +Extend existing tests and helpers rather than building another test harness. + +- Compare remote reads against a local read-only CTable from the same archive: + empty/nonempty columns, multilingual and long values, empty strings, sentinel + and mask nulls, chunk/block boundaries, deleted rows and spare capacity. + Exercise scalar, contiguous, strided, reverse and fancy reads, iteration, + filtered views, predicates and local materialization. Include nested column + paths and both standalone and RemoteStore-nested tables. +- Cover missing/malformed companions through existing validation conventions, + unsupported sibling isolation, mutation rejection before pending-state changes, + parent-store close, table close, refresh invalidation and partial-open cleanup. +- Use the instrumented fsspec memory filesystem to exercise NONE/MEMORY/DISK, + aggregate cache limits and disk reopen. Check opening/size reporting does not + scan string payloads, allowing existing bounded metadata prefetch. Verify a + small slice transfers a bounded subset of a sufficiently large archive and a + repeated cached slice requires no new payload requests when it fits the budget. +- Use deliberately suitable chunk/block geometry and data to exercise existing + block selection for both backing arrays, plus whole-chunk fallback. Reuse + source-level transport tests and add focused integration coverage; do not + assert block mode for every tiny or highly compressible dataset. Include the + existing deterministic HTTP range-server setup to verify actual transport. +- Run focused UTF-8, CTable, RemoteCTable, RemoteStore and relevant B2Z/remote-array + tests, then the default suite. Run Ruff on changed Python files. Warnings remain + errors; no C/Cython changes are expected without a demonstrated need. +- Generate a small example archive and verify its schema, note values, masks and + attributes locally and through `memory://`. Record a larger cold/warm experiment + with archive size, requests, transferred bytes and retained cache bytes. Cloud + uploads are optional manual validation, not a requirement for automated tests. + +Completion means UTF-8 reads and supported queries match local behavior, remain +read-only and bounded on demand, and the updated example demonstrates them. +Update the original plan's follow-up status once this implementation is verified. + +## Implementation and verification results + +- RemoteTableStorage now opens UTF-8 offsets/data as RemoteArrays and validates + their dimensions and dtypes. Partial-open failures release acquired handles. +- The existing UTF8Array handles remote reads and queries; its public mutations + reject writes before changing pending state. Metadata and empty reads check + backing-array lifetime. Size reporting reads frame metadata, retaining the + existing padded-storage meaning of UTF8Array.nbytes. +- Native range sources expose stored and compressed sizes; RemoteArray.cbytes + reports source compressed size (and explicitly rejects sources without it). +- The example now writes nullable multilingual notes, verifies their values and + null counts, and measures first/repeated reads of a distant note slice. Its + extra demonstration is conditional on the source having a UTF-8 note column, + so older fixed-width archives remain usable. +- Regression coverage includes all cache policies, sentinel/mask nulls, deleted + rows, nested/empty tables, copy/export, close/refresh, invalid companions, + bounded transfers, block and whole-chunk reads, disk reopen and real HTTP ranges. + Block-path tests lower the cost threshold for small deterministic fixtures; + production selection thresholds are unchanged. +- Focused CTable, RemoteStore, RemoteArray and B2Z suites: 3,066 passed, 5 skipped. + Ruff passed for all changed Python files. +- Default suite: 10,240 passed, 36 skipped. The first run had one RSS-based + `TestCodec.test_no_leaks` failure; it passed in isolation and the complete + default-suite rerun passed without code changes to that test or compression. + +### Cold/warm example experiment + +The updated writer generated 1,000,000 rows and a 5,296,657-byte archive, then the +example opened that exact archive through fsspec `memory://` with default MEMORY +caching. These are byte-range traffic measurements, not cloud latency results. + +| Operation | Requests | Transferred KiB | +| --- | ---: | ---: | +| Metadata/size/schema/attributes | 15 | 125.98 | +| First five rows | 9 | 246.53 | +| Repeat first five rows | 0 | 0 | +| Notes at rows 999995:1000000 | 2 | 134.57 | +| Repeat note slice | 0 | 0 | +| Total | 26 | 507.07 | + +Retained payload after these reads was 494.50 KiB. In a representative local run, +metadata took 10.3 ms, the first/repeated row reads 7.9/7.0 ms and the +first/repeated note slices 0.8/0.3 ms; these timings do not measure network latency. + +The same example also passed a 100-row round trip with smaller write batches; +small-table note reads can already be warm from the initial row/metadata reads, +which the example explicitly labels. diff --git a/plans/remote-ctable.md b/plans/remote-ctable.md index a0369ce33..7ceb653e4 100644 --- a/plans/remote-ctable.md +++ b/plans/remote-ctable.md @@ -1,7 +1,8 @@ # RemoteCTable implementation plan Status: initial fixed-width, read-only implementation completed on 2026-09-17; -UTF-8, batch-backed columns, persisted indexes and portable references remain follow-ups. +UTF-8 support was added in the v2 extension (see `remote-ctable-v2.md`); +batch-backed columns, persisted indexes and portable references remain follow-ups. ## Objective and architecture @@ -55,8 +56,9 @@ second implementation or building a general remote-object framework. NDArrays. The schema and table metadata live in `_meta` SChunk vlmeta. - `b2z_source.py`: `B2ZNDSource` reads external, unencrypted ZIP_STORED `.b2nd` members. `member_vlmeta()` and `B2ZEmbeddedMetadata` provide metadata reads. -- `remote_store.py`: discovery already recognizes CTable roots and nested table - boundaries, but deliberately marks them unsupported and hides their internals. +- `remote_store.py`: discovery recognizes CTable roots and nested table + boundaries, opens supported table nodes as RemoteCTable objects, and keeps + their internals opaque during hierarchy traversal. - Persisted index descriptors are currently resolved through local paths and ZIP-offset registration. Batch-backed columns use local `.b2b` opening paths. @@ -106,7 +108,9 @@ with blosc2.RemoteStore("s3://bucket/archive.b2z") as store: node: `_cols`, `_meta` and index files are not ordinary public children. - A standalone table archive is opened with RemoteCTable. RemoteStore retains its group-root requirement and gives a diagnostic directing users to it. -- No automatic change to `blosc2.open()` or `CTable.open()` URL dispatch. +- The initial release did not change `blosc2.open()` or `CTable.open()` URL + dispatch. A subsequent extension adds remote table/group dispatch to + `blosc2.open()`; `CTable.open()` remains the local table opener. ### Supported data and operations @@ -249,7 +253,8 @@ and correct results, not a claim that scan queries avoid reading their operands. ## Follow-ups, separately scoped -1. UTF-8 columns through remote offsets and bytes, with null/query/size reporting. +1. UTF-8 columns through remote offsets and bytes, with null/query/size reporting: + implemented in the v2 extension described in `remote-ctable-v2.md`. 2. Remote batch reads for lists, variable-length values and dictionary stores. 3. Persisted indexes through a remote-aware sidecar resolver, starting with SUMMARY indexes and measuring query transfer savings. diff --git a/src/blosc2/_utf8_array.py b/src/blosc2/_utf8_array.py index a0d4b4017..356c3c6be 100644 --- a/src/blosc2/_utf8_array.py +++ b/src/blosc2/_utf8_array.py @@ -443,6 +443,7 @@ class UTF8Array: """ def __init__(self, spec, offsets=None, data=None) -> None: + from blosc2.remote_array import RemoteArray from blosc2.schema import UTF8Spec if not isinstance(spec, UTF8Spec): @@ -455,6 +456,7 @@ def __init__(self, spec, offsets=None, data=None) -> None: offsets, data = _new_backend_arrays() self._offsets = offsets self._data = data + self._remote = isinstance(offsets, RemoteArray) or isinstance(data, RemoteArray) self._persisted_rows: int = int(offsets.shape[0]) - 1 # End byte position of the persisted region; resolved lazily because it # needs a chunk read from the offsets array. @@ -467,6 +469,16 @@ def __init__(self, spec, offsets=None, data=None) -> None: # Private helpers # ------------------------------------------------------------------ + def _check_open(self) -> None: + if self._remote: + self._offsets._check_open() + self._data._check_open() + + def _check_writable(self) -> None: + self._check_open() + if self._remote: + raise ValueError("Remote UTF8Array is read-only") + @property def _bytes_used(self) -> int: if self._bytes_used_cache is None: @@ -636,6 +648,7 @@ def _rewrite_from(self, pos: int, values: list[str]) -> None: def append(self, value: Any) -> None: """Append one string row (``None`` maps to the null sentinel).""" + self._check_writable() value = self._coerce(value) self._pending.append(value) self._pending_chars += len(value) @@ -649,6 +662,7 @@ def extend(self, values: Iterable[Any]) -> None: unusual batch of many multi-MB strings can therefore overshoot ``_FLUSH_CHARS`` by up to one chunk before a flush is triggered. """ + self._check_writable() it = iter(values) while True: chunk = list(itertools.islice(it, _FLUSH_ROWS)) @@ -666,8 +680,10 @@ def extend(self, values: Iterable[Any]) -> None: def flush(self) -> None: """Write pending rows to the backing offsets/data NDArrays.""" + self._check_open() if not self._pending: return + self._check_writable() values, self._pending = self._pending, [] self._pending_chars = 0 self._rewrite_from(self._persisted_rows, values) @@ -680,6 +696,7 @@ def set_all(self, values: Iterable[Any]) -> None: in-memory ``UTF8Array``). Used by ``sort_by(inplace=True)`` and ``compact()`` to rewrite a column in a new row order. """ + self._check_writable() coerced = [self._coerce(v) for v in values] self._pending = [] self._pending_chars = 0 @@ -690,12 +707,14 @@ def set_all(self, values: Iterable[Any]) -> None: # ------------------------------------------------------------------ def __len__(self) -> int: + self._check_open() return self._persisted_rows + len(self._pending) def __iter__(self) -> Iterator[str]: yield from self[:] def __getitem__(self, index: int | slice | list | tuple | np.ndarray): + self._check_open() if isinstance(index, (int, np.integer)): n = len(self) index = int(index) @@ -730,6 +749,7 @@ def __setitem__(self, index: int, value: Any) -> None: row rewrites the byte blob and offsets of all subsequent rows — an O(n - index) operation. """ + self._check_writable() if not isinstance(index, (int, np.integer)): raise TypeError(f"UTF8Array assignment index must be int, got {type(index)!r}") value = self._coerce(value) @@ -826,11 +846,13 @@ def __ge__(self, other: Any, /): @property def spec(self): + self._check_open() return self._spec @property def dtype(self): """The ``StringDType`` used for materialized reads.""" + self._check_open() return self._dtype @property @@ -841,6 +863,7 @@ def shape(self) -> tuple[int, ...]: @property def ndim(self) -> int: """Always 1: a utf8 array is a flat sequence of strings.""" + self._check_open() return 1 @property @@ -862,28 +885,35 @@ def __array__(self, dtype=None, copy=None) -> np.ndarray: @property def offsets(self): """The underlying ``int64`` NDArray of row offsets (length ``n + 1``).""" + self._check_open() return self._offsets @property def data(self): """The underlying ``uint8`` NDArray with the concatenated UTF-8 bytes.""" + self._check_open() return self._data @property def schunk(self): + self._check_open() return self._offsets.schunk @property def urlpath(self) -> str | None: + self._check_open() return getattr(self._offsets, "urlpath", None) @property def nbytes(self) -> int: + self._check_open() + if self._remote: + return self._offsets.src.storage_nbytes + self._data.src.storage_nbytes return self._offsets.schunk.nbytes + self._data.schunk.nbytes @property def cbytes(self) -> int: - return self._offsets.schunk.cbytes + self._data.schunk.cbytes + return self._offsets.cbytes + self._data.cbytes @property def cratio(self) -> float: diff --git a/src/blosc2/b2z_source.py b/src/blosc2/b2z_source.py index 605f0ac10..180b85aeb 100644 --- a/src/blosc2/b2z_source.py +++ b/src/blosc2/b2z_source.py @@ -241,6 +241,14 @@ def _read_archive(self, offset, size): return data +class B2ZArrayNotFoundError(ValueError): + """The selected archive path is not an external NDArray member.""" + + def __init__(self, dataset, traffic): + super().__init__(f"No supported external NDArray at {dataset!r}; specify an external array leaf") + self.traffic = traffic + + class B2ZNDSource(ByteRangeNDSource): """Read a stored external NDArray from an immutable B2Z archive via fsspec. @@ -290,9 +298,7 @@ def __init__( object_info = archive.object_info matches = [info for info in archive.members if info.filename == dataset + ".b2nd"] if not matches: - raise ValueError( - f"No supported external NDArray at {dataset!r}; specify an external array leaf" - ) + raise B2ZArrayNotFoundError(dataset, self.traffic) if len(matches) != 1: raise ValueError("duplicate B2Z array member") self.member_offset, self.member_length = archive.member_window(matches[0], prefetch=True) diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index 774356541..507f2a239 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -15,7 +15,7 @@ * :class:`FileTableStorage` — arrays are stored inside a :class:`blosc2.TreeStore` rooted at ``urlpath``; logical object metadata lives in ``/_meta`` and table data lives under ``/_valid_rows`` and ``/_cols/``. -* :class:`RemoteTableStorage` — fixed-width arrays are opened lazily as +* :class:`RemoteTableStorage` — fixed-width and UTF-8 backing arrays are opened lazily as :class:`blosc2.RemoteArray` objects from a remote B2Z archive. """ @@ -645,7 +645,7 @@ def index_anchor_path(self, col_name: str) -> str | None: class RemoteTableStorage(TableStorage): - """Read-only fixed-width CTable storage over a shared RemoteStore owner.""" + """Read-only CTable storage over a shared RemoteStore owner.""" def __init__(self, owner, root_key: str) -> None: self._owner = owner @@ -699,6 +699,24 @@ def open_list_column(self, name: str) -> ListArray: raise NotImplementedError(f"Remote CTable list column {name!r} is not supported") def open_varlen_scalar_column(self, name: str, spec) -> _ScalarVarLenArray: + if isinstance(spec, UTF8Spec): + key = f"{_COLS_DIR}/{_column_name_to_relpath(name)}" + first = len(self._arrays) + offsets = self._open_array(key) + data = None + try: + data = self._open_array(key + _UTF8_DATA_SUFFIX) + if offsets.ndim != 1 or offsets.dtype != np.dtype("int64") or offsets.shape[0] < 1: + raise ValueError(f"Invalid offsets array for remote UTF-8 column {name!r}") + if data.ndim != 1 or data.dtype != np.dtype("uint8"): + raise ValueError(f"Invalid data array for remote UTF-8 column {name!r}") + return UTF8Array(spec, offsets, data) + except BaseException: + for array in (offsets, data): + if array is not None: + array.close() + del self._arrays[first:] + raise raise NotImplementedError( f"Remote CTable variable-length column {name!r} ({type(spec).__name__}) is not supported" ) diff --git a/src/blosc2/proxy_source.py b/src/blosc2/proxy_source.py index 186d9d404..f53fa23eb 100644 --- a/src/blosc2/proxy_source.py +++ b/src/blosc2/proxy_source.py @@ -1115,6 +1115,16 @@ def read_ranges(self, spans: Sequence[tuple[int, int]]) -> list[bytes]: """ return [self.read_range(offset, size) for offset, size in spans] + @property + def storage_nbytes(self) -> int: + """Uncompressed stored bytes, including padding, as for SChunk.nbytes.""" + return int(self._header[4]) + + @property + def cbytes(self) -> int: + """Compressed payload bytes recorded in the native frame header.""" + return int(self._header[5]) + def wants_blocks( self, nchunk: int, diff --git a/src/blosc2/remote_array.py b/src/blosc2/remote_array.py index 72fffedc2..0c525caf6 100644 --- a/src/blosc2/remote_array.py +++ b/src/blosc2/remote_array.py @@ -1288,6 +1288,15 @@ def traffic(self): self._check_open() return getattr(self.src, "traffic", None) + @property + def cbytes(self) -> int: + """Compressed source size, when supplied by the source format.""" + self._check_open() + value = getattr(self.src, "cbytes", None) + if value is None: + raise NotImplementedError("This remote source does not report its compressed size") + return int(value) + @property def nbytes(self) -> int: """The uncompressed size of the remote array.""" diff --git a/src/blosc2/remote_ctable.py b/src/blosc2/remote_ctable.py index e4b6edafe..1ddac7cb3 100644 --- a/src/blosc2/remote_ctable.py +++ b/src/blosc2/remote_ctable.py @@ -14,7 +14,7 @@ class RemoteCTable(CTable): - """A read-only CTable whose fixed-width columns are fetched on demand.""" + """A read-only CTable whose fixed-width and UTF-8 columns are fetched on demand.""" def __new__( cls, diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index b93ff72f1..2721714b1 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -78,8 +78,13 @@ def __init__( _source_validator=None, _manifest_validator=None, _max_nodes=None, + _source_format=None, ): self.urlpath, dataset, self.format = parse_container_url(urlpath, dataset) + if _source_format is not None: + self.format = _source_format + elif manifest is not None: + self.format = manifest["source"]["kind"] self.root = (dataset or "").strip("/") self._validate(self.root) self.storage_options = storage_options or {} @@ -797,6 +802,7 @@ def __init__( _source_validator=None, _manifest_validator=None, _max_nodes=None, + _source_format=None, ): if isinstance(urlpath, os.PathLike): urlpath = os.fspath(urlpath) @@ -819,7 +825,11 @@ def __init__( from blosc2.remote_store_cache import StoreDiskCache base_url, root, kind = parse_container_url(urlpath, dataset) - source = {"urlpath": base_url, "dataset": (root or "").strip("/"), "kind": kind} + source = { + "urlpath": base_url, + "dataset": (root or "").strip("/"), + "kind": _source_format or kind, + } fingerprint = storage_options_fingerprint(storage_options) if fingerprint: # The same URL through another endpoint or account must not @@ -838,6 +848,7 @@ def __init__( _source_validator=_source_validator, _manifest_validator=_manifest_validator, _max_nodes=_max_nodes, + _source_format=_source_format, ) except BaseException: if disk is not None: @@ -1155,6 +1166,7 @@ def refresh(self): _source_validator=owner.source_validator, _manifest_validator=owner.manifest_validator, _max_nodes=owner.max_nodes, + _source_format=owner.format, ) try: if not replacement.is_tree: diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index 6ce2e4a8b..e35d1a8a2 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2260,6 +2260,8 @@ def _open_fsspec_url(urlpath: str, mode: str, offset: int, kwargs: dict): lazy = False if lazy is False and hdf5_index is not None: raise NotImplementedError("hdf5_index is only supported with lazy=True") + if lazy is None and source_format == "b2z": + lazy = True # Auto-infer lazy=True only when the caller left the choice unspecified. lazy = _resolve_lazy(lazy, dataset, source_format, urlpath) @@ -2282,6 +2284,8 @@ def _open_fsspec_url(urlpath: str, mode: str, offset: int, kwargs: dict): requested = [k for k, v in kwargs.items() if v is not None] if requested: raise NotImplementedError(f"{', '.join(requested)} is not supported with lazy=True") + if source_format == "b2z": + return _open_remote_b2z(urlpath, remote_array_options) return blosc2.RemoteArray(urlpath, **remote_array_options) _validate_non_lazy_fsspec_options(immutable_present, remote_array_options, cache_path, max_concurrency) @@ -2292,8 +2296,8 @@ def _open_fsspec_url(urlpath: str, mode: str, offset: int, kwargs: dict): if source_format == "b2z": raise NotImplementedError( - "Remote B2Z containers require b2view for browsing, lazy=True with a dataset for arrays, " - "or cache_dir= for explicit localization" + "Remote B2Z containers require lazy=True for on-demand access, " + "or cache_dir= with lazy=False for explicit localization" ) if offset != 0: @@ -2311,6 +2315,44 @@ def _open_fsspec_url(urlpath: str, mode: str, offset: int, kwargs: dict): return blosc2.from_cframe(f.read()) +def _open_remote_b2z(urlpath, options): + """Keep the array fast path and discover table/group nodes through RemoteStore.""" + from blosc2.b2z_source import B2ZArrayNotFoundError + + dataset = options.get("dataset") + array_error = None + if dataset and dataset.strip("/"): + try: + return blosc2.RemoteArray(urlpath, **options) + except B2ZArrayNotFoundError as exc: + # Tables and groups have no corresponding .b2nd member. + array_error = exc + if options["cache_path"] is not None: + raise NotImplementedError("Remote tables and stores use cache_dir, not cache_path") + if options["max_concurrency"] is not None: + raise NotImplementedError("max_concurrency is only supported for remote arrays") + if options["assume_immutable"] is not True: + raise NotImplementedError("Remote tables and stores require assume_immutable=True") + store_options = { + key: value + for key, value in options.items() + if key in {"dataset", "storage_options", "cache_dir", "cache_policy", "max_cache_bytes"} + } + try: + with blosc2.RemoteStore( + urlpath, _allow_array_root=True, _source_format="b2z", **store_options + ) as store: + if array_error is not None: + # Include the initial array lookup in shared transfer accounting. + store.traffic.nbytes += array_error.traffic.nbytes + store.traffic.requests += array_error.traffic.requests + return store[""] + except KeyError: + if array_error is not None: + raise array_error from None + raise + + def _is_hdf5_open_request(urlpath: str, kwargs: dict) -> bool: if kwargs.get("source_format") == "hdf5" or "hdf5_index" in kwargs: return True @@ -2391,6 +2433,9 @@ def open( | blosc2.ObjectArray | blosc2.C2Array | blosc2.RemoteArray + | blosc2.CTable + | blosc2.RemoteCTable + | blosc2.RemoteStore | blosc2.LazyArray | blosc2.Proxy | blosc2.DictStore @@ -2399,7 +2444,7 @@ def open( ): """Open a persistent :ref:`SChunk`, :ref:`NDArray`, a remote :ref:`C2Array`, :ref:`RemoteArray`, :ref:`Proxy`, a :ref:`DictStore`, :ref:`EmbedStore`, or - :ref:`TreeStore`. + :ref:`TreeStore`, :class:`CTable`, :class:`RemoteCTable`, or :class:`RemoteStore`. See the `Notes` section for more info on opening `Proxy` objects. @@ -2436,15 +2481,16 @@ def open( kwargs: dict, optional lazy: bool or None, optional ``None`` (the default) automatically selects the access mode. ``True`` - returns a lazy :ref:`RemoteArray`; ``False`` requests eager access. - Known remote `.b2nd` arrays default to lazy access, including when + returns a lazy remote object; ``False`` requests eager access. + Known remote `.b2nd` arrays and `.b2z` archives default to lazy access, including when ``cache_dir=`` is supplied. Pass ``lazy=False`` to download the whole container under ``cache_dir`` instead. ``mmap_mode=`` or a nonzero ``offset=`` forces the eager path. Dataset paths currently require ``lazy=True`` and reject explicit ``False``. For an fsspec URL or a Caterva2 :ref:`URLPath`, return a :ref:`RemoteArray` over - the remote dataset and read the byte ranges a slice touches. Neither form opens - a whole remote store hierarchy. A slice landing in a small part of a large + the remote array dataset and read the byte ranges a slice touches. + B2Z table and group nodes return :class:`RemoteCTable` and :class:`RemoteStore` instead. + A slice landing in a small part of a large chunk costs only the *blocks* it touches when ranges are available; chunks small enough to be one cheap request are still fetched whole. What arrives is kept in memory (defaulting to :attr:`CachePolicy.MEMORY`), @@ -2508,7 +2554,7 @@ def open( an fsspec URL (for instance credentials, endpoint URL, token, client_kwargs, etc.). dataset: str, optional Array path within HDF5, Zarr, or B2Z containers (e.g. ``dataset="d0/d1/a2"``). - B2Z supports external NDArray leaves in immutable archives. + B2Z also supports table and group paths in immutable archives. Requires ``lazy=True``. hdf5_index: dict | str | PathLike, optional Pre-computed native HDF5 index or path to a JSON index file. @@ -2516,7 +2562,7 @@ def open( Format of a lazy remote source. A ``.zarr`` URL path component selects Zarr automatically; a ``.h5`` or ``.hdf5`` path selects HDF5 automatically; a ``.b2z`` path selects B2Z automatically. An explicit value supports - suffix-free array paths. Zarr and HDF5 sources automatically enable + suffix-free paths. Zarr, HDF5 and B2Z sources automatically enable ``lazy=True``. assume_immutable: bool, optional With ``lazy=True``, skip remote identity checks before reads. Defaults @@ -2525,7 +2571,8 @@ def open( Returns ------- out: :ref:`SChunk`, :ref:`NDArray`, :ref:`C2Array`, :ref:`RemoteArray`, - :ref:`Proxy`, :ref:`DictStore`, :ref:`EmbedStore`, or :ref:`TreeStore` + :ref:`Proxy`, :ref:`DictStore`, :ref:`EmbedStore`, :ref:`TreeStore`, + :class:`CTable`, :class:`RemoteCTable`, or :class:`RemoteStore` The object found in the path. Notes @@ -2554,12 +2601,14 @@ def open( endpoint URL, region, etc.) can be passed directly via ``storage_options``. ``mode != 'r'`` always raises, as object stores have no rename and no locks. A plain URL read rebuilds the object from a cframe held in memory, so it - covers ``.b2nd``, ``.b2f`` and ``.b2e`` only -- a ``.b2z`` store is a zip - archive rather than a cframe, and needs ``cache_dir`` like the directory - formats do. With ``lazy=True`` and a dataset path, a B2Z archive serves - its selected external NDArray by byte range. Lazy opening returns a :ref:`RemoteArray` (using - ``CachePolicy.DISK`` with ``cache_dir`` or ``cache_path``, and - ``CachePolicy.MEMORY`` otherwise). + covers ``.b2nd``, ``.b2f`` and ``.b2e`` only. Remote ``.b2z`` archives + default to lazy discovery: table nodes return :class:`RemoteCTable`, + groups return :class:`RemoteStore`, and external array leaves return + :ref:`RemoteArray`. Select a nested node with ``dataset=`` or ``::path``. + Tables and groups use ``cache_dir`` for DISK caching and default to MEMORY + otherwise; array leaves additionally support ``cache_path``. + Use ``lazy=False, cache_dir=...`` to download a complete archive and open + its root locally. Local table archives return :class:`CTable`. * Persistent data handling follows a no-hidden-writes rule except for an explicitly self-caching :ref:`RemoteArray`: diff --git a/tests/b2view/test_hierarchy.py b/tests/b2view/test_hierarchy.py index e415f29c9..239bd7086 100644 --- a/tests/b2view/test_hierarchy.py +++ b/tests/b2view/test_hierarchy.py @@ -50,9 +50,10 @@ def counted(self, path, start=None, end=None, **kwargs): return cat_file(self, path, start=start, end=end, **kwargs) monkeypatch.setattr(type(fs), "cat_file", counted) - with pytest.raises(NotImplementedError, match="B2Z containers"): - blosc2.open(url) - assert reads == [] + with blosc2.open(url) as store: + assert isinstance(store, blosc2.RemoteStore) + assert reads + reads.clear() with StoreBrowser(url) as browser: assert browser.is_tree assert [n.name for n in browser.list_children()] == ["group"] @@ -541,7 +542,7 @@ def test_b2z_large_embedded_chunk_notice(): def test_b2z_explicit_localization(tmp_path): url, data = b2z_url(tmp_path) - with blosc2.open(url, cache_dir=tmp_path / "localized", mode="r") as store: + with blosc2.open(url, cache_dir=tmp_path / "localized", mode="r", lazy=False) as store: assert isinstance(store, blosc2.TreeStore) np.testing.assert_array_equal(store["/group/a"][:2, :3], data[:2, :3]) diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 9c758b85b..14135df81 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -3,6 +3,8 @@ from __future__ import annotations import dataclasses +import itertools +import zipfile import numpy as np import pytest @@ -101,6 +103,76 @@ class Mixed: table["text"][:] +@pytest.mark.parametrize("policy", list(blosc2.CachePolicy)) +@pytest.mark.parametrize("null_storage", ["mask", "sentinel"]) +@pytest.mark.parametrize("deleted", [False, True]) +def test_remote_ctable_utf8(tmp_path, policy, null_storage, deleted): + @dataclasses.dataclass + class TextRow: + x: int + text: str = blosc2.field(blosc2.utf8(null_storage=null_storage)) + + values = ["", "café", "東京", "🌦️", None, "long" * 2000, "last"] + local = blosc2.CTable(TextRow, list(enumerate(values)), create_summary_index=False) + if deleted: + local.delete([1, 3]) + url = remote_table_url(tmp_path, local) + kwargs = {"cache_policy": policy} + if policy == blosc2.CachePolicy.DISK: + kwargs["cache_dir"] = tmp_path / "cache" + with blosc2.CTable.open(tmp_path / "table.b2z") as local, blosc2.RemoteCTable(url, **kwargs) as remote: + assert remote.nbytes == local.nbytes + assert remote.cbytes == local.cbytes + assert remote.cratio == local.cratio + raw = remote["text"].raw + assert isinstance(raw.offsets, blosc2.RemoteArray) + assert isinstance(raw.data, blosc2.RemoteArray) + for item in ( + 0, + -1, + slice(None), + slice(1, 4), + slice(None, None, 2), + slice(None, None, -1), + [2, 0, 2], + ): + np.testing.assert_array_equal(remote["text"][item], local["text"][item]) + np.testing.assert_array_equal(remote["text"].is_null(), local["text"].is_null()) + assert remote["text"].null_count() == local["text"].null_count() + assert list(remote["text"]) == list(local["text"]) + for expression in ('text == "東京"', 'text != ""', '(text >= "café") & (x > 0)'): + np.testing.assert_array_equal(remote.where(expression)["x"][:], local.where(expression)["x"][:]) + assert str(remote[:3]) + before = (len(raw), raw._pending[:], raw._pending_chars) + for write in ( + lambda: raw.append("new"), + lambda: raw.extend(["new"]), + lambda: raw.set_all(["new"]), + lambda: raw.__setitem__(0, "new"), + ): + with pytest.raises(ValueError, match="read-only"): + write() + assert (len(raw), raw._pending, raw._pending_chars) == before + copied = remote.copy() + assert type(copied) is blosc2.CTable + np.testing.assert_array_equal(copied["text"][:], local["text"][:]) + copied["text"][0] = "changed" + raw[:2] + requests = remote.traffic.requests + raw[:2] + if policy != blosc2.CachePolicy.NONE: + assert remote.traffic.requests == requests + for read in ( + lambda: raw[:0], + lambda: len(raw), + lambda: raw.dtype, + lambda: raw.nbytes, + lambda: raw.flush(), + ): + with pytest.raises(RuntimeError, match="closed"): + read() + + def test_remote_ctable_deleted_rows_and_disk_cache(tmp_path): source = str(tmp_path / "deleted.b2d") table = blosc2.CTable(Row, urlpath=source, mode="w", expected_size=8, create_summary_index=False) @@ -122,6 +194,154 @@ def test_remote_ctable_deleted_rows_and_disk_cache(tmp_path): np.testing.assert_array_equal(reopened["x"][:], [0, 2, 3, 5]) +@pytest.mark.parametrize("empty", [False, True]) +def test_remote_utf8_nested_lifetime(tmp_path, empty): + @dataclasses.dataclass + class TextRow: + text: str = blosc2.field(blosc2.utf8()) + + local = blosc2.CTable(TextRow, [] if empty else [("café",), ("東京",)], create_summary_index=False) + local.rename_column("text", "nested.text") + path = tmp_path / "tree.b2z" + with blosc2.TreeStore(path, mode="w", threshold=0) as tree: + tree["group/table"] = local + url = f"memory://{tmp_path.name}-tree.b2z" + fsspec.filesystem("memory").pipe(url, path.read_bytes()) + with blosc2.RemoteStore(url) as store: + table = store["group/table"] + raw = table["nested.text"].raw + np.testing.assert_array_equal(raw[:], local["nested.text"][:]) + assert raw.nbytes == local["nested.text"].raw.nbytes + view = table[:1] + store.refresh() + for read in (lambda: raw[:0], lambda: raw.shape, lambda: raw.cbytes, lambda: view["nested.text"][:]): + with pytest.raises(RuntimeError, match="stale"): + read() + table.close() + table = store["group/table"] + try: + np.testing.assert_array_equal(table["nested.text"][:], local["nested.text"][:]) + dest = tmp_path / "copy.b2z" + table.to_b2z(dest) + with blosc2.CTable.open(dest) as copied: + np.testing.assert_array_equal(copied["nested.text"][:], local["nested.text"][:]) + finally: + table.close() + + +@pytest.mark.parametrize("bad_data", [None, np.arange(4, dtype="int64")]) +def test_remote_utf8_invalid_companion(tmp_path, bad_data): + @dataclasses.dataclass + class TextRow: + x: int + text: str = blosc2.field(blosc2.utf8()) + + local = blosc2.CTable(TextRow, [(1, "abc")], create_summary_index=False) + path = tmp_path / "bad.b2z" + local.to_b2z(path) + # Repack only the companion, retaining all table metadata unchanged. + with zipfile.ZipFile(path) as archive: + members = {info.filename: archive.read(info) for info in archive.infolist()} + member = "_cols/text.utf8.b2nd" + if bad_data is None: + del members[member] + else: + members[member] = blosc2.asarray(bad_data).to_cframe() + broken = tmp_path / "broken.b2z" + with zipfile.ZipFile(broken, "w", zipfile.ZIP_STORED) as archive: + for name, data in members.items(): + archive.writestr(name, data) + url = f"memory://{tmp_path.name}-broken.b2z" + fsspec.filesystem("memory").pipe(url, broken.read_bytes()) + with blosc2.RemoteCTable(url) as table: + assert table["x"][0] == 1 + handles = len(table._storage._arrays) + for _ in range(2): + with pytest.raises((ValueError, NotImplementedError), match=r"(data array|unavailable)"): + table["text"][:] + assert len(table._storage._arrays) == handles + + +@pytest.mark.parametrize("policy", list(blosc2.CachePolicy)) +@pytest.mark.parametrize("blocks", [False, True]) +def test_remote_utf8_bounded_transfer(tmp_path, monkeypatch, policy, blocks): + from blosc2 import proxy_source + from blosc2._utf8_array import UTF8Array + + # Lower only the cost threshold so a small fixture exercises both native paths. + monkeypatch.setattr(proxy_source, "BLOCK_MIN_CBYTES", 0 if blocks else 1 << 30) + + @dataclasses.dataclass + class TextRow: + text: str = blosc2.field(blosc2.utf8()) + + rng = np.random.default_rng(7) + lengths = rng.integers(80, 120, size=20000) + offsets = np.concatenate(([0], np.cumsum(lengths))) + data = rng.integers(32, 127, size=offsets[-1], dtype="uint8") + values = [data[a:b].tobytes().decode() for a, b in itertools.pairwise(offsets)] + local = blosc2.CTable(TextRow, [(s,) for s in values], create_summary_index=False) + local._cols["text"] = UTF8Array( + blosc2.utf8(), + blosc2.asarray(offsets, chunks=(8192,), blocks=(256,)), + blosc2.asarray(data, chunks=(262144,), blocks=(4096,)), + ) + url = remote_table_url(tmp_path, local) + # Table export reconstructs UTF-8 arrays with default grids. Install the + # smaller fixture grids in the archive to exercise multi-chunk reads. + raw = local["text"].raw + replacements = { + "_cols/text.b2nd": raw.offsets.to_cframe(), + "_cols/text.utf8.b2nd": raw.data.to_cframe(), + } + path = tmp_path / "grids.b2z" + with zipfile.ZipFile(tmp_path / "table.b2z") as source, zipfile.ZipFile(path, "w") as dest: + for info in source.infolist(): + dest.writestr(info, replacements.get(info.filename, source.read(info))) + fsspec.filesystem("memory").pipe(url, path.read_bytes()) + kwargs = {"cache_policy": policy} + if policy != blosc2.CachePolicy.NONE: + kwargs["max_cache_bytes"] = 1 << 20 + if policy == blosc2.CachePolicy.DISK: + kwargs["cache_dir"] = tmp_path / "cache" + plans = [] + original = proxy_source.ByteRangeNDSource.block_plan + + def record_plan(source, *args): + plans.append(source.dataset) + return original(source, *args) + + monkeypatch.setattr(proxy_source.ByteRangeNDSource, "block_plan", record_plan) + with blosc2.RemoteCTable(url, **kwargs) as table: + assert not table._storage._arrays[1:] # Only validity is opened eagerly. + assert table.nbytes > 0 + assert table.cbytes > 0 + assert table.traffic.nbytes < (tmp_path / "table.b2z").stat().st_size // 4 + # Metadata prefetch can warm the entire small offsets member. + table["text"].raw.offsets.trim_cache(0) + table["text"].raw.data.trim_cache(0) + before = table.traffic.nbytes + np.testing.assert_array_equal(table["text"][10000:10003], values[10000:10003]) + assert table.traffic.nbytes - before < len(data) // 4 + if blocks: + assert set(plans) >= {"_cols/text", "_cols/text.utf8"} + else: + assert not plans + before = table.traffic.requests + np.testing.assert_array_equal(table["text"][10000:10003], values[10000:10003]) + if policy != blosc2.CachePolicy.NONE: + assert table.traffic.requests == before + assert table.cache_bytes <= kwargs.get("max_cache_bytes", 0) + boundary = int(np.searchsorted(offsets, 262144)) + for item in (slice(8190, 8195), slice(boundary - 1, boundary + 2), slice(9990, 10010, 3)): + np.testing.assert_array_equal(table["text"][item], values[item]) + if policy == blosc2.CachePolicy.DISK: + with blosc2.RemoteCTable(url, **kwargs) as table: + before = table.traffic.requests + np.testing.assert_array_equal(table["text"][10000:10003], values[10000:10003]) + assert table.traffic.requests == before + + def test_remote_store_rejects_table_root(tmp_path): url = remote_table_url( tmp_path, @@ -130,3 +350,79 @@ def test_remote_store_rejects_table_root(tmp_path): ) with pytest.raises(ValueError, match="use RemoteCTable"): blosc2.RemoteStore(url) + + +@pytest.mark.parametrize("policy", list(blosc2.CachePolicy)) +def test_open_dispatches_local_and_remote_tables(tmp_path, policy): + @dataclasses.dataclass + class TextRow: + text: str = blosc2.field(blosc2.utf8(null_storage="mask")) + + local = blosc2.CTable(TextRow, [("café",), (None,), ("東京",)], create_summary_index=False) + url = remote_table_url(tmp_path, local) + with blosc2.open(tmp_path / "table.b2z") as table: + assert type(table) is blosc2.CTable + assert table["text"][0] == "café" + options = {"cache_policy": policy} + if policy == blosc2.CachePolicy.DISK: + options["cache_dir"] = tmp_path / "cache" + with blosc2.open(url, **options) as table: + assert isinstance(table, blosc2.RemoteCTable) + assert table.cache_policy == policy + assert table["text"][-1] == "東京" + assert table["text"].null_count() == 1 + with pytest.raises(RuntimeError, match="closed"): + table["text"][:] + with blosc2.open(url) as table: + assert table.cache_policy == blosc2.CachePolicy.MEMORY + + +@pytest.mark.parametrize("suffix", [".b2z", ""]) +def test_open_dispatches_remote_table_hierarchy(tmp_path, suffix, monkeypatch): + local = blosc2.CTable(Row, [(1, [1, 2], "one")], create_summary_index=False) + path = tmp_path / "tree.b2z" + with blosc2.TreeStore(path, mode="w", threshold=0) as tree: + tree["group/table"] = local + tree["group/array"] = blosc2.arange(5) + url = f"memory://{tmp_path.name}-tree{suffix}" + format_options = {} if suffix else {"source_format": "b2z"} + fsspec.filesystem("memory").pipe(url, path.read_bytes()) + reads = [] + filesystem_type = type(fsspec.filesystem("memory")) + original = filesystem_type.cat_file + + def counted(self, path, start=None, end=None, **kwargs): + result = original(self, path, start=start, end=end, **kwargs) + reads.append(len(result)) + return result + + monkeypatch.setattr(filesystem_type, "cat_file", counted) + with blosc2.open(url, **format_options) as store: + assert isinstance(store, blosc2.RemoteStore) + table = store["group/table"] + assert table["x"][0] == 1 + table.close() + for target, options in ( + (url, {"dataset": "group/table"}), + (url + "::group/table", {}), + (url + "/group/table", {"lazy": True}), + ): + if not suffix and target != url: + continue # URL suffix parsing requires a recognizable archive suffix. + reads.clear() + with blosc2.open(target, **options, **format_options) as table: + assert isinstance(table, blosc2.RemoteCTable) + assert table["x"][0] == 1 + assert table.traffic.requests == len(reads) + assert table.traffic.nbytes == sum(reads) + with blosc2.open(url, dataset="group", **format_options) as group: + assert isinstance(group, blosc2.RemoteStore) + assert group.keys() == ["array", "table"] + group.refresh() + assert group.keys() == ["array", "table"] + with blosc2.open(url, dataset="group/array", **format_options) as array: + assert isinstance(array, blosc2.RemoteArray) + np.testing.assert_array_equal(array[:], np.arange(5)) + with blosc2.open(url, lazy=False, cache_dir=tmp_path / "localized", **format_options) as store: + assert isinstance(store, blosc2.TreeStore) + assert store["group/table"]["x"][0] == 1 diff --git a/tests/test_b2z_source.py b/tests/test_b2z_source.py index eb9ea4922..76c6eb021 100644 --- a/tests/test_b2z_source.py +++ b/tests/test_b2z_source.py @@ -91,7 +91,8 @@ def counted(self, path, start=None, end=None, **kwargs): @pytest.mark.parametrize("suffix", ["supported", "ignored", "rejected", "malformed"]) @pytest.mark.parametrize("small", [False, True]) -def test_http_tail_bootstrap(suffix, small): +@pytest.mark.parametrize("ctable", [False, True]) +def test_http_tail_bootstrap(suffix, small, ctable, tmp_path): import http.server import threading @@ -100,6 +101,18 @@ def test_http_tail_bootstrap(suffix, small): pytest.importorskip("aiohttp") url, data = memory_archive(np.zeros((40, 250), dtype="uint8") if small else None) body = fsspec.filesystem("memory").cat_file(url) + if ctable: + import dataclasses + + @dataclasses.dataclass + class TextRow: + text: str = blosc2.field(blosc2.utf8()) + + values = ["", "café", "東京", "🌦️"] * (1 if small else 10000) + table = blosc2.CTable(TextRow, [(v,) for v in values], create_summary_index=False) + path = tmp_path / "table.b2z" + table.to_b2z(path) + body = path.read_bytes() requests = [] class Handler(http.server.BaseHTTPRequestHandler): @@ -141,13 +154,24 @@ def do_GET(self): try: url = f"http://127.0.0.1:{server.server_port}/array.b2z" options = {"headers": {"X-Test": "preserved"}, "skip_instance_cache": True} - with blosc2.open(url + "::/d0/a", storage_options=options) as arr: + remote = ( + blosc2.RemoteCTable(url, storage_options=options) + if ctable + else blosc2.open(url + "::/d0/a", storage_options=options) + ) + with remote as arr: assert requests[0] == ("GET", "bytes=-8192") assert sum(method == "HEAD" for method, _ in requests) == (suffix != "supported") - if suffix == "supported": + if suffix == "supported" and not ctable: assert len(requests) == (1 if small else 2) assert arr.traffic.nbytes == (len(body) if small else 8192 + 16384) - np.testing.assert_array_equal(arr[:], data) + if ctable: + np.testing.assert_array_equal(arr["text"][-4:], values[-4:]) + before = len(requests) + np.testing.assert_array_equal(arr["text"][-4:], values[-4:]) + assert len(requests) == before + else: + np.testing.assert_array_equal(arr[:], data) # A persisted suffix bootstrap must have the same identity as HEAD. metadata = {} @@ -291,13 +315,20 @@ def test_opening_buffer_fallbacks(variant): np.testing.assert_array_equal(arr[:], data) -@pytest.mark.parametrize("dataset", [None, "", "/", "d0", "missing", "../d0/a", "d0//a", "d0/./a", "d0/\na"]) +@pytest.mark.parametrize("dataset", ["missing", "../d0/a", "d0//a", "d0/./a", "d0/\na"]) def test_bad_datasets(dataset): url, _ = memory_archive() with pytest.raises(ValueError): blosc2.open(url, lazy=True, dataset=dataset) +@pytest.mark.parametrize("dataset", [None, "", "/", "d0"]) +def test_open_b2z_groups(dataset): + url, _ = memory_archive() + with blosc2.open(url, dataset=dataset) as store: + assert isinstance(store, blosc2.RemoteStore) + + def test_options_and_bad_archives(): url, _ = memory_archive() with pytest.raises(NotImplementedError, match="mutable B2Z"): diff --git a/tests/test_fsspec.py b/tests/test_fsspec.py index 92c12ebe1..a3d1ba8e1 100644 --- a/tests/test_fsspec.py +++ b/tests/test_fsspec.py @@ -906,17 +906,17 @@ def test_http_store_disk_reopen_and_transport_close(tmp_path): assert len(requests) == count -def test_zip_store_needs_cache(tmp_path): - # A .b2z store is a zip archive, not a cframe, so there is nothing for the - # in-memory read to rebuild +def test_zip_store_lazy_default_and_explicit_localization(tmp_path): localpath = str(tmp_path / "t.b2z") with blosc2.TreeStore(localpath, mode="w") as tstore: tstore["/a"] = blosc2.arange(10, dtype="i4") fsspec.filesystem("memory").pipe_file("/t.b2z", pathlib.Path(localpath).read_bytes()) - with pytest.raises(RuntimeError): - blosc2.open("memory://t.b2z") - with blosc2.open("memory://t.b2z", cache_dir=tmp_path / "cache") as tstore: + with blosc2.open("memory://t.b2z") as tstore: + assert isinstance(tstore, blosc2.RemoteStore) + np.testing.assert_array_equal(tstore["/a"][:], np.arange(10, dtype="i4")) + with blosc2.open("memory://t.b2z", cache_dir=tmp_path / "cache", lazy=False) as tstore: + assert isinstance(tstore, blosc2.TreeStore) assert np.array_equal(tstore["/a"][:], np.arange(10, dtype="i4")) From 68e42b65b1483868b8743e1bf88f4f64382bc6ad Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 18 Sep 2026 08:29:50 +0200 Subject: [PATCH 04/82] Add bounded parallel reads for RemoteCTable --- bench/remote_ctable_metadata.md | 167 ++++++++++++ bench/remote_ctable_metadata.py | 145 ++++++++++ doc/guides/remote_arrays.md | 34 +++ examples/ctable/remote_handling.py | 23 +- plans/remote-ctable-parallel.md | 410 +++++++++++++++++++++++++++++ src/blosc2/b2z_source.py | 36 ++- src/blosc2/ctable.py | 159 +++++++++-- src/blosc2/ctable_remote_read.py | 362 +++++++++++++++++++++++++ src/blosc2/ctable_storage.py | 24 +- src/blosc2/proxy_source.py | 30 ++- src/blosc2/remote_ctable.py | 64 ++++- src/blosc2/schunk.py | 27 +- tests/ctable/test_remote_ctable.py | 301 +++++++++++++++++++++ 13 files changed, 1743 insertions(+), 39 deletions(-) create mode 100644 bench/remote_ctable_metadata.md create mode 100644 bench/remote_ctable_metadata.py create mode 100644 plans/remote-ctable-parallel.md create mode 100644 src/blosc2/ctable_remote_read.py diff --git a/bench/remote_ctable_metadata.md b/bench/remote_ctable_metadata.md new file mode 100644 index 000000000..87034a055 --- /dev/null +++ b/bench/remote_ctable_metadata.md @@ -0,0 +1,167 @@ +# Parallel RemoteCTable metadata and row-read experiment + +The script now benchmarks the production public API. Run: + +```bash +conda run -n blosc2 python bench/remote_ctable_metadata.py --workers 1 2 4 8 --repeats 3 +``` + +One worker is the production serial control, including header reuse. The old +`workers=0` baseline and replay implementation are no longer in the script; +their historical results below are retained for comparison. `--serial-rows` +now selects one worker for rows while preserving metadata concurrency. + +Measured on 2026-09-17 in the `blosc2` conda environment, against +`https://f001.backblazeb2.com/file/blosc2/readings.b2z` (200,000 rows, seven columns, +including nullable UTF-8 notes). Library baseline: commit `626abef9`. + +## Reproduce + +```bash +conda run -n blosc2 python bench/remote_ctable_metadata.py --workers 0 1 2 4 8 --repeats 3 --serial-rows +``` + +The script emits per-run JSON followed by median timings. Every run opens a new +table with MEMORY caching and a new HTTP filesystem/session. Run order alternates +ascending/descending concurrency. DNS, TLS session caches, server caches and +network conditions are not controlled. No remote files are created or changed. + +Workers=0 uses the unmodified library. Workers=1 buffers the same eight column +header ranges but retrieves them serially, separating the effects of reuse and +concurrency. Workers=2/4/8 retrieve those same ranges concurrently. Timings include +the temporary buffer's creation and all HTTP requests needed by metadata display. + +## Metadata-only results (2026-09-17) + +Medians of three runs, seconds: + +| Mode | Bare open | Column metadata | Open + metadata | First five rows | Total to first rows | +| --- | ---: | ---: | ---: | ---: | ---: | +| Library baseline | 0.765 | 2.840 | 3.613 | 2.673 | 6.306 | +| Buffered, serial | 0.776 | 1.958 | 2.735 | 2.649 | 5.383 | +| Buffered, 2 workers | 0.769 | 1.255 | 2.026 | 2.960 | 4.985 | +| Buffered, 4 workers | 0.767 | 0.995 | 1.781 | 3.198 | 4.978 | +| Buffered, 8 workers | 0.768 | 0.830 | 1.608 | 3.555 | 5.163 | + +Combined columns are medians of each run's sums, not sums of medians. Total excludes +the subsequent warm read, handle cleanup and interpreter startup. All columns of +the first five rows are materialized; pandas display truncation cannot silently +reduce the work as it can in a `str(table[:5])` benchmark. + +Opening itself consistently used two requests / 9,899 bytes. Metadata inspection +then used 13 requests / 132,987 bytes in the library baseline, versus eight +requests / 132,581 bytes in every buffered case. Temporary header reuse therefore +eliminated five small requests independently of parallelism. Maximum simultaneous +range calls was measured as 1, 1, 2, 4 and 8 respectively. + +Cold row reads consistently used ten requests / 941,623 bytes in every mode. +Repeated row reads issued no requests and transferred no bytes. Retained payload +after the row reads was identical across modes: 990,248 bytes. Metadata and row +values were checked for equality across all fifteen runs. + +Eight workers improved column-metadata inspection by 3.42x versus the library +baseline, or 2.36x versus the serial buffered control. Including bare opening, +the improvement was 2.25x (3.613 to 1.608 seconds). + +However, the later cold row reads were slower after higher-concurrency metadata +reads. This was repeatable in these samples, but the cause was not diagnosed. +The best total time here was effectively tied between two and four workers: +about 4.98 seconds versus 6.31 seconds, a 21% reduction. Eight workers were best +for metadata alone, not for time to first rows. More concurrency is not an +unqualified improvement for the whole access sequence. + +## Metadata-only prototype boundaries + +This experiment changes no library code or default behavior. It holds the +existing owner lock, fetches independent byte ranges with a standard-library +thread pool, then constructs sources and populates caches serially. It does not +run ZIP readers concurrently or weaken the shared cache/lifetime locks. + +The temporary buffer follows the existing small-member/header prefetch sizes, +is limited to 8 MiB, and is released after metadata inspection. It is deliberately +limited to fresh MEMORY-cached tables with external fixed-width/UTF-8 members. +It does not prefetch null masks, row payloads, or the entire archive. The byte +ranges may include small payloads, exactly as ordinary header prefetch does. + +A production implementation should batch metadata range reads only when several +columns are actually requested, keeping individual column access lazy. Reuse +overlapping header bytes through the batch, then parse/register sources on the +owner thread. It should bound batch memory, handle DISK/NONE policies and failed +opens, and measure the observed row-read slowdown before choosing a default +concurrency. Cross-column row payload batching remains a separate experiment. + +Validation: all 33 RemoteCTable tests passed, including a deterministic delayed +memory-filesystem check for overlapping requests, result/traffic equivalence and +cleanup after an injected range failure. Ruff check and formatting passed. + +## Parallel row-read extension (2026-09-18) + +Omit `--serial-rows` to parallelize row reads as well as metadata. Workers=0 +still uses the unchanged library baseline; workers=1 runs the same buffered row +algorithm serially. JSON now includes `row_workers` and `peak_row_reads`. + +The row experiment uses the existing column readers to discover the next missing +range for each column, fetches these ranges concurrently, then resumes those +readers on the calling thread. This repeats until all requested columns are +cached. UTF-8 offsets therefore arrive before their dependent byte ranges; the +existing compressed-block selection and null handling remain in use. Row +assembly, source parsing and cache updates never run in worker threads. + +This deliberately avoids duplicating the library's block planner. It is a +benchmark-only replay mechanism, not a production API: temporary row buffers +are capped at 32 MiB, MEMORY caching is required, and reads are limited to the +first five rows. Production integration should use explicit batched planning +instead of intercepting and replaying reads. Existing library/example behavior +is unchanged. + +Medians of three fresh-session runs, seconds: + +| Workers | Open + metadata | Cold rows | Total to first rows | Peak row requests | +| --- | ---: | ---: | ---: | ---: | +| 0 (library baseline) | 3.736 | 2.735 | 6.603 | 1 | +| 1 (serial buffered control) | 2.778 | 2.933 | 5.617 | 1 | +| 2 | 2.012 | 2.134 | 4.173 | 2 | +| 4 | 1.782 | 1.810 | 3.592 | 3 | +| 8 | 1.603 | 2.020 | 3.605 | 3 | + +Four workers reduced total latency by 46% versus the library baseline (1.84x +faster), and cold-row latency by 34%. Eight workers did not improve overall +latency: the row phase reached only three simultaneous range calls, with +dependencies between waves. Network variability still applies. + +A subsequent three-run `--workers 4 --serial-rows` control measured 1.741 seconds +for open + metadata, 3.386 seconds for cold rows, and 5.126 seconds total. Against +that metadata-only control, parallel rows cut row latency by 47% (1.87x faster) +and total latency by 30%. These control runs followed rather than interleaved +with the parallel runs, so the comparison includes possible network drift. + +All modes fetched exactly ten row requests / 941,623 bytes and retained 990,248 +cache bytes. All fifteen runs returned identical metadata and rows; every warm +read used zero requests. The extended delayed-memory-filesystem test uses large +uncompressed columns so row payloads are not swallowed by header prefetch; it +checks overlap, serial/parallel values and traffic, cache sizes, and restoration +and successful retry after a transport error. All 33 RemoteCTable tests and Ruff +checks passed. + +## Production implementation results (2026-09-18) + +Three-run HTTPS medians, using `blosc2.open(..., max_concurrency=workers)`: + +| Workers | Open + metadata | Cold rows | Total to first rows | +| --- | ---: | ---: | ---: | +| 1 | 2.842 | 2.701 | 5.544 | +| 2 | 2.260 | 2.313 | 4.538 | +| 4 | 1.736 | 1.914 | 3.656 | +| 8 | 1.613 | 2.105 | 3.701 | + +All outputs matched. Metadata inspection used eight requests / 132,581 bytes; +rows used ten requests / 941,623 bytes. Warm reads made no requests. Retained +cache was 990,248 bytes in every mode. With eight workers the peak temporary +reservations were 132,581 metadata bytes and 925,252 row bytes; peak concurrent +requests were eight and three respectively. These counters exclude decoded +output, cache memory and transport/native overhead. + +The production planner uses explicit dependency waves, not monkeypatching or +exception-driven replay. Instrumentation still wraps `cat_file` solely to count +concurrent requests. See `plans/remote-ctable-parallel.md` for scope, correctness +checks and the separate 70-column fresh-process RSS experiment. diff --git a/bench/remote_ctable_metadata.py b/bench/remote_ctable_metadata.py new file mode 100644 index 000000000..afa22a585 --- /dev/null +++ b/bench/remote_ctable_metadata.py @@ -0,0 +1,145 @@ +#!/usr/bin/env python3 +####################################################################### +# Copyright (c) 2019-present, Blosc Development Team +# All rights reserved. +# SPDX-License-Identifier: BSD-3-Clause +####################################################################### +"""Benchmark public RemoteCTable parallel reads with fresh HTTPS sessions. + + python bench/remote_ctable_metadata.py URL --workers 1 2 4 8 --repeats 3 + +One worker includes metadata reuse; it is not the historical library baseline. +""" + +import argparse +import json +import statistics +import threading +import time +from contextlib import contextmanager +from unittest.mock import patch + +import blosc2 + +DEFAULT_URL = "https://f001.backblazeb2.com/file/blosc2/readings.b2z" + + +def metadata(table): + return { + "nbytes": table.nbytes, + "chunks": table.chunks, + "blocks": table.blocks, + "schema": table.schema_dict(), + "attrs": dict(table.attrs), + } + + +@contextmanager +def count_concurrent_reads(filesystem): + """Observe actual requests without changing planning or buffering.""" + original = filesystem.cat_file + lock = threading.Lock() + counts = {"active": 0, "peak": 0} + + def counted(*args, **kwargs): + with lock: + counts["active"] += 1 + counts["peak"] = max(counts["peak"], counts["active"]) + try: + return original(*args, **kwargs) + finally: + with lock: + counts["active"] -= 1 + + with patch.object(filesystem, "cat_file", counted): + yield counts + + +def measured(table, operation): + before_bytes, before_requests = table.traffic.nbytes, table.traffic.requests + started = time.perf_counter() + result = operation() + return result, { + "seconds": time.perf_counter() - started, + "requests": table.traffic.requests - before_requests, + "bytes": table.traffic.nbytes - before_bytes, + } + + +def run(url, workers, row_workers=None): + started = time.perf_counter() + with blosc2.open(url, max_concurrency=workers, storage_options={"skip_instance_cache": True}) as table: + if not isinstance(table, blosc2.RemoteCTable): + raise ValueError("URL must identify a remote CTable") + storage = table._remote_storage() + opened = { + "seconds": time.perf_counter() - started, + "requests": table.traffic.requests, + "bytes": table.traffic.nbytes, + } + with count_concurrent_reads(storage._owner.filesystem) as counts: + description, inspected = measured(table, lambda: metadata(table)) + metadata_peak = storage._peak_metadata_buffer_bytes + table.max_concurrency = workers if row_workers is None else row_workers + with count_concurrent_reads(storage._owner.filesystem) as row_counts: + rows, cold = measured(table, lambda: list(table[:5])) + row_peak = storage._peak_row_buffer_bytes + warm_rows, warm = measured(table, lambda: list(table[:5])) + assert repr(rows) == repr(warm_rows) + assert warm["requests"] == warm["bytes"] == 0 + result = { + "workers": workers, + "row_workers": table.max_concurrency, + "open": opened, + "metadata": inspected, + "cold_rows": cold, + "warm_rows": warm, + "peak_metadata_reads": counts["peak"], + "peak_row_reads": row_counts["peak"], + "peak_metadata_buffer_bytes": metadata_peak, + "peak_row_buffer_bytes": row_peak, + "cache_bytes": table.cache_bytes, + } + result["total_seconds"] = sum(result[s]["seconds"] for s in ("open", "metadata", "cold_rows")) + return result, (description, repr(rows)) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("url", nargs="?", default=DEFAULT_URL) + parser.add_argument("--workers", nargs="+", type=int, default=[1, 2, 4, 8]) + parser.add_argument("--repeats", type=int, default=3) + parser.add_argument("--serial-rows", action="store_true", help="Use one worker for row reads") + args = parser.parse_args() + if args.repeats < 1 or any(n < 1 for n in args.workers): + parser.error("repeats and workers must be positive (1 = production serial control)") + results, expected = [], None + for repeat in range(args.repeats): + order = args.workers if repeat % 2 == 0 else args.workers[::-1] + for workers in order: + result, values = run(args.url, workers, row_workers=1 if args.serial_rows else None) + if expected is None: + expected = values + assert values == expected, "Metadata or row values changed between runs" + result["repeat"] = repeat + 1 + results.append(result) + print(json.dumps(result), flush=True) + print("\nMedians (seconds; fresh table and HTTP session per run)") + print("workers=1: production serial control (includes header reuse)") + print("workers open column metadata open+metadata cold rows total peak meta/rows") + for workers in args.workers: + runs = [r for r in results if r["workers"] == workers] + med = { + s: statistics.median(r[s]["seconds"] for r in runs) for s in ("open", "metadata", "cold_rows") + } + opening = statistics.median(r["open"]["seconds"] + r["metadata"]["seconds"] for r in runs) + total = statistics.median(r["total_seconds"] for r in runs) + peak = max(r["peak_metadata_reads"] for r in runs) + print( + f"{workers:7d} {med['open']:7.3f} {med['metadata']:16.3f} {opening:14.3f} " + f"{med['cold_rows']:10.3f} {total:8.3f} {peak:6d}/{max(r['peak_row_reads'] for r in runs)}" + ) + + +if __name__ == "__main__": + main() diff --git a/doc/guides/remote_arrays.md b/doc/guides/remote_arrays.md index df7bbbc32..b0d75cdd8 100644 --- a/doc/guides/remote_arrays.md +++ b/doc/guides/remote_arrays.md @@ -567,6 +567,40 @@ metadata without scanning strings. Small-member metadata prefetch may also fetch some payload. Repeated reads can reuse cached blocks; filtering scans the required columns because persisted indexes are not used remotely. +Multi-column metadata inspection and row materialization overlap independent +requests by default, up to eight at once. This includes row iteration, display, +and batched Arrow/pandas export; single-column access remains lazy. UTF-8 byte +requests wait for their offsets. Cache publication and decoding stay serialized. + +```python +with blosc2.open(url, max_concurrency=1) as table: # serial control + rows = list(table[:10]) + +with blosc2.RemoteCTable( + url, + max_concurrency=8, + metadata_buffer_bytes=8 << 20, + row_buffer_bytes=64 << 20, +) as table: + table.row_buffer_bytes = 256 << 20 # optional explicit override + rows = list(table[:10]) +``` + +The fixed defaults are 8 MiB of temporary metadata and 64 MiB of temporary row +data, allocated on demand. They do not depend on CPU count or available RAM. +Wider reads use bounded batches; a single oversized required unit runs alone. +These are soft transport budgets, not total RAM limits: decoded output, native +scratch, HTTP overhead and retained caches are additional. DISK caching does not +remove the need to bound temporary reads or consume large outputs in batches. +The buffer keywords belong to RemoteCTable, not `blosc2.open()`; settings may +also be changed on a returned table, including one obtained from RemoteStore. + +The existing 1 MiB compressed-chunk threshold is a block-selection heuristic, +not a maximum response size. Large selections and unsupported partial-block +layouts can still fetch whole chunks. Cross-process shared-cache handles and +read-only artifacts retain their existing guarded row-read paths; standalone +RemoteArray concurrency and RemoteStore discovery behavior are unchanged. + Columns and views are borrowed from the root table and require it to remain open. Closing a parent RemoteStore leaves a returned table usable; refreshing the store invalidates previously returned tables and their columns. Copies and data exports diff --git a/examples/ctable/remote_handling.py b/examples/ctable/remote_handling.py index 7f2efa927..b6f94aed2 100644 --- a/examples/ctable/remote_handling.py +++ b/examples/ctable/remote_handling.py @@ -138,6 +138,11 @@ def access_table(args) -> None: "attrs": dict(table.attrs), } if remote: + metadata["max_concurrency"] = table.max_concurrency + metadata["temporary_buffers"] = { + "metadata": table.metadata_buffer_bytes, + "rows": table.row_buffer_bytes, + } metadata["cache_policy"] = table.cache_policy.name metadata["cache_bytes"] = f"{table.cache_bytes} ({table.cache_bytes / 1024:.2f} KiB)" metadata_time = time.perf_counter() - started @@ -151,16 +156,19 @@ def access_table(args) -> None: rendered = "{\n " + rendered[1:] print(f"{name:<13}: {rendered}") - print("\nSample rows (1st fetch):") + sample_start = max(0, table.nrows // 2 - 2) + sample_stop = min(sample_start + 5, table.nrows) + sample_slice = slice(sample_start, sample_stop) + print(f"\nSample rows [{sample_start}:{sample_stop}] around the midpoint (1st fetch):") started = time.perf_counter() - sample = str(table[:5]) + sample = str(table[sample_slice]) first_time = time.perf_counter() - started first_bytes = table.traffic.nbytes - metadata_bytes if remote else 0 first_requests = table.traffic.requests - metadata_requests if remote else 0 print(sample) started = time.perf_counter() - _ = str(table[:5]) + _ = str(table[sample_slice]) second_time = time.perf_counter() - started second_bytes = table.traffic.nbytes - metadata_bytes - first_bytes if remote else 0 second_requests = table.traffic.requests - metadata_requests - first_requests if remote else 0 @@ -179,6 +187,8 @@ def access_table(args) -> None: f" - 2nd row fetch : {second_time * 1000:7.1f} ms " f"({second_requests} requests, {second_bytes / 1024:8.2f} KB transferred){cache_hit}" ) + # Sum operation wall times (including decoding/cache work), not printing. + total_time = metadata_time + first_time + second_time if "note" in table.col_names and table["note"].is_utf8: start = max(0, table.nrows - 5) print(f"\nUTF-8 note slice [{start}:{table.nrows}] (may overlap warmed blocks in small tables):") @@ -188,6 +198,7 @@ def access_table(args) -> None: started = time.perf_counter() notes = table["note"][start:] elapsed = time.perf_counter() - started + total_time += elapsed transferred = table.traffic.nbytes - before_bytes if remote else 0 requests = table.traffic.requests - before_requests if remote else 0 print( @@ -198,10 +209,10 @@ def access_table(args) -> None: print(f" {notes}") if remote: print( - f" - Total network : {table.traffic.requests} requests, " - f"{table.traffic.nbytes / 1024:8.2f} KB transferred" + f"\nTotal network : {total_time * 1000:7.1f} ms " + f"({table.traffic.requests} requests, {table.traffic.nbytes / 1024:8.2f} KB transferred)" ) - print(f" - Retained cache: {table.cache_bytes / 1024:8.2f} KB") + print(f"Retained cache: {table.cache_bytes / 1024:8.2f} KB") def main() -> int: diff --git a/plans/remote-ctable-parallel.md b/plans/remote-ctable-parallel.md new file mode 100644 index 000000000..e0e357d93 --- /dev/null +++ b/plans/remote-ctable-parallel.md @@ -0,0 +1,410 @@ +# RemoteCTable parallel metadata and row reads + +Status: production implementation completed on 2026-09-18. Verification and +explicit scope notes are recorded below; the design sections retain the agreed +implementation requirements. + +## Goal and boundaries + +Overlap independent remote requests across the columns a RemoteCTable operation +actually needs. Reduce metadata inspection and time to first rows without +changing values, transfer granularity, read-only semantics, or lazy selection. + +Scope is RemoteCTable backed by the existing fsspec B2Z reader, including tables +obtained from `blosc2.open()` and nested tables obtained through RemoteStore. +Fixed-width, shaped fixed-width and UTF-8 columns, nullable companions, and live +row selection retain their existing semantics. No archive format changes. + +RemoteArray already supports concurrency within an array. This work adds the +cross-column information that an individual array cannot see. Shared private +helpers may be factored out, but RemoteArray and RemoteStore public behavior, +defaults and independent reads must remain unchanged. Local CTable operations +must remain unchanged. Do not add eager store-wide loading, a general scheduler, +automatic concurrency tuning, new dependencies, or parallel decompression. + +## Evidence and agreed defaults + +See `bench/remote_ctable_metadata.py` and `bench/remote_ctable_metadata.md`. +The public HTTPS `readings.b2z` contains 200,000 rows and seven columns, with +eight primary array members because UTF-8 uses offsets and bytes separately. + +Three-run prototype medians: library baseline 6.60 seconds to first rows, +four workers 3.59 seconds, eight workers 3.61 seconds. Four-worker cold rows +took 1.81 seconds versus 3.39 seconds in a subsequent metadata-only control. +These are illustrative network measurements, not contractual speedups. + +Metadata can expose eight independent header requests. Row reads on this table +reach three concurrent requests because small members were already prefetched; +remaining indexing/payload requests have dependencies. Scale with ready requests, +not schema column count or machine CPU count. + +| Setting | Default | Meaning | +| --- | ---: | --- | +| `max_concurrency` | 8 | Maximum in-flight requests for one table operation | +| `metadata_buffer_bytes` | 8 MiB (`8 << 20`) | Temporary metadata scheduling budget | +| `row_buffer_bytes` | 64 MiB (`64 << 20`) | Temporary compressed row-data scheduling budget | + +Allocate on demand; never reserve these amounts up front. Concurrency is at most +`min(max_concurrency, ready_request_count)` and can be lower to satisfy memory +budgets. One worker is a serial implementation of the same production planner, +including header reuse, not a switch back to the old implementation. + +Keep these defaults fixed across machines: no RAM detection, CPU-based sizing, +or container-memory heuristics. The 64 MiB row default replaces the initially +proposed 256 MiB to leave more headroom on the target minimum 4-core, 1 GB +machines. Users may explicitly raise it to 256 MiB or another value when larger +reads and measurements justify that tradeoff. Metadata defaults to 8 MiB; +larger metadata workloads use bounded batches rather than accumulating all +prefixes at once. Both budgets remain configurable and are separate +per-operation allowances, not a combined process-memory cap. + +These budgets are not total RSS limits. Retained caches, decoded output, native +scratch space, HTTP buffering and concurrent operations on unrelated owners are +additional. In the benchmark the measured retained cache was 990,248 bytes; +row traffic was 941,623 bytes. Five returned rows are small, but neither their +size nor compressed traffic measures peak decoding allocations. The conservative +default does not guarantee that arbitrary reads fit a 1 GB machine: retained +cache limits and bounded result consumption still matter. + +### Transfer granularity is separate from buffer budgets + +Preserve the existing HTTP/B2Z block-selection policy. Its 1 MiB compressed-chunk +threshold is a minimum for considering partial-block reads, not a maximum +download size. Blocks are normally selected only when at most half the chunk's +blocks are needed and the layout and range-request cost permit it. Larger +selections, unsupported layouts (including MEMCPYED chunks and codec dictionary +cases), or supported whole-chunk fallback can still require a large chunk. +Small members of at most 64 KiB may already be fully prefetched during opening. + +For example, eight requests for one 256 KiB block each require about 2 MiB of +response payload if compression gives little reduction; eight whole 8 MiB chunk +responses require about 64 MiB instead. These estimates exclude completed +responses awaiting consumption, assembly and other client memory. Budget actual +compressed ranges and outstanding assembly, not column count times 1 MiB. +UTF-8 offsets and bytes are separate backing arrays with dependencies. Table +width is not capped by the temporary allowance: process wider tables in batches +and release consumed responses, rather than buffer every column at once. + +## Public configuration and dispatch + +1. Add the three keyword options above to RemoteCTable construction. Require + positive integer values (reject booleans); `max_concurrency=1` is supported. + Preserve existing cache options and keep `max_cache_bytes` independent. +2. Expose validated runtime settings on RemoteCTable so a table obtained from + `store["table"]` can be configured without adding RemoteStore options. + Keep settings in table storage, not the shared owner: independent table + handles sharing an archive must not overwrite each other's configuration. + Borrowed views use their base table's settings; snapshot settings at operation + entry under the existing lock. Do not persist them in archive metadata or + cache identity. Fresh handles use defaults. +3. Extend only `max_concurrency` in `blosc2.open()` to RemoteCTable targets. + Audit option extraction and `_open_remote_b2z()` in `schunk.py`: it currently + rejects table/store `max_concurrency`. Preserve the existing RemoteArray + interpretation of that option. Do not add `metadata_buffer_bytes` or + `row_buffer_bytes` to the general opener; reject them there rather than + forwarding or silently discarding them, even for table targets. Recommend + direct RemoteCTable construction for advanced buffer tuning, or validated + instance settings after `blosc2.open()`/RemoteStore lookup. A group target + must retain its current concurrency-option behavior. +4. Thread configuration through `_from_owner()` and RemoteTableStorage before + table reads begin. RemoteStore's ordinary table lookup uses defaults without + changing store discovery or eager-loading behavior. + +Intended usage after implementation: + +```python +# General opener: shared remote concurrency control, default table budgets. +table = blosc2.open(url, max_concurrency=8) + +# Table-specific tuning belongs on the table interface. +table = blosc2.RemoteCTable( + url, + max_concurrency=8, + metadata_buffer_bytes=8 << 20, + row_buffer_bytes=64 << 20, +) +``` + +For standalone remote arrays, `max_concurrency` already bounds independent +fetch tasks within an array read: whole chunks, or block-layout/payload range +tasks in their respective dependency waves. It does not change block-selection +thresholds, decompression thread counts or cache budgets. Preserve that behavior; +the new table path extends request overlap across selected columns instead. + +## Production architecture + +### Ownership and threading + +Use a standard-library ThreadPoolExecutor scoped to an active table operation; +skip pool creation when no parallel work exists. Reuse it across that operation's +waves. Keep the existing owner lock held across planning, transport completion +and cache publication. Workers execute only immutable, validated transport range +tasks; they must not acquire the owner lock, manipulate a shared ZIP cursor, +construct sources, decode arrays, or mutate caches and source registries. + +The coordinator performs parsing and writes on the calling thread. Keep at most +the configured number of requests submitted; do not enqueue every column/chunk +in an unbounded executor map. Consume completed work and release references +promptly. Output column/row ordering must not depend on completion ordering. +Do not nest Proxy pools inside this pool. All table-scheduled array requests use +the operation's concurrency budget; unrelated array reads retain their own path. + +An owner lock serializes operations sharing that owner, including refresh/close. +Do not weaken it in this project. Separate owners can operate concurrently and +their budgets add up; no process-wide limiter is proposed. + +### Metadata batching + +Add a private batch-open path in RemoteTableStorage for an explicit set of +required columns. Reuse member validation, column-path mapping, small-member +prefetch sizes and existing lazy opening routines. Include the relevant UTF-8 +companions; include null/validity metadata only when the operation needs it. + +Build immutable absolute-range tasks on the coordinator, fetch independent +prefixes concurrently, then parse/register sources serially. Retain overlapping +prefix bytes through the batch to avoid the redundant follow-up reads eliminated +by the prototype. Account downloaded bytes once, and hand prefetched payloads +to the existing cache ownership path without double charging them. + +Use an explicit private range-buffer parameter/context in the archive reader if +needed; do not monkeypatch methods, use `unittest.mock`, or share mutable +`capture_metadata`/opening-range state with workers. Preserve validation of +ZIP_STORED, unencrypted members, frame bounds and malformed companion handling. + +Integrate at points where the requested column set is known, including +`_LazyColumnDict._load_all()` and table-wide size inspection. Projection views +must batch only their selected leaves, not the base table's entire schema. +Bare opening, schema/name inspection and single-column access stay lazy. + +### Row planning and dependency waves + +Trace CTable scalar-row access, iteration, view materialization and bulk exports +to their shared column-read points before editing. Install the smallest optional +remote-storage hook there; do not implement only the example's first-five-rows +path or change view creation into eager materialization. + +For one bounded row batch: + +1. Resolve logical selection and live-row mapping using existing validity/view + machinery. Fetch any required validity data before deriving physical indices. +2. Determine requested stored columns and masks; open missing metadata through + the metadata batch path. Do not open unsupported, unselected siblings. +3. Reuse Proxy's existing missing-block/chunk selection, block-vs-whole cost + decisions, layout parsing and range plans. Factor the minimum private + plan/fetch/apply stages out of `Proxy._fetch_by_block()` and its whole-chunk + sibling if necessary. Keep the ordinary Proxy path using the same logic. +4. Schedule independent index/layout and payload ranges across columns. Resolve + dependency waves explicitly; no exception-driven discovery/replay as used by + the benchmark. Source metadata mutation stays on the coordinator. +5. Decode required UTF-8 offsets before planning dependent string-byte spans. + Reuse UTF8Array's existing contiguous/sparse span selection and null semantics; + do not load all string bytes to avoid this dependency. Masks and independent + fixed-width payloads may progress alongside ready UTF-8 work. +6. Apply validated compressed responses through existing cache write paths, + decode results and assemble rows serially. Preserve selection order, + duplicates, shaped values and nulls. Release temporary buffers after their + last consumer, not at the end of the entire table scan. + +Support scalar, contiguous, stepped/reverse and fancy selections through existing +selection semantics. Iteration and bulk materialization use bounded row batches, +not prefetch of all rows before the first result. Keep predicate evaluation and +computed-column dependency handling on existing paths where no explicit batch +of stored operands is available; do not create a new query engine. Such paths +must remain correct even if they do not gain parallelism in this implementation. + +### Memory budgets and cache policies + +Reserve expected response bytes before submission. Count in-flight reservations, +completed-but-unconsumed responses and pending compressed block assembly against +the relevant temporary budget; do not count only the final buffer dictionary. +Avoid duplicate copies where practical and document unavoidable assembly copies. +Do not let a completed future retain a response after its consumer is done. + +Split work into batches when a budget is reached. Release metadata batches before +starting row payload batches where possible. Prefer smaller valid range groups +when a combined request is too large. One indivisible required response or +assembly unit can exceed a user budget: handle it alone via the existing serial +path, without other outstanding tasks, and document this explicit soft-budget +exception. Do not reject an otherwise valid large string/chunk, loop without +progress, or advertise a hard RSS cap. Test this exception with small budgets. + +Preserve all cache policies: + +- MEMORY: publish through the shared cache coordinator and honor aggregate + `max_cache_bytes`. Do not prefetch data only to evict and immediately refetch + it during row assembly; consume bounded batches before advancing. +- DISK: network reads can overlap, but mutation/manifest transactions and cache + publication retain their existing locking and failure semantics. +- NONE: keep responses/decoded values only for the active operation and assemble + its output directly. Do not warm throwaway proxies and then invoke a second + reader that downloads the same data again. No persistent payload retention. + +Repeated MEMORY/DISK reads require zero payload requests only when the required +data fits the retained cache. NONE and deliberately undersized caches have no +such guarantee. Temporary settings must not silently increase retained limits. + +DISK is useful for a large reusable working set, but does not replace temporary +budgets: downloads, decoding, output arrays and native scratch still use RAM, +and the OS may cache disk pages. NONE can suit one-pass scans, while bounded +MEMORY can suit a small hot subset of a large table. Do not automatically select +a cache policy based on table size or machine RAM. Document cache choice, +temporary budgets and batched output consumption as separate controls. + +### Errors, lifetime and accounting + +Preserve current range-response validation, identity assumptions, source errors +and NotRanged whole-chunk fallback. On the first failure, stop scheduling, cancel +not-yet-running work and join active workers before releasing owner resources. +Do not return partial row results as a successful read. Already validated cache +entries may remain reusable, but partial blocks must not be marked complete. + +Ensure close, refresh, stale views and parent-store lifetime rules still apply +on warm as well as cold reads. Release acquired handles and temporary buffers on +all failure paths. Keep traffic counters thread-safe and charge actual requests, +including completed requests after a sibling failure. Do not invent new retry +policies or swallow errors to fall back to a full archive download. + +## Implementation sequence + +1. Add/validate table settings and dispatch plumbing, with regression tests for + unchanged array, store and local open behavior. +2. Implement bounded metadata range batching and serial source registration; + wire explicit multi-column metadata consumers and verify lazy projections. +3. Extract only the necessary Proxy planning/application helpers, preserving + standalone array behavior. Add bounded cross-column row waves and UTF-8 + dependencies through shared CTable materialization hooks. +4. Complete NONE/DISK, eviction, oversized-unit and failure/lifetime handling + before enabling the default parallel path for all RemoteCTable reads. +5. Update API docs and `examples/ctable/remote_handling.py` to describe the table + defaults and options while preserving local-file support. Convert the + benchmark to exercise the real public API, retain an explicit serial control, + and remove its method-patching/replay implementation once no longer needed. + +## Verification and acceptance + +Use the `blosc2` conda environment for all Python/tests/build commands. Extend +existing tests rather than introduce a second concurrency framework. + +- Deterministic delayed memory/HTTP range tests must demonstrate actual overlap, + enforce the configured maximum, and verify serial behavior at one worker. + Include more pending columns than workers and UTF-8 dependency waves. +- Compare values with the same local archive: numeric, fixed strings, shaped + fields, UTF-8 Unicode/emoji/empty/long values, both null representations, empty + tables, deleted rows, spare capacity, chunk boundaries, projections, scalar, + slices, reverse/strided and duplicate/out-of-order fancy selections. +- Test metadata reuse and traffic without assuming every file has identical + request geometry. Suitable fixtures must exercise partial blocks and whole + chunks. Parallelism alone must not inflate payload transfers or bypass the + source's block-selection policy. Check warm cache behavior and NONE reads. +- Use small configurable budgets to force multiple batches and an oversized + indivisible unit. Assert scheduler byte accounting, prompt buffer release, + bounded outstanding futures and progress. Include a cache smaller than the + requested result to catch prefetch/eviction/refetch loops. +- Test delayed/erroring requests, truncated responses, mid-wave failure, + partial-open cleanup, successful subsequent reads, refresh/close interactions, + shared-owner handles and settings isolation. Ensure no worker waits on the + owner lock held by the coordinator. +- Cover direct RemoteCTable construction, unified open, nested store lookup, + settings on borrowed views, invalid options, and unchanged RemoteArray, + RemoteStore and local CTable defaults/laziness. +- Verify `blosc2.open(..., max_concurrency=...)` for table and array targets, + and explicit rejection of table-only buffer keywords by the general opener. + Verify buffer tuning through direct construction and returned table settings. +- Assert fixed defaults of eight requests, 8 MiB for metadata and 64 MiB for rows; + cover an explicit 256 MiB row-budget override. Default selection must not + inspect available RAM or CPU count. Exercise both block-readable and + whole-chunk fallback fixtures under small budgets, across cache policies. +- Run focused RemoteCTable, CTable/UTF-8, RemoteStore, RemoteArray, Proxy and B2Z + tests, then the default suite and Ruff. Network performance is not a CI gate. + +Repeat the fresh-session HTTPS benchmark at 1/2/4/8 workers, alternating order, +with at least three runs per setting. Preserve historical baseline results; +production one-worker runs include reuse and are not the old baseline. Report +bare open, metadata, cold/warm rows, total latency, request/byte counts, peak +concurrency, retained cache and peak temporary reservations. Check identical +outputs across modes. Measure RSS separately (including native allocations) on +a wider/larger fixture; do not present reservation counters or tracemalloc alone +as total process memory. No cloud writes are needed. + +Completion requires real public-API parallelism, bounded scheduling with the +documented single-unit exception, all cache policies and lifecycle tests passing, +and no behavioral changes to standalone arrays/stores or local tables. Record +results here and update the benchmark report when production work is verified. + +## Implementation and verification results + +- RemoteCTable defaults to eight workers, 8 MiB metadata and 64 MiB row budgets. + Settings are validated, mutable on the table, inherited by borrowed reads and + isolated between independent handles. Only `max_concurrency` is accepted by + the general opener; table buffer keywords produce an explicit diagnostic. +- `ctable_remote_read.py` drives explicit range/dependency generators in bounded + waves. Workers perform transport only; index/layout parsing, cache mutation, + decoding and assembly remain on the owner thread. A single oversized unit + runs alone. It reuses existing Proxy block selection/application and source + layout helpers; standalone Proxy's execution path is not replaced. +- Metadata prefixes are retained for one bounded batch. Small neighboring user + attributes are decoded while already-fetched bytes cover the entire member, + avoiding five redundant requests in the example. No whole-schema payload + prefetch is introduced for single-column/projection reads. +- Row scalar access, iteration, display, Arrow batches and pandas export use the + column batch reader. Iteration/pandas use 1,024-row decoded batches; Arrow uses + its requested batch size. UTF-8 uses sorted span clusters and slice-based byte + access without allocating an integer index for every encoded byte. +- NONE uses temporary per-chunk proxies and returns decoded results directly. + MEMORY/DISK cache publication precedes decoding and eviction, avoiding a second + fetch when the retained limit is smaller than the result. Shaped columns are + consumed one physical chunk at a time before another column can evict them. +- Shared cross-process cache handles and read-only artifacts deliberately retain + their existing guarded row readers. Batching those requires cross-process + leases and is not introduced here. Computed expressions and other consumers + without an explicit multi-column batch retain their existing paths. Ordinary + NONE/MEMORY/DISK table handles use the parallel reader; no store/array default + behavior changes. Pools are scoped to a bounded batch, not a full lazy export. +- Automated tests cover actual overlap, byte reservations, oversized serial + progress, worker failure cleanup, retry, all cache policies, small retained + limits, block reads, disk reopen, nullable UTF-8, timestamps, shaped fields, + deleted/fancy/reverse rows, projection laziness and export equality. + +HTTPS medians over three fresh-session runs per setting (seconds): + +| Workers | Open + metadata | Cold first five rows | Total | +| --- | ---: | ---: | ---: | +| 1 | 2.842 | 2.701 | 5.544 | +| 2 | 2.260 | 2.313 | 4.538 | +| 4 | 1.736 | 1.914 | 3.656 | +| 8 | 1.613 | 2.105 | 3.701 | + +The production serial control includes metadata reuse. Eight workers reduced +total latency by about 33% versus that control; four/eight remain close and +network variability prevents treating this as a universal optimum. Every run +used eight column-metadata requests / 132,581 bytes and ten row requests / +941,623 bytes, with zero warm-read requests and 990,248 retained cache bytes. +Eight-worker peak reservations were 132,581 metadata bytes and 925,252 row bytes; +observed peak requests were eight for metadata and three for rows. + +A separate fresh-process RSS check used a 70-column, 200,000-row archive: +58 random float64 columns and 12 short multilingual UTF-8 columns, 83,124,830 +archive bytes. A file-backed transport exercised the same remote B2Z reader +without loading the archive into an in-memory filesystem. After metadata +inspection, ten rows were materialized. Native-inclusive process high-water RSS +was measured with `resource.getrusage` on macOS; baseline RSS used psutil. + +| Cache policy | Baseline RSS | Peak process RSS | Retained compressed cache | +| --- | ---: | ---: | ---: | +| NONE | 60.8 MiB | 77.9 MiB | 0 | +| MEMORY | 60.3 MiB | 106.1 MiB | 10.62 MiB | +| DISK | 59.9 MiB | 107.7 MiB | 10.62 MiB | + +Temporary reservations were 1,342,180 metadata bytes and 703,932 row bytes. +These are fixture-specific measurements, not RAM guarantees or an HTTPS memory +profile. DISK can retain OS-backed pages and native allocations; it does not +promise lower RSS for a small one-shot read. The temporary fixture/cache was +removed automatically after measurement. + +Final validation: 10,258 default-suite tests passed, 36 skipped; all 45 focused +RemoteCTable tests passed. Ruff lint and formatting checks passed. The full +suite required local-server/multiprocessing permissions; sandbox-only repeats +hit permission errors and were superseded by the successful permitted run. +The HTTPS example was verified with the default settings (no parallel opt-in). +Its printed sample can omit middle columns to fit the terminal, so use the +benchmark's full-row materialization for comparable request counts. diff --git a/src/blosc2/b2z_source.py b/src/blosc2/b2z_source.py index 180b85aeb..6986c2193 100644 --- a/src/blosc2/b2z_source.py +++ b/src/blosc2/b2z_source.py @@ -10,6 +10,7 @@ import operator import re import zipfile +from contextlib import contextmanager from blosc2.core import _import_fsspec from blosc2.proxy_source import REMOTE_MAX_CONCURRENCY, ByteRangeNDSource, Traffic @@ -122,6 +123,7 @@ def __init__(self, urlpath, *, storage_options=None, _filesystem=None, _traffic= raise ValueError("Invalid cached B2Z metadata range") self.capture_metadata = True self._opening_ranges = [] + self._batch_ranges = [] # ponytail: small directories fit in 8 KiB; larger ones use exact reads. tail_start = max(0, size - 8192) if bootstrap is None: @@ -185,17 +187,34 @@ def _read_archive(self, offset, size): for start, data in self.metadata["ranges"] if self.capture_metadata else (): if start <= offset and offset + size <= start + len(data): return data[offset - start : offset - start + size] - for start, data in self._opening_ranges: + for start, data in (*self._opening_ranges, *self._batch_ranges): if start <= offset and offset + size <= start + len(data): return data[offset - start : offset - start + size] + data = self.read_transport(offset, size) + if self.capture_metadata and self.persist_metadata: + self.metadata["ranges"].append((offset, data)) + return data + + def read_transport(self, offset, size): + """Read immutable bytes only; safe while the owner parses other responses.""" + if not 0 <= offset <= self.size or not 0 <= size <= self.size - offset: + raise ValueError("B2Z range exceeds archive bounds") data = self._fs.cat_file(self._path, start=offset, end=offset + size) - if len(data) > size: + if len(data) != size: raise ValueError("B2Z transport did not honor the requested byte range") self.traffic.charge(len(data)) - if self.capture_metadata and self.persist_metadata: - self.metadata["ranges"].append((offset, data)) return data + @contextmanager + def buffered_ranges(self, ranges): + """Owner-thread-only prefix reuse while opening a batch of members.""" + previous = self._batch_ranges + self._batch_ranges = ranges + try: + yield + finally: + self._batch_ranges = previous + def close(self): self.archive.close() self.file.close() @@ -227,6 +246,11 @@ def _read_archive(self, offset, size): if self.prefix_start <= offset and offset + size <= self.prefix_start + len(self.prefix): start = offset - self.prefix_start return self.prefix[start : start + size] + self.prepare_transport() + return self.read_transport(offset, size) + + def prepare_transport(self): + """Resolve a seeded transport on the owner thread before parallel reads.""" if self._fs is None: if self._filesystem is None: fs, path = _import_fsspec(self.urlpath).url_to_fs(self.urlpath, **self.storage_options) @@ -234,8 +258,10 @@ def _read_archive(self, offset, size): fs = self._filesystem path = fs._strip_protocol(self.urlpath) self._path, self._fs = path, fs + + def read_transport(self, offset, size): data = self._fs.cat_file(self._path, start=offset, end=offset + size) - if len(data) > size: + if len(data) != size: raise ValueError("B2Z transport did not honor the requested byte range") self.traffic.charge(len(data)) return data diff --git a/src/blosc2/ctable.py b/src/blosc2/ctable.py index 7abfad227..a8276ad74 100644 --- a/src/blosc2/ctable.py +++ b/src/blosc2/ctable.py @@ -4353,6 +4353,10 @@ def _load(self, name: str): return dict.__getitem__(self, name) def _load_all(self) -> None: + storage = self._table._remote_read_storage() + if storage is not None: + storage.open_columns(self._table, self._col_names, self._load) + return for name in self._col_names: self._load(name) @@ -6097,15 +6101,20 @@ def _prewarm_display_cache(self, display_cols: list[str], head_pos, tail_pos) -> Only pays off when both slices are non-empty. """ cache = getattr(self, "_display_fetch_cache", None) - if cache is None or len(head_pos) == 0 or len(tail_pos) == 0: + remote = self._remote_read_storage() is not None + if cache is None or (not remote and (len(head_pos) == 0 or len(tail_pos) == 0)): return real_cols = [n for n in display_cols if n != "..." and (n in self._cols or n in self._computed_cols)] if not real_cols: return nh = len(head_pos) combined = np.concatenate([head_pos, tail_pos]) + if remote: + from blosc2.ctable_remote_read import column_values + + columns = column_values(self, real_cols, combined) for name in real_cols: - vals = self._fetch_col_at_positions_uncached(name, combined) + vals = columns[name] if remote else self._fetch_col_at_positions_uncached(name, combined) cache[(name, id(head_pos))] = vals[:nh] cache[(name, id(tail_pos))] = vals[nh:] @@ -6457,6 +6466,21 @@ def __len__(self): def __iter__(self): """Iterate over live rows in insertion order, yielding namedtuple-like row objects.""" + storage = self._remote_read_storage() + if storage is not None: + from blosc2.ctable_remote_read import column_values + + # Bound decoded output independently of the transport budget. + for start, _, batch in self._remote_position_batches(1024): + values = column_values(self, self.col_names, batch) + for j, pos in enumerate(batch): + storage._check_open() + yield self._materialize_row( + start + j, + _physical=int(pos), + _values={name: values[name][j] for name in self.col_names}, + ) + return for i in range(self.nrows): yield self._materialize_row(i) @@ -6467,6 +6491,26 @@ def _row_namedtuple_type(self): self._row_namedtuple_type_cache_cols = visible return self._row_namedtuple_type_cache + def _remote_read_storage(self): + from blosc2.ctable_storage import RemoteTableStorage + + table = self + while table.base is not None: + table = table.base + storage = getattr(table, "_storage", None) + return storage if isinstance(storage, RemoteTableStorage) else None + + def _remote_position_batches(self, batch_size): + """Bounded live selections, preserving a gathered/sorted view's order.""" + cached = getattr(self, "_cached_live_positions", None) + groups = (cached,) if cached is not None else self._iter_live_positions_chunks() + logical = 0 + for positions in groups: + for start in range(0, len(positions), batch_size): + batch = positions[start : start + batch_size] + yield logical, logical + len(batch), batch + logical += len(batch) + def _row_namedtuple_type_for_fields(self, fields: tuple[str, ...]): cache = getattr(self, "_row_namedtuple_type_cache_by_fields", None) if cache is None: @@ -6502,27 +6546,47 @@ def _physical_row_value(self, col_name: str, pos: int): return np.datetime64(int(value), spec.unit) return value - def _materialize_row(self, index: int): + def _materialize_row(self, index: int, *, _physical=None, _values=None): n_rows = self.nrows if index < 0: index += n_rows if not (0 <= index < n_rows): raise IndexError(f"row index {index} is out of bounds for table with {n_rows} rows") _slp = getattr(self, "_cached_live_positions", None) - if _slp is not None and self.base is not None: + if _physical is not None: + pos = _physical + elif _slp is not None and self.base is not None: pos = int(_slp[index]) else: pos = _find_physical_index(self._valid_rows, index) + values = _values + if values is None and self._remote_read_storage() is not None: + from blosc2.ctable_remote_read import column_values + + columns = column_values(self, self.col_names, np.array([pos], dtype=np.int64)) + values = {name: columns[name][0] for name in self.col_names} + + def row_value(name): + if values is None: + return self._physical_row_value(name, int(pos)) + value = values[name] + spec = self._schema.columns_by_name.get(name) + if value is not None and spec is not None and isinstance(spec.spec, timestamp): + return ( + value if isinstance(value, np.datetime64) else np.datetime64(int(value), spec.spec.unit) + ) + return self._normalize_scalar_value(value) + nested_meta = self._schema.metadata.get("nested") if self._schema.metadata else None reconstruct = isinstance(nested_meta, dict) and bool(nested_meta.get("reconstruct_rows", False)) if not reconstruct: row_type = self._row_namedtuple_type() - return row_type(*(self._physical_row_value(name, int(pos)) for name in self.col_names)) + return row_type(*(row_value(name) for name in self.col_names)) row_dict: dict[str, Any] = {} for name in self.col_names: - value = self._physical_row_value(name, int(pos)) + value = row_value(name) parts = split_field_path(name) if len(parts) <= 1: row_dict[name] = value @@ -8092,8 +8156,40 @@ def iter_arrow_batches( # noqa: C901 if any(name in self.col_names and self[name].is_dictionary for name in names): dict_real_pos = blosc2.where(self._valid_rows, _arange(len(self._valid_rows))).compute() - for start in range(0, self._n_rows, batch_size): - stop = min(start + batch_size, self._n_rows) + remote = self._remote_read_storage() + parallel = ( + remote is not None and remote._owner.is_mutable and not getattr(remote._owner, "shared", False) + ) + batches = ( + self._remote_position_batches(batch_size) + if parallel + else ( + (start, min(start + batch_size, self._n_rows), None) + for start in range(0, self._n_rows, batch_size) + ) + ) + for start, stop, positions in batches: + remote_values, remote_nulls = {}, {} + if parallel: + from blosc2.ctable_remote_read import column_values + + leaves = [name for name in names if name in self.col_names] + remote_values = column_values(self, leaves, positions, null_masks=remote_nulls) + + def read_values(name, remote_values=remote_values, start=start, stop=stop): + return remote_values[name] if name in remote_values else self[name][start:stop] + + def read_nulls( + name, values, remote_values=remote_values, remote_nulls=remote_nulls, start=start, stop=stop + ): + if name in remote_values: + return ( + remote_nulls.get(name) + if self[name]._nulls.kind == NULL_MASK + else self[name]._nulls.mask_for_values(values) + ) + return self[name]._nulls.null_mask_slice(values, start, stop) + arrays = [] for name in names: cc = self._schema.columns_by_name.get(name) @@ -8110,7 +8206,12 @@ def iter_arrow_batches( # noqa: C901 spec = self._schema.columns_by_name[name].spec arr8 = self._cols[name] nv = col.null_value - if self.base is None and self._last_pos == self._n_rows and stop <= arr8._persisted_rows: + if ( + not parallel + and self.base is None + and self._last_pos == self._n_rows + and stop <= arr8._persisted_rows + ): # Dense root table: logical rows == persisted rows, so # export straight from the offsets/bytes buffers with # no per-row decode (storage is already Arrow layout). @@ -8118,8 +8219,8 @@ def iter_arrow_batches( # noqa: C901 arr8.arrow_slice(pa, start, stop, nv, valid=col._nulls.valid_slice(start, stop)) ) continue - values = col[start:stop] # StringDType array; nulls per this column's channel - null_mask = col._nulls.null_mask_slice(values, start, stop) + values = read_values(name) # StringDType array; nulls per this column's channel + null_mask = read_nulls(name, values) arrays.append( pa.array( values.astype(object), @@ -8162,12 +8263,12 @@ def iter_arrow_batches( # noqa: C901 continue if col.is_ndarray: spec = self._schema.columns_by_name[name].spec - values = np.asarray(col[start:stop]) + values = np.asarray(read_values(name)) # Row-level under mask storage. A sentinel ndarray column # keeps the older, lossier rule -- a row is null only when # *every* element equals the sentinel -- because that is the # only thing its storage can express. - null_mask = col._nulls.null_mask_slice(values, start, stop) + null_mask = read_nulls(name, values) pa_type = self._pa_type_from_spec(pa, spec) flat_values = np.ascontiguousarray(values.reshape(-1)) pa_values = pa.array(flat_values, type=pa_type.value_type) @@ -8179,8 +8280,8 @@ def iter_arrow_batches( # noqa: C901 ) ) continue - arr = np.asarray(col[start:stop]) - null_mask = col._nulls.null_mask_slice(arr, start, stop) + arr = np.asarray(read_values(name)) + null_mask = read_nulls(name, arr) if arr.dtype.kind in "US": # pyarrow reads the mask alongside the values, so the null # slots need no substitution here — under mask storage they @@ -10188,6 +10289,29 @@ def to_pandas(self): """ import pandas as pd + remote = self._remote_read_storage() + if ( + remote is not None + and remote._owner.is_mutable + and not getattr(remote._owner, "shared", False) + and self.nrows + ): + from blosc2.ctable_remote_read import column_values + + frames = [] + for _, _, positions in self._remote_position_batches(1024): + nulls = {} + values = column_values(self, self.col_names, positions, null_masks=nulls) + data = {} + for name in self.col_names: + col = self[name] + raw = list(values[name]) if col.is_ndarray else values[name] + data[name] = self._pandas_values( + pd, col, raw, nulls=nulls.get(name, np.zeros(len(positions), dtype=bool)) + ) + frames.append(pd.DataFrame(data)) + return pd.concat(frames, ignore_index=True) + data = {} for name in self.col_names: col = self[name] @@ -10247,7 +10371,7 @@ def missing(value): return cells @staticmethod - def _pandas_values(pd, col, values): + def _pandas_values(pd, col, values, *, nulls=None): """*values*, with this column's nulls turned into something pandas reads as NA. A null slot holds the fill under mask storage and the sentinel under a @@ -10264,7 +10388,8 @@ def _pandas_values(pd, col, values): channel = col._nulls kind = channel.kind if kind == NULL_MASK: - nulls = channel.null_mask() # one byte per row, off the sidecar + if nulls is None: + nulls = channel.null_mask() # one byte per row, off the sidecar elif kind == NULL_SENTINEL: nulls = channel.mask_for_values(values) # in band, from what we just read else: diff --git a/src/blosc2/ctable_remote_read.py b/src/blosc2/ctable_remote_read.py new file mode 100644 index 000000000..4b331b3a4 --- /dev/null +++ b/src/blosc2/ctable_remote_read.py @@ -0,0 +1,362 @@ +####################################################################### +# Copyright (c) 2019-present, Blosc Development Team +# All rights reserved. +# SPDX-License-Identifier: BSD-3-Clause +####################################################################### +"""Bounded transport waves for RemoteCTable; all parsing stays on the caller.""" + +import itertools +import struct +from collections import deque +from concurrent.futures import ThreadPoolExecutor, as_completed +from contextlib import ExitStack + +import numpy as np + +import blosc2 +from blosc2.b2z_source import _LOCAL_HEADER_HEADROOM, _WHOLE_MEMBER_PREFETCH_MAX +from blosc2.proxy_source import NotRanged + + +def run_reads(readers, workers, budget): # noqa: C901 + """Drive readers yielding (callable, arguments, reserved bytes). + + Only one bounded wave is submitted at a time. A single oversized response + runs alone. Readers parse, cache and decode responses on this thread. + """ + readers = iter(readers) + ready = deque() + results = {} + peak = 0 + live = set() + with ExitStack() as stack: + pool = None + + def advance(key, reader, answer=None, error=None): + try: + task = reader.send(answer) if error is None else reader.throw(error) + ready.append((key, reader, task)) + except StopIteration as done: + results[key] = done.value + live.discard(reader) + + try: + exhausted = False + while ready or not exhausted: + while len(ready) < workers and not exhausted: + try: + key, reader = next(readers) + except StopIteration: + exhausted = True + else: + live.add(reader) + advance(key, reader) + wave, size = [], 0 + while ready and len(wave) < workers: + item = ready[0] + cost = item[2][2] + if wave and size + cost > budget: + break + wave.append(ready.popleft()) + size += cost + if size >= budget: + break + peak = max(peak, size) + if len(wave) == 1: + key, reader, (func, args, _) = wave.pop() + try: + answer = func(*args) + except Exception as error: + advance(key, reader, error=error) + else: + advance(key, reader, answer) + del answer + elif wave: + if pool is None: + pool = stack.enter_context(ThreadPoolExecutor(max_workers=workers)) + futures = { + pool.submit(func, *args): (key, reader) for key, reader, (func, args, _) in wave + } + try: + for future in as_completed(futures): + key, reader = futures.pop(future) + try: + answer = future.result() + except Exception as error: + advance(key, reader, error=error) + else: + advance(key, reader, answer) + del answer + del future + finally: + for future in futures: + future.cancel() + finally: + # Join workers before closing generators that own cache transactions. + if pool is not None: + pool.shutdown(wait=True, cancel_futures=True) + for reader in live: + reader.close() + return results, peak + + +def open_columns(storage, table, names, load): # noqa: C901 + """Fetch column prefixes in bounded groups, then open each column serially.""" + from blosc2.ctable_storage import _column_name_to_relpath + from blosc2.schema import UTF8Spec + + owner = storage._owner + with owner.lock: + storage._check_open() + archive = owner.archive + members = {} + for info in archive.members: + members.setdefault(info.filename, []).append(info) + + def ranges_for(name): + key = storage._full_key(f"_cols/{_column_name_to_relpath(name)}") + spec = table._schema.columns_by_name[name].spec + keys = (key, key + ".utf8") if isinstance(spec, UTF8Spec) else (key,) + ranges = [] + for key in keys: + if key in owner.sources: + continue + matches = members.get(key + ".b2nd", ()) + if len(matches) != 1: + # The ordinary opener supplies the appropriate diagnostic. + return [] + info = matches[0] + if info.flag_bits & 1 or info.compress_type != 0: + return [] + if info.compress_size != info.file_size or not 0 <= info.header_offset <= archive.size - 30: + return [] + want = ( + info.file_size + _LOCAL_HEADER_HEADROOM + if info.file_size <= _WHOLE_MEMBER_PREFETCH_MAX + else 16384 + ) + ranges.append((info.header_offset, min(want, archive.size - info.header_offset))) + return ranges + + def reader(offset, size): + return (yield archive.read_transport, (offset, size), size) + + def consume(batch, ranges): + prefixes, _ = run_reads( + ((offset, reader(offset, size)) for offset, size in ranges), + storage.max_concurrency, + storage.metadata_buffer_bytes, + ) + with archive.buffered_ranges(list(prefixes.items())): + for name in batch: + load(name) + # A small neighboring attributes member often arrived with the + # last column. Decode it now only if its entire member is here. + for info in members.get(storage._full_key("_vlmeta") + ".b2f", ()): + for offset, data in prefixes.items(): + start = info.header_offset - offset + if 0 <= start <= len(data) - 30: + head = data[start : start + 30] + end = start + 30 + int.from_bytes(head[26:28], "little") + end += int.from_bytes(head[28:30], "little") + info.file_size + if end <= len(data): + storage.load_user_attrs() + break + + batch, ranges, size, peak = [], [], 0, 0 + for name in names: + spans = ranges_for(name) + cost = sum(length for _, length in spans) + if batch and size + cost > storage.metadata_buffer_bytes: + consume(batch, ranges) + batch, ranges, size = [], [], 0 + if cost > storage.metadata_buffer_bytes: + # Ordinary opening consumes one member at a time, without a batch. + load(name) + peak = max(peak, max((length for _, length in spans), default=0)) + continue + batch.append(name) + ranges.extend(spans) + size += cost + peak = max(peak, size) + if batch: + consume(batch, ranges) + storage._peak_metadata_buffer_bytes = peak + + +def _range_task(source, offset, size): + # Workers bypass mutable opening buffers and operate only on archive bytes. + if offset < 0 or size < 0 or offset > source.member_length: + raise ValueError("B2Z range exceeds member bounds") + prepare = getattr(source._archive, "prepare_transport", None) + if prepare is not None: + prepare() + size = min(size, source.member_length - offset) + return source._archive.read_transport, (source.member_offset + offset, size), size + + +def _array_items(source, positions): + """Yield output/source selections without an integer index per UTF-8 byte.""" + chunk_rows = source.chunks[0] + if isinstance(positions, slice): + start = positions.start + while start < positions.stop: + stop = min((start // chunk_rows + 1) * chunk_rows, positions.stop) + yield (slice(start - positions.start, stop - positions.start),), (slice(start, stop),) + start = stop + return + trailing = [ + range(0, size, chunk) for size, chunk in zip(source.shape[1:], source.chunks[1:], strict=True) + ] + for group, *starts in itertools.product(np.unique(positions // chunk_rows), *trailing): + selected = np.flatnonzero(positions // chunk_rows == group) + spans = tuple( + slice(start, min(start + chunk, size)) + for start, chunk, size in zip(starts, source.chunks[1:], source.shape[1:], strict=True) + ) + yield (selected, *spans), (positions[selected], *spans) + + +def _array_values(array, positions): # noqa: C901 + """Fetch and decode one chunk's selected rows before cache eviction can run.""" + source = array.src + owner = array._store_owner + array._check_open() + if owner.cache_policy == blosc2.CachePolicy.NONE: + proxy = None + else: + proxy = owner.get_cache(source) + if not owner.is_mutable: + # Read-only artifacts retain their existing fallback behavior. + return array[positions] + count = positions.stop - positions.start if isinstance(positions, slice) else len(positions) + out = np.empty((count, *source.shape[1:]), dtype=source.dtype) + for target, item in _array_items(source, positions): + if owner.cache_policy == blosc2.CachePolicy.NONE: + # No operation-scoped proxy accumulates payloads from previous chunks. + proxy = blosc2.Proxy(source, _refresh_source=False) + missing = proxy._missing_blocks(item) + if missing: + steps = source.frame_index_reads() + answer = None + while True: + try: + offset, size = steps.send(answer) + except StopIteration: + break + answer = yield _range_task(source, offset, size) + del answer + proxy._begin_persistent_mutation() + try: + wanted = proxy._asking_blocks(missing, None) if proxy._blocks_per_chunk > 1 else {} + for nchunk, blocks in missing.items(): + layout = None + if nchunk in wanted: + if nchunk not in source._layouts: + section = blosc2.MAX_OVERHEAD + 4 * source.blocks_per_chunk + head = yield _range_task(source, int(source._offsets[nchunk]), section) + source._layouts[nchunk] = source._parse_layout(head, section) + del head + layout = source._layouts[nchunk] + if layout is not None: + runs = source.block_plan(nchunk, blocks) + tasks = [_range_task(source, offset, size) for offset, size, _ in runs] + + def fetch(tasks=tasks): + return [func(*args) for func, args, _ in tasks] + + try: + answers = yield fetch, (), sum(run[1] for run in runs) + except NotRanged: + layout = None + else: + payloads = { + block: data[offset : offset + size] + for run, data in zip(runs, answers, strict=True) + for block, offset, size in run[2] + } + proxy._write_blocks(nchunk, payloads, layout[0]) + del answers, payloads + if layout is None: + offset = int(source._offsets[nchunk]) + if offset < 0: + data = source._special_chunk(offset) + else: + data = yield _range_task(source, offset, int(source._extents[nchunk])) + data = data[: struct.unpack(" _GATHER_GAP) + 1 + start = 0 + for cluster in np.split(sorted_pos, splits): + if not len(cluster): + continue + lo, hi = int(cluster[0]), int(cluster[-1]) + offsets = yield from _array_values(col._offsets, slice(lo, hi + 2)) + first, last = int(offsets[0]), int(offsets[-1]) + data = yield from _array_values(col._data, slice(first, last)) + blob = data.tobytes() + for j, pos in enumerate(cluster): + a, b = offsets[pos - lo : pos - lo + 2] - first + values[order[start + j]] = blob[a:b].decode("utf-8") + start += len(cluster) + else: + values = yield from _array_values(col, positions) + spec = table._schema.columns_by_name[name].spec + if isinstance(spec, timestamp): + values = values.astype(f"datetime64[{spec.unit}]") + mask = masks[name] + if mask is not None: + valid = yield from _array_values(mask, positions) + if null_masks is not None: + null_masks[name] = ~valid + elif not valid.all(): + values = list(values) + for i in np.flatnonzero(~valid): + values[i] = None + return values + + values, peak = run_reads( + ((name, reader(name)) for name in stored), storage.max_concurrency, storage.row_buffer_bytes + ) + storage._peak_row_buffer_bytes = peak + for name in names: + if name not in values: + values[name] = table._fetch_col_at_positions_uncached(name, positions) + return values diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index 507f2a239..ff128c4a7 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -647,7 +647,19 @@ def index_anchor_path(self, col_name: str) -> str | None: class RemoteTableStorage(TableStorage): """Read-only CTable storage over a shared RemoteStore owner.""" - def __init__(self, owner, root_key: str) -> None: + def __init__( + self, + owner, + root_key: str, + *, + max_concurrency=8, + metadata_buffer_bytes=8 << 20, + row_buffer_bytes=64 << 20, + ) -> None: + self.max_concurrency = max_concurrency + self.metadata_buffer_bytes = metadata_buffer_bytes + self.row_buffer_bytes = row_buffer_bytes + self._peak_metadata_buffer_bytes = self._peak_row_buffer_bytes = 0 self._owner = owner self._root_key = root_key.strip("/") self._generation = owner.generation @@ -655,6 +667,11 @@ def __init__(self, owner, root_key: str) -> None: self._closed = False owner.acquire() + def open_columns(self, table, names, load): + from blosc2.ctable_remote_read import open_columns + + return open_columns(self, table, names, load) + def _check_open(self) -> None: if self._closed: raise RuntimeError("RemoteCTable handle is closed") @@ -750,6 +767,8 @@ def load_schema(self) -> dict[str, Any]: def load_user_attrs(self) -> dict: self._check_open() + if hasattr(self, "_user_attrs"): + return dict(self._user_attrs) member = self._full_key("_vlmeta") + ".b2f" matches = [info for info in self._owner.archive.members if info.filename == member] if not matches: @@ -758,7 +777,8 @@ def load_user_attrs(self) -> dict: raise ValueError(f"Duplicate Remote CTable metadata member {member!r}") from blosc2.b2z_source import member_vlmeta - return dict(member_vlmeta(self._owner.archive, matches[0])) + self._user_attrs = dict(member_vlmeta(self._owner.archive, matches[0])) + return dict(self._user_attrs) def table_exists(self) -> bool: try: diff --git a/src/blosc2/proxy_source.py b/src/blosc2/proxy_source.py index f53fa23eb..ce79e95ed 100644 --- a/src/blosc2/proxy_source.py +++ b/src/blosc2/proxy_source.py @@ -585,6 +585,17 @@ def _read_frame_header(read_range, head=None) -> tuple[bytes, list, bytes]: def _read_frame_offsets(read_range, header: list, head: bytes, header_len: int) -> np.ndarray: + steps = _frame_offset_reads(header, head, header_len) + answer = None + while True: + try: + offset, size = steps.send(answer) + except StopIteration as done: + return done.value + answer = read_range(offset, size) + + +def _frame_offset_reads(header, head, header_len): """The absolute position of every chunk of a frame whose header is in hand. A negative position is not a position at all: it encodes a run-length chunk @@ -609,10 +620,10 @@ def _read_frame_offsets(read_range, header: list, head: bytes, header_len: int) if len(head) >= frame_len: index = head[index_pos:] # the whole frame arrived in the first read else: - index = read_range(index_pos, min(frame_len - index_pos, _INDEX_PREFETCH)) + index = yield index_pos, min(frame_len - index_pos, _INDEX_PREFETCH) index_cbytes = struct.unpack(" len(index): - index = read_range(index_pos, index_cbytes) + index = yield index_pos, index_cbytes offsets = np.frombuffer(blosc2.decompress2(index[:index_cbytes]), dtype=np.int64) # Offsets are relative to the end of the header return np.where(offsets >= 0, offsets + header_len, offsets) @@ -963,6 +974,21 @@ def _offsets(self) -> np.ndarray: """Where each chunk begins, negative for one that lives in its offset.""" return self._frame_index()[0] + def frame_index_reads(self): + """Yield index ranges for an immutable source; parse on the caller thread. + + The caller serializes source access and sends each response back. Ordinary + array reads retain their existing locked, synchronous index path. + """ + if self._stale: + raise RuntimeError("Batched index reads require an immutable source") + if self._index is None: + offsets = yield from _frame_offset_reads(self._header, self._head, self._header_len) + _check_specials(offsets, self.urlpath) + self._index = (offsets, _chunk_extents(offsets, self._header)) + self._head = None + return self._index + @property def _extents(self) -> np.ndarray: """How many bytes to read at each chunk's offset to be sure of covering it.""" diff --git a/src/blosc2/remote_ctable.py b/src/blosc2/remote_ctable.py index 1ddac7cb3..2884ac9bf 100644 --- a/src/blosc2/remote_ctable.py +++ b/src/blosc2/remote_ctable.py @@ -8,13 +8,53 @@ from __future__ import annotations +import operator + from blosc2.ctable import CTable from blosc2.ctable_storage import RemoteTableStorage from blosc2.remote_array import CACHE_POLICY_DEFAULT, RemoteMetadataMapping +def _positive_integer(name, value): + if isinstance(value, bool): + raise TypeError(f"{name} must be a positive integer") + try: + value = operator.index(value) + except TypeError: + raise TypeError(f"{name} must be a positive integer") from None + if value < 1: + raise ValueError(f"{name} must be a positive integer") + return value + + +def _read_setting(name): + def get(self): + return getattr(self._remote_storage(), name) + + def set(self, value): + value = _positive_integer(name, value) + storage = self._remote_storage() + with storage._owner.lock: + storage._check_open() + setattr(storage, name, value) + + return property(get, set) + + class RemoteCTable(CTable): - """A read-only CTable whose fixed-width and UTF-8 columns are fetched on demand.""" + """A read-only CTable whose fixed-width and UTF-8 columns are fetched on demand. + + Independent column requests overlap by default. ``max_concurrency`` defaults + to 8; use 1 for serial reads. ``metadata_buffer_bytes`` (8 MiB) and + ``row_buffer_bytes`` (64 MiB) bound temporary transport batches, not retained + caches or total RAM. An indivisible oversized unit is read alone. These + positive-integer settings can also be changed on an open table; views use + their base table's settings. There is no automatic CPU/RAM-based tuning. + + ``blosc2.open`` accepts ``max_concurrency`` but not the table-specific buffer + keywords. Use this constructor or the returned table's settings to tune + buffers. Cache policies and ``max_cache_bytes`` remain independent. + """ def __new__( cls, @@ -25,10 +65,21 @@ def __new__( cache_policy=CACHE_POLICY_DEFAULT, max_cache_bytes=CACHE_POLICY_DEFAULT, cache_dir=None, + max_concurrency=8, + metadata_buffer_bytes=8 << 20, + row_buffer_bytes=64 << 20, _filesystem=None, ): if urlpath is None: raise TypeError("RemoteCTable requires a remote B2Z URL") + settings = { + name: _positive_integer(name, value) + for name, value in { + "max_concurrency": max_concurrency, + "metadata_buffer_bytes": metadata_buffer_bytes, + "row_buffer_bytes": row_buffer_bytes, + }.items() + } from blosc2.remote_store import RemoteStore @@ -49,7 +100,7 @@ def __new__( if kind == "unsupported": raise NotImplementedError(str(diagnostic)) raise ValueError("RemoteCTable requires a CTable node") - return cls._from_owner(store._owner, full) + return cls._from_owner(store._owner, full, **settings) finally: store.close() @@ -58,14 +109,19 @@ def __init__(self, *args, **kwargs): pass @classmethod - def _from_owner(cls, owner, full_path): - storage = RemoteTableStorage(owner, full_path) + def _from_owner(cls, owner, full_path, **settings): + settings = {name: _positive_integer(name, value) for name, value in settings.items()} + storage = RemoteTableStorage(owner, full_path, **settings) try: return cls._open_from_storage(storage) except BaseException: storage.close() raise + max_concurrency = _read_setting("max_concurrency") + metadata_buffer_bytes = _read_setting("metadata_buffer_bytes") + row_buffer_bytes = _read_setting("row_buffer_bytes") + def _remote_storage(self) -> RemoteTableStorage: storage = getattr(self, "_storage", None) if not isinstance(storage, RemoteTableStorage): diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index e35d1a8a2..08205101d 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2329,8 +2329,6 @@ def _open_remote_b2z(urlpath, options): array_error = exc if options["cache_path"] is not None: raise NotImplementedError("Remote tables and stores use cache_dir, not cache_path") - if options["max_concurrency"] is not None: - raise NotImplementedError("max_concurrency is only supported for remote arrays") if options["assume_immutable"] is not True: raise NotImplementedError("Remote tables and stores require assume_immutable=True") store_options = { @@ -2346,6 +2344,17 @@ def _open_remote_b2z(urlpath, options): # Include the initial array lookup in shared transfer accounting. store.traffic.nbytes += array_error.traffic.nbytes store.traffic.requests += array_error.traffic.requests + if options["max_concurrency"] is not None: + _, full = store._resolve("") + if store._owner.nodes[full][0] != "ctable": + raise NotImplementedError( + "max_concurrency is only supported for remote arrays and tables" + ) + return blosc2.RemoteCTable._from_owner( + store._owner, + full, + max_concurrency=options["max_concurrency"], + ) return store[""] except KeyError: if array_error is not None: @@ -2419,6 +2428,14 @@ def _normalize_open_target(urlpath, kwargs, dataset, hdf5_index): return urlpath +def _reject_table_buffer_options(kwargs): + table_options = kwargs.keys() & {"metadata_buffer_bytes", "row_buffer_bytes"} + if table_options: + raise TypeError( + f"{', '.join(sorted(table_options))} are RemoteCTable options; use RemoteCTable directly" + ) + + def open( urlpath: str | pathlib.Path | blosc2.URLPath, mode: str = "r", @@ -2496,12 +2513,15 @@ def open( What arrives is kept in memory (defaulting to :attr:`CachePolicy.MEMORY`), under ``cache_dir``, or at the exact ``cache_path`` (as :attr:`CachePolicy.DISK`). max_concurrency: int, optional - Only with ``lazy``: how many fetches to run at once, in a thread + For lazy remote arrays and RemoteCTable: how many fetches to run at once, in a thread pool. A slice against an object store is almost entirely round-trip latency, so overlapping the requests is what makes a wide slice bearable. Defaults to 8; pass 1 for a protocol with no latency to hide, where the pool costs about 10 microseconds per chunk and saves nothing. + Remote tables overlap independent requests across selected columns; + their table-specific temporary buffer settings are available through + ``RemoteCTable``, not through this general opener. cache_dir: str | pathlib.Path, optional For fsspec URLs and lazy Caterva2 :ref:`URLPath` objects, a directory holding this container's local copy — either the whole thing, or just the chunks and blocks ``lazy`` has fetched so far @@ -2663,6 +2683,7 @@ def open( >>> all(sc_open.decompress_chunk(i, dest1) == sc_open_mmap.decompress_chunk(i, dest1) for i in range(nchunks)) True """ + _reject_table_buffer_options(kwargs) if isinstance(urlpath, blosc2.URLPath): return _open_c2_urlpath(urlpath, mode, offset, kwargs) diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 14135df81..2605cda62 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -31,6 +31,37 @@ def remote_table_url(tmp_path, table, name="table"): return url +@pytest.mark.parametrize("include_note", [False, True]) +@pytest.mark.parametrize("nrows", [0, 2, 20, 21]) +def test_remote_example_total_timing(tmp_path, capsys, include_note, nrows): + import runpy + from pathlib import Path + from types import SimpleNamespace + + fields = [("x", int)] + rows = [(i,) for i in range(nrows)] + if include_note: + fields.append(("note", str, blosc2.field(blosc2.utf8()))) + rows = [(i, f"café 東京 #{i}") for i in range(nrows)] + local = blosc2.CTable(dataclasses.make_dataclass("Sample", fields), rows, create_summary_index=False) + url = remote_table_url(tmp_path, local) + example = runpy.run_path(str(Path(__file__).resolve().parents[2] / "examples/ctable/remote_handling.py")) + ticks = iter(range(10)) + access = example["access_table"] + access.__globals__["time"] = SimpleNamespace(perf_counter=lambda: next(ticks)) + access(SimpleNamespace(url=url)) + output = capsys.readouterr().out + expected_ms = 5000 if include_note else 3000 + assert f"\n\nTotal network : {expected_ms:7.1f} ms (" in output + assert "\nRetained cache:" in output + assert " - Total network" not in output + start = max(0, nrows // 2 - 2) + stop = min(start + 5, nrows) + heading = f"Sample rows [{start}:{stop}] around the midpoint (1st fetch):\n" + sample = output.split(heading)[1].split("\nTiming & Network Traffic:")[0].strip() + assert sample == str(local[start:stop]).strip() + + def test_remote_ctable_fixed_width_reads_and_queries(tmp_path): rows = [ (1, np.array([1, 2], dtype=np.float32), "one"), @@ -426,3 +457,273 @@ def counted(self, path, start=None, end=None, **kwargs): with blosc2.open(url, lazy=False, cache_dir=tmp_path / "localized", **format_options) as store: assert isinstance(store, blosc2.TreeStore) assert store["group/table"]["x"][0] == 1 + + +def test_parallel_metadata_benchmark(tmp_path, monkeypatch): + import runpy + import time + from pathlib import Path + + benchmark = runpy.run_path(str(Path(__file__).resolve().parents[2] / "bench/remote_ctable_metadata.py")) + + @dataclasses.dataclass + class TextRow: + x: int + y: float + text: str = blosc2.field(blosc2.utf8(null_storage="mask")) + + local = blosc2.CTable(TextRow, [(1, 2.0, "café"), (2, 3.0, None)], create_summary_index=False) + url = remote_table_url(tmp_path, local) + filesystem_type = type(fsspec.filesystem("memory")) + original = filesystem_type.cat_file + + def delayed(self, *args, **kwargs): + time.sleep(0.01) + return original(self, *args, **kwargs) + + monkeypatch.setattr(filesystem_type, "cat_file", delayed) + serial, expected = benchmark["run"](url, 1) + buffered, buffered_values = benchmark["run"](url, 1) + parallel, actual = benchmark["run"](url, 4) + assert actual == expected + assert buffered_values == expected + assert serial["peak_metadata_reads"] == 1 + assert parallel["peak_metadata_reads"] > 1 + assert parallel["metadata"]["bytes"] <= serial["metadata"]["bytes"] + assert buffered["metadata"]["bytes"] == parallel["metadata"]["bytes"] + assert buffered["metadata"]["requests"] == parallel["metadata"]["requests"] + with blosc2.open(url) as table: + archive = table._remote_storage()._owner.archive + original_read = archive._read_archive + + def fail(*args): + raise OSError("injected range failure") + + with monkeypatch.context() as patch: + patch.setattr(archive, "_read_archive", fail) + with pytest.raises(OSError, match="injected range failure"): + benchmark["metadata"](table) + assert archive._read_archive is fail + assert archive._read_archive == original_read + assert benchmark["metadata"](table) == expected[0] + + # Large, uncompressed columns keep payloads out of the metadata prefixes. + local = blosc2.CTable( + TextRow, + [(i, i / 7, None if i % 3 == 0 else f"東京 café {i}") for i in range(20000)], + cparams={"clevel": 0}, + create_summary_index=False, + ) + url = remote_table_url(tmp_path, local, name="row-payloads") + serial, expected = benchmark["run"](url, 1, row_workers=1) + buffered, buffered_values = benchmark["run"](url, 1) + parallel, actual = benchmark["run"](url, 4) + assert actual == buffered_values == expected + assert serial["peak_row_reads"] == 1 + assert parallel["peak_row_reads"] > 1 + for key in ("requests", "bytes"): + assert parallel["cold_rows"][key] == buffered["cold_rows"][key] == serial["cold_rows"][key] + assert parallel["cache_bytes"] == serial["cache_bytes"] + with blosc2.open(url) as table: + benchmark["metadata"](table) + archive = table._remote_storage()._owner.archive + original_read = archive.read_transport + with monkeypatch.context() as patch: + patch.setattr(archive, "read_transport", fail) + with pytest.raises(OSError, match="injected range failure"): + list(table[:5]) + assert archive.read_transport is fail + assert archive.read_transport == original_read + assert repr(list(table[:5])) == expected[1] + + +@pytest.mark.parametrize( + "policy", [blosc2.CachePolicy.NONE, blosc2.CachePolicy.MEMORY, blosc2.CachePolicy.DISK] +) +@pytest.mark.parametrize("budget", [128, 1 << 20]) +def test_parallel_rows_cache_policies(tmp_path, policy, budget): + @dataclasses.dataclass + class Mixed: + x: int = blosc2.field(blosc2.int64(null_storage="mask")) + y: float = blosc2.field(blosc2.float64()) + text: str = blosc2.field(blosc2.utf8(null_storage="mask")) + + local = blosc2.CTable( + Mixed, + [ + (None if i % 7 == 0 else i, i / 3, None if i % 5 == 0 else f"🌦 café 東京 {i}") + for i in range(20000) + ], + cparams={"clevel": 0}, + create_summary_index=False, + ) + local.delete([2, 9]) + url = remote_table_url(tmp_path, local) + options = {"cache_policy": policy, "row_buffer_bytes": budget, "metadata_buffer_bytes": budget} + if policy == blosc2.CachePolicy.DISK: + options["cache_dir"] = tmp_path / "cache" + if policy != blosc2.CachePolicy.NONE: + options["max_cache_bytes"] = 1024 + with blosc2.RemoteCTable(url, **options) as table: + for selection in (slice(0, 5), slice(3, 12, 2), slice(8, 1, -2), [15, 0, 15, 3]): + assert repr(list(table[selection])) == repr(list(local[selection])) + assert repr(table[3]) == repr(local[3]) + assert table[:4].to_string() == local[:4].to_string() + pd = pytest.importorskip("pandas") + pd.testing.assert_frame_equal(table[:12].to_pandas(), local[:12].to_pandas()) + pytest.importorskip("pyarrow") + assert table[:12].to_arrow().equals(local[:12].to_arrow()) + assert table.cache_bytes <= 1024 + if policy == blosc2.CachePolicy.NONE: + assert table.cache_bytes == 0 + + +def test_parallel_table_settings(tmp_path): + local = blosc2.CTable(Row, [(1, [1, 2], "one")], create_summary_index=False) + url = remote_table_url(tmp_path, local) + with blosc2.open(url, max_concurrency=2) as table: + assert table.max_concurrency == 2 + assert table.metadata_buffer_bytes == 8 << 20 + assert table.row_buffer_bytes == 64 << 20 + table.row_buffer_bytes = 256 << 20 + assert table[:1]._remote_read_storage().row_buffer_bytes == 256 << 20 + for name in ("max_concurrency", "metadata_buffer_bytes", "row_buffer_bytes"): + for value, error in ((0, ValueError), (-1, ValueError), (True, TypeError), (1.5, TypeError)): + with pytest.raises(error): + setattr(table, name, value) + with pytest.raises(error): + blosc2.RemoteCTable(url, **{name: value}) + assert repr(list(table[:1])) == repr(list(local[:1])) + for name in ("metadata_buffer_bytes", "row_buffer_bytes"): + with pytest.raises((TypeError, NotImplementedError), match=name): + blosc2.open(url, **{name: 1024}) + with blosc2.RemoteStore(url, _allow_array_root=True) as store: + first, second = store[""], store[""] + first.max_concurrency = 1 + assert second.max_concurrency == 8 + first.close() + second.close() + + +def test_parallel_scheduler_budget_and_cleanup(): + import threading + import time + + from blosc2.ctable_remote_read import run_reads + + lock = threading.Lock() + active = peak = 0 + closed = [] + + def fetch(size): + nonlocal active, peak + with lock: + active += size + peak = max(peak, active) + try: + time.sleep(0.01) + return bytes(size) + finally: + with lock: + active -= size + + def reader(i, size): + try: + data = yield fetch, (size,), size + return len(data) + finally: + closed.append(i) + + values, reserved = run_reads(((i, reader(i, 4)) for i in range(12)), 8, 10) + assert values == dict.fromkeys(range(12), 4) + assert 4 < peak <= reserved <= 10 + assert sorted(closed) == list(range(12)) + values, reserved = run_reads(((i, reader(i, 20)) for i in range(2)), 8, 10) + assert reserved == 20 # One oversized unit alone, never two together. + assert active == 0 + + closed.clear() + + def fail(): + raise OSError("injected worker failure") + + def broken(): + try: + yield fail, (), 4 + finally: + closed.append("broken") + + with pytest.raises(OSError, match="injected worker failure"): + run_reads([(0, broken()), (1, reader(1, 4))], 2, 10) + assert active == 0 + assert set(closed) == {"broken", 1} + + +@pytest.mark.parametrize( + "policy", [blosc2.CachePolicy.NONE, blosc2.CachePolicy.MEMORY, blosc2.CachePolicy.DISK] +) +def test_parallel_rows_blocks_and_reopen(tmp_path, monkeypatch, policy): + import threading + + @dataclasses.dataclass + class Numbers: + x: float + y: float + + rng = np.random.default_rng(42) + local = blosc2.CTable( + Numbers, + {"x": rng.normal(size=400000), "y": rng.normal(size=400000)}, + create_summary_index=False, + ) + url = remote_table_url(tmp_path, local) + writes = [] + original = blosc2.Proxy._write_blocks + + def write(self, *args): + writes.append(threading.current_thread()) + return original(self, *args) + + monkeypatch.setattr(blosc2.Proxy, "_write_blocks", write) + options = {"cache_policy": policy} + if policy == blosc2.CachePolicy.DISK: + options["cache_dir"] = tmp_path / "cache" + with blosc2.RemoteCTable(url, **options) as table: + before = table.traffic.nbytes + assert list(table[:10]) == list(local[:10]) + assert table.traffic.nbytes - before < 2 << 20 + assert writes + assert all(t is threading.current_thread() for t in writes) + if policy == blosc2.CachePolicy.DISK: + with blosc2.RemoteCTable(url, **options) as table: + before = table.traffic.requests + assert list(table[:10]) == list(local[:10]) + assert table.traffic.requests == before + assert list(table[399990:]) == list(local[399990:]) + + +def test_parallel_timestamp_and_projection(tmp_path): + @dataclasses.dataclass + class Timed: + time: object = blosc2.field(blosc2.timestamp(unit="ns", null_storage="mask")) + vec: object = blosc2.field(blosc2.ndarray((2,), dtype=blosc2.float32(), null_storage="mask")) + text: str = blosc2.field(blosc2.utf8()) + + local = blosc2.CTable( + Timed, + [(np.datetime64("2025-01-01", "ns"), [1, 2], "東京"), (None, None, "")], + create_summary_index=False, + ) + url = remote_table_url(tmp_path, local) + with blosc2.open(url) as table: + assert repr(list(table)) == repr(list(local)) + assert table.to_string() == local.to_string() + pd = pytest.importorskip("pandas") + pd.testing.assert_frame_equal(table.to_pandas(), local.to_pandas()) + pytest.importorskip("pyarrow") + assert table.to_arrow().equals(local.to_arrow()) + with blosc2.open(url) as table: + view = table[["text"]] + assert list(view) == list(local[["text"]]) + assert "_cols/time" not in table._remote_storage()._owner.sources + assert "_cols/vec" not in table._remote_storage()._owner.sources From 849197af8fe39eee73550021dd6793f00ff0a0a6 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 18 Sep 2026 08:31:00 +0200 Subject: [PATCH 05/82] Improve persistent remote table caching --- doc/guides/remote_arrays.md | 7 ++ examples/ctable/remote_handling.py | 6 +- src/blosc2/b2z_source.py | 29 +++++++- src/blosc2/ctable_remote_read.py | 2 + src/blosc2/ctable_storage.py | 5 ++ src/blosc2/remote_store.py | 30 ++++++-- tests/ctable/test_remote_ctable.py | 116 ++++++++++++++++++++++++++++- tests/test_b2z_source.py | 4 +- tests/test_remote_store.py | 30 ++++++++ 9 files changed, 216 insertions(+), 13 deletions(-) diff --git a/doc/guides/remote_arrays.md b/doc/guides/remote_arrays.md index b0d75cdd8..f6943baf9 100644 --- a/doc/guides/remote_arrays.md +++ b/doc/guides/remote_arrays.md @@ -633,6 +633,13 @@ For a replaceable `.b2nd` or Caterva2 source, pass `assume_immutable=False` to c ### Refreshing a RemoteStore Remote containers (B2Z, Zarr, and HDF5) are assumed immutable by default. +For B2Z tables and stores, reopening a populated disk cache trusts its saved +archive identity and metadata: no HEAD/identity request is made. Uncached data +still requires remote reads. Older caches may perform one identity lookup to +upgrade their metadata. Do not replace the remote archive while using its cache. +Use `store.refresh()` after a replacement; for a standalone `RemoteCTable`, +which has no `refresh()` method, open with a fresh `cache_dir` instead. + If a remote container is updated on the server—such as adding new datasets or appending data—call `store.refresh()` to update discovery: ```python diff --git a/examples/ctable/remote_handling.py b/examples/ctable/remote_handling.py index b6f94aed2..5927d69d5 100644 --- a/examples/ctable/remote_handling.py +++ b/examples/ctable/remote_handling.py @@ -121,7 +121,8 @@ def access_table(args) -> None: print(f"Accessing: {args.url}") started = time.perf_counter() - with blosc2.open(args.url, storage_options=storage_options or None) as table: + cache_options = {"cache_dir": args.cache_dir} if args.cache_dir is not None else {} + with blosc2.open(args.url, storage_options=storage_options or None, **cache_options) as table: if not isinstance(table, blosc2.CTable): raise ValueError("input must be a CTable archive") remote = isinstance(table, blosc2.RemoteCTable) @@ -221,6 +222,9 @@ def main() -> int: "url", nargs="?", help="Local .b2z CTable path or remote URL (s3://, http://, https://)" ) parser.add_argument("--write", type=Path, metavar="FILE", help="Create a local .b2z CTable instead") + parser.add_argument( + "--cache-dir", type=Path, metavar="DIR", help="Persist remote data in DIR (default: in-memory cache)" + ) parser.add_argument("--rows", type=int, default=1_000_000) parser.add_argument("--batch-size", type=int, default=100_000) parser.add_argument("--overwrite", action="store_true") diff --git a/src/blosc2/b2z_source.py b/src/blosc2/b2z_source.py index 6986c2193..c1d74e465 100644 --- a/src/blosc2/b2z_source.py +++ b/src/blosc2/b2z_source.py @@ -103,9 +103,14 @@ def __init__(self, urlpath, *, storage_options=None, _filesystem=None, _traffic= if isinstance(self._fs, HTTPFileSystem): bootstrap = sync(self._fs.loop, _http_tail, self._fs, self._path) - object_info = self._fs.info(self._path) if bootstrap is None else bootstrap[0] - self.object_info = object_info + legacy_metadata = bool(_metadata) and "object_info" not in _metadata + object_info = (_metadata or {}).get("object_info") + if object_info is None: + # Old manifests need one identity lookup to acquire this bootstrap. + object_info = self._fs.info(self._path) if bootstrap is None else bootstrap[0] size = object_info["size"] + if not isinstance(size, int) or isinstance(size, bool) or size < 0: + raise ValueError("Invalid cached B2Z archive size") self.size = size self.metadata = _metadata if _metadata is not None else {} self.persist_metadata = _metadata is not None @@ -113,6 +118,17 @@ def __init__(self, urlpath, *, storage_options=None, _filesystem=None, _traffic= if self.metadata and self.metadata.get("identity") != identity: raise ValueError("B2Z source changed; refresh the store cache") self.metadata.setdefault("identity", identity) + # Backend info can contain datetime or other non-msgpack values. Only + # size needs its native type; retain per-member stamps before normalizing. + self.metadata.setdefault( + "object_info", {**{k: str(v) for k, v in object_info.items()}, "size": size} + ) + # New readers (including concurrent sparse-cache handles) must derive + # identical stamps from the live and serialized forms. During upgrades, + # preserve the old stamp for existing payloads in member_stamps below. + self.object_info = ( + object_info if legacy_metadata or _metadata is None else self.metadata["object_info"] + ) self.metadata.setdefault("ranges", []) for offset, data in self.metadata["ranges"]: if ( @@ -331,7 +347,14 @@ def __init__( prefix_start, prefix = archive._opening_ranges[-1] from fsspec.utils import tokenize - self.stamp = tokenize(urlpath, object_info, dataset, self.member_offset, self.member_length) + # tokenize() uses dict repr: HEAD and range responses can contain + # identical fields in different insertion orders. + self.stamp = archive.metadata.setdefault("member_stamps", {}).setdefault( + dataset, + tokenize( + urlpath, sorted(object_info.items()), dataset, self.member_offset, self.member_length + ), + ) super().__init__(urlpath, max_concurrency, traffic=self.traffic) if b"b2o" in self._header[13][1]: raise NotImplementedError("B2Z object carriers are not supported; select a plain NDArray") diff --git a/src/blosc2/ctable_remote_read.py b/src/blosc2/ctable_remote_read.py index 4b331b3a4..39162e10a 100644 --- a/src/blosc2/ctable_remote_read.py +++ b/src/blosc2/ctable_remote_read.py @@ -121,6 +121,8 @@ def ranges_for(name): for key in keys: if key in owner.sources: continue + if key in archive.metadata.get("ctable_seeds", {}): + continue matches = members.get(key + ".b2nd", ()) if len(matches) != 1: # The ordinary opener supplies the appropriate diagnostic. diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index ff128c4a7..7c4b3beef 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -769,6 +769,9 @@ def load_user_attrs(self) -> dict: self._check_open() if hasattr(self, "_user_attrs"): return dict(self._user_attrs) + if self._root_key in self._owner.attrs: + self._user_attrs = dict(self._owner.attrs[self._root_key]) + return dict(self._user_attrs) member = self._full_key("_vlmeta") + ".b2f" matches = [info for info in self._owner.archive.members if info.filename == member] if not matches: @@ -778,6 +781,8 @@ def load_user_attrs(self) -> dict: from blosc2.b2z_source import member_vlmeta self._user_attrs = dict(member_vlmeta(self._owner.archive, matches[0])) + self._owner.attrs[self._root_key] = self._user_attrs + self._owner.save_manifest() return dict(self._user_attrs) def table_exists(self) -> bool: diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 2721714b1..d1ee288a9 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -520,9 +520,7 @@ def open_source(self, path): value if isinstance(value, str) else "Array access is unavailable for this node" ) if self.format == "b2z": - from blosc2.b2z_source import B2ZNDSource - - source = B2ZNDSource(self.urlpath, full, _archive=self.archive) + source = self._open_b2z_source(full) elif self.format == "hdf5": from blosc2.hdf5_source import HDF5NDSource @@ -561,15 +559,35 @@ def open_ctable_array(self, table_path, logical_key): raise NotImplementedError( f"Remote CTable array {full!r} requires an unencrypted ZIP_STORED member" ) - from blosc2.b2z_source import B2ZNDSource - - source = B2ZNDSource(self.urlpath, full, _archive=self.archive) + source = self._open_b2z_source(full) if self.source_validator is not None: self.source_validator(source) self.nodes[full] = ("ndarray", None) self.sources[full] = source return source + def _open_b2z_source(self, full): + from blosc2.b2z_source import B2ZNDSource + + # Immutable archives trust their persisted identity when restoring leaves. + seeds = self.archive.metadata.get("ctable_seeds", {}) + if full in seeds: + return B2ZNDSource( + self.urlpath, + full, + _seed=seeds[full], + storage_options=self.storage_options, + _filesystem=self.filesystem, + _traffic=self.traffic, + ) + source = B2ZNDSource(self.urlpath, full, _archive=self.archive) + if self.persist_metadata and any( + kind == "ctable" and full.startswith(root + "/" if root else "") + for root, (kind, _) in self.nodes.items() + ): + self.archive.metadata.setdefault("ctable_seeds", {})[full] = source._seed + return source + def remote_array(self, full): """Return a RemoteArray sharing this discovery owner's resources.""" relative = full[len(self.root) + 1 :] if self.root else full diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 2605cda62..e16ab5b70 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -49,7 +49,7 @@ def test_remote_example_total_timing(tmp_path, capsys, include_note, nrows): ticks = iter(range(10)) access = example["access_table"] access.__globals__["time"] = SimpleNamespace(perf_counter=lambda: next(ticks)) - access(SimpleNamespace(url=url)) + access(SimpleNamespace(url=url, cache_dir=None)) output = capsys.readouterr().out expected_ms = 5000 if include_note else 3000 assert f"\n\nTotal network : {expected_ms:7.1f} ms (" in output @@ -62,6 +62,120 @@ def test_remote_example_total_timing(tmp_path, capsys, include_note, nrows): assert sample == str(local[start:stop]).strip() +def test_remote_example_cache_dir(tmp_path, capsys, monkeypatch): + import runpy + import sys + from pathlib import Path + + local = blosc2.CTable(dataclasses.make_dataclass("Sample", [("x", int)]), [(i,) for i in range(20)]) + url = remote_table_url(tmp_path, local) + cache_dir = tmp_path / "cache" + script = Path(__file__).resolve().parents[2] / "examples/ctable/remote_handling.py" + example = runpy.run_path(str(script)) + monkeypatch.setattr(sys, "argv", [str(script), url, "--cache-dir", str(cache_dir)]) + for _ in range(2): + assert example["main"]() == 0 + output = capsys.readouterr().out + assert "cache_policy : DISK" in output + assert any(cache_dir.iterdir()) + assert "(0 requests," in output.split("Total network :")[1] + + +def test_disk_cache_metadata_key_order(tmp_path, monkeypatch): + local = blosc2.CTable(dataclasses.make_dataclass("Sample", [("x", int)]), [(i,) for i in range(20)]) + url = remote_table_url(tmp_path, local) + cache_dir = tmp_path / "cache" + with blosc2.open(url, cache_dir=cache_dir) as table: + np.testing.assert_array_equal(table["x"][:], np.arange(20)) + + fs = fsspec.filesystem("memory") + original_info = type(fs).info + + def reordered_info(self, path, **kwargs): + return dict(reversed(list(original_info(self, path, **kwargs).items()))) + + monkeypatch.setattr(type(fs), "info", reordered_info) + with blosc2.open(url, cache_dir=cache_dir) as table: + np.testing.assert_array_equal(table["x"][:], np.arange(20)) + assert table.traffic.requests == 0 + + def changed_info(self, path, **kwargs): + return {**original_info(self, path, **kwargs), "ETag": "changed"} + + monkeypatch.setattr(type(fs), "info", changed_info) + # Immutable reopening trusts the persisted identity, even if remote metadata changes. + with blosc2.open(url, cache_dir=cache_dir) as table: + np.testing.assert_array_equal(table["x"][:], np.arange(20)) + + +@pytest.mark.parametrize("max_concurrency", [1, 8]) +@pytest.mark.parametrize("legacy", [False, True]) +def test_disk_cache_reuses_ctable_bootstrap(tmp_path, monkeypatch, max_concurrency, legacy): + schema = dataclasses.make_dataclass("Sample", [("x", float), ("note", str, blosc2.field(blosc2.utf8()))]) + values = np.random.default_rng(42).random(20_000) + local = blosc2.CTable( + schema, + [(x, f"東京 café #{i}") for i, x in enumerate(values)], + create_summary_index=False, + ) + local._cols["x"] = blosc2.asarray(values, chunks=(2000,), blocks=(200,)) + local.attrs["description"] = "Persistent UTF-8 metadata" + url = remote_table_url(tmp_path, local) + # Export rechunks columns; retain a multi-chunk fixture for cold-row checks. + path = tmp_path / "grids.b2z" + with zipfile.ZipFile(tmp_path / "table.b2z") as source, zipfile.ZipFile(path, "w") as dest: + for info in source.infolist(): + data = local._cols["x"].to_cframe() if info.filename == "_cols/x.b2nd" else source.read(info) + dest.writestr(info, data) + fsspec.filesystem("memory").pipe(url, path.read_bytes()) + cache_dir = tmp_path / "cache" + options = {"cache_dir": cache_dir, "max_concurrency": max_concurrency} + with blosc2.open(url, **options) as table: + nbytes = table.nbytes + attrs = dict(table.attrs) + sample = list(table[9998:10003]) + notes = table["note"][-5:].tolist() + if legacy: + from fsspec.utils import tokenize + + owner = table._remote_storage()._owner + # Before object_info was persisted, stamps used native backend types + # (including memory:// creation datetimes), not normalized strings. + info = owner.archive._fs.info(owner.archive._path) + for key, source in owner.sources.items(): + stamp = tokenize(url, sorted(info.items()), key, source.member_offset, source.member_length) + owner.caches[key].schunk.vlmeta["proxy-stamp"] = stamp + owner.archive.metadata.pop("ctable_seeds") + owner.archive.metadata.pop("object_info") + owner.archive.metadata.pop("member_stamps") + owner.attrs.clear() + + if legacy: + # Old caches acquire bootstraps and attributes on their next access. + with blosc2.open(url, **options) as table: + assert table.nbytes == nbytes + assert dict(table.attrs) == attrs + + def unexpected_read(*args, **kwargs): + pytest.fail("Warm metadata/rows must not download archive bytes") + + with monkeypatch.context() as patch: + patch.setattr(type(fsspec.filesystem("memory")), "cat_file", unexpected_read) + patch.setattr(type(fsspec.filesystem("memory")), "info", unexpected_read) + with blosc2.open(url, **options) as table: + assert table.nbytes == nbytes + assert dict(table.attrs) == attrs + assert list(table[9998:10003]) == sample + assert table["note"][-5:].tolist() == notes + assert table.traffic.requests == 0 + + # Restored bootstraps must still support transport for uncached rows. + with blosc2.open(url, **options) as table: + np.testing.assert_array_equal(table["x"][:5], values[:5]) + assert table["note"][:5].tolist() == [f"東京 café #{i}" for i in range(5)] + assert table.traffic.requests > 0 + + def test_remote_ctable_fixed_width_reads_and_queries(tmp_path): rows = [ (1, np.array([1, 2], dtype=np.float32), "one"), diff --git a/tests/test_b2z_source.py b/tests/test_b2z_source.py index 76c6eb021..ab92f5d46 100644 --- a/tests/test_b2z_source.py +++ b/tests/test_b2z_source.py @@ -173,14 +173,14 @@ def do_GET(self): else: np.testing.assert_array_equal(arr[:], data) - # A persisted suffix bootstrap must have the same identity as HEAD. + # Immutable archive discovery reopens entirely from its saved bootstrap. metadata = {} archive = B2ZArchive(url, storage_options=options, _metadata=metadata) archive.close() requests.clear() archive = B2ZArchive(url, storage_options=options, _metadata=metadata) archive.close() - assert requests == [("HEAD", None)] + assert requests == [] finally: server.shutdown() server.server_close() diff --git a/tests/test_remote_store.py b/tests/test_remote_store.py index 93ca6f9fa..ef76fdc29 100644 --- a/tests/test_remote_store.py +++ b/tests/test_remote_store.py @@ -126,6 +126,34 @@ def test_store_none_rejects_limit(): ) +def test_b2z_disk_cache_trusts_snapshot_until_refresh(tmp_path, monkeypatch): + path = tmp_path / "source.b2z" + url = f"memory://{tmp_path.name}/source.b2z" + fs = fsspec.filesystem("memory") + cache = tmp_path / "cache" + for value in (1, 2): + with blosc2.TreeStore(path, mode="w", threshold=0) as tree: + tree["a"] = blosc2.asarray(np.full(10, value, dtype="i4")) + fs.pipe(url, path.read_bytes()) + if value == 1: + with blosc2.RemoteStore(url, cache_dir=cache) as store, store["a"] as array: + np.testing.assert_array_equal(array[:], 1) + + def forbidden(*args, **kwargs): + pytest.fail("Immutable reopening must not contact the remote source") + + with monkeypatch.context() as patch: + patch.setattr(type(fs), "info", forbidden) + patch.setattr(type(fs), "cat_file", forbidden) + with blosc2.RemoteStore(url, cache_dir=cache) as store, store["a"] as array: + np.testing.assert_array_equal(array[:], 1) + + with blosc2.RemoteStore(url, cache_dir=cache) as store: + store.refresh() + with store["a"] as array: + np.testing.assert_array_equal(array[:], 2) + + def test_disk_reopen_all_leaves_and_refresh(hierarchy, tmp_path, monkeypatch): url, data = hierarchy parent = tmp_path / "cache" @@ -145,6 +173,8 @@ def forbidden(*args, **kwargs): raise AssertionError("reopen fetched remote bytes") patch.setattr(type(fsspec.filesystem("memory")), "cat_file", forbidden) + if url.endswith(".b2z"): + patch.setattr(type(fsspec.filesystem("memory")), "info", forbidden) reopened = blosc2.RemoteStore(url, cache_dir=parent) reopened.close() with blosc2.RemoteStore(url, cache_dir=parent) as store: From 7c44f13972b7f8a01572b5a06509dc71e603277b Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 18 Sep 2026 08:32:32 +0200 Subject: [PATCH 06/82] Align RemoteCTable cache lifecycle APIs --- doc/guides/remote_arrays.md | 20 ++++++- src/blosc2/ctable.py | 15 ++++- src/blosc2/remote_ctable.py | 50 ++++++++++++++++ src/blosc2/remote_store.py | 50 ++++++++++------ tests/ctable/test_remote_ctable.py | 91 ++++++++++++++++++++++++++++++ 5 files changed, 205 insertions(+), 21 deletions(-) diff --git a/doc/guides/remote_arrays.md b/doc/guides/remote_arrays.md index f6943baf9..2f20eb797 100644 --- a/doc/guides/remote_arrays.md +++ b/doc/guides/remote_arrays.md @@ -602,6 +602,9 @@ read-only artifacts retain their existing guarded row-read paths; standalone RemoteArray concurrency and RemoteStore discovery behavior are unchanged. Columns and views are borrowed from the root table and require it to remain open. +`table.is_cache_mutable` reports whether the local cache is writable, matching +the corresponding RemoteStore and RemoteArray property. It is read-only and does +not imply that the remote table can be modified. Closing a parent RemoteStore leaves a returned table usable; refreshing the store invalidates previously returned tables and their columns. Copies and data exports produce local tables. Remote writes, batch-backed `vlstring`/lists/objects, @@ -637,8 +640,21 @@ For B2Z tables and stores, reopening a populated disk cache trusts its saved archive identity and metadata: no HEAD/identity request is made. Uncached data still requires remote reads. Older caches may perform one identity lookup to upgrade their metadata. Do not replace the remote archive while using its cache. -Use `store.refresh()` after a replacement; for a standalone `RemoteCTable`, -which has no `refresh()` method, open with a fresh `cache_dir` instead. +Use `store.refresh()` after a replacement, or `table.refresh()` for a standalone +`RemoteCTable`. Neither operation writes to the remote source. + +```python +with blosc2.RemoteCTable(url, cache_dir="table-cache") as table: + table.refresh() # rediscover the remote table and replace its cache generation + print(table[:5]) +``` + +Table refresh reloads the schema, row information and column backends while +preserving cache policy, cache limits and parallel-read settings. Previously +obtained columns, raw arrays and views become unusable; retrieve them again from +the refreshed table. Discovery or table initialization failure leaves the old +table usable. A table obtained from a `RemoteStore` must be refreshed through the +root store, then retrieved again; it cannot independently refresh shared discovery. If a remote container is updated on the server—such as adding new datasets or appending data—call `store.refresh()` to update discovery: diff --git a/src/blosc2/ctable.py b/src/blosc2/ctable.py index a8276ad74..bea31f463 100644 --- a/src/blosc2/ctable.py +++ b/src/blosc2/ctable.py @@ -1400,10 +1400,17 @@ class Column: _REPR_PREVIEW_ITEMS = 8 def __init__(self, table: CTable, col_name: str, mask=None): - self._table = table + self._table_ref = table + self._remote_storage_ref = table._remote_read_storage() self._col_name = col_name self._mask = mask + @property + def _table(self): + if self._remote_storage_ref is not None: + self._remote_storage_ref._check_open() + return self._table_ref + @property def _nulls(self) -> NullChannel: """This column's validity channel (see :mod:`blosc2.ctable_nulls`). @@ -6494,6 +6501,10 @@ def _row_namedtuple_type(self): def _remote_read_storage(self): from blosc2.ctable_storage import RemoteTableStorage + saved = getattr(self, "_remote_view_storage", None) + if saved is not None: + saved._check_open() + return saved table = self while table.base is not None: table = table.base @@ -7365,6 +7376,7 @@ def _make_view(cls, parent: CTable, new_valid_rows: blosc2.NDArray) -> CTable: obj._table_dparams = parent._table_dparams obj._storage = None obj._read_only = parent._read_only # inherit: only True for mode="r" disk tables + obj._remote_view_storage = parent._remote_read_storage() obj._schema = parent._schema obj._cols = parent._cols # shared — views cannot change row structure obj._computed_cols = parent._computed_cols # shared — LazyExpr refs remain valid @@ -7678,6 +7690,7 @@ def select(self, cols: list[str]) -> CTable: obj._table_dparams = self._table_dparams obj._storage = None obj._read_only = self._read_only + obj._remote_view_storage = self._remote_read_storage() obj._valid_rows = self._valid_rows obj._n_rows = self._known_n_rows() obj._last_pos = self._last_pos diff --git a/src/blosc2/remote_ctable.py b/src/blosc2/remote_ctable.py index 2884ac9bf..cd6762a05 100644 --- a/src/blosc2/remote_ctable.py +++ b/src/blosc2/remote_ctable.py @@ -134,6 +134,51 @@ def close(self) -> None: if isinstance(storage, RemoteTableStorage): storage.close() + def refresh(self) -> None: + """Reload a standalone table and invalidate its old columns and views. + + Preserve the table and its cache on discovery/initialization failure. + For tables obtained from a RemoteStore, refresh the root store instead. + """ + storage = self._remote_storage() + owner = storage._owner + with owner.lock: + storage._check_open() + if self.base is not None or owner.root != storage._root_key or owner.is_tree: + raise ValueError("Refresh the root RemoteStore, then retrieve this table again") + replacement = owner.prepare_refresh("ctable") + replacement.acquire() # Keep failed initialization from closing the borrowed disk cache. + fresh = None + try: + fresh = type(self)._from_owner( + replacement, + replacement.root, + max_concurrency=storage.max_concurrency, + metadata_buffer_bytes=storage.metadata_buffer_bytes, + row_buffer_bytes=storage.row_buffer_bytes, + ) + replacement.restoring = False + replacement.save_manifest() + except BaseException: + replacement.disk = None + if fresh is not None: + fresh.close() + replacement.release() + raise + replacement.release() + replacement._cleanup_dir, owner._cleanup_dir = owner._cleanup_dir, None + replacement.artifact_path = owner.artifact_path + owner.disk = None + owner.generation = replacement.generation + state = fresh.__dict__.copy() + fresh._storage = None # Ownership is transferred to this handle. + self.__dict__ = state + self._cols._table = self + storage.close() + owner.close() + if replacement.disk is not None: + replacement.disk.discard_old_generations(replacement.generation) + @property def vlmeta(self): return RemoteMetadataMapping(self._remote_storage().load_user_attrs()) @@ -161,6 +206,11 @@ def cache_policy(self): def max_cache_bytes(self): return self._remote_storage()._owner.max_cache_bytes + @property + def is_cache_mutable(self) -> bool: + """Whether the current local cache is writable, not the remote table.""" + return self._remote_storage()._owner.is_mutable + @property def cache_bytes(self): return self._remote_storage()._owner.cache_coordinator.cache_bytes diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index d1ee288a9..dc3573970 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -717,6 +717,36 @@ def get_cache(self, source, *, seed=None): self.save_manifest() return self.caches[key] + def prepare_refresh(self, kind): + """Prepare fresh discovery without publishing or retiring the current generation.""" + if not self.is_mutable: + raise ValueError("Cannot refresh an immutable remote artifact; use a writable cache") + replacement = RemoteDiscovery( + self.urlpath, + self.storage_options, + dataset=self.root, + persist_metadata=self.disk is not None, + _filesystem=self._external_filesystem, + _source_validator=self.source_validator, + _manifest_validator=self.manifest_validator, + _max_nodes=self.max_nodes, + _source_format=self.format, + ) + try: + if replacement.nodes[replacement.root][0] != kind: + raise ValueError(f"Refreshed source is no longer a {kind}") + replacement.cache_policy = self.cache_policy + replacement.max_cache_bytes = self.max_cache_bytes + replacement.cache_coordinator = CacheCoordinator(self.max_cache_bytes) + replacement.shared = getattr(self, "shared", False) + replacement.mutable = self.mutable + replacement.restoring = True + replacement.disk = self.disk + return replacement + except BaseException: + replacement.close() + raise + def close(self): if self._closed: return @@ -1175,25 +1205,9 @@ def refresh(self): if self._path: raise ValueError("refresh must be called on the root store handle") owner = self._owner - replacement = RemoteDiscovery( - owner.urlpath, - owner.storage_options, - dataset=owner.root, - persist_metadata=owner.disk is not None, - _filesystem=owner._external_filesystem, - _source_validator=owner.source_validator, - _manifest_validator=owner.manifest_validator, - _max_nodes=owner.max_nodes, - _source_format=owner.format, - ) + replacement = owner.prepare_refresh("group") try: - if not replacement.is_tree: - raise ValueError("Refreshed source is no longer a group") - replacement.disk = owner.disk - replacement.cache_policy = owner.cache_policy - replacement.max_cache_bytes = owner.max_cache_bytes - replacement.shared = getattr(owner, "shared", False) - replacement.mutable = owner.mutable + replacement.restoring = False replacement.save_manifest() if replacement.disk is not None: replacement.disk.discard_old_generations(replacement.generation) diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index e16ab5b70..1ae3b07f4 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -176,6 +176,95 @@ def unexpected_read(*args, **kwargs): assert table.traffic.requests > 0 +@pytest.mark.parametrize( + "policy", [blosc2.CachePolicy.NONE, blosc2.CachePolicy.MEMORY, blosc2.CachePolicy.DISK] +) +def test_remote_ctable_refresh(tmp_path, policy): + schema = dataclasses.make_dataclass("Sample", [("x", int), ("note", str, blosc2.field(blosc2.utf8()))]) + local = blosc2.CTable(schema, [(1, "old"), (2, "café")], create_summary_index=False) + url = remote_table_url(tmp_path, local) + options = {"cache_policy": policy, "max_concurrency": 2, "row_buffer_bytes": 1 << 20} + if policy is blosc2.CachePolicy.DISK: + options["cache_dir"] = tmp_path / "cache" + with blosc2.RemoteCTable(url, **options) as table: + column, raw = table["x"], table["x"].raw + view, projection = table[:1], table.select("x") + assert column[:].tolist() == [1, 2] + changed_schema = dataclasses.make_dataclass( + "Changed", [("x", int), ("note", str, blosc2.field(blosc2.utf8())), ("active", bool)] + ) + changed_path = str(tmp_path / "changed.b2d") + with blosc2.CTable( + changed_schema, [(9, "東京", True)], create_summary_index=False, urlpath=changed_path, mode="w" + ) as changed: + changed.attrs["version"] = 2 + with blosc2.CTable.open(changed_path) as changed: + remote_table_url(tmp_path, changed, "changed") + fsspec.filesystem("memory").pipe(url, (tmp_path / "changed.b2z").read_bytes()) + assert table.refresh() is None + assert table.col_names == ["x", "note", "active"] + assert table.nrows == 1 + assert table["x"][:].tolist() == [9] + assert table["note"][:].tolist() == ["東京"] + assert dict(table.attrs) == {"version": 2} + assert table.max_concurrency == 2 + assert table.row_buffer_bytes == 1 << 20 + for read in (lambda: column[:], lambda: raw[:], lambda: list(view), lambda: list(projection)): + with pytest.raises(RuntimeError, match=r"stale|closed"): + read() + table.refresh() + assert table["x"][:].tolist() == [9] + if policy is blosc2.CachePolicy.DISK: + with blosc2.RemoteCTable(url, **options) as table: + assert table["note"][:].tolist() == ["東京"] + + +@pytest.mark.parametrize("failure", ["discovery", "initialization", "publication"]) +def test_remote_ctable_failed_refresh(tmp_path, monkeypatch, failure): + local = blosc2.CTable(dataclasses.make_dataclass("Sample", [("x", int)]), [(1,), (2,)]) + url = remote_table_url(tmp_path, local) + cache = tmp_path / "cache" + with blosc2.RemoteCTable(url, cache_dir=cache) as table: + column = table["x"] + assert column[:].tolist() == [1, 2] + owner = table._remote_storage()._owner + generation = owner.generation + + def fail(*args, **kwargs): + raise OSError("refresh failed") + + with monkeypatch.context() as patch: + if failure == "discovery": + patch.setattr(type(fsspec.filesystem("memory")), "info", fail) + elif failure == "initialization": + from blosc2.ctable_storage import RemoteTableStorage + + patch.setattr(RemoteTableStorage, "open_valid_rows", fail) + else: + patch.setattr(owner.disk, "publish", fail) + with pytest.raises(OSError, match="refresh failed"): + table.refresh() + assert owner.generation == generation + assert column[:].tolist() == [1, 2] + assert table["x"][:].tolist() == [1, 2] + table.refresh() + assert table["x"][:].tolist() == [1, 2] + + +def test_remote_ctable_is_cache_mutable(tmp_path, monkeypatch): + local = blosc2.CTable(dataclasses.make_dataclass("Sample", [("x", int)]), [(1,)]) + url = remote_table_url(tmp_path, local) + with blosc2.RemoteCTable(url) as table: + assert table.is_cache_mutable is True + with monkeypatch.context() as patch: + patch.setattr(table._remote_storage()._owner, "is_mutable", False) + assert table.is_cache_mutable is False + with pytest.raises(AttributeError): + table.is_cache_mutable = False + with pytest.raises(RuntimeError, match="closed"): + _ = table.is_cache_mutable + + def test_remote_ctable_fixed_width_reads_and_queries(tmp_path): rows = [ (1, np.array([1, 2], dtype=np.float32), "one"), @@ -358,6 +447,8 @@ class TextRow: np.testing.assert_array_equal(raw[:], local["nested.text"][:]) assert raw.nbytes == local["nested.text"].raw.nbytes view = table[:1] + with pytest.raises(ValueError, match="root RemoteStore"): + table.refresh() store.refresh() for read in (lambda: raw[:0], lambda: raw.shape, lambda: raw.cbytes, lambda: view["nested.text"][:]): with pytest.raises(RuntimeError, match="stale"): From 4a174d2df8e142d0f439f222da022b02c555cabf Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 18 Sep 2026 08:32:59 +0200 Subject: [PATCH 07/82] Plan RemoteCTable reference saving --- plans/remote-ctable-save.md | 182 ++++++++++++++++++++++++++++++++++++ 1 file changed, 182 insertions(+) create mode 100644 plans/remote-ctable-save.md diff --git a/plans/remote-ctable-save.md b/plans/remote-ctable-save.md new file mode 100644 index 000000000..6166505fe --- /dev/null +++ b/plans/remote-ctable-save.md @@ -0,0 +1,182 @@ +# RemoteCTable reference saving + +Status: deferred proposal for later consideration. This document does not change +the current save behavior. + +## Goal + +Make `RemoteCTable.save()` preserve a remote reference and optional retained +cache, consistent with `RemoteStore.save()`. Keep explicit independent local +exports available through `to_b2z()`, `to_b2d()` and `copy()`. + +Do not change `CTable.save()` for ordinary local tables. Reuse the existing +RemoteStore artifact format, cache ownership and reader infrastructure rather +than introduce a second table-specific reference format. + +## Current behavior and dependencies + +- RemoteCTable inherits CTable.save(), which materializes live rows into a local + table and returns None. +- RemoteStore.save() writes a portable .b2z reference archive containing its + source descriptor, discovery metadata and optionally already-retained data. + It returns the output path and does not fetch all missing data. +- RemoteStore._collect_export_nodes() explicitly rejects CTable nodes. Its + existing export logic also does not preserve CTable schema metadata as needed + for reconstructing those nodes. Removing the rejection alone is insufficient. +- CTable.to_b2z() and to_b2d() dispatch to self.save() on remote root tables. + These must retain materializing semantics when RemoteCTable overrides save(). +- No direct RemoteCTable.save() callers were found in the repository during the + current audit. test_remote_utf8_nested_lifetime indirectly relies on save() + through table.to_b2z(), and verifies an independent local table. External + callers are unknown; document the change rather than assuming none exist. +- RemoteCTable now exposes refresh() and is_cache_mutable. Artifact readers + must preserve their cache-writability and stale-handle contracts. + +## Proposed public contract + +Proposed signature, to confirm when implementing: + +```python +def save(destination, *, include_cache=True, mutable=None, overwrite=False) -> str: ... +``` + +Match RemoteStore.save() for accepted options, defaults, .b2z destinations, +return value, overwrite checks and destination safety. This changes both the +meaning and return value of the inherited method; note the destination keyword +change from CTable.save(urlpath=...) as well. + +- include_cache=True exports only already-retained data, not missing chunks. +- include_cache=False exports enough metadata to reconstruct the table without + its cached column payloads. Bootstrap metadata is still necessary; do not + promise a file containing literally zero source bytes. +- mutable controls whether the reopened local artifact/cache can grow, not + whether the remote table becomes writable. Reuse the store's default rules. +- blosc2.open(saved_path) returns RemoteCTable for a table-root reference and + RemoteStore for a group-root reference. Ordinary materialized CTable archives + continue to open as local CTable instances. +- A standalone table and a full table selected from RemoteStore can be saved. + Row-filtered/projected/computed views are not new remote-reference objects in + this phase; their existing explicit materialization paths remain available. +- Cached reads use local data; missing reads contact the original source. + A saved reference is not a guaranteed complete offline copy or a frozen copy + of the original remote object. + +## Implementation steps + +### 1. Preserve explicit materialization before overriding save + +Audit every self.save() dispatch in CTable conversion helpers. Make the root +remote to_b2z()/to_b2d() paths call the materializing implementation explicitly, +or use a narrowly shared internal writer if that avoids duplication. Keep local +fast pack/unpack paths and view compaction unchanged. + +Do not require copy().save() merely to bypass virtual dispatch: that can add an +unnecessary whole-table in-memory copy. Retain the current writer's resource +behavior. Test the explicit export methods before changing reference saving. + +### 2. Extend the shared artifact exporter + +Allow a selected CTable root and CTable descendants of exported groups. Preserve +the CTable node metadata, schema, user attributes and source dataset selection. +Retain original archive member paths consistently with source descriptors, +bootstrap seeds and fingerprints; avoid casually rebasing paths. + +Account for all physical arrays, not just visible logical column names: + +- Fixed-width and shaped columns. +- UTF-8 offsets and bytes companions. +- Null-mask sidecars and the live-row/deletion mask. + +Export existing cached portions through the existing carrier-copy machinery. +Do not eagerly open every column, decode every string or fetch missing chunks +to produce an include_cache=True artifact. Metadata-only and never-read tables +must also export correctly. Keep cache-budget checks and atomic file publication. + +Keep destination exclusions: no overwriting a live cache, the source artifact, +or another existing destination unless explicitly authorized by overwrite=True. +Use the existing source-descriptor/storage-options fingerprint handling; never +introduce serialized credentials or live filesystem/session objects. + +### 3. Extend shared artifact validation and dispatch + +Trace artifact loading end to end: manifest validation, immutable/mutable +opening, owner initialization, root-kind checks, blosc2.open() dispatch and +nested lookup. Support table roots without loosening group/array validation or +accepting malformed CTable schema and member mappings. + +Reuse RemoteCTable._from_owner() and RemoteTableStorage. Preserve owner lifetime +rules so closing a parent handle does not prematurely close a returned table. +Ensure cache-only export/reopen does not reintroduce remote identity requests +under the immutable-source assumption when all bootstrap metadata is present. + +### 4. Implement RemoteCTable.save() + +Delegate to the shared exporter with the table's selected dataset; do not +duplicate ZIP construction and manifest serialization. Keep root/selection +checks explicit and preserve closed/stale-handle errors. + +Runtime max_concurrency and buffer settings are not a new serialized artifact +contract in this phase: use defaults or caller overrides on reopen, consistent +with the existing readers. Do not extend blosc2.open() with table buffer options. + +### 5. Integrate artifact lifecycle + +Verify immutable artifacts report is_cache_mutable=False, never modify their +files during reads and reject refresh(). Missing data may still be fetched into +transient memory, as with store artifacts. + +Verify mutable artifacts use the existing writable-cache machinery, preserve +cached data across reopening and support standalone table.refresh(). Refresh +must retain settings, replace the cache generation and invalidate borrowed +columns/views. Tables obtained through a group still refresh via the root store. + +### 6. Documentation and migration + +Document reference saving versus materialization side by side: + +```python +remote.save("reference.b2z") # source + metadata + retained cache +remote.save("cold-reference.b2z", include_cache=False) +remote.to_b2z("complete-local.b2z") # independent table data +remote.to_b2d("complete-local.b2d") +local = remote.copy() # independent in-memory CTable +``` + +Explain network access, immutability, storage credentials on another machine, +cache writability and overwrite behavior. State explicitly that local +CTable.save() remains unchanged. Update the guide's current statement that +portable table-reference export is unsupported only once implemented and tested. + +## Verification + +Use focused fixtures and parametrization rather than live-network-only tests: + +1. Root and nested table reference round trips through blosc2.open(); mixed + groups containing arrays and tables continue to return the appropriate types. +2. Fixed-width/shaped/nullable columns, multilingual UTF-8, empty tables and + deleted rows preserve schema, attributes and values. +3. include_cache=True preserves partial warm reads; include_cache=False leaves + data cold. Instrument info and byte-range transport separately. Saving does + not fetch missing payloads, and a fully cached selection reopens locally. +4. Uncached reads after reopening still work, including UTF-8 spans requiring + both offsets and bytes. Exporting a nested table does not include unrelated + sibling data or confuse absolute dataset paths. +5. Immutable and mutable artifacts obey existing cache policy/budget rules; + refresh, closing, parent lifetime and stale handles behave consistently. +6. Failure/overwrite/path-safety tests leave existing artifacts and caches + intact. Invalid manifests and unsupported table kinds fail clearly. +7. to_b2z()/to_b2d() remain standalone local exports: remove the test remote + source or forbid transport before reopening and reading the exported table. +8. Existing RemoteStore and RemoteArray reference artifacts remain readable. + Version the manifest only if its actual compatibility requirements demand it. + +Run focused CTable, B2Z and RemoteStore tests, then the full suite and Ruff in +the blosc2 conda environment. Finish only when reference saving no longer +changes the semantics of explicit materialization methods. + +## Boundaries + +No remote writes, automatic change detection, new remote formats, automatic +full-cache population, reference export of arbitrary table views, or changes +to local CTable.save(). The proposal is intentionally deferred; revisit the +public signature and compatibility notes before starting implementation. From cd399a0f1270e122083d4bba2b45eff1fca2133f Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 18 Sep 2026 12:34:44 +0200 Subject: [PATCH 08/82] Add initial RemoteObject API --- doc/reference/classes.rst | 12 +++++ doc/reference/remotearray.rst | 8 +-- doc/reference/remoteobject.rst | 30 +++++++++++ src/blosc2/__init__.py | 2 + src/blosc2/ctable.py | 24 ++++++++- src/blosc2/remote_array.py | 71 +++++++++++++++++--------- src/blosc2/remote_ctable.py | 20 +++++++- src/blosc2/remote_object.py | 31 +++++++++++ src/blosc2/remote_store.py | 9 ++-- tests/ctable/test_remote_ctable.py | 23 +++++++++ tests/ctable/test_table_persistency.py | 11 ++++ tests/test_b2z_source.py | 8 ++- tests/test_locking.py | 21 +++++--- tests/test_python_blosc.py | 25 +++++---- tests/test_remote_array.py | 38 ++++++++++++-- tests/test_remote_store.py | 8 ++- 16 files changed, 281 insertions(+), 60 deletions(-) create mode 100644 doc/reference/remoteobject.rst create mode 100644 src/blosc2/remote_object.py diff --git a/doc/reference/classes.rst b/doc/reference/classes.rst index 77f9c5c66..64913e31b 100644 --- a/doc/reference/classes.rst +++ b/doc/reference/classes.rst @@ -115,6 +115,17 @@ codecs, filters, and remote paths. URLPath +Remote Data +----------- + +.. autosummary:: + + RemoteObject + RemoteArray + RemoteStore + RemoteCTable + + Ancillary / Advanced Classes ---------------------------- @@ -142,6 +153,7 @@ container APIs above. list_array objectarray proxy + remoteobject remotearray remotestore proxysource diff --git a/doc/reference/remotearray.rst b/doc/reference/remotearray.rst index 262485645..257e87ff7 100644 --- a/doc/reference/remotearray.rst +++ b/doc/reference/remotearray.rst @@ -144,10 +144,10 @@ is always returned: specifying ``cache_dir`` or ``cache_path`` configures it wit By default, :meth:`RemoteArray.save ` and :meth:`RemoteArray.to_cframe ` include valid warm -chunks for DISK proxies; MEMORY proxies always export cold carriers. -Pass ``include_cache=False`` for a cold carrier without changing the -warm original. The cache policy and limit remain in both forms; local paths and -authentication data are not serialized. +chunks already retained by DISK or MEMORY proxies. Pass ``include_cache=False`` +for a cold carrier without changing the warm original. The cache policy and +limit remain in both forms; local paths and authentication data are not +serialized. Pass ``cache_policy=blosc2.CachePolicy.NONE`` (or another policy) to either export method to produce a cold carrier with an explicit policy, leaving the diff --git a/doc/reference/remoteobject.rst b/doc/reference/remoteobject.rst new file mode 100644 index 000000000..d2ec4e8f3 --- /dev/null +++ b/doc/reference/remoteobject.rst @@ -0,0 +1,30 @@ +.. _RemoteObject: + +RemoteObject +============ + +``RemoteObject`` is the public base class for :class:`blosc2.RemoteArray`, +:class:`blosc2.RemoteStore`, and :class:`blosc2.RemoteCTable`. It is useful for +type checks; construct a concrete type or use :func:`blosc2.open` instead of +constructing ``RemoteObject`` directly. + +.. code-block:: python + + import blosc2 + + obj = blosc2.open("https://host/data.b2z") + assert isinstance(obj, blosc2.RemoteObject) + +Remote objects expose a credential-free ``source``, read-only ``attrs``, source +``traffic``, cache policy and accounting properties, an export-default +``mutable`` flag, and ``close()`` with context-manager support. Cache accounting +has concrete-type scope: arrays report their own retained payload, while stores +and tables report their shared owner's retained payload. + +``RemoteObject`` is not a factory, storage backend, serialization format, or +remote-write API. Data-specific operations remain on the concrete classes. +Arrays and tables provide ``materialize()``; stores are navigated to a leaf that +can be materialized. + +.. autoclass:: blosc2.RemoteObject + :members: diff --git a/src/blosc2/__init__.py b/src/blosc2/__init__.py index f299c93af..72cc28ab8 100644 --- a/src/blosc2/__init__.py +++ b/src/blosc2/__init__.py @@ -619,6 +619,7 @@ def _raise(exc): jit, as_simpleproxy, ) +from .remote_object import RemoteObject from .remote_array import RemoteMetadataMapping, RemoteArray from .remote_store import RemoteNode, RemoteStore from . import linalg @@ -915,6 +916,7 @@ def _raise(exc): "ProxySource", "Ref", "RemoteMetadataMapping", + "RemoteObject", "RemoteArray", "RemoteCTable", "RemoteNode", diff --git a/src/blosc2/ctable.py b/src/blosc2/ctable.py index bea31f463..3c2f10c18 100644 --- a/src/blosc2/ctable.py +++ b/src/blosc2/ctable.py @@ -6784,7 +6784,7 @@ def to_b2z(self, urlpath: str, *, overwrite: bool = False, compact: bool = False materialized = self.copy(compact=True) materialized.save(urlpath, overwrite=overwrite) else: - self.save(urlpath, overwrite=overwrite) + CTable.save(self, urlpath, overwrite=overwrite) return os.path.abspath(urlpath) def to_b2d(self, urlpath: str, *, overwrite: bool = False, compact: bool = False) -> str: @@ -6840,7 +6840,7 @@ def to_b2d(self, urlpath: str, *, overwrite: bool = False, compact: bool = False materialized = self.copy(compact=True) materialized.save(urlpath, overwrite=overwrite) else: - self.save(urlpath, overwrite=overwrite) + CTable.save(self, urlpath, overwrite=overwrite) return os.path.abspath(urlpath) def to_cframe(self) -> bytes: @@ -14331,6 +14331,26 @@ def _sorted_copy_from_positions(self, sorted_pos: np.ndarray, n: int) -> CTable: result._last_pos = n return result + def materialize( + self, + *, + urlpath: str | os.PathLike[str] | None = None, + overwrite: bool = False, + compact: bool = True, + chunks: int | tuple[int, ...] | None = None, + blocks: int | tuple[int, ...] | None = None, + cparams: dict[str, Any] | None = None, + ) -> CTable: + """Return an independent local copy of this table or view.""" + return self.copy( + compact=compact, + urlpath=urlpath, + overwrite=overwrite, + chunks=chunks, + blocks=blocks, + cparams=cparams, + ) + def copy( # noqa: C901 self, compact: bool = True, diff --git a/src/blosc2/remote_array.py b/src/blosc2/remote_array.py index 0c525caf6..7065eaccf 100644 --- a/src/blosc2/remote_array.py +++ b/src/blosc2/remote_array.py @@ -14,6 +14,7 @@ import json import math import os +import tempfile import threading import weakref from collections.abc import Mapping @@ -36,6 +37,7 @@ ) from blosc2.core import fsspec_cache_path, parse_container_url, storage_options_fingerprint from blosc2.info import InfoReporter, format_nbytes_info +from blosc2.remote_object import RemoteObject DEFAULT_DISK_CACHE_BYTES = 256 * 2**20 @@ -506,7 +508,7 @@ def _resolve_init_dataset_and_url(urlpath, dataset, source_format, hdf5_index=No return urlpath, resolved_dataset, resolved_format -class RemoteArray(blosc2.Operand): +class RemoteArray(RemoteObject, blosc2.Operand): """A persistable, optionally self-caching reference to a remote array. With :attr:`CachePolicy.DISK`, the public constructor uses the persisted @@ -1056,11 +1058,20 @@ def _attach_carrier_cache(self): _persistent_dirty=self._shared_runtime_cache, ) elif self.cache_policy is blosc2.CachePolicy.MEMORY: - self._proxy = blosc2.Proxy( + proxy = blosc2.Proxy( self.src, _refresh_source=False, _max_cache_bytes=self._cache_limit, ) + if self._carrier is not None: + self._import_warm_seed(self._carrier, proxy._cache) + proxy = blosc2.Proxy( + self.src, + _cache=proxy._cache, + _refresh_source=False, + _max_cache_bytes=self._cache_limit, + ) + self._proxy = proxy else: self._proxy = None @@ -1689,7 +1700,7 @@ def _export_carrier(self, include_cache: bool, cache_policy=None, mutable=None): if key.startswith("proxy-") or key == _B2OBJECT_USER_VLMETA_KEY: carrier.schunk.vlmeta[key] = runtime_schunk.vlmeta[key] return carrier - if include_cache and self._store_owner is not None and self._proxy is not None: + if include_cache and self._proxy is not None: carrier = self._to_b2object_carrier(mutable=effective_mutable) for nchunk in self._proxy._cache_sizes: carrier.schunk.update_chunk(nchunk, self._proxy.schunk.get_chunk(nchunk)) @@ -1703,7 +1714,7 @@ def _export_carrier(self, include_cache: bool, cache_policy=None, mutable=None): def to_cframe( self, *, include_cache: bool = True, cache_policy=None, mutable: bool | None = None ) -> bytes: - """Export a carrier. Only DISK preserves warm chunks by default. + """Export a carrier containing retained chunks by default. An explicit cache_policy exports a cold carrier with that policy. """ @@ -1714,33 +1725,55 @@ def to_cframe( @_serialized_operation def save( self, - urlpath: str | os.PathLike, + destination: str | os.PathLike | None = None, contiguous: bool = True, *, + urlpath: str | os.PathLike | None = None, include_cache: bool = True, cache_policy=None, mutable: bool | None = None, + overwrite: bool = False, **kwargs, ) -> str: - """Save a carrier; MEMORY exports are cold. See :meth:`to_cframe`. + """Save a carrier containing retained chunks by default. - Return the written ``urlpath``. + ``urlpath`` is retained as a compatibility alias for ``destination``. + Return the written destination. """ if mutable is not None and not isinstance(mutable, bool): raise TypeError("mutable must be a boolean") - urlpath = os.fspath(urlpath) + if destination is None: + if urlpath is None: + raise TypeError("save() missing required destination") + destination = urlpath + elif urlpath is not None: + raise TypeError("destination and urlpath cannot both be specified") + destination = os.fspath(destination) if (cache_policy is not None or not include_cache) and any( - path is not None and os.path.abspath(path) == os.path.abspath(urlpath) + path is not None and os.path.abspath(path) == os.path.abspath(destination) for path in (self.cache_path, self.runtime_cache_path) ): raise ValueError("cold or policy-changing export requires a different destination") carrier = self._export_carrier(include_cache, cache_policy, mutable=mutable) source_path = getattr(carrier.schunk, "urlpath", None) - if source_path is not None and os.path.abspath(source_path) == os.path.abspath(urlpath): - return urlpath - blosc2.blosc2_ext.check_access_mode(urlpath, "w") - carrier.save(urlpath, contiguous=contiguous, **kwargs) - return urlpath + same_live_carrier = source_path is not None and os.path.abspath(source_path) == os.path.abspath( + destination + ) + if same_live_carrier: + if overwrite: + raise ValueError("cannot overwrite the attached live cache") + raise ValueError(f"destination {destination!r} already exists; use overwrite=True to replace it") + if os.path.exists(destination) and not overwrite: + raise ValueError(f"destination {destination!r} already exists; use overwrite=True to replace it") + blosc2.blosc2_ext.check_access_mode(destination, "w") + if os.path.exists(destination): + with tempfile.TemporaryDirectory(dir=os.path.dirname(os.path.abspath(destination))) as temp_dir: + staged = os.path.join(temp_dir, os.path.basename(destination)) + carrier.save(staged, contiguous=contiguous, **kwargs) + os.replace(staged, destination) + else: + carrier.save(destination, contiguous=contiguous, **kwargs) + return destination @classmethod def _from_payload(cls, payload, carrier): @@ -1766,7 +1799,7 @@ def _from_payload(cls, payload, carrier): source_kind, urlpath = _parse_source_from_payload(source) expected = (carrier.shape, carrier.dtype, carrier.chunks, carrier.blocks) kwargs = {} if policy is blosc2.CachePolicy.NONE else {"max_cache_bytes": limit} - carrier_arg = carrier if policy is blosc2.CachePolicy.DISK else None + carrier_arg = carrier if policy in {blosc2.CachePolicy.DISK, blosc2.CachePolicy.MEMORY} else None hdf5_index = _hdf5_index_from_carrier(carrier) if source_kind == "hdf5" else None carrier_mode = getattr(carrier.schunk, "mode", "r") if carrier is not None else "r" is_disk_file = carrier is not None and bool(getattr(carrier.schunk, "urlpath", None)) @@ -1798,13 +1831,5 @@ def _from_payload(cls, payload, carrier): obj._cached_vlmeta = read_b2object_user_vlmeta(carrier) return obj - def __enter__(self): - self._check_open() - return self - - def __exit__(self, exc_type, exc_val, exc_tb): - self.close() - return False - def __str__(self): return f"RemoteArray({self._display_identity()!r}, cache_policy={self.cache_policy.name})" diff --git a/src/blosc2/remote_ctable.py b/src/blosc2/remote_ctable.py index cd6762a05..19f0950a4 100644 --- a/src/blosc2/remote_ctable.py +++ b/src/blosc2/remote_ctable.py @@ -13,6 +13,7 @@ from blosc2.ctable import CTable from blosc2.ctable_storage import RemoteTableStorage from blosc2.remote_array import CACHE_POLICY_DEFAULT, RemoteMetadataMapping +from blosc2.remote_object import RemoteObject def _positive_integer(name, value): @@ -41,7 +42,7 @@ def set(self, value): return property(get, set) -class RemoteCTable(CTable): +class RemoteCTable(RemoteObject, CTable): """A read-only CTable whose fixed-width and UTF-8 columns are fetched on demand. Independent column requests overlap by default. ``max_concurrency`` defaults @@ -129,6 +130,9 @@ def _remote_storage(self) -> RemoteTableStorage: storage._check_open() return storage + def _check_open(self) -> None: + self._remote_storage() + def close(self) -> None: storage = getattr(self, "_storage", None) if isinstance(storage, RemoteTableStorage): @@ -206,6 +210,20 @@ def cache_policy(self): def max_cache_bytes(self): return self._remote_storage()._owner.max_cache_bytes + @property + def mutable(self) -> bool: + """The export default mutability for future reference exports.""" + return self._remote_storage()._owner.mutable + + @mutable.setter + def mutable(self, value: bool) -> None: + if not isinstance(value, bool): + raise TypeError("mutable must be a boolean") + storage = self._remote_storage() + with storage._owner.lock: + storage._check_open() + storage._owner.mutable = value + @property def is_cache_mutable(self) -> bool: """Whether the current local cache is writable, not the remote table.""" diff --git a/src/blosc2/remote_object.py b/src/blosc2/remote_object.py new file mode 100644 index 000000000..49f4ba954 --- /dev/null +++ b/src/blosc2/remote_object.py @@ -0,0 +1,31 @@ +####################################################################### +# Copyright (c) 2019-present, Blosc Development Team +# All rights reserved. +# +# SPDX-License-Identifier: BSD-3-Clause +####################################################################### + +"""Common public API for remote Blosc2 objects.""" + + +class RemoteObject: + """Base class for remote arrays, stores, and tables. + + Construct a concrete remote type or use :func:`blosc2.open`; this class is + intended for type checks and the shared remote-object contract. + """ + + def _check_open(self) -> None: + """Raise when this handle cannot be used.""" + + def close(self) -> None: + """Release resources owned by this handle.""" + raise NotImplementedError + + def __enter__(self): + self._check_open() + return self + + def __exit__(self, exc_type, exc_value, traceback): + self.close() + return False diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index dc3573970..037779b78 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -24,6 +24,7 @@ normalize_cache_limit, validate_persistable_url, ) +from blosc2.remote_object import RemoteObject RESERVED_NAMES = {"embed.b2e", "__vlmeta__"} @@ -790,7 +791,7 @@ def _close_resources(self): self.filesystem = None -class RemoteStore: +class RemoteStore(RemoteObject): """Read-only remote B2Z, Zarr or HDF5 hierarchy. Discovery and returned array handles share source resources and traffic. @@ -1230,12 +1231,8 @@ def close(self): with self._owner.lock: self._finalizer() - def __enter__(self): + def _check_open(self): self._resolve("") - return self - - def __exit__(self, exc_type, exc_value, traceback): - self.close() def save( self, diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 1ae3b07f4..45828d8ee 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -255,6 +255,13 @@ def test_remote_ctable_is_cache_mutable(tmp_path, monkeypatch): local = blosc2.CTable(dataclasses.make_dataclass("Sample", [("x", int)]), [(1,)]) url = remote_table_url(tmp_path, local) with blosc2.RemoteCTable(url) as table: + assert isinstance(table, blosc2.RemoteObject) + assert isinstance(table, blosc2.CTable) + assert table.mutable is False + table.mutable = True + assert table.mutable is True + with pytest.raises(TypeError, match="boolean"): + table.mutable = "invalid" assert table.is_cache_mutable is True with monkeypatch.context() as patch: patch.setattr(table._remote_storage()._owner, "is_mutable", False) @@ -263,6 +270,8 @@ def test_remote_ctable_is_cache_mutable(tmp_path, monkeypatch): table.is_cache_mutable = False with pytest.raises(RuntimeError, match="closed"): _ = table.is_cache_mutable + with pytest.raises(RuntimeError, match="closed"): + _ = table.mutable def test_remote_ctable_fixed_width_reads_and_queries(tmp_path): @@ -424,9 +433,23 @@ def test_remote_ctable_deleted_rows_and_disk_cache(tmp_path): assert remote["x"].sum() == 10 assert remote.cache_policy is blosc2.CachePolicy.DISK + materialized = remote.materialize() + assert type(materialized) is blosc2.CTable + np.testing.assert_array_equal(materialized["x"][:], [0, 2, 3, 5]) + + destination = tmp_path / "materialized.b2z" + persisted = remote.materialize(urlpath=destination) + assert type(persisted) is blosc2.CTable + np.testing.assert_array_equal(persisted["x"][:], [0, 2, 3, 5]) + with blosc2.RemoteCTable(url, cache_dir=cache) as reopened: np.testing.assert_array_equal(reopened["x"][:], [0, 2, 3, 5]) + np.testing.assert_array_equal(materialized["x"][:], [0, 2, 3, 5]) + np.testing.assert_array_equal(persisted["x"][:], [0, 2, 3, 5]) + materialized.close() + persisted.close() + @pytest.mark.parametrize("empty", [False, True]) def test_remote_utf8_nested_lifetime(tmp_path, empty): diff --git a/tests/ctable/test_table_persistency.py b/tests/ctable/test_table_persistency.py index 71abf81b5..34c1dc200 100644 --- a/tests/ctable/test_table_persistency.py +++ b/tests/ctable/test_table_persistency.py @@ -275,6 +275,17 @@ def test_copy_to_b2z_uses_urlpath_extension(): assert list(copied["id"][:]) == [10, 20] +def test_materialize_view_returns_independent_table(): + t = CTable(Row, new_data=[(1, 10.0, True), (2, 20.0, False), (3, 30.0, True)]) + view = t.where(t["id"] > 1) + + materialized = view.materialize() + t.close() + + assert type(materialized) is CTable + assert list(materialized["id"][:]) == [2, 3] + + def test_to_b2d_unpacks_persistent_b2z(): src_b2d = table_path("to_b2d_src.b2d") src_b2z = table_path("to_b2d_src.b2z") diff --git a/tests/test_b2z_source.py b/tests/test_b2z_source.py index ab92f5d46..82c12b114 100644 --- a/tests/test_b2z_source.py +++ b/tests/test_b2z_source.py @@ -263,7 +263,13 @@ def test_policies_exports_and_identity(tmp_path): assert none.traffic.nbytes > before memory = blosc2.open(url, dataset="d0/a", lazy=True) memory[:2] - memory.save(tmp_path / "cold.b2nd") + memory.save(tmp_path / "warm.b2nd") + warm = blosc2.open(tmp_path / "warm.b2nd") + assert warm.cache_bytes > 0 + warm.traffic.reset() + np.testing.assert_array_equal(warm[:2], data[:2]) + assert warm.traffic.requests == 0 + memory.save(tmp_path / "cold.b2nd", include_cache=False) cold = blosc2.open(tmp_path / "cold.b2nd") assert cold.cache_bytes == 0 np.testing.assert_array_equal(cold.materialize((slice(0, 2),))[:], data[:2]) diff --git a/tests/test_locking.py b/tests/test_locking.py index 469198c33..ffde52eab 100644 --- a/tests/test_locking.py +++ b/tests/test_locking.py @@ -598,6 +598,8 @@ def test_cross_process_multiwriter_ndarray(tmp_path): import numpy as np import blosc2 +blosc2.set_nthreads(1) + urlpath, writer_id, appends_per_writer, items_per_append = ( sys.argv[1], int(sys.argv[2]), int(sys.argv[3]), int(sys.argv[4]) ) @@ -650,8 +652,8 @@ def test_cross_process_multiwriter_ndarray_append(tmp_path): ) del a - writers = [ - subprocess.Popen( + writers = { + wid: subprocess.Popen( [ sys.executable, "-c", @@ -660,16 +662,21 @@ def test_cross_process_multiwriter_ndarray_append(tmp_path): str(wid), str(appends_per_writer), str(items_per_append), - ] + ], + stderr=subprocess.PIPE, + text=True, ) for wid in range(nwriters) - ] + } deadline = time.monotonic() + 180 - for w in writers: + failures = [] + for wid, writer in writers.items(): remaining = deadline - time.monotonic() assert remaining > 0, "writer processes did not finish in time" - w.wait(timeout=remaining) - assert all(w.returncode == 0 for w in writers), "a writer process failed" + _, stderr = writer.communicate(timeout=remaining) + if writer.returncode != 0: + failures.append((wid, writer.returncode, stderr)) + assert not failures, f"writer processes failed: {failures}" reader = blosc2.open(urlpath, mode="r", locking=True) expected_len = nwriters * appends_per_writer * items_per_append diff --git a/tests/test_python_blosc.py b/tests/test_python_blosc.py index 87c5a098a..55e2b7aa7 100644 --- a/tests/test_python_blosc.py +++ b/tests/test_python_blosc.py @@ -206,18 +206,25 @@ def test_no_leaks(self): def leaks(operation, repeats=3): # Fill reusable codec/Python allocator arenas before taking the RSS - # baseline. A real per-call leak keeps growing in the second batch. + # baseline. A real per-call leak keeps growing in successive batches; + # a single jump can just be malloc retaining a freed output buffer. for _ in range(repeats): operation() gc.collect() - used_mem_before = psutil.Process(os.getpid()).memory_info()[0] - for _ in range(repeats): - operation() - gc.collect() - used_mem_after = psutil.Process(os.getpid()).memory_info()[0] + process = psutil.Process(os.getpid()) + used_mem_before = process.memory_info().rss # We multiply by an additional factor of .01 to account for # storage overhead of Python classes - return (used_mem_after - used_mem_before) >= num_elements * 8.01 + threshold = num_elements * 8.01 + growth = [] + for _ in range(2): + for _ in range(repeats): + operation() + gc.collect() + used_mem_after = process.memory_info().rss + growth.append(used_mem_after - used_mem_before) + used_mem_before = used_mem_after + return growth if all(delta >= threshold for delta in growth) else [] def compress(): blosc2.compress(array, typesize, clevel=1) @@ -227,8 +234,8 @@ def compress(): def decompress(): blosc2.decompress(compressed) - assert not leaks(compress), "compress leaks memory" - assert not leaks(decompress), "decompress leaks memory" + assert not (growth := leaks(compress)), f"compress leaks memory: RSS growth {growth}" + assert not (growth := leaks(decompress)), f"decompress leaks memory: RSS growth {growth}" def test_get_blocksize(self): s = b"0123456789" * 1000 diff --git a/tests/test_remote_array.py b/tests/test_remote_array.py index 3accc4dbe..c3ab38749 100644 --- a/tests/test_remote_array.py +++ b/tests/test_remote_array.py @@ -178,6 +178,9 @@ def test_remote_array_operand_interface(): assert type(proxy).__module__ == "blosc2.remote_array" assert type(proxy).__name__ == "RemoteArray" + assert isinstance(proxy, blosc2.RemoteObject) + assert isinstance(proxy, blosc2.Operand) + assert "RemoteObject" in blosc2.__all__ assert "RemoteArray" in blosc2.__all__ assert proxy.ndim == 1 assert len(proxy) == 100 @@ -500,6 +503,15 @@ def test_save_returns_written_path(tmp_path): destination = tmp_path / "out.b2nd" assert original.save(destination) == str(destination) + alias = tmp_path / "alias.b2nd" + assert original.save(urlpath=alias) == str(alias) + with pytest.raises(TypeError, match="cannot both"): + original.save(destination, urlpath=alias) + + with pytest.raises(ValueError, match="overwrite=True"): + original.save(destination) + assert original.save(destination, overwrite=True) == str(destination) + def test_dict_store_externalizes_disk_remote_array(tmp_path): from blosc2.dict_store import DictStore @@ -1071,15 +1083,31 @@ def test_legacy_cache_url_open_preserves_file(tmp_path): np.testing.assert_array_equal(blosc2.open(path)[:], data) -def test_memory_warm_export_is_cold(): - url, _ = _remote_array("warm-memory.b2nd", nchunks=1, chunk_size=100) +def test_memory_warm_export_restores_retained_cache(tmp_path): + url, data = _remote_array("warm-memory.b2nd", nchunks=2, chunk_size=100) proxy = blosc2.open(url, lazy=True) - proxy[:] - carrier = blosc2.ndarray_from_cframe(proxy.to_cframe()) + np.testing.assert_array_equal(proxy[:100], data[:100]) + + frame = proxy.to_cframe() + carrier = blosc2.ndarray_from_cframe(frame) assert carrier.schunk.vlmeta["b2o"]["cache_policy"] == "memory" - assert not carrier.schunk.vlmeta.get("proxy-fetched") + assert carrier.schunk.vlmeta.get("proxy-fetched") assert proxy.cache_bytes > 0 + restored = blosc2.from_cframe(frame) + restored.traffic.reset() + np.testing.assert_array_equal(restored[:100], data[:100]) + assert restored.traffic.requests == 0 + np.testing.assert_array_equal(restored[100:], data[100:]) + assert restored.traffic.requests > 0 + + path = tmp_path / "warm-memory.b2nd" + proxy.save(path) + reopened = blosc2.open(path) + reopened.traffic.reset() + np.testing.assert_array_equal(reopened[:100], data[:100]) + assert reopened.traffic.requests == 0 + def test_cold_export_cannot_overwrite_live_carrier(tmp_path): url, _ = _remote_array("same-export.b2nd", nchunks=1, chunk_size=100) diff --git a/tests/test_remote_store.py b/tests/test_remote_store.py index ef76fdc29..4c7fde178 100644 --- a/tests/test_remote_store.py +++ b/tests/test_remote_store.py @@ -923,9 +923,13 @@ def test_hdf5_array_attrs_survive_manifest_and_export(tmp_path): def test_mutability_property_and_inheritance(hierarchy): url, _ = hierarchy with blosc2.RemoteStore(url) as store: + assert isinstance(store, blosc2.RemoteObject) assert store.mutable is False - assert store["group"].mutable is False - assert store["group/a"].mutable is False + with store["group"] as group, store["group/a"] as array: + assert isinstance(group, blosc2.RemoteObject) + assert isinstance(array, blosc2.RemoteObject) + assert group.mutable is False + assert array.mutable is False store.mutable = True assert store.mutable is True From 7b9d5e9290fd28bd6cb582693a87c37a6c8467d0 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 18 Sep 2026 12:52:23 +0200 Subject: [PATCH 09/82] Add RemoteCTable reference saving --- doc/guides/remote_arrays.md | 15 ++++- doc/reference/classes.rst | 1 + doc/reference/remotectable.rst | 33 ++++++++++ doc/reference/remotestore.rst | 13 ++-- src/blosc2/ctable.py | 11 ++++ src/blosc2/ctable_storage.py | 15 +---- src/blosc2/remote_ctable.py | 40 ++++++++++++ src/blosc2/remote_object.py | 55 ++++++++++++++++ src/blosc2/remote_store.py | 50 ++++++++++++--- tests/ctable/test_remote_ctable.py | 86 ++++++++++++++++++++++++++ tests/ctable/test_table_persistency.py | 13 ++++ 11 files changed, 300 insertions(+), 32 deletions(-) create mode 100644 doc/reference/remotectable.rst diff --git a/doc/guides/remote_arrays.md b/doc/guides/remote_arrays.md index 2f20eb797..ce17edd92 100644 --- a/doc/guides/remote_arrays.md +++ b/doc/guides/remote_arrays.md @@ -607,8 +607,19 @@ the corresponding RemoteStore and RemoteArray property. It is read-only and does not imply that the remote table can be modified. Closing a parent RemoteStore leaves a returned table usable; refreshing the store invalidates previously returned tables and their columns. Copies and data exports -produce local tables. Remote writes, batch-backed `vlstring`/lists/objects, -dictionary columns and portable table-reference export remain unsupported. +produce local tables. `save()` writes a portable remote reference containing +bootstrap metadata and any retained cache; `materialize()`, `copy()`, `to_b2z()` +and `to_b2d()` produce independent local tables: + +```python +table.save("table-reference.b2z") +table.save("cold-reference.b2z", include_cache=False) +local = table.materialize(urlpath="complete-local.b2z") +table.to_b2d("complete-local.b2d") +``` + +Saving a reference does not fetch missing table data. Remote writes and +batch-backed `vlstring`/lists/objects and dictionary columns remain unsupported. See `examples/ctable/remote_handling.py` for a batched archive writer with a nullable multilingual UTF-8 column, plus sample row and string-slice traffic measurements. diff --git a/doc/reference/classes.rst b/doc/reference/classes.rst index 64913e31b..b4adc0b8b 100644 --- a/doc/reference/classes.rst +++ b/doc/reference/classes.rst @@ -156,6 +156,7 @@ container APIs above. remoteobject remotearray remotestore + remotectable proxysource proxyndsource byterangendsource diff --git a/doc/reference/remotectable.rst b/doc/reference/remotectable.rst new file mode 100644 index 000000000..c459cc803 --- /dev/null +++ b/doc/reference/remotectable.rst @@ -0,0 +1,33 @@ +.. _RemoteCTable: + +RemoteCTable +============ + +``RemoteCTable`` is a read-only :class:`blosc2.CTable` backed by a remote B2Z +archive. Fixed-width, shaped, nullable, and UTF-8 columns are fetched on demand. +Standalone tables can be opened directly; tables inside a hierarchy can be +selected with ``dataset=`` or through :class:`blosc2.RemoteStore`. + +Saving and materializing have different meanings: + +.. code-block:: python + + with blosc2.RemoteCTable(url) as table: + table["temperature"][:100] # warm part of one column + table.save("reference.b2z") + table.save("cold-reference.b2z", include_cache=False) + local = table.materialize(urlpath="complete.b2z") + +``save()`` writes the source descriptor, table metadata, and by default only +payload already retained in the cache. Missing data is still read from the +original source after reopening. ``materialize()`` returns an independent local +table and reads all data needed for it. ``copy()``, ``to_b2z()``, and ``to_b2d()`` +remain local materialization operations inherited from :class:`blosc2.CTable`. + +Table cache bytes, limits, and traffic are scoped to the shared remote owner and +may include sibling leaves. A table selected from a store owns an independent +handle, but its columns and views remain borrowed from that table. Refresh a +nested table through its root store; a standalone table can call ``refresh()``. + +.. autoclass:: blosc2.RemoteCTable + :members: diff --git a/doc/reference/remotestore.rst b/doc/reference/remotestore.rst index d2b0cdac0..1b99d520c 100644 --- a/doc/reference/remotestore.rst +++ b/doc/reference/remotestore.rst @@ -4,8 +4,9 @@ RemoteStore =========== ``RemoteStore`` discovers a read-only B2Z, Zarr or HDF5 hierarchy and returns -:ref:`RemoteArray` leaves. Groups and arrays share one source session: a B2Z -archive, a native HDF5 index, or a Zarr store. Zarr listing remains lazy. +:ref:`RemoteArray` and :ref:`RemoteCTable` leaves. Groups and leaves share one +source session: a B2Z archive, a native HDF5 index, or a Zarr store. Zarr listing +remains lazy. The default ``CachePolicy.MEMORY`` shares a 256 MiB allowance across all leaves. Set ``max_cache_bytes`` to a positive integer to change it. ``CachePolicy.NONE`` @@ -41,10 +42,10 @@ An array root must be opened with ``RemoteArray`` instead. ``keys()`` and ``get_info()`` do not construct leaf readers or payload caches. Discovery can read archive prefixes, attributes and small HDF5 inline values. ``get_info()`` returns a ``RemoteNode`` with a relative path, a kind (``group``, -``ndarray`` or ``unsupported``), known attributes and a diagnostic. Unknown array -attributes are ``None``; open the array to retrieve them. Unsupported nodes stay -discoverable and raise ``NotImplementedError`` when selected. Missing paths raise -``KeyError``. +``ndarray``, ``ctable`` or ``unsupported``), known attributes and a diagnostic. +Unknown array attributes are ``None``; open the array to retrieve them. +Unsupported nodes stay discoverable and raise ``NotImplementedError`` when +selected. Missing paths raise ``KeyError``. Group ``attrs`` mappings are read-only. ``source`` returns the credential-free container descriptor and full group path. ``traffic`` is one shared source diff --git a/src/blosc2/ctable.py b/src/blosc2/ctable.py index 3c2f10c18..46b8e22c2 100644 --- a/src/blosc2/ctable.py +++ b/src/blosc2/ctable.py @@ -7094,6 +7094,12 @@ def _save_to_storage( # noqa: C901 disk_mask[:n_live] = mask[:n_live] if no_deletions else mask[live_pos] storage.save_schema(self._schema_dict_with_computed()) + attrs = self.attrs[:] + if attrs: + vlmeta = blosc2.SChunk() + for key, value in attrs.items(): + vlmeta.vlmeta[key] = value + storage.save_vlmeta(vlmeta) def save(self, urlpath: str, *, overwrite: bool = False) -> None: """Persist this table to disk at *urlpath*. @@ -14536,6 +14542,9 @@ def copy( # noqa: C901 result._n_rows = n_live result._last_pos = None # recomputed lazily on next append + for key, value in self.attrs[:].items(): + result.attrs[key] = value + return result def _empty_copy( @@ -14749,6 +14758,8 @@ def vlmeta(self): """ storage = getattr(self, "_storage", None) if storage is None: + if self.base is not None: + return self.base.vlmeta raise AttributeError("CTable has no storage backend") if not hasattr(storage, "_open_meta"): # In-memory table: create a simple SChunk to hold vlmeta lazily diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index 7c4b3beef..3ded6b608 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -769,20 +769,7 @@ def load_user_attrs(self) -> dict: self._check_open() if hasattr(self, "_user_attrs"): return dict(self._user_attrs) - if self._root_key in self._owner.attrs: - self._user_attrs = dict(self._owner.attrs[self._root_key]) - return dict(self._user_attrs) - member = self._full_key("_vlmeta") + ".b2f" - matches = [info for info in self._owner.archive.members if info.filename == member] - if not matches: - return {} - if len(matches) != 1: - raise ValueError(f"Duplicate Remote CTable metadata member {member!r}") - from blosc2.b2z_source import member_vlmeta - - self._user_attrs = dict(member_vlmeta(self._owner.archive, matches[0])) - self._owner.attrs[self._root_key] = self._user_attrs - self._owner.save_manifest() + self._user_attrs = self._owner.load_ctable_attrs(self._root_key) return dict(self._user_attrs) def table_exists(self) -> bool: diff --git a/src/blosc2/remote_ctable.py b/src/blosc2/remote_ctable.py index 19f0950a4..388059f9b 100644 --- a/src/blosc2/remote_ctable.py +++ b/src/blosc2/remote_ctable.py @@ -9,6 +9,7 @@ from __future__ import annotations import operator +import os # noqa: TC003 from blosc2.ctable import CTable from blosc2.ctable_storage import RemoteTableStorage @@ -187,6 +188,11 @@ def refresh(self) -> None: def vlmeta(self): return RemoteMetadataMapping(self._remote_storage().load_user_attrs()) + @property + def attrs(self): + """Read-only user attributes.""" + return self.vlmeta + @property def source(self): storage = self._remote_storage() @@ -238,3 +244,37 @@ def metadata_bytes(self): storage = self._remote_storage() storage._owner.save_manifest() return storage._owner.metadata_bytes + + def save( + self, + destination: str | os.PathLike | None = None, + *, + urlpath: str | os.PathLike | None = None, + include_cache: bool = True, + mutable: bool | None = None, + overwrite: bool = False, + ) -> str: + """Export this table as a portable remote-reference archive.""" + if destination is None: + if urlpath is None: + raise TypeError("save() missing required destination") + destination = urlpath + elif urlpath is not None: + raise TypeError("destination and urlpath cannot both be specified") + + from blosc2.remote_store import RemoteStore + + storage = self._remote_storage() + owner = storage._owner + relative = storage._root_key[len(owner.root) + 1 :] if owner.root else storage._root_key + selected = object.__new__(RemoteStore) + selected._attach(owner, relative) + try: + return selected.save( + destination, + include_cache=include_cache, + mutable=mutable, + overwrite=overwrite, + ) + finally: + selected.close() diff --git a/src/blosc2/remote_object.py b/src/blosc2/remote_object.py index 49f4ba954..fb57eaa01 100644 --- a/src/blosc2/remote_object.py +++ b/src/blosc2/remote_object.py @@ -18,6 +18,61 @@ class RemoteObject: def _check_open(self) -> None: """Raise when this handle cannot be used.""" + @property + def source(self): + """Credential-free descriptor for the selected remote object.""" + raise NotImplementedError + + @property + def attrs(self): + """Read-only user metadata for the selected remote object.""" + raise NotImplementedError + + @property + def traffic(self): + """Source-read accounting for this object or its shared owner.""" + raise NotImplementedError + + @property + def cache_policy(self): + """Configured payload-retention policy.""" + raise NotImplementedError + + @property + def max_cache_bytes(self): + """Configured compressed-payload allowance.""" + raise NotImplementedError + + @property + def cache_bytes(self): + """Compressed payload currently retained at this object's scope.""" + raise NotImplementedError + + @property + def mutable(self) -> bool: + """Default mutability for future reference exports.""" + raise NotImplementedError + + @mutable.setter + def mutable(self, value: bool) -> None: + raise NotImplementedError + + @property + def is_cache_mutable(self) -> bool: + """Whether the currently attached local cache is writable.""" + raise NotImplementedError + + def save( + self, + destination, + *, + include_cache: bool = True, + mutable: bool | None = None, + overwrite: bool = False, + ) -> str: + """Write a portable remote reference and return its output path.""" + raise NotImplementedError + def close(self) -> None: """Release resources owned by this handle.""" raise NotImplementedError diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 037779b78..92f61b7fc 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -567,6 +567,23 @@ def open_ctable_array(self, table_path, logical_key): self.sources[full] = source return source + def load_ctable_attrs(self, table_path): + """Load one table's user attributes without opening its data arrays.""" + if table_path in self.attrs: + return dict(self.attrs[table_path]) + member = "/".join(part for part in (table_path, "_vlmeta.b2f") if part) + matches = [info for info in self.archive.members if info.filename == member] + if len(matches) > 1: + raise ValueError(f"Duplicate Remote CTable metadata member {member!r}") + if matches: + from blosc2.b2z_source import member_vlmeta + + self.attrs[table_path] = dict(member_vlmeta(self.archive, matches[0])) + else: + self.attrs[table_path] = {} + self.save_manifest() + return dict(self.attrs[table_path]) + def _open_b2z_source(self, full): from blosc2.b2z_source import B2ZNDSource @@ -1353,11 +1370,9 @@ def _validate_save_destination(self, destination, overwrite): def _collect_export_nodes(self, include_cache): group_full = "/".join(p for p in (self._owner.root, self._path.strip("/")) if p) prefix = (group_full + "/") if group_full else "" - if any( - kind == "ctable" and (path == group_full or not prefix or path.startswith(prefix)) - for path, (kind, _) in self._owner.nodes.items() - ): - raise NotImplementedError("RemoteStore artifacts do not yet support RemoteCTable nodes") + for path, (kind, _) in self._owner.nodes.items(): + if kind == "ctable" and (path == group_full or not prefix or path.startswith(prefix)): + self._owner.load_ctable_attrs(path) exported_source = { "urlpath": self._owner.urlpath, "dataset": group_full, @@ -1368,7 +1383,7 @@ def _collect_export_nodes(self, include_cache): exported_source["storage_options"] = fingerprint if prefix: exported_nodes = { - k: (v[0], v[1] if v[0] == "unsupported" else None) + k: (v[0], v[1] if v[0] in {"ctable", "unsupported"} else None) for k, v in self._owner.nodes.items() if k == group_full or k.startswith(prefix) } @@ -1383,7 +1398,8 @@ def _collect_export_nodes(self, include_cache): ) else: exported_nodes = { - k: (v[0], v[1] if v[0] == "unsupported" else None) for k, v in self._owner.nodes.items() + k: (v[0], v[1] if v[0] in {"ctable", "unsupported"} else None) + for k, v in self._owner.nodes.items() } exported_attrs = dict(self._owner.attrs) exported_listed = {k: list(v) for k, v in self._owner.listed.items()} @@ -1479,7 +1495,7 @@ def _load_artifact_manifest(cls, urlpath): return manifest, artifact_offsets @staticmethod - def _validate_artifact_manifest(manifest): + def _validate_artifact_manifest(manifest): # noqa: C901 from blosc2.remote_store_cache import validate_generation validate_generation(manifest.get("generation")) @@ -1497,9 +1513,18 @@ def _validate_artifact_manifest(manifest): if ( not isinstance(entry, (list, tuple)) or len(entry) != 2 - or entry[0] not in {"group", "ndarray", "unsupported"} + or entry[0] not in {"group", "ndarray", "ctable", "unsupported"} ): raise ValueError("Invalid RemoteStore node") + if entry[0] == "ctable": + metadata = entry[1] + if ( + source.get("kind") != "b2z" + or not isinstance(metadata, dict) + or metadata.get("kind") not in {"ctable", b"ctable"} + or not isinstance(metadata.get("schema"), (str, bytes)) + ): + raise ValueError("Invalid RemoteStore CTable node") if root not in nodes: raise ValueError("Missing RemoteStore root") if any(path not in nodes for field in ("attrs", "listed") for path in manifest[field]): @@ -1667,9 +1692,14 @@ def _open_artifact(cls, urlpath, mode="r", **kwargs): urlpath, manifest, artifact_offsets, storage_options, cache_policy, limit, cache_dir ) + req_dataset = kwargs.get("dataset") + if owner.nodes[owner.root][0] == "ctable": + if req_dataset: + owner.close() + raise ValueError("dataset cannot select below a RemoteCTable artifact root") + return blosc2.RemoteCTable._from_owner(owner, owner.root) store = cls.__new__(cls) store._attach(owner, "") - req_dataset = kwargs.get("dataset") if req_dataset: return store[req_dataset] return store diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 45828d8ee..0e1ddeeb0 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -4,6 +4,7 @@ import dataclasses import itertools +import os import zipfile import numpy as np @@ -329,6 +330,91 @@ def test_remote_store_returns_table_with_independent_lifetime(tmp_path): np.testing.assert_array_equal(direct["x"][:], [1, 2]) +def test_remote_ctable_reference_save_roundtrip(tmp_path): + @dataclasses.dataclass + class TextRow: + x: int = blosc2.field(blosc2.int64(null_storage="mask")) + text: str = blosc2.field(blosc2.utf8()) + + local = blosc2.CTable( + TextRow, + [(1, "café"), (None, "東京"), (3, "three")], + create_summary_index=False, + ) + local.attrs["title"] = "remote table" + url = remote_table_url(tmp_path, local, "reference-source") + warm_path = tmp_path / "warm-reference.b2z" + cold_path = tmp_path / "cold-reference.b2z" + mutable_path = tmp_path / "mutable-reference.b2z" + + with blosc2.RemoteCTable(url) as remote: + np.testing.assert_array_equal(remote["x"][:], [1, 0, 3]) + assert remote.attrs["title"] == "remote table" + before = remote.traffic.requests + assert remote.save(warm_path) == os.path.abspath(warm_path) + assert remote.save(urlpath=cold_path, include_cache=False) == os.path.abspath(cold_path) + remote.save(mutable_path, mutable=True) + assert remote.traffic.requests == before + + with blosc2.open(warm_path) as warm: + assert isinstance(warm, blosc2.RemoteCTable) + assert warm.attrs["title"] == "remote table" + warm.traffic.reset() + np.testing.assert_array_equal(warm["x"][:], [1, 0, 3]) + assert warm.traffic.requests == 0 + assert warm["text"][:].tolist() == ["café", "東京", "three"] + assert warm.traffic.requests > 0 + + with blosc2.open(cold_path) as cold: + assert isinstance(cold, blosc2.RemoteCTable) + assert cold.cache_bytes == 0 + cold.traffic.reset() + np.testing.assert_array_equal(cold["x"][:], [1, 0, 3]) + assert cold.traffic.requests > 0 + + with blosc2.open(mutable_path) as mutable: + assert isinstance(mutable, blosc2.RemoteCTable) + assert mutable.is_cache_mutable + mutable.refresh() + assert mutable.attrs["title"] == "remote table" + + with blosc2.RemoteCTable(url) as remote, pytest.raises(FileExistsError): + remote.save(warm_path) + + +def test_nested_remote_ctable_reference_save(tmp_path): + source = tmp_path / "tree.b2z" + table = blosc2.CTable(Row, [(1, [1, 2], "one"), (2, [3, 4], "two")]) + with blosc2.TreeStore(source, mode="w", threshold=0) as root: + root["group/table"] = table + root["group/sibling"] = blosc2.arange(10) + url = f"memory://{tmp_path.name}-reference-tree.b2z" + fsspec.filesystem("memory").pipe(url, source.read_bytes()) + destination = tmp_path / "nested-reference.b2z" + group_destination = tmp_path / "group-reference.b2z" + + with blosc2.RemoteStore(url) as store, store["group/table"] as remote: + np.testing.assert_array_equal(remote["x"][:], [1, 2]) + remote.save(destination) + with store["group"] as group: + group.save(group_destination) + + with blosc2.open(destination) as reopened: + assert isinstance(reopened, blosc2.RemoteCTable) + assert reopened.source["dataset"] == "group/table" + reopened.traffic.reset() + np.testing.assert_array_equal(reopened["x"][:], [1, 2]) + assert reopened.traffic.requests == 0 + + with blosc2.open(group_destination) as group: + assert isinstance(group, blosc2.RemoteStore) + with group["table"] as reopened: + assert isinstance(reopened, blosc2.RemoteCTable) + np.testing.assert_array_equal(reopened["x"][:], [1, 2]) + with group["sibling"] as sibling: + assert isinstance(sibling, blosc2.RemoteArray) + + def test_remote_ctable_unsupported_column_is_lazy(tmp_path): @dataclasses.dataclass class Mixed: diff --git a/tests/ctable/test_table_persistency.py b/tests/ctable/test_table_persistency.py index 34c1dc200..40f70288f 100644 --- a/tests/ctable/test_table_persistency.py +++ b/tests/ctable/test_table_persistency.py @@ -100,6 +100,19 @@ def test_ctable_vlmeta_in_memory(): assert t.vlmeta[:]["active"] is True +def test_ctable_copy_preserves_vlmeta(tmp_path): + t = CTable(Row, [(1, 10.0, True)]) + t.attrs["author"] = "test" + + copied = t.copy() + assert copied.attrs[:] == {"author": "test"} + view_copy = t.where(t["id"] == 1).copy() + assert view_copy.attrs[:] == {"author": "test"} + t.save(tmp_path / "copy.b2z") + with CTable.open(tmp_path / "copy.b2z") as reopened: + assert reopened.attrs[:] == {"author": "test"} + + def test_ctable_vlmeta_persistent(tmp_path): """CTable.vlmeta round-trips through close/reopen.""" path = str(tmp_path / "vlmeta.b2z") From 65916a1e4d4da34fd731ef67302161b3c609d4ae Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 18 Sep 2026 13:17:27 +0200 Subject: [PATCH 10/82] Complete RemoteObject documentation --- RELEASE_NOTES.md | 24 ++ doc/guides/index.rst | 2 + doc/guides/remote_arrays.md | 451 ++--------------------------- doc/guides/remote_objects.md | 385 ++++++++++++++++++++++++ doc/guides/remote_tables.md | 110 +++++++ doc/reference/remotectable.rst | 3 + doc/reference/remoteobject.rst | 3 + doc/reference/remotestore.rst | 3 + tests/ctable/test_remote_ctable.py | 47 +++ 9 files changed, 600 insertions(+), 428 deletions(-) create mode 100644 doc/guides/remote_objects.md create mode 100644 doc/guides/remote_tables.md diff --git a/RELEASE_NOTES.md b/RELEASE_NOTES.md index 2ea4aab65..d058edfe7 100644 --- a/RELEASE_NOTES.md +++ b/RELEASE_NOTES.md @@ -4,6 +4,30 @@ XXX version-specific blurb XXX +### Improvements + +#### Common remote-object API + +- Added the public `RemoteObject` base for `RemoteArray`, `RemoteStore`, + and `RemoteCTable`. It documents their shared source, attributes, traffic, + cache accounting, export mutability, reference saving, and lifetime contract. +- Remote references now preserve valid warm MEMORY cache chunks by default. + Pass `include_cache=False` to produce a cold reference without clearing the + live cache. +- `RemoteCTable.save()` now writes a portable `.b2z` remote reference with + retained cache data. `materialize()`, `copy()`, `to_b2z()`, and + `to_b2d()` remain the independent local-table operations. + +### Compatibility notes + +- `RemoteCTable.save()` previously inherited `CTable.save()` and returned + `None` after materializing local data. It now returns the reference path. + Use `materialize(urlpath=...)` or the table conversion methods when a + complete local table is required. Local `CTable.save()` is unchanged. +- Remote reference destinations are no longer replaced implicitly. Pass + `overwrite=True` when replacement is intended; live cache and source + artifacts remain protected. + ## Changes from 4.13.0 to 4.13.1 A maintenance and performance follow-up to 4.13.0 focused on remote data diff --git a/doc/guides/index.rst b/doc/guides/index.rst index 25d40f8f6..8b4e25516 100644 --- a/doc/guides/index.rst +++ b/doc/guides/index.rst @@ -13,7 +13,9 @@ Topics benchmarks optimization_tips + remote_objects remote_arrays + remote_tables sharing_across_processes pandas_engine diff --git a/doc/guides/remote_arrays.md b/doc/guides/remote_arrays.md index ce17edd92..f86db9a85 100644 --- a/doc/guides/remote_arrays.md +++ b/doc/guides/remote_arrays.md @@ -1,31 +1,10 @@ # Working with Remote Arrays -Blosc2 can open remote arrays and stores (arrays on a hierarchical container) without downloading them first. -Source metadata is read at open time; array data is fetched when a slice needs it and retained according to the cache policy. - -Python-Blosc2 provides two primary entry points for remote data: -- {ref}`RemoteArray`: Access and slice an individual remote array (a standalone `.b2nd` file or a specific dataset in a container). -- {ref}`RemoteStore`: Discover, navigate, and access multi-dataset hierarchies in B2Z, Zarr, or HDF5 containers, sharing a single cache budget across all leaves. - -```python -import blosc2 - -# Discover and read from a remote container (B2Z, Zarr, or HDF5) -with blosc2.RemoteStore("https://datasets.example.org/data.h5") as store: - print(store.keys()) # discover groups and datasets - group = store["experiment"] - array = group["temperature"] # yields a RemoteArray leaf - values = array[:100] # fetches only the requested slice - attrs = array.attrs[:] # fetches user metadata - -# Or open a single remote array directly -a = blosc2.open("s3://bucket/big.b2nd", lazy=True) -values = a[:100] -attrs = a.attrs[:] # fetches user metadata -``` - -The `b2view` terminal browser uses these public types with a 64 MiB allowance by default to let you explore remote containers interactively. -For a script showing hierarchy discovery, leaf previews, and persistent caching, see `examples/remote/store-browse.py`. +`RemoteArray` opens one remote array without downloading it first. Source +metadata is read at open time; payload is fetched when indexing or computation +needs it. See {doc}`remote_objects` for RemoteStore navigation, shared cache and +traffic semantics, credentials, reference mutability, and object lifetimes. See +{doc}`remote_tables` for CTable access. ## Choose a remote route @@ -159,197 +138,19 @@ What differs between the transports is the types of remote objects each can open ## Explore remote hierarchies with RemoteStore -When working with containers that hold multiple groups and datasets—such as `.b2z`, `.zarr`, or `.h5` files—use {ref}`RemoteStore` to discover, navigate, and access the hierarchy: - -```python -import blosc2 - -with blosc2.RemoteStore("https://datasets.example.org/data.h5") as store: - # 1. Discover immediate child groups and datasets (metadata only, no array data downloaded) - print(store.keys()) - - # 2. Inspect a node's kind and attributes - info = store.get_info("experiment") - print(info.kind) # "group", "ndarray", or "unsupported" - - # 3. Read user metadata on groups or arrays - print(store["experiment"].attrs[:]) - - # 4. Access an array leaf and slice it - temp = store["experiment/temperature"] - values = temp[:100] # fetches and caches only the requested slice -``` - -### Hierarchy navigation and inspection - -- **Child enumeration**: `store.keys()` and `for name in store:` list immediate children of the current store or group level without fetching array data. -- **Relative paths**: Lookups can use slash paths or chained indexing interchangeably (`store["experiment/temperature"]` is equivalent to `store["experiment"]["temperature"]`). Both return a {ref}`RemoteArray` leaf. -- **Node inspection with {ref}`RemoteNode`**: Call `store.get_info(name)` to inspect a node without creating leaf readers or allocating cache memory. A `RemoteNode` provides: - - `path`: relative dataset path. - - `kind`: `"group"`, `"ndarray"`, or `"unsupported"`. - - `attrs`: user metadata mapping (or `None` if array attributes require opening the leaf). - - `diagnostic`: explanation for unsupported nodes (e.g. non-array objects or unsupported codecs). -- **Graceful degradation**: Unsupported nodes remain visible during discovery and raise an informative `NotImplementedError` only when selected as arrays, allowing you to browse mixed containers without errors. - -### Persistent disk caching with `cache_dir` - -Specify `cache_dir` when creating a `RemoteStore` to persist discovery metadata and downloaded chunks to local disk: - -```python -with blosc2.RemoteStore( - "https://datasets.example.org/data.h5", - cache_dir="./b2store_cache", - max_cache_bytes=512 * 2**20, # 512 MiB shared disk limit -) as store: - temp = store["experiment/temperature"] - values = temp[:100] -``` - -When reopening the same store later with the same `cache_dir`: -- Discovery metadata (such as B2Z member offsets or native HDF5 indexes) is restored from local disk, avoiding repeated remote scans. `store.metadata_bytes` reports the encoded manifest size. -- Retained leaf chunks are available immediately from disk without network transfers. -- Single-owner locks ensure that concurrent processes do not corrupt the shared cache. - -### Lifetime and clean shutdown - -- Use context managers (`with blosc2.RemoteStore(...) as store:`) for clean lifecycle management. -- Child handles (`RemoteArray` leaves or group views) remain usable even after the parent `store` handle closes. -- Transport sessions, HTTP connections, and disk cache locks are released automatically once the last dependent handle is closed or garbage collected. +RemoteStore discovery and navigation now live in {doc}`remote_objects`. This +heading remains as a pointer for existing links. ## Access HTTP/HTTPS, S3, and cloud storage -Because Python-Blosc2 uses [fsspec](https://filesystem-spec.readthedocs.io/) under the hood, any remote protocol supported by fsspec can be used to open arrays lazily. - -### HTTP and HTTPS - -Publicly accessible arrays on any web server, CDN, or object store URL can be opened directly over HTTP or HTTPS without requiring cloud-specific libraries or credentials: - -```python -import blosc2 - -# Standalone array over HTTPS: -a = blosc2.open("https://datasets.example.org/big.b2nd", lazy=True) - -# Container dataset over HTTPS: -b = blosc2.open( - "https://f001.backblazeb2.com/file/blosc2/hierarchy.b2z::/d0/a3", - lazy=True, -) -``` - -### S3 and cloud object stores - -For arrays stored on Amazon S3 or S3-compatible cloud object stores (Backblaze B2, MinIO, Cloudflare R2, Ceph, Wasabi, etc.), install `s3fs` and open the `s3://` URL: - -```python -# Using default credentials from environment or ~/.aws/credentials -a = blosc2.open("s3://bucket/big.b2nd", lazy=True) -``` - -Other cloud stores work similarly by installing their respective fsspec driver (e.g. `gcsfs` for Google Cloud `gs://` or `adlfs` for Azure `abfs://`). - -### Storage options and authentication - -Pass a `storage_options` dictionary to configure headers, credentials, or custom endpoints. -Options are forwarded directly to the underlying `fsspec` filesystem: - -```python -# For HTTP/HTTPS: custom headers or authentication tokens -a = blosc2.open( - "https://datasets.example.org/private.b2nd", - lazy=True, - storage_options={"headers": {"Authorization": "Bearer "}}, -) - -# For S3: AWS profiles, credentials, or custom endpoints -storage_options = { - "profile": "blosc2", # named profile from ~/.aws/credentials - "endpoint_url": "https://s3.us-west-001.backblazeb2.com", # custom endpoint - # Or explicit keys: - # "key": "AWS_ACCESS_KEY_ID", - # "secret": "AWS_SECRET_ACCESS_KEY", - # Or anonymous public access: - # "anon": True, -} -a = blosc2.open("s3://bucket/big.b2nd", lazy=True, storage_options=storage_options) -``` - -### Remote performance: latency, caching, and concurrency - -Remote requests over HTTP or object stores typically incur 20–100 ms of latency per range request. -Python-Blosc2 addresses this in two ways: - -1. **Caching**: Chunks and blocks fetched for a slice are kept in the local cache (in RAM by default, or persisted to disk with `cache_dir=` or `cache_path=`). - Re-fetching previously read regions requires zero network round trips and zero bytes transferred. -2. **Concurrent fetches**: Independent range requests for required chunks and blocks are issued concurrently in a thread pool (configured via `max_concurrency=`, default 8). - -The runnable script `examples/remote/s3-access.py` demonstrates opening `.b2nd`, `.b2z`, `.zarr`, and `.h5` datasets over remote URLs (both S3 and HTTPS), timing metadata discovery vs. slice fetching, measuring network traffic with {ref}`Traffic`, and showing the impact of chunk caching. +Shared fsspec, cloud authentication, and concurrency guidance is in +{doc}`remote_objects`. Array-specific supported formats remain in +[Choose a remote route](#choose-a-remote-route). ## Cache policies and memory management -Every lazy open uses a cache policy. -By default, fetched data is cached in memory with a bound on retained compressed payload. - -### In-memory caching (`CachePolicy.MEMORY` — Default) - -When opened without disk options, `blosc2.open(..., lazy=True)` retains fetched chunks in RAM as a {ref}`RemoteArray` with {attr}`CachePolicy.MEMORY `: - -```python -a = blosc2.open("s3://bucket/big.b2nd", lazy=True) -a[10:12, 500:600] # fetched and cached in RAM -a[10:12, 500:600] # served from memory cache (no network traffic) -``` - -In-memory caches use `max_cache_bytes` (defaults to 256 MiB) with automatic LRU eviction after operations, including failed fetches. -This is not a peak RAM limit: metadata, in-flight transfers, decompression buffers, and results are excluded. -Large operations can exceed it substantially. - -```python -# Custom in-memory limit (e.g. 512 MiB): -a = blosc2.open("s3://bucket/big.b2nd", lazy=True, max_cache_bytes=512 * 2**20) -``` - -### Persistent disk caching (`CachePolicy.DISK`) - -Set `cache_dir` or `cache_path` to persist fetched data across sessions ({attr}`CachePolicy.DISK `): - -```python -url = "s3://bucket/big.b2nd" - -# Blosc2 manages a cache file inside a directory: -a = blosc2.open(url, lazy=True, cache_dir="./b2cache") -a[100:110, :50] # fetched and stored under ./b2cache - -# A later process can reuse the same cache: -a = blosc2.open(url, lazy=True, cache_dir="./b2cache") -a[100:110, :50] # served from local disk (no network traffic) -``` - -- For an individual {ref}`RemoteArray`, pass `cache_dir` (Blosc2 creates the cache carrier inside that directory) or `cache_path` (to specify an exact carrier filename, such as `big-cache.b2nd`). -- For a {ref}`RemoteStore`, pass `cache_dir` to store discovered hierarchy metadata and all leaf caches together under that directory. -- In both cases, compressed chunks are retained up to `max_cache_bytes` (defaults to 256 MiB; pass `max_cache_bytes=None` for an unbounded disk cache that never evicts). -- If you want the on-disk carrier to be visible and predictable, set `cache_path="big-cache.b2nd"` explicitly; otherwise Blosc2 may create its own cache filename under the configured `cache_dir`. - -Authenticated Caterva2 caches must be private to one user. -Reopen them under an equivalent authenticated {func}`blosc2.c2context`; do not share a cache directory between users. - -### Stateless streaming (`CachePolicy.NONE`) - -To stream data without retaining any chunks after each operation, specify {attr}`CachePolicy.NONE `: - -```python -stream = blosc2.open( - "s3://bucket/big.b2nd", - lazy=True, - cache_policy=blosc2.CachePolicy.NONE, -) -``` - -Each read pulls only the bytes required for the slice and retains no cache payload. - -> [!NOTE] -> `max_cache_bytes` is applied after each operation completes. -> It bounds the retained compressed cache payload; it does not limit the temporary working set or the decompressed NumPy array requested by the caller. +The MEMORY, DISK, and NONE policies are described in {doc}`remote_objects`. +Array reads, prefetching, and materialization below all use that shared policy. ## Only what a slice touches @@ -399,31 +200,10 @@ Separate handles or processes sharing a disk carrier require external locking. ## Measure network traffic -{ref}`RemoteArray`, {ref}`RemoteStore`, {ref}`C2Array`, and {ref}`Proxy` objects expose cumulative request and byte counts through {ref}`Traffic`. -The count starts when the remote source is opened, so it includes metadata as well as array data: - -```python -a = blosc2.open("s3://bucket/big.b2nd", lazy=True) - -a.traffic.reset() -corner = a[0, :100, :100] -print(a.traffic) # requests and bytes fetched - -a.traffic.reset() -corner = a[0, :100, :100] -print(a.traffic) # Traffic(requests=0, nbytes=0) -> cache hit! -``` - -Use `reset()` or subtract two readings to measure one operation. -`traffic` is `None` for a local source because no network transport exists. -For a `RemoteStore`, `store.traffic` reports cumulative traffic across discovery and all leaf accesses in the session. - -`examples/remote/c2array-traffic.py` compares block, chunk, and cached reads against a live Caterva2 dataset. +Shared traffic accounting and examples are in {doc}`remote_objects`. ## Persist and reopen remote references -Python-Blosc2 allows you to save remote references and their cached data to disk as portable files, and reopen them later without needing the original remote URL. - ### Persist a remote array reference (.b2nd) Use {ref}`RemoteArray` directly when a `.b2nd` file should carry a portable remote descriptor and, optionally, its own bounded persistent cache: @@ -449,80 +229,11 @@ The saved object contains source and geometry metadata but no credentials. A `.b2nd` carrier is a local cache/reference file for a remote array, not a second copy of the remote dataset itself: it stores the source locator plus any warm compressed chunks that have already been fetched. This is why a file such as `big-cache.b2nd` can be reopened later and continue serving cached reads without re-fetching the remote source. To create a visible file with a predictable name, set `cache_path="big-cache.b2nd"` when opening the remote array or call `save("big-cache.b2nd")` on the live handle. With `CachePolicy.NONE`, repeated reads contact the source and do not mutate the carrier. With `CachePolicy.DISK`, the carrier file itself is the cache and retains compressed chunks up to its payload limit. -Disk arrays preserve warm chunks by default; memory arrays export cold carriers. +DISK and MEMORY arrays preserve valid warm chunks by default. Pass `include_cache=False` to export a cold copy without mutating the warm carrier. -### Export a remote store snapshot (.b2z) - -To export an entire remote hierarchy—including discovered groups, array geometry, source locators, and optional cached chunks—call `save()` on a {ref}`RemoteStore`: - -```python -with blosc2.RemoteStore("https://datasets.example.org/data.h5") as store: - temp = store["experiment/temperature"] - temp[:100] # warms the cache for this slice - - # Save a portable .b2z reference archive containing discovery and warm chunks - store.save("snapshot.b2z") - - # Or export a cold reference containing only metadata and locators (no chunks) - store.save("cold_ref.b2z", include_cache=False) - - # Or export only a specific subtree - store["experiment"].save("experiment_sub.b2z") -``` - -- **Portable reference**: The `.b2z` archive contains the discovered hierarchy, attributes, and source locators (such as the native HDF5 index or B2Z member offsets), but no secrets or credentials. -- **`include_cache=True` (default)**: Bundles warm cached chunks along with metadata so reading previously fetched slices requires zero network traffic. -- **`include_cache=False`**: Omits cached payload chunks, producing a minimal reference archive for remote streaming. -- **Subtree export**: Calling `save()` on a group view exports that subtree with relative child keys and the appropriate source root. - -### Reopen reference files with `blosc2.open()` - -Both `.b2nd` array carriers and `.b2z` store snapshots can be reopened directly with `blosc2.open()`: - -```python -# 1. Reopen an exported RemoteStore hierarchy: -with blosc2.open("snapshot.b2z") as restored: - print(restored.keys()) - temp = restored["experiment/temperature"] - values = temp[:100] # served from archive if cached; fetched remotely if missing - -# 2. Reopen a standalone RemoteArray carrier: -arr = blosc2.open("big-cache.b2nd", mode="a") -values = arr[:100] -``` - -Opening an exported `.b2z` archive automatically recognizes the remote store marker and constructs a {ref}`RemoteStore`. All leaves opened from it share one cache coordinator and budget. -Opening an on-disk carrier with `mode="a"` returns a {ref}`RemoteArray` and lets newly fetched regions extend the cache; opening with `mode="r"` keeps the cache file unchanged. -Legacy proxy caches created by older Blosc2 versions are also detected and reopened as a {ref}`Proxy`. - -Independent reopening works for fsspec URLs, Caterva2 datasets, and persistent local Blosc2 sources. -The required runtime environment must still be available: fsspec backends and their configuration must be installed, local source paths must remain valid, and authenticated Caterva2 caches must be reopened inside an equivalent {func}`blosc2.c2context`. -Caterva2 credentials are not stored in the cache file. - -An arbitrary custom {ref}`ProxyNDSource` cannot be reconstructed because its Python class and runtime state are not serialized. -In that case, recreate the source explicitly and attach the existing cache with `blosc2.Proxy(source, urlpath="big-cache.b2nd", mode="a")`. - -### Cache mutability: immutable vs. mutable snapshots - -When saving an export, you can configure whether the resulting snapshot operates in **immutable** or **mutable** mode via the `mutable` argument (or the `.mutable` property on `RemoteStore` / `RemoteArray`): - -```python -# Default is mutable=False (immutable snapshot) -store.save("read_only.b2z", mutable=False) - -# Or export a mutable snapshot -store.save("writable.b2z", mutable=True) -``` - -| Mode | Behavior when reopened | Cache misses | Modifying operations | -| --- | --- | --- | --- | -| **Immutable** (`mutable=False`, default) | Reads directly in-place from `.b2z` without disk writes. Safe on read-only media (`chmod 0o444`). | Fetched transiently into RAM to satisfy the read; never written to disk or the archive. | `fetch()`, `afetch()`, `trim_cache()`, and `refresh()` are disallowed. | -| **Mutable** (`mutable=True`) | Staged into an independent writable runtime cache directory. Original `.b2z` stays untouched. | Fetched and cached to disk under standard LRU eviction rules. | Fully supported. Can be opened with a smaller budget, trimming excess chunks. | - -```{tip} -Use **immutable snapshots** (`mutable=False`) for sharing reproducible, read-only reference archives or distributing datasets that should never modify local storage. Use **mutable snapshots** (`mutable=True`) when users should be able to expand the local cache with newly fetched regions over time. -``` +Store and table `.b2z` references, cache mutability, and the distinction between +`save()` and `materialize()` are documented in {doc}`remote_objects`. ## Retrieve scattered points @@ -538,91 +249,8 @@ Prefer direct `C2Array` indexing for sparse, one-off point retrieval; prefer a { ## Remote tables -`RemoteCTable` opens a read-only CTable in an immutable remote `.b2z` archive. -Fixed-width and `blosc2.utf8()` columns are fetched on demand, including their -null masks. A table inside a hierarchy can also be opened through `RemoteStore`. - -`blosc2.open()` dispatches local table archives to `CTable` and remote table -archives to `RemoteCTable`. Remote `.b2z` groups return `RemoteStore` by default; -array leaves retain their `RemoteArray` behavior. Use `dataset="group/table"` -or a `::group/table` URL suffix to select a nested table. For a complete local -download instead, pass `lazy=False, cache_dir="download-cache"`. - -```python -with blosc2.open("https://example.org/readings.b2z") as table: - notes = table["note"][:5] - selected = table.where(table["note"] == "café") - ids = selected["id"][:] -``` - -UTF-8 strings use two compressed arrays: row offsets and encoded bytes. A slice -first reads its offsets, then its byte span. Both reads use the existing range -transport, fetching compressed blocks when worthwhile or whole compressed chunks -otherwise, and decompressing locally. Archive members are ZIP_STORED, as produced -by the Blosc2 writers; the arrays inside remain Blosc2-compressed. - -The backing arrays share the table's cache budget and traffic counters. MEMORY, -DISK (with `cache_dir`) and NONE policies are supported. Size reporting uses source -metadata without scanning strings. Small-member metadata prefetch may also fetch -some payload. Repeated reads can reuse cached blocks; filtering scans the required -columns because persisted indexes are not used remotely. - -Multi-column metadata inspection and row materialization overlap independent -requests by default, up to eight at once. This includes row iteration, display, -and batched Arrow/pandas export; single-column access remains lazy. UTF-8 byte -requests wait for their offsets. Cache publication and decoding stay serialized. - -```python -with blosc2.open(url, max_concurrency=1) as table: # serial control - rows = list(table[:10]) - -with blosc2.RemoteCTable( - url, - max_concurrency=8, - metadata_buffer_bytes=8 << 20, - row_buffer_bytes=64 << 20, -) as table: - table.row_buffer_bytes = 256 << 20 # optional explicit override - rows = list(table[:10]) -``` - -The fixed defaults are 8 MiB of temporary metadata and 64 MiB of temporary row -data, allocated on demand. They do not depend on CPU count or available RAM. -Wider reads use bounded batches; a single oversized required unit runs alone. -These are soft transport budgets, not total RAM limits: decoded output, native -scratch, HTTP overhead and retained caches are additional. DISK caching does not -remove the need to bound temporary reads or consume large outputs in batches. -The buffer keywords belong to RemoteCTable, not `blosc2.open()`; settings may -also be changed on a returned table, including one obtained from RemoteStore. - -The existing 1 MiB compressed-chunk threshold is a block-selection heuristic, -not a maximum response size. Large selections and unsupported partial-block -layouts can still fetch whole chunks. Cross-process shared-cache handles and -read-only artifacts retain their existing guarded row-read paths; standalone -RemoteArray concurrency and RemoteStore discovery behavior are unchanged. - -Columns and views are borrowed from the root table and require it to remain open. -`table.is_cache_mutable` reports whether the local cache is writable, matching -the corresponding RemoteStore and RemoteArray property. It is read-only and does -not imply that the remote table can be modified. -Closing a parent RemoteStore leaves a returned table usable; refreshing the store -invalidates previously returned tables and their columns. Copies and data exports -produce local tables. `save()` writes a portable remote reference containing -bootstrap metadata and any retained cache; `materialize()`, `copy()`, `to_b2z()` -and `to_b2d()` produce independent local tables: - -```python -table.save("table-reference.b2z") -table.save("cold-reference.b2z", include_cache=False) -local = table.materialize(urlpath="complete-local.b2z") -table.to_b2d("complete-local.b2d") -``` - -Saving a reference does not fetch missing table data. Remote writes and -batch-backed `vlstring`/lists/objects and dictionary columns remain unsupported. - -See `examples/ctable/remote_handling.py` for a batched archive writer with a nullable -multilingual UTF-8 column, plus sample row and string-slice traffic measurements. +Remote CTable access, filtering, buffering, saving, and materialization now live +in {doc}`remote_tables`. This heading remains as a pointer for existing links. ## Handle remote changes @@ -644,43 +272,8 @@ Use a fresh cache when such a source may have changed without changing its geome For a replaceable `.b2nd` or Caterva2 source, pass `assume_immutable=False` to check for updates and invalidate stale cached chunks before each operation. -### Refreshing a RemoteStore - -Remote containers (B2Z, Zarr, and HDF5) are assumed immutable by default. -For B2Z tables and stores, reopening a populated disk cache trusts its saved -archive identity and metadata: no HEAD/identity request is made. Uncached data -still requires remote reads. Older caches may perform one identity lookup to -upgrade their metadata. Do not replace the remote archive while using its cache. -Use `store.refresh()` after a replacement, or `table.refresh()` for a standalone -`RemoteCTable`. Neither operation writes to the remote source. - -```python -with blosc2.RemoteCTable(url, cache_dir="table-cache") as table: - table.refresh() # rediscover the remote table and replace its cache generation - print(table[:5]) -``` - -Table refresh reloads the schema, row information and column backends while -preserving cache policy, cache limits and parallel-read settings. Previously -obtained columns, raw arrays and views become unusable; retrieve them again from -the refreshed table. Discovery or table initialization failure leaves the old -table usable. A table obtained from a `RemoteStore` must be refreshed through the -root store, then retrieved again; it cannot independently refresh shared discovery. - -If a remote container is updated on the server—such as adding new datasets or appending data—call `store.refresh()` to update discovery: - -```python -with blosc2.RemoteStore("https://datasets.example.org/data.h5") as store: - # Refresh remote metadata atomically - store.refresh() - - # Re-access datasets from the refreshed store - array = store["experiment/temperature"] -``` - -- **Atomic update**: `store.refresh()` fetches fresh discovery from the remote source before updating the active generation. If discovery fails, the existing store state remains unchanged. -- **Stale handle safety**: Any child handles (`RemoteArray` leaves or group views) opened *before* `refresh()` become stale. Accessing them raises a `RuntimeError`, prompting you to look them up again from the refreshed store. -- **Immutability rule**: Calling `refresh()` on an immutable reference snapshot (`mutable=False`) is disallowed and raises an error. +RemoteStore and RemoteCTable refresh behavior is documented in +{doc}`remote_objects` and {doc}`remote_tables`. ## Fill a Caterva2 array concurrently @@ -779,4 +372,6 @@ For ordinary remote access, use `blosc2.open("https://...", lazy=True)` or `blos - `examples/remote/proxy-carray.py` — creating a persistent local disk proxy of a remote Caterva2 array. - `examples/remote/rw-fsspec.py` — fsspec reading and writing examples. - {doc}`b2view ` — interactive terminal browser for local and remote containers. -- {ref}`RemoteArray`, {ref}`RemoteStore`, {ref}`RemoteNode`, {ref}`C2Array`, {ref}`B2ZNDSource`, {ref}`ZarrNDSource`, {ref}`HDF5NDSource`, {ref}`FsspecNDSource`, {ref}`ByteRangeNDSource`, {ref}`Proxy`, and {ref}`Traffic` — API reference pages. +- {doc}`remote_objects` — shared remote caching, traffic, references, and hierarchy navigation. +- {doc}`remote_tables` — remote CTable access. +- {ref}`RemoteArray`, {ref}`C2Array`, {ref}`B2ZNDSource`, {ref}`ZarrNDSource`, {ref}`HDF5NDSource`, {ref}`FsspecNDSource`, {ref}`ByteRangeNDSource`, and {ref}`Proxy` — API reference pages. diff --git a/doc/guides/remote_objects.md b/doc/guides/remote_objects.md new file mode 100644 index 000000000..a1540f095 --- /dev/null +++ b/doc/guides/remote_objects.md @@ -0,0 +1,385 @@ +# Working with Remote Data + +Remote data is represented by three concrete types. {ref}`RemoteArray` reads one +array, {ref}`RemoteCTable` reads one table, and {ref}`RemoteStore` discovers a +hierarchy and returns array or table leaves. All three inherit +{ref}`RemoteObject`, which provides the shared source, metadata, traffic, cache, +reference-saving, and lifetime contract. Construct a concrete type or use +{func}`blosc2.open`; `RemoteObject` is useful for type checks, not as a factory. + +```python +import blosc2 + +with blosc2.open("https://datasets.example.org/data.b2z") as remote: + assert isinstance(remote, blosc2.RemoteObject) +``` + +Use {doc}`remote_arrays` for slicing, expressions, prefetching, and array +transports. Use {doc}`remote_tables` for table columns, filtering, string data, +and table materialization. + +## Choose an object + +| Source | Result | +| --- | --- | +| Standalone `.b2nd`, or one B2Z/Zarr/HDF5 array dataset | `RemoteArray` | +| One standalone or nested B2Z CTable | `RemoteCTable` | +| A B2Z, Zarr, or HDF5 hierarchy | `RemoteStore` | + +Remote objects read immutable source data. `mutable` controls whether a future +reference export may grow its local cache; it never permits remote writes. +Runtime credentials and live filesystem objects are not serialized. + +`RemoteArray` remains an array expression operand because a remote reference is +not mutable NDArray storage. `RemoteCTable` remains a CTable so its schema, +columns, queries, views, and local copy operations work normally. + +## Explore remote hierarchies with RemoteStore + +When working with containers that hold multiple groups and datasets—such as `.b2z`, `.zarr`, or `.h5` files—use {ref}`RemoteStore` to discover, navigate, and access the hierarchy: + +```python +import blosc2 + +with blosc2.RemoteStore("https://datasets.example.org/data.h5") as store: + # 1. Discover immediate child groups and datasets (metadata only, no array data downloaded) + print(store.keys()) + + # 2. Inspect a node's kind and attributes + info = store.get_info("experiment") + print(info.kind) # "group", "ndarray", "ctable", or "unsupported" + + # 3. Read user metadata on groups or arrays + print(store["experiment"].attrs[:]) + + # 4. Access an array leaf and slice it + temp = store["experiment/temperature"] + values = temp[:100] # fetches and caches only the requested slice +``` + +### Hierarchy navigation and inspection + +- **Child enumeration**: `store.keys()` and `for name in store:` list immediate children of the current store or group level without fetching array data. +- **Relative paths**: Lookups can use slash paths or chained indexing interchangeably (`store["experiment/temperature"]` is equivalent to `store["experiment"]["temperature"]`). Leaves return a {ref}`RemoteArray` or {ref}`RemoteCTable` according to their kind. +- **Node inspection with `RemoteNode`**: Call `store.get_info(name)` to inspect a node without creating leaf readers or allocating cache memory. A `RemoteNode` provides: + - `path`: relative dataset path. + - `kind`: `"group"`, `"ndarray"`, `"ctable"`, or `"unsupported"`. + - `attrs`: user metadata mapping (or `None` if array attributes require opening the leaf). + - `diagnostic`: explanation for unsupported nodes (e.g. non-array objects or unsupported codecs). +- **Graceful degradation**: Unsupported nodes remain visible during discovery and raise an informative `NotImplementedError` only when selected as arrays, allowing you to browse mixed containers without errors. + +### Persistent disk caching with `cache_dir` + +Specify `cache_dir` when creating a `RemoteStore` to persist discovery metadata and downloaded chunks to local disk: + +```python +with blosc2.RemoteStore( + "https://datasets.example.org/data.h5", + cache_dir="./b2store_cache", + max_cache_bytes=512 * 2**20, # 512 MiB shared disk limit +) as store: + temp = store["experiment/temperature"] + values = temp[:100] +``` + +When reopening the same store later with the same `cache_dir`: +- Discovery metadata (such as B2Z member offsets or native HDF5 indexes) is restored from local disk, avoiding repeated remote scans. `store.metadata_bytes` reports the encoded manifest size. +- Retained leaf chunks are available immediately from disk without network transfers. +- Single-owner locks ensure that concurrent processes do not corrupt the shared cache. + +### Lifetime and clean shutdown + +- Use context managers (`with blosc2.RemoteStore(...) as store:`) for clean lifecycle management. +- Child handles (`RemoteArray` leaves or group views) remain usable even after the parent `store` handle closes. +- Transport sessions, HTTP connections, and disk cache locks are released automatically once the last dependent handle is closed or garbage collected. + +## Access HTTP/HTTPS, S3, and cloud storage + +For standalone B2ND URLs and B2Z, Zarr, or HDF5 containers, Python-Blosc2 uses +[fsspec](https://filesystem-spec.readthedocs.io/) and supports its remote protocols. + +### HTTP and HTTPS + +Publicly accessible arrays on any web server, CDN, or object store URL can be opened directly over HTTP or HTTPS without requiring cloud-specific libraries or credentials: + +```python +import blosc2 + +# Standalone array over HTTPS: +a = blosc2.open("https://datasets.example.org/big.b2nd", lazy=True) + +# Container dataset over HTTPS: +b = blosc2.open( + "https://f001.backblazeb2.com/file/blosc2/hierarchy.b2z::/d0/a3", + lazy=True, +) +``` + +### S3 and cloud object stores + +For arrays stored on Amazon S3 or S3-compatible cloud object stores (Backblaze B2, MinIO, Cloudflare R2, Ceph, Wasabi, etc.), install `s3fs` and open the `s3://` URL: + +```python +# Using default credentials from environment or ~/.aws/credentials +a = blosc2.open("s3://bucket/big.b2nd", lazy=True) +``` + +Other cloud stores work similarly by installing their respective fsspec driver (e.g. `gcsfs` for Google Cloud `gs://` or `adlfs` for Azure `abfs://`). + +### Storage options and authentication + +Pass a `storage_options` dictionary to configure headers, credentials, or custom endpoints. +Options are forwarded directly to the underlying `fsspec` filesystem: + +```python +# For HTTP/HTTPS: custom headers or authentication tokens +a = blosc2.open( + "https://datasets.example.org/private.b2nd", + lazy=True, + storage_options={"headers": {"Authorization": "Bearer "}}, +) + +# For S3: AWS profiles, credentials, or custom endpoints +storage_options = { + "profile": "blosc2", # named profile from ~/.aws/credentials + "endpoint_url": "https://s3.us-west-001.backblazeb2.com", # custom endpoint + # Or explicit keys: + # "key": "AWS_ACCESS_KEY_ID", + # "secret": "AWS_SECRET_ACCESS_KEY", + # Or anonymous public access: + # "anon": True, +} +a = blosc2.open("s3://bucket/big.b2nd", lazy=True, storage_options=storage_options) +``` + +### Remote performance: latency, caching, and concurrency + +Remote requests over HTTP or object stores typically incur 20–100 ms of latency per range request. +Python-Blosc2 addresses this in two ways: + +1. **Caching**: Chunks and blocks fetched for a slice are kept in the local cache (in RAM by default, or persisted to disk with `cache_dir=` or `cache_path=`). + Re-fetching previously read regions requires zero network round trips and zero bytes transferred. +2. **Concurrent fetches**: Independent range requests for required chunks and blocks are issued concurrently in a thread pool (configured via `max_concurrency=`, default 8). + +The runnable script `examples/remote/s3-access.py` demonstrates opening `.b2nd`, `.b2z`, `.zarr`, and `.h5` datasets over remote URLs (both S3 and HTTPS), timing metadata discovery vs. slice fetching, measuring network traffic with {ref}`Traffic`, and showing the impact of chunk caching. + +## Cache policies and memory management + +Every lazy open uses a cache policy. +By default, fetched data is cached in memory with a bound on retained compressed payload. + +### In-memory caching (`CachePolicy.MEMORY` — Default) + +When opened without disk options, `blosc2.open(..., lazy=True)` retains fetched chunks in RAM as a {ref}`RemoteArray` with {attr}`CachePolicy.MEMORY `: + +```python +a = blosc2.open("s3://bucket/big.b2nd", lazy=True) +a[10:12, 500:600] # fetched and cached in RAM +a[10:12, 500:600] # served from memory cache (no network traffic) +``` + +In-memory caches use `max_cache_bytes` (defaults to 256 MiB) with automatic LRU eviction after operations, including failed fetches. +This is not a peak RAM limit: metadata, in-flight transfers, decompression buffers, and results are excluded. +Large operations can exceed it substantially. + +```python +# Custom in-memory limit (e.g. 512 MiB): +a = blosc2.open("s3://bucket/big.b2nd", lazy=True, max_cache_bytes=512 * 2**20) +``` + +### Persistent disk caching (`CachePolicy.DISK`) + +Set `cache_dir` or `cache_path` to persist fetched data across sessions ({attr}`CachePolicy.DISK `): + +```python +url = "s3://bucket/big.b2nd" + +# Blosc2 manages a cache file inside a directory: +a = blosc2.open(url, lazy=True, cache_dir="./b2cache") +a[100:110, :50] # fetched and stored under ./b2cache + +# A later process can reuse the same cache: +a = blosc2.open(url, lazy=True, cache_dir="./b2cache") +a[100:110, :50] # served from local disk (no network traffic) +``` + +- For an individual {ref}`RemoteArray`, pass `cache_dir` (Blosc2 creates the cache carrier inside that directory) or `cache_path` (to specify an exact carrier filename, such as `big-cache.b2nd`). +- For a {ref}`RemoteStore`, pass `cache_dir` to store discovered hierarchy metadata and all leaf caches together under that directory. +- In both cases, compressed chunks are retained up to `max_cache_bytes` (defaults to 256 MiB; pass `max_cache_bytes=None` for an unbounded disk cache that never evicts). +- If you want the on-disk carrier to be visible and predictable, set `cache_path="big-cache.b2nd"` explicitly; otherwise Blosc2 may create its own cache filename under the configured `cache_dir`. + +Authenticated Caterva2 caches must be private to one user. +Reopen them under an equivalent authenticated {func}`blosc2.c2context`; do not share a cache directory between users. + +### Stateless streaming (`CachePolicy.NONE`) + +To stream data without retaining any chunks after each operation, specify {attr}`CachePolicy.NONE `: + +```python +stream = blosc2.open( + "s3://bucket/big.b2nd", + lazy=True, + cache_policy=blosc2.CachePolicy.NONE, +) +``` + +Each read pulls only the bytes required for the slice and retains no cache payload. + +> [!NOTE] +> `max_cache_bytes` is applied after each operation completes. +> It bounds the retained compressed cache payload; it does not limit the temporary working set or the decompressed NumPy array requested by the caller. + +## Measure network traffic + +{ref}`RemoteArray`, {ref}`RemoteStore`, {ref}`C2Array`, and {ref}`Proxy` objects expose cumulative request and byte counts through {ref}`Traffic`. +The count starts when the remote source is opened, so it includes metadata as well as array data: + +```python +a = blosc2.open("s3://bucket/big.b2nd", lazy=True) + +a.traffic.reset() +corner = a[0, :100, :100] +print(a.traffic) # requests and bytes fetched + +a.traffic.reset() +corner = a[0, :100, :100] +print(a.traffic) # Traffic(requests=0, nbytes=0) -> cache hit! +``` + +Use `reset()` or subtract two readings to measure one operation. +`traffic` is `None` for a local source because no network transport exists. +For a `RemoteStore`, `store.traffic` reports cumulative traffic across discovery and all leaf accesses in the session. + +`examples/remote/c2array-traffic.py` compares block, chunk, and cached reads against a live Caterva2 dataset. + +## Save references and materialize data + +`save()` writes a remote reference plus, by default, payload already retained in +the selected cache. It does not fetch missing payload. `include_cache=False` +writes a cold reference containing only the source and bootstrap metadata. + +Arrays and tables also provide `materialize()`, which reads everything required +for an independent local object. Stores have no recursive materialization API; +navigate to an array or table leaf first. + +```python +# Table example: .b2z references and local materialization +table.save("reference.b2z") +table.save("cold.b2z", include_cache=False) +local = table.materialize(urlpath="local.b2z") +table.to_b2d("local.b2d") + +# Array example: .b2nd reference and selected materialization +array.save("reference.b2nd") +subset = array.materialize(item=slice(0, 100)) +``` + +An array reports its own retained payload. Stores and tables report their shared +owner's cache and traffic, which may include siblings. Saving a nested selection +still exports only that selected subtree. + +### Export a remote store snapshot (.b2z) + +To export an entire remote hierarchy—including discovered groups, array geometry, source locators, and optional cached chunks—call `save()` on a {ref}`RemoteStore`: + +```python +with blosc2.RemoteStore("https://datasets.example.org/data.h5") as store: + temp = store["experiment/temperature"] + temp[:100] # warms the cache for this slice + + # Save a portable .b2z reference archive containing discovery and warm chunks + store.save("snapshot.b2z") + + # Or export a cold reference containing only metadata and locators (no chunks) + store.save("cold_ref.b2z", include_cache=False) + + # Or export only a specific subtree + store["experiment"].save("experiment_sub.b2z") +``` + +- **Portable reference**: The `.b2z` archive contains the discovered hierarchy, attributes, and source locators (such as the native HDF5 index or B2Z member offsets), but no secrets or credentials. +- **`include_cache=True` (default)**: Bundles warm cached chunks along with metadata so reading previously fetched slices requires zero network traffic. +- **`include_cache=False`**: Omits cached payload chunks, producing a minimal reference archive for remote streaming. +- **Subtree export**: Calling `save()` on a group view exports that subtree with relative child keys and the appropriate source root. + +### Reopen reference files with `blosc2.open()` + +Array `.b2nd` carriers and store or table `.b2z` references can be reopened directly with `blosc2.open()`: + +```python +# 1. Reopen an exported RemoteStore hierarchy: +with blosc2.open("snapshot.b2z") as restored: + print(restored.keys()) + temp = restored["experiment/temperature"] + values = temp[:100] # served from archive if cached; fetched remotely if missing + +# 2. Reopen a standalone RemoteArray carrier: +arr = blosc2.open("big-cache.b2nd", mode="a") +values = arr[:100] +``` + +Opening an exported `.b2z` archive recognizes its remote marker and constructs a +{ref}`RemoteStore` or {ref}`RemoteCTable` according to the saved root. Leaves +opened from a store share one cache coordinator and budget. +Opening an on-disk carrier with `mode="a"` returns a {ref}`RemoteArray` and lets newly fetched regions extend the cache; opening with `mode="r"` keeps the cache file unchanged. +Legacy proxy caches created by older Blosc2 versions are also detected and reopened as a {ref}`Proxy`. + +Independent reopening works for fsspec URLs, Caterva2 datasets, and persistent local Blosc2 sources. +The required runtime environment must still be available: fsspec backends and their configuration must be installed, local source paths must remain valid, and authenticated Caterva2 caches must be reopened inside an equivalent {func}`blosc2.c2context`. +Caterva2 credentials are not stored in the cache file. + +An arbitrary custom {ref}`ProxyNDSource` cannot be reconstructed because its Python class and runtime state are not serialized. +In that case, recreate the source explicitly and attach the existing cache with `blosc2.Proxy(source, urlpath="big-cache.b2nd", mode="a")`. + +### Cache mutability: immutable vs. mutable snapshots + +When saving an export, you can configure whether the resulting snapshot operates in **immutable** or **mutable** mode via the `mutable` argument (or the `.mutable` property on `RemoteStore`, `RemoteArray`, or `RemoteCTable`): + +```python +# Default is mutable=False (immutable snapshot) +store.save("read_only.b2z", mutable=False) + +# Or export a mutable snapshot +store.save("writable.b2z", mutable=True) +``` + +| Mode | Behavior when reopened | Cache misses | Modifying operations | +| --- | --- | --- | --- | +| **Immutable** (`mutable=False`, default) | Reads directly in-place from `.b2z` without disk writes. Safe on read-only media (`chmod 0o444`). | Fetched transiently into RAM to satisfy the read; never written to disk or the archive. | `fetch()`, `afetch()`, `trim_cache()`, and `refresh()` are disallowed. | +| **Mutable** (`mutable=True`) | Staged into an independent writable runtime cache directory. Original `.b2z` stays untouched. | Fetched and cached to disk under standard LRU eviction rules. | Fully supported. Can be opened with a smaller budget, trimming excess chunks. | + +```{tip} +Use **immutable snapshots** (`mutable=False`) for sharing reproducible, read-only reference archives or distributing datasets that should never modify local storage. Use **mutable snapshots** (`mutable=True`) when users should be able to expand the local cache with newly fetched regions over time. +``` + +## Refresh a remote hierarchy + +Remote containers (B2Z, Zarr, and HDF5) are assumed immutable by default. +For B2Z tables and stores, reopening a populated disk cache trusts its saved +archive identity and metadata: no HEAD/identity request is made. Uncached data +still requires remote reads. Older caches may perform one identity lookup to +upgrade their metadata. Do not replace the remote archive while using its cache. +Use `store.refresh()` after replacing a container. It never writes to the remote +source. Standalone and nested table refresh behavior is covered in +{doc}`remote_tables`. + +If a remote container is updated on the server—such as adding new datasets or appending data—call `store.refresh()` to update discovery: + +```python +with blosc2.RemoteStore("https://datasets.example.org/data.h5") as store: + # Refresh remote metadata atomically + store.refresh() + + # Re-access datasets from the refreshed store + array = store["experiment/temperature"] +``` + +- **Atomic update**: `store.refresh()` fetches fresh discovery from the remote source before updating the active generation. If discovery fails, the existing store state remains unchanged. +- **Stale handle safety**: Any child handles (`RemoteArray` leaves or group views) opened *before* `refresh()` become stale. Accessing them raises a `RuntimeError`, prompting you to look them up again from the refreshed store. +- **Immutability rule**: Calling `refresh()` on an immutable reference snapshot (`mutable=False`) is disallowed and raises an error. + +## See also + +- {doc}`remote_arrays` — array formats, slicing, expressions, and prefetching. +- {doc}`remote_tables` — table columns, filtering, reference saving, and materialization. +- {ref}`RemoteObject`, {ref}`RemoteStore`, {ref}`RemoteArray`, and {ref}`RemoteCTable` — API reference. diff --git a/doc/guides/remote_tables.md b/doc/guides/remote_tables.md new file mode 100644 index 000000000..ccfd70c7a --- /dev/null +++ b/doc/guides/remote_tables.md @@ -0,0 +1,110 @@ +# Working with Remote Tables + +`RemoteCTable` opens a read-only CTable in an immutable remote `.b2z` archive. +Fixed-width and `blosc2.utf8()` columns are fetched on demand, including their +null masks. A table inside a hierarchy can also be opened through `RemoteStore`. + +`blosc2.open()` dispatches local table archives to `CTable` and remote table +archives to `RemoteCTable`. Remote `.b2z` groups return `RemoteStore` by default; +array leaves retain their `RemoteArray` behavior. Use `dataset="group/table"` +or a `::group/table` URL suffix to select a nested table. For a complete local +download instead, pass `lazy=False, cache_dir="download-cache"`. + +```python +with blosc2.open("https://example.org/readings.b2z") as table: + notes = table["note"][:5] + selected = table.where(table["note"] == "café") + ids = selected["id"][:] +``` + +UTF-8 strings use two compressed arrays: row offsets and encoded bytes. A slice +first reads its offsets, then its byte span. Both reads use the existing range +transport, fetching compressed blocks when worthwhile or whole compressed chunks +otherwise, and decompressing locally. Archive members are ZIP_STORED, as produced +by the Blosc2 writers; the arrays inside remain Blosc2-compressed. + +The backing arrays share the table's cache budget and traffic counters. MEMORY, +DISK (with `cache_dir`) and NONE policies are supported. Size reporting uses source +metadata without scanning strings. Small-member metadata prefetch may also fetch +some payload. Repeated reads can reuse cached blocks; filtering scans the required +columns because persisted indexes are not used remotely. + +Multi-column metadata inspection and row materialization overlap independent +requests by default, up to eight at once. This includes row iteration, display, +and batched Arrow/pandas export; single-column access remains lazy. UTF-8 byte +requests wait for their offsets. Cache publication and decoding stay serialized. + +```python +with blosc2.open(url, max_concurrency=1) as table: # serial control + rows = list(table[:10]) + +with blosc2.RemoteCTable( + url, + max_concurrency=8, + metadata_buffer_bytes=8 << 20, + row_buffer_bytes=64 << 20, +) as table: + table.row_buffer_bytes = 256 << 20 # optional explicit override + rows = list(table[:10]) +``` + +The fixed defaults are 8 MiB of temporary metadata and 64 MiB of temporary row +data, allocated on demand. They do not depend on CPU count or available RAM. +Wider reads use bounded batches; a single oversized required unit runs alone. +These are soft transport budgets, not total RAM limits: decoded output, native +scratch, HTTP overhead and retained caches are additional. DISK caching does not +remove the need to bound temporary reads or consume large outputs in batches. +The buffer keywords belong to RemoteCTable, not `blosc2.open()`; settings may +also be changed on a returned table, including one obtained from RemoteStore. + +The existing 1 MiB compressed-chunk threshold is a block-selection heuristic, +not a maximum response size. Large selections and unsupported partial-block +layouts can still fetch whole chunks. Cross-process shared-cache handles and +read-only artifacts retain their existing guarded row-read paths; standalone +RemoteArray concurrency and RemoteStore discovery behavior are unchanged. + +Columns and views are borrowed from the root table and require it to remain open. +`table.is_cache_mutable` reports whether the local cache is writable, matching +the corresponding RemoteStore and RemoteArray property. It is read-only and does +not imply that the remote table can be modified. +Closing a parent RemoteStore leaves a returned table usable; refreshing the store +invalidates previously returned tables and their columns. Copies and data exports +produce local tables. `save()` writes a portable remote reference containing +bootstrap metadata and any retained cache; `materialize()`, `copy()`, `to_b2z()` +and `to_b2d()` produce independent local tables: + +```python +table.save("table-reference.b2z") +table.save("cold-reference.b2z", include_cache=False) +local = table.materialize(urlpath="complete-local.b2z") +table.to_b2d("complete-local.b2d") +``` + +Saving a reference does not fetch missing table data. Remote writes and +batch-backed `vlstring`/lists/objects and dictionary columns remain unsupported. + +See `examples/ctable/remote_handling.py` for a batched archive writer with a nullable +multilingual UTF-8 column, plus sample row and string-slice traffic measurements. + +## Refresh a remote table + +Remote containers are assumed immutable. A standalone table with a writable +cache can call `refresh()` to rediscover its schema and replace the cache +generation while preserving cache limits and parallel-read settings: + +```python +with blosc2.RemoteCTable(url, cache_dir="table-cache") as table: + table.refresh() + print(table[:5]) +``` + +Previously obtained columns, arrays, and views become stale after a successful +refresh. Refresh a table obtained from a {ref}`RemoteStore` through the root +store, then retrieve the table again. Immutable reference artifacts reject +`refresh()`. + +## See also + +- {doc}`remote_objects` — shared caching, traffic, reference, and lifetime behavior. +- {doc}`remote_arrays` — remote array formats and operations. +- {ref}`RemoteCTable` and {ref}`CTable` — API reference. diff --git a/doc/reference/remotectable.rst b/doc/reference/remotectable.rst index c459cc803..ec325caac 100644 --- a/doc/reference/remotectable.rst +++ b/doc/reference/remotectable.rst @@ -29,5 +29,8 @@ may include sibling leaves. A table selected from a store owns an independent handle, but its columns and views remain borrowed from that table. Refresh a nested table through its root store; a standalone table can call ``refresh()``. +See :doc:`Working with Remote Tables <../guides/remote_tables>` for column +access, filtering, buffering, reference saving, and materialization examples. + .. autoclass:: blosc2.RemoteCTable :members: diff --git a/doc/reference/remoteobject.rst b/doc/reference/remoteobject.rst index d2ec4e8f3..8dba8c333 100644 --- a/doc/reference/remoteobject.rst +++ b/doc/reference/remoteobject.rst @@ -26,5 +26,8 @@ remote-write API. Data-specific operations remain on the concrete classes. Arrays and tables provide ``materialize()``; stores are navigated to a leaf that can be materialized. +See :doc:`Working with Remote Data <../guides/remote_objects>` for the shared +cache, traffic, reference-saving, and lifetime behavior. + .. autoclass:: blosc2.RemoteObject :members: diff --git a/doc/reference/remotestore.rst b/doc/reference/remotestore.rst index 1b99d520c..d24b0b8ee 100644 --- a/doc/reference/remotestore.rst +++ b/doc/reference/remotestore.rst @@ -151,6 +151,9 @@ checks. ``save`` exports ordinary portable warm/cold archives. Private sparse directories are not portable store artifacts. This protocol targets processes sharing a local filesystem, not distributed or network-filesystem ownership. +See :doc:`Working with Remote Data <../guides/remote_objects>` for navigation, +shared caching, traffic, credentials, and portable reference examples. + .. autoclass:: blosc2.RemoteStore :members: :special-members: __getitem__, __iter__ diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 0e1ddeeb0..6c4b12557 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -382,6 +382,44 @@ class TextRow: remote.save(warm_path) +@pytest.mark.parametrize("mutation", ["metadata", "kind", "schema", "source_kind"]) +def test_remote_ctable_reference_rejects_invalid_manifest(tmp_path, mutation): + local = blosc2.CTable(Row, [(1, [1, 2], "one")], create_summary_index=False) + url = remote_table_url(tmp_path, local, f"invalid-{mutation}") + artifact = tmp_path / f"reference-{mutation}.b2z" + with blosc2.RemoteCTable(url) as remote: + remote.save(artifact) + + unpacked = tmp_path / f"unpacked-{mutation}" + with zipfile.ZipFile(artifact) as archive: + archive.extractall(unpacked) + embed = blosc2.blosc2_ext.open(str(unpacked / "embed.b2e"), "a", 0) + manifest = dict(embed.vlmeta["b2remote_manifest"]) + root = manifest["source"]["dataset"] + nodes = dict(manifest["nodes"]) + metadata = dict(nodes[root][1]) + if mutation == "metadata": + metadata = None + elif mutation == "kind": + metadata["kind"] = "group" + elif mutation == "schema": + metadata["schema"] = None + else: + manifest["source"] = {**manifest["source"], "kind": "hdf5"} + nodes[root] = ("ctable", metadata) + manifest["nodes"] = nodes + embed.vlmeta["b2remote_manifest"] = manifest + del embed + + broken = tmp_path / f"broken-{mutation}.b2z" + with zipfile.ZipFile(broken, "w", zipfile.ZIP_STORED) as archive: + for member in unpacked.rglob("*"): + if member.is_file(): + archive.write(member, member.relative_to(unpacked)) + with pytest.raises(ValueError, match="Invalid RemoteStore CTable node"): + blosc2.open(broken) + + def test_nested_remote_ctable_reference_save(tmp_path): source = tmp_path / "tree.b2z" table = blosc2.CTable(Row, [(1, [1, 2], "one"), (2, [3, 4], "two")]) @@ -395,10 +433,19 @@ def test_nested_remote_ctable_reference_save(tmp_path): with blosc2.RemoteStore(url) as store, store["group/table"] as remote: np.testing.assert_array_equal(remote["x"][:], [1, 2]) + with store["group/sibling"] as sibling: + np.testing.assert_array_equal(sibling[:], np.arange(10)) remote.save(destination) with store["group"] as group: group.save(group_destination) + with zipfile.ZipFile(destination) as archive: + names = archive.namelist() + assert any(name.startswith("group/table/") for name in names) + assert not any(name.startswith("group/sibling") for name in names) + with zipfile.ZipFile(group_destination) as archive: + assert any(name.startswith("group/sibling") for name in archive.namelist()) + with blosc2.open(destination) as reopened: assert isinstance(reopened, blosc2.RemoteCTable) assert reopened.source["dataset"] == "group/table" From 998cb91627aef3236ccbc4de95cbd933207d85ae Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 18 Sep 2026 13:33:40 +0200 Subject: [PATCH 11/82] Fix remote reference exports and table reopening --- src/blosc2/remote_array.py | 48 ++-- src/blosc2/remote_ctable.py | 14 +- src/blosc2/remote_store.py | 425 +++++++++++++++-------------- src/blosc2/schunk.py | 4 + tests/ctable/test_remote_ctable.py | 23 ++ tests/test_remote_array.py | 48 ++++ tests/test_remote_store.py | 4 +- 7 files changed, 335 insertions(+), 231 deletions(-) diff --git a/src/blosc2/remote_array.py b/src/blosc2/remote_array.py index 7065eaccf..5a0b7d87f 100644 --- a/src/blosc2/remote_array.py +++ b/src/blosc2/remote_array.py @@ -1749,30 +1749,38 @@ def save( elif urlpath is not None: raise TypeError("destination and urlpath cannot both be specified") destination = os.fspath(destination) - if (cache_policy is not None or not include_cache) and any( - path is not None and os.path.abspath(path) == os.path.abspath(destination) - for path in (self.cache_path, self.runtime_cache_path) - ): - raise ValueError("cold or policy-changing export requires a different destination") - carrier = self._export_carrier(include_cache, cache_policy, mutable=mutable) - source_path = getattr(carrier.schunk, "urlpath", None) - same_live_carrier = source_path is not None and os.path.abspath(source_path) == os.path.abspath( - destination - ) - if same_live_carrier: - if overwrite: - raise ValueError("cannot overwrite the attached live cache") - raise ValueError(f"destination {destination!r} already exists; use overwrite=True to replace it") + dest_real = os.path.realpath(destination) + for attached in (self._carrier, self._runtime_cache): + path = None if attached is None else getattr(attached.schunk, "urlpath", None) + if path is None: + continue + live_real = os.path.realpath(path) + if ( + dest_real == live_real + or (os.path.exists(destination) and os.path.samefile(destination, path)) + or (os.path.isdir(path) and dest_real.startswith(live_real + os.sep)) + ): + raise ValueError( + "cannot overwrite the attached live cache; export requires a different destination" + ) if os.path.exists(destination) and not overwrite: raise ValueError(f"destination {destination!r} already exists; use overwrite=True to replace it") blosc2.blosc2_ext.check_access_mode(destination, "w") - if os.path.exists(destination): - with tempfile.TemporaryDirectory(dir=os.path.dirname(os.path.abspath(destination))) as temp_dir: - staged = os.path.join(temp_dir, os.path.basename(destination)) - carrier.save(staged, contiguous=contiguous, **kwargs) + carrier = self._export_carrier(include_cache, cache_policy, mutable=mutable) + with tempfile.TemporaryDirectory(dir=os.path.dirname(os.path.abspath(destination))) as temp_dir: + staged = os.path.join(temp_dir, "payload") + carrier.save(staged, contiguous=contiguous, **kwargs) + if os.path.exists(destination) and (os.path.isdir(destination) or not contiguous): + # Keep the previous directory until publication succeeds. + previous = os.path.join(temp_dir, "previous") + os.replace(destination, previous) + try: + os.replace(staged, destination) + except BaseException: + os.replace(previous, destination) + raise + else: os.replace(staged, destination) - else: - carrier.save(destination, contiguous=contiguous, **kwargs) return destination @classmethod diff --git a/src/blosc2/remote_ctable.py b/src/blosc2/remote_ctable.py index 388059f9b..6e5a385e0 100644 --- a/src/blosc2/remote_ctable.py +++ b/src/blosc2/remote_ctable.py @@ -262,19 +262,13 @@ def save( elif urlpath is not None: raise TypeError("destination and urlpath cannot both be specified") - from blosc2.remote_store import RemoteStore - storage = self._remote_storage() - owner = storage._owner - relative = storage._root_key[len(owner.root) + 1 :] if owner.root else storage._root_key - selected = object.__new__(RemoteStore) - selected._attach(owner, relative) - try: - return selected.save( + with storage._owner.lock: + storage._check_open() + return storage._owner.save_selection( + storage._root_key, destination, include_cache=include_cache, mutable=mutable, overwrite=overwrite, ) - finally: - selected.close() diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 92f61b7fc..fdc5e18fc 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -807,6 +807,204 @@ def _close_resources(self): close(self.filesystem.loop, session) self.filesystem = None + def save_selection( + self, + full_path, + destination: str | os.PathLike, + *, + include_cache: bool = True, + mutable: bool | None = None, + overwrite: bool = False, + ) -> str: + """Export the current store or subtree to a portable .b2z reference archive.""" + if not isinstance(include_cache, bool): + raise TypeError("include_cache must be a boolean") + if mutable is not None and not isinstance(mutable, bool): + raise TypeError("mutable must be a boolean") + dest_abs, dest_dir = self._validate_save_destination(destination, overwrite) + + with self.lock: + effective_mutable = self.mutable if mutable is None else mutable + if include_cache: + retained = self.cache_coordinator.cache_bytes + if self.max_cache_bytes is not None and retained > self.max_cache_bytes: + raise ValueError( + f"Retained cache ({retained} bytes) exceeds max_cache_bytes ({self.max_cache_bytes})" + ) + + if self.format == "hdf5": + self._validate_hdf5_index() + metadata = self.hdf5_index + elif self.format == "b2z": + metadata = self.archive.metadata + else: + metadata = self.metadata + + src_desc, nodes, attrs, listed, candidates = self._collect_export_nodes(full_path, include_cache) + + staging_dir = tempfile.mkdtemp(prefix="b2z-export-", dir=dest_dir) + fd, tmp_zip = tempfile.mkstemp(prefix="export-", suffix=".b2z.tmp", dir=dest_dir) + os.close(fd) + try: + exported_caches = [] + for orig_key in candidates: + proxy = self.caches.get(orig_key) + if proxy is None or not proxy._cache_sizes: + continue + self._copy_leaf_carrier(orig_key, proxy, staging_dir) + exported_caches.append(orig_key) + + exported_manifest = { + "version": 1, + "source": src_desc, + "generation": self.generation, + "nodes": nodes, + "attrs": attrs, + "listed": listed, + "notice": self.notice, + "metadata": metadata, + "caches": sorted(exported_caches), + "cache_policy": self.cache_policy.value, + "max_cache_bytes": self.max_cache_bytes, + "mutable": effective_mutable, + } + + embed_dst = os.path.join(staging_dir, "embed.b2e") + st = blosc2.Storage(contiguous=True, urlpath=embed_dst, mode="w") + st.meta = {"b2tree": {"version": 1}, "b2remote_store": {"version": 1}} + embed = blosc2.SChunk(chunksize=2**13, data=None, storage=st) + embed.vlmeta["b2remote_manifest"] = exported_manifest + del embed + + filepaths = [] + for root, _, files in os.walk(staging_dir): + for file in files: + fp = os.path.join(root, file) + if os.path.abspath(fp) != os.path.abspath(embed_dst): + filepaths.append(fp) + filepaths.sort(key=os.path.getsize, reverse=True) + + with zipfile.ZipFile(tmp_zip, "w", zipfile.ZIP_STORED) as zf: + for fp in filepaths: + arcname = os.path.relpath(fp, staging_dir) + zf.write(fp, arcname) + zf.write(embed_dst, "embed.b2e") + + os.replace(tmp_zip, dest_abs) + return dest_abs + finally: + if os.path.exists(tmp_zip): + with contextlib.suppress(OSError): + os.unlink(tmp_zip) + shutil.rmtree(staging_dir, ignore_errors=True) + + def _validate_save_destination(self, destination, overwrite): + destination = os.fspath(destination) + if not destination.endswith(".b2z"): + raise ValueError("destination must have a .b2z extension") + dest_abs = os.path.abspath(destination) + if os.path.isdir(dest_abs): + raise ValueError("destination must name a file, not a directory") + if os.path.exists(dest_abs) and not overwrite: + raise FileExistsError(f"'{dest_abs}' already exists. Use overwrite=True to overwrite.") + dest_dir = os.path.dirname(dest_abs) + if not os.path.exists(dest_dir): + raise FileNotFoundError(f"Destination directory '{dest_dir}' does not exist") + if self.disk is not None: + live_dir = os.path.abspath(str(self.disk.path)) + cache_root = os.path.abspath(str(self.disk.path.parent)) + if ( + dest_abs in (live_dir, cache_root) + or dest_abs.startswith(live_dir + os.sep) + or dest_abs.startswith(cache_root + os.sep) + ): + raise ValueError("destination cannot be inside live cache storage") + if self.artifact_path is not None and dest_abs == os.path.abspath(self.artifact_path): + raise ValueError("destination cannot be the source artifact") + return dest_abs, dest_dir + + def _collect_export_nodes(self, group_full, include_cache): + prefix = (group_full + "/") if group_full else "" + for path, (kind, _) in self.nodes.items(): + if kind == "ctable" and (path == group_full or not prefix or path.startswith(prefix)): + self.load_ctable_attrs(path) + exported_source = { + "urlpath": self.urlpath, + "dataset": group_full, + "kind": self.format, + } + fingerprint = storage_options_fingerprint(getattr(self, "storage_options", None)) + if fingerprint: + exported_source["storage_options"] = fingerprint + if prefix: + exported_nodes = { + k: (v[0], v[1] if v[0] in {"ctable", "unsupported"} else None) + for k, v in self.nodes.items() + if k == group_full or k.startswith(prefix) + } + exported_attrs = {k: v for k, v in self.attrs.items() if k == group_full or k.startswith(prefix)} + exported_listed = { + k: list(v) for k, v in self.listed.items() if k == group_full or k.startswith(prefix) + } + candidate_caches = [k for k in self.caches if k.startswith(prefix)] if include_cache else [] + else: + exported_nodes = { + k: (v[0], v[1] if v[0] in {"ctable", "unsupported"} else None) for k, v in self.nodes.items() + } + exported_attrs = dict(self.attrs) + exported_listed = {k: list(v) for k, v in self.listed.items()} + candidate_caches = list(self.caches) if include_cache else [] + return exported_source, exported_nodes, exported_attrs, exported_listed, candidate_caches + + def _copy_leaf_carrier(self, orig_key, proxy, staging_dir): + leaf_filename = f"{orig_key}.b2nd" + leaf_dst = os.path.join(staging_dir, leaf_filename) + os.makedirs(os.path.dirname(leaf_dst), exist_ok=True) + if self.artifact_offsets is not None: + with zipfile.ZipFile(self.artifact_path, "r") as zf: + zf.extract(leaf_filename, staging_dir) + elif self.disk is not None and not getattr(self, "shared", False): + src_file = self.disk.payload_path(self.generation, orig_key) + if src_file.exists(): + shutil.copy2(src_file, leaf_dst) + elif self.artifact_path is not None and os.path.isdir(self.artifact_path): + src_file = os.path.join(self.artifact_path, leaf_filename) + if os.path.exists(src_file): + shutil.copy2(src_file, leaf_dst) + else: + # An in-memory proxy cache is an NDArray, so the leaf must be built the + # same way a disk leaf is: a matching NDArray carrier, not a bare SChunk. + # The ``remote-store`` identity and ``proxy-source`` metalayers are what + # the mutable-reopen path validates when it adopts the extracted leaf. + meta = {name: proxy._schunk_cache.meta[name] for name in proxy._schunk_cache.meta} + meta.pop("b2nd", None) + meta["remote-store"] = {"generation": self.generation, "dataset": orig_key} + cache = proxy._cache + if isinstance(cache, blosc2.NDArray): + leaf = blosc2.empty( + cache.shape, + cache.dtype, + chunks=cache.chunks, + blocks=cache.blocks, + cparams=cache.cparams, + urlpath=leaf_dst, + mode="w", + meta=meta, + ) + leaf_schunk = leaf.schunk + else: + st = blosc2.Storage(contiguous=True, urlpath=leaf_dst, mode="w") + st.meta = meta + leaf = blosc2.SChunk(chunksize=proxy._schunk_cache.chunksize, storage=st) + leaf_schunk = leaf + for nchunk in sorted(proxy._cache_sizes): + chunk = proxy._schunk_cache.get_chunk(nchunk) + if chunk is not None: + leaf_schunk.update_chunk(nchunk, chunk) + for key, value in proxy._schunk_cache.vlmeta.items(): + leaf_schunk.vlmeta[key] = value + del leaf + class RemoteStore(RemoteObject): """Read-only remote B2Z, Zarr or HDF5 hierarchy. @@ -1260,200 +1458,11 @@ def save( overwrite: bool = False, ) -> str: """Export the current store or subtree to a portable .b2z reference archive.""" - if not isinstance(include_cache, bool): - raise TypeError("include_cache must be a boolean") - if mutable is not None and not isinstance(mutable, bool): - raise TypeError("mutable must be a boolean") - dest_abs, dest_dir = self._validate_save_destination(destination, overwrite) - with self._owner.lock: - self._resolve("") - effective_mutable = self.mutable if mutable is None else mutable - if include_cache: - retained = self.cache_bytes - if self.max_cache_bytes is not None and retained > self.max_cache_bytes: - raise ValueError( - f"Retained cache ({retained} bytes) exceeds max_cache_bytes ({self.max_cache_bytes})" - ) - - if self._owner.format == "hdf5": - self._owner._validate_hdf5_index() - metadata = self._owner.hdf5_index - elif self._owner.format == "b2z": - metadata = self._owner.archive.metadata - else: - metadata = self._owner.metadata - - src_desc, nodes, attrs, listed, candidates = self._collect_export_nodes(include_cache) - - staging_dir = tempfile.mkdtemp(prefix="b2z-export-", dir=dest_dir) - fd, tmp_zip = tempfile.mkstemp(prefix="export-", suffix=".b2z.tmp", dir=dest_dir) - os.close(fd) - try: - exported_caches = [] - for orig_key in candidates: - proxy = self._owner.caches.get(orig_key) - if proxy is None or not proxy._cache_sizes: - continue - self._copy_leaf_carrier(orig_key, proxy, staging_dir) - exported_caches.append(orig_key) - - exported_manifest = { - "version": 1, - "source": src_desc, - "generation": self._owner.generation, - "nodes": nodes, - "attrs": attrs, - "listed": listed, - "notice": self._owner.notice, - "metadata": metadata, - "caches": sorted(exported_caches), - "cache_policy": self.cache_policy.value, - "max_cache_bytes": self.max_cache_bytes, - "mutable": effective_mutable, - } - - embed_dst = os.path.join(staging_dir, "embed.b2e") - st = blosc2.Storage(contiguous=True, urlpath=embed_dst, mode="w") - st.meta = {"b2tree": {"version": 1}, "b2remote_store": {"version": 1}} - embed = blosc2.SChunk(chunksize=2**13, data=None, storage=st) - embed.vlmeta["b2remote_manifest"] = exported_manifest - del embed - - filepaths = [] - for root, _, files in os.walk(staging_dir): - for file in files: - fp = os.path.join(root, file) - if os.path.abspath(fp) != os.path.abspath(embed_dst): - filepaths.append(fp) - filepaths.sort(key=os.path.getsize, reverse=True) - - with zipfile.ZipFile(tmp_zip, "w", zipfile.ZIP_STORED) as zf: - for fp in filepaths: - arcname = os.path.relpath(fp, staging_dir) - zf.write(fp, arcname) - zf.write(embed_dst, "embed.b2e") - - os.replace(tmp_zip, dest_abs) - return dest_abs - finally: - if os.path.exists(tmp_zip): - with contextlib.suppress(OSError): - os.unlink(tmp_zip) - shutil.rmtree(staging_dir, ignore_errors=True) - - def _validate_save_destination(self, destination, overwrite): - destination = os.fspath(destination) - if not destination.endswith(".b2z"): - raise ValueError("destination must have a .b2z extension") - dest_abs = os.path.abspath(destination) - if os.path.isdir(dest_abs): - raise ValueError("destination must name a file, not a directory") - if os.path.exists(dest_abs) and not overwrite: - raise FileExistsError(f"'{dest_abs}' already exists. Use overwrite=True to overwrite.") - dest_dir = os.path.dirname(dest_abs) - if not os.path.exists(dest_dir): - raise FileNotFoundError(f"Destination directory '{dest_dir}' does not exist") - if self._owner.disk is not None: - live_dir = os.path.abspath(str(self._owner.disk.path)) - cache_root = os.path.abspath(str(self._owner.disk.path.parent)) - if ( - dest_abs in (live_dir, cache_root) - or dest_abs.startswith(live_dir + os.sep) - or dest_abs.startswith(cache_root + os.sep) - ): - raise ValueError("destination cannot be inside live cache storage") - if self._owner.artifact_path is not None and dest_abs == os.path.abspath(self._owner.artifact_path): - raise ValueError("destination cannot be the source artifact") - return dest_abs, dest_dir - - def _collect_export_nodes(self, include_cache): - group_full = "/".join(p for p in (self._owner.root, self._path.strip("/")) if p) - prefix = (group_full + "/") if group_full else "" - for path, (kind, _) in self._owner.nodes.items(): - if kind == "ctable" and (path == group_full or not prefix or path.startswith(prefix)): - self._owner.load_ctable_attrs(path) - exported_source = { - "urlpath": self._owner.urlpath, - "dataset": group_full, - "kind": self._owner.format, - } - fingerprint = storage_options_fingerprint(getattr(self._owner, "storage_options", None)) - if fingerprint: - exported_source["storage_options"] = fingerprint - if prefix: - exported_nodes = { - k: (v[0], v[1] if v[0] in {"ctable", "unsupported"} else None) - for k, v in self._owner.nodes.items() - if k == group_full or k.startswith(prefix) - } - exported_attrs = { - k: v for k, v in self._owner.attrs.items() if k == group_full or k.startswith(prefix) - } - exported_listed = { - k: list(v) for k, v in self._owner.listed.items() if k == group_full or k.startswith(prefix) - } - candidate_caches = ( - [k for k in self._owner.caches if k.startswith(prefix)] if include_cache else [] + _, full = self._resolve("") + return self._owner.save_selection( + full, destination, include_cache=include_cache, mutable=mutable, overwrite=overwrite ) - else: - exported_nodes = { - k: (v[0], v[1] if v[0] in {"ctable", "unsupported"} else None) - for k, v in self._owner.nodes.items() - } - exported_attrs = dict(self._owner.attrs) - exported_listed = {k: list(v) for k, v in self._owner.listed.items()} - candidate_caches = list(self._owner.caches) if include_cache else [] - return exported_source, exported_nodes, exported_attrs, exported_listed, candidate_caches - - def _copy_leaf_carrier(self, orig_key, proxy, staging_dir): - leaf_filename = f"{orig_key}.b2nd" - leaf_dst = os.path.join(staging_dir, leaf_filename) - os.makedirs(os.path.dirname(leaf_dst), exist_ok=True) - if self._owner.artifact_offsets is not None: - with zipfile.ZipFile(self._owner.artifact_path, "r") as zf: - zf.extract(leaf_filename, staging_dir) - elif self._owner.disk is not None and not getattr(self._owner, "shared", False): - src_file = self._owner.disk.payload_path(self._owner.generation, orig_key) - if src_file.exists(): - shutil.copy2(src_file, leaf_dst) - elif self._owner.artifact_path is not None and os.path.isdir(self._owner.artifact_path): - src_file = os.path.join(self._owner.artifact_path, leaf_filename) - if os.path.exists(src_file): - shutil.copy2(src_file, leaf_dst) - else: - # An in-memory proxy cache is an NDArray, so the leaf must be built the - # same way a disk leaf is: a matching NDArray carrier, not a bare SChunk. - # The ``remote-store`` identity and ``proxy-source`` metalayers are what - # the mutable-reopen path validates when it adopts the extracted leaf. - meta = {name: proxy._schunk_cache.meta[name] for name in proxy._schunk_cache.meta} - meta.pop("b2nd", None) - meta["remote-store"] = {"generation": self._owner.generation, "dataset": orig_key} - cache = proxy._cache - if isinstance(cache, blosc2.NDArray): - leaf = blosc2.empty( - cache.shape, - cache.dtype, - chunks=cache.chunks, - blocks=cache.blocks, - cparams=cache.cparams, - urlpath=leaf_dst, - mode="w", - meta=meta, - ) - leaf_schunk = leaf.schunk - else: - st = blosc2.Storage(contiguous=True, urlpath=leaf_dst, mode="w") - st.meta = meta - leaf = blosc2.SChunk(chunksize=proxy._schunk_cache.chunksize, storage=st) - leaf_schunk = leaf - for nchunk in sorted(proxy._cache_sizes): - chunk = proxy._schunk_cache.get_chunk(nchunk) - if chunk is not None: - leaf_schunk.update_chunk(nchunk, chunk) - for key, value in proxy._schunk_cache.vlmeta.items(): - leaf_schunk.vlmeta[key] = value - del leaf @classmethod def _load_artifact_manifest(cls, urlpath): @@ -1692,14 +1701,32 @@ def _open_artifact(cls, urlpath, mode="r", **kwargs): urlpath, manifest, artifact_offsets, storage_options, cache_policy, limit, cache_dir ) - req_dataset = kwargs.get("dataset") + return cls._select_artifact(owner, kwargs.get("dataset"), kwargs.get("max_concurrency")) + + @classmethod + def _select_artifact(cls, owner, req_dataset, max_concurrency): if owner.nodes[owner.root][0] == "ctable": if req_dataset: owner.close() raise ValueError("dataset cannot select below a RemoteCTable artifact root") - return blosc2.RemoteCTable._from_owner(owner, owner.root) - store = cls.__new__(cls) - store._attach(owner, "") - if req_dataset: - return store[req_dataset] - return store + result = blosc2.RemoteCTable._from_owner(owner, owner.root) + else: + result = cls.__new__(cls) + result._attach(owner, "") + if req_dataset: + store = result + try: + result = store[req_dataset] + finally: + store.close() + try: + if max_concurrency is not None: + if not isinstance(result, blosc2.RemoteCTable): + raise NotImplementedError( + "max_concurrency is only supported for table artifact selections" + ) + result.max_concurrency = max_concurrency + return result + except BaseException: + result.close() + raise diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index 08205101d..9b682095b 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2381,6 +2381,10 @@ def _is_container_open_request(urlpath: str, kwargs: dict) -> bool: _, parsed_dataset, hint = parse_container_url(urlpath, kwargs.get("dataset")) if hint == "hdf5": return True + if hint == "b2z" and os.path.exists(urlpath): + meta = _meta_from_store(urlpath, 0) + if meta is not None and "b2remote_store" in meta: + return False return (hint in {"zarr", "b2z"} or kwargs.get("source_format") == "b2z") and ( kwargs.get("lazy") or parsed_dataset is not None or "dataset" in kwargs ) diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 6c4b12557..c709b16cf 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -382,6 +382,29 @@ class TextRow: remote.save(warm_path) +@pytest.mark.parametrize("nested", [False, True]) +@pytest.mark.parametrize("mutable", [False, True]) +def test_reference_max_concurrency(tmp_path, nested, mutable): + local = blosc2.CTable(Row, [(1, [1, 2], "one")]) + source = tmp_path / "source.b2z" + if nested: + with blosc2.TreeStore(source, mode="w") as tree: + tree["table"] = local + else: + local.to_b2z(source) + url = f"memory://{tmp_path.name}-settings.b2z" + fsspec.filesystem("memory").pipe(url, source.read_bytes()) + artifact = tmp_path / "reference.b2z" + with blosc2.open(url) as remote: + remote.save(artifact, mutable=mutable) + options = {"dataset": "table"} if nested else {} + with blosc2.open(artifact, max_concurrency=1, **options) as reopened: + assert reopened.max_concurrency == 1 + np.testing.assert_array_equal(reopened["x"][:], [1]) + with pytest.raises(ValueError, match="max_concurrency must be a positive integer"): + blosc2.open(artifact, max_concurrency=0, **options) + + @pytest.mark.parametrize("mutation", ["metadata", "kind", "schema", "source_kind"]) def test_remote_ctable_reference_rejects_invalid_manifest(tmp_path, mutation): local = blosc2.CTable(Row, [(1, [1, 2], "one")], create_summary_index=False) diff --git a/tests/test_remote_array.py b/tests/test_remote_array.py index c3ab38749..b5c357393 100644 --- a/tests/test_remote_array.py +++ b/tests/test_remote_array.py @@ -1121,6 +1121,54 @@ def test_cold_export_cannot_overwrite_live_carrier(tmp_path): assert path.read_bytes() == before +@pytest.mark.parametrize("mutable", [None, False, True]) +def test_warm_export_cannot_overwrite_live_carrier(tmp_path, mutable): + url, data = _remote_array("live-export.b2nd", nchunks=2, chunk_size=100) + path = tmp_path / "live.b2nd" + proxy = blosc2.open(url, lazy=True, cache_path=path) + proxy[:100] + before = path.read_bytes() + with pytest.raises(ValueError, match="attached live cache"): + proxy.save(path, overwrite=True, mutable=mutable) + assert path.read_bytes() == before + np.testing.assert_array_equal(proxy[:], data) + + +@pytest.mark.parametrize("contiguous", [False, True]) +@pytest.mark.parametrize("initial_contiguous", [False, True]) +def test_array_export_overwrite(tmp_path, monkeypatch, contiguous, initial_contiguous): + import os + + url, data = _remote_array("sparse-export.b2nd", nchunks=2, chunk_size=100) + proxy = blosc2.open(url, lazy=True) + proxy[:100] + path = tmp_path / "reference.b2nd" + proxy.save(path, initial_contiguous) + + def snapshot(): + return path.read_bytes() if path.is_file() else {p.name: p.read_bytes() for p in path.iterdir()} + + before = snapshot() + proxy[:] + replace = os.replace + + def fail_publication(src, dst): + if os.path.basename(src) == "payload": + raise OSError("publication failed") + return replace(src, dst) + + with monkeypatch.context() as patch: + patch.setattr(os, "replace", fail_publication) + with pytest.raises(OSError, match="publication failed"): + proxy.save(path, contiguous, overwrite=True) + assert snapshot() == before + proxy.save(path, contiguous, overwrite=True) + with blosc2.open(path) as reopened: + reopened.traffic.reset() + np.testing.assert_array_equal(reopened[:], data) + assert reopened.traffic.requests == 0 + + def test_unlimited_disk_cache_does_not_evict(tmp_path): url, data = _remote_array("unlimited-disk.b2nd", nchunks=5, chunk_size=20) carrier_path = tmp_path / "unlimited.b2nd" diff --git a/tests/test_remote_store.py b/tests/test_remote_store.py index 4c7fde178..15e852d65 100644 --- a/tests/test_remote_store.py +++ b/tests/test_remote_store.py @@ -1164,7 +1164,7 @@ def fail(self, *args, **kwargs): destination = tmp_path / "out.b2z" destination.write_bytes(b"sentinel") with monkeypatch.context() as patch: - patch.setattr(blosc2.RemoteStore, "_copy_leaf_carrier", fail) + patch.setattr(type(store._owner), "_copy_leaf_carrier", fail) with pytest.raises(OSError, match="export failed"): store.save(destination, overwrite=True) assert destination.read_bytes() == b"sentinel" @@ -1172,7 +1172,7 @@ def fail(self, *args, **kwargs): fresh = tmp_path / "fresh.b2z" with monkeypatch.context() as patch: - patch.setattr(blosc2.RemoteStore, "_copy_leaf_carrier", fail) + patch.setattr(type(store._owner), "_copy_leaf_carrier", fail) with pytest.raises(OSError, match="export failed"): store.save(fresh) assert not fresh.exists() From c5e176088c239f663c666ca56b32e7c3c197d90b Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 18 Sep 2026 17:50:09 +0200 Subject: [PATCH 12/82] Prove remote batch range reads --- src/blosc2/proxy_source.py | 4 +-- tests/test_b2z_source.py | 55 ++++++++++++++++++++++++++++++++++++++ 2 files changed, 57 insertions(+), 2 deletions(-) diff --git a/src/blosc2/proxy_source.py b/src/blosc2/proxy_source.py index ce79e95ed..2f8f494f8 100644 --- a/src/blosc2/proxy_source.py +++ b/src/blosc2/proxy_source.py @@ -608,7 +608,7 @@ def _frame_offset_reads(header, head, header_len): """ # An empty frame has no chunks, so it has no offsets chunk either: what sits # at index_pos is the trailer, and reading it as one fails obscurely - if header[8] == 0: # chunksize + if header[4] == 0: # nbytes; variable-sized chunks use chunksize == 0 return np.empty(0, dtype=np.int64) # The offsets live in a Blosc2 chunk of their own, right after the data ones, @@ -707,7 +707,7 @@ def _chunk_extents(offsets: np.ndarray, header: list) -> np.ndarray: index_pos = header[1] + header[5] bounds = np.sort(np.append(offsets[offsets >= 0], index_pos)) extents = bounds[np.searchsorted(bounds, offsets, side="right")] - offsets - return np.minimum(extents, header[8] + blosc2.MAX_OVERHEAD) + return np.minimum(extents, header[8] + blosc2.MAX_OVERHEAD) if header[8] else extents class ByteRangeNDSource(ProxyNDSource): diff --git a/tests/test_b2z_source.py b/tests/test_b2z_source.py index 82c12b114..a7ac52b84 100644 --- a/tests/test_b2z_source.py +++ b/tests/test_b2z_source.py @@ -1,5 +1,6 @@ """Native remote reads inside B2Z archives.""" +import dataclasses import io import subprocess import sys @@ -28,6 +29,60 @@ def memory_archive(data=None, *, compression=zipfile.ZIP_STORED, zip64=False): return "memory://v10.b2z", data +def test_remote_batch_member_range_experiment(tmp_path, monkeypatch): + """A distant BatchArray chunk can be decoded without fetching its member.""" + from blosc2.b2z_source import B2ZArchive, member_vlmeta + from blosc2.msgpack_utils import msgpack_unpackb + from blosc2.proxy_source import _chunk_extents, _read_frame_header, _read_frame_offsets + + @dataclasses.dataclass + class Row: + text: str = blosc2.field(blosc2.vlstring(nullable=True, batch_rows=64)) + + rng = np.random.default_rng(42) + values = [None if i % 17 == 0 else rng.bytes(512).hex() for i in range(640)] + table = blosc2.CTable(Row, [(value,) for value in values], create_summary_index=False) + path = tmp_path / "remote-batches.b2z" + table.to_b2z(path) + url = "memory://remote-batches.b2z" + fs = fsspec.filesystem("memory") + fs.pipe_file(url, path.read_bytes()) + + reads = [] + original = type(fs).cat_file + + def counted(self, path, start=None, end=None, **kwargs): + reads.append((start, end)) + return original(self, path, start=start, end=end, **kwargs) + + monkeypatch.setattr(type(fs), "cat_file", counted) + archive = B2ZArchive(url) + try: + info = next(info for info in archive.members if info.filename == "_cols/text.b2b") + member_offset, member_length = archive.member_window(info) + + def read_range(offset, size): + return archive._read_archive(member_offset + offset, size) + + raw, header, head = _read_frame_header(read_range) + offsets = _read_frame_offsets(read_range, header, head, len(raw)) + metadata = member_vlmeta(archive, info)["_batch_array_metadata"] + batch_index = len(metadata["batch_lengths"]) - 2 + extents = _chunk_extents(offsets, header) + chunk = read_range(int(offsets[batch_index]), int(extents[batch_index])) + chunk = chunk[: int.from_bytes(chunk[12:16], "little")] + decoded = [ + item for block in blosc2.blosc2_ext.vldecompress(chunk) for item in msgpack_unpackb(block) + ] + + start = sum(metadata["batch_lengths"][:batch_index]) + assert decoded == values[start : start + metadata["batch_lengths"][batch_index]] + assert member_length > 64 * 1024 + assert sum(end - start for start, end in reads) < member_length + finally: + archive.close() + + @pytest.mark.parametrize("address", ["::/d0/a", "/d0/a", "keyword"]) def test_addressing_and_hits(address, monkeypatch): url, data = memory_archive() From c941ed9445f02030df8dd7c717487b29e3402673 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 18 Sep 2026 17:52:02 +0200 Subject: [PATCH 13/82] Add internal remote batch reader --- src/blosc2/b2z_source.py | 52 ++++++++++++++++++ src/blosc2/remote_batch.py | 109 +++++++++++++++++++++++++++++++++++++ tests/test_b2z_source.py | 26 +++++++++ 3 files changed, 187 insertions(+) create mode 100644 src/blosc2/remote_batch.py diff --git a/src/blosc2/b2z_source.py b/src/blosc2/b2z_source.py index c1d74e465..0a9a7cdc3 100644 --- a/src/blosc2/b2z_source.py +++ b/src/blosc2/b2z_source.py @@ -439,6 +439,58 @@ def read_range(self, offset, size): return self._read_archive(self.member_offset + offset, size) +class B2ZBatchSource: + """Internal byte-range source for one external BatchArray member.""" + + def __init__(self, archive, dataset): + from blosc2.proxy_source import ( + _chunk_extents, + _read_frame_header, + _read_frame_metalayers, + _read_frame_offsets, + ) + + self.archive = archive + self.dataset = dataset.strip("/") + matches = [info for info in archive.members if info.filename == self.dataset + ".b2b"] + if len(matches) != 1: + raise NotImplementedError(f"Remote CTable batch member {self.dataset!r} is unavailable") + self.info = matches[0] + self.member_offset, self.member_length = archive.member_window(self.info, prefetch=True) + prefix_start, prefix = archive._opening_ranges[-1] + start = self.member_offset - prefix_start + head = prefix[start:] if 0 <= start < len(prefix) else None + raw, self.header, head = _read_frame_header(self.read_range, head=head) + if not len(raw) <= self.header[2] <= self.member_length: + raise ValueError(f"Batch frame for {self.dataset!r} exceeds its B2Z member bounds") + self.meta = _read_frame_metalayers(raw, self.header) + self.vlmeta = member_vlmeta(archive, self.info) + self.offsets = _read_frame_offsets(self.read_range, self.header, head, len(raw)) + self.extents = _chunk_extents(self.offsets, self.header) + archive._opening_ranges.clear() + + def read_range(self, offset, size): + offset, size = operator.index(offset), operator.index(size) + if offset < 0 or size < 0: + raise ValueError("invalid B2Z batch range") + size = max(0, min(size, self.member_length - offset)) + return self.archive._read_archive(self.member_offset + offset, size) + + def get_chunk(self, index): + offset = int(self.offsets[index]) + if offset < 0: + raise ValueError(f"Batch {index} of {self.dataset!r} has an unsupported special offset") + chunk = self.read_range(offset, int(self.extents[index])) + if len(chunk) < 16: + raise ValueError(f"Truncated batch {index} in {self.dataset!r}") + cbytes = int.from_bytes(chunk[12:16], "little") + if not 16 <= cbytes <= len(chunk): + raise ValueError(f"Invalid compressed size for batch {index} in {self.dataset!r}") + # ponytail: remote variable-length reads fetch whole batches; add block + # transport only if oversized batches prove this granularity insufficient. + return chunk[:cbytes] + + def member_vlmeta(archive, info): """Read a member's frame trailer without loading its embedded payload.""" from blosc2.proxy_source import _parse_trailer_vlmeta diff --git a/src/blosc2/remote_batch.py b/src/blosc2/remote_batch.py new file mode 100644 index 000000000..71246bdae --- /dev/null +++ b/src/blosc2/remote_batch.py @@ -0,0 +1,109 @@ +"""Internal read-only BatchArray adapter over remote compressed batches.""" + +from __future__ import annotations + +import blosc2 +from blosc2.batch_array import ( + _BATCHARRAY_VLMETA_KEY, + Batch, + BatchArray, + BatchArrayItems, +) + + +class _RemoteBatch(Batch): + def _payloads(self): + payloads = getattr(self, "_remote_payloads", None) + if payloads is None: + payloads = blosc2.blosc2_ext.vldecompress(self._lazybatch) + self._remote_payloads = payloads + return payloads + + def _decode_items(self): + if self._items is None: + self._items = [ + item for payload in self._payloads() for item in self._parent._deserialize_block(payload) + ] + return self._items + + def _get_block(self, block_index): + if self._cached_block_index != block_index or self._cached_block is None: + self._cached_block = self._parent._deserialize_block(self._payloads()[block_index]) + self._cached_block_index = block_index + return self._cached_block + + def _get_block_item(self, block_index, item_index): + if self._parent._serializer == "arrow": + return self._parent._deserialize_arrow_block_item(self._payloads()[block_index], item_index) + return self._get_block(block_index)[item_index] + + +class _RemoteBatchArray(BatchArray): + """The BatchArray read surface needed by remote CTable wrappers.""" + + def __init__(self, source, column): + try: + metadata = source.meta["batcharray"] + except KeyError as exc: + raise ValueError(f"Remote batch column {column!r} is not tagged as a BatchArray") from exc + self._serializer = metadata.get("serializer", "msgpack") + self._items_per_block = metadata.get("items_per_block") + self._arrow_schema = metadata.get("arrow_schema") + self._arrow_schema_obj = None + self._source = source + self.schunk = _RemoteBatchSChunk(source) + self.mode = "r" + self.mmap_mode = None + self._batch_lengths = self._validated_lengths(column) + self._items = BatchArrayItems(self) + self._item_prefix_sums = None + self._validate_tag() + + def _validated_lengths(self, column): + metadata = self.schunk.vlmeta.get(_BATCHARRAY_VLMETA_KEY, {}) + lengths = metadata.get("batch_lengths") + count = self.schunk.nchunks + if count == 0 and lengths in (None, []): + return [] + if not isinstance(lengths, list) or len(lengths) != count: + raise ValueError(f"Remote batch column {column!r} requires one persisted batch length per batch") + if any(isinstance(length, bool) or not isinstance(length, int) or length < 0 for length in lengths): + raise ValueError(f"Remote batch column {column!r} has invalid persisted batch lengths") + if sum(lengths) > self.schunk.nbytes: + raise ValueError(f"Remote batch column {column!r} has invalid persisted batch length bounds") + return lengths + + def _get_batch(self, index): + return _RemoteBatch(self, index, self._source.get_chunk(index)) + + def _check_writable(self): + raise ValueError("Cannot modify a remote BatchArray") + + +class _RemoteBatchSChunk: + def __init__(self, source): + self._source = source + self.meta = source.meta + self.vlmeta = source.vlmeta + self.mode = "r" + self.mmap_mode = None + self.nchunks = len(source.offsets) + self.nbytes = int(source.header[4]) + self.cbytes = int(source.header[5]) + self.typesize = 1 + self.urlpath = None + self.contiguous = True + self.cparams = blosc2.CParams(typesize=1) + self.dparams = blosc2.DParams() + + @property + def cratio(self): + return self.nbytes / self.cbytes if self.cbytes else 0.0 + + def get_chunk(self, index): + return self._source.get_chunk(index) + + get_lazychunk = get_chunk + + def get_vlblock(self, chunk, block): + return blosc2.blosc2_ext.vldecompress(self.get_chunk(chunk))[block] diff --git a/tests/test_b2z_source.py b/tests/test_b2z_source.py index a7ac52b84..fcbc39eee 100644 --- a/tests/test_b2z_source.py +++ b/tests/test_b2z_source.py @@ -83,6 +83,32 @@ def read_range(offset, size): archive.close() +def test_internal_remote_batch_reader(tmp_path): + from blosc2.b2z_source import B2ZArchive, B2ZBatchSource + from blosc2.remote_batch import _RemoteBatchArray + + batches = [[f"batch {batch}: item {item}" for item in range(batch + 1)] for batch in range(8)] + source = blosc2.BatchArray(items_per_block=2) + source.extend(batches) + path = tmp_path / "batch-reader.b2z" + with zipfile.ZipFile(path, "w") as archive: + archive.writestr("data.b2b", source.to_cframe()) + fs = fsspec.filesystem("memory") + fs.pipe_file("batch-reader.b2z", path.read_bytes()) + + archive = B2ZArchive("memory://batch-reader.b2z") + try: + remote = _RemoteBatchArray(B2ZBatchSource(archive, "data"), "data") + assert len(remote) == len(batches) + assert remote.items_per_block == 2 + assert remote.nbytes == source.nbytes + assert remote.cbytes == source.cbytes + assert remote[-2][:] == batches[-2] + assert remote.items[-1] == batches[-1][-1] + finally: + archive.close() + + @pytest.mark.parametrize("address", ["::/d0/a", "/d0/a", "keyword"]) def test_addressing_and_hits(address, monkeypatch): url, data = memory_archive() From 3bbb744ab40133c51111132720e4929e7ad5f652 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 18 Sep 2026 17:54:46 +0200 Subject: [PATCH 14/82] Read remote variable-length strings --- src/blosc2/ctable_storage.py | 10 +++++--- src/blosc2/proxy_source.py | 3 ++- src/blosc2/remote_store.py | 9 +++++++ tests/ctable/test_remote_ctable.py | 41 +++++++++++++++++++++++++++--- 4 files changed, 56 insertions(+), 7 deletions(-) diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index 3ded6b608..afa790a53 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -734,9 +734,13 @@ def open_varlen_scalar_column(self, name: str, spec) -> _ScalarVarLenArray: array.close() del self._arrays[first:] raise - raise NotImplementedError( - f"Remote CTable variable-length column {name!r} ({type(spec).__name__}) is not supported" - ) + from blosc2.remote_batch import _RemoteBatchArray + + key = f"{_COLS_DIR}/{_column_name_to_relpath(name)}" + full = self._full_key(key) + backend = _RemoteBatchArray(self._owner.open_ctable_batch(full), name) + _validate_role_metadata(backend, spec) + return _ScalarVarLenArray(spec, backend) def open_dictionary_column(self, name: str, spec) -> DictionaryColumn: raise NotImplementedError(f"Remote CTable dictionary column {name!r} is not supported") diff --git a/src/blosc2/proxy_source.py b/src/blosc2/proxy_source.py index 2f8f494f8..5c5c48e23 100644 --- a/src/blosc2/proxy_source.py +++ b/src/blosc2/proxy_source.py @@ -707,7 +707,8 @@ def _chunk_extents(offsets: np.ndarray, header: list) -> np.ndarray: index_pos = header[1] + header[5] bounds = np.sort(np.append(offsets[offsets >= 0], index_pos)) extents = bounds[np.searchsorted(bounds, offsets, side="right")] - offsets - return np.minimum(extents, header[8] + blosc2.MAX_OVERHEAD) if header[8] else extents + cap = header[8] + 2 * blosc2.MAX_OVERHEAD + return np.minimum(extents, cap) if header[8] else extents class ByteRangeNDSource(ProxyNDSource): diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index fdc5e18fc..d62378555 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -567,6 +567,15 @@ def open_ctable_array(self, table_path, logical_key): self.sources[full] = source return source + def open_ctable_batch(self, full): + """Open one external BatchArray member hidden below a CTable node.""" + if self.format != "b2z": + raise NotImplementedError("Remote CTable access currently requires a B2Z source") + self._validate(full) + from blosc2.b2z_source import B2ZBatchSource + + return B2ZBatchSource(self.archive, full) + def load_ctable_attrs(self, table_path): """Load one table's user attributes without opening its data arrays.""" if table_path in self.attrs: diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index c709b16cf..cfd826dc6 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -310,6 +310,42 @@ def test_remote_ctable_fixed_width_reads_and_queries(tmp_path): table["tag"][:] +def test_remote_ctable_nullable_vlstring_none_cache(tmp_path, monkeypatch): + @dataclasses.dataclass + class TextRow: + text: str = blosc2.field(blosc2.vlstring(nullable=True, batch_rows=32)) + + rng = np.random.default_rng(42) + values = np.asarray( + [ + None if i % 19 == 0 else "" if i % 23 == 0 else f"café 東京 {i} " + rng.bytes(1024).hex() + for i in range(160) + ], + dtype=object, + ) + local = blosc2.CTable(TextRow, [(value,) for value in values], create_summary_index=False) + url = remote_table_url(tmp_path, local, "vlstring-none") + fs = fsspec.filesystem("memory") + reads = [] + original = type(fs).cat_file + + def counted(self, path, start=None, end=None, **kwargs): + reads.append((start, end)) + return original(self, path, start=start, end=end, **kwargs) + + monkeypatch.setattr(type(fs), "cat_file", counted) + with blosc2.RemoteCTable(url, cache_policy=blosc2.CachePolicy.NONE) as remote: + assert remote.col_names == ["text"] + column = remote["text"] + reads.clear() + assert column[-1] == values[-1] + transferred = sum(end - start for start, end in reads) + assert transferred < (tmp_path / "vlstring-none.b2z").stat().st_size + assert column[0] is None + assert column[23] == "" + assert column[31:34] == values[31:34].tolist() + + def test_remote_store_returns_table_with_independent_lifetime(tmp_path): source = tmp_path / "tree.b2z" table = blosc2.CTable(Row, [(1, [1, 2], "one"), (2, [3, 4], "two")], create_summary_index=False) @@ -485,7 +521,7 @@ def test_nested_remote_ctable_reference_save(tmp_path): assert isinstance(sibling, blosc2.RemoteArray) -def test_remote_ctable_unsupported_column_is_lazy(tmp_path): +def test_remote_ctable_batch_column_is_lazy(tmp_path): @dataclasses.dataclass class Mixed: x: int = 0 @@ -498,8 +534,7 @@ class Mixed: ) with blosc2.RemoteCTable(url, cache_policy=blosc2.CachePolicy.NONE) as table: np.testing.assert_array_equal(table["x"][:], [1, 2]) - with pytest.raises(NotImplementedError, match="variable-length column 'text'"): - table["text"][:] + assert table["text"][:] == ["a", "bb"] @pytest.mark.parametrize("policy", list(blosc2.CachePolicy)) From 3c4f182111e0c6d1732ebca0d9ec74b30ddff676 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 18 Sep 2026 17:57:17 +0200 Subject: [PATCH 15/82] Integrate remote batch caching --- src/blosc2/ctable_remote_read.py | 17 ++++--- src/blosc2/remote_batch.py | 71 ++++++++++++++++++++++++++++++ src/blosc2/remote_store.py | 19 +++++++- src/blosc2/remote_store_cache.py | 9 ++++ tests/ctable/test_remote_ctable.py | 30 +++++++++++++ 5 files changed, 139 insertions(+), 7 deletions(-) diff --git a/src/blosc2/ctable_remote_read.py b/src/blosc2/ctable_remote_read.py index 39162e10a..ef0ac7efd 100644 --- a/src/blosc2/ctable_remote_read.py +++ b/src/blosc2/ctable_remote_read.py @@ -103,7 +103,7 @@ def advance(key, reader, answer=None, error=None): def open_columns(storage, table, names, load): # noqa: C901 """Fetch column prefixes in bounded groups, then open each column serially.""" from blosc2.ctable_storage import _column_name_to_relpath - from blosc2.schema import UTF8Spec + from blosc2.schema import ListSpec, ObjectSpec, StructSpec, UTF8Spec, VLBytesSpec, VLStringSpec owner = storage._owner with owner.lock: @@ -116,14 +116,19 @@ def open_columns(storage, table, names, load): # noqa: C901 def ranges_for(name): key = storage._full_key(f"_cols/{_column_name_to_relpath(name)}") spec = table._schema.columns_by_name[name].spec - keys = (key, key + ".utf8") if isinstance(spec, UTF8Spec) else (key,) + if isinstance(spec, UTF8Spec): + members_for_column = ((key, ".b2nd"), (key + ".utf8", ".b2nd")) + elif isinstance(spec, (VLStringSpec, VLBytesSpec, StructSpec, ObjectSpec, ListSpec)): + members_for_column = ((key, ".b2b"),) + else: + members_for_column = ((key, ".b2nd"),) ranges = [] - for key in keys: - if key in owner.sources: + for key, suffix in members_for_column: + if key in owner.sources or key in owner.batch_caches: continue - if key in archive.metadata.get("ctable_seeds", {}): + if suffix == ".b2nd" and key in archive.metadata.get("ctable_seeds", {}): continue - matches = members.get(key + ".b2nd", ()) + matches = members.get(key + suffix, ()) if len(matches) != 1: # The ordinary opener supplies the appropriate diagnostic. return [] diff --git a/src/blosc2/remote_batch.py b/src/blosc2/remote_batch.py index 71246bdae..df2539be9 100644 --- a/src/blosc2/remote_batch.py +++ b/src/blosc2/remote_batch.py @@ -2,6 +2,10 @@ from __future__ import annotations +import os +from collections import OrderedDict +from pathlib import Path + import blosc2 from blosc2.batch_array import ( _BATCHARRAY_VLMETA_KEY, @@ -80,6 +84,73 @@ def _check_writable(self): raise ValueError("Cannot modify a remote BatchArray") +class _RemoteBatchCache: + """Compressed batch retention using the owner's aggregate cache budget.""" + + def __init__(self, source, key, coordinator, path=None): + self._source = source + self._cache_key = key + self._cache_coordinator = coordinator + self._path = None if path is None else Path(path) + self._memory = {} + self._cache_sizes = {} + self._cache_lru = OrderedDict() + if self._path is not None: + self._path.mkdir(parents=True, exist_ok=True) + for file in sorted(self._path.glob("*.chunk"), key=lambda item: int(item.stem)): + index = int(file.stem) + if index < len(source.offsets): + self._cache_sizes[index] = file.stat().st_size + self._cache_lru[index] = None + coordinator.register(self) + + def __getattr__(self, name): + return getattr(self._source, name) + + def _file(self, index): + return self._path / f"{index}.chunk" + + def get_chunk(self, index): + if index in self._cache_sizes: + chunk = self._file(index).read_bytes() if self._path is not None else self._memory[index] + else: + chunk = self._source.get_chunk(index) + if self._path is None: + self._memory[index] = chunk + else: + from blosc2.remote_store_cache import atomic_write + + atomic_write(self._file(index), chunk) + self._cache_sizes[index] = len(chunk) + self._cache_lru.pop(index, None) + self._cache_lru[index] = None + self._cache_coordinator.touch(self, index) + self._cache_coordinator.enforce() + return chunk + + def _sync_evictions(self): + pass + + def _retained_cache_bytes(self): + return sum(self._cache_sizes.values()) + + def _trim_cache(self, target_bytes, *, max_chunks=None): + removed = [] + while self._retained_cache_bytes() > target_bytes and self._cache_lru: + if max_chunks is not None and len(removed) >= max_chunks: + break + index = next(iter(self._cache_lru)) + if self._path is None: + self._memory.pop(index, None) + else: + os.unlink(self._file(index)) + self._cache_lru.pop(index) + self._cache_sizes.pop(index) + self._cache_coordinator.forget(self, index) + removed.append(index) + return tuple(removed) + + class _RemoteBatchSChunk: def __init__(self, source): self._source = source diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index d62378555..d2e90c7d7 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -98,6 +98,7 @@ def __init__( self.zstore = None self.sources = {} self.caches = {} + self.batch_caches = {} self.disk = None self.generation = manifest["generation"] if manifest else uuid.uuid4().hex self.metadata = manifest["metadata"] if manifest else {} @@ -574,7 +575,22 @@ def open_ctable_batch(self, full): self._validate(full) from blosc2.b2z_source import B2ZBatchSource - return B2ZBatchSource(self.archive, full) + self.archive.capture_metadata = True + try: + source = B2ZBatchSource(self.archive, full) + finally: + self.archive.capture_metadata = False + self.archive._opening_ranges.clear() + if self.cache_policy is blosc2.CachePolicy.NONE: + return source + if full not in self.batch_caches: + from blosc2.remote_batch import _RemoteBatchCache + + path = None + if self.disk is not None: + path = self.disk.batch_payload_path(self.generation, full) + self.batch_caches[full] = _RemoteBatchCache(source, full, self.cache_coordinator, path) + return self.batch_caches[full] def load_ctable_attrs(self, table_path): """Load one table's user attributes without opening its data arrays.""" @@ -802,6 +818,7 @@ def _close_resources(self): source.close() self.sources.clear() self.caches.clear() + getattr(self, "batch_caches", {}).clear() self.nodes.clear() self.attrs.clear() self.listed.clear() diff --git a/src/blosc2/remote_store_cache.py b/src/blosc2/remote_store_cache.py index 5d801d2ad..fdb7d8658 100644 --- a/src/blosc2/remote_store_cache.py +++ b/src/blosc2/remote_store_cache.py @@ -134,6 +134,15 @@ def payload_path(self, generation, key): leaf_path.parent.mkdir(parents=True, exist_ok=True) return leaf_path + def batch_payload_path(self, generation, key): + from blosc2.remote_store import RemoteDiscovery + + validate_generation(generation) + RemoteDiscovery._validate(key) + path = self.path / f"{generation}.b2d" / f"{key}.b2b.cache" + path.parent.mkdir(parents=True, exist_ok=True) + return path + def discard_old_generations(self, active): active_name = f"{active}.b2d" for path in self.path.iterdir(): diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index cfd826dc6..1693ff9fd 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -346,6 +346,36 @@ def counted(self, path, start=None, end=None, **kwargs): assert column[31:34] == values[31:34].tolist() +@pytest.mark.parametrize("policy", [blosc2.CachePolicy.MEMORY, blosc2.CachePolicy.DISK]) +def test_remote_ctable_vlstring_cache_and_reopen(tmp_path, policy): + @dataclasses.dataclass + class Mixed: + x: int + text: str = blosc2.field(blosc2.vlstring(batch_rows=16)) + + rng = np.random.default_rng(7) + values = [rng.bytes(1024).hex() for _ in range(128)] + local = blosc2.CTable(Mixed, list(enumerate(values)), create_summary_index=False) + url = remote_table_url(tmp_path, local, f"vlstring-{policy.value}") + options = {"cache_policy": policy, "max_cache_bytes": 48 << 10} + if policy is blosc2.CachePolicy.DISK: + options["cache_dir"] = tmp_path / "cache" + + with blosc2.RemoteCTable(url, **options) as remote: + assert remote["text"][80:85] == values[80:85] + requests = remote.traffic.requests + assert remote["text"][80:85] == values[80:85] + assert remote.traffic.requests == requests + np.testing.assert_array_equal(remote["x"][80:85], np.arange(80, 85)) + assert remote.cache_bytes <= options["max_cache_bytes"] + + if policy is blosc2.CachePolicy.DISK: + with blosc2.RemoteCTable(url, **options) as remote: + requests = remote.traffic.requests + assert remote["text"][80:85] == values[80:85] + assert remote.traffic.requests == requests + + def test_remote_store_returns_table_with_independent_lifetime(tmp_path): source = tmp_path / "tree.b2z" table = blosc2.CTable(Row, [(1, [1, 2], "one"), (2, [3, 4], "two")], create_summary_index=False) From 3f58deedf4c34ae7e58c9cec7bec0f323c39a569 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 18 Sep 2026 18:01:11 +0200 Subject: [PATCH 16/82] Enable remote batch-backed columns --- src/blosc2/ctable_remote_read.py | 12 ++++- src/blosc2/ctable_storage.py | 53 ++++++++++++++------ src/blosc2/list_array.py | 18 +++++++ src/blosc2/msgpack_utils.py | 35 +++++++++++++ src/blosc2/remote_batch.py | 5 ++ tests/ctable/test_remote_ctable.py | 80 ++++++++++++++++++++++++++++++ 6 files changed, 188 insertions(+), 15 deletions(-) diff --git a/src/blosc2/ctable_remote_read.py b/src/blosc2/ctable_remote_read.py index ef0ac7efd..131661d0b 100644 --- a/src/blosc2/ctable_remote_read.py +++ b/src/blosc2/ctable_remote_read.py @@ -103,7 +103,15 @@ def advance(key, reader, answer=None, error=None): def open_columns(storage, table, names, load): # noqa: C901 """Fetch column prefixes in bounded groups, then open each column serially.""" from blosc2.ctable_storage import _column_name_to_relpath - from blosc2.schema import ListSpec, ObjectSpec, StructSpec, UTF8Spec, VLBytesSpec, VLStringSpec + from blosc2.schema import ( + DictionarySpec, + ListSpec, + ObjectSpec, + StructSpec, + UTF8Spec, + VLBytesSpec, + VLStringSpec, + ) owner = storage._owner with owner.lock: @@ -118,6 +126,8 @@ def ranges_for(name): spec = table._schema.columns_by_name[name].spec if isinstance(spec, UTF8Spec): members_for_column = ((key, ".b2nd"), (key + ".utf8", ".b2nd")) + elif isinstance(spec, DictionarySpec): + members_for_column = ((key, ".b2nd"), (key + "_dict", ".b2b")) elif isinstance(spec, (VLStringSpec, VLBytesSpec, StructSpec, ObjectSpec, ListSpec)): members_for_column = ((key, ".b2b"),) else: diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index afa790a53..2e50db616 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -25,7 +25,7 @@ import copy import json import os -from typing import TYPE_CHECKING, Any +from typing import Any import numpy as np @@ -40,12 +40,9 @@ _ScalarVarLenArray, _validate_role_metadata, ) -from blosc2.schema import UTF8Spec +from blosc2.schema import ListSpec, UTF8Spec from blosc2.schunk import process_opened_object -if TYPE_CHECKING: - from blosc2.schema import ListSpec - # Directory inside the table root that holds per-column index sidecar files. _INDEXES_DIR = "_indexes" @@ -713,7 +710,26 @@ def open_column(self, name: str) -> blosc2.RemoteArray: return self._open_array(f"{_COLS_DIR}/{_column_name_to_relpath(name)}") def open_list_column(self, name: str) -> ListArray: - raise NotImplementedError(f"Remote CTable list column {name!r} is not supported") + spec = self._schema_spec(name) + if spec.storage != "batch": + raise NotImplementedError( + f"Remote CTable list column {name!r} with storage='vl' is not supported" + ) + return ListArray._from_batch_backend(spec, self._open_batch(name, spec)) + + def _schema_spec(self, name): + from blosc2.schema_compiler import schema_from_dict + + return schema_from_dict(self.load_schema()).columns_by_name[name].spec + + def _open_batch(self, name, spec, *, suffix=""): + from blosc2.remote_batch import _RemoteBatchArray + + key = f"{_COLS_DIR}/{_column_name_to_relpath(name)}{suffix}" + backend = _RemoteBatchArray(self._owner.open_ctable_batch(self._full_key(key)), name) + if not isinstance(spec, ListSpec): + _validate_role_metadata(backend, spec) + return backend def open_varlen_scalar_column(self, name: str, spec) -> _ScalarVarLenArray: if isinstance(spec, UTF8Spec): @@ -734,16 +750,25 @@ def open_varlen_scalar_column(self, name: str, spec) -> _ScalarVarLenArray: array.close() del self._arrays[first:] raise - from blosc2.remote_batch import _RemoteBatchArray - - key = f"{_COLS_DIR}/{_column_name_to_relpath(name)}" - full = self._full_key(key) - backend = _RemoteBatchArray(self._owner.open_ctable_batch(full), name) - _validate_role_metadata(backend, spec) - return _ScalarVarLenArray(spec, backend) + return _ScalarVarLenArray(spec, self._open_batch(name, spec)) def open_dictionary_column(self, name: str, spec) -> DictionaryColumn: - raise NotImplementedError(f"Remote CTable dictionary column {name!r} is not supported") + from blosc2.schema import VLStringSpec + + first = len(self._arrays) + codes = self.open_column(name) + try: + if codes.ndim != 1 or codes.dtype != np.dtype("int32"): + raise ValueError( + f"Remote dictionary column {name!r} requires a one-dimensional int32 code array" + ) + dict_spec = VLStringSpec(nullable=False) + backend = self._open_batch(name, dict_spec, suffix=_DICT_SUFFIX) + return DictionaryColumn(spec, codes, _ScalarVarLenArray(dict_spec, backend)) + except BaseException: + codes.close() + del self._arrays[first:] + raise def open_valid_rows(self) -> blosc2.RemoteArray: return self._open_array("_valid_rows") diff --git a/src/blosc2/list_array.py b/src/blosc2/list_array.py index 25143aea4..7ad4d19fa 100644 --- a/src/blosc2/list_array.py +++ b/src/blosc2/list_array.py @@ -296,6 +296,24 @@ def _init_from_schunk(self, schunk) -> None: self._backend = BatchArray(_from_schunk=schunk) self._persisted_row_count = self._persisted_rows_count() + @classmethod + def _from_batch_backend(cls, spec, backend): + """Build an internal read wrapper around an already-open BatchArray.""" + if spec.storage != "batch": + raise NotImplementedError("Remote ListArray storage='vl' is not supported") + stored = backend.meta.get("listarray") + if stored != spec.to_listarray_metadata(): + raise ValueError("Remote ListArray metadata does not match its table schema") + obj = object.__new__(cls) + obj.spec = spec + obj._pending_cells = [] + obj._persisted_prefix_cache = None + obj._cached_batch_index = None + obj._cached_batch_values = None + obj._backend = backend + obj._persisted_row_count = obj._persisted_rows_count() + return obj + def _invalidate_batch_caches(self) -> None: self._persisted_prefix_cache = None self._cached_batch_index = None diff --git a/src/blosc2/msgpack_utils.py b/src/blosc2/msgpack_utils.py index 8802d864b..692489872 100644 --- a/src/blosc2/msgpack_utils.py +++ b/src/blosc2/msgpack_utils.py @@ -151,3 +151,38 @@ def _decode_msgpack_ext(code, data): def msgpack_unpackb(payload): return unpackb(payload, list_hook=decode_tuple_list_hook, ext_hook=_decode_msgpack_ext) + + +def _safe_msgpack_unpackb(payload): + """Decode passive values while rejecting executable or referential extensions.""" + + def decode_ext(code, data): + if code == _BLOSC2_COMPLEX_EXT_CODE: + real, imag = struct.unpack(">dd", data) + return complex(real, imag) + if code == _BLOSC2_SET_EXT_CODE: + return set(_safe_msgpack_unpackb(data)) + if code == _BLOSC2_NDARRAY_EXT_CODE: + value = _safe_msgpack_unpackb(data) + shape = value.get("shape") + if not isinstance(shape, list) or any( + isinstance(size, bool) or not isinstance(size, int) or size < 0 for size in shape + ): + raise ValueError("Unsafe remote NumPy extension shape") + count = int(np.prod(shape, dtype=np.int64)) + if "values" in value: + if not isinstance(value["values"], list) or len(value["values"]) != count: + raise ValueError("Invalid remote object-array extension") + result = np.empty(shape, dtype=object) + result.reshape(-1)[:] = value["values"] + return result + from blosc2.hdf5_source import dtype_from_value + + dtype = dtype_from_value(value["dtype"]) + data = value["data"] + if dtype.hasobject or not isinstance(data, bytes) or len(data) != count * dtype.itemsize: + raise ValueError("Invalid remote NumPy extension payload") + return np.frombuffer(data, dtype=dtype).reshape(shape) + raise ValueError(f"Unsafe remote MessagePack extension code {code}") + + return unpackb(payload, list_hook=decode_tuple_list_hook, ext_hook=decode_ext) diff --git a/src/blosc2/remote_batch.py b/src/blosc2/remote_batch.py index df2539be9..a560fffbb 100644 --- a/src/blosc2/remote_batch.py +++ b/src/blosc2/remote_batch.py @@ -80,6 +80,11 @@ def _validated_lengths(self, column): def _get_batch(self, index): return _RemoteBatch(self, index, self._source.get_chunk(index)) + def _deserialize_msgpack_block(self, payload): + from blosc2.msgpack_utils import _safe_msgpack_unpackb + + return _safe_msgpack_unpackb(payload) + def _check_writable(self): raise ValueError("Cannot modify a remote BatchArray") diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 1693ff9fd..d5f91ec02 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -376,6 +376,86 @@ class Mixed: assert remote.traffic.requests == requests +def test_remote_ctable_batch_wrappers_and_dictionary(tmp_path): + @dataclasses.dataclass + class Rich: + data: bytes = blosc2.field(blosc2.vlbytes(nullable=True, batch_rows=2)) + tags: list[int] = blosc2.field( # noqa: RUF009 + blosc2.list(blosc2.int64(), nullable=True, batch_rows=2) + ) + props: dict = blosc2.field( # noqa: RUF009 + blosc2.struct({"count": blosc2.int32(), "name": blosc2.vlstring()}, nullable=True) + ) + payload: object = blosc2.field(blosc2.object(nullable=True, batch_rows=2)) + category: str = blosc2.field(blosc2.dictionary(nullable=True)) + + rows = [ + (b"one", [1, 2], {"count": 1, "name": "one"}, {"x": [1, 2]}, "a"), + (None, [], None, ("tuple", 2), None), + (b"", None, {"count": 3, "name": "東京"}, np.arange(3), "b"), + (b"four", [4], {"count": 4, "name": "four"}, {1, 2}, "a"), + ] + local = blosc2.CTable(Rich, rows, create_summary_index=False) + url = remote_table_url(tmp_path, local, "batch-wrappers") + with blosc2.RemoteCTable(url) as remote: + assert remote["data"][:] == [row[0] for row in rows] + assert remote["tags"][:] == [row[1] for row in rows] + assert remote["props"][:] == [row[2] for row in rows] + assert remote["payload"][:2] == [row[3] for row in rows[:2]] + np.testing.assert_array_equal(remote["payload"][2], np.arange(3)) + assert remote["payload"][3] == {1, 2} + assert remote["category"][:] == [row[4] for row in rows] + assert remote.where(remote["category"] == "a")["data"][:] == [b"one", b"four"] + + +def test_remote_ctable_rejects_unsafe_object_extension(tmp_path): + @dataclasses.dataclass + class Unsafe: + payload: object = blosc2.field(blosc2.object()) + + nested = blosc2.arange(3) + local = blosc2.CTable(Unsafe, [(nested,)], create_summary_index=False) + url = remote_table_url(tmp_path, local, "unsafe-object") + with blosc2.RemoteCTable(url) as remote: + with pytest.raises(ValueError, match="Unsafe remote MessagePack extension code 42"): + remote["payload"][0] + + +def test_remote_ctable_rejects_vl_list_storage(tmp_path): + @dataclasses.dataclass + class Lists: + values: list[int] = blosc2.field(blosc2.list(blosc2.int64(), storage="vl")) # noqa: RUF009 + + local = blosc2.CTable(Lists, [([1, 2],), ([],)], create_summary_index=False) + url = remote_table_url(tmp_path, local, "vl-list") + with blosc2.RemoteCTable(url) as remote: + with pytest.raises(NotImplementedError, match="storage='vl'"): + remote["values"][:] + + +def test_remote_ctable_missing_dictionary_companion_isolated(tmp_path): + @dataclasses.dataclass + class Mixed: + x: int + category: str = blosc2.field(blosc2.dictionary()) + + local = blosc2.CTable(Mixed, [(1, "a"), (2, "b")], create_summary_index=False) + remote_table_url(tmp_path, local, "missing-dictionary") + path = tmp_path / "missing-dictionary.b2z" + rewritten = tmp_path / "missing-dictionary-rewritten.b2z" + with zipfile.ZipFile(path) as source, zipfile.ZipFile(rewritten, "w") as target: + for info in source.infolist(): + if info.filename != "_cols/category_dict.b2b": + target.writestr(info, source.read(info)) + url = "memory://missing-dictionary.b2z" + fsspec.filesystem("memory").pipe(url, rewritten.read_bytes()) + + with blosc2.RemoteCTable(url) as remote: + np.testing.assert_array_equal(remote["x"][:], [1, 2]) + with pytest.raises(NotImplementedError, match="category_dict"): + remote["category"][:] + + def test_remote_store_returns_table_with_independent_lifetime(tmp_path): source = tmp_path / "tree.b2z" table = blosc2.CTable(Row, [(1, [1, 2], "one"), (2, [3, 4], "two")], create_summary_index=False) From b3dfaaf938e24003d3e0d4c1868dc2ba9b086ae5 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 18 Sep 2026 18:10:45 +0200 Subject: [PATCH 17/82] Harden remote batch reads and lifetime --- src/blosc2/b2z_source.py | 7 +- src/blosc2/ctable_remote_read.py | 22 +++- src/blosc2/ctable_storage.py | 7 +- src/blosc2/dictionary_column.py | 2 + src/blosc2/list_array.py | 17 +++- src/blosc2/remote_batch.py | 57 ++++++++++- src/blosc2/remote_store.py | 10 +- src/blosc2/scalar_array.py | 11 ++ tests/ctable/test_remote_ctable.py | 157 +++++++++++++++++++++++++++++ 9 files changed, 277 insertions(+), 13 deletions(-) diff --git a/src/blosc2/b2z_source.py b/src/blosc2/b2z_source.py index 0a9a7cdc3..6c9848e02 100644 --- a/src/blosc2/b2z_source.py +++ b/src/blosc2/b2z_source.py @@ -442,7 +442,7 @@ def read_range(self, offset, size): class B2ZBatchSource: """Internal byte-range source for one external BatchArray member.""" - def __init__(self, archive, dataset): + def __init__(self, archive, dataset, check_open=None): from blosc2.proxy_source import ( _chunk_extents, _read_frame_header, @@ -452,6 +452,7 @@ def __init__(self, archive, dataset): self.archive = archive self.dataset = dataset.strip("/") + self._check = check_open or (lambda: None) matches = [info for info in archive.members if info.filename == self.dataset + ".b2b"] if len(matches) != 1: raise NotImplementedError(f"Remote CTable batch member {self.dataset!r} is unavailable") @@ -466,6 +467,9 @@ def __init__(self, archive, dataset): self.meta = _read_frame_metalayers(raw, self.header) self.vlmeta = member_vlmeta(archive, self.info) self.offsets = _read_frame_offsets(self.read_range, self.header, head, len(raw)) + index_pos = self.header[1] + self.header[5] + if ((self.offsets < self.header[1]) | (self.offsets >= index_pos)).any(): + raise ValueError(f"Batch frame for {self.dataset!r} contains an invalid chunk offset") self.extents = _chunk_extents(self.offsets, self.header) archive._opening_ranges.clear() @@ -477,6 +481,7 @@ def read_range(self, offset, size): return self.archive._read_archive(self.member_offset + offset, size) def get_chunk(self, index): + self._check() offset = int(self.offsets[index]) if offset < 0: raise ValueError(f"Batch {index} of {self.dataset!r} has an unsupported special offset") diff --git a/src/blosc2/ctable_remote_read.py b/src/blosc2/ctable_remote_read.py index 131661d0b..46fbf607a 100644 --- a/src/blosc2/ctable_remote_read.py +++ b/src/blosc2/ctable_remote_read.py @@ -314,10 +314,18 @@ def fetch(tasks=tasks): return out -def column_values(table, names, positions, *, null_masks=None): +def column_values(table, names, positions, *, null_masks=None): # noqa: C901 """Read a bounded selection of stored columns, returning decoded values.""" from blosc2._utf8_array import _GATHER_GAP, UTF8Array - from blosc2.schema import timestamp + from blosc2.schema import ( + DictionarySpec, + ListSpec, + ObjectSpec, + StructSpec, + VLBytesSpec, + VLStringSpec, + timestamp, + ) storage = table._remote_read_storage() with storage._owner.lock: @@ -329,12 +337,15 @@ def column_values(table, names, positions, *, null_masks=None): return {name: table._fetch_col_at_positions_uncached(name, positions) for name in names} storage.open_columns(table, stored, table._cols.__getitem__) masks = { - name: table._null_mask(name) if table._schema.columns_by_name[name].spec.uses_mask else None + name: table._null_mask(name) + if getattr(table._schema.columns_by_name[name].spec, "uses_mask", False) + else None for name in stored } def reader(name): col = table._cols[name] + spec = table._schema.columns_by_name[name].spec if isinstance(col, UTF8Array): values = np.empty(len(positions), dtype=col.dtype) order = np.argsort(positions, kind="stable") @@ -353,9 +364,12 @@ def reader(name): a, b = offsets[pos - lo : pos - lo + 2] - first values[order[start + j]] = blob[a:b].decode("utf-8") start += len(cluster) + elif isinstance( + spec, (VLStringSpec, VLBytesSpec, StructSpec, ObjectSpec, ListSpec, DictionarySpec) + ): + values = col[positions] else: values = yield from _array_values(col, positions) - spec = table._schema.columns_by_name[name].spec if isinstance(spec, timestamp): values = values.astype(f"datetime64[{spec.unit}]") mask = masks[name] diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index 2e50db616..fb522da33 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -726,7 +726,9 @@ def _open_batch(self, name, spec, *, suffix=""): from blosc2.remote_batch import _RemoteBatchArray key = f"{_COLS_DIR}/{_column_name_to_relpath(name)}{suffix}" - backend = _RemoteBatchArray(self._owner.open_ctable_batch(self._full_key(key)), name) + backend = _RemoteBatchArray( + self._owner.open_ctable_batch(self._full_key(key)), name, self._check_open + ) if not isinstance(spec, ListSpec): _validate_role_metadata(backend, spec) return backend @@ -1645,7 +1647,8 @@ def create_varlen_scalar_column( return UTF8Array(spec, offsets, data) urlpath = self._list_col_path(name) os.makedirs(os.path.dirname(urlpath), exist_ok=True) - return _make_persistent_backend(spec, urlpath, "w", cparams=cparams, dparams=dparams) + backend = _make_persistent_backend(spec, urlpath, "w", cparams=cparams, dparams=dparams) + return _ScalarVarLenArray(spec, backend) def open_varlen_scalar_column(self, name: str, spec) -> _ScalarVarLenArray: if isinstance(spec, UTF8Spec): diff --git a/src/blosc2/dictionary_column.py b/src/blosc2/dictionary_column.py index 375b87e85..efed8aec4 100644 --- a/src/blosc2/dictionary_column.py +++ b/src/blosc2/dictionary_column.py @@ -286,6 +286,7 @@ def _equality_mask(self, other, *, invert: bool): def __setitem__(self, key, value) -> None: """Encode *value* (str/None or list thereof) and write the code(s).""" + self._dict_store._backend._check_writable() if isinstance(key, (int, np.integer)): self._codes[int(key)] = np.int32(self.encode(value)) elif isinstance(key, slice): @@ -302,6 +303,7 @@ def __setitem__(self, key, value) -> None: def resize(self, shape: tuple) -> None: """Resize the underlying codes NDArray (delegates to the NDArray).""" + self._dict_store._backend._check_writable() self._codes.resize(shape) # ------------------------------------------------------------------ diff --git a/src/blosc2/list_array.py b/src/blosc2/list_array.py index 7ad4d19fa..5b92ab784 100644 --- a/src/blosc2/list_array.py +++ b/src/blosc2/list_array.py @@ -379,6 +379,7 @@ def _flush_full_batches(self) -> None: def append(self, value: Any) -> int: """Append one list cell and return the new number of rows.""" + self._backend._check_writable() cell = coerce_list_cell(self.spec, value) if self.spec.storage == "vl": self._backend.append(cell) @@ -394,6 +395,7 @@ def extend(self, values: Iterable[Any], *, validate: bool = True) -> None: Set ``validate=False`` only for trusted values that already match this array's schema. """ + self._backend._check_writable() if validate: cells = [coerce_list_cell(self.spec, v) for v in values] else: @@ -417,6 +419,7 @@ def extend_arrow(self, arrow_array) -> None: This requires batch storage with ``serializer='arrow'`` and is intended for trusted Arrow/Parquet import paths. """ + self._backend._check_writable() pa = _require_pyarrow() if isinstance(arrow_array, pa.ChunkedArray): chunks = arrow_array.chunks @@ -441,6 +444,7 @@ def flush(self) -> None: if self.spec.storage != "batch": return if self._pending_cells: + self._backend._check_writable() batch = list(self._pending_cells) self._backend.append(batch) self._persisted_row_count += len(batch) @@ -514,7 +518,11 @@ def _get_many(self, indices: list[int]) -> list[Any]: # For small selections from block-addressable batches, scalar access is # much cheaper than materializing the full containing batch. This is # common for filtered column previews and small logical slices. - if getattr(self._backend, "items_per_block", None) is not None and len(indices) <= 1024: + if ( + not getattr(self._backend, "_remote", False) + and getattr(self._backend, "items_per_block", None) is not None + and len(indices) <= 1024 + ): return [self[index] for index in indices] if len(indices) <= 1: return self._get_many_grouped(indices) @@ -531,6 +539,9 @@ def _get_many(self, indices: list[int]) -> list[Any]: def __getitem__(self, index: int | slice | list[int] | tuple[int, ...] | np.ndarray) -> Any: """Return one cell or a list of cells selected by index, slice, or mask.""" + check = getattr(self._backend, "_check_open", None) + if check is not None: + check() if isinstance(index, slice): indices = list(range(*index.indices(len(self)))) return self._get_many(indices) @@ -554,6 +565,7 @@ def __getitem__(self, index: int | slice | list[int] | tuple[int, ...] | np.ndar def __setitem__(self, index: int, value: Any) -> None: """Replace one list cell.""" + self._backend._check_writable() cell = coerce_list_cell(self.spec, value) index = self._normalize_index(index) if self.spec.storage == "vl": @@ -571,6 +583,9 @@ def __setitem__(self, index: int, value: Any) -> None: def __len__(self) -> int: """Return the number of rows.""" + check = getattr(self._backend, "_check_open", None) + if check is not None: + check() if self.spec.storage == "vl": return len(self._backend) return self._persisted_row_count + len(self._pending_cells) diff --git a/src/blosc2/remote_batch.py b/src/blosc2/remote_batch.py index a560fffbb..6a3087087 100644 --- a/src/blosc2/remote_batch.py +++ b/src/blosc2/remote_batch.py @@ -16,7 +16,16 @@ class _RemoteBatch(Batch): + def __getitem__(self, index): + self._parent._check_open() + return super().__getitem__(index) + + def __len__(self): + self._parent._check_open() + return super().__len__() + def _payloads(self): + self._parent._check_open() payloads = getattr(self, "_remote_payloads", None) if payloads is None: payloads = blosc2.blosc2_ext.vldecompress(self._lazybatch) @@ -45,7 +54,9 @@ def _get_block_item(self, block_index, item_index): class _RemoteBatchArray(BatchArray): """The BatchArray read surface needed by remote CTable wrappers.""" - def __init__(self, source, column): + def __init__(self, source, column, check_open=None): + self._remote = True + self._owner_check = check_open or (lambda: None) try: metadata = source.meta["batcharray"] except KeyError as exc: @@ -63,6 +74,10 @@ def __init__(self, source, column): self._item_prefix_sums = None self._validate_tag() + def _check_open(self): + self._owner_check() + self._source._check() + def _validated_lengths(self, column): metadata = self.schunk.vlmeta.get(_BATCHARRAY_VLMETA_KEY, {}) lengths = metadata.get("batch_lengths") @@ -78,6 +93,7 @@ def _validated_lengths(self, column): return lengths def _get_batch(self, index): + self._check_open() return _RemoteBatch(self, index, self._source.get_chunk(index)) def _deserialize_msgpack_block(self, payload): @@ -86,7 +102,37 @@ def _deserialize_msgpack_block(self, payload): return _safe_msgpack_unpackb(payload) def _check_writable(self): - raise ValueError("Cannot modify a remote BatchArray") + self._check_open() + raise ValueError("Remote CTable batch columns are read-only") + + def __len__(self): + self._check_open() + return super().__len__() + + @property + def meta(self): + self._check_open() + return super().meta + + @property + def vlmeta(self): + self._check_open() + return super().vlmeta + + @property + def nbytes(self): + self._check_open() + return super().nbytes + + @property + def cbytes(self): + self._check_open() + return super().cbytes + + @property + def cratio(self): + self._check_open() + return super().cratio class _RemoteBatchCache: @@ -116,6 +162,7 @@ def _file(self, index): return self._path / f"{index}.chunk" def get_chunk(self, index): + self._source._check() if index in self._cache_sizes: chunk = self._file(index).read_bytes() if self._path is not None else self._memory[index] else: @@ -158,9 +205,11 @@ def _trim_cache(self, target_bytes, *, max_chunks=None): class _RemoteBatchSChunk: def __init__(self, source): + from blosc2.remote_array import RemoteMetadataMapping + self._source = source - self.meta = source.meta - self.vlmeta = source.vlmeta + self.meta = RemoteMetadataMapping(source.meta) + self.vlmeta = RemoteMetadataMapping(source.vlmeta) self.mode = "r" self.mmap_mode = None self.nchunks = len(source.offsets) diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index d2e90c7d7..284a2189d 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -575,9 +575,17 @@ def open_ctable_batch(self, full): self._validate(full) from blosc2.b2z_source import B2ZBatchSource + generation = self.generation + + def check_open(): + if self._closed: + raise RuntimeError("RemoteCTable handle is closed") + if self.generation != generation: + raise RuntimeError("RemoteCTable handle is stale; look it up again after refresh") + self.archive.capture_metadata = True try: - source = B2ZBatchSource(self.archive, full) + source = B2ZBatchSource(self.archive, full, check_open) finally: self.archive.capture_metadata = False self.archive._opening_ranges.clear() diff --git a/src/blosc2/scalar_array.py b/src/blosc2/scalar_array.py index c9d5c0bd5..cbfbbc3e2 100644 --- a/src/blosc2/scalar_array.py +++ b/src/blosc2/scalar_array.py @@ -239,11 +239,13 @@ def _flush_full_batches(self) -> None: def append(self, value: Any) -> None: """Append one scalar row.""" + self._backend._check_writable() self._pending.append(self._coerce(value)) self._flush_full_batches() def extend(self, values: Iterable[Any]) -> None: """Append many scalar rows.""" + self._backend._check_writable() for v in values: self._pending.append(self._coerce(v)) if len(self._pending) >= self._batch_rows: @@ -252,6 +254,7 @@ def extend(self, values: Iterable[Any]) -> None: def flush(self) -> None: """Flush any remaining pending rows to the backend as one batch.""" if self._pending: + self._backend._check_writable() batch = list(self._pending) self._backend.append(batch) self._persisted_row_count += len(batch) @@ -265,6 +268,7 @@ def set_all(self, values: Iterable[Any]) -> None: :meth:`__setitem__` would rewrite a whole batch per row. Mirrors ``UTF8Array.set_all`` so callers can treat both the same way. """ + self._backend._check_writable() coerced = [self._coerce(v) for v in values] if len(coerced) != len(self): raise ValueError(f"set_all() expects {len(self)} values, got {len(coerced)}.") @@ -279,6 +283,9 @@ def set_all(self, values: Iterable[Any]) -> None: # ------------------------------------------------------------------ def __len__(self) -> int: + check = getattr(self._backend, "_check_open", None) + if check is not None: + check() return self._persisted_row_count + len(self._pending) def __iter__(self) -> Iterator[Any]: @@ -311,6 +318,9 @@ def __ne__(self, other): return np.asarray(self[:], dtype=object) != other def __getitem__(self, index: int | slice | list | tuple) -> Any | list[Any]: + check = getattr(self._backend, "_check_open", None) + if check is not None: + check() if isinstance(index, int): n = len(self) if index < 0: @@ -346,6 +356,7 @@ def __getitem__(self, index: int | slice | list | tuple) -> Any | list[Any]: raise TypeError(f"_ScalarVarLenArray indices must be int, slice, or array; got {type(index)!r}") def __setitem__(self, index: int, value: Any) -> None: + self._backend._check_writable() value = self._coerce(value) n = len(self) if index < 0: diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index d5f91ec02..84caebeed 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -408,6 +408,22 @@ class Rich: assert remote.where(remote["category"] == "a")["data"][:] == [b"one", b"four"] +def test_remote_ctable_arrow_list(tmp_path): + pytest.importorskip("pyarrow") + + @dataclasses.dataclass + class Lists: + values: list[int] = blosc2.field( # noqa: RUF009 + blosc2.list(blosc2.int64(), serializer="arrow", nullable=True, batch_rows=2) + ) + + rows = [([1, 2],), (None,), ([],), ([3],)] + local = blosc2.CTable(Lists, rows, create_summary_index=False) + url = remote_table_url(tmp_path, local, "arrow-list") + with blosc2.RemoteCTable(url) as remote: + assert remote["values"][:] == [row[0] for row in rows] + + def test_remote_ctable_rejects_unsafe_object_extension(tmp_path): @dataclasses.dataclass class Unsafe: @@ -456,6 +472,147 @@ class Mixed: remote["category"][:] +def test_remote_batch_reads_mutation_lifetime_and_copy(tmp_path, monkeypatch): + @dataclasses.dataclass + class Mixed: + text: str = blosc2.field(blosc2.vlstring(batch_rows=3)) + tags: list[int] = blosc2.field(blosc2.list(blosc2.int64(), batch_rows=3)) # noqa: RUF009 + category: str = blosc2.field(blosc2.dictionary()) + + rows = [(f"text {i}", [i, i + 1], "even" if i % 2 == 0 else "odd") for i in range(8)] + local = blosc2.CTable(Mixed, rows, create_summary_index=False) + url = remote_table_url(tmp_path, local, "batch-read-lifetime") + fs = fsspec.filesystem("memory") + reads = [] + original = type(fs).cat_file + + def counted(self, path, start=None, end=None, **kwargs): + reads.append((start, end)) + return original(self, path, start=start, end=end, **kwargs) + + monkeypatch.setattr(type(fs), "cat_file", counted) + remote = blosc2.RemoteCTable(url, cache_policy=blosc2.CachePolicy.NONE) + text, tags, category = (remote[name].raw for name in ("text", "tags", "category")) + reads.clear() + assert tags[[0, 2, 1]] == [[0, 1], [2, 3], [1, 2]] + assert len(reads) == 1 + for item in (slice(None, None, 2), slice(None, None, -1), [7, 0, 7, 3]): + assert text[item] == local["text"].raw[item] + assert tags[item] == local["tags"].raw[item] + assert list(text) == [row[0] for row in rows] + assert list(tags) == [row[1] for row in rows] + assert remote["category"][:] == [row[2] for row in rows] + assert text.nbytes > 0 + assert text.cbytes > 0 + assert "text 0" in str(remote[:2]) + + for mutate in ( + lambda: text.append("new"), + lambda: text.extend(["new"]), + lambda: text.set_all(["new"] * len(text)), + lambda: text.__setitem__(0, "new"), + lambda: tags.append([9]), + lambda: tags.extend([[9]]), + lambda: tags.__setitem__(0, [9]), + lambda: category.__setitem__(0, "new"), + ): + with pytest.raises(ValueError, match="read-only"): + mutate() + assert text._pending == [] + assert tags._pending_cells == [] + assert category._value_to_code is not None + assert "new" not in category._value_to_code + with pytest.raises(TypeError): + text._backend.vlmeta["new"] = 1 + + detached = remote.copy() + cached_batch = tags._backend[0] + assert cached_batch[:] == [[0, 1], [1, 2], [2, 3]] + remote.close() + with pytest.raises(RuntimeError, match="closed"): + text[[]] + with pytest.raises(RuntimeError, match="closed"): + tags[:1] + with pytest.raises(RuntimeError, match="closed"): + cached_batch[:] + detached["text"][0] = "changed" + detached["tags"][0] = [99] + detached["category"][0] = "changed" + assert detached[0].text == "changed" + assert detached[0].tags == [99] + assert detached[0].category == "changed" + + +def test_remote_batch_refresh_invalidates_raw_wrapper(tmp_path): + @dataclasses.dataclass + class Text: + value: str = blosc2.field(blosc2.vlstring(batch_rows=2)) + + url = remote_table_url( + tmp_path, blosc2.CTable(Text, [("old",), ("values",)], create_summary_index=False), "batch-refresh" + ) + with blosc2.RemoteCTable(url) as remote: + raw = remote["value"].raw + assert raw[:] == ["old", "values"] + replacement = blosc2.CTable(Text, [("new",)], create_summary_index=False) + path = tmp_path / "replacement.b2z" + replacement.to_b2z(path) + fsspec.filesystem("memory").pipe(url, path.read_bytes()) + remote.refresh() + assert remote["value"][:] == ["new"] + with pytest.raises(RuntimeError, match=r"stale|closed"): + raw[:] + + +def test_remote_store_batch_table_outlives_parent(tmp_path): + @dataclasses.dataclass + class Text: + value: str = blosc2.field(blosc2.vlstring(batch_rows=2)) + + source = tmp_path / "batch-tree.b2z" + with blosc2.TreeStore(source, mode="w", threshold=0) as tree: + tree["table"] = blosc2.CTable(Text, [("a",), ("b",)], create_summary_index=False) + url = "memory://batch-tree.b2z" + fsspec.filesystem("memory").pipe(url, source.read_bytes()) + store = blosc2.RemoteStore(url) + remote = store["table"] + store.close() + assert remote["value"][:] == ["a", "b"] + remote.close() + + +@pytest.mark.parametrize("lengths", [None, [-1]]) +def test_remote_batch_invalid_lengths_are_column_local(tmp_path, lengths): + @dataclasses.dataclass + class Mixed: + x: int + text: str = blosc2.field(blosc2.vlstring(batch_rows=2)) + + local = blosc2.CTable(Mixed, [(1, "a"), (2, "b")], create_summary_index=False) + remote_table_url(tmp_path, local, "invalid-lengths") + source_path = tmp_path / "invalid-lengths.b2z" + broken_path = tmp_path / "invalid-lengths-broken.b2z" + with zipfile.ZipFile(source_path) as source: + frame = source.read("_cols/text.b2b") + schunk = blosc2.schunk_from_cframe(frame, copy=True) + if lengths is None: + del schunk.vlmeta["_batch_array_metadata"] + else: + schunk.vlmeta["_batch_array_metadata"] = {"batch_lengths": lengths} + replacement = schunk.to_cframe() + with zipfile.ZipFile(broken_path, "w") as target: + for info in source.infolist(): + target.writestr( + info, replacement if info.filename == "_cols/text.b2b" else source.read(info) + ) + url = "memory://invalid-lengths.b2z" + fsspec.filesystem("memory").pipe(url, broken_path.read_bytes()) + with blosc2.RemoteCTable(url) as remote: + np.testing.assert_array_equal(remote["x"][:], [1, 2]) + with pytest.raises(ValueError, match=r"column 'text'.*batch length"): + remote["text"][:] + + def test_remote_store_returns_table_with_independent_lifetime(tmp_path): source = tmp_path / "tree.b2z" table = blosc2.CTable(Row, [(1, [1, 2], "one"), (2, [3, 4], "two")], create_summary_index=False) From 4619082c8822a3e0d70a2c17c18952c7d6ff48fc Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 18 Sep 2026 18:17:00 +0200 Subject: [PATCH 18/82] Document remote batch-backed columns --- doc/guides/remote_tables.md | 30 ++++++++++++--- doc/reference/remotectable.rst | 6 ++- examples/ctable/remote_handling.py | 62 +++++++++++++++++++++++++++++- plans/remote-ctable.md | 21 +++++++++- src/blosc2/remote_ctable.py | 7 +++- tests/ctable/test_remote_ctable.py | 24 ++++++++++++ 6 files changed, 138 insertions(+), 12 deletions(-) diff --git a/doc/guides/remote_tables.md b/doc/guides/remote_tables.md index ccfd70c7a..e64928a6f 100644 --- a/doc/guides/remote_tables.md +++ b/doc/guides/remote_tables.md @@ -1,8 +1,9 @@ # Working with Remote Tables `RemoteCTable` opens a read-only CTable in an immutable remote `.b2z` archive. -Fixed-width and `blosc2.utf8()` columns are fetched on demand, including their -null masks. A table inside a hierarchy can also be opened through `RemoteStore`. +Fixed-width, `blosc2.utf8()`, batch-backed variable-length, list, struct/object, +and dictionary columns are fetched on demand, including their null masks. A table +inside a hierarchy can also be opened through `RemoteStore`. `blosc2.open()` dispatches local table archives to `CTable` and remote table archives to `RemoteCTable`. Remote `.b2z` groups return `RemoteStore` by default; @@ -29,6 +30,18 @@ metadata without scanning strings. Small-member metadata prefetch may also fetch some payload. Repeated reads can reuse cached blocks; filtering scans the required columns because persisted indexes are not used remotely. +Batch-backed columns transfer one whole compressed batch per required batch, then +decode it locally. A small row selection can therefore fetch and allocate a large +batch. The compressed payload shares the remote owner's cache limit, but decoded +Python lists, strings, objects, dictionary maps, result arrays, and decoder scratch +space do not. Nonempty batch columns require a valid persisted batch-length catalog; +missing, negative, or inconsistent lengths raise an error naming the column. + +Dictionary codes use the usual selective fixed-width reads. The first decoded +value or string predicate loads the complete vocabulary and builds Python lookup +maps, costing O(dictionary cardinality) transfer and decoded memory. This is +separate from the code-array read and remains cached by the column wrapper. + Multi-column metadata inspection and row materialization overlap independent requests by default, up to eight at once. This includes row iteration, display, and batched Arrow/pandas export; single-column access remains lazy. UTF-8 byte @@ -80,11 +93,16 @@ local = table.materialize(urlpath="complete-local.b2z") table.to_b2d("complete-local.b2d") ``` -Saving a reference does not fetch missing table data. Remote writes and -batch-backed `vlstring`/lists/objects and dictionary columns remain unsupported. +Saving a reference does not fetch missing table data. Remote writes and persisted +indexes remain unsupported. Lists configured with `storage="vl"` are rejected; +use the default batch storage. Remote MessagePack object values support passive +data forms, while embedded Blosc2 containers and serialized references are +rejected instead of being reconstructed from untrusted remote data. -See `examples/ctable/remote_handling.py` for a batched archive writer with a nullable -multilingual UTF-8 column, plus sample row and string-slice traffic measurements. +See `examples/ctable/remote_handling.py` for a batched archive writer with nullable +multilingual UTF-8 and variable-length strings, a batch-backed list, and a +dictionary. It reports ordinary batch cold/warm reads and dictionary code/vocabulary +costs separately. ## Refresh a remote table diff --git a/doc/reference/remotectable.rst b/doc/reference/remotectable.rst index ec325caac..50c7a870e 100644 --- a/doc/reference/remotectable.rst +++ b/doc/reference/remotectable.rst @@ -4,10 +4,14 @@ RemoteCTable ============ ``RemoteCTable`` is a read-only :class:`blosc2.CTable` backed by a remote B2Z -archive. Fixed-width, shaped, nullable, and UTF-8 columns are fetched on demand. +archive. Fixed-width, shaped, nullable, UTF-8, batch-backed variable-length, +batch-backed list, struct/object, and dictionary columns are fetched on demand. Standalone tables can be opened directly; tables inside a hierarchy can be selected with ``dataset=`` or through :class:`blosc2.RemoteStore`. +Batch-backed reads transfer and decode whole compressed batches. Dictionary +codes remain selective, while the full vocabulary is loaded on first use. + Saving and materializing have different meanings: .. code-block:: python diff --git a/examples/ctable/remote_handling.py b/examples/ctable/remote_handling.py index 5927d69d5..8b61adb4b 100644 --- a/examples/ctable/remote_handling.py +++ b/examples/ctable/remote_handling.py @@ -6,7 +6,7 @@ # SPDX-License-Identifier: BSD-3-Clause ####################################################################### -"""Create or access a mask-nullable CTable with fixed-width and UTF-8 columns locally or remotely.""" +"""Create or access a CTable with fixed-width, UTF-8, batch-backed and dictionary columns.""" import argparse import pprint @@ -32,6 +32,11 @@ class Reading: status: str = blosc2.field(blosc2.string(max_length=8, null_storage="mask")) active: bool = blosc2.field(blosc2.bool()) note: str = blosc2.field(blosc2.utf8(null_storage="mask")) + message: str = blosc2.field(blosc2.vlstring(nullable=True, batch_rows=4096)) + tags: list[int] = blosc2.field( # noqa: RUF009 + blosc2.list(blosc2.int16(), nullable=True, batch_rows=4096) + ) + region: str = blosc2.field(blosc2.dictionary(nullable=True)) def make_notes(ids): @@ -81,6 +86,10 @@ def write_table(args) -> None: temperature[ids % 17 == 0] = None humidity[ids % 29 == 0] = None status[ids % 41 == 0] = None + messages = [None if i % 47 == 0 else f"sensor {i}: café 東京" for i in ids] + tags = [None if i % 53 == 0 else [int(i % 7), int(i % 11)] for i in ids] + regions = np.array(["north", "south", "east", "west"], dtype=object)[ids % 4] + regions[ids % 59 == 0] = None table.extend( { "id": ids, @@ -90,6 +99,9 @@ def write_table(args) -> None: "status": status, "active": ids % 5 != 0, "note": make_notes(ids), + "message": messages, + "tags": tags, + "region": regions, }, validate=False, ) @@ -149,6 +161,7 @@ def access_table(args) -> None: metadata_time = time.perf_counter() - started metadata_bytes = table.traffic.nbytes if remote else 0 metadata_requests = table.traffic.requests if remote else 0 + extra_time = 0.0 print(f"\n[Format: Blosc2 B2Z ({type(table).__name__})]") for name, value in metadata.items(): @@ -157,6 +170,51 @@ def access_table(args) -> None: rendered = "{\n " + rendered[1:] print(f"{name:<13}: {rendered}") + probe = slice(0, min(5, table.nrows)) + if "message" in table.col_names: + print("\nBatch-backed message slice (whole compressed batches are the transfer unit):") + for label in ("cold", "warm"): + before_bytes = table.traffic.nbytes if remote else 0 + before_requests = table.traffic.requests if remote else 0 + started = time.perf_counter() + messages = table["message"][probe] + elapsed = time.perf_counter() - started + extra_time += elapsed + transferred = table.traffic.nbytes - before_bytes if remote else 0 + requests = table.traffic.requests - before_requests if remote else 0 + print( + f" - {label:<4} batch read: {elapsed * 1000:7.1f} ms " + f"({requests} requests, {transferred / 1024:8.2f} KB transferred)" + ) + print(f" {messages}") + + if "region" in table.col_names: + dictionary = table["region"].raw + print("\nDictionary costs (codes first, then full vocabulary on first decode):") + before_bytes = table.traffic.nbytes if remote else 0 + before_requests = table.traffic.requests if remote else 0 + started = time.perf_counter() + _ = dictionary.codes[probe] + elapsed = time.perf_counter() - started + extra_time += elapsed + print( + f" - code read : {elapsed * 1000:7.1f} ms " + f"({table.traffic.requests - before_requests if remote else 0} requests, " + f"{(table.traffic.nbytes - before_bytes if remote else 0) / 1024:8.2f} KB transferred)" + ) + before_bytes = table.traffic.nbytes if remote else 0 + before_requests = table.traffic.requests if remote else 0 + started = time.perf_counter() + regions = table["region"][probe] + elapsed = time.perf_counter() - started + extra_time += elapsed + print( + f" - first decode : {elapsed * 1000:7.1f} ms " + f"({table.traffic.requests - before_requests if remote else 0} requests, " + f"{(table.traffic.nbytes - before_bytes if remote else 0) / 1024:8.2f} KB transferred)" + ) + print(f" {regions}") + sample_start = max(0, table.nrows // 2 - 2) sample_stop = min(sample_start + 5, table.nrows) sample_slice = slice(sample_start, sample_stop) @@ -189,7 +247,7 @@ def access_table(args) -> None: f"({second_requests} requests, {second_bytes / 1024:8.2f} KB transferred){cache_hit}" ) # Sum operation wall times (including decoding/cache work), not printing. - total_time = metadata_time + first_time + second_time + total_time = metadata_time + first_time + second_time + extra_time if "note" in table.col_names and table["note"].is_utf8: start = max(0, table.nrows - 5) print(f"\nUTF-8 note slice [{start}:{table.nrows}] (may overlap warmed blocks in small tables):") diff --git a/plans/remote-ctable.md b/plans/remote-ctable.md index 7ceb653e4..ccf4a30f2 100644 --- a/plans/remote-ctable.md +++ b/plans/remote-ctable.md @@ -2,7 +2,9 @@ Status: initial fixed-width, read-only implementation completed on 2026-09-17; UTF-8 support was added in the v2 extension (see `remote-ctable-v2.md`); -batch-backed columns, persisted indexes and portable references remain follow-ups. +batch-backed columns were added in the batch extension (see +`remote-ctable-batches.md`); persisted indexes and portable references remain +follow-ups. ## Objective and architecture @@ -251,11 +253,26 @@ and correct results, not a claim that scan queries avoid reading their operands. sample rows were read on demand; the repeated sample read issued zero requests and transferred zero bytes from the warm memory cache. +### Results recorded for the batch-backed extension + +- The default suite passed with 10,315 tests and 36 skips. The focused remote + table suite passed with 90 tests, and Ruff passed for all changed Python files. +- An instrumented `memory://` experiment used a 55,026,931-byte archive with + 100,000 rows, 1,024 rows per variable-length batch, and a 20,000-value + dictionary. A distant five-row read used 5 requests and 575,310 bytes, retained + 558,335 compressed bytes, and repeated with 0 requests and 0 bytes. +- The dictionary code slice used 2 requests and 4,612 bytes. First decode then + loaded the vocabulary with 10 requests and 95,494 bytes. The resulting Python + strings and lookup maps occupied approximately 4,028,168 bytes by + `sys.getsizeof`, outside the compressed transport cache budget. Metadata opening + used 2 requests and 9,297 bytes. + ## Follow-ups, separately scoped 1. UTF-8 columns through remote offsets and bytes, with null/query/size reporting: implemented in the v2 extension described in `remote-ctable-v2.md`. -2. Remote batch reads for lists, variable-length values and dictionary stores. +2. Remote batch reads for lists, variable-length values and dictionary stores: + implemented in the extension described in `remote-ctable-batches.md`. 3. Persisted indexes through a remote-aware sidecar resolver, starting with SUMMARY indexes and measuring query transfer savings. 4. Portable RemoteCTable references, RemoteStore artifact inclusion and sparse diff --git a/src/blosc2/remote_ctable.py b/src/blosc2/remote_ctable.py index 6e5a385e0..fc3b29ba8 100644 --- a/src/blosc2/remote_ctable.py +++ b/src/blosc2/remote_ctable.py @@ -44,7 +44,12 @@ def set(self, value): class RemoteCTable(RemoteObject, CTable): - """A read-only CTable whose fixed-width and UTF-8 columns are fetched on demand. + """A read-only CTable whose columns are fetched on demand. + + Supported columns include fixed-width, UTF-8, batch-backed variable-length, + batch-backed list, struct/object and dictionary columns. Batch reads transfer + one whole compressed batch; dictionary decoding loads the full vocabulary on + first use. Independent column requests overlap by default. ``max_concurrency`` defaults to 8; use 1 for serial reads. ``metadata_buffer_bytes`` (8 MiB) and diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 84caebeed..2d547d08f 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -82,6 +82,30 @@ def test_remote_example_cache_dir(tmp_path, capsys, monkeypatch): assert "(0 requests," in output.split("Total network :")[1] +def test_remote_example_batch_columns(tmp_path, capsys): + import runpy + from pathlib import Path + from types import SimpleNamespace + + script = Path(__file__).resolve().parents[2] / "examples/ctable/remote_handling.py" + example = runpy.run_path(str(script)) + path = tmp_path / "example-batches.b2z" + example["write_table"](SimpleNamespace(write=path, rows=100, batch_size=37, overwrite=False)) + with blosc2.open(path) as local: + assert local["message"][47] is None + assert local["tags"][53] is None + assert local["region"][59] is None + + url = f"memory://{tmp_path.name}-example-batches.b2z" + fsspec.filesystem("memory").pipe(url, path.read_bytes()) + capsys.readouterr() + example["access_table"](SimpleNamespace(url=url, cache_dir=None)) + output = capsys.readouterr().out + assert "cold batch read" in output + assert "warm batch read" in output + assert "Dictionary costs (codes first, then full vocabulary on first decode)" in output + + def test_disk_cache_metadata_key_order(tmp_path, monkeypatch): local = blosc2.CTable(dataclasses.make_dataclass("Sample", [("x", int)]), [(i,) for i in range(20)]) url = remote_table_url(tmp_path, local) From a9c84508f1e586d333aad1902b38f0864b58fc8e Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 18 Sep 2026 18:26:54 +0200 Subject: [PATCH 19/82] Mark remote batch plan complete --- plans/remote-ctable-batches.md | 251 +++++++++++++++++++++++++++++++++ 1 file changed, 251 insertions(+) create mode 100644 plans/remote-ctable-batches.md diff --git a/plans/remote-ctable-batches.md b/plans/remote-ctable-batches.md new file mode 100644 index 000000000..baeafe71b --- /dev/null +++ b/plans/remote-ctable-batches.md @@ -0,0 +1,251 @@ +# RemoteCTable: batch-backed columns + +Status: implemented and verified on 2026-09-18 as follow-up 2 in +`remote-ctable.md`. + +## Objective and scope + +Extend the existing read-only RemoteCTable to read external `.b2b` members on +demand. Reuse BatchArray serialization and row lookup, ListArray, +_ScalarVarLenArray and DictionaryColumn rather than introducing remote versions +of their table algorithms. + +The intended column coverage is: + +- Batch-backed variable-length strings and bytes. +- Batch-backed lists, with existing MessagePack and optional Arrow serializers. +- Struct/object scalar values supported by the existing serializer, subject to + the remote decoding safety audit below. +- Dictionary columns: RemoteArray integer codes plus a batch-backed vocabulary. + +Keep existing schema, null, deleted-row and spare-capacity semantics. Unsupported +representations must not prevent schema inspection or reading supported siblings. +Errors must identify the affected column and limitation. + +No new public remote-container class, dependency, archive format or constructor +option is planned. Persisted indexes, portable references, remote writes and +variable-length block transport remain separately scoped. ListArray +`storage="vl"` uses ObjectArray rather than BatchArray and is outside this +extension; retain an explicit diagnostic for it. + +## Evidence and implementation boundaries + +- `ctable_storage.py`: RemoteTableStorage already supplies the storage boundary; + its list, non-UTF-8 variable-length and dictionary openers remain unsupported. +- `batch_array.py`: BatchArray handles batch metadata, item lookup and + MessagePack/Arrow decoding, but its reads currently use a local SChunk. +- `scalar_array.py`: _ScalarVarLenArray constructs its persisted row count from + batch lengths; missing lengths can currently cause every batch to be decoded. +- `list_array.py`: ListArray has batch-backed reads and a separate ObjectArray + representation. Its cached batches and local copy optimizations need auditing. +- `dictionary_column.py`: DictionaryColumn lazily loads the full vocabulary to + construct value-to-code and code-to-value caches on first use. +- `b2z_source.py` and `proxy_source.py`: B2Z member windows and frame-parsing + helpers provide reusable byte-range machinery. The existing NDArray block + parser explicitly excludes variable-length blocks. +- `ctable_remote_read.py`: metadata and row transport already have bounded + scheduling. Batch integration must reuse that mechanism where applicable. + +Trace the callers of each shared helper before changing it. Preserve local +optimized paths and avoid a broad source/protocol refactor. No C/Cython changes +are expected unless the initial transport experiment demonstrates a need. + +## Storage, transfer and caching decisions + +Use a whole compressed batch (one BatchArray chunk) as the initial payload +transfer unit. Locate it using frame metadata and chunk offsets, fetch only that +range, and decode locally using existing Blosc2 and BatchArray machinery. Do not +route variable-length blocks through the fixed-width NDArray block parser. + +A small row selection may require whole batches. A single unusually large batch +therefore requires a correspondingly large read and decode allocation. Transport +buffer limits bound scheduling; they do not promise to subdivide a batch or bound +its decoded Python representation. Document this ceiling and mark the deliberate +whole-batch simplification with a `ponytail:` comment at its implementation. + +Require valid persisted batch lengths for nonempty remote batch columns in this +first implementation. Validate their count, nonnegative integer values and +prefix-sum bounds before row lookup. Reject missing or invalid lengths with a +column-specific diagnostic instead of silently decoding the column during open. +Empty columns remain valid without a nonempty lengths catalog. Schema inspection +and supported sibling reads must still work for rejected columns. + +Reuse archive identity, member paths and the existing storage-options fingerprint +for cache identity. NONE, MEMORY and DISK must work. Retained compressed batch +payloads must share the owner's aggregate budget with NDArray leaves, traffic +accounting and eviction machinery; do not add an independent per-column budget. +Batch counts, lengths and frame indexes are metadata and should be accounted for +consistently with existing remote metadata. Do not persist credentials. + +Group requested rows by batch where needed so NONE does not repeatedly fetch the +same batch within one operation. Audit decoded wrapper caches separately from the +compressed transport cache; do not imply that the latter bounds all Python memory. +Retain the existing bounded small-member prefetch allowance. + +## Dictionary behavior + +Open dictionary codes through RemoteArray and the vocabulary through the same +batch backend used by variable-length string columns. Reuse DictionaryColumn's +existing lookup and predicate behavior. + +Preserve full-vocabulary loading on first use for this extension. Opening the +table or inspecting its schema must remain lazy, but decoding rows or resolving +query literals may read the complete vocabulary. This costs O(dictionary +cardinality) transfer and decoded memory, not O(table row count). The lookup maps +are decoded Python state outside the compressed transport cache budget. + +Document that cost and measure it separately from code-array reads. Selective +vocabulary lookup is deferred until measurements establish a need. Persisted +indexes remain disabled, including dictionary indexes. + +## Implementation sequence + +1. **Prove one remote batch read.** Generate a nullable `vlstring` table with + several batches, archive it and expose it through the existing instrumented + fsspec `memory://` filesystem. Read the `.b2b` member header, metadata and chunk + offsets using the existing B2Z window and frame helpers. Fetch one distant + compressed batch and decode it with existing machinery. Compare against the + local archive and record the ranges transferred. Use a member large enough + that bounded metadata prefetch cannot mask a whole-member download. + +2. **Connect a minimal internal batch reader.** Choose the smallest backend + boundary demonstrated by the experiment: reuse BatchArray's serialization, + batch lengths and item mapping, adapting only its local-SChunk assumptions + needed for reads. Support metadata, batch counts and lengths, compressed-batch + access, decoding and size reporting. Reuse owner leases, generation checks, + archive transport and frame helpers. Do not emulate unrelated writable SChunk + APIs or add a public RemoteBatchArray as part of this task. + +3. **Deliver the first end-to-end column.** Add remote variable-length opening + through _ScalarVarLenArray and verify one nullable `vlstring` column with NONE + caching. A distant scalar read must match local behavior without downloading + the whole column or archive. Include batch boundaries and null/empty-string + distinction. This is the first implementation milestone and the checkpoint + for confirming the backend design before expanding type coverage. + +4. **Complete shared caching and transport integration.** Connect compressed + batch retention to the existing owner cache coordinator for MEMORY and DISK, + including eviction and reopen. Extend `.b2nd`-specific member selection in + remote table metadata opening to include required `.b2b` members and dictionary + companions. Reuse the existing bounded transport scheduler; keep parsing, + decoding and cache mutation on the owning thread. Check repeated reads and + mixed NDArray/batch pressure against the aggregate cache limit. + +5. **Enable the remaining wrappers.** Add variable-length bytes, batch-backed + lists, and supported struct/object scalars using their existing wrappers. + Preserve optional Arrow dependency behavior. Then add dictionary opening with + RemoteArray codes and the remote vocabulary store. Reuse existing storage + suffixes, schema reconstruction and role-metadata validation. Validate + companion existence and dictionary code layout; release every acquired handle + on partial-open failure. + +6. **Audit shared reads, mutation and lifetime.** Trace scalar, slice, strided, + reverse and fancy reads, iteration, filtered views, supported predicates, + null handling, size reporting, repr and local copy/save. Repair shared helpers + where they assume a native SChunk, local path or extension input. Reject + mutations before pending lists or dictionary caches change, including through + raw wrappers and metadata handles. Check lifetime even on empty reads and + cached results. Root-table close and RemoteStore refresh must invalidate + borrowed columns/views; parent-store close must leave an acquired table usable. + Detached materializations must be ordinary writable local tables. + +7. **Document and demonstrate the extension.** Update API support descriptions + and the remote handling example with a small representative set of batch-backed + columns, including a dictionary. Report ordinary batch cold/warm reads and + dictionary first-use costs separately. Document whole-batch granularity, + missing-length diagnostics, decoded-memory costs and remaining unsupported + representations. Update follow-up 2 in the original plan only after verification. + +## Validation and safe decoding + +Validate remote frame/member boundaries, offsets, metadata versions, serializer +names, batch lengths and required companions before using them. Preserve existing +rejections for encrypted, ZIP-compressed and unsupported embedded payloads. +Do not silently localize an entire archive to satisfy an unsupported read. + +Before enabling object values, audit `msgpack_utils.py` and its extension decoders. +They can reconstruct Blosc2 objects and serialized expressions, so ordinary +MessagePack decoding cannot simply be assumed to be passive for every value. +Remote reads must not execute arbitrary serialized callables, import code named +by remote metadata or unexpectedly resolve external references. Reuse safe data +decoding where possible; explicitly reject unsafe extension forms with a useful +diagnostic. Keep local decoding behavior unchanged. Record any resulting object +value limitations in the supported-type documentation and regression tests. + +## Verification and acceptance + +Use the `blosc2` conda environment for all Python, test and build commands. Extend +existing helpers in `tests/ctable/test_remote_ctable.py`, `tests/test_batch_array.py`, +`tests/test_b2z_source.py` and `tests/test_remote_store.py` as appropriate rather +than adding another transport test harness. + +- Compare against a local read-only CTable from the exact same archive. Cover + empty/nonempty columns, unequal and partial final batches, batch boundaries, + null versus empty values, multilingual strings, bytes, supported lists and + objects, dictionary nulls, deleted rows and spare capacity. +- Exercise scalar, contiguous, strided, reverse and fancy reads, iteration, + filtered views, mixed fixed-width/UTF-8/batch predicates, existing supported + computations and local materialization. Preserve local errors for unsupported + operations. Include nested column paths and RemoteStore-nested tables. +- Verify lazy opening and metadata-only size reporting without payload scans, + allowing bounded small-member prefetch. Missing batch-length metadata must fail + explicitly for nonempty columns without affecting supported siblings. +- Cover malformed metadata, invalid lengths/offsets, missing companions, + unsupported serializers/representations, unsafe object extension payloads and + partial-open cleanup. Exercise Arrow when installed and its existing missing + dependency diagnostic otherwise. +- Exercise NONE, MEMORY and DISK, mixed-leaf aggregate eviction, disk reopen and + source identity isolation. A small selection must transfer only the required + batches plus bounded metadata. A repeated cached selection that fits the budget + must need no new payload requests. NONE must avoid duplicate batch fetches + within a grouped read without retaining a persistent payload cache. +- Check read-only enforcement before pending-state changes, cached reads after + close/refresh, parent-store close, borrowed views and detached local exports. +- Reuse the deterministic HTTP range server to verify actual range transport. + Cloud access is optional manual validation, not an automated-test requirement. +- Record a larger cold/warm experiment with archive size, batch geometry, + requested rows, requests, transferred bytes and retained compressed payload. + Report dictionary vocabulary loading separately and identify decoded-memory + costs. No universal latency target is required. +- Run focused batch/list/scalar/dictionary, CTable, remote table/store and B2Z + tests, then the default suite. Run Ruff on changed Python files. Warnings remain + errors. + +Completion means supported batch-backed reads and queries match local behavior, +remain read-only, obey existing lifetime guarantees, and fetch payloads on demand +at documented batch granularity. Whole-column query scans and first-use dictionary +vocabulary loading must be explicit costs, not presented as selective row I/O. + +## Implementation results + +The seven implementation steps were completed as separate commits: + +1. `c5e17608` — prove remote batch range reads. +2. `c941ed94` — add the internal remote batch reader. +3. `3bbb744a` — read remote variable-length strings. +4. `3c4f1821` — integrate remote batch caching. +5. `3f58deed` — enable the remaining batch-backed columns. +6. `b3dfaaf9` — harden shared reads, mutation rejection and lifetime handling. +7. `4619082c` — document and demonstrate the extension. + +The resulting reader supports variable-length strings and bytes, MessagePack and +Arrow batch-backed lists, passive struct/object values, and dictionary columns. +Compressed batches participate in the owner's shared NONE, MEMORY or DISK cache +policy. Remote MessagePack decoding rejects embedded Blosc2 containers and +serialized references. Nonempty columns require valid persisted batch lengths, +and `storage="vl"` lists remain explicitly unsupported. + +The default suite passed with 10,315 tests and 36 skips. The focused remote table +suite passed with 90 tests, the related batch/list/dictionary/remote suites passed +with 413 tests and 5 skips, Ruff passed, and the normal Sphinx documentation build +completed. The strict `sphinx -W` build remains affected by pre-existing +repository-wide orphan-page, theme-option and ambiguous-reference warnings. + +An instrumented `memory://` experiment used a 55,026,931-byte archive containing +100,000 rows, 1,024 rows per variable-length batch and a 20,000-value dictionary. +A distant five-row batch read used 5 requests and transferred 575,310 bytes, +retaining 558,335 compressed bytes; the warm repeat used 0 requests and 0 bytes. +The dictionary code slice used 2 requests and 4,612 bytes. First decode loaded the +vocabulary with 10 requests and 95,494 bytes. Its decoded Python strings and maps +occupied approximately 4,028,168 bytes by `sys.getsizeof`, outside the compressed +transport cache budget. Metadata opening used 2 requests and 9,297 bytes. From fd4f3bcc9ae20866f3f7cd2438a967ed56ed2439 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 06:13:43 +0200 Subject: [PATCH 20/82] Support nullable ListArray elements --- src/blosc2/list_array.py | 17 ++++++++++------- src/blosc2/schema.py | 14 +++++++++----- tests/ctable/test_remote_ctable.py | 8 ++++---- tests/test_list_array.py | 29 +++++++++++++++++++++++++++++ 4 files changed, 52 insertions(+), 16 deletions(-) diff --git a/src/blosc2/list_array.py b/src/blosc2/list_array.py index 5b92ab784..88a527057 100644 --- a/src/blosc2/list_array.py +++ b/src/blosc2/list_array.py @@ -122,21 +122,21 @@ def _coerce_struct_item(spec: StructSpec, value: Any) -> dict[str, Any]: for name, child_spec in spec.fields.items(): if name not in value: raise ValueError(f"Struct list item is missing field {name!r}") - result[name] = None if value[name] is None else _coerce_scalar_item(child_spec, value[name]) + result[name] = _coerce_scalar_item(child_spec, value[name]) return result def _coerce_scalar_item(spec: SchemaSpec, value: Any) -> Any: # noqa: C901 if value is None: - raise ValueError("ListArray does not support nullable items inside a list in V1") + if getattr(spec, "nullable", False): + return None + raise ValueError(f"Null {_spec_label(spec)} list items are not allowed") if isinstance(spec, StructSpec): return _coerce_struct_item(spec, value) if isinstance(spec, ListSpec): return coerce_list_cell(spec, value) if isinstance(spec, DictionarySpec): - if value is None: - raise ValueError("ListArray does not support nullable items inside a list in V1") if not isinstance(value, str): value = str(value) return value @@ -211,7 +211,7 @@ def __init__( nullable: bool = False, storage: str = "batch", serializer: str = "msgpack", - batch_rows: int | None = None, + batch_rows: int | None = 2048, items_per_block: int | None = None, _from_schunk=None, **kwargs: Any, @@ -220,7 +220,10 @@ def __init__( Parameters may be supplied either as a complete ``spec`` or as an ``item_spec`` plus list/storage options. Storage-related keyword - arguments are passed to :class:`blosc2.Storage`. + arguments are passed to :class:`blosc2.Storage`. Batch storage flushes + automatically every ``batch_rows`` list cells (2048 by default). Pass + ``None`` for caller-managed batches; :meth:`flush`, :meth:`close`, or + context-manager exit persists the partial final batch. """ if _from_schunk is not None: if spec is not None or item_spec is not None or kwargs: @@ -801,7 +804,7 @@ def from_arrow( nullable: bool = True, storage: str = "batch", serializer: str = "msgpack", - batch_rows: int | None = None, + batch_rows: int | None = 2048, items_per_block: int | None = None, **kwargs: Any, ) -> ListArray: diff --git a/src/blosc2/schema.py b/src/blosc2/schema.py index 94c105644..6979f77f1 100644 --- a/src/blosc2/schema.py +++ b/src/blosc2/schema.py @@ -617,7 +617,12 @@ def from_metadata_dict(cls, data: dict[str, Any]) -> StructSpec: class ListSpec(SchemaSpec): - """Logical schema descriptor for a list-valued column.""" + """Logical schema descriptor for a list-valued column. + + ``batch_rows`` defaults to 2048 list cells for batch storage. Pass ``None`` + to keep pending cells in one caller-managed batch until ListArray.flush(). + The option has no effect on ``storage="vl"``. + """ python_type = _builtin_list dtype = None @@ -629,7 +634,7 @@ def __init__( nullable: bool = False, storage: str = "batch", serializer: str = "msgpack", - batch_rows: int | None = None, + batch_rows: int | None = 2048, items_per_block: int | None = None, ): if not isinstance(item_spec, SchemaSpec): @@ -662,8 +667,7 @@ def to_metadata_dict(self) -> dict[str, Any]: "storage": self.storage, "serializer": self.serializer, } - if self.batch_rows is not None: - d["batch_rows"] = self.batch_rows + d["batch_rows"] = self.batch_rows if self.items_per_block is not None: d["items_per_block"] = self.items_per_block return d @@ -1268,7 +1272,7 @@ def list( nullable: bool = False, storage: str = "batch", serializer: str = "msgpack", - batch_rows: int | None = None, + batch_rows: int | None = 2048, items_per_block: int | None = None, ) -> ListSpec: """Build a list-valued schema descriptor for CTable and ListArray.""" diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 2d547d08f..d9af61f74 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -405,7 +405,7 @@ def test_remote_ctable_batch_wrappers_and_dictionary(tmp_path): class Rich: data: bytes = blosc2.field(blosc2.vlbytes(nullable=True, batch_rows=2)) tags: list[int] = blosc2.field( # noqa: RUF009 - blosc2.list(blosc2.int64(), nullable=True, batch_rows=2) + blosc2.list(blosc2.int64(nullable=True), nullable=True, batch_rows=2) ) props: dict = blosc2.field( # noqa: RUF009 blosc2.struct({"count": blosc2.int32(), "name": blosc2.vlstring()}, nullable=True) @@ -414,7 +414,7 @@ class Rich: category: str = blosc2.field(blosc2.dictionary(nullable=True)) rows = [ - (b"one", [1, 2], {"count": 1, "name": "one"}, {"x": [1, 2]}, "a"), + (b"one", [1, None, 2], {"count": 1, "name": "one"}, {"x": [1, 2]}, "a"), (None, [], None, ("tuple", 2), None), (b"", None, {"count": 3, "name": "東京"}, np.arange(3), "b"), (b"four", [4], {"count": 4, "name": "four"}, {1, 2}, "a"), @@ -438,10 +438,10 @@ def test_remote_ctable_arrow_list(tmp_path): @dataclasses.dataclass class Lists: values: list[int] = blosc2.field( # noqa: RUF009 - blosc2.list(blosc2.int64(), serializer="arrow", nullable=True, batch_rows=2) + blosc2.list(blosc2.int64(nullable=True), serializer="arrow", nullable=True, batch_rows=2) ) - rows = [([1, 2],), (None,), ([],), ([3],)] + rows = [([1, None, 2],), (None,), ([],), ([3],)] local = blosc2.CTable(Lists, rows, create_summary_index=False) url = remote_table_url(tmp_path, local, "arrow-list") with blosc2.RemoteCTable(url) as remote: diff --git a/tests/test_list_array.py b/tests/test_list_array.py index 8342ff3d6..1ff62498a 100644 --- a/tests/test_list_array.py +++ b/tests/test_list_array.py @@ -70,6 +70,35 @@ def test_listarray_rejects_invalid_cells(): arr.append([1, None]) +@pytest.mark.parametrize("storage", ["vl", "batch"]) +def test_listarray_nullable_items(storage): + arr = blosc2.ListArray(item_spec=blosc2.int32(nullable=True), storage=storage, batch_rows=2) + values = [[1, None, 3], [], [None]] + arr.extend(values) + arr.flush() + assert arr[:] == values + + +def test_listarray_default_batch_rows_and_explicit_none(): + spec = blosc2.list(blosc2.int32()) + assert spec.batch_rows == 2048 + assert spec.to_metadata_dict()["batch_rows"] == 2048 + + legacy = blosc2.schema.ListSpec.from_metadata_dict({"kind": "list", "item": {"kind": "int32"}}) + assert legacy.batch_rows is None + + managed = blosc2.list(blosc2.int32(), batch_rows=None) + assert managed.to_metadata_dict()["batch_rows"] is None + + +def test_listarray_default_batch_boundaries(): + arr = blosc2.ListArray(item_spec=blosc2.int32()) + arr.extend([[i] for i in range(4097)]) + assert arr._backend._load_or_compute_batch_lengths() == [2048, 2048] + arr.flush() + assert arr._backend._load_or_compute_batch_lengths() == [2048, 2048, 1] + + def test_listarray_boolean_fancy_indexing(): arr = blosc2.ListArray(item_spec=blosc2.int32(), nullable=True, storage="batch", batch_rows=2) arr.extend([[1], None, [], [2, 3]]) From e49753d5c70c316c0319f67b3212c808ecb15183 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 06:16:00 +0200 Subject: [PATCH 21/82] Support nested ListArray values --- src/blosc2/list_array.py | 114 +++++++++----------- src/blosc2/schema.py | 5 +- src/blosc2/schema_compiler.py | 3 + tests/ctable/test_remote_ctable.py | 23 ++++ tests/ctable/test_varlen_schema_compiler.py | 15 +++ tests/test_list_array.py | 20 ++++ 6 files changed, 112 insertions(+), 68 deletions(-) diff --git a/src/blosc2/list_array.py b/src/blosc2/list_array.py index 88a527057..e020e7586 100644 --- a/src/blosc2/list_array.py +++ b/src/blosc2/list_array.py @@ -49,9 +49,15 @@ def _require_pyarrow(): def _arrow_type_for_spec(pa, spec: SchemaSpec): + if isinstance(spec, ListSpec): + child = pa.field("item", _arrow_type_for_spec(pa, spec.item_spec), nullable=spec.item_spec.nullable) + return pa.list_(child) if isinstance(spec, StructSpec): return pa.struct( - [pa.field(name, _arrow_type_for_spec(pa, child)) for name, child in spec.fields.items()] + [ + pa.field(name, _arrow_type_for_spec(pa, child), nullable=child.nullable) + for name, child in spec.fields.items() + ] ) mapping = { "int8": pa.int8(), @@ -71,7 +77,7 @@ def _arrow_type_for_spec(pa, spec: SchemaSpec): return mapping.get(spec.to_metadata_dict()["kind"]) -def _arrow_list_item_type_to_spec(pa, value_type): +def _arrow_list_item_type_to_spec(pa, value_type, *, nullable=False): import blosc2.schema as b2s mapping = { @@ -91,14 +97,28 @@ def _arrow_list_item_type_to_spec(pa, value_type): pa.binary(): b2s.bytes(), pa.large_binary(): b2s.bytes(), } + if pa.types.is_list(value_type) or pa.types.is_large_list(value_type): + field = value_type.value_field + return b2s.list( + _arrow_list_item_type_to_spec(pa, field.type, nullable=field.nullable), nullable=nullable + ) if pa.types.is_struct(value_type): return b2s.struct( - {field.name: _arrow_list_item_type_to_spec(pa, field.type) for field in value_type} + { + field.name: _arrow_list_item_type_to_spec(pa, field.type, nullable=field.nullable) + for field in value_type + }, + nullable=nullable, ) - return mapping.get(value_type) + spec = mapping.get(value_type) + if spec is not None: + spec.nullable = nullable + return spec -def _validate_list_spec(spec: ListSpec) -> None: +def _validate_list_spec(spec: ListSpec, *, _depth=0) -> None: + if _depth >= 32: + raise ValueError("ListArray nesting exceeds the supported depth of 32") if spec.storage not in _SUPPORTED_STORAGES: raise ValueError(f"Unsupported list storage: {spec.storage!r}") if spec.serializer not in _SUPPORTED_SERIALIZERS: @@ -108,7 +128,7 @@ def _validate_list_spec(spec: ListSpec) -> None: if spec.serializer == "arrow" and spec.storage != "batch": raise ValueError("ListArray serializer='arrow' requires storage='batch'") if isinstance(spec.item_spec, ListSpec): - raise TypeError("Nested list item specs are not supported in V1") + _validate_list_spec(spec.item_spec, _depth=_depth + 1) if spec.batch_rows is not None and spec.batch_rows <= 0: raise ValueError("batch_rows must be a positive integer") if spec.items_per_block is not None and spec.items_per_block <= 0: @@ -375,7 +395,7 @@ def _flush_full_batches(self) -> None: return while len(self._pending_cells) >= batch_rows: batch = self._pending_cells[:batch_rows] - self._backend.append(batch) + self._backend.append(self._typed_batch(batch)) self._pending_cells = self._pending_cells[batch_rows:] self._persisted_row_count += len(batch) self._invalidate_batch_caches() @@ -436,11 +456,13 @@ def extend_arrow(self, arrow_array) -> None: # backend, which would otherwise reorder them ahead of pending cells. self.flush() for chunk in chunks: - if len(chunk) == 0: - continue - self._backend.append(chunk) - self._persisted_row_count += len(chunk) - self._invalidate_batch_caches() + step = self.batch_rows or len(chunk) or 1 + for start in range(0, len(chunk), step): + part = chunk.slice(start, step) + typed = self._typed_batch(part.to_pylist()) + self._backend.append(typed) + self._persisted_row_count += len(part) + self._invalidate_batch_caches() def flush(self) -> None: """Persist any pending rows when using the batch backend.""" @@ -449,11 +471,19 @@ def flush(self) -> None: if self._pending_cells: self._backend._check_writable() batch = list(self._pending_cells) - self._backend.append(batch) + self._backend.append(self._typed_batch(batch)) self._persisted_row_count += len(batch) self._pending_cells.clear() self._invalidate_batch_caches() + def _typed_batch(self, values): + """Give Arrow batches their declared type so empty/null batches stay stable.""" + if self.spec.serializer != "arrow": + return values + pa = _require_pyarrow() + item = pa.field("item", self._arrow_item_type(), nullable=self.spec.item_spec.nullable) + return pa.array(values, type=pa.list_(item)) + def close(self) -> None: """Flush pending rows and close the logical container.""" self.flush() @@ -760,31 +790,7 @@ def to_cframe(self) -> bytes: return self._backend.to_cframe() def _arrow_item_type(self): - pa = _require_pyarrow() - kind = self.spec.item_spec.to_metadata_dict()["kind"] - mapping = { - "int8": pa.int8(), - "int16": pa.int16(), - "int32": pa.int32(), - "int64": pa.int64(), - "uint8": pa.uint8(), - "uint16": pa.uint16(), - "uint32": pa.uint32(), - "uint64": pa.uint64(), - "float32": pa.float32(), - "float64": pa.float64(), - "bool": pa.bool_(), - "string": pa.string(), - "bytes": pa.large_binary(), - } - if isinstance(self.spec.item_spec, StructSpec): - return pa.struct( - [ - pa.field(name, _arrow_type_for_spec(pa, child_spec)) - for name, child_spec in self.spec.item_spec.fields.items() - ] - ) - return mapping.get(kind) + return _arrow_type_for_spec(_require_pyarrow(), self.spec.item_spec) def to_arrow(self): """Return the data as a PyArrow list array.""" @@ -792,7 +798,8 @@ def to_arrow(self): self.flush() item_type = self._arrow_item_type() if item_type is not None: - return pa.array(list(self), type=pa.list_(item_type)) + item = pa.field("item", item_type, nullable=self.spec.item_spec.nullable) + return pa.array(list(self), type=pa.list_(item)) return pa.array(list(self)) @classmethod @@ -813,32 +820,9 @@ def from_arrow( if isinstance(arrow_array, pa.ChunkedArray): arrow_array = arrow_array.combine_chunks() if item_spec is None: - value_type = arrow_array.type.value_type - import blosc2.schema as b2s - - mapping = { - pa.int8(): b2s.int8(), - pa.int16(): b2s.int16(), - pa.int32(): b2s.int32(), - pa.int64(): b2s.int64(), - pa.uint8(): b2s.uint8(), - pa.uint16(): b2s.uint16(), - pa.uint32(): b2s.uint32(), - pa.uint64(): b2s.uint64(), - pa.float32(): b2s.float32(), - pa.float64(): b2s.float64(), - pa.bool_(): b2s.bool(), - pa.string(): b2s.string(), - pa.large_string(): b2s.string(), - pa.binary(): b2s.bytes(), - pa.large_binary(): b2s.bytes(), - } - if pa.types.is_struct(value_type): - item_spec = b2s.struct( - {field.name: _arrow_list_item_type_to_spec(pa, field.type) for field in value_type} - ) - else: - item_spec = mapping.get(value_type) + field = arrow_array.type.value_field + value_type = field.type + item_spec = _arrow_list_item_type_to_spec(pa, value_type, nullable=field.nullable) if item_spec is None: raise TypeError(f"Unsupported Arrow list item type {value_type!r}") arr = cls( diff --git a/src/blosc2/schema.py b/src/blosc2/schema.py index 6979f77f1..9626cc726 100644 --- a/src/blosc2/schema.py +++ b/src/blosc2/schema.py @@ -639,8 +639,6 @@ def __init__( ): if not isinstance(item_spec, SchemaSpec): raise TypeError("ListSpec item_spec must be a SchemaSpec instance") - if isinstance(item_spec, ListSpec): - raise TypeError("Nested list item specs are not supported in V1") if storage not in {"batch", "vl"}: raise ValueError("storage must be 'batch' or 'vl'") if serializer not in {"msgpack", "arrow"}: @@ -678,7 +676,8 @@ def to_listarray_metadata(self) -> dict[str, Any]: return d def display_label(self) -> str: - item_kind = self.item_spec.to_metadata_dict().get("kind", type(self.item_spec).__name__) + item = self.item_spec.display_label() if isinstance(self.item_spec, ListSpec) else None + item_kind = item or self.item_spec.to_metadata_dict().get("kind", type(self.item_spec).__name__) return f"list[{item_kind}]" @classmethod diff --git a/src/blosc2/schema_compiler.py b/src/blosc2/schema_compiler.py index 2c312f81d..cba9572fd 100644 --- a/src/blosc2/schema_compiler.py +++ b/src/blosc2/schema_compiler.py @@ -238,6 +238,9 @@ def validate_annotation_matches_spec(name: str, annotation: Any, spec: SchemaSpe if len(args) != 1: raise TypeError(f"Column {name!r}: list annotations must specify exactly one item type.") item_annotation = args[0] + if isinstance(spec.item_spec, ListSpec): + validate_annotation_matches_spec(f"{name}[]", item_annotation, spec.item_spec) + return expected = spec.item_spec.python_type if item_annotation is not expected: raise TypeError( diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index d9af61f74..cd921ef38 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -448,6 +448,29 @@ class Lists: assert remote["values"][:] == [row[0] for row in rows] +@pytest.mark.parametrize("serializer", ["msgpack", "arrow"]) +def test_remote_ctable_nested_list(tmp_path, serializer): + if serializer == "arrow": + pytest.importorskip("pyarrow") + + @dataclasses.dataclass + class Lists: + values: list[list[int]] = blosc2.field( # noqa: RUF009 + blosc2.list( + blosc2.list(blosc2.int64(nullable=True), nullable=True), + nullable=True, + serializer=serializer, + batch_rows=2, + ) + ) + + rows = [(None,), ([],), ([None, [], [1, None, 2]],), ([[3]],)] + local = blosc2.CTable(Lists, rows, create_summary_index=False) + url = remote_table_url(tmp_path, local, f"nested-list-{serializer}") + with blosc2.RemoteCTable(url) as remote: + assert remote["values"][:] == [row[0] for row in rows] + + def test_remote_ctable_rejects_unsafe_object_extension(tmp_path): @dataclasses.dataclass class Unsafe: diff --git a/tests/ctable/test_varlen_schema_compiler.py b/tests/ctable/test_varlen_schema_compiler.py index 80e8d3cb8..ba3a56f8b 100644 --- a/tests/ctable/test_varlen_schema_compiler.py +++ b/tests/ctable/test_varlen_schema_compiler.py @@ -46,3 +46,18 @@ class Bad: with pytest.raises(TypeError, match="list spec"): compile_schema(Bad) + + +def test_nested_list_annotation_and_schema_roundtrip(): + @dataclass + class Nested: + values: list[list[int]] = blosc2.field( # noqa: RUF009 + blosc2.list(blosc2.list(blosc2.int32(nullable=True), nullable=True)) + ) + + schema = compile_schema(Nested) + restored = schema_from_dict(schema_to_dict(schema)) + spec = restored.columns_by_name["values"].spec + assert isinstance(spec.item_spec, ListSpec) + assert spec.item_spec.nullable + assert spec.item_spec.item_spec.nullable diff --git a/tests/test_list_array.py b/tests/test_list_array.py index 1ff62498a..70ffc9fba 100644 --- a/tests/test_list_array.py +++ b/tests/test_list_array.py @@ -115,6 +115,26 @@ def test_listarray_arrow_roundtrip(): assert arr.to_arrow().to_pylist() == [["a"], None, ["b", "c"]] +@pytest.mark.parametrize("storage", ["vl", "batch"]) +def test_listarray_nested_lists(storage): + item = blosc2.list(blosc2.int32(nullable=True), nullable=True) + arr = blosc2.ListArray(item_spec=item, nullable=True, storage=storage, batch_rows=2) + values = [None, [], [None, [], [1, None, 3]], [[4]]] + arr.extend(values) + arr.flush() + assert arr[:] == values + assert arr.spec.display_label() == "list[list[int32]]" + + +def test_listarray_nested_arrow_roundtrip(): + pa = pytest.importorskip("pyarrow") + values = [None, [], [None, [], [1, None, 3]], [[4]]] + arrow = pa.array(values, type=pa.list_(pa.field("item", pa.list_(pa.int32()), nullable=True))) + arr = blosc2.ListArray.from_arrow(arrow) + assert arr[:] == values + assert arr.to_arrow().to_pylist() == values + + def test_listarray_extend_no_validate_keeps_none(): arr = blosc2.ListArray(item_spec=blosc2.int32(), nullable=True, storage="batch", batch_rows=2) arr.extend([[1], None, [2, 3]], validate=False) From 1ce260b7cf50729418052a2edfbd7272b8a0a22d Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 06:17:54 +0200 Subject: [PATCH 22/82] Add ListArray membership predicates --- src/blosc2/ctable.py | 20 ++++++++++++ src/blosc2/list_array.py | 52 ++++++++++++++++++++++++++++++ tests/ctable/test_remote_ctable.py | 2 ++ tests/test_list_array.py | 27 ++++++++++++++++ 4 files changed, 101 insertions(+) diff --git a/src/blosc2/ctable.py b/src/blosc2/ctable.py index 46b8e22c2..84f2a0dfb 100644 --- a/src/blosc2/ctable.py +++ b/src/blosc2/ctable.py @@ -1589,6 +1589,26 @@ def __getitem__(self, key: int | slice | list | np.ndarray): """ return self._values_from_key(key) + def contains(self, value): + """Return a Boolean row predicate for list cells containing *value*.""" + if not self.is_list: + raise TypeError("Column.contains() is only supported for list columns") + physical = self._raw_col.contains(value) + positions = self._resolve_live_positions() + mask = np.zeros(len(self._table._valid_rows), dtype=np.bool_) + mask[positions] = physical[positions] + return blosc2.asarray(mask) + + def overlaps(self, values): + """Return a Boolean row predicate for list cells sharing any value.""" + if not self.is_list: + raise TypeError("Column.overlaps() is only supported for list columns") + physical = self._raw_col.overlaps(values) + positions = self._resolve_live_positions() + mask = np.zeros(len(self._table._valid_rows), dtype=np.bool_) + mask[positions] = physical[positions] + return blosc2.asarray(mask) + def _values_from_key(self, key, *, check_stale: bool = True): # noqa: C901 """Materialise values for a logical index key.""" if check_stale: diff --git a/src/blosc2/list_array.py b/src/blosc2/list_array.py index e020e7586..a60a53c1c 100644 --- a/src/blosc2/list_array.py +++ b/src/blosc2/list_array.py @@ -8,6 +8,7 @@ from __future__ import annotations import copy +import math from bisect import bisect_right from collections import defaultdict from collections.abc import Iterable, Iterator @@ -216,6 +217,29 @@ def coerce_list_cell(spec: ListSpec, value: Any) -> list[Any] | None: return [_coerce_scalar_item(spec.item_spec, item) for item in list(value)] +def _list_item_equal(left: Any, right: Any) -> bool: + """Structural equality used by list membership predicates.""" + if isinstance(left, list) or isinstance(right, list): + return ( + isinstance(left, list) + and isinstance(right, list) + and len(left) == len(right) + and all(_list_item_equal(a, b) for a, b in zip(left, right, strict=True)) + ) + if isinstance(left, dict) or isinstance(right, dict): + return ( + isinstance(left, dict) + and isinstance(right, dict) + and left.keys() == right.keys() + and all(_list_item_equal(left[key], right[key]) for key in left) + ) + if isinstance(left, (float, np.floating)) and math.isnan(left): + return False + if isinstance(right, (float, np.floating)) and math.isnan(right): + return False + return bool(left == right) + + class ListArray: """A row-oriented container for list-valued data. @@ -436,6 +460,34 @@ def extend(self, values: Iterable[Any], *, validate: bool = True) -> None: self._pending_cells.extend(cells) self._flush_full_batches() + def contains(self, value: Any) -> np.ndarray: + """Return one Boolean per row indicating immediate-child membership. + + Null outer lists and empty lists return false. Nested lists compare + structurally and are not flattened. + """ + needle = _coerce_scalar_item(self.spec.item_spec, value) + return np.fromiter( + (cell is not None and any(_list_item_equal(item, needle) for item in cell) for cell in self), + dtype=np.bool_, + count=len(self), + ) + + def overlaps(self, values: Iterable[Any]) -> np.ndarray: + """Return one Boolean per row when any immediate child is in *values*.""" + if isinstance(values, (str, bytes, bytearray, memoryview)) or not isinstance(values, Iterable): + raise TypeError("ListArray.overlaps() expects an iterable of list items") + needles = [_coerce_scalar_item(self.spec.item_spec, value) for value in values] + return np.fromiter( + ( + cell is not None + and any(_list_item_equal(item, needle) for item in cell for needle in needles) + for cell in self + ), + dtype=np.bool_, + count=len(self), + ) + def extend_arrow(self, arrow_array) -> None: """Append a PyArrow list array without materializing Python cells. diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index cd921ef38..9e5c2386b 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -469,6 +469,8 @@ class Lists: url = remote_table_url(tmp_path, local, f"nested-list-{serializer}") with blosc2.RemoteCTable(url) as remote: assert remote["values"][:] == [row[0] for row in rows] + assert remote[remote["values"].contains([1, None, 2])]["values"][:] == [[None, [], [1, None, 2]]] + assert remote[remote["values"].overlaps([[3], [4]])]["values"][:] == [[[3]]] def test_remote_ctable_rejects_unsafe_object_extension(tmp_path): diff --git a/tests/test_list_array.py b/tests/test_list_array.py index 70ffc9fba..17c40422e 100644 --- a/tests/test_list_array.py +++ b/tests/test_list_array.py @@ -9,6 +9,7 @@ from dataclasses import dataclass +import numpy as np import pytest import blosc2 @@ -135,6 +136,32 @@ def test_listarray_nested_arrow_roundtrip(): assert arr.to_arrow().to_pylist() == values +def test_listarray_contains_and_overlaps(): + item = blosc2.list(blosc2.int32(nullable=True), nullable=True) + arr = blosc2.ListArray(item_spec=item, nullable=True, batch_rows=2) + arr.extend([None, [], [None, [1, 2]], [[3]], [[1, 2], [3]]]) + + np.testing.assert_array_equal(arr.contains([1, 2]), [False, False, True, False, True]) + np.testing.assert_array_equal(arr.contains(None), [False, False, True, False, False]) + np.testing.assert_array_equal(arr.overlaps([[3], [4]]), [False, False, False, True, True]) + np.testing.assert_array_equal(arr.overlaps([]), np.zeros(5, dtype=np.bool_)) + + +def test_ctable_list_predicates_compose(): + @dataclass + class Rows: + tags: list[int] = blosc2.field( # noqa: RUF009 + blosc2.list(blosc2.int32(nullable=True), nullable=True, batch_rows=2) + ) + value: int = blosc2.field(blosc2.int32()) + + rows = [([1, None], 0), (None, 1), ([], 2), ([2, 3], 3), ([3], 4)] + table = blosc2.CTable(Rows, rows, create_summary_index=False) + assert table[table["tags"].contains(None)]["value"][:].tolist() == [0] + assert table[table["tags"].overlaps([2, 9]) & (table["value"] > 2)]["value"][:].tolist() == [3] + assert table[~table["tags"].contains(3)]["value"][:].tolist() == [0, 1, 2] + + def test_listarray_extend_no_validate_keeps_none(): arr = blosc2.ListArray(item_spec=blosc2.int32(), nullable=True, storage="batch", batch_rows=2) arr.extend([[1], None, [2, 3]], validate=False) From 1caa94c7a9c0bee0944b5adacdf259f1399e5fa2 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 06:22:36 +0200 Subject: [PATCH 23/82] Add remote ListArray membership indexes --- src/blosc2/ctable.py | 15 +++++ src/blosc2/ctable_indexing.py | 101 ++++++++++++++++++++++++++++- src/blosc2/ctable_storage.py | 31 ++++++++- src/blosc2/list_array.py | 17 +++++ tests/ctable/test_remote_ctable.py | 36 ++++++++++ tests/test_list_array.py | 23 +++++++ 6 files changed, 220 insertions(+), 3 deletions(-) diff --git a/src/blosc2/ctable.py b/src/blosc2/ctable.py index 84f2a0dfb..f437ccca9 100644 --- a/src/blosc2/ctable.py +++ b/src/blosc2/ctable.py @@ -1593,6 +1593,12 @@ def contains(self, value): """Return a Boolean row predicate for list cells containing *value*.""" if not self.is_list: raise TypeError("Column.contains() is only supported for list columns") + positions = self._table._membership_positions(self._col_name, [value]) + if positions is not None: + mask = np.zeros(len(self._table._valid_rows), dtype=np.bool_) + mask[positions] = True + mask &= self._valid_rows[:] + return blosc2.asarray(mask) physical = self._raw_col.contains(value) positions = self._resolve_live_positions() mask = np.zeros(len(self._table._valid_rows), dtype=np.bool_) @@ -1603,6 +1609,15 @@ def overlaps(self, values): """Return a Boolean row predicate for list cells sharing any value.""" if not self.is_list: raise TypeError("Column.overlaps() is only supported for list columns") + if isinstance(values, (str, bytes, bytearray, memoryview)) or not isinstance(values, Iterable): + raise TypeError("Column.overlaps() expects an iterable of list items") + values = list(values) + positions = self._table._membership_positions(self._col_name, values) + if positions is not None: + mask = np.zeros(len(self._table._valid_rows), dtype=np.bool_) + mask[positions] = True + mask &= self._valid_rows[:] + return blosc2.asarray(mask) physical = self._raw_col.overlaps(values) positions = self._resolve_live_positions() mask = np.zeros(len(self._table._valid_rows), dtype=np.bool_) diff --git a/src/blosc2/ctable_indexing.py b/src/blosc2/ctable_indexing.py index 6680223b2..71171a254 100644 --- a/src/blosc2/ctable_indexing.py +++ b/src/blosc2/ctable_indexing.py @@ -385,7 +385,7 @@ def _validate_index_descriptor(col_name: str, descriptor: dict) -> None: if not isinstance(token, str) or not token: raise ValueError(f"Malformed index metadata for column {col_name!r}: missing token.") kind = descriptor.get("kind") - if kind not in {"summary", "bucket", "partial", "full", "opsi"}: + if kind not in {"summary", "bucket", "partial", "full", "opsi", "membership"}: raise ValueError(f"Malformed index metadata for column {col_name!r}: invalid kind {kind!r}.") if kind == "bucket" and not isinstance(descriptor.get("bucket"), dict): raise ValueError(f"Malformed index metadata for column {col_name!r}: missing bucket payload.") @@ -393,6 +393,10 @@ def _validate_index_descriptor(col_name: str, descriptor: dict) -> None: raise ValueError(f"Malformed index metadata for column {col_name!r}: missing partial payload.") if kind == "full" and not isinstance(descriptor.get("full"), dict): raise ValueError(f"Malformed index metadata for column {col_name!r}: missing full payload.") + if kind == "membership" and not isinstance(descriptor.get("membership"), dict): + raise ValueError( + f"Malformed index metadata for column {col_name!r}: missing membership payload." + ) def _drop_index_descriptor(self, col_name: str, descriptor: dict) -> None: """Delete sidecars/cache for a catalog descriptor without touching the column mapping.""" @@ -408,6 +412,14 @@ def _drop_index_descriptor(self, col_name: str, descriptor: dict) -> None: ) token = descriptor["token"] + if descriptor.get("kind") == "membership": + path = descriptor["membership"].get("postings_path") + if path: + with contextlib.suppress(OSError): + os.remove(path) + with contextlib.suppress(OSError): + os.rmdir(os.path.dirname(path)) + return col_arr = None with contextlib.suppress(Exception): col_arr = self._index_target_array(col_name, descriptor) @@ -459,6 +471,82 @@ def _index_create_kwargs_from_descriptor(self, descriptor: dict) -> dict[str, An kwargs["expression"] = target.get("expression") return kwargs + def _create_membership_index(self, col_name: str, *, name: str | None = None): + """Build one compressed posting-list batch per distinct scalar item.""" + from blosc2.list_array import list_item_key + + spec = self._schema.columns_by_name[col_name].spec + if not isinstance(spec, ListSpec): + raise ValueError("kind='membership' is only supported for list columns") + if isinstance(spec.item_spec, (ListSpec, StructSpec)): + raise ValueError("Membership indexes currently require a scalar list item type") + + postings: dict[bytes, list[int]] = {} + for row, cell in enumerate(self._cols[col_name]): + if cell is None: + continue + row_keys = {key for item in cell if (key := list_item_key(spec.item_spec, item)) is not None} + for key in row_keys: + postings.setdefault(key, []).append(row) + + entries = sorted(postings.items()) + anchor = self._storage.index_anchor_path(col_name) + path = None if anchor is None else os.path.join(os.path.dirname(anchor), "membership.b2b") + membership: dict[str, Any] = {"version": 1, "keys": []} + if path is None: + membership["postings"] = [rows for _, rows in entries] + else: + os.makedirs(os.path.dirname(path), exist_ok=True) + store = blosc2.BatchArray(urlpath=path, mode="w") + for _, rows in entries: + store.append(rows) + membership["postings_path"] = path + membership["keys"] = [[key, index, len(rows)] for index, (key, rows) in enumerate(entries)] + + value_epoch, _ = self._storage.get_epoch_counters() + descriptor = { + "kind": "membership", + "token": col_name, + "name": name or "", + "membership": membership, + "built_value_epoch": value_epoch, + "stale": False, + } + catalog = self._get_index_catalog() + catalog[col_name] = descriptor + self._storage.save_index_catalog(catalog) + self._invalidate_index_catalog_cache() + return blosc2.Index._from_table(self, col_name, descriptor) + + def _membership_positions(self, col_name: str, values) -> np.ndarray | None: + """Return indexed physical rows, or None when a scan is required.""" + from blosc2.list_array import list_item_key + + descriptor = self._get_index_catalog().get(col_name) + if not descriptor or descriptor.get("kind") != "membership" or descriptor.get("stale"): + return None + self._validate_index_descriptor(col_name, descriptor) + payload = descriptor["membership"] + directory = {key: (int(index), int(count)) for key, index, count in payload.get("keys", [])} + spec = self._schema.columns_by_name[col_name].spec + wanted = {key for value in values if (key := list_item_key(spec.item_spec, value)) is not None} + indexes = sorted({directory[key][0] for key in wanted if key in directory}) + if not indexes: + return np.empty(0, dtype=np.int64) + + inline = payload.get("postings") + if inline is not None: + chunks = [inline[index] for index in indexes] + elif hasattr(self._storage, "open_membership_postings"): + chunks = self._storage.open_membership_postings(col_name, descriptor, indexes) + else: + path = payload.get("postings_path") + if not isinstance(path, str): + raise ValueError(f"Malformed membership index for column {col_name!r}: missing postings") + store = blosc2.open(path, mode="r") + chunks = [store[index][:] for index in indexes] + return np.unique(np.concatenate([np.asarray(chunk, dtype=np.int64) for chunk in chunks])) + def _normalize_table_expression_target( self, expression: str, operands: dict | None = None ) -> tuple[dict, np.dtype]: @@ -818,7 +906,7 @@ def create_index( # noqa: C901 field: str | None = None, expression: str | None = None, operands: dict | None = None, - kind: blosc2.IndexKind | None = None, + kind: blosc2.IndexKind | str | None = None, optlevel: int = 5, name: str | None = None, build: str = "auto", @@ -892,6 +980,15 @@ def create_index( # noqa: C901 if col_name is not None: col_name = self._logical_to_physical_name(col_name) + if kind == "membership": + if expression is not None or col_name is None: + raise ValueError("Membership indexes require a stored list column") + if col_name not in self._cols: + raise KeyError(f"No column named {col_name!r}. Available: {self.col_names}") + if col_name in self._get_index_catalog(): + raise ValueError(f"Index already exists for column {col_name!r}.") + return self._create_membership_index(col_name, name=name) + from blosc2.indexing import ( _IN_MEMORY_INDEXES, _copy_descriptor, diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index fb522da33..23d8f27b6 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -25,6 +25,7 @@ import copy import json import os +import pathlib from typing import Any import numpy as np @@ -843,7 +844,35 @@ def close(self) -> None: def load_index_catalog(self) -> dict: self._check_open() - return {} + raw = self._metadata().get("index_catalog") + if not isinstance(raw, dict): + return {} + catalog = {} + for name, descriptor in raw.items(): + if not isinstance(descriptor, dict) or descriptor.get("kind") != "membership": + continue + payload = descriptor.get("membership") + if not isinstance(payload, dict): + raise ValueError(f"Malformed remote membership index for column {name!r}") + path = payload.get("postings_path") + if path is not None and ( + not isinstance(path, str) or os.path.isabs(path) or ".." in pathlib.PurePosixPath(path).parts + ): + raise ValueError(f"Unsafe remote membership index path for column {name!r}") + catalog[name] = copy.deepcopy(descriptor) + return catalog + + def open_membership_postings(self, name: str, descriptor: dict, indexes: list[int]): + """Read only the compressed posting-list batches needed by a query.""" + from blosc2.remote_batch import _RemoteBatchArray + + path = descriptor["membership"].get("postings_path") + if not isinstance(path, str) or not path.endswith(".b2b"): + raise ValueError(f"Malformed remote membership index for column {name!r}") + logical = path[:-4].strip("/") + source = self._owner.open_ctable_batch(self._full_key(logical)) + backend = _RemoteBatchArray(source, f"{name} membership index", self._check_open) + return [backend[index][:] for index in indexes] save_index_catalog = _not_supported diff --git a/src/blosc2/list_array.py b/src/blosc2/list_array.py index a60a53c1c..05bac6bd9 100644 --- a/src/blosc2/list_array.py +++ b/src/blosc2/list_array.py @@ -240,6 +240,23 @@ def _list_item_equal(left: Any, right: Any) -> bool: return bool(left == right) +def list_item_key(spec: SchemaSpec, value: Any) -> bytes | None: + """Return the stable persisted membership key for one validated list item. + + NaN has no key because membership follows ordinary equality and never + matches it. Signed zero is normalized before MessagePack encoding. + """ + from blosc2.msgpack_utils import msgpack_packb + + value = _coerce_scalar_item(spec, value) + if isinstance(value, (float, np.floating)): + if math.isnan(value): + return None + if value == 0: + value = 0.0 + return msgpack_packb(value) + + class ListArray: """A row-oriented container for list-valued data. diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 9e5c2386b..9008fe3d2 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -473,6 +473,42 @@ class Lists: assert remote[remote["values"].overlaps([[3], [4]])]["values"][:] == [[[3]]] +def test_remote_ctable_membership_index_avoids_list_payload(tmp_path, monkeypatch): + @dataclasses.dataclass + class Rows: + tags: list[int] = blosc2.field( # noqa: RUF009 + blosc2.list(blosc2.int32(nullable=True), nullable=True, batch_rows=2) + ) + value: int = blosc2.field(blosc2.int32()) + + rows = [([1, None], 0), (None, 1), ([], 2), ([2, 3], 3), ([3], 4)] + local = blosc2.CTable( + Rows, + rows, + urlpath=tmp_path / "membership-index.b2d", + mode="w", + create_summary_index=False, + ) + local.create_index("tags", kind="membership") + url = remote_table_url(tmp_path, local, "membership-index") + + fs = fsspec.filesystem("memory") + reads = [] + original = type(fs).cat_file + + def counted(self, path, start=None, end=None, **kwargs): + reads.append((start, end)) + return original(self, path, start=start, end=end, **kwargs) + + monkeypatch.setattr(type(fs), "cat_file", counted) + with blosc2.RemoteCTable(url, cache_policy=blosc2.CachePolicy.NONE) as remote: + reads.clear() + assert remote._get_index_catalog()["tags"]["kind"] == "membership" + assert remote[remote["tags"].contains(3)]["value"][:].tolist() == [3, 4] + assert reads + assert not dict.__contains__(remote._cols, "tags") + + def test_remote_ctable_rejects_unsafe_object_extension(tmp_path): @dataclasses.dataclass class Unsafe: diff --git a/tests/test_list_array.py b/tests/test_list_array.py index 17c40422e..5de9ebd3b 100644 --- a/tests/test_list_array.py +++ b/tests/test_list_array.py @@ -162,6 +162,29 @@ class Rows: assert table[~table["tags"].contains(3)]["value"][:].tolist() == [0, 1, 2] +def test_ctable_membership_index_matches_scan(tmp_path): + @dataclass + class Rows: + tags: list[int] = blosc2.field( # noqa: RUF009 + blosc2.list(blosc2.int32(nullable=True), nullable=True, batch_rows=2) + ) + value: int = blosc2.field(blosc2.int32()) + + rows = [([1, None, 1], 0), (None, 1), ([], 2), ([2, 3], 3), ([3], 4)] + path = tmp_path / "membership.b2d" + table = blosc2.CTable(Rows, rows, urlpath=path, mode="w", create_summary_index=False) + expected = table[table["tags"].overlaps([None, 3])]["value"][:].tolist() + + index = table.create_index("tags", kind="membership") + assert index.kind == "membership" + assert table[table["tags"].overlaps([None, 3])]["value"][:].tolist() == expected + assert table[table["tags"].contains(9)]["value"][:].tolist() == [] + + table["tags"][0] = [9] + assert table._get_index_catalog()["tags"]["stale"] is True + assert table[table["tags"].contains(9)]["value"][:].tolist() == [0] + + def test_listarray_extend_no_validate_keeps_none(): arr = blosc2.ListArray(item_spec=blosc2.int32(), nullable=True, storage="batch", batch_rows=2) arr.extend([[1], None, [2, 3]], validate=False) From eb20bd3e87e4e5fbc0f49518936fe8beac987462 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 06:30:08 +0200 Subject: [PATCH 24/82] Document ListArray V2 support --- RELEASE_NOTES.md | 15 + doc/reference/ctable.rst | 16 + doc/reference/list_array.rst | 58 ++- doc/reference/remotectable.rst | 12 + doc/tutorials/11.containers.ipynb | 106 +++++- examples/ctable/remote_handling.py | 19 + plans/ctable-varlen-v2.md | 402 ++++++++++++++++++++ src/blosc2/ctable.py | 6 +- src/blosc2/ctable_indexing.py | 34 +- src/blosc2/schema_compiler.py | 3 + tests/ctable/test_varlen_schema_compiler.py | 10 + 11 files changed, 654 insertions(+), 27 deletions(-) create mode 100644 plans/ctable-varlen-v2.md diff --git a/RELEASE_NOTES.md b/RELEASE_NOTES.md index d058edfe7..be2d493ac 100644 --- a/RELEASE_NOTES.md +++ b/RELEASE_NOTES.md @@ -6,6 +6,17 @@ XXX version-specific blurb XXX ### Improvements +#### ListArray V2 + +- List elements can be nullable, and ListSpec values can be nested recursively. +- ListArray and CTable list columns provide `contains()` and `overlaps()` row + predicates. +- Optional `kind="membership"` indexes accelerate flat scalar-list predicates + locally and through RemoteCTable. +- New ListArray schemas use `batch_rows=2048` by default. Explicit `None` keeps + caller-managed batching, and existing stored schemas without the field retain + their previous behavior when reopened. + #### Common remote-object API - Added the public `RemoteObject` base for `RemoteArray`, `RemoteStore`, @@ -20,6 +31,10 @@ XXX version-specific blurb XXX ### Compatibility notes +- The ListArray construction default changed from caller-managed batches to + 2048 rows per batch. Pass `batch_rows=None` to retain the previous behavior. + Existing arrays are not rewritten and keep their stored boundaries. + - `RemoteCTable.save()` previously inherited `CTable.save()` and returned `None` after materializing local data. It now returns the reference path. Use `materialize(urlpath=...)` or the table conversion methods when a diff --git a/doc/reference/ctable.rst b/doc/reference/ctable.rst index f9974f965..038671da7 100644 --- a/doc/reference/ctable.rst +++ b/doc/reference/ctable.rst @@ -10,6 +10,22 @@ independently; rows are never materialised in their entirety unless you explicitly call :meth:`~blosc2.CTable.to_arrow` or iterate with :meth:`~blosc2.CTable.__iter__`. +List columns +------------ + +Declare list columns with :func:`blosc2.list`. Batch storage uses 2048 outer +rows per compressed batch by default; override ``batch_rows`` to trade remote +read granularity against request count and compression. ``batch_rows=None`` +keeps caller-managed boundaries, while CTable still flushes the pending final +batch when it persists or closes the table. Arrow/Parquet input batch size and +the persisted ListArray ``batch_rows`` setting are separate controls. + +List item specs may themselves be nullable or lists. Use +:meth:`Column.contains` and :meth:`Column.overlaps` for membership filtering, +and ``create_index(name, kind="membership")`` for repeated selective queries on +flat scalar lists. Membership indexes are marked stale after table mutation; +queries then use the scan path until the index is rebuilt. + .. currentmodule:: blosc2 .. autoclass:: blosc2.CTable diff --git a/doc/reference/list_array.rst b/doc/reference/list_array.rst index 267c9f539..547f9396b 100644 --- a/doc/reference/list_array.rst +++ b/doc/reference/list_array.rst @@ -22,19 +22,19 @@ Quick example import blosc2 - arr = blosc2.ListArray( - item_spec=blosc2.string(max_length=16), + with blosc2.ListArray( + item_spec=blosc2.string(max_length=16, nullable=True), nullable=True, storage="batch", urlpath="ingredients.b2b", mode="w", - ) - arr.append(["salt", "sugar"]) - arr.append([]) - arr.append(None) + ) as arr: + arr.append(["salt", None, "sugar"]) + arr.append([]) + arr.append(None) - print(arr[0]) - print(arr[1:]) + print(arr[0]) + print(arr[1:]) reopened = blosc2.open("ingredients.b2b", mode="r") print(type(reopened).__name__) @@ -43,6 +43,46 @@ Quick example Returned Python lists are detached values. Mutating them locally does not write back to the container; reassign the whole cell instead. +Batching +-------- +Batch storage writes up to 2048 list cells per compressed chunk by default. +``batch_rows`` counts outer list cells, not child elements or bytes. A partial +final batch is persisted by :meth:`ListArray.flush`, :meth:`ListArray.close`, or +context-manager exit. Pass a smaller positive value to reduce remote overfetch, +usually at the cost of more requests and weaker compression. + +``batch_rows=None`` opts into caller-managed boundaries: appends remain pending +until a flush, which writes every pending cell as one batch. This can consume +substantial memory and make a remote scalar read download a large batch. The +setting has no effect with ``storage="vl"``. Existing arrays retain their stored +batch boundaries when reopened. + +Nested values and predicates +---------------------------- +Nullability is independent at each level. The following schema accepts a null +outer list, null inner lists, and null integers inside an inner list:: + + nested = blosc2.list( + blosc2.list(blosc2.int32(nullable=True), nullable=True), + nullable=True, + ) + +Membership predicates inspect immediate children and do not flatten nesting:: + + arr.contains([1, 2]) + arr.overlaps([[1, 2], [3]]) + +On a CTable column these methods return Boolean row predicates that compose with +other column expressions. An optional membership index accelerates flat lists of +scalar Boolean, numeric, string, or bytes values:: + + table.create_index("tags", kind="membership") + selected = table[table["tags"].overlaps(["python", "numpy"])] + +Null outer lists and empty lists do not match. A nullable child matches an +explicit ``None``. Nested-list and struct membership uses the scan path; creating +a membership index for those child types is not supported. + .. currentmodule:: blosc2 .. autoclass:: ListArray @@ -76,3 +116,5 @@ Quick example -------------- .. automethod:: to_arrow .. automethod:: to_cframe + .. automethod:: contains + .. automethod:: overlaps diff --git a/doc/reference/remotectable.rst b/doc/reference/remotectable.rst index 50c7a870e..4fe67b271 100644 --- a/doc/reference/remotectable.rst +++ b/doc/reference/remotectable.rst @@ -11,6 +11,18 @@ selected with ``dataset=`` or through :class:`blosc2.RemoteStore`. Batch-backed reads transfer and decode whole compressed batches. Dictionary codes remain selective, while the full vocabulary is loaded on first use. +ListArray batches contain 2048 list cells by default. Smaller batches reduce +overfetch for sparse reads; larger batches usually improve scans and compression. +The row limit is not a byte limit, so one unusually large list can still require +a large transfer. See :ref:`ListArray` for batching controls. + +Nullable list elements, nested lists, and ``contains``/``overlaps`` predicates +have the same semantics as local tables. Without an index a predicate scans the +remote list batches. A persisted ``kind="membership"`` index on a flat scalar +list fetches only the compressed posting batches for requested values. If the +result projects only other columns, the list payload remains unopened. Indexes +on nested lists and structs are not supported. ListArray ``storage="vl"`` also +remains unavailable through RemoteCTable. Saving and materializing have different meanings: diff --git a/doc/tutorials/11.containers.ipynb b/doc/tutorials/11.containers.ipynb index 7f63d370c..a9ca29f56 100644 --- a/doc/tutorials/11.containers.ipynb +++ b/doc/tutorials/11.containers.ipynb @@ -20,6 +20,12 @@ "ExecuteTime": { "end_time": "2026-05-21T09:37:13.776332Z", "start_time": "2026-05-21T09:37:13.709517Z" + }, + "execution": { + "iopub.execute_input": "2026-09-20T04:26:06.450451Z", + "iopub.status.busy": "2026-09-20T04:26:06.450037Z", + "iopub.status.idle": "2026-09-20T04:26:06.533528Z", + "shell.execute_reply": "2026-09-20T04:26:06.533098Z" } }, "outputs": [], @@ -94,6 +100,12 @@ "ExecuteTime": { "end_time": "2026-05-21T09:37:13.794555Z", "start_time": "2026-05-21T09:37:13.782817Z" + }, + "execution": { + "iopub.execute_input": "2026-09-20T04:26:06.535070Z", + "iopub.status.busy": "2026-09-20T04:26:06.534980Z", + "iopub.status.idle": "2026-09-20T04:26:06.537914Z", + "shell.execute_reply": "2026-09-20T04:26:06.537547Z" } }, "outputs": [ @@ -148,6 +160,12 @@ "ExecuteTime": { "end_time": "2026-05-21T09:37:13.811021Z", "start_time": "2026-05-21T09:37:13.795349Z" + }, + "execution": { + "iopub.execute_input": "2026-09-20T04:26:06.539091Z", + "iopub.status.busy": "2026-09-20T04:26:06.538998Z", + "iopub.status.idle": "2026-09-20T04:26:06.544357Z", + "shell.execute_reply": "2026-09-20T04:26:06.543909Z" } }, "outputs": [ @@ -192,7 +210,11 @@ "\n", "`ListArray` is a compact container for one variable-length typed list per row. It is useful when every row contains a list of items with the same logical item type, but the list length changes from row to row.\n", "\n", - "Use it for ragged typed data such as token ids, tags, nested numeric observations, or nullable lists. Compared with `ObjectArray`, it keeps more type information and can interoperate with Arrow-style list arrays.\n" + "Use it for ragged typed data such as token ids, tags, nested numeric observations, or nullable lists. Compared with `ObjectArray`, it keeps more type information and can interoperate with Arrow-style list arrays.\n", + "\n", + "Batch storage defaults to 2048 outer list rows per compressed batch. A smaller `batch_rows` reduces remote overfetch; `batch_rows=None` leaves boundaries under caller control and writes all pending rows when `flush()` is called. `flush()`, `close()`, or context-manager exit always persists the partial final batch. The row count is not a byte limit, and it has no effect on `storage=\"vl\"`.\n", + "\n", + "The item spec controls element nullability and may itself be another list spec. `contains()` and `overlaps()` inspect immediate children; CTable columns expose the same predicates and can add a `kind=\"membership\"` index for flat scalar lists.\n" ] }, { @@ -203,6 +225,12 @@ "ExecuteTime": { "end_time": "2026-05-21T09:37:13.826164Z", "start_time": "2026-05-21T09:37:13.811655Z" + }, + "execution": { + "iopub.execute_input": "2026-09-20T04:26:06.545373Z", + "iopub.status.busy": "2026-09-20T04:26:06.545299Z", + "iopub.status.idle": "2026-09-20T04:26:06.549817Z", + "shell.execute_reply": "2026-09-20T04:26:06.549467Z" } }, "outputs": [ @@ -211,24 +239,24 @@ "output_type": "stream", "text": [ "length: 4\n", - "all rows: [['red', 'fast'], [], None, ['blue']]\n", - "row 0: ['red', 'fast']\n", + "all rows: [['red', None, 'fast'], [], None, ['blue']]\n", + "row 0: ['red', None, 'fast']\n", "reopened type: ListArray\n", - "reopened rows: [['red', 'fast'], [], None, ['blue']]\n" + "reopened rows: [['red', None, 'fast'], [], None, ['blue']]\n" ] } ], "source": [ "list_path = reset(\"tags.b2b\")\n", "tags = blosc2.ListArray(\n", - " item_spec=blosc2.string(max_length=16),\n", + " item_spec=blosc2.string(max_length=16, nullable=True),\n", " nullable=True,\n", " storage=\"batch\",\n", " batch_rows=2,\n", " urlpath=list_path,\n", " mode=\"w\",\n", ")\n", - "tags.extend([[\"red\", \"fast\"], [], None, [\"blue\"]])\n", + "tags.extend([[\"red\", None, \"fast\"], [], None, [\"blue\"]])\n", "tags.flush()\n", "reopened = blosc2.open(list_path, mode=\"r\")\n", "\n", @@ -259,6 +287,12 @@ "ExecuteTime": { "end_time": "2026-05-21T09:37:13.846391Z", "start_time": "2026-05-21T09:37:13.826835Z" + }, + "execution": { + "iopub.execute_input": "2026-09-20T04:26:06.550821Z", + "iopub.status.busy": "2026-09-20T04:26:06.550759Z", + "iopub.status.idle": "2026-09-20T04:26:06.554131Z", + "shell.execute_reply": "2026-09-20T04:26:06.553814Z" } }, "outputs": [ @@ -311,6 +345,12 @@ "ExecuteTime": { "end_time": "2026-05-21T09:37:13.860323Z", "start_time": "2026-05-21T09:37:13.847148Z" + }, + "execution": { + "iopub.execute_input": "2026-09-20T04:26:06.555107Z", + "iopub.status.busy": "2026-09-20T04:26:06.555040Z", + "iopub.status.idle": "2026-09-20T04:26:06.558562Z", + "shell.execute_reply": "2026-09-20T04:26:06.558228Z" } }, "outputs": [ @@ -360,6 +400,12 @@ "ExecuteTime": { "end_time": "2026-05-21T09:37:13.927792Z", "start_time": "2026-05-21T09:37:13.861189Z" + }, + "execution": { + "iopub.execute_input": "2026-09-20T04:26:06.559493Z", + "iopub.status.busy": "2026-09-20T04:26:06.559432Z", + "iopub.status.idle": "2026-09-20T04:26:06.595966Z", + "shell.execute_reply": "2026-09-20T04:26:06.595556Z" } }, "outputs": [ @@ -369,15 +415,13 @@ "text": [ "columns: ['trip_id', 'distance_km', 'company', 'tags']\n", "rows: 3\n", - "distance storage: ndarray\n", - "tags storage: list\n", + "distance storage: persistent\n", + "tags storage: persistent\n", "data:\n", " trip_id distance_km company tags\n", "0 1 2.500000 Blue Cab ['airport', 'card']\n", "1 2 0.800000 Green Cab []\n", "2 3 12.100000 Yellow Cab None\n", - "\n", - "[3 rows x 4 columns]\n", "reopened type: CTable\n" ] } @@ -436,6 +480,12 @@ "ExecuteTime": { "end_time": "2026-05-21T09:37:13.945499Z", "start_time": "2026-05-21T09:37:13.928890Z" + }, + "execution": { + "iopub.execute_input": "2026-09-20T04:26:06.596994Z", + "iopub.status.busy": "2026-09-20T04:26:06.596927Z", + "iopub.status.idle": "2026-09-20T04:26:06.600635Z", + "shell.execute_reply": "2026-09-20T04:26:06.600312Z" } }, "outputs": [ @@ -482,6 +532,12 @@ "ExecuteTime": { "end_time": "2026-05-21T09:37:13.968390Z", "start_time": "2026-05-21T09:37:13.953403Z" + }, + "execution": { + "iopub.execute_input": "2026-09-20T04:26:06.601678Z", + "iopub.status.busy": "2026-09-20T04:26:06.601613Z", + "iopub.status.idle": "2026-09-20T04:26:06.606921Z", + "shell.execute_reply": "2026-09-20T04:26:06.606542Z" } }, "outputs": [ @@ -530,6 +586,12 @@ "ExecuteTime": { "end_time": "2026-05-21T09:37:14.027512Z", "start_time": "2026-05-21T09:37:13.983736Z" + }, + "execution": { + "iopub.execute_input": "2026-09-20T04:26:06.607883Z", + "iopub.status.busy": "2026-09-20T04:26:06.607809Z", + "iopub.status.idle": "2026-09-20T04:26:06.613216Z", + "shell.execute_reply": "2026-09-20T04:26:06.612870Z" } }, "outputs": [ @@ -583,6 +645,12 @@ "ExecuteTime": { "end_time": "2026-05-21T09:37:14.138368Z", "start_time": "2026-05-21T09:37:14.028932Z" + }, + "execution": { + "iopub.execute_input": "2026-09-20T04:26:06.614173Z", + "iopub.status.busy": "2026-09-20T04:26:06.614116Z", + "iopub.status.idle": "2026-09-20T04:26:06.692164Z", + "shell.execute_reply": "2026-09-20T04:26:06.691720Z" } }, "outputs": [ @@ -665,6 +733,12 @@ "ExecuteTime": { "end_time": "2026-05-21T09:37:14.163560Z", "start_time": "2026-05-21T09:37:14.139950Z" + }, + "execution": { + "iopub.execute_input": "2026-09-20T04:26:06.693274Z", + "iopub.status.busy": "2026-09-20T04:26:06.693200Z", + "iopub.status.idle": "2026-09-20T04:26:06.695375Z", + "shell.execute_reply": "2026-09-20T04:26:06.695020Z" } }, "outputs": [ @@ -672,7 +746,7 @@ "name": "stdout", "output_type": "stream", "text": [ - "remote URLPath: \n", + "remote URLPath: \n", "Set RUN_REMOTE = True to open a live C2Array from a Caterva2 service.\n" ] } @@ -749,6 +823,12 @@ "ExecuteTime": { "end_time": "2026-05-21T09:37:14.194158Z", "start_time": "2026-05-21T09:37:14.173549Z" + }, + "execution": { + "iopub.execute_input": "2026-09-20T04:26:06.696417Z", + "iopub.status.busy": "2026-09-20T04:26:06.696348Z", + "iopub.status.idle": "2026-09-20T04:26:06.698477Z", + "shell.execute_reply": "2026-09-20T04:26:06.698196Z" } }, "outputs": [ @@ -756,7 +836,7 @@ "name": "stdout", "output_type": "stream", "text": [ - "removed workdir: /var/folders/tb/7hwq2y354bb_68xwxjwjwwlr0000gn/T/blosc2-containers-nugha8ad\n" + "removed workdir: /var/folders/tb/7hwq2y354bb_68xwxjwjwwlr0000gn/T/blosc2-containers-tpemrqn5\n" ] } ], @@ -783,7 +863,7 @@ "name": "python", "nbconvert_exporter": "python", "pygments_lexer": "ipython3", - "version": "3.13.5" + "version": "3.14.4" } }, "nbformat": 4, diff --git a/examples/ctable/remote_handling.py b/examples/ctable/remote_handling.py index 8b61adb4b..c2e00d0a6 100644 --- a/examples/ctable/remote_handling.py +++ b/examples/ctable/remote_handling.py @@ -105,6 +105,7 @@ def write_table(args) -> None: }, validate=False, ) + table.create_index("tags", kind="membership") with blosc2.CTable.open(str(output)) as table: assert len(table) == args.rows @@ -188,6 +189,24 @@ def access_table(args) -> None: ) print(f" {messages}") + if ( + "tags" in table.col_names + and table._get_index_catalog().get("tags", {}).get("kind") == "membership" + ): + before_bytes = table.traffic.nbytes if remote else 0 + before_requests = table.traffic.requests if remote else 0 + started = time.perf_counter() + matching_ids = table[table["tags"].contains(3)]["id"][:5] + elapsed = time.perf_counter() - started + extra_time += elapsed + print("\nIndexed list membership (posting batches; tags stay unopened for an id projection):") + print( + f" - contains(3) : {elapsed * 1000:7.1f} ms " + f"({table.traffic.requests - before_requests if remote else 0} requests, " + f"{(table.traffic.nbytes - before_bytes if remote else 0) / 1024:8.2f} KB transferred)" + ) + print(f" first ids: {matching_ids}") + if "region" in table.col_names: dictionary = table["region"].raw print("\nDictionary costs (codes first, then full vocabulary on first decode):") diff --git a/plans/ctable-varlen-v2.md b/plans/ctable-varlen-v2.md new file mode 100644 index 000000000..5ab96941a --- /dev/null +++ b/plans/ctable-varlen-v2.md @@ -0,0 +1,402 @@ +# CTable variable-length columns: ListArray V2 + +Status: implemented and verified on 2026-09-20. + +## Objective and scope + +Extend ListArray and list-valued CTable columns with four features, available +locally and through read-only RemoteCTable: + +1. Nullable elements inside a list, independently of whole-cell nullability. +2. Nested lists with recursive schema validation and serialization. +3. List-aware `contains` and `overlaps` predicates. +4. Optional membership indexes that avoid scanning list payloads for selective + queries, including remote queries. + +Build on [remote-ctable-batches.md](remote-ctable-batches.md). Reuse ListArray, +BatchArray, the existing CTable query/index lifecycle and remote transport/cache +machinery. No new public remote array class or required dependency is intended. +V2 here names this feature proposal; persist a new format version only where +needed to prevent older readers from misinterpreting new semantics. + +Remote coverage in this proposal is `storage="batch"`, with MessagePack and +optional Arrow serializers. Local `storage="vl"` must support the new logical +values and scan predicates too. Remote ObjectArray transport is a separate +extension: it is not required to deliver these four features remotely. + +Sorting/grouping of list values, arbitrary list expressions, automatic byte-based +batch tuning, and a general Arrow-native execution engine remain follow-ups. + +## Current implementation and boundaries + +- `schema.py`: ListSpec rejects nested ListSpec items and carries storage, + serializer and batch settings. `schema_compiler.py` validates annotations; + recursive list annotations need corresponding handling. +- `list_array.py`: scalar coercion rejects `None` items; schema validation rejects + nesting. Arrow type conversion is not recursive for lists. Reads, copy and + mutation must preserve the new values through both existing backends. +- `batch_array.py` and `msgpack_utils.py`: reuse serializers and safe remote + decoding. MessagePack already represents nested lists and null values; schema + support and validation are still required. +- `ctable.py`: Column/query integration, physical-row selection, Arrow/Parquet + conversion, copies, deletes and compaction are shared local/remote boundaries. +- `ctable_indexing.py`: index descriptors currently accept summary, bucket, + partial, full and opsi kinds. Membership needs its own descriptor and lookup + semantics, while reusing lifecycle and storage facilities. +- `ctable_storage.py`: RemoteTableStorage currently returns an empty index + catalog. Selectively expose supported membership descriptors and resolve their + companions inside the archive; do not enable existing local-path index openers. +- `remote_batch.py`, `ctable_remote_read.py`, `remote_array.py`, `b2z_source.py` + and the owner cache coordinator provide range reads, scheduling and caching. + +Trace shared callers before editing. Avoid a general storage or query framework +refactor. The index transport work is the largest part of the remote extension; +nullable and nested values mainly extend schema and decoding behavior. + +## Schema and compatibility + +Use the child spec's `nullable` flag for element nullability and ListSpec's own +`nullable` flag for the list value at that level. Proposed examples: + +```python +# None, [], [1, None, 3] +blosc2.list(blosc2.int32(nullable=True), nullable=True, batch_rows=2048) + +# None, [], [None, [], [1, None, 3]] +blosc2.list( + blosc2.list(blosc2.int32(nullable=True), nullable=True), + nullable=True, + batch_rows=2048, +) +``` + +Inner ListSpec nodes describe values, not independently stored containers. Only +the outer column's storage, serializer, batch_rows and items_per_block control +physical storage. Reject nondefault inner storage options with an actionable +error rather than silently ignoring an attempted layout configuration. + +Validate recursively, including nullable struct fields inside list items, with +errors identifying the column and nested item position. Keep `None`, `[]`, +`[None]`, `[[]]` and `[None, []]` distinct. Update annotation checking, schema +metadata round trips, display labels and validation bypass paths used by trusted +imports. Do not let table-wide sentinel null policies replace null list items. + +Read existing V1 data unchanged. Audit existing child-nullable metadata before +deciding how to normalize it: V1's validator rejected null items even where a +child spec carried a nullable flag. V2 writers must mark schemas requiring new +semantics explicitly; unknown versions/features must fail clearly. Verify actual +V1 reader rejection of the chosen marker, rather than assuming a version bump +alone enforces it. Never silently rewrite an existing archive on read. + +## Serialization and batch geometry + +MessagePack retains recursive values as passive data. Extend Arrow schema +conversion recursively and preserve list value-field nullability, nested list +nullability and struct field nullability during import/export. Handle regular and +large list inputs explicitly; reject unsupported Arrow layouts with a useful +error. Arrow remains optional, and MessagePack must not import it. + +Keep one persisted batch per compressed BatchArray chunk. Nested child lists +remain inside that chunk. Remote scalar reads therefore still fetch the entire +containing batch; reading one member inside a list does not imply partial-list +transport. `storage="batch"` does not imply Arrow: MessagePack is the current +default serializer. + +Change the default to `batch_rows=2048` for newly constructed ListSpec, +`blosc2.list()` and ListArray values, including `ListArray.from_arrow` and CTable +construction paths. Keep Arrow/Parquet import defaults consistent, documenting +any intentional import-specific override. Inner ListSpec defaults do not create +separate batches; only the outer storage configuration applies. This option has +no batching effect on local `storage="vl"` columns. + +Retain explicit `batch_rows=None` as an opt-in to caller-managed boundaries: +ordinary append/extend buffers rows until a flush, and the flush persists all +pending rows as one batch. Explain the resulting memory and remote overfetch +costs. With the finite default, full batches flush automatically, but remaining +rows still need `flush()`, `close()` or context-manager exit before reopening a +standalone ListArray. Do not imply that automatic batching bounds the temporary +input allocation of a large `extend()` call. + +Keep construction defaults separate from metadata decoding: reopening an old +schema with no batch_rows field must preserve its previous None semantics and +existing batch boundaries. Preserve explicit None through metadata round trips +(encode it explicitly or use an equally unambiguous version-aware rule). Do not +silently apply 2048 when decoding legacy metadata or copying an existing spec. + +Audit `extend_arrow`, import and rewrite paths so a configured row target is +honored consistently, splitting incoming batches where necessary. Flush may +produce a smaller batch; persist actual lengths rather than deriving row lookup +from the target. A single huge list can still produce a huge chunk. No byte or +decoded-memory bound follows from a row-count target. + +## Required documentation updates + +Update existing documentation as part of implementation, not just this plan: + +- `doc/reference/list_array.rst` and public docstrings in `list_array.py` and + `schema.py`: document the 2048-row default, positive overrides, explicit None, + automatic full-batch flushing and persistence of the partial final batch. + Repair the reference example to flush/close or exit a context manager before + reopening the file. Explain that the unit is list cells, not child elements + or bytes, and that batch_rows does not control VL storage. +- `doc/tutorials/11.containers.ipynb`: explain the default alongside the existing + small explicit batch example. Demonstrate safe reopening with a partial final + batch and explain when caller-managed None is useful. Execute the changed + example and refresh its outputs. +- `doc/reference/ctable.rst` and relevant existing CTable/import tutorials: + document list-column defaults, overrides and table-managed persistence; + distinguish input/import batch size from persisted list batch size. Audit + existing examples for assumptions about None or reopen without flushing. +- `doc/reference/remotectable.rst`: explain whole-batch transfer, the finite + default, the smaller-batch versus throughput tradeoff, and why one very large + list can still require a large download. Link to the ListArray batching guide. +- Add a release/migration note for the changed construction default and preserved + legacy metadata behavior. Keep signatures, generated API docs and examples + consistent; old archives do not acquire new physical boundaries on reopen. + +## Predicate contract + +The following is proposed API, not currently implemented syntax: + +```python +has_python = table["tags"].contains("python") +has_either = table["tags"].overlaps(["python", "numpy"]) +selected = table[has_either] +``` + +Provide equivalent behavior on ListArray and CTable Column. Column predicates +must enter the existing table expression/selection machinery lazily and compose +with `&`, `|` and `~`, including predicates on scalar columns. Reuse existing +expression classes where possible; string-expression syntax is not required. + +Define membership over immediate children, with no implicit recursive flattening. +For `[[1, 2], [3]]`, `contains([1, 2])` is true and `contains([1])` is false; +`contains(1)` is a type error. `overlaps([[3], [4]])` is true. Structural equality +is recursive, ordered and exact for list children, including null positions. +Scalar literals must be validated against the child type without lossy coercions. +Define float NaN as unequal, consistent with ordinary numeric equality, and +normalize signed zero for indexing. Validate literals even on empty inputs. + +Use explicit two-valued membership results: a null outer list or an empty list +matches neither operation; null child values match an explicit `None` literal +when the child schema permits it. `overlaps([])` is false for every row. +Repeated child values or query literals never duplicate result rows. Negation is +ordinary Boolean negation, so negating a false result includes a null outer row. +Document this behavior separately from scalar-column null comparisons and test +mixed expressions. Existing whole-cell null predicates remain available. + +Implement the scan baseline first. Iterate batches once per operation, preserve +physical row positions internally, and apply live-row/view selection correctly. +On remote tables, a scan downloads all relevant uncached list batches; it is +client-side execution, not a server-side filter. Bound transient decoded batches +and avoid materializing the entire list column as Python values. + +## Membership indexes + +Add an opt-in membership index kind through the existing index API. Proposed use: + +```python +table.create_index("tags", kind="membership") +``` + +Initially index flat lists of supported scalar child types: Boolean, integers, +floats, strings and bytes, including nullable items. Finalize a typed canonical +key encoding shared by scans and indexes; never use Python's randomized hash as +a persisted key. Exclude NaN from postings because it never matches. Keep a +distinct null-item key. Struct and nested-list membership uses the scan path in +V2; index creation for those types reports the limitation explicitly. All four +features are available remotely, but indexing every possible nested value is +not a V2 requirement. + +Store each key's sorted, unique physical row IDs as a posting list. Repeated +values within one cell generate one posting. `contains` looks up one posting; +`overlaps` unions postings. Apply the table's live-row mask and the current view +selection. Complements use the live-row universe, never unused capacity. + +Use existing compressed storage primitives for index companions. Persist a +versioned descriptor recording the stable typed key encoding, source revision, +row extent and companion identity. The initial layout keeps the sorted typed key +directory in CTable metadata and stores one compressed BatchArray posting batch +per distinct value. Exact key comparison is required; hashes alone never +establish membership. This makes selective remote lookup one metadata read plus +the requested posting batches. Paging the key directory is deferred until real +high-cardinality measurements justify the extra format and lookup machinery. + +The initial builder accumulates unique physical row IDs per typed key in memory, +then publishes the complete posting store before activating its descriptor. +Reuse create/drop/rebuild, packaging and stale-index machinery. +Initially mark indexes stale after changes rather than adding incremental +posting updates. Audit append, assignment (including raw column access), schema +changes, deletes, compaction, copy and save. If a mutation path cannot reliably +invalidate a descriptor, fix that before enabling indexed queries. Compaction +changes physical row IDs and must invalidate or rebuild postings. + +Missing or stale indexes use a scan. Malformed advertised indexes must raise an +actionable integrity error rather than return incomplete results. Offer the +existing scan-forcing/debug idiom where applicable so parity is testable. + +## Efficient RemoteCTable execution + +Expose valid supported membership descriptors lazily through RemoteTableStorage. +Resolve companions relative to the table's archive namespace, including tables +nested in RemoteStore. Validate member uniqueness, role, bounds, versions, +schema/revision agreement, key ordering, posting offsets and row-ID ranges. +Do not interpret remote metadata as local paths, URLs or executable objects. +Other persisted index kinds remain disabled unless separately implemented. + +The selective query path is: + +1. Read the descriptor and bounded directory metadata. +2. Fetch only key pages needed for the literals and chunks covering their postings. +3. Merge/intersect positions with live rows and any other query candidates. +4. Fetch only requested output columns, grouping selected rows by physical batch + or NDArray block and restoring requested order/duplicates at the output. + +An exact index-only membership selection must not fetch list data payloads merely +to recheck membership. If the result projects the list column, fetch its matching +batches. A query projecting only scalar columns need not fetch list batches at +all. Missing terms need no posting or list payload reads. Counts may be derived +from postings after live-row/view filtering, without materializing list cells. + +Compose candidates conservatively: intersections can narrow an AND scan, but OR +must include both branches and NOT requires the correct live-row universe. +Provide a simple cost decision using posting lengths and candidate batch coverage +to avoid expensive index traversal for broad predicates. Index availability is +not a promise of lower transfer: scattered matches may touch every output batch. + +Reuse the owner's NONE/MEMORY/DISK caches, identity/generation keys, eviction, +traffic accounting and bounded scheduler for key pages and postings. No separate +unbounded index cache or per-column payload budget. Group repeated literals and +chunk reads within one operation even under NONE. Decode and mutate caches on +the owning thread. Large postings must be streamable; scheduling limits do not +bound a caller-requested full result mask or decoded Python result. + +Preserve read-only enforcement, leases, close/refresh invalidation and safe +MessagePack decoding at all nesting levels. Cached predicates and index handles +must check lifetime too. No implicit full-archive localization, remote index +creation or remote writes. Local materialization produces ordinary writable +tables with valid rebuilt indexes or no index, never stale source descriptors. + +## Implementation sequence + +1. **Define recursive schemas and compatibility.** Implement recursive ListSpec + validation, child nullability, annotation handling and metadata feature/version + checks. Add V1 fixtures and rejection tests before emitting V2 metadata. + +2. **Deliver nullable and nested values locally.** Extend coercion and both + backends; update recursive Arrow conversion and import/export. Audit batching, + mutation, copy/save and null handling. Verify distinct null/empty forms and + configured batch geometry across Python and Arrow writes. Implement the 2048 + construction default and explicit None behavior, preserving legacy metadata. + Update batching docstrings and the reference/tutorial examples with this step. + +3. **Deliver nullable and nested remote reads.** Reuse the remote batch reader + for both serializers. Verify sparse reads, metadata-only opening, safe recursive + decoding, wrapper lifetime and local exports against the same local archive. + This milestone delivers the first two features remotely. + +4. **Add shared list predicate scans.** Implement the stated truth table and + nested structural comparisons, lazy Column integration and Boolean composition. + Use batch-grouped scans locally and remotely, with live-row/view correctness. + This supplies the correctness oracle for every indexed path. + +5. **Build and persist local membership indexes.** Add the descriptor kind, typed + keys, paged catalog, chunked postings and bounded builder. Integrate lifecycle, + invalidation and archive packaging; verify indexed/scan parity, including null + items, duplicates and updates. Record the exact on-disk layout in developer docs. + +6. **Prove selective remote index lookup.** Resolve membership companions through + the archive and expose only supported descriptors. On a large instrumented + fixture, look up distant terms and fetch their postings without downloading + the catalog, list column or archive in full. Integrate aggregate cache budgets, + scheduling, safe metadata validation and generation checks. + +7. **Integrate remote indexed query planning.** Connect positions to projections, + mixed predicates, views, counts and local materialization. Handle AND/OR/NOT, + scan fallback and broad-query cost decisions. Demonstrate an indexed selection + projecting scalar columns with zero list payload reads, then list projection + fetching only required batches. + +8. **Document and validate the complete feature set.** Update ListArray/CTable + schema, predicate, index and remote support documentation, completing every + item in Required documentation updates. Add a runnable local + and remote example, state nesting/index restrictions and batch costs, and record + the cold/warm measurements below. Update related plans only to reflect features + actually delivered. Run focused checks and the default suite. + +## Verification and acceptance + +Use the `blosc2` conda environment for Python, tests and builds. Extend existing +list, schema, Arrow/Parquet, CTable indexing and remote test modules; reuse the +instrumented fsspec memory filesystem and deterministic HTTP range server. + +- Round-trip null outer lists, null scalar items, null inner lists, empty lists + at every level, nullable struct children, multilingual strings and bytes. + Cover both serializers, local VL, persistence, reopen and V1 compatibility. +- Verify default batch lengths using 4097 rows: two automatic 2048-row batches + plus a final one-row batch after flushing. Cover explicit positive overrides, + explicit None, legacy missing-field decoding, new metadata round trips, and + Python/Arrow/CTable creation paths. Confirm schema copy preserves the setting. + Execute updated tutorial examples and check that reopened data includes the + partial final batch; review generated docs for consistent defaults and links. +- Test invalid depths/types, child constraints, unsupported versions, malformed + metadata/payloads, unsafe MessagePack extensions and missing optional Arrow. + Bound schema recursion and validate decoded structure before trusting offsets + or lengths; document the supported depth limit. +- Compare scans and indexes with a simple reference implementation for empty, + duplicate-heavy and randomized data. Include integer limits, signed zero, NaN, + null literals, structural comparisons and invalid literals on empty tables. +- Cover deleted rows, capacity, scalar/slice/strided/reverse/fancy selections, + filtered views, mixed expressions, projections, counts and detached exports. + Test each mutation's index invalidation and interrupted index publication. +- Check lazy open, NONE/MEMORY/DISK, mixed index/column eviction, disk reopen, + source identity isolation, stale descriptors, missing companions, malformed + postings and close/refresh behavior even on warm cached results. +- Use payloads larger than metadata prefetch thresholds. Assert selective index + lookup reads only needed pages/posting chunks plus bounded metadata; scalar-only + projection reads no list payload, list projection reads only matching batches, + and a warm query fitting cache needs no new payload requests. Include absent + terms, huge postings and matches scattered across most batches. +- Record dataset/archive/index sizes, cardinality, list length distribution, + null rate, batch geometry, selectivity, cold/warm requests and bytes, decoded + memory and elapsed time. Compare scan and index locally and remotely, separating + index traffic from output-column traffic. No universal latency target is needed. +- Run focused list/schema/serialization/index/remote tests, then the default + suite. Run Ruff for changed Python and build affected documentation. Record + any pre-existing failures separately. + +Completion means the four features meet their documented type coverage locally +and remotely, indexed and scan semantics agree, and selective remote membership +queries demonstrably avoid list scans. Remote VL access, indexes on structural +children, sorting/grouping and finer-than-batch list transport remain explicit +follow-ups. + +## Implementation results + +The four requested features were delivered as separate commits: + +1. `fd4f3bcc` — nullable list elements, plus the 2048-row construction default + and backward-compatible explicit/legacy None handling. +2. `e49753d5` — recursive nested lists, recursive Arrow types and remote reads for + MessagePack and Arrow batches. +3. `1ce260b7` — immediate-child `contains` and `overlaps` predicates shared by + standalone ListArray, CTable and RemoteCTable. +4. `1caa94c7` — stable typed membership keys, compressed posting batches, + mutation invalidation and selective RemoteCTable posting reads. + +The implementation reuses the existing remote BatchArray source and aggregate +NONE/MEMORY/DISK cache for posting payloads. A remote indexed predicate that +projects only scalar columns does not open or transfer the source list column. +Absent terms return an empty posting result without scanning list batches. + +The final implementation deliberately keeps the sorted key directory in table +metadata and builds postings in memory. This is the smallest format that provides +selective remote payload reads and exact typed-key semantics. Directory paging +and spill/merge construction remain follow-ups for measured high-cardinality +indexes; they do not change the public predicate or index API. + +Focused ListArray, schema, Arrow/Parquet, CTable indexing and RemoteCTable suites +passed with 400 tests after one explicit-None import regression was repaired. +The default suite passed with 10,330 tests and 36 skips. Ruff and pre-commit +passed, the container tutorial executed successfully, and the normal Sphinx HTML +build completed with the repository's existing warnings. diff --git a/src/blosc2/ctable.py b/src/blosc2/ctable.py index f437ccca9..032f12670 100644 --- a/src/blosc2/ctable.py +++ b/src/blosc2/ctable.py @@ -9497,13 +9497,11 @@ def from_arrow( # noqa: C901 if cls._is_list_column(col): if getattr(col.spec, "storage", None) == "batch": col.spec.serializer = list_serializer - if blosc2_batch_size is not None: - col.spec.batch_rows = blosc2_batch_size + col.spec.batch_rows = blosc2_batch_size if blosc2_items_per_block is not None: col.spec.items_per_block = blosc2_items_per_block elif cls._is_varlen_scalar_column(col): - if blosc2_batch_size is not None: - col.spec.batch_rows = blosc2_batch_size + col.spec.batch_rows = blosc2_batch_size if blosc2_items_per_block is not None: col.spec.items_per_block = blosc2_items_per_block metadata = cls._arrow_schema_metadata(schema) diff --git a/src/blosc2/ctable_indexing.py b/src/blosc2/ctable_indexing.py index 71171a254..e04e490e0 100644 --- a/src/blosc2/ctable_indexing.py +++ b/src/blosc2/ctable_indexing.py @@ -527,7 +527,29 @@ def _membership_positions(self, col_name: str, values) -> np.ndarray | None: return None self._validate_index_descriptor(col_name, descriptor) payload = descriptor["membership"] - directory = {key: (int(index), int(count)) for key, index, count in payload.get("keys", [])} + raw_keys = payload.get("keys", []) + if not isinstance(raw_keys, list): + raise ValueError(f"Malformed membership index for column {col_name!r}: invalid key directory") + directory = {} + previous = None + for entry in raw_keys: + if ( + not isinstance(entry, (list, tuple)) + or len(entry) != 3 + or not isinstance(entry[0], bytes) + or isinstance(entry[1], bool) + or not isinstance(entry[1], int) + or entry[1] < 0 + or isinstance(entry[2], bool) + or not isinstance(entry[2], int) + or entry[2] < 0 + or (previous is not None and entry[0] <= previous) + ): + raise ValueError( + f"Malformed membership index for column {col_name!r}: invalid key directory" + ) + directory[entry[0]] = (entry[1], entry[2]) + previous = entry[0] spec = self._schema.columns_by_name[col_name].spec wanted = {key for value in values if (key := list_item_key(spec.item_spec, value)) is not None} indexes = sorted({directory[key][0] for key in wanted if key in directory}) @@ -545,7 +567,15 @@ def _membership_positions(self, col_name: str, values) -> np.ndarray | None: raise ValueError(f"Malformed membership index for column {col_name!r}: missing postings") store = blosc2.open(path, mode="r") chunks = [store[index][:] for index in indexes] - return np.unique(np.concatenate([np.asarray(chunk, dtype=np.int64) for chunk in chunks])) + counts = dict(directory.values()) + for index, chunk in zip(indexes, chunks, strict=True): + expected = counts[index] + if len(chunk) != expected: + raise ValueError(f"Malformed membership index for column {col_name!r}: posting length") + positions = np.unique(np.concatenate([np.asarray(chunk, dtype=np.int64) for chunk in chunks])) + if positions.size and (positions[0] < 0 or positions[-1] >= len(self._valid_rows)): + raise ValueError(f"Malformed membership index for column {col_name!r}: posting row range") + return positions def _normalize_table_expression_target( self, expression: str, operands: dict | None = None diff --git a/src/blosc2/schema_compiler.py b/src/blosc2/schema_compiler.py index cba9572fd..745d47f05 100644 --- a/src/blosc2/schema_compiler.py +++ b/src/blosc2/schema_compiler.py @@ -433,6 +433,9 @@ def spec_from_metadata_dict(data: dict[str, Any]) -> SchemaSpec: data["null_value"] = _json_to_bytes(data["null_value"]) if kind == "list": item_spec = spec_from_metadata_dict(data.pop("item")) + # Before ListArray V2 the field was omitted to mean caller-managed + # batching. Preserve that meaning when reading legacy schemas. + data.setdefault("batch_rows", None) return ListSpec(item_spec, **data) if kind == "struct": return StructSpec.from_metadata_dict({"fields": data.pop("fields"), **data}) diff --git a/tests/ctable/test_varlen_schema_compiler.py b/tests/ctable/test_varlen_schema_compiler.py index ba3a56f8b..560e323c3 100644 --- a/tests/ctable/test_varlen_schema_compiler.py +++ b/tests/ctable/test_varlen_schema_compiler.py @@ -39,6 +39,16 @@ def test_list_schema_roundtrip(): assert restored.columns_by_name["tags"].spec.batch_rows == 32 +def test_legacy_list_schema_without_batch_rows_keeps_caller_managed_batches(): + restored = schema_from_dict( + { + "version": 1, + "columns": [{"name": "tags", "kind": "list", "item": {"kind": "int32"}}], + } + ) + assert restored.columns_by_name["tags"].spec.batch_rows is None + + def test_list_annotation_mismatch_rejected(): @dataclass class Bad: From 7a53f6e164ead3f3ada393bb1fd0a42bcd8dde1c Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 18:12:10 +0200 Subject: [PATCH 25/82] Add source-bound CTable columns --- src/blosc2/ctable.py | 103 +++++++++++++++++++++++++++- tests/ctable/test_remote_columns.py | 77 +++++++++++++++++++++ 2 files changed, 177 insertions(+), 3 deletions(-) create mode 100644 tests/ctable/test_remote_columns.py diff --git a/src/blosc2/ctable.py b/src/blosc2/ctable.py index 032f12670..4f34eb51f 100644 --- a/src/blosc2/ctable.py +++ b/src/blosc2/ctable.py @@ -4849,11 +4849,12 @@ def _live_positions_from_valid_rows_chunks(self) -> np.ndarray: self._cached_live_positions = result return result - def __init__( + def __init__( # noqa: C901 self, row_type: type[RowT], new_data=None, *, + sources: Mapping[str, blosc2.NDArray | blosc2.RemoteArray] | None = None, urlpath: str | None = None, mode: str = "a", expected_size: int | None = None, @@ -4867,6 +4868,12 @@ def __init__( Parameters ---------- + sources: + Mapping from every stored column name to an existing + :class:`NDArray` or :class:`RemoteArray`. The arrays are bound + without copying for an in-memory table. Their dtypes and shapes + must exactly match the schema, and the resulting table is + read-only. create_summary_index: If ``True`` (default), SUMMARY indexes are automatically built for all eligible scalar columns. These indexes are extremely cheap to @@ -4887,6 +4894,11 @@ def __init__( logical copy and do **not** trigger the build; index the source table (or the reopened result) explicitly if you need it. """ + if sources is not None and new_data is not None: + raise ValueError("sources and new_data are mutually exclusive") + if sources is not None and not isinstance(sources, Mapping): + raise TypeError("sources must be a mapping from column names to arrays") + # Auto-size: if the caller didn't specify expected_size and new_data has a # known length, pre-allocate just enough (×2 for headroom, min 64). # Fall back to 1 M when new_data has no __len__ or is absent. @@ -4928,9 +4940,10 @@ def __init__( if storage.table_exists() and mode != "w": # ---- Open existing persistent table ---- - if new_data is not None: + if new_data is not None or sources is not None: raise ValueError( - "Cannot pass new_data when opening an existing table. Use mode='w' to overwrite." + "Cannot pass new_data or sources when opening an existing table. " + "Use mode='w' to overwrite." ) storage.check_kind() schema_dict = storage.load_schema() @@ -4978,6 +4991,27 @@ def __init__( self._schema = _compile_pydantic_schema(row_type) self._resolve_nullable_specs(self._schema) + if sources is not None: + n_rows = self._validate_sources(sources) + capacity = max(n_rows, 1) + default_chunks, default_blocks = compute_chunks_blocks((capacity,)) + self._valid_rows = storage.create_valid_rows( + shape=(capacity,), chunks=default_chunks, blocks=default_blocks + ) + if n_rows: + self._valid_rows[:n_rows] = True + self._n_rows = n_rows + self._last_pos = n_rows + for col in self._schema.columns: + self.col_names.append(col.name) + self._col_widths[col.name] = max(len(col.name), col.display_width) + self._cols[col.name] = storage.install_column(col.name, sources[col.name]) + self._read_only = True + self._create_summary_index = False + self._summary_indexes_built = True + storage.save_schema(self._schema_dict_with_computed()) + return + self._n_rows = 0 self._last_pos = 0 @@ -5010,6 +5044,69 @@ def __init__( # _valid_rows intersection in where(). self._save_n_rows_to_meta() + def _validate_sources(self, sources: Mapping[str, Any]) -> int: # noqa: C901 + """Validate source bindings without reading their payloads.""" + expected = {col.name for col in self._schema.columns} + supplied = set(sources) + missing = expected - supplied + unknown = supplied - expected + if missing or unknown: + details = [] + if missing: + details.append(f"missing: {', '.join(sorted(missing))}") + if unknown: + details.append(f"unknown: {', '.join(sorted(unknown))}") + raise ValueError( + "sources must bind every stored column exactly once (" + "; ".join(details) + ")" + ) + if self._table_cparams is not None or self._table_dparams is not None: + raise ValueError("cparams and dparams cannot be specified with sources") + + n_rows = None + for col in self._schema.columns: + source = sources[col.name] + if not isinstance(source, (blosc2.NDArray, blosc2.RemoteArray)): + raise TypeError( + f"Source for column {col.name!r} must be an NDArray or RemoteArray, " + f"got {type(source).__name__}" + ) + if ( + self._is_list_column(col) + or self._is_varlen_scalar_column(col) + or self._is_dictionary_column(col) + ): + raise TypeError(f"Source binding for column {col.name!r} requires a fixed-width schema") + if getattr(col.spec, "uses_mask", False): + raise TypeError(f"Source binding for nullable mask column {col.name!r} is not supported") + if self._validate and any( + getattr(col.spec, constraint, None) is not None for constraint in ("ge", "gt", "le", "lt") + ): + raise ValueError( + f"Source binding for constrained column {col.name!r} requires validate=False; " + "source values are not scanned during construction" + ) + if any( + option is not None + for option in (col.config.chunks, col.config.blocks, col.config.cparams, col.config.dparams) + ): + raise ValueError(f"Storage options for source-bound column {col.name!r} are not supported") + + shape = tuple(source.shape) + wanted = self._column_physical_shape(col, shape[0] if shape else 0) + if shape != wanted: + raise ValueError(f"Source for column {col.name!r} has shape {shape}, expected {wanted}") + if np.dtype(source.dtype) != np.dtype(col.dtype): + raise TypeError( + f"Source for column {col.name!r} has dtype {source.dtype}, expected {np.dtype(col.dtype)}" + ) + if n_rows is None: + n_rows = shape[0] + elif shape[0] != n_rows: + raise ValueError( + f"Source columns have different row counts: {col.name!r} has {shape[0]}, expected {n_rows}" + ) + return 0 if n_rows is None else n_rows + def close(self) -> None: """Close any persistent backing store held by this table. diff --git a/tests/ctable/test_remote_columns.py b/tests/ctable/test_remote_columns.py new file mode 100644 index 000000000..1e5a7d38e --- /dev/null +++ b/tests/ctable/test_remote_columns.py @@ -0,0 +1,77 @@ +from dataclasses import dataclass +from uuid import uuid4 + +import numpy as np +import pytest + +import blosc2 + +fsspec = pytest.importorskip("fsspec") + + +@dataclass +class Row: + local: int = blosc2.field(blosc2.int32()) + remote: float = blosc2.field(blosc2.float32()) + + +def remote_array(values, *, chunks=None, cache_policy=blosc2.CachePolicy.NONE): + values = np.asarray(values) + array = blosc2.asarray(values, chunks=chunks) + name = f"ctable-sources-{uuid4().hex}.b2nd" + fsspec.filesystem("memory").pipe_file(name, array.to_cframe()) + return blosc2.RemoteArray(f"memory://{name}", cache_policy=cache_policy) + + +def test_mixed_sources_read_query_and_borrowed_lifetime(): + local = blosc2.asarray(np.arange(8, dtype=np.int32), chunks=(3,)) + remote = remote_array(np.arange(8, dtype=np.float32) / 2, chunks=(5,)) + + table = blosc2.CTable(Row, sources={"local": local, "remote": remote}) + + np.testing.assert_array_equal(table.local[1:7:2], np.array([1, 3, 5], dtype=np.int32)) + np.testing.assert_array_equal(table[table.local >= 4].remote[:], np.array([2, 2.5, 3, 3.5])) + np.testing.assert_array_equal((table.local + table.remote)[:], np.arange(8) * 1.5) + with pytest.raises(ValueError, match="read-only"): + table.append((8, 4.0)) + + table.close() + np.testing.assert_array_equal(remote[:2], np.array([0, 0.5], dtype=np.float32)) + + +def test_source_validation_and_empty_sources(): + empty = blosc2.asarray(np.array([], dtype=np.int32)) + empty_remote = remote_array(np.array([], dtype=np.float32)) + table = blosc2.CTable(Row, sources={"local": empty, "remote": empty_remote}) + assert len(table) == 0 + + with pytest.raises(ValueError, match="missing: remote"): + blosc2.CTable(Row, sources={"local": empty}) + with pytest.raises(TypeError, match="dtype"): + blosc2.CTable( + Row, + sources={ + "local": blosc2.asarray(np.array([], dtype=np.int64)), + "remote": empty_remote, + }, + ) + with pytest.raises(ValueError, match="different row counts"): + blosc2.CTable( + Row, + sources={ + "local": blosc2.asarray(np.arange(2, dtype=np.int32)), + "remote": empty_remote, + }, + ) + + +def test_constrained_source_requires_validation_opt_out(): + @dataclass + class Constrained: + value: int = blosc2.field(blosc2.int32(ge=0)) + + source = blosc2.asarray(np.arange(3, dtype=np.int32)) + with pytest.raises(ValueError, match="validate=False"): + blosc2.CTable(Constrained, sources={"value": source}) + table = blosc2.CTable(Constrained, sources={"value": source}, validate=False) + np.testing.assert_array_equal(table.value[:], np.arange(3, dtype=np.int32)) From b2b57c9a68036e6b3d3d40ed91c9946e0549903c Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 18:14:52 +0200 Subject: [PATCH 26/82] Persist CTable source bindings --- src/blosc2/ctable.py | 136 ++++++++++++++++++++++++---- src/blosc2/ctable_storage.py | 14 ++- tests/ctable/test_remote_columns.py | 50 ++++++++++ 3 files changed, 179 insertions(+), 21 deletions(-) diff --git a/src/blosc2/ctable.py b/src/blosc2/ctable.py index 4f34eb51f..eba24466c 100644 --- a/src/blosc2/ctable.py +++ b/src/blosc2/ctable.py @@ -4923,8 +4923,19 @@ def __init__( # noqa: C901 self.auto_compact = compact self._create_summary_index = create_summary_index self._summary_indexes_built = False + self._source_bound = False + self._source_columns: set[str] = set() self.base = None + source_n_rows = None + if sources is not None: + if dataclasses.is_dataclass(row_type) and isinstance(row_type, type): + self._schema = compile_schema(row_type) + else: + self._schema = _compile_pydantic_schema(row_type) + self._resolve_nullable_specs(self._schema) + source_n_rows = self._validate_sources(sources) + # Choose storage backend if urlpath is not None: if mode == "w" and os.path.exists(urlpath): @@ -4947,6 +4958,7 @@ def __init__( # noqa: C901 ) storage.check_kind() schema_dict = storage.load_schema() + self._load_source_binding_metadata(schema_dict) self._schema: CompiledSchema = schema_from_dict(schema_dict) self._schema = CompiledSchema( row_cls=row_type, @@ -4974,6 +4986,10 @@ def __init__( # noqa: C901 # Restore auto-index preference from the schema. self._create_summary_index = schema_dict.get("create_summary_index", True) self._summary_indexes_built = schema_dict.get("summary_indexes_built", False) + if self._source_bound: + self._read_only = True + self._create_summary_index = False + self._summary_indexes_built = True else: # ---- Create new table ---- if storage.is_read_only(): @@ -4984,15 +5000,18 @@ def __init__( # noqa: C901 "use mode='w' to create a new one." ) - # Build compiled schema from either a dataclass or a legacy Pydantic model - if dataclasses.is_dataclass(row_type) and isinstance(row_type, type): - self._schema = compile_schema(row_type) - else: - self._schema = _compile_pydantic_schema(row_type) - self._resolve_nullable_specs(self._schema) + # Build compiled schema from either a dataclass or a legacy Pydantic model. + # Source-bound schemas were compiled before storage selection so a + # validation failure cannot overwrite a destination. + if sources is None: + if dataclasses.is_dataclass(row_type) and isinstance(row_type, type): + self._schema = compile_schema(row_type) + else: + self._schema = _compile_pydantic_schema(row_type) + self._resolve_nullable_specs(self._schema) if sources is not None: - n_rows = self._validate_sources(sources) + n_rows = source_n_rows capacity = max(n_rows, 1) default_chunks, default_blocks = compute_chunks_blocks((capacity,)) self._valid_rows = storage.create_valid_rows( @@ -5006,6 +5025,10 @@ def __init__( # noqa: C901 self.col_names.append(col.name) self._col_widths[col.name] = max(len(col.name), col.display_width) self._cols[col.name] = storage.install_column(col.name, sources[col.name]) + self._source_bound = True + self._source_columns = { + name for name, source in sources.items() if isinstance(source, blosc2.RemoteArray) + } self._read_only = True self._create_summary_index = False self._summary_indexes_built = True @@ -5107,6 +5130,17 @@ def _validate_sources(self, sources: Mapping[str, Any]) -> int: # noqa: C901 ) return 0 if n_rows is None else n_rows + def _load_source_binding_metadata(self, schema_dict: Mapping[str, Any]) -> None: + version = schema_dict.get("source_bindings_version") + if version is None: + self._source_bound = False + self._source_columns = set() + return + if version != 1: + raise ValueError(f"Unsupported CTable source bindings version: {version!r}") + self._source_bound = True + self._source_columns = set(schema_dict.get("source_columns", ())) + def close(self) -> None: """Close any persistent backing store held by this table. @@ -6862,7 +6896,14 @@ def open(cls, urlpath: str, *, mode: str = "r", mmap_mode: str | None = None) -> raise FileNotFoundError(f"No CTable found at {urlpath!r}") return cls._open_from_storage(storage) - def to_b2z(self, urlpath: str, *, overwrite: bool = False, compact: bool = False) -> str: + def to_b2z( + self, + urlpath: str, + *, + overwrite: bool = False, + compact: bool = False, + preserve_sources: bool = False, + ) -> str: """Write this table to a compact ``.b2z`` container. ``.b2z`` is the compact zip-backed CTable format. For persistent, @@ -6896,6 +6937,8 @@ def to_b2z(self, urlpath: str, *, overwrite: bool = False, compact: bool = False """ if not str(urlpath).endswith(".b2z"): raise ValueError("urlpath must have a .b2z extension") + if preserve_sources and self.base is not None: + raise ValueError("preserve_sources requires an unfiltered root table") storage = getattr(self, "_storage", None) can_physical_pack = ( @@ -6903,6 +6946,7 @@ def to_b2z(self, urlpath: str, *, overwrite: bool = False, compact: bool = False and self.base is None and isinstance(storage, FileTableStorage) and not str(storage._root).endswith(".b2z") + and (preserve_sources or not self._source_bound) ) if can_physical_pack: self._flush_varlen_columns() @@ -6916,10 +6960,17 @@ def to_b2z(self, urlpath: str, *, overwrite: bool = False, compact: bool = False materialized = self.copy(compact=True) materialized.save(urlpath, overwrite=overwrite) else: - CTable.save(self, urlpath, overwrite=overwrite) + CTable.save(self, urlpath, overwrite=overwrite, preserve_sources=preserve_sources) return os.path.abspath(urlpath) - def to_b2d(self, urlpath: str, *, overwrite: bool = False, compact: bool = False) -> str: + def to_b2d( + self, + urlpath: str, + *, + overwrite: bool = False, + compact: bool = False, + preserve_sources: bool = False, + ) -> str: """Write this table to a directory-backed store. Directory-backed CTable stores may use any path that does not end in @@ -6960,6 +7011,7 @@ def to_b2d(self, urlpath: str, *, overwrite: bool = False, compact: bool = False and isinstance(storage, FileTableStorage) and str(storage._root).endswith(".b2z") and storage.open_mode() == "r" + and (preserve_sources or not self._source_bound) ) if can_physical_unpack: store = blosc2.TreeStore(storage._root, mode="r") @@ -6972,10 +7024,10 @@ def to_b2d(self, urlpath: str, *, overwrite: bool = False, compact: bool = False materialized = self.copy(compact=True) materialized.save(urlpath, overwrite=overwrite) else: - CTable.save(self, urlpath, overwrite=overwrite) + CTable.save(self, urlpath, overwrite=overwrite, preserve_sources=preserve_sources) return os.path.abspath(urlpath) - def to_cframe(self) -> bytes: + def to_cframe(self, *, preserve_sources: bool = False) -> bytes: """Serialize this table to a bytes buffer (a CFrame). This is the Blosc2-bytes counterpart of :meth:`to_b2z`, mirroring @@ -7020,7 +7072,11 @@ def to_cframe(self) -> bytes: meta = blosc2.SChunk() meta.vlmeta["kind"] = "ctable" meta.vlmeta["version"] = 1 - meta.vlmeta["schema"] = json.dumps(src._schema_dict_with_computed()) + schema_dict = src._schema_dict_with_computed() + if not preserve_sources: + schema_dict.pop("source_bindings_version", None) + schema_dict.pop("source_columns", None) + meta.vlmeta["schema"] = json.dumps(schema_dict) estore["/_meta"] = meta estore["/_valid_rows"] = src._valid_rows @@ -7045,7 +7101,14 @@ def to_cframe(self) -> bytes: estore[key] = arr._backend else: # Scalar NDArray or ListArray — both serialize via to_cframe(). - estore[key] = arr + if isinstance(arr, blosc2.RemoteArray): + estore[key] = ( + arr._export_carrier(include_cache=False) + if preserve_sources + else blosc2.asarray(arr[:]) + ) + else: + estore[key] = arr # Validity sidecars travel beside their columns; a column without one # simply contributes no entry, which reconstructs as all-valid. @@ -7233,7 +7296,27 @@ def _save_to_storage( # noqa: C901 vlmeta.vlmeta[key] = value storage.save_vlmeta(vlmeta) - def save(self, urlpath: str, *, overwrite: bool = False) -> None: + def _save_sources_to_storage(self, storage: TableStorage) -> None: + """Persist identity-mapped source bindings without reading remote payloads.""" + if self.base is not None: + raise ValueError("preserve_sources requires an unfiltered root table") + n_rows = len(self) + capacity = max(n_rows, 1) + chunks, blocks = compute_chunks_blocks((capacity,)) + valid = storage.create_valid_rows(shape=(capacity,), chunks=chunks, blocks=blocks) + if n_rows: + valid[:n_rows] = True + for name in self.col_names: + storage.install_column(name, self._cols[name]) + storage.save_schema(self._schema_dict_with_computed()) + attrs = self.attrs[:] + if attrs: + vlmeta = blosc2.SChunk() + for key, value in attrs.items(): + vlmeta.vlmeta[key] = value + storage.save_vlmeta(vlmeta) + + def save(self, urlpath: str, *, overwrite: bool = False, preserve_sources: bool = False) -> None: """Persist this table to disk at *urlpath*. This writes a standalone copy and returns ``None``; use :meth:`copy` @@ -7259,6 +7342,8 @@ def save(self, urlpath: str, *, overwrite: bool = False) -> None: ValueError If *urlpath* already exists and ``overwrite=False``. """ + if preserve_sources and self.base is not None: + raise ValueError("preserve_sources requires an unfiltered root table") if self.base is not None: materialized = self.copy(compact=True) materialized.save(urlpath, overwrite=overwrite) @@ -7274,7 +7359,12 @@ def save(self, urlpath: str, *, overwrite: bool = False) -> None: else: os.remove(target_path) - self._save_to_storage(file_storage) + if preserve_sources: + if not self._source_bound: + raise ValueError("preserve_sources requires a source-bound CTable") + self._save_sources_to_storage(file_storage) + else: + self._save_to_storage(file_storage) file_storage.close() @classmethod @@ -7297,6 +7387,9 @@ def _open_from_storage(cls, storage: TableStorage) -> CTable: obj._table_dparams = None obj._storage = storage obj._read_only = storage.is_read_only() + obj._load_source_binding_metadata(schema_dict) + if obj._source_bound: + obj._read_only = True obj._schema = schema obj._cols = {} obj._col_widths = {} @@ -7304,6 +7397,9 @@ def _open_from_storage(cls, storage: TableStorage) -> CTable: obj.auto_compact = False obj._create_summary_index = schema_dict.get("create_summary_index", True) obj._summary_indexes_built = schema_dict.get("summary_indexes_built", False) + if obj._source_bound: + obj._create_summary_index = False + obj._summary_indexes_built = True obj.base = None obj._valid_rows = storage.open_valid_rows() @@ -7337,7 +7433,10 @@ def _save_to_treestore(self, store: blosc2.TreeStore, full_key: str) -> None: materialized._save_to_treestore(store, full_key) return storage = TreeStoreTableStorage(store, full_key, mode="a", owns_store=False) - self._save_to_storage(storage) + if self._source_bound: + self._save_sources_to_storage(storage) + else: + self._save_to_storage(storage) # storage is non-owning; outer store handles persistence @classmethod @@ -11925,6 +12024,9 @@ def _row_values_with_nulls(self, name: str, spec, positions: np.ndarray, values) def _schema_dict_with_computed(self) -> dict: """Return the schema dict extended with computed/materialized metadata.""" d = schema_to_dict(self._schema) + if getattr(self, "_source_bound", False): + d["source_bindings_version"] = 1 + d["source_columns"] = sorted(self._source_columns) n_rows = self._known_n_rows() if n_rows is not None: d["n_rows"] = n_rows diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index 23d8f27b6..93aab3d4e 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -69,7 +69,9 @@ def create_column( ) -> blosc2.NDArray: raise NotImplementedError - def install_column(self, name: str, ndarray: blosc2.NDArray) -> blosc2.NDArray: + def install_column( + self, name: str, ndarray: blosc2.NDArray | blosc2.RemoteArray + ) -> blosc2.NDArray | blosc2.RemoteArray: """Store a pre-built NDArray as column *name*, preserving its storage config. Faster than create_column + fill when the caller already has the fully @@ -268,7 +270,7 @@ def create_column(self, name, *, dtype, shape, chunks, blocks, cparams, dparams) kwargs["dparams"] = dparams return blosc2.zeros(shape, dtype=dtype, **kwargs) - def install_column(self, name, ndarray: blosc2.NDArray) -> blosc2.NDArray: + def install_column(self, name, ndarray: blosc2.NDArray | blosc2.RemoteArray): """Store a pre-built NDArray as column *name* (skips the zeros+fill pattern).""" return ndarray @@ -1011,7 +1013,7 @@ def create_column(self, name, *, dtype, shape, chunks, blocks, cparams, dparams) store[self._col_key(name)] = col return store[self._col_key(name)] - def install_column(self, name, ndarray: blosc2.NDArray) -> blosc2.NDArray: + def install_column(self, name, ndarray: blosc2.NDArray | blosc2.RemoteArray): """Store a pre-built NDArray as column *name* (skips the zeros+fill pattern).""" store = self._open_store() store[self._col_key(name)] = ndarray @@ -1596,7 +1598,11 @@ def create_column( self._store._modified = True return col - def install_column(self, name: str, ndarray: blosc2.NDArray) -> blosc2.NDArray: + def install_column(self, name: str, ndarray: blosc2.NDArray | blosc2.RemoteArray): + if isinstance(ndarray, blosc2.RemoteArray): + key = self._table_key(self._col_logical_key(name)) + self._store[key] = ndarray + return self._store[key] dest_path = self._dest_path(self._col_logical_key(name), ".b2nd") os.makedirs(os.path.dirname(dest_path), exist_ok=True) saved = ndarray.copy(urlpath=dest_path) diff --git a/tests/ctable/test_remote_columns.py b/tests/ctable/test_remote_columns.py index 1e5a7d38e..efc37b907 100644 --- a/tests/ctable/test_remote_columns.py +++ b/tests/ctable/test_remote_columns.py @@ -75,3 +75,53 @@ class Constrained: blosc2.CTable(Constrained, sources={"value": source}) table = blosc2.CTable(Constrained, sources={"value": source}, validate=False) np.testing.assert_array_equal(table.value[:], np.arange(3, dtype=np.int32)) + + +def test_save_materializes_by_default_and_can_preserve_sources(tmp_path): + values = np.arange(6, dtype=np.float32) + table = blosc2.CTable( + Row, + sources={ + "local": blosc2.asarray(np.arange(6, dtype=np.int32)), + "remote": remote_array(values), + }, + ) + + materialized_path = tmp_path / "materialized.b2z" + table.save(materialized_path) + materialized = blosc2.open(materialized_path) + assert isinstance(materialized._cols["remote"], blosc2.NDArray) + assert not isinstance(materialized._cols["remote"], blosc2.RemoteArray) + np.testing.assert_array_equal(materialized.remote[:], values) + + referenced_path = tmp_path / "referenced.b2z" + table.save(referenced_path, preserve_sources=True) + referenced = blosc2.open(referenced_path) + assert isinstance(referenced._cols["remote"], blosc2.RemoteArray) + assert referenced._read_only + np.testing.assert_array_equal(referenced.remote[:], values) + with pytest.raises(ValueError, match="read-only"): + referenced.append((6, 6.0)) + + +def test_treestore_and_cframe_preserve_sources(tmp_path): + values = np.arange(4, dtype=np.float32) + table = blosc2.CTable( + Row, + sources={ + "local": blosc2.asarray(np.arange(4, dtype=np.int32)), + "remote": remote_array(values), + }, + ) + + path = tmp_path / "table-tree.b2z" + with blosc2.TreeStore(path, mode="w") as tree: + tree["table"] = table + with blosc2.TreeStore(path, mode="r") as tree: + reopened = tree["table"] + assert isinstance(reopened._cols["remote"], blosc2.RemoteArray) + np.testing.assert_array_equal(reopened.remote[:], values) + + restored = blosc2.ctable_from_cframe(table.to_cframe(preserve_sources=True)) + assert isinstance(restored._cols["remote"], blosc2.RemoteArray) + np.testing.assert_array_equal(restored.remote[:], values) From d362c3e74696b3388b04040ab657d6efc26a06c9 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 18:22:11 +0200 Subject: [PATCH 27/82] Open remote CTable source columns --- src/blosc2/ctable_remote_read.py | 4 ++ src/blosc2/ctable_storage.py | 6 ++ src/blosc2/remote_array.py | 103 +++++++++++++++++++++++----- src/blosc2/remote_store.py | 29 +++++++- tests/ctable/test_remote_columns.py | 38 ++++++++++ 5 files changed, 162 insertions(+), 18 deletions(-) diff --git a/src/blosc2/ctable_remote_read.py b/src/blosc2/ctable_remote_read.py index 46fbf607a..9de43bab7 100644 --- a/src/blosc2/ctable_remote_read.py +++ b/src/blosc2/ctable_remote_read.py @@ -122,6 +122,8 @@ def open_columns(storage, table, names, load): # noqa: C901 members.setdefault(info.filename, []).append(info) def ranges_for(name): + if name in getattr(table, "_source_columns", ()): + return [] key = storage._full_key(f"_cols/{_column_name_to_relpath(name)}") spec = table._schema.columns_by_name[name].spec if isinstance(spec, UTF8Spec): @@ -346,6 +348,8 @@ def column_values(table, names, positions, *, null_masks=None): # noqa: C901 def reader(name): col = table._cols[name] spec = table._schema.columns_by_name[name].spec + if name in getattr(table, "_source_columns", ()): + return col[positions] if isinstance(col, UTF8Array): values = np.empty(len(positions), dtype=col.dtype) order = np.argsort(positions, kind="stable") diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index 93aab3d4e..b1bba2856 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -710,6 +710,12 @@ def _not_supported(*args, **kwargs): raise RuntimeError("RemoteTableStorage is read-only") def open_column(self, name: str) -> blosc2.RemoteArray: + source_columns = set(self.load_schema().get("source_columns", ())) + if name in source_columns: + logical_key = f"{_COLS_DIR}/{_column_name_to_relpath(name)}" + array = self._owner.open_ctable_carrier(self._root_key, logical_key) + self._arrays.append(array) + return array return self._open_array(f"{_COLS_DIR}/{_column_name_to_relpath(name)}") def open_list_column(self, name: str) -> ListArray: diff --git a/src/blosc2/remote_array.py b/src/blosc2/remote_array.py index 5a0b7d87f..ab1248f41 100644 --- a/src/blosc2/remote_array.py +++ b/src/blosc2/remote_array.py @@ -31,6 +31,7 @@ from blosc2.b2objects import ( _B2OBJECT_USER_VLMETA_KEY, make_b2object_carrier, + read_b2object_payload, read_b2object_user_vlmeta, write_b2object_payload, write_b2object_user_vlmeta, @@ -283,34 +284,49 @@ def _validate_authorized_source(urlpath, storage_options, source_descriptor, *, if storage_options is not None: raise ValueError("storage_options cannot be used with an authorized source") hdf5_cls = getattr(blosc2, "HDF5NDSource", ()) - if not isinstance(urlpath, (blosc2.FsspecNDSource, blosc2.ZarrNDSource, hdf5_cls, blosc2.B2ZNDSource)): + if not isinstance( + urlpath, + (blosc2.C2Array, blosc2.FsspecNDSource, blosc2.ZarrNDSource, hdf5_cls, blosc2.B2ZNDSource), + ): raise TypeError( - "source_descriptor requires an authorized FsspecNDSource, ZarrNDSource, HDF5NDSource, or B2ZNDSource" + "source_descriptor requires an authorized C2Array, FsspecNDSource, ZarrNDSource, " + "HDF5NDSource, or B2ZNDSource" ) assume_immutable = _validate_assume_immutable( source_descriptor.get("assume_immutable"), "source_descriptor assume_immutable" ) - expected = { - "kind": ( - "b2z" - if isinstance(urlpath, blosc2.B2ZNDSource) - else "hdf5" - if isinstance(urlpath, hdf5_cls) - else "zarr" - if isinstance(urlpath, blosc2.ZarrNDSource) - else "fsspec" - ), - "version": 1, - "urlpath": urlpath.urlpath, - "assume_immutable": assume_immutable, - } + if isinstance(urlpath, blosc2.C2Array): + expected = { + "kind": "caterva2", + "version": 1, + "path": urlpath.path, + "urlbase": urlpath.urlbase, + "assume_immutable": assume_immutable, + } + else: + expected = { + "kind": ( + "b2z" + if isinstance(urlpath, blosc2.B2ZNDSource) + else "hdf5" + if isinstance(urlpath, hdf5_cls) + else "zarr" + if isinstance(urlpath, blosc2.ZarrNDSource) + else "fsspec" + ), + "version": 1, + "urlpath": urlpath.urlpath, + "assume_immutable": assume_immutable, + } if isinstance(urlpath, (hdf5_cls, blosc2.B2ZNDSource)): expected["dataset"] = urlpath.dataset if isinstance(urlpath, blosc2.B2ZNDSource) and urlpath._archive.urlpath != urlpath.urlpath: raise ValueError("B2Z source URL does not match its archive") if source_descriptor != expected: raise ValueError("source_descriptor does not match the supplied source") - validate_persistable_url(urlpath.urlpath) + persisted_url = urlpath.urlbase if isinstance(urlpath, blosc2.C2Array) else urlpath.urlpath + if persisted_url is not None: + validate_persistable_url(persisted_url) return urlpath, dict(expected) @@ -1783,6 +1799,59 @@ def save( os.replace(staged, destination) return destination + @classmethod + def _from_carrier_with_owner(cls, carrier, owner, cache_key): + """Open a persisted carrier under a RemoteStore owner's cache policy.""" + payload = read_b2object_payload(carrier) + allowed = {"kind", "version", "source", "cache_policy", "max_cache_bytes", "mutable"} + if not set(payload).issubset(allowed) or payload.get("kind") != "remote_array": + raise ValueError("CTable source column is not a RemoteArray carrier") + if payload.get("version") != 1: + raise ValueError(f"Unsupported persisted Blosc2 object version: {payload.get('version')!r}") + source = payload.get("source") + source_kind, urlpath = _parse_source_from_payload(source) + hdf5_index = _hdf5_index_from_carrier(carrier) if source_kind == "hdf5" else None + seed = ( + _zarr_metadata_from_carrier(carrier) + if source_kind == "zarr" + else _b2z_seed_from_carrier(carrier) + if source_kind == "b2z" + else None + ) + src, descriptor = cls._open_source( + urlpath, + None, + traffic=owner.traffic, + source_format=source_kind if source_kind in {"zarr", "hdf5", "b2z"} else None, + assume_immutable=source["assume_immutable"], + dataset=source.get("dataset") if source_kind in {"hdf5", "b2z"} else None, + hdf5_index=hdf5_index, + seed=seed, + blocks=carrier.blocks if source_kind in {"zarr", "hdf5"} else None, + cparams=carrier.cparams if source_kind in {"zarr", "hdf5"} else None, + ) + expected = ( + tuple(carrier.shape), + np.dtype(carrier.dtype), + tuple(carrier.chunks), + tuple(carrier.blocks), + ) + actual = cls._geometry(src) + if actual != expected: + raise ValueError(f"RemoteArray source geometry no longer matches its carrier: {actual!r}") + owner.sources[cache_key] = src + kwargs = {} + if owner.cache_policy is not blosc2.CachePolicy.NONE: + kwargs["max_cache_bytes"] = owner.max_cache_bytes + return cls( + src, + cache_policy=owner.cache_policy, + _carrier=carrier, + _source_descriptor=descriptor, + _store_owner=owner, + **kwargs, + ) + @classmethod def _from_payload(cls, payload, carrier): allowed = {"kind", "version", "source", "cache_policy", "max_cache_bytes", "mutable"} diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 284a2189d..bb16d45cc 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -568,6 +568,19 @@ def open_ctable_array(self, table_path, logical_key): self.sources[full] = source return source + def open_ctable_carrier(self, table_path, logical_key): + """Open a RemoteArray carrier stored as a CTable column.""" + if self.format != "b2z": + raise NotImplementedError("Remote CTable access currently requires a B2Z source") + full = "/".join(part.strip("/") for part in (table_path, logical_key) if part.strip("/")) + self._validate(full) + matches = [info for info in self.archive.members if info.filename == full + ".b2nd"] + if len(matches) != 1: + raise NotImplementedError(f"Remote CTable source carrier {full!r} is unavailable") + offset, length = self.archive.member_window(matches[0]) + carrier = blosc2.ndarray_from_cframe(self.archive._read_archive(offset, length), copy=True) + return blosc2.RemoteArray._from_carrier_with_owner(carrier, self, full + ".source") + def open_ctable_batch(self, full): """Open one external BatchArray member hidden below a CTable node.""" if self.format != "b2z": @@ -682,6 +695,11 @@ def restore_caches(self, manifest): self.restoring = True for path in manifest["caches"]: self._validate(path) + if path.endswith(".source") and path not in self.nodes: + # External CTable columns are registered lazily when their + # persisted carrier is opened. get_cache() adopts this + # artifact leaf after that source has been authorized. + continue if path not in self.nodes or self.nodes[path][0] != "ndarray": raise ValueError("Invalid cached RemoteStore leaf") relative = path[len(self.root) + 1 :] if self.root else path @@ -1583,6 +1601,8 @@ def _validate_artifact_manifest(manifest): # noqa: C901 raise ValueError("Invalid RemoteStore caches") for path in caches: RemoteDiscovery._validate(path) + if path.endswith(".source") and path not in nodes: + continue if ( path not in nodes or nodes[path][0] != "ndarray" @@ -1712,7 +1732,14 @@ def _open_artifact(cls, urlpath, mode="r", **kwargs): "use a cold export or a cache policy that permits retained payload." ) else: - cache_policy = blosc2.CachePolicy.DISK + try: + cache_policy = ( + blosc2.CachePolicy.DISK + if manifest.get("mutable", False) + else blosc2.CachePolicy(manifest.get("cache_policy", "disk")) + ) + except ValueError as exc: + raise ValueError("RemoteStore artifact has an unsupported cache policy") from exc if cache_policy is blosc2.CachePolicy.NONE: if max_cache_bytes is not None: diff --git a/tests/ctable/test_remote_columns.py b/tests/ctable/test_remote_columns.py index efc37b907..0dd56816f 100644 --- a/tests/ctable/test_remote_columns.py +++ b/tests/ctable/test_remote_columns.py @@ -125,3 +125,41 @@ def test_treestore_and_cframe_preserve_sources(tmp_path): restored = blosc2.ctable_from_cframe(table.to_cframe(preserve_sources=True)) assert isinstance(restored._cols["remote"], blosc2.RemoteArray) np.testing.assert_array_equal(restored.remote[:], values) + + +@pytest.mark.parametrize("policy", list(blosc2.CachePolicy)) +def test_remote_ctable_owns_external_column_cache_policy(tmp_path, policy): + values = np.arange(20, dtype=np.float32) + source = remote_array(values, chunks=(5,), cache_policy=blosc2.CachePolicy.MEMORY) + table = blosc2.CTable( + Row, + sources={ + "local": blosc2.asarray(np.arange(20, dtype=np.int32)), + "remote": source, + }, + ) + archive = tmp_path / "referenced.b2z" + table.save(archive, preserve_sources=True) + archive_name = f"ctable-archive-{uuid4().hex}.b2z" + fsspec.filesystem("memory").pipe_file(archive_name, archive.read_bytes()) + + kwargs = {"cache_policy": policy} + if policy is not blosc2.CachePolicy.NONE: + kwargs["max_cache_bytes"] = 1 << 20 + if policy is blosc2.CachePolicy.DISK: + kwargs["cache_dir"] = tmp_path / "cache" + remote = blosc2.RemoteCTable(f"memory://{archive_name}", **kwargs) + external = remote._cols["remote"] + + assert external.cache_policy is policy + assert source.cache_policy is blosc2.CachePolicy.MEMORY + np.testing.assert_array_equal(remote.remote[3:13:2], values[3:13:2]) + np.testing.assert_array_equal(remote[remote.local >= 17].remote[:], values[17:]) + if policy is not blosc2.CachePolicy.NONE: + assert remote.cache_bytes <= remote.max_cache_bytes + if policy is blosc2.CachePolicy.MEMORY: + artifact = tmp_path / "remote-artifact.b2z" + remote.save(artifact) + restored = blosc2.open(artifact) + assert restored.cache_policy is policy + np.testing.assert_array_equal(restored.remote[:], values) From d3a828f5df09f4fec47628eb2c379820db6a466b Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 18:24:18 +0200 Subject: [PATCH 28/82] Document remote CTable columns --- doc/reference/ctable.rst | 21 +++++++++++++++++++++ doc/reference/remotectable.rst | 10 +++++++--- examples/ctable/remote_columns.py | 27 +++++++++++++++++++++++++++ src/blosc2/ctable.py | 4 ++++ src/blosc2/proxy.py | 8 ++++++++ src/blosc2/remote_array.py | 7 +++++++ src/blosc2/remote_ctable.py | 5 +++++ 7 files changed, 79 insertions(+), 3 deletions(-) create mode 100644 examples/ctable/remote_columns.py diff --git a/doc/reference/ctable.rst b/doc/reference/ctable.rst index 038671da7..2d7dde228 100644 --- a/doc/reference/ctable.rst +++ b/doc/reference/ctable.rst @@ -10,6 +10,27 @@ independently; rows are never materialised in their entirety unless you explicitly call :meth:`~blosc2.CTable.to_arrow` or iterate with :meth:`~blosc2.CTable.__iter__`. +Source-bound columns +-------------------- + +Pass ``sources=`` to bind every fixed-width schema column to an existing +:class:`~blosc2.NDArray` or :class:`~blosc2.RemoteArray` without copying it:: + + table = blosc2.CTable( + Reading, + sources={ + "station": local_station_ids, + "temperature": remote_temperatures, + }, + ) + +Names, dtypes, item shapes, and row counts must match the compiled schema +exactly. Source-bound tables are read-only. A local table keeps each +``RemoteArray`` cache policy; a :class:`~blosc2.RemoteCTable` applies its outer +policy and shared cache budget to every column. ``save()`` materializes an +independent table by default; use ``preserve_sources=True`` to retain remote +references for an unfiltered root table. + List columns ------------ diff --git a/doc/reference/remotectable.rst b/doc/reference/remotectable.rst index 4fe67b271..29c47f3d1 100644 --- a/doc/reference/remotectable.rst +++ b/doc/reference/remotectable.rst @@ -41,9 +41,13 @@ table and reads all data needed for it. ``copy()``, ``to_b2z()``, and ``to_b2d() remain local materialization operations inherited from :class:`blosc2.CTable`. Table cache bytes, limits, and traffic are scoped to the shared remote owner and -may include sibling leaves. A table selected from a store owns an independent -handle, but its columns and views remain borrowed from that table. Refresh a -nested table through its root store; a standalone table can call ``refresh()``. +may include sibling leaves. This also applies when a table column is itself a +``RemoteArray`` reference to another fsspec or Caterva2 URL: the outer table +policy overrides the carrier policy for that handle, and all such columns share +one budget. Standalone instances of those arrays keep their original policies. +A table selected from a store owns an independent handle, but its columns and +views remain borrowed from that table. Refresh a nested table through its root +store; a standalone table can call ``refresh()``. See :doc:`Working with Remote Tables <../guides/remote_tables>` for column access, filtering, buffering, reference saving, and materialization examples. diff --git a/examples/ctable/remote_columns.py b/examples/ctable/remote_columns.py new file mode 100644 index 000000000..f6e7526d2 --- /dev/null +++ b/examples/ctable/remote_columns.py @@ -0,0 +1,27 @@ +#!/usr/bin/env python3 +"""Bind local, fsspec, and Caterva2 arrays as read-only CTable columns.""" + +from dataclasses import dataclass + +import blosc2 + + +@dataclass +class Reading: + station_id: int = blosc2.field(blosc2.int32()) + temperature: float = blosc2.field(blosc2.float32()) + humidity: float = blosc2.field(blosc2.float32()) + + +table = blosc2.CTable( + Reading, + sources={ + "station_id": blosc2.open("s3://weather/station-id.b2nd", lazy=True), + "temperature": blosc2.RemoteArray("https://data.example.org/temperature.b2nd"), + "humidity": blosc2.RemoteArray( + blosc2.URLPath("weather/humidity.b2nd", urlbase="https://caterva.example.org") + ), + }, +) + +print(table[table.temperature > 20][["station_id", "humidity"]][:10]) diff --git a/src/blosc2/ctable.py b/src/blosc2/ctable.py index eba24466c..a4c23eb21 100644 --- a/src/blosc2/ctable.py +++ b/src/blosc2/ctable.py @@ -7336,6 +7336,10 @@ def save(self, urlpath: str, *, overwrite: bool = False, preserve_sources: bool overwrite: If ``False`` (default), raise :exc:`ValueError` when *urlpath* already exists. Set to ``True`` to replace an existing table. + preserve_sources: + Keep RemoteArray columns as references instead of materializing + them. This is valid only for an unfiltered source-bound root table. + The default writes an independent local copy. Raises ------ diff --git a/src/blosc2/proxy.py b/src/blosc2/proxy.py index 6aea403f9..626b4f2d0 100644 --- a/src/blosc2/proxy.py +++ b/src/blosc2/proxy.py @@ -153,6 +153,14 @@ class Proxy(blosc2.Operand): This can be used to cache chunks of a regular data container which follows the :ref:`ProxySource` or :ref:`ProxyNDSource` interfaces. + + .. note:: + + Use :ref:`RemoteArray` for supported remote URLs: it manages source + descriptors, cache policies and portable save/reopen behavior using Proxy + internally. Use Proxy directly for custom sources, including local or + generated data. Implementing the source interface enables reads, but does + not by itself define how to reconstruct that source from a saved cache. """ _stamped = False diff --git a/src/blosc2/remote_array.py b/src/blosc2/remote_array.py index ab1248f41..8d5db9f25 100644 --- a/src/blosc2/remote_array.py +++ b/src/blosc2/remote_array.py @@ -534,6 +534,13 @@ class RemoteArray(RemoteObject, blosc2.Operand): retained in process memory up to a bounded size. With :attr:`CachePolicy.NONE`, reads retain no data. + .. note:: + + RemoteArray manages supported remote sources and their portable + descriptors, using :ref:`Proxy` internally for reads and caching. For a + custom source implementing :ref:`ProxyNDSource` or :ref:`ProxySource`, + use Proxy directly; RemoteArray does not accept arbitrary source objects. + Parameters ---------- urlpath: str, URLPath, or C2Array diff --git a/src/blosc2/remote_ctable.py b/src/blosc2/remote_ctable.py index fc3b29ba8..3f970a88d 100644 --- a/src/blosc2/remote_ctable.py +++ b/src/blosc2/remote_ctable.py @@ -61,6 +61,11 @@ class RemoteCTable(RemoteObject, CTable): ``blosc2.open`` accepts ``max_concurrency`` but not the table-specific buffer keywords. Use this constructor or the returned table's settings to tune buffers. Cache policies and ``max_cache_bytes`` remain independent. + + A RemoteCTable's cache policy applies to every column read through it, + including columns backed by independent RemoteArray carriers. Those columns + share the table owner's cache budget and traffic accounting; their persisted + standalone policies are not used or modified by the table. """ def __new__( From 84ea5a73e1b40dd366e8ae9f171ae5b0cd960c4b Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 18:25:50 +0200 Subject: [PATCH 29/82] Finalize remote CTable column plan --- plans/ctable-remote-cols.md | 305 ++++++++++++++++++++++++++++ src/blosc2/ctable.py | 5 +- tests/ctable/test_remote_columns.py | 4 +- 3 files changed, 312 insertions(+), 2 deletions(-) create mode 100644 plans/ctable-remote-cols.md diff --git a/plans/ctable-remote-cols.md b/plans/ctable-remote-cols.md new file mode 100644 index 000000000..06a6e7784 --- /dev/null +++ b/plans/ctable-remote-cols.md @@ -0,0 +1,305 @@ +# RemoteArray columns in CTable and RemoteCTable + +Status: implemented for fixed-width NDArray and RemoteArray columns. + +## Objective + +Allow a CTable column to use an existing RemoteArray without copying its data. +Keep the row schema independent of source locations and bind arrays through a +new keyword-only `sources=` constructor argument. Support Caterva2 and fsspec +through the existing RemoteArray interface. Persist these references using the +same carriers already supported by TreeStore leaves. + +The first implementation is a read-only table with fixed-width columns. It must +support slices, row access, projections, filtering, expressions, save/reopen and +explicit materialization. It does not add remote writes or a new transport. + +## Public API + +```python +from dataclasses import dataclass +import blosc2 as b2 + + +@dataclass +class Measurement: + station_id: int = b2.field(b2.int32()) + temperature: float = b2.field(b2.float32()) + + +table = b2.CTable( + Measurement, + sources={ + "station_id": b2.RemoteArray("s3://weather/station_id.b2nd"), + "temperature": b2.RemoteArray( + b2.URLPath( + "weather/temperature.b2nd", + urlbase="https://caterva.example.org", + ), + cache_policy=b2.CachePolicy.MEMORY, + ), + }, +) + +table["temperature"][:10] +``` + +This is proposed syntax. `RemoteArray` and `URLPath` already exist; `sources=` +does not. Do not add `source=` to `field()` or a remote schema type. + +Define `sources` as a mapping from compiled column names to `RemoteArray` or +`NDArray`. Allow NDArray values so local and remote columns can coexist without +another ingestion API. Require every stored schema column exactly once; reject +unknown or missing names. For nested schemas, use existing flattened leaf names. +Do not accept bare URL strings, arbitrary Proxy objects or NumPy arrays initially. +Callers can construct RemoteArray or use `blosc2.asarray` explicitly. + +`sources=` is a creation path and is mutually exclusive with `new_data`. Reject +it when reopening an existing table. Treat an empty mapping as an explicit +request, not as omission. Preserve the existing constructor behavior when +`sources is None`. + +## Source and cache contract + +- Caterva2: accept RemoteArray constructed from an explicit URLPath. A plain + HTTPS string retains its existing fsspec meaning. +- fsspec: support the formats RemoteArray already supports: B2ND, B2Z array + datasets, HDF5 datasets and Zarr arrays. Protocol drivers and range-access + requirements remain those of the existing readers. +- Direct `RemoteArray(...)` defaults to NONE. `blosc2.open(..., lazy=True)` + defaults to MEMORY without an explicit disk cache location. A local CTable + preserves supplied arrays' policies; it has no table-wide cache policy. +- MEMORY retains compressed data in process memory, with its existing default + limit of 256 MiB per independent array. Its policy survives serialization; + its warm payload does not. DISK and NONE keep existing RemoteArray semantics. +- RemoteCTable controls caching for everything read through it, including + external columns. Its policy, cache location and shared `max_cache_bytes` + budget override column carrier settings for table-owned runtime handles. + Standalone arrays retain their own settings. Do not mutate supplied handles, + original carriers or independently opened arrays when applying this override. +- Outer NONE retains no fetched column payload, MEMORY uses one shared memory + budget, and DISK uses the table's local cache directory and one shared disk + budget. These are retained compressed-payload bounds, not total process RAM + or limits on temporary read buffers and results. Preserve existing metadata + accounting and read-only artifact rules. +- External columns must use the table's cache coordinator rather than allocate + independent per-array caches. Aggregate their requests into table traffic + accounting without double counting. Cache keys must distinguish source + identity/version, dataset and array geometry; keep existing credential-scope + isolation and source validation. Equal chunk numbers in different sources + must never collide. +- A DISK column needs table-owned writable local cache storage; a carrier inside + a remote archive cannot be updated in place. Ignore serialized runtime cache + locations and reuse existing owner-managed cache allocation. Do not create + disk caches under outer NONE or MEMORY because a carrier requests DISK. +- Credentials remain runtime configuration. Reuse existing carrier URL checks + and credential-free serialization; do not add secrets to table metadata. + Persisted private sources use the existing RemoteArray authentication mechanisms. + Do not forward outer archive credentials automatically to another host. + +## Construction and validation + +Add the sources branch before ordinary capacity allocation and column creation +in `CTable.__init__`. Compile the normal schema first, then validate the entire +mapping before creating or overwriting destination storage. + +1. Accept fixed-width scalar columns and fixed-shape NDArraySpec columns. + Reject UTF-8, variable-length, list, dictionary and object column bindings in + this first version; their storage spans multiple coordinated components. +2. Require exact storage dtype compatibility. Scalars require shape `(N,)`; + NDArraySpec columns require `(N, *item_shape)`. Do not silently cast, fetch, + rechunk or recompress a reference to satisfy the schema. +3. Infer N from the sources and require equal first-axis lengths. Accept N=0. + Metadata checks must not scan payloads. Schema value constraints such as + `ge` and `le` cannot be verified without reads: reject such constrained source + bindings with `validate=True`, explaining that `validate=False` skips value + validation but never dtype/shape checks. +4. Initially reject nullable bindings requiring a separate validity mask. Permit + supported explicit sentinel null representations using existing decoding. + Add remote validity-sidecar binding in a subsequent extension. +5. Create the normal valid-row array with N live entries, set row-count/last-pos + metadata consistently, and install the source objects as columns. Any minimum + backing capacity required for an empty validity array must stay marked false. + Do not apply the usual default million-row allocation to source columns. +6. Source geometry is authoritative. Reject conflicting per-column chunks, + blocks or compression settings rather than silently ignoring them. Table + storage settings may still configure table-owned metadata arrays. + +Source row i represents table physical row i. Matching lengths do not prove that +two datasets describe corresponding observations; alignment is the caller's +responsibility. No joins or implicit key matching occur. + +Treat supplied handles as borrowed: closing the table must not close a caller's +RemoteArray. Handles opened from persisted carriers are table-owned. Document +that independently mutating a supplied local NDArray is outside this read-only +table contract. Reuse existing source identity checks; independently changing +remote datasets do not form a transactional snapshot. + +## Storage and persistence + +Reuse TableStorage and its existing column paths. Add only the small column +installation operation needed by the supported storage backends: in-memory +storage retains a handle, persistent storage writes a RemoteArray carrier or +copies a local NDArray through existing store facilities. Never persist +`remote.cache` as though it were a complete data column. + +Record which columns are external references in table storage metadata. Reuse +the carrier as the authoritative source descriptor, including its standalone +cache policy. RemoteCTable's runtime override does not replace that descriptor; +do not duplicate its URL/authentication/format schema in table metadata. Add a +versioned table capability marker, with readers rejecting unknown versions, +so older code cannot interpret an incomplete carrier as ordinary column data. +Inspect both schema and table-kind version checks before choosing the marker. + +Audit all backends: InMemoryTableStorage, FileTableStorage, +EmbedStoreTableStorage and TreeStoreTableStorage. Use their normal leaf open +paths and existing object reconstruction. A persistent constructor with +`sources=` must write references and all required metadata before exposing the +table as read-only. Reopen restores that restriction even with mode='a'. + +Keep existing materializing export behavior explicit: + +- Add `preserve_sources=False` to CTable.save/to_b2z; their existing defaults + continue writing independent local data. `preserve_sources=True` writes + carriers and the table capability marker without scanning source columns. +- TreeStore assignment of an unfiltered reference-backed root table preserves + its references, as ordinary RemoteArray leaf assignment already does. +- Preserve references only for the identity row mapping, including column-only + projections. Reject reference-preserving exports of filtered/reordered views + or requests to rechunk referenced columns. Their ordinary materializing + exports remain available. Do not serialize a view as an unfiltered source. +- `copy()` continues to produce independent materialized columns, removing the + external-reference marker. Document its read and storage cost. +- RemoteCTable.save already exports a remote-reference artifact. Extend that + path to preserve nested column carriers; do not replace its existing contract + with the local CTable.save default. Persist the effective outer policy and + shared budget in the owner manifest, retaining nested standalone descriptors. + Reopening the artifact applies the outer policy to all columns again. Export + retained external payload through the owner cache using existing include_cache + semantics; MEMORY exports reopen cold. Never serialize runtime cache paths. + +In particular, `_save_to_storage()` currently compacts live rows and allocates +new arrays. Reference preservation needs an explicit branch before that copying +path. Cover `_save_to_treestore`, `to_cframe` and embedded storage too; no export +path may accidentally serialize only missing-cache placeholders as local data. + +## RemoteCTable opening and reads + +RemoteTableStorage currently opens each column as an array member of one B2Z +archive. Extend discovery/opening to distinguish an ordinary array member from +a RemoteArray carrier. Resolve the latter through existing carrier decoding, +so its target may be Caterva2 or a different fsspec object. Do not treat the +carrier's placeholder chunks as the column's actual values. + +Construct external runtime handles under the outer owner before any cache is +allocated. Extend the existing cache coordinator/owner integration in +remote_store.py and remote_array.py to cover their source types, including +Caterva2. Resolve identity and seed-cache validity through existing source +checks. Warm carrier payload may be reused only under the outer policy and +shared budget; it must not enable hidden retention under NONE. Avoid constructing +an independently cached RemoteArray and changing its policy after the fact. + +`ctable_remote_read.py` currently assumes B2Z member offsets, an archive transport +and a shared store owner. Keep that optimized path for ordinary archive columns. +For independent external sources, use a table-owned array's public indexing path +with the shared cache coordinator and existing concurrency behavior. Ensure this +fallback participates in eviction, traffic accounting and owner lifecycle checks. +Start with a correct fallback; do not generalize +the B2Z scheduler into a new transport framework. Preserve column order and +physical row selections when combining local/archive/external reads. + +Audit Column access, `_fetch_col_at_positions_uncached`, expression construction, +reductions, null handling and NDArray-only type checks. Broaden only checks where +RemoteArray satisfies the actual read contract. Differently partitioned sources +must use existing general evaluation paths rather than assuming aligned chunks. +Do not access raw incomplete caches to satisfy an NDArray-only fast path. + +External arrays opened by RemoteCTable must participate in close/stale-handle +checks. Refreshing the outer table discards owned external handles and reopens +them lazily. Refresh must not claim atomic version alignment across sources. +An unavailable external column should fail when accessed without preventing an +unrelated column from being read where metadata permits lazy opening. + +## Mutation and indexing + +Make the whole reference-backed table logically read-only in this first version, +including mixed local/remote tables. Reuse the existing read-only guards, and +audit append/extend, assignment, deletion, resize, compact, add/drop columns and +materialized-column maintenance for bypasses. Raise before any partial mutation. +Changing cache contents remains allowed according to the effective cache policy: +per-array for local CTable, outer-owner policy for RemoteCTable. + +Disable automatic SUMMARY-index creation for source-backed construction and +reopen: closing a handle must not download columns. Initially reject explicit +persistent index creation on these tables and ignore/reject stale index metadata +for external columns. Read-only scans and transient expression evaluation remain +supported. Source-version-aware persisted indexes are a later feature. + +## Implementation sequence and verification + +1. [x] Implement constructor binding, validation, read-only behavior and fixed-width + read/query paths in ctable.py and ctable_storage.py. Preserve ordinary-table + behavior. Start tests with deterministic fsspec memory sources. +2. [x] Add versioned metadata and carrier-preserving local/store exports and reopen. + Reuse dict_store.py/RemoteArray serialization; avoid a parallel Ref resolver. +3. [x] Extend RemoteStore discovery, RemoteTableStorage and the remote read fallback + to open and query external column carriers inside remote B2Z tables. Integrate + all external source types with the outer owner's cache budget and traffic + accounting; this is required functionality, not a later optimization. +4. [x] Document the constructor, source immutability/alignment, local per-array versus + RemoteCTable-wide policies and budgets, + credential handling and reference versus materialized exports. Add a small + example under examples/ctable with both Caterva2 and fsspec bindings. + +The implementation landed in four focused commits: + +- ``7a53f6e1`` adds source-bound construction, validation, reads, queries and + read-only behavior. +- ``b2b57c9a`` adds versioned metadata, materializing and reference-preserving + exports, and TreeStore/CFrame round trips. +- ``d362c3e7`` opens external carriers through RemoteCTable and applies the + outer cache owner, budget, traffic accounting and artifact policy. +- ``d3a828f5`` documents the API, cache ownership and mixed remote sources. + +Verification used focused pytest coverage under the ``blosc2`` conda environment: + +- Same-length and empty sources; mixed NDArray/RemoteArray columns; scalar and + fixed-shape cells; mismatched names, dtype, shape and conflicting options. +- Slices, stepped/fancy selections, rows, projection, filtering and expressions + across columns with different chunk/block grids, compared with NumPy values. +- Mutation rejection before writes, and borrowed versus owned handle lifetimes. +- Local CTable NONE and MEMORY policy preservation; a warm MEMORY source reopens cold; + DISK retains only the cache allowed by existing export semantics. +- Parameterize outer NONE/MEMORY/DISK against each persisted column policy. + Verify the outer policy wins without changing standalone handles/descriptors. + Under outer NONE/MEMORY, an inner DISK policy must not create disk caches. +- Read multiple ordinary and external columns past one shared budget and verify + cross-column eviction. Test distinct sources with matching geometry/chunk + numbers for cache isolation, plus changed-source invalidation. Temporary + working buffers remain outside the retained-cache budget. +- Verify aggregate table traffic includes external requests exactly once, warm + reads reuse the owner cache, DISK stays under the owner cache directory, and + saved remote-reference artifacts restore the outer policy/shared budget. +- Construction and reference exports make no payload reads after source metadata + discovery. Repeated MEMORY reads reuse fetched data; NONE does not retain it. + Account for archive metadata prefetch separately from full-column reads. +- Local save/open, TreeStore B2Z and embedded round trips, explicit materializing + export/copy, and rejection of nonidentity reference-preserving views. +- A remote B2Z table whose columns reference independent fsspec and mocked + Caterva2 sources; close/refresh invalidation and one unavailable source. +- Serialized artifacts contain no credentials or runtime cache paths; legacy + tables continue to open without a new marker. + +The new focused suite passes with all three outer cache policies. The existing +RemoteCTable, RemoteStore and RemoteArray suites pass (350 passed, 5 skipped), +as do the persistence/CFrame suites (82 passed). Ruff passes for every changed +Python file. The Sphinx build reaches completion but ``-W`` remains non-zero +because the repository already emits unrelated autosummary, duplicate-target, +toctree and theme warnings. + +## Deferred work + +Remote writes, mutable hybrid tables, row remapping of references, nullable mask +bindings, composite/variable-length remote columns, persisted remote indexes, +and a unified parallel transport scheduler are +outside the first implementation. No new dependency is required. diff --git a/src/blosc2/ctable.py b/src/blosc2/ctable.py index a4c23eb21..049650bc0 100644 --- a/src/blosc2/ctable.py +++ b/src/blosc2/ctable.py @@ -7288,7 +7288,10 @@ def _save_to_storage( # noqa: C901 if n_live > 0: disk_mask[:n_live] = mask[:n_live] if no_deletions else mask[live_pos] - storage.save_schema(self._schema_dict_with_computed()) + schema_dict = self._schema_dict_with_computed() + schema_dict.pop("source_bindings_version", None) + schema_dict.pop("source_columns", None) + storage.save_schema(schema_dict) attrs = self.attrs[:] if attrs: vlmeta = blosc2.SChunk() diff --git a/tests/ctable/test_remote_columns.py b/tests/ctable/test_remote_columns.py index 0dd56816f..a066618ca 100644 --- a/tests/ctable/test_remote_columns.py +++ b/tests/ctable/test_remote_columns.py @@ -89,10 +89,12 @@ def test_save_materializes_by_default_and_can_preserve_sources(tmp_path): materialized_path = tmp_path / "materialized.b2z" table.save(materialized_path) - materialized = blosc2.open(materialized_path) + materialized = blosc2.open(materialized_path, mode="a") assert isinstance(materialized._cols["remote"], blosc2.NDArray) assert not isinstance(materialized._cols["remote"], blosc2.RemoteArray) + assert not materialized._read_only np.testing.assert_array_equal(materialized.remote[:], values) + materialized.append((6, 6.0)) referenced_path = tmp_path / "referenced.b2z" table.save(referenced_path, preserve_sources=True) From c4017a89a66afdf7b706c2817912e5187236c22b Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 18:49:57 +0200 Subject: [PATCH 30/82] Add sparse caching for remote tables --- doc/guides/remote_tables.md | 11 ++++++ doc/reference/remotectable.rst | 6 +++ plans/remote-ctable-save.md | 58 ++++++++++++++++------------ plans/remote-ctable.md | 31 ++++++++++++--- src/blosc2/remote_array.py | 1 + src/blosc2/remote_ctable.py | 60 ++++++++++++++++++++++++++++- src/blosc2/remote_store.py | 20 ++++++---- src/blosc2/remote_store_cache.py | 1 + tests/ctable/test_remote_columns.py | 25 ++++++++++++ tests/ctable/test_remote_ctable.py | 24 ++++++++++++ 10 files changed, 199 insertions(+), 38 deletions(-) diff --git a/doc/guides/remote_tables.md b/doc/guides/remote_tables.md index e64928a6f..74ccdd56e 100644 --- a/doc/guides/remote_tables.md +++ b/doc/guides/remote_tables.md @@ -121,6 +121,17 @@ refresh. Refresh a table obtained from a {ref}`RemoteStore` through the root store, then retrieve the table again. Immutable reference artifacts reject `refresh()`. +For a cache shared by multiple server processes, use the sparse constructor: + +```python +with blosc2.RemoteCTable.with_sparse_cache(url, "shared-table-cache") as table: + print(table[:5]) +``` + +The outer table owns one aggregate cache budget for its ordinary and referenced +`RemoteArray` columns. Every process using the cache directory must use this +constructor. + ## See also - {doc}`remote_objects` — shared caching, traffic, reference, and lifetime behavior. diff --git a/doc/reference/remotectable.rst b/doc/reference/remotectable.rst index 29c47f3d1..b27ae6c60 100644 --- a/doc/reference/remotectable.rst +++ b/doc/reference/remotectable.rst @@ -49,6 +49,12 @@ A table selected from a store owns an independent handle, but its columns and views remain borrowed from that table. Refresh a nested table through its root store; a standalone table can call ``refresh()``. +Server processes can share one bounded sparse disk cache with +``RemoteCTable.with_sparse_cache(url, runtime_cache_path)``. The constructor +uses the same process-safe cache and aggregate byte limit as +``RemoteStore.with_sparse_cache()``; all processes using that directory must +open it through the sparse-cache constructor. + See :doc:`Working with Remote Tables <../guides/remote_tables>` for column access, filtering, buffering, reference saving, and materialization examples. diff --git a/plans/remote-ctable-save.md b/plans/remote-ctable-save.md index 6166505fe..2fb8c94b6 100644 --- a/plans/remote-ctable-save.md +++ b/plans/remote-ctable-save.md @@ -1,7 +1,9 @@ # RemoteCTable reference saving -Status: deferred proposal for later consideration. This document does not change -the current save behavior. +Status: reference saving and RemoteStore artifact inclusion are implemented. +Status reviewed on 2026-09-20. The implementation steps and verification checklist +below record the design requirements, rather than results of a new verification +run during this documentation update. ## Goal @@ -15,35 +17,36 @@ than introduce a second table-specific reference format. ## Current behavior and dependencies -- RemoteCTable inherits CTable.save(), which materializes live rows into a local - table and returns None. +- RemoteCTable overrides CTable.save(), writes a remote-reference artifact and + returns its absolute output path. CTable.save() materializes by default. - RemoteStore.save() writes a portable .b2z reference archive containing its source descriptor, discovery metadata and optionally already-retained data. It returns the output path and does not fetch all missing data. -- RemoteStore._collect_export_nodes() explicitly rejects CTable nodes. Its - existing export logic also does not preserve CTable schema metadata as needed - for reconstructing those nodes. Removing the rejection alone is insufficient. -- CTable.to_b2z() and to_b2d() dispatch to self.save() on remote root tables. - These must retain materializing semantics when RemoteCTable overrides save(). -- No direct RemoteCTable.save() callers were found in the repository during the - current audit. test_remote_utf8_nested_lifetime indirectly relies on save() - through table.to_b2z(), and verifies an independent local table. External - callers are unknown; document the change rather than assuming none exist. +- RemoteStore._collect_export_nodes() includes CTable nodes and their metadata. + Shared artifact validation and dispatch recognize table-root references and + tables inside exported groups. +- CTable.to_b2z() and to_b2d() explicitly call the materializing CTable.save() + implementation on remote root tables. copy() also produces a local table. +- Tests in tests/ctable/test_remote_ctable.py cover root and nested reference + saving, warm/cold and mutable artifacts, reopen settings, malformed metadata + and independent materialization. - RemoteCTable now exposes refresh() and is_cache_mutable. Artifact readers must preserve their cache-writability and stale-handle contracts. -## Proposed public contract +## Implemented public contract -Proposed signature, to confirm when implementing: +Current signature (type annotations abbreviated): ```python -def save(destination, *, include_cache=True, mutable=None, overwrite=False) -> str: ... +def save( + destination=None, *, urlpath=None, include_cache=True, mutable=None, overwrite=False +) -> str: ... ``` -Match RemoteStore.save() for accepted options, defaults, .b2z destinations, -return value, overwrite checks and destination safety. This changes both the -meaning and return value of the inherited method; note the destination keyword -change from CTable.save(urlpath=...) as well. +The method matches RemoteStore.save() for defaults, .b2z destinations, return +value, overwrite checks and destination safety. It accepts `urlpath=` as an alias +for `destination`; callers must supply exactly one. Its meaning and return value +differ from the materializing CTable.save(). - include_cache=True exports only already-retained data, not missing chunks. - include_cache=False exports enough metadata to reconstruct the table without @@ -144,8 +147,9 @@ local = remote.copy() # independent in-memory CTable Explain network access, immutability, storage credentials on another machine, cache writability and overwrite behavior. State explicitly that local -CTable.save() remains unchanged. Update the guide's current statement that -portable table-reference export is unsupported only once implemented and tested. +CTable.save() retains its materializing default. The reference-saving examples +are now documented in `doc/reference/remotectable.rst` and +`doc/guides/remote_tables.md`. ## Verification @@ -178,5 +182,11 @@ changes the semantics of explicit materialization methods. No remote writes, automatic change detection, new remote formats, automatic full-cache population, reference export of arbitrary table views, or changes -to local CTable.save(). The proposal is intentionally deferred; revisit the -public signature and compatibility notes before starting implementation. +to the materializing default of local CTable.save(). A subsequent extension +adds opt-in `preserve_sources=True` for source-bound local tables; see +`ctable-remote-cols.md` for that separate contract. + +General remote SUMMARY/scalar index resolution remains a separate follow-up in +`remote-ctable.md`. Shared sparse runtime caching is available directly through +`RemoteCTable.with_sparse_cache()` and includes referenced RemoteArray columns +under the outer table's cache owner and aggregate budget. diff --git a/plans/remote-ctable.md b/plans/remote-ctable.md index ccf4a30f2..e97c75efd 100644 --- a/plans/remote-ctable.md +++ b/plans/remote-ctable.md @@ -3,8 +3,13 @@ Status: initial fixed-width, read-only implementation completed on 2026-09-17; UTF-8 support was added in the v2 extension (see `remote-ctable-v2.md`); batch-backed columns were added in the batch extension (see -`remote-ctable-batches.md`); persisted indexes and portable references remain -follow-ups. +`remote-ctable-batches.md`). Portable references and RemoteStore artifact +inclusion are implemented (see `remote-ctable-save.md`). Remote list membership +indexes are supported; SUMMARY and other scalar persisted indexes remain +follow-ups. Status reviewed on 2026-09-20. + +The implementation notes and first-release decisions below describe the initial +scope. The follow-up status at the end reflects subsequent extensions. ## Objective and architecture @@ -274,10 +279,26 @@ and correct results, not a claim that scan queries avoid reading their operands. 2. Remote batch reads for lists, variable-length values and dictionary stores: implemented in the extension described in `remote-ctable-batches.md`. 3. Persisted indexes through a remote-aware sidecar resolver, starting with - SUMMARY indexes and measuring query transfer savings. + SUMMARY indexes and measuring query transfer savings: still outstanding for + SUMMARY and other scalar index kinds. List membership indexes already work: + `RemoteTableStorage.load_index_catalog()` admits membership descriptors and + `open_membership_postings()` fetches selected posting batches. The test + `test_remote_ctable_membership_index_avoids_list_payload` covers this path. 4. Portable RemoteCTable references, RemoteStore artifact inclusion and sparse - runtime-cache APIs if needed by actual consumers. + runtime-cache APIs if needed by actual consumers: reference saving and artifact + inclusion are implemented. `RemoteCTable.save()` delegates to the shared + exporter; `blosc2.open()` reconstructs table-root artifacts and nested tables. + Root/nested round-trip tests live in `tests/ctable/test_remote_ctable.py`. + `RemoteStore.with_sparse_cache()` and `RemoteCTable.with_sparse_cache()` supply + shared sparse runtime-cache APIs. Table tests cover concurrent handles, + cross-handle cache reuse, refresh invalidation and referenced RemoteArray + column reuse. + +Additional extensions include standalone `refresh()`, bounded parallel column +reads (`remote-ctable-parallel.md`), and `sources=` bindings for NDArray and +RemoteArray columns (`ctable-remote-cols.md`). External column reads use the +outer RemoteCTable cache owner and budget. There are no blocking API questions left for the initial fixed-width scope. The -remaining items above are deliberately separate extensions rather than blockers +remaining work is general remote persisted-index support. It is a separate extension rather than a blocker for the implemented read-only API. diff --git a/src/blosc2/remote_array.py b/src/blosc2/remote_array.py index 8d5db9f25..b59074470 100644 --- a/src/blosc2/remote_array.py +++ b/src/blosc2/remote_array.py @@ -1847,6 +1847,7 @@ def _from_carrier_with_owner(cls, carrier, owner, cache_key): if actual != expected: raise ValueError(f"RemoteArray source geometry no longer matches its carrier: {actual!r}") owner.sources[cache_key] = src + owner.source_descriptors[cache_key] = descriptor kwargs = {} if owner.cache_policy is not blosc2.CachePolicy.NONE: kwargs["max_cache_bytes"] = owner.max_cache_bytes diff --git a/src/blosc2/remote_ctable.py b/src/blosc2/remote_ctable.py index 3f970a88d..5f0aa7945 100644 --- a/src/blosc2/remote_ctable.py +++ b/src/blosc2/remote_ctable.py @@ -120,6 +120,59 @@ def __init__(self, *args, **kwargs): # Construction is completed by CTable._open_from_storage() in __new__. pass + @classmethod + def with_sparse_cache( + cls, + urlpath, + runtime_cache_path, + *, + dataset=None, + manifest=None, + max_cache_bytes=None, + carrier=None, + max_concurrency=8, + metadata_buffer_bytes=8 << 20, + row_buffer_bytes=64 << 20, + _filesystem=None, + _source_validator=None, + _manifest_validator=None, + _max_nodes=None, + ): + """Attach a remote CTable to a sparse disk cache shared across processes.""" + settings = { + name: _positive_integer(name, value) + for name, value in { + "max_concurrency": max_concurrency, + "metadata_buffer_bytes": metadata_buffer_bytes, + "row_buffer_bytes": row_buffer_bytes, + }.items() + } + + from blosc2.remote_store import RemoteStore + + store = RemoteStore.with_sparse_cache( + urlpath, + runtime_cache_path, + dataset=dataset, + manifest=manifest, + max_cache_bytes=max_cache_bytes, + carrier=carrier, + _filesystem=_filesystem, + _source_validator=_source_validator, + _manifest_validator=_manifest_validator, + _max_nodes=_max_nodes, + ) + try: + _, full = store._resolve("") + kind, diagnostic = store._owner.nodes[full] + if kind != "ctable": + if kind == "unsupported": + raise NotImplementedError(str(diagnostic)) + raise ValueError("RemoteCTable requires a CTable node") + return cls._from_owner(store._owner, full, **settings) + finally: + store.close() + @classmethod def _from_owner(cls, owner, full_path, **settings): settings = {name: _positive_integer(name, value) for name, value in settings.items()} @@ -181,9 +234,14 @@ def refresh(self) -> None: replacement.release() raise replacement.release() + if getattr(replacement, "shared", False): + from blosc2.remote_store_cache import SharedStoreOperation + + replacement.lock = SharedStoreOperation(replacement) replacement._cleanup_dir, owner._cleanup_dir = owner._cleanup_dir, None replacement.artifact_path = owner.artifact_path - owner.disk = None + if not getattr(owner, "shared", False): + owner.disk = None owner.generation = replacement.generation state = fresh.__dict__.copy() fresh._storage = None # Ownership is transferred to this handle. diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index bb16d45cc..13de6f895 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -97,6 +97,7 @@ def __init__( self.archive = None self.zstore = None self.sources = {} + self.source_descriptors = {} self.caches = {} self.batch_caches = {} self.disk = None @@ -712,14 +713,16 @@ def get_cache(self, source, *, seed=None): key = next(path for path, value in self.sources.items() if value is source) if key not in self.caches: if getattr(self, "shared", False): - descriptor = { - "kind": self.format, - "version": 1, - "urlpath": source.urlpath, - "assume_immutable": True, - } - if self.format in {"b2z", "hdf5"}: - descriptor["dataset"] = key + descriptor = self.source_descriptors.get(key) + if descriptor is None: + descriptor = { + "kind": self.format, + "version": 1, + "urlpath": source.urlpath, + "assume_immutable": True, + } + if self.format in {"b2z", "hdf5"}: + descriptor["dataset"] = key runtime = blosc2.RemoteArray.with_sparse_cache( source, self.disk.payload_path(self.generation, key), @@ -843,6 +846,7 @@ def _close_resources(self): if isinstance(source, blosc2.HDF5NDSource): source.close() self.sources.clear() + getattr(self, "source_descriptors", {}).clear() self.caches.clear() getattr(self, "batch_caches", {}).clear() self.nodes.clear() diff --git a/src/blosc2/remote_store_cache.py b/src/blosc2/remote_store_cache.py index fdb7d8658..020502a1b 100644 --- a/src/blosc2/remote_store_cache.py +++ b/src/blosc2/remote_store_cache.py @@ -244,6 +244,7 @@ def __enter__(self): owner.zstore.close() owner.zstore = None owner.sources.clear() + owner.source_descriptors.clear() owner.generation = manifest["generation"] owner.metadata = manifest["metadata"] owner.nodes.clear() diff --git a/tests/ctable/test_remote_columns.py b/tests/ctable/test_remote_columns.py index a066618ca..579d806f4 100644 --- a/tests/ctable/test_remote_columns.py +++ b/tests/ctable/test_remote_columns.py @@ -165,3 +165,28 @@ def test_remote_ctable_owns_external_column_cache_policy(tmp_path, policy): restored = blosc2.open(artifact) assert restored.cache_policy is policy np.testing.assert_array_equal(restored.remote[:], values) + + +def test_remote_ctable_sparse_cache_reuses_external_column(tmp_path): + values = np.arange(20, dtype=np.float32) + table = blosc2.CTable( + Row, + sources={ + "local": blosc2.asarray(np.arange(20, dtype=np.int32)), + "remote": remote_array(values, chunks=(5,)), + }, + ) + archive = tmp_path / "referenced.b2z" + table.save(archive, preserve_sources=True) + archive_name = f"ctable-sparse-{uuid4().hex}.b2z" + url = f"memory://{archive_name}" + fsspec.filesystem("memory").pipe_file(archive_name, archive.read_bytes()) + cache = tmp_path / "sparse-cache" + + with blosc2.RemoteCTable.with_sparse_cache(url, cache) as first: + np.testing.assert_array_equal(first.remote[:], values) + with blosc2.RemoteCTable.with_sparse_cache(url, cache) as second: + remote = second._cols["remote"] + second.traffic.reset() + np.testing.assert_array_equal(remote[:], values) + assert second.traffic.requests == 0 diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 9008fe3d2..0d1e9cece 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -32,6 +32,25 @@ def remote_table_url(tmp_path, table, name="table"): return url +def test_sparse_cache_shared_handles_and_refresh(tmp_path): + local = blosc2.CTable(Row, [(i, (i, i + 1), f"r{i}") for i in range(20)]) + url = remote_table_url(tmp_path, local, "sparse-shared") + cache = tmp_path / "sparse-cache" + + with blosc2.RemoteCTable.with_sparse_cache(url, cache) as first: + with blosc2.RemoteCTable.with_sparse_cache(url, cache) as second: + np.testing.assert_array_equal(first.x[:], np.arange(20)) + second.traffic.reset() + np.testing.assert_array_equal(second.x[:], np.arange(20)) + assert second.traffic.requests == 0 + assert first.cache_bytes == second.cache_bytes > 0 + assert second.cache_policy is blosc2.CachePolicy.DISK + + second.refresh() + with pytest.raises(RuntimeError, match="stale"): + first.x[:] + + @pytest.mark.parametrize("include_note", [False, True]) @pytest.mark.parametrize("nrows", [0, 2, 20, 21]) def test_remote_example_total_timing(tmp_path, capsys, include_note, nrows): @@ -717,6 +736,11 @@ def test_remote_store_returns_table_with_independent_lifetime(tmp_path): with blosc2.RemoteCTable(url, dataset="group/table") as direct: np.testing.assert_array_equal(direct["x"][:], [1, 2]) + with blosc2.RemoteCTable.with_sparse_cache( + url, tmp_path / "nested-sparse-cache", dataset="group/table" + ) as sparse: + np.testing.assert_array_equal(sparse["x"][:], [1, 2]) + def test_remote_ctable_reference_save_roundtrip(tmp_path): @dataclasses.dataclass From 058e126d46fa3b2d1bf5230a3007d407b8f2fea9 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 19:11:59 +0200 Subject: [PATCH 31/82] Resolve remote CTable index sidecars --- src/blosc2/ctable_storage.py | 31 ++++++++++++++++++++++- src/blosc2/indexing.py | 10 +++++++- tests/ctable/test_remote_ctable.py | 40 ++++++++++++++++++++++++++++++ 3 files changed, 79 insertions(+), 2 deletions(-) diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index b1bba2856..9ed1ace19 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -664,6 +664,7 @@ def __init__( self._root_key = root_key.strip("/") self._generation = owner.generation self._arrays: list[blosc2.RemoteArray] = [] + self._registered_index_paths: list[str] = [] self._closed = False owner.acquire() @@ -828,6 +829,11 @@ def close(self) -> None: if self._closed: return self._closed = True + from blosc2.indexing import _SIDECAR_REMOTE_REGISTRY + + for path in self._registered_index_paths: + _SIDECAR_REMOTE_REGISTRY.pop(path, None) + self._registered_index_paths.clear() for array in self._arrays: array.close() self._arrays.clear() @@ -857,7 +863,30 @@ def load_index_catalog(self) -> dict: return {} catalog = {} for name, descriptor in raw.items(): - if not isinstance(descriptor, dict) or descriptor.get("kind") != "membership": + if not isinstance(descriptor, dict): + continue + kind = descriptor.get("kind") + if kind in {"summary", "bucket", "partial", "full", "opsi"}: + if descriptor.get("version") != 1 or not isinstance(descriptor.get("token"), str): + raise ValueError(f"Malformed remote index for column {name!r}") + resolved = copy.deepcopy(descriptor) + from blosc2.indexing import _SIDECAR_REMOTE_REGISTRY + + for obj, key in FileTableStorage._walk_descriptor_paths(resolved): + path = obj[key] + parts = pathlib.PurePosixPath(path).parts + if os.path.isabs(path) or ".." in parts or not path.endswith(".b2nd"): + raise ValueError(f"Unsafe remote index path for column {name!r}") + logical = path[:-5].strip("/") + if not self._has_array(logical): + raise ValueError(f"Missing remote index sidecar for column {name!r}") + remote_path = f"remote-index://{id(self)}/{path}" + _SIDECAR_REMOTE_REGISTRY[remote_path] = (self, logical) + self._registered_index_paths.append(remote_path) + obj[key] = remote_path + catalog[name] = resolved + continue + if kind != "membership": continue payload = descriptor.get("membership") if not isinstance(payload, dict): diff --git a/src/blosc2/indexing.py b/src/blosc2/indexing.py index 21e044af7..915826503 100644 --- a/src/blosc2/indexing.py +++ b/src/blosc2/indexing.py @@ -128,6 +128,10 @@ def nbytes(self) -> int: # Populated by the storage layer so indexing code can open sidecars without # extracting them to a temporary directory first. _SIDECAR_ZIP_REGISTRY: dict[str, tuple[str, int]] = {} +# Remote CTable sidecars use the same readers as local sidecars. The storage +# object owns the returned RemoteArray and unregisters the synthetic path when +# the table closes. +_SIDECAR_REMOTE_REGISTRY: dict[str, tuple[object, str]] = {} _HOT_CACHE_GLOBAL_SCOPE = ("global", 0) FULL_OOC_RUN_ITEMS = 10_000_000 @@ -302,8 +306,12 @@ def _owned_path(path) -> bool: handles.pop(path, None) -def _open_sidecar_file(path: str, mmap_mode=None) -> blosc2.NDArray: +def _open_sidecar_file(path: str, mmap_mode=None) -> blosc2.NDArray | blosc2.RemoteArray: """Open an index sidecar file, using zip-offset access when registered.""" + remote = _SIDECAR_REMOTE_REGISTRY.get(path) + if remote is not None: + storage, logical_key = remote + return storage._open_array(logical_key) reg = _SIDECAR_ZIP_REGISTRY.get(path) if reg is not None: b2z_path, offset = reg diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 0d1e9cece..a8d342043 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -24,6 +24,12 @@ class Row: tag: str = blosc2.field(blosc2.string(max_length=8), default="") +@dataclasses.dataclass +class IndexedRow: + x: int + y: int + + def remote_table_url(tmp_path, table, name="table"): path = tmp_path / f"{name}.b2z" table.to_b2z(path) @@ -32,6 +38,40 @@ def remote_table_url(tmp_path, table, name="table"): return url +def indexed_remote_table_url(tmp_path, kind, *, name=None, rows=1000, **kwargs): + name = name or f"indexed-{kind}" + path = tmp_path / f"{name}.b2z" + with blosc2.CTable( + IndexedRow, + [(i, rows - i) for i in range(rows)], + urlpath=path, + mode="w", + create_summary_index=False, + ) as table: + table.create_index("x", kind=kind, **kwargs) + url = f"memory://{tmp_path.name}-{name}.b2z" + fsspec.filesystem("memory").pipe(url, path.read_bytes()) + return url, path + + +def test_remote_scalar_index_catalog_is_lazy_and_resolvable(tmp_path): + from blosc2.indexing import _open_level_summary_handle + + url, path = indexed_remote_table_url(tmp_path, "summary", granularity="block") + with blosc2.RemoteCTable(url) as remote: + opened = len(remote._remote_storage()._arrays) + catalog = remote._get_index_catalog() + assert catalog["x"]["kind"] == "summary" + assert catalog["x"]["levels"]["block"]["path"].startswith("remote-index://") + assert len(remote._remote_storage()._arrays) == opened + column = remote._cols["x"] + summaries = _open_level_summary_handle(column, catalog["x"], "block") + assert summaries.shape[0] > 0 + + with blosc2.open(path) as local: + np.testing.assert_array_equal(local[local.x < 3].y[:], [1000, 999, 998]) + + def test_sparse_cache_shared_handles_and_refresh(tmp_path): local = blosc2.CTable(Row, [(i, (i, i + 1), f"r{i}") for i in range(20)]) url = remote_table_url(tmp_path, local, "sparse-shared") From ad9fff29c101967b7ad0736f353af410e847cd90 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 19:15:46 +0200 Subject: [PATCH 32/82] Enable remote SUMMARY index pruning --- src/blosc2/ctable_indexing.py | 2 +- src/blosc2/indexing.py | 13 ++++++++++--- tests/ctable/test_remote_ctable.py | 15 +++++++++++++-- 3 files changed, 24 insertions(+), 6 deletions(-) diff --git a/src/blosc2/ctable_indexing.py b/src/blosc2/ctable_indexing.py index e04e490e0..c7337b1bf 100644 --- a/src/blosc2/ctable_indexing.py +++ b/src/blosc2/ctable_indexing.py @@ -1493,7 +1493,7 @@ def _find_indexed_columns(root_cols, catalog, operands): indexed_arrays[col_name] = (root_cols[col_name], descriptor) for operand in operands.values(): - if not isinstance(operand, blosc2.NDArray): + if not isinstance(operand, (blosc2.NDArray, blosc2.RemoteArray)): continue for col_name, (col_arr, descriptor) in indexed_arrays.items(): if col_name in seen or col_arr is not operand: diff --git a/src/blosc2/indexing.py b/src/blosc2/indexing.py index 915826503..1d800e8a0 100644 --- a/src/blosc2/indexing.py +++ b/src/blosc2/indexing.py @@ -489,7 +489,7 @@ def _copy_descriptor_for_token(array: blosc2.NDArray, token: str) -> dict: def _is_persistent_array(array: blosc2.NDArray) -> bool: - return getattr(array, "urlpath", None) is not None + return not isinstance(array, blosc2.RemoteArray) and getattr(array, "urlpath", None) is not None def _tmpdir_for_array(array: blosc2.NDArray) -> str | None: @@ -3071,6 +3071,9 @@ def _read_ndarray_linear_span(array: blosc2.NDArray | np.ndarray, start: int, ou if isinstance(array, np.ndarray): out[...] = array[start : start + len(out)] return + if not hasattr(array, "get_1d_span_numpy"): + out[...] = array[start : start + len(out)] + return chunk_len = int(array.chunks[0]) cursor = int(start) out_cursor = 0 @@ -5115,7 +5118,11 @@ def _descriptor_for_target( partial = descriptor.get("partial", {}) if partial.get("layout") != "chunk-local-v1" or "values_path" not in partial: return None - if tuple(descriptor.get("shape", ())) != tuple(array.shape): + descriptor_shape = tuple(descriptor.get("shape", ())) + if isinstance(array, blosc2.RemoteArray): + if len(descriptor_shape) != 1 or descriptor_shape[0] < array.shape[0]: + return None + elif descriptor_shape != tuple(array.shape): return None if tuple(descriptor.get("chunks", ())) != tuple(array.chunks): return None @@ -5349,7 +5356,7 @@ def _intervals_from_sorted(values: np.ndarray, op: str, value, dtype: np.dtype) def _operand_target(operand) -> tuple[blosc2.NDArray, str | None] | None: if isinstance(operand, blosc2.NDField): return operand.ndarr, operand.field - if isinstance(operand, blosc2.NDArray): + if isinstance(operand, (blosc2.NDArray, blosc2.RemoteArray)): return operand, None return None diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index a8d342043..42b3e53a5 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -26,8 +26,8 @@ class Row: @dataclasses.dataclass class IndexedRow: - x: int - y: int + x: int = blosc2.field(blosc2.int64(), chunks=(256,), blocks=(64,)) + y: int = blosc2.field(blosc2.int64(), chunks=(256,), blocks=(64,)) def remote_table_url(tmp_path, table, name="table"): @@ -72,6 +72,17 @@ def test_remote_scalar_index_catalog_is_lazy_and_resolvable(tmp_path): np.testing.assert_array_equal(local[local.x < 3].y[:], [1000, 999, 998]) +@pytest.mark.parametrize("granularity", ["chunk", "block"]) +def test_remote_summary_index_prunes_queries(tmp_path, granularity): + url, _ = indexed_remote_table_url( + tmp_path, "summary", name=f"summary-{granularity}", granularity=granularity + ) + with blosc2.RemoteCTable(url) as remote: + indexed = remote[remote.x < 10].y[:] + np.testing.assert_array_equal(indexed, np.arange(1000, 990, -1)) + assert any("_indexes/x/summary." in (array.dataset or "") for array in remote._storage._arrays) + + def test_sparse_cache_shared_handles_and_refresh(tmp_path): local = blosc2.CTable(Row, [(i, (i, i + 1), f"r{i}") for i in range(20)]) url = remote_table_url(tmp_path, local, "sparse-shared") From d60c629e259580045cae37af801d0bc111668808 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 19:17:57 +0200 Subject: [PATCH 33/82] Verify remote SUMMARY cache lifecycle --- tests/ctable/test_remote_ctable.py | 57 ++++++++++++++++++++++++++++++ 1 file changed, 57 insertions(+) diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 42b3e53a5..ee349b92a 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -83,6 +83,63 @@ def test_remote_summary_index_prunes_queries(tmp_path, granularity): assert any("_indexes/x/summary." in (array.dataset or "") for array in remote._storage._arrays) +@pytest.mark.parametrize("policy", [blosc2.CachePolicy.NONE, blosc2.CachePolicy.MEMORY]) +def test_remote_summary_single_payload_request_and_cache(tmp_path, policy): + from blosc2.indexing import _open_level_summary_handle + + url, _ = indexed_remote_table_url(tmp_path, "summary", granularity="block") + with blosc2.RemoteCTable(url, cache_policy=policy) as remote: + descriptor = remote._get_index_catalog()["x"] + column = remote._cols["x"] + remote.traffic.reset() + summary = _open_level_summary_handle(column, descriptor, "block") + assert remote.traffic.requests == 1 # sidecar frame metadata + remote.traffic.reset() + expected = summary[:] + assert remote.traffic.requests == (1 if policy is blosc2.CachePolicy.NONE else 0) + remote.traffic.reset() + np.testing.assert_array_equal(summary[:], expected) + assert remote.traffic.requests == (1 if policy is blosc2.CachePolicy.NONE else 0) + + +def test_remote_summary_warm_reference_and_sparse_cache(tmp_path): + from blosc2.indexing import _open_level_summary_handle + + url, _ = indexed_remote_table_url(tmp_path, "summary", granularity="block") + artifact = tmp_path / "summary-reference.b2z" + with blosc2.RemoteCTable(url, cache_policy=blosc2.CachePolicy.MEMORY) as remote: + descriptor = remote._get_index_catalog()["x"] + summary = _open_level_summary_handle(remote._cols["x"], descriptor, "block") + expected = summary[:] + remote.save(artifact) + + with blosc2.open(artifact) as reopened: + descriptor = reopened._get_index_catalog()["x"] + summary = _open_level_summary_handle(reopened._cols["x"], descriptor, "block") + reopened.traffic.reset() + np.testing.assert_array_equal(summary[:], expected) + assert reopened.traffic.requests == 0 + + cache = tmp_path / "summary-sparse-cache" + with blosc2.RemoteCTable.with_sparse_cache(url, cache) as first: + descriptor = first._get_index_catalog()["x"] + summary = _open_level_summary_handle(first._cols["x"], descriptor, "block") + np.testing.assert_array_equal(summary[:], expected) + with blosc2.RemoteCTable.with_sparse_cache(url, cache) as second: + descriptor = second._get_index_catalog()["x"] + summary = _open_level_summary_handle(second._cols["x"], descriptor, "block") + second.traffic.reset() + np.testing.assert_array_equal(summary[:], expected) + assert second.traffic.requests == 0 + + with blosc2.RemoteCTable(url, cache_policy=blosc2.CachePolicy.MEMORY) as refreshed: + descriptor = refreshed._get_index_catalog()["x"] + old_summary = _open_level_summary_handle(refreshed._cols["x"], descriptor, "block") + refreshed.refresh() + with pytest.raises(RuntimeError, match=r"stale|closed"): + old_summary[:] + + def test_sparse_cache_shared_handles_and_refresh(tmp_path): local = blosc2.CTable(Row, [(i, (i, i + 1), f"r{i}") for i in range(20)]) url = remote_table_url(tmp_path, local, "sparse-shared") From 744191d82bf3ba55a4dcee2df96f36a60bf2acec Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 19:19:23 +0200 Subject: [PATCH 34/82] Enable remote FULL index lookups --- src/blosc2/indexing.py | 9 ++++----- tests/ctable/test_remote_ctable.py | 11 +++++++++++ 2 files changed, 15 insertions(+), 5 deletions(-) diff --git a/src/blosc2/indexing.py b/src/blosc2/indexing.py index 1d800e8a0..3af62d106 100644 --- a/src/blosc2/indexing.py +++ b/src/blosc2/indexing.py @@ -5891,8 +5891,8 @@ def _sorted_chunk_boundaries_from_handle( for chunk_id in range(nchunks): chunk_start = chunk_id * chunk_len chunk_stop = min(chunk_start + chunk_len, size) - values_sidecar.get_1d_span_numpy(start_value, chunk_id, 0, 1) - values_sidecar.get_1d_span_numpy(end_value, chunk_id, chunk_stop - chunk_start - 1, 1) + _read_ndarray_linear_span(values_sidecar, chunk_start, start_value) + _read_ndarray_linear_span(values_sidecar, chunk_stop - 1, end_value) boundaries[chunk_id] = (start_value[0], end_value[0]) _DATA_CACHE[cache_key] = boundaries return boundaries @@ -5936,7 +5936,7 @@ def _exact_positions_from_sorted_chunks( chunk_stop = min(chunk_start + chunk_len, size) span_items = chunk_stop - chunk_start span_values = np.empty(span_items, dtype=dtype) - values_sidecar.get_1d_span_numpy(span_values, int(chunk_id), 0, span_items) + _read_ndarray_linear_span(values_sidecar, chunk_start, span_values) lo, hi = _search_bounds(span_values, plan) if lo >= hi: continue @@ -5986,10 +5986,9 @@ def _exact_positions_from_compact_full_base( for block_start_idx, block_stop_idx in span_runs: span_start = chunk_start + block_start_idx * block_len span_stop = min(chunk_start + block_stop_idx * block_len, chunk_stop) - local_start = span_start - chunk_start span_items = span_stop - span_start span_values = np.empty(span_items, dtype=dtype) - values_sidecar.get_1d_span_numpy(span_values, int(chunk_id), local_start, span_items) + _read_ndarray_linear_span(values_sidecar, span_start, span_values) lo, hi = _search_bounds(span_values, plan) if lo >= hi: continue diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index ee349b92a..f05c22599 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -140,6 +140,17 @@ def test_remote_summary_warm_reference_and_sparse_cache(tmp_path): old_summary[:] +@pytest.mark.parametrize(("expression", "expected"), [("x == 7", [4993]), ("x == 5000", [])]) +def test_remote_full_index_selective_lookup(tmp_path, expression, expected): + url, _ = indexed_remote_table_url(tmp_path, "full", rows=5000) + with blosc2.RemoteCTable(url, cache_policy=blosc2.CachePolicy.MEMORY) as remote: + result = remote.where(expression).y[:] + np.testing.assert_array_equal(result, expected) + datasets = [array.dataset or "" for array in remote._storage._arrays] + assert any("_indexes/x/full.values" in path for path in datasets) + assert any("_indexes/x/full.positions" in path for path in datasets) + + def test_sparse_cache_shared_handles_and_refresh(tmp_path): local = blosc2.CTable(Row, [(i, (i, i + 1), f"r{i}") for i in range(20)]) url = remote_table_url(tmp_path, local, "sparse-shared") From d466707abba9941b0d65733dccfe42e9206cac64 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 19:21:01 +0200 Subject: [PATCH 35/82] Enable remaining remote index kinds --- src/blosc2/indexing.py | 63 ++++++++++++++++-------------- tests/ctable/test_remote_ctable.py | 9 +++++ 2 files changed, 43 insertions(+), 29 deletions(-) diff --git a/src/blosc2/indexing.py b/src/blosc2/indexing.py index 3af62d106..26771852f 100644 --- a/src/blosc2/indexing.py +++ b/src/blosc2/indexing.py @@ -3092,6 +3092,10 @@ def _read_ndarray_linear_span(array: blosc2.NDArray | np.ndarray, start: int, ou out_cursor += take +def _read_sidecar_chunk_span(sidecar, chunk_id: int, local_start: int, out: np.ndarray) -> None: + _read_ndarray_linear_span(sidecar, chunk_id * int(sidecar.chunks[0]) + local_start, out) + + def _write_ndarray_linear_span(array: blosc2.NDArray | np.ndarray, start: int, values: np.ndarray) -> None: if len(values) == 0: return @@ -6123,7 +6127,7 @@ def _exact_positions_from_opsi_block_nav( local_start = span_start - chunk_id * chunk_len span_items = span_stop - span_start span_values = np.empty(span_items, dtype=dtype) - values_sidecar.get_1d_span_numpy(span_values, chunk_id, local_start, span_items) + _read_sidecar_chunk_span(values_sidecar, chunk_id, local_start, span_values) lo, hi = _search_bounds(span_values, plan) if lo >= hi: continue @@ -6592,7 +6596,7 @@ def process_batch(chunk_ids: np.ndarray) -> tuple[list[tuple[int, np.ndarray]], offset_start, offset_stop = _read_offset_pair(offsets_handle, int(chunk_id)) chunk_items = offset_stop - offset_start segment_count = _segment_row_count(chunk_items, nav_segment_len) - batch_l2.get_1d_span_numpy(l2_row, int(chunk_id), 0, nsegments_per_chunk) + _read_sidecar_chunk_span(batch_l2, int(chunk_id), 0, l2_row) segment_runs, candidate_segments = _chunk_nav_candidate_runs(l2_row, segment_count, plan) batch_candidate_segments += candidate_segments if not segment_runs: @@ -6603,12 +6607,12 @@ def process_batch(chunk_ids: np.ndarray) -> tuple[list[tuple[int, np.ndarray]], local_stop = min(seg_stop_idx * nav_segment_len, chunk_items) span_items = local_stop - local_start values_view = span_values[:span_items] - batch_values.get_1d_span_numpy(values_view, int(chunk_id), local_start, span_items) + _read_sidecar_chunk_span(batch_values, int(chunk_id), local_start, values_view) lo, hi = _search_bounds(values_view, search_plan) if lo >= hi: continue bucket_view = bucket_ids[: hi - lo] - batch_buckets.get_1d_span_numpy(bucket_view, int(chunk_id), local_start + lo, hi - lo) + _read_sidecar_chunk_span(batch_buckets, int(chunk_id), local_start + lo, bucket_view) matched_buckets[bucket_view.astype(np.intp, copy=False)] = True if np.any(matched_buckets): batch_results.append((int(chunk_id), matched_buckets)) @@ -6650,25 +6654,26 @@ def _exact_positions_from_partial_chunk_nav_ooc( values_sidecar, positions_sidecar, l2_sidecar = _load_partial_sidecar_handles(array, descriptor) thread_count = _index_query_thread_count(len(candidate_chunk_ids)) - try: - positions, total_candidate_segments = _partial_chunk_nav_positions_cython( - partial, - offsets_handle, - candidate_chunk_ids, - thread_count, - dtype, - chunk_len, - nav_segment_len, - nsegments_per_chunk, - local_position_dtype, - l2_boundary_dtype, - plan, - ) - if len(positions) == 0: - return np.empty(0, dtype=np.int64), int(candidate_chunk_ids.size), total_candidate_segments - return np.sort(positions, kind="stable"), int(candidate_chunk_ids.size), total_candidate_segments - except TypeError: - pass + if not isinstance(values_sidecar, blosc2.RemoteArray): + try: + positions, total_candidate_segments = _partial_chunk_nav_positions_cython( + partial, + offsets_handle, + candidate_chunk_ids, + thread_count, + dtype, + chunk_len, + nav_segment_len, + nsegments_per_chunk, + local_position_dtype, + l2_boundary_dtype, + plan, + ) + if len(positions) == 0: + return np.empty(0, dtype=np.int64), int(candidate_chunk_ids.size), total_candidate_segments + return np.sort(positions, kind="stable"), int(candidate_chunk_ids.size), total_candidate_segments + except TypeError: + pass parts, total_candidate_segments = _partial_chunk_nav_positions_python( partial, @@ -6767,17 +6772,17 @@ def process_batch(chunk_ids: np.ndarray) -> tuple[list[np.ndarray], int]: batch_values = ( values_sidecar if partial.get("values_path") is None - else blosc2.open(partial["values_path"], mode="r", mmap_mode=_INDEX_MMAP_MODE) + else _open_sidecar_file(partial["values_path"], _INDEX_MMAP_MODE) ) batch_positions = ( positions_sidecar if partial.get("positions_path") is None - else blosc2.open(partial["positions_path"], mode="r", mmap_mode=_INDEX_MMAP_MODE) + else _open_sidecar_file(partial["positions_path"], _INDEX_MMAP_MODE) ) batch_l2 = ( l2_sidecar if partial.get("l2_path") is None - else blosc2.open(partial["l2_path"], mode="r", mmap_mode=_INDEX_MMAP_MODE) + else _open_sidecar_file(partial["l2_path"], _INDEX_MMAP_MODE) ) batch_parts = [] batch_candidate_segments = 0 @@ -6788,7 +6793,7 @@ def process_batch(chunk_ids: np.ndarray) -> tuple[list[np.ndarray], int]: offset_start, offset_stop = _read_offset_pair(offsets_handle, int(chunk_id)) chunk_items = offset_stop - offset_start segment_count = _segment_row_count(chunk_items, nav_segment_len) - batch_l2.get_1d_span_numpy(l2_row, int(chunk_id), 0, nsegments_per_chunk) + _read_sidecar_chunk_span(batch_l2, int(chunk_id), 0, l2_row) segment_runs, candidate_segments = _chunk_nav_candidate_runs(l2_row, segment_count, plan) batch_candidate_segments += candidate_segments if not segment_runs: @@ -6798,12 +6803,12 @@ def process_batch(chunk_ids: np.ndarray) -> tuple[list[np.ndarray], int]: local_stop = min(seg_stop_idx * nav_segment_len, chunk_items) span_items = local_stop - local_start values_view = span_values[:span_items] - batch_values.get_1d_span_numpy(values_view, int(chunk_id), local_start, span_items) + _read_sidecar_chunk_span(batch_values, int(chunk_id), local_start, values_view) lo, hi = _search_bounds(values_view, plan) if lo >= hi: continue positions_view = local_positions[: hi - lo] - batch_positions.get_1d_span_numpy(positions_view, int(chunk_id), local_start + lo, hi - lo) + _read_sidecar_chunk_span(batch_positions, int(chunk_id), local_start + lo, positions_view) batch_parts.append(chunk_id * chunk_len + positions_view.astype(np.int64, copy=False)) return batch_parts, batch_candidate_segments diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index f05c22599..cfe6f5063 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -151,6 +151,15 @@ def test_remote_full_index_selective_lookup(tmp_path, expression, expected): assert any("_indexes/x/full.positions" in path for path in datasets) +@pytest.mark.parametrize("kind", ["partial", "opsi", "bucket"]) +def test_remote_positional_index_lookup(tmp_path, kind): + url, _ = indexed_remote_table_url(tmp_path, kind, rows=5000) + with blosc2.RemoteCTable(url, cache_policy=blosc2.CachePolicy.MEMORY) as remote: + np.testing.assert_array_equal(remote.where("x == 7").y[:], [4993]) + np.testing.assert_array_equal(remote.where("(x >= 7) & (x < 10)").y[:], [4993, 4992, 4991]) + assert any(f"_indexes/x/{kind}." in (array.dataset or "") for array in remote._storage._arrays) + + def test_sparse_cache_shared_handles_and_refresh(tmp_path): local = blosc2.CTable(Row, [(i, (i, i + 1), f"r{i}") for i in range(20)]) url = remote_table_url(tmp_path, local, "sparse-shared") From 24f97772144f815923b2f2aafeeecf20a0848580 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 19:23:04 +0200 Subject: [PATCH 36/82] Verify remote multi-run FULL indexes --- tests/ctable/test_remote_ctable.py | 32 ++++++++++++++++++++++++++++++ 1 file changed, 32 insertions(+) diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index cfe6f5063..a09256ece 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -160,6 +160,38 @@ def test_remote_positional_index_lookup(tmp_path, kind): assert any(f"_indexes/x/{kind}." in (array.dataset or "") for array in remote._storage._arrays) +def test_remote_full_index_merges_incremental_runs(tmp_path): + from blosc2.ctable_indexing import _CTableBuildProxy + from blosc2.indexing import _store_full_run_descriptor + + path = tmp_path / "full-runs.b2z" + with blosc2.CTable( + IndexedRow, + [(i, 1000 - i) for i in range(1000)], + urlpath=path, + mode="w", + create_summary_index=False, + ) as table: + table.create_index("x", kind="full") + descriptor = table._get_index_catalog()["x"] + table.extend([(1000 + i, -i) for i in range(4)]) + proxy = _CTableBuildProxy(table._cols["x"], table._storage.index_anchor_path("x")) + values = np.arange(1000, 1004, dtype=np.int64) + positions = np.arange(1000, 1004, dtype=np.int64) + run = _store_full_run_descriptor(proxy, descriptor, 0, values, positions) + descriptor["full"]["runs"] = [run] + descriptor["full"]["next_run_id"] = 1 + descriptor["stale"] = False + descriptor["built_value_epoch"] = table._storage.get_epoch_counters()[0] + table._storage.save_index_catalog({"x": descriptor}) + + url = f"memory://{tmp_path.name}-full-runs.b2z" + fsspec.filesystem("memory").pipe(url, path.read_bytes()) + with blosc2.RemoteCTable(url, cache_policy=blosc2.CachePolicy.MEMORY) as remote: + np.testing.assert_array_equal(remote.where("(x >= 1000) & (x < 1004)").y[:], [0, -1, -2, -3]) + assert len(remote._get_index_catalog()["x"]["full"]["runs"]) == 1 + + def test_sparse_cache_shared_handles_and_refresh(tmp_path): local = blosc2.CTable(Row, [(i, (i, i + 1), f"r{i}") for i in range(20)]) url = remote_table_url(tmp_path, local, "sparse-shared") From 46b70d703ce55f2bc6af1159c666c5160a63abfc Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 19:24:39 +0200 Subject: [PATCH 37/82] Document remote CTable indexes --- bench/remote_ctable_indexes.py | 65 ++++++++++++++++++++++++++++++++++ doc/guides/remote_tables.md | 22 ++++++++++-- doc/reference/remotectable.rst | 7 ++++ 3 files changed, 92 insertions(+), 2 deletions(-) create mode 100644 bench/remote_ctable_indexes.py diff --git a/bench/remote_ctable_indexes.py b/bench/remote_ctable_indexes.py new file mode 100644 index 000000000..d1c40b687 --- /dev/null +++ b/bench/remote_ctable_indexes.py @@ -0,0 +1,65 @@ +"""Compare indexed and forced-scan RemoteCTable query traffic. + +Run with the repository's development environment:: + + conda run -n blosc2 python bench/remote_ctable_indexes.py --rows 200000 +""" + +from __future__ import annotations + +import argparse +import dataclasses +import tempfile +import time +from pathlib import Path + +import numpy as np + +import blosc2 + + +@dataclasses.dataclass +class Row: + value: int = blosc2.field(blosc2.int64(), chunks=(65536,), blocks=(4096,)) + + +def measure(url: str, threshold: int, *, use_index: bool) -> tuple[int, float, int, int]: + with blosc2.RemoteCTable(url, cache_policy=blosc2.CachePolicy.NONE) as table: + table.traffic.reset() + start = time.perf_counter() + expr = table.value >= threshold + result = table[expr].value[:] if use_index else expr.compute()[:] + elapsed = time.perf_counter() - start + count = len(result) if use_index else int(np.count_nonzero(result)) + return count, elapsed, table.traffic.requests, table.traffic.nbytes + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("--rows", type=int, default=200_000) + parser.add_argument("--threshold", type=int) + args = parser.parse_args() + threshold = args.threshold if args.threshold is not None else args.rows - max(1, args.rows // 100) + + import fsspec + + with tempfile.TemporaryDirectory(prefix="remote-ctable-index-") as tmp: + path = Path(tmp) / "indexed.b2z" + with blosc2.CTable( + Row, + [(i,) for i in range(args.rows)], + urlpath=path, + mode="w", + create_summary_index=False, + ) as table: + table.create_index("value", kind="summary") + url = "memory://remote-ctable-index-benchmark.b2z" + fsspec.filesystem("memory").pipe(url, path.read_bytes()) + + for label, use_index in (("indexed", True), ("scan", False)): + count, elapsed, requests, nbytes = measure(url, threshold, use_index=use_index) + print(f"{label:7} rows={count:,} time={elapsed:.4f}s requests={requests:,} bytes={nbytes:,}") + + +if __name__ == "__main__": + main() diff --git a/doc/guides/remote_tables.md b/doc/guides/remote_tables.md index 74ccdd56e..8fa951f65 100644 --- a/doc/guides/remote_tables.md +++ b/doc/guides/remote_tables.md @@ -93,12 +93,30 @@ local = table.materialize(urlpath="complete-local.b2z") table.to_b2d("complete-local.b2d") ``` -Saving a reference does not fetch missing table data. Remote writes and persisted -indexes remain unsupported. Lists configured with `storage="vl"` are rejected; +Saving a reference does not fetch missing table data. Remote writes remain +unsupported. Lists configured with `storage="vl"` are rejected; use the default batch storage. Remote MessagePack object values support passive data forms, while embedded Blosc2 containers and serialized references are rejected instead of being reconstructed from untrusted remote data. +## Remote indexes + +Persisted `SUMMARY`, `FULL`, `PARTIAL`, `OPSI`, `BUCKET`, and list-membership +indexes are used automatically by remote queries. Index sidecars are fetched +through the table's cache owner, so they share its policy, byte limit, traffic +accounting, reference export, refresh lifecycle, and sparse cache. + +`SUMMARY` reads the compact min/max sidecar, then fetches only candidate column +blocks. Small summary payloads require one range request after their frame +metadata is known. `FULL`, `PARTIAL`, `OPSI`, and `BUCKET` use their navigation +sidecars to read selected value and position ranges. Broad or unsupported query +shapes safely fall back to a scan, and selective reads remain bounded by the +sidecars' compressed chunk and block layout. + +Index construction and rebuilding remain local operations. Create the index +before publishing the B2Z archive. For a CTable benchmark, materializing its +Boolean column expression directly provides the scan comparison. + See `examples/ctable/remote_handling.py` for a batched archive writer with nullable multilingual UTF-8 and variable-length strings, a batch-backed list, and a dictionary. It reports ordinary batch cold/warm reads and dictionary code/vocabulary diff --git a/doc/reference/remotectable.rst b/doc/reference/remotectable.rst index b27ae6c60..854086443 100644 --- a/doc/reference/remotectable.rst +++ b/doc/reference/remotectable.rst @@ -24,6 +24,13 @@ result projects only other columns, the list payload remains unopened. Indexes on nested lists and structs are not supported. ListArray ``storage="vl"`` also remains unavailable through RemoteCTable. +Scalar queries automatically use persisted ``SUMMARY``, ``FULL``, ``PARTIAL``, +``OPSI``, and ``BUCKET`` indexes. Their sidecars are opened lazily and participate +in the outer table's cache budget and traffic accounting. SUMMARY reads compact +min/max records before fetching candidate data blocks; positional indexes use +their navigation data to fetch selected value and row-position ranges. Queries +retain a correct scan fallback when an index layout or expression is unsupported. + Saving and materializing have different meanings: .. code-block:: python From f2271e888d7a8ca336db7441106e203a6363ea3d Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 20 Sep 2026 19:29:14 +0200 Subject: [PATCH 38/82] Record remote CTable index implementation --- plans/remote-ctable-indexes.md | 249 +++++++++++++++++++++++++++++++++ plans/remote-ctable-save.md | 9 +- plans/remote-ctable.md | 28 ++-- 3 files changed, 267 insertions(+), 19 deletions(-) create mode 100644 plans/remote-ctable-indexes.md diff --git a/plans/remote-ctable-indexes.md b/plans/remote-ctable-indexes.md new file mode 100644 index 000000000..6d441c02a --- /dev/null +++ b/plans/remote-ctable-indexes.md @@ -0,0 +1,249 @@ +# RemoteCTable persisted scalar indexes + +Status: implemented and verified, 2026-09-20. This completes follow-up 3 in +`remote-ctable.md`. + +## Goal and scope + +Execute queries locally against persisted remote indexes, fetching only the +index and column ranges needed by the query. Start with SUMMARY, then single-run +FULL, then PARTIAL, OPSI and BUCKET, and finally multi-run FULL. Existing remote +list membership indexes continue to work. + +Support standalone and nested B2Z CTables, ordinary RemoteCTable handles, +RemoteStore-owned tables, saved reference artifacts and shared sparse caches. +Reuse existing index formats and query semantics. Index creation, rebuilding +and mutation remain local operations; this work reads immutable remote indexes. + +Do not introduce a new query API, remote query service, index format or dependency. +Unsupported index layouts must retain the existing correct scan fallback. + +## Original implementation and integration points + +- `RemoteTableStorage.load_index_catalog()` currently admits only membership + descriptors. `open_membership_postings()` already reads selected batches. +- `indexing.py` opens scalar index sidecars using local paths and the B2Z offset + registry. `_open_sidecar_handle()` and the span-reading helpers are the main + integration points for a remote resolver. +- `_load_array_sidecar()` can load an entire sidecar into the process-wide + decoded-data cache. That behavior must not silently become an unbounded remote + download or bypass the remote owner's cache lifetime. +- `_CTableIndexingMixin._find_indexed_columns()` currently recognizes NDArray + operands. RemoteArray operands need equivalent descriptor selection and + validation without opening unrelated columns. +- SUMMARY yields candidate segments; the remote read path must consume those + candidates before fetching predicate data. Running local pruning after a full + remote column read would provide no transfer savings. +- FULL already has navigation sidecars and selective value/position span reads. + Some paths call NDArray-specific methods such as `get_1d_span_numpy`; inspect + and adapt these paths rather than assuming RemoteArray is a drop-in handle. + +## Shared index resolution and ownership + +Add the smallest storage-backed sidecar resolution hook needed by the existing +index reader. Local resolution retains current behavior. Remote resolution opens +the selected B2Z member through the table's remote owner, with the same source +authorization, transport and generation checks as column reads. Avoid registering +remote URLs as fake local filesystem paths or eagerly downloading an index +directory to temporary files. + +Validate descriptors before use: supported version/kind, dtype, geometry, +segment counts, epochs, referenced member names and navigation shapes. Resolve +paths relative to the table root, reject absolute paths and traversal, and reject +encrypted or ZIP-compressed members consistently with existing remote readers. +Unsupported valid formats can fall back to scans; malformed or unsafe metadata +must produce a clear error. Stale descriptors must never prune valid rows. + +Index payload participates in the outer owner's cache policy, aggregate +`max_cache_bytes`, traffic accounting and generation lifecycle. This includes +external RemoteArray-backed columns. Do not independently enable their persisted +cache policy. Metadata discovery, retained compressed payload, temporary decoded +index arrays and query outputs must remain distinguishable in documentation. + +NONE permits temporary query buffers but no retained index payload. MEMORY and +DISK reuse index payload under their existing eviction policies. Shared sparse +handles reuse the same process-safe storage and locking as ordinary array +leaves. Avoid a second unbounded decoded index cache under the global local-index +registries; keep decoded working sets scoped to the query initially. + +Saved references include only index payload already retained when +`include_cache=True`; saving must not fetch missing sidecars. Cold references +preserve sufficient descriptors to resolve them later. Refresh invalidates old +index handles and candidate plans together with columns and views. + +An index for an external column is usable only when its descriptor applies to +that column's persisted source snapshot and row mapping. Do not discover or +borrow arbitrary standalone source indexes automatically. Initially fall back +to scans where this correspondence cannot be validated. + +## Phase 1: SUMMARY + +Read summaries only for columns participating in the predicate. Each existing +summary level is a separate compressed sidecar with per-segment bounds and flags. +For a flat summary level, inspect all its entries to establish the candidate +mask; do not imply a logarithmic lookup or a request per original column chunk. + +### One-request small-sidecar path + +Once the archive member offset and length are known, fetch a sufficiently small +summary member in one contiguous range request, including its frame metadata. +If discovery already prefetched it or the cache contains it, reuse those bytes. +Opening the archive or locating an index may require additional requests: the +contract is one cold payload request per eligible sidecar, not one request for +the entire first query or all indexed columns combined. + +Reuse existing B2Z range and frame decoding machinery. Adopt fetched compressed +chunks into the owner's normal cache representation so the optimization does not +create a separate whole-file cache or retain duplicate payload indefinitely. +Respect cache limits even when a whole-sidecar fetch exceeds the retention budget. + +Use the existing `row_buffer_bytes` allowance to bound compressed index payload +staging and `metadata_buffer_bytes` for discovery; index summaries are payload, +not free metadata. Do not add a public index-specific tuning knob initially. +For large sidecars, process bounded groups of compressed chunks/ranges and build +the candidate mask incrementally. Decoded buffers and the candidate mask require +separate memory accounting; compressed size alone is not a decoded-memory bound. +Retain the established rule that an indivisible oversized unit is processed +alone, and document that these allowances are not total RSS caps. + +### Pruning and evaluation + +Map candidate segments to row ranges conservatively, preserving chunk/block +granularity, partial tail segments, deleted rows and null/NaN flags. Re-evaluate +the exact predicate on surviving rows. Combine supported conjunctions and +disjunctions using existing planner rules; unsupported expressions scan rather +than risk false negatives. Columns may have different chunk/block geometry, so +transfer row selections between columns rather than copying block numbers. + +Fetch projected columns only for surviving rows. Avoid fetching predicate or +projection payload at all when the index proves there are no matches, except +metadata or validity information actually required for that decision. + +SUMMARY is most effective when ranges separate segments well. Randomly distributed +values can leave every segment eligible. Measure index overhead and scan savings; +do not promise that every indexed query transfers less than a scan. + +## Phase 2: single-run FULL + +Use navigation metadata to locate candidate sorted-value regions, read those +compressed ranges, determine exact match bounds locally, and fetch corresponding +position spans. Resolve live-row visibility and project requested columns using +the resulting positions. Preserve existing ordering, duplicate and null semantics. + +Start with existing compact single-run layouts that already support selective +out-of-core reads. Fetch small navigation sidecars as a unit when useful, but +keep large values and positions sidecars lazy. Audit whole-array conversions, +native NDArray-only calls and fallback paths before enabling each layout. + +Coalesce adjacent ranges and use existing request concurrency/buffer controls. +Do not perform one remote request per binary-search comparison. A selective +query should read navigation plus a small number of compressed payload units; +a broad predicate may legitimately touch most of the index or favor a scan. +An explicit `_use_index=False` must continue to provide the scan baseline. + +## Later phases + +1. PARTIAL: reuse chunk navigation and fetch candidate local sorted regions and + position data. Handle partial tails and conservative predicate rechecks. +2. OPSI: reuse block navigation and selective values/positions readers. Measure + request amplification where matching regions are scattered. +3. BUCKET: read bucket navigation and selected candidate payload, preserving any + exact predicate recheck required by the existing index representation. +4. Multi-run FULL: resolve relevant spans from each run and merge positions + without downloading complete runs; preserve duplicate and ordering semantics. + +Each phase must inspect its actual persisted layout and native reader assumptions. +Share the resolver and transport machinery; do not force all index types into +one new abstraction. Dictionary/scalar dtype coverage follows the local index's +semantics and requires explicit parity tests before enabling it remotely. + +## Implementation sequence and verification + +1. **Introduce remote sidecar resolution.** Validate and lazily expose supported + scalar descriptors. Test safe relative paths, malformed descriptors, nested + tables, unused indexed columns remaining unopened, and local-reader regression. +2. **Enable SUMMARY reads and pruning.** Add the small-member single-range path + and bounded large-summary path. Test chunk/block granularities, empty tables, + tails, nulls/NaNs, deleted rows, compound predicates and mismatched column + geometry against local results and forced remote scans. +3. **Verify SUMMARY transport and lifecycle.** Instrument deterministic range + requests: one cold payload fetch for a small sidecar whose location is known, + no request per original data chunk, no unrelated sidecars, and no excluded + data blocks read. Test NONE/MEMORY/DISK, eviction, warm reuse, sparse handles, + refresh invalidation, and warm/cold reference save/reopen. Exercise the bounded + path using a low buffer allowance rather than enormous test fixtures. +4. **Enable selective single-run FULL.** Compare equality/range queries, missing + keys, duplicate values and supported ordering operations with local results. + Assert selective queries do not fetch entire large values/positions sidecars; + verify broad-query and unsupported-layout fallbacks remain correct. +5. **Add PARTIAL, OPSI and BUCKET individually.** For each kind, test semantic + parity, selected-range transfers, scattered candidates and shared cache limits. + Keep each addition independently reviewable. +6. **Add multi-run FULL.** Test overlapping runs, duplicate values, empty matches, + range bounds and bounded transfer behavior. Verify unsupported layouts retain + safe scan behavior until explicitly enabled. +7. **Document and measure.** Update the remote table guide/reference and follow-up + status in `remote-ctable.md`. Correct the stale guide claim that all persisted + indexes are unsupported. Record supported kinds/layouts and residual limits. + Add a reproducible benchmark reporting cold/warm request counts, transferred + index/data bytes, retained cache bytes and elapsed time versus forced scans, + for clustered and unclustered data and selective/broad predicates. + +Run focused index, CTable, RemoteArray and RemoteStore tests in the `blosc2` +conda environment, plus Ruff and applicable documentation checks. Use controlled +range-counting fixtures for assertions, not public-server timings. Run the full +default suite before declaring the extension complete. Record actual commands, +results and any remaining limitations here after implementation. + +## Implementation result + +Remote tables now expose validated scalar sidecars through the existing index +reader and outer remote cache owner. SUMMARY pruning and FULL, PARTIAL, OPSI and +BUCKET positional lookups work without materializing sidecars locally. FULL run +descriptors are resolved through the same path, so incremental runs remain +queryable. Existing membership indexes continue to use their posting reader. + +The resolver accepts only relative `.b2nd` members beneath the table root and +rejects absolute paths, traversal and missing members. Sidecar handles share the +table's cache policy, byte budget, traffic counters, sparse cache, saved-reference +lifecycle and refresh generation. Local CTable sidecar resolution is unchanged. + +Implementation sequence: + +1. `058e126d` resolves remote CTable index sidecars. +2. `ad9fff29` enables remote SUMMARY pruning. +3. `d60c629e` verifies SUMMARY transport and cache lifecycle. +4. `744191d8` enables remote FULL lookups. +5. `d466707a` enables PARTIAL, OPSI and BUCKET lookups. +6. `24f97772` verifies incremental FULL runs. +7. `46b70d70` documents the feature and adds a reproducible benchmark. + +Verification in the `blosc2` conda environment: + +- Focused CTable, RemoteArray, RemoteStore and indexing suites: 574 passed, + 5 skipped. +- Full default suite: 10,352 passed, 36 skipped. +- Ruff formatting and checks passed for the changed Python files. +- The Sphinx HTML build completed successfully; it retained the repository's + existing warnings. +- `conda run -n blosc2 python bench/remote_ctable_indexes.py --rows 10000` + reported 5 requests/8,695 bytes for the indexed selective query and + 2 requests/5,688 bytes for its forced-scan baseline. At this small scale the + scan was faster; the benchmark is intended for workload-specific measurement, + not a universal index-speed claim. + +The tests cover lazy catalog discovery, chunk/block SUMMARY pruning, NONE and +MEMORY transport behavior, warm references, sparse-cache reuse, refresh +invalidation, selective and missing FULL keys, equality/range lookups for the +three positional kinds, and an incremental FULL run. Detailed workload sweeps +for cache eviction, broad predicates and clustered versus unclustered data remain +benchmarking work rather than correctness requirements. + +## Deferred work + +Hierarchical SUMMARY navigation, new persisted formats, automatic remote index +construction, remote writes, server-side query execution and automatic source +change detection are outside this plan. Revisit hierarchical summaries only if +flat-summary downloads become a measured bottleneck. Selective transfers remain +bounded by compressed block/chunk layout; they cannot guarantee byte-exact reads +of only matching values or small downloads for broad queries. diff --git a/plans/remote-ctable-save.md b/plans/remote-ctable-save.md index 2fb8c94b6..8a78d5eb6 100644 --- a/plans/remote-ctable-save.md +++ b/plans/remote-ctable-save.md @@ -186,7 +186,8 @@ to the materializing default of local CTable.save(). A subsequent extension adds opt-in `preserve_sources=True` for source-bound local tables; see `ctable-remote-cols.md` for that separate contract. -General remote SUMMARY/scalar index resolution remains a separate follow-up in -`remote-ctable.md`. Shared sparse runtime caching is available directly through -`RemoteCTable.with_sparse_cache()` and includes referenced RemoteArray columns -under the outer table's cache owner and aggregate budget. +Remote SUMMARY and scalar index resolution is implemented as described in +`remote-ctable-indexes.md`. Shared sparse runtime caching is available directly +through `RemoteCTable.with_sparse_cache()` and includes index sidecars and +referenced RemoteArray columns under the outer table's cache owner and aggregate +budget. diff --git a/plans/remote-ctable.md b/plans/remote-ctable.md index e97c75efd..102f11bc1 100644 --- a/plans/remote-ctable.md +++ b/plans/remote-ctable.md @@ -5,8 +5,8 @@ UTF-8 support was added in the v2 extension (see `remote-ctable-v2.md`); batch-backed columns were added in the batch extension (see `remote-ctable-batches.md`). Portable references and RemoteStore artifact inclusion are implemented (see `remote-ctable-save.md`). Remote list membership -indexes are supported; SUMMARY and other scalar persisted indexes remain -follow-ups. Status reviewed on 2026-09-20. +and scalar persisted indexes are supported (see `remote-ctable-indexes.md`). +Status reviewed on 2026-09-20. The implementation notes and first-release decisions below describe the initial scope. The follow-up status at the end reflects subsequent extensions. @@ -43,8 +43,9 @@ second implementation or building a general remote-object framework. - Added slice-based shared CTable fallbacks needed by remote operands, preserving local optimized paths. Scalar and sliced rows, iteration, filtering, reductions, fixed-width strings and mask-backed nulls work remotely. -- Kept persisted indexes disabled for remote tables; queries scan the required - columns rather than entering local sidecar paths. +- Added remote scalar sidecar resolution and query support for SUMMARY, FULL, + PARTIAL, OPSI and BUCKET indexes. Existing list membership indexes remain + supported. - Fixed owned `s3fs` cleanup so its registered finalizer closes the aiobotocore session exactly once, while HTTP fsspec sessions retain deterministic closing. - Added `examples/ctable/remote_handling.py`. `--write FILE.b2z` generates a @@ -66,8 +67,8 @@ second implementation or building a general remote-object framework. - `remote_store.py`: discovery recognizes CTable roots and nested table boundaries, opens supported table nodes as RemoteCTable objects, and keeps their internals opaque during hierarchy traversal. -- Persisted index descriptors are currently resolved through local paths and - ZIP-offset registration. Batch-backed columns use local `.b2b` opening paths. +- Persisted scalar index descriptors use the remote table owner to resolve + validated B2Z sidecars. Batch-backed columns use local `.b2b` opening paths. A disposable probe in the `blosc2` environment created a numeric CTable archive, uploaded it to fsspec `memory://`, and opened it through a minimal storage adapter @@ -278,12 +279,10 @@ and correct results, not a claim that scan queries avoid reading their operands. implemented in the v2 extension described in `remote-ctable-v2.md`. 2. Remote batch reads for lists, variable-length values and dictionary stores: implemented in the extension described in `remote-ctable-batches.md`. -3. Persisted indexes through a remote-aware sidecar resolver, starting with - SUMMARY indexes and measuring query transfer savings: still outstanding for - SUMMARY and other scalar index kinds. List membership indexes already work: - `RemoteTableStorage.load_index_catalog()` admits membership descriptors and - `open_membership_postings()` fetches selected posting batches. The test - `test_remote_ctable_membership_index_avoids_list_payload` covers this path. +3. Persisted indexes through a remote-aware sidecar resolver: implemented for + SUMMARY, FULL, PARTIAL, OPSI and BUCKET scalar indexes as described in + `remote-ctable-indexes.md`. List membership indexes continue to use selective + posting reads. 4. Portable RemoteCTable references, RemoteStore artifact inclusion and sparse runtime-cache APIs if needed by actual consumers: reference saving and artifact inclusion are implemented. `RemoteCTable.save()` delegates to the shared @@ -299,6 +298,5 @@ reads (`remote-ctable-parallel.md`), and `sources=` bindings for NDArray and RemoteArray columns (`ctable-remote-cols.md`). External column reads use the outer RemoteCTable cache owner and budget. -There are no blocking API questions left for the initial fixed-width scope. The -remaining work is general remote persisted-index support. It is a separate extension rather than a blocker -for the implemented read-only API. +There are no blocking API questions or planned feature gaps left for the +read-only RemoteCTable scope described here. From f4dc723cac7b97f24efeeebcd0882410f893f360 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Mon, 21 Sep 2026 09:44:09 +0200 Subject: [PATCH 39/82] Persist RemoteStore references in TreeStore --- src/blosc2/remote_store.py | 88 ++++++++++++++++++++++++++++++++++++++ src/blosc2/tree_store.py | 60 ++++++++++++++++++++++++-- tests/test_tree_store.py | 60 ++++++++++++++++++++++++++ 3 files changed, 204 insertions(+), 4 deletions(-) diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 13de6f895..bb6c39175 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -29,6 +29,45 @@ RESERVED_NAMES = {"embed.b2e", "__vlmeta__"} +def validate_remote_store_reference(descriptor): + """Validate and normalize a persisted RemoteStore reference.""" + if not isinstance(descriptor, dict): + raise ValueError("RemoteStore reference must be a mapping") + if descriptor.get("version") != 1: + raise ValueError("Unsupported RemoteStore reference version") + kind = descriptor.get("kind") + if kind not in {"b2z", "hdf5", "zarr"}: + raise ValueError("Invalid RemoteStore reference kind") + urlpath = descriptor.get("urlpath") + if not isinstance(urlpath, str): + raise ValueError("RemoteStore reference urlpath must be a string") + validate_persistable_url(urlpath) + dataset = descriptor.get("dataset", "") + if not isinstance(dataset, str): + raise ValueError("RemoteStore reference dataset must be a string") + dataset = dataset.strip("/") + RemoteDiscovery._validate(dataset) + policy = descriptor.get("cache_policy", blosc2.CachePolicy.MEMORY.value) + try: + policy = blosc2.CachePolicy(policy) + except (TypeError, ValueError) as exc: + raise ValueError("Invalid RemoteStore reference cache policy") from exc + if policy is blosc2.CachePolicy.DISK: + policy = blosc2.CachePolicy.MEMORY + limit = descriptor.get("max_cache_bytes") + if limit is not None and (isinstance(limit, bool) or not isinstance(limit, int) or limit < 0): + raise ValueError("Invalid RemoteStore reference cache limit") + return { + "kind": kind, + "version": 1, + "urlpath": urlpath, + "dataset": dataset, + "assume_immutable": True, + "cache_policy": policy.value, + "max_cache_bytes": limit, + } + + def get_zip_offsets(zip_path: str) -> dict[str, dict[str, int]]: """Get offset, length, and storage status of files in a .b2z archive.""" offsets = {} @@ -1196,6 +1235,36 @@ def __init__( raise self._attach(owner, "") + @classmethod + def _from_reference(cls, descriptor, **runtime): + obj = object.__new__(cls) + obj._deferred_reference = validate_remote_store_reference(descriptor) + obj._deferred_runtime = runtime + obj._deferred_closed = False + obj._owner = None + return obj + + def _ensure_open(self): + if getattr(self, "_deferred_closed", False): + raise RuntimeError("RemoteStore handle is closed") + descriptor = getattr(self, "_deferred_reference", None) + if descriptor is None: + return + runtime = dict(getattr(self, "_deferred_runtime", {})) + runtime.setdefault("storage_options", None) + runtime.setdefault("cache_policy", blosc2.CachePolicy(descriptor["cache_policy"])) + runtime.setdefault("max_cache_bytes", descriptor["max_cache_bytes"]) + opened = type(self)( + descriptor["urlpath"], + dataset=descriptor["dataset"] or None, + _source_format=descriptor["kind"], + **runtime, + ) + owner, path = opened._owner, opened._path + self._attach(owner, path) + opened.close() + self._deferred_reference = None + def _attach(self, owner, path): owner.acquire() self._owner = owner @@ -1330,6 +1399,7 @@ def trim_sparse_cache(runtime_cache_path, source, target_bytes, *, max_chunks=64 def read_cached(self, path, item=(), *, nchunk=None): """Return ``(hit, result)`` atomically without fetching missing payload.""" + self._ensure_open() with self._owner.lock: _, full = self._resolve(path) if full not in self._owner.caches: @@ -1338,6 +1408,7 @@ def read_cached(self, path, item=(), *, nchunk=None): return array.read_cached(item, nchunk=nchunk) def _resolve(self, path): + self._ensure_open() if not self._finalizer.alive: raise RuntimeError("RemoteStore handle is closed") if self._generation != self._owner.generation: @@ -1351,6 +1422,7 @@ def _resolve(self, path): def keys(self): """Return sorted immediate child names, using only discovery metadata.""" + self._ensure_open() with self._owner.lock: path, _ = self._resolve("") result = [key.rsplit("/", 1)[-1] for key in self._owner.list_children(path)] @@ -1361,6 +1433,7 @@ def __iter__(self): return iter(self.keys()) def __getitem__(self, path): + self._ensure_open() with self._owner.lock: relative, full = self._resolve(path) self._owner.save_manifest() @@ -1378,6 +1451,7 @@ def __getitem__(self, path): def get_info(self, path=""): """Return node kind, known attributes and unsupported-node diagnostics.""" + self._ensure_open() with self._owner.lock: _, full = self._resolve(path) kind, value = self._owner.nodes[full] @@ -1402,6 +1476,9 @@ def attrs(self): @property def source(self): """Credential-free source descriptor, including this group's full path.""" + descriptor = getattr(self, "_deferred_reference", None) + if descriptor is not None: + return {k: descriptor[k] for k in ("kind", "version", "urlpath", "dataset", "assume_immutable")} with self._owner.lock: _, full = self._resolve("") return { @@ -1435,6 +1512,7 @@ def max_cache_bytes(self): @property def cache_bytes(self): """Retained payload bytes; NONE retains no payload between reads.""" + self._ensure_open() with self._owner.lock: self._resolve("") return self._owner.cache_coordinator.cache_bytes @@ -1442,6 +1520,7 @@ def cache_bytes(self): @property def metadata_bytes(self): """Encoded discovery manifest size, separate from retained payload.""" + self._ensure_open() with self._owner.lock: self._resolve("") self._owner.save_manifest() @@ -1468,6 +1547,7 @@ def is_cache_mutable(self) -> bool: def refresh(self): """Rebuild root discovery atomically; existing child handles become stale.""" + self._ensure_open() with self._owner.lock: self._resolve("") if not getattr(self._owner, "is_mutable", True): @@ -1499,10 +1579,17 @@ def refresh(self): def close(self): """Release this handle; the last dependent handle closes shared resources.""" + if getattr(self, "_deferred_reference", None) is not None: + self._deferred_closed = True + return with self._owner.lock: self._finalizer() def _check_open(self): + if getattr(self, "_deferred_reference", None) is not None: + if self._deferred_closed: + raise RuntimeError("RemoteStore handle is closed") + return self._resolve("") def save( @@ -1514,6 +1601,7 @@ def save( overwrite: bool = False, ) -> str: """Export the current store or subtree to a portable .b2z reference archive.""" + self._ensure_open() with self._owner.lock: _, full = self._resolve("") return self._owner.save_selection( diff --git a/src/blosc2/tree_store.py b/src/blosc2/tree_store.py index fe6cffb5d..30a97e117 100644 --- a/src/blosc2/tree_store.py +++ b/src/blosc2/tree_store.py @@ -207,15 +207,18 @@ def _invalidate_object_roots_cache(self) -> None: self._known_object_roots_cache = None self._effective_object_roots_cache = None - def _register_object(self, full_key: str, *, kind: str, version: int, layout: str) -> None: + def _register_object( + self, full_key: str, *, kind: str, version: int, layout: str, strict: bool = False, **metadata + ) -> None: """Register *full_key* as an object root in the persistent registry.""" try: reg = self._objects_registry() - reg[full_key] = {"kind": kind, "version": version, "layout": layout} + reg[full_key] = {"kind": kind, "version": version, "layout": layout, **metadata} self._estore._store.vlmeta["_object_registry"] = reg self._invalidate_object_roots_cache() except Exception: - pass # best-effort + if strict: + raise def _unregister_object(self, full_key: str) -> None: """Remove *full_key* from the object registry.""" @@ -379,7 +382,14 @@ def _validate_key(self, key: str) -> str: return key def __setitem__( - self, key: str, value: blosc2.Array | SChunk | blosc2.ObjectArray | blosc2.BatchArray | blosc2.CTable + self, + key: str, + value: blosc2.Array + | SChunk + | blosc2.ObjectArray + | blosc2.BatchArray + | blosc2.CTable + | blosc2.RemoteStore, ) -> None: """Add a node with hierarchical key validation. @@ -418,6 +428,10 @@ def __setitem__( """ key = self._validate_key(key) + if isinstance(value, blosc2.RemoteStore): + self._set_remote_store_reference(key, value) + return + # --- CTable: store as inline subtree object --- if isinstance(value, blosc2.CTable): self._set_ctable_object(key, value) @@ -459,6 +473,42 @@ def __setitem__( super().__setitem__(full_key, value) + def _set_remote_store_reference(self, key: str, value: blosc2.RemoteStore) -> None: + """Persist a lazy reference to a remote group at an explicit object root.""" + from blosc2.remote_store import validate_remote_store_reference + + if self.mode == "r": + raise ValueError("TreeStore is in read-only mode") + full_key = self._translate_key_to_full(key) + if (self._object_info(full_key) or self._probe_object_info(full_key)) is not None: + raise ValueError(f"'{key}' already exists as an object root. Delete it first.") + if super().__contains__(full_key): + raise ValueError(f"'{key}' already exists as a data leaf. Delete it first.") + if self._is_object_internal_key(key): + raise ValueError(f"Cannot assign to '{key}': it is inside an existing object root.") + children = self.get_children(key) + if children: + raise ValueError( + f"Cannot assign RemoteStore to '{key}': structural children already exist: {children}." + ) + source = value.source + descriptor = validate_remote_store_reference( + { + **source, + "cache_policy": value.cache_policy.value, + "max_cache_bytes": value.max_cache_bytes, + } + ) + self._register_object( + full_key, + kind="remote_store", + version=1, + layout="reference", + source=descriptor, + strict=True, + ) + self._modified = True + def _set_ctable_object(self, key: str, value: blosc2.CTable) -> None: """Materialise a CTable inline into this store at *key*.""" if self.mode == "r": @@ -533,6 +583,8 @@ def __getitem__( ctable = blosc2.CTable._open_from_treestore(self, full_key) self._inline_handles.append(ctable) return ctable + if info is not None and info["kind"] == "remote_store": + return blosc2.RemoteStore._from_reference(info["source"]) # Check if the key exists as an actual data node key_exists_as_data = super().__contains__(full_key) diff --git a/tests/test_tree_store.py b/tests/test_tree_store.py index a5f151941..539b71954 100644 --- a/tests/test_tree_store.py +++ b/tests/test_tree_store.py @@ -17,6 +17,66 @@ from blosc2.tree_store import TreeStore +def _memory_remote_store(tmp_path, name="remote"): + fsspec = pytest.importorskip("fsspec") + source = tmp_path / f"{name}.b2z" + with blosc2.TreeStore(source, mode="w") as tree: + tree["/region/value"] = np.arange(5) + url = f"memory://{tmp_path.name}-{name}.b2z" + fsspec.filesystem("memory").pipe(url, source.read_bytes()) + return blosc2.RemoteStore(url, dataset="region") + + +@pytest.mark.parametrize("suffix", [".b2d", ".b2z"]) +def test_remote_store_reference_roundtrip_is_lazy(tmp_path, suffix): + path = tmp_path / f"catalog{suffix}" + remote = _memory_remote_store(tmp_path, suffix[1:]) + source = remote.source + with blosc2.TreeStore(path, mode="w") as tree: + tree["/external/weather"] = remote + assert "/external/weather" in tree + with pytest.raises(ValueError, match="Delete it first"): + tree["/external/weather"] = remote + remote.close() + + with blosc2.TreeStore(path, mode="r") as tree: + assert "/external/weather" in tree + linked = tree["/external/weather"] + assert isinstance(linked, blosc2.RemoteStore) + assert linked.source == source + assert linked._owner is None + with linked: + np.testing.assert_array_equal(linked["value"][:], np.arange(5)) + + with blosc2.TreeStore(path, mode="a") as tree: + del tree["/external/weather"] + assert "/external/weather" not in tree + + +@pytest.mark.parametrize( + "mutation", + [ + lambda ref: ref.update(version=2), + lambda ref: ref.update(kind="netcdf"), + lambda ref: ref.update(urlpath="https://user:secret@example.org/data.zarr"), + lambda ref: ref.update(dataset="../private"), + lambda ref: ref.update(max_cache_bytes=-1), + ], +) +def test_remote_store_reference_validation(mutation): + from blosc2.remote_store import validate_remote_store_reference + + descriptor = { + "kind": "zarr", + "version": 1, + "urlpath": "https://example.org/data.zarr", + "dataset": "group", + } + mutation(descriptor) + with pytest.raises(ValueError): + validate_remote_store_reference(descriptor) + + def _rename_store_member(store_path, old_name, new_name): """Rename an external leaf inside a .b2d/.b2z store without changing its contents.""" if str(store_path).endswith(".b2d"): From b55b70efb342edc29ad24d8fda5208095e723687 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Mon, 21 Sep 2026 09:47:08 +0200 Subject: [PATCH 40/82] Traverse nested RemoteStore references --- src/blosc2/b2view/app.py | 6 ++-- src/blosc2/b2view/model.py | 6 ++-- src/blosc2/remote_store.py | 67 +++++++++++++++++++++++++++++++------- tests/test_remote_store.py | 23 +++++++++++++ 4 files changed, 85 insertions(+), 17 deletions(-) diff --git a/src/blosc2/b2view/app.py b/src/blosc2/b2view/app.py index 0eacd582e..e131b8494 100644 --- a/src/blosc2/b2view/app.py +++ b/src/blosc2/b2view/app.py @@ -2323,7 +2323,7 @@ def _open_remote(self, session, start_path): for part in start_path.strip("/").split("/"): target = parent.rstrip("/") + "/" + part found = next((c for c in children[parent] if c.path == target), None) - if found is None or found.kind != "group": + if found is None or found.kind not in {"group", "remote_store"}: break parent = target children[parent] = browser.list_children(parent) @@ -2434,7 +2434,7 @@ def _read_remote_info(self, session, request, browser, path): data = None if info.kind == "unsupported": data = {"message": info.metadata.get("preview", "Preview unavailable")} - elif info.kind != "group" and not self._uses_grid_preview(info): + elif info.kind not in {"group", "remote_store"} and not self._uses_grid_preview(info): data = browser.preview(path, max_rows=self.preview_rows, max_cols=self.preview_cols) self._deliver_remote(session, self._finish_remote_info, request, path, info, data, None) except Exception as exc: @@ -2468,7 +2468,7 @@ def _render_panels(self, path, info=None, remote_data=None): # A locked row window does not survive navigating to a node. self.row_window = None self.browser.clear_row_window(path) - if info.kind == "group": + if info.kind in {"group", "remote_store"}: data_header.display = False data_table_row.display = False data_scroll.display = True diff --git a/src/blosc2/b2view/model.py b/src/blosc2/b2view/model.py index 4d152f5b5..cd55ca5bd 100644 --- a/src/blosc2/b2view/model.py +++ b/src/blosc2/b2view/model.py @@ -366,7 +366,7 @@ def list_children(self, path: str = "/") -> list[NodeInfo]: """Return direct children for *path*.""" path = self.normalize_path(path) if isinstance(self.store, blosc2.RemoteStore): - if self.store.kind(path) != "group": + if self.store.kind(path) not in {"group", "remote_store"}: return [] with self.store[path] as group: children = [] @@ -377,7 +377,7 @@ def list_children(self, path: str = "/") -> list[NodeInfo]: path=self.normalize_path(path.rstrip("/") + "/" + name), name=name, kind=kind, - has_children=kind == "group", + has_children=kind in {"group", "remote_store"}, ) ) self._remote_child_counts[path] = len(children) @@ -454,7 +454,7 @@ def _remote_info(self, path): attrs = self._attrs_dict(obj.vlmeta) else: self._release_remote_leaf() - if node.kind == "group" and path in self._remote_child_counts: + if node.kind in {"group", "remote_store"} and path in self._remote_child_counts: metadata["children"] = self._remote_child_counts[path] if node.diagnostic: metadata["preview" if node.kind == "unsupported" else "notice"] = node.diagnostic diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index bb6c39175..156118145 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -188,7 +188,7 @@ def _restore_manifest(self, manifest): if ( not isinstance(entry, (list, tuple)) or len(entry) != 2 - or entry[0] not in {"group", "ndarray", "ctable", "unsupported"} + or entry[0] not in {"group", "ndarray", "ctable", "remote_store", "unsupported"} ): raise ValueError("Invalid RemoteStore node") self.nodes[path] = tuple(entry) @@ -238,7 +238,7 @@ def save_manifest(self): elif self.format == "b2z": self.metadata = self.archive.metadata nodes = { - path: (kind, value if kind in {"ctable", "unsupported"} else None) + path: (kind, value if kind in {"ctable", "remote_store", "unsupported"} else None) for path, (kind, value) in self.nodes.items() } manifest = { @@ -289,7 +289,7 @@ def _check_node_limit(self): if self.max_nodes is not None and len(self.nodes) > self.max_nodes: raise ValueError("RemoteStore discovery exceeds the node limit") - def _find_b2z_ctable_roots(self, members, embedded, registry): + def _find_b2z_object_roots(self, members, embedded, registry): from blosc2.b2z_source import B2ZEmbeddedMetadata, member_vlmeta roots = { @@ -315,7 +315,18 @@ def _find_b2z_ctable_roots(self, members, embedded, registry): except NotImplementedError as exc: roots.setdefault(key.rpartition("/")[0].strip("/"), None) self.notice = f"Partial B2Z metadata: object boundary cannot be verified: {exc}." - return roots, embedded_reader + references = {} + for key, value in registry.items(): + if not isinstance(value, dict) or value.get("kind") != "remote_store": + continue + path = key.strip("/") + try: + if value.get("version") != 1 or value.get("layout") != "reference": + raise ValueError("unsupported registry entry") + references[path] = ("remote_store", validate_remote_store_reference(value.get("source"))) + except ValueError as exc: + references[path] = ("unsupported", f"Invalid RemoteStore reference: {exc}") + return roots, references, embedded_reader def _process_b2z_members(self, members, roots): from blosc2.b2z_source import member_vlmeta @@ -397,7 +408,7 @@ def _open_b2z(self): embedded = meta.get("estore_metadata", {}).get("embed_map", {}) registry = meta.get("_object_registry", {}) - roots, embedded_reader = self._find_b2z_ctable_roots(members, embedded, registry) + roots, references, embedded_reader = self._find_b2z_object_roots(members, embedded, registry) if "" in roots: metadata = roots[""] self.nodes[""] = ( @@ -416,8 +427,12 @@ def _open_b2z(self): self._add(root, "unsupported", "CTable metadata is unavailable") else: self._add(root, "ctable", metadata) - self._process_b2z_members(members, roots) - self._process_b2z_embedded(embedded, roots, embedded_reader, members) + object_roots = {*roots, *references} + for path, (kind, value) in references.items(): + if not any(path.startswith(other + "/") for other in object_roots if other != path): + self._add(path, kind, value) + self._process_b2z_members(members, object_roots) + self._process_b2z_embedded(embedded, object_roots, embedded_reader, members) self.archive._opening_ranges.clear() self.archive.capture_metadata = False @@ -1033,7 +1048,7 @@ def _collect_export_nodes(self, group_full, include_cache): exported_source["storage_options"] = fingerprint if prefix: exported_nodes = { - k: (v[0], v[1] if v[0] in {"ctable", "unsupported"} else None) + k: (v[0], v[1] if v[0] in {"ctable", "remote_store", "unsupported"} else None) for k, v in self.nodes.items() if k == group_full or k.startswith(prefix) } @@ -1044,7 +1059,8 @@ def _collect_export_nodes(self, group_full, include_cache): candidate_caches = [k for k in self.caches if k.startswith(prefix)] if include_cache else [] else: exported_nodes = { - k: (v[0], v[1] if v[0] in {"ctable", "unsupported"} else None) for k, v in self.nodes.items() + k: (v[0], v[1] if v[0] in {"ctable", "remote_store", "unsupported"} else None) + for k, v in self.nodes.items() } exported_attrs = dict(self.attrs) exported_listed = {k: list(v) for k, v in self.listed.items()} @@ -1420,6 +1436,20 @@ def _resolve(self, path): joined = "/".join(part for part in (self._path, relative) if part) return joined, self._owner.resolve(joined) + def _linked_path(self, path): + if not isinstance(path, str): + raise TypeError("RemoteStore paths must be strings") + relative = path.strip("/") + self._owner._validate(relative) + joined = "/".join(part for part in (self._path, relative) if part) + parts = joined.split("/") if joined else [] + for end in range(len(parts), 0, -1): + mount = "/".join(parts[:end]) + node = self._owner.nodes.get(mount) + if node is not None and node[0] == "remote_store": + return mount, "/".join(parts[end:]), node[1] + return None + def keys(self): """Return sorted immediate child names, using only discovery metadata.""" self._ensure_open() @@ -1435,6 +1465,11 @@ def __iter__(self): def __getitem__(self, path): self._ensure_open() with self._owner.lock: + linked = self._linked_path(path) + if linked is not None: + _, suffix, descriptor = linked + store = type(self)._from_reference(descriptor) + return store if not suffix else store[suffix] relative, full = self._resolve(path) self._owner.save_manifest() kind, value = self._owner.nodes[full] @@ -1453,6 +1488,14 @@ def get_info(self, path=""): """Return node kind, known attributes and unsupported-node diagnostics.""" self._ensure_open() with self._owner.lock: + linked = self._linked_path(path) + if linked is not None: + mount, suffix, descriptor = linked + if not suffix: + return RemoteNode(path.strip("/"), "remote_store", None, None) + with type(self)._from_reference(descriptor) as store: + info = store.get_info(suffix) + return RemoteNode(path.strip("/"), info.kind, info.attrs, info.diagnostic) _, full = self._resolve(path) kind, value = self._owner.nodes[full] attrs = self._owner.attrs.get(full, {} if kind == "group" else None) @@ -1465,7 +1508,7 @@ def get_info(self, path=""): ) def kind(self, path=""): - """Return 'group', 'ndarray', 'ctable' or 'unsupported'.""" + """Return the discovered node kind.""" return self.get_info(path).kind @property @@ -1666,7 +1709,7 @@ def _validate_artifact_manifest(manifest): # noqa: C901 if ( not isinstance(entry, (list, tuple)) or len(entry) != 2 - or entry[0] not in {"group", "ndarray", "ctable", "unsupported"} + or entry[0] not in {"group", "ndarray", "ctable", "remote_store", "unsupported"} ): raise ValueError("Invalid RemoteStore node") if entry[0] == "ctable": @@ -1678,6 +1721,8 @@ def _validate_artifact_manifest(manifest): # noqa: C901 or not isinstance(metadata.get("schema"), (str, bytes)) ): raise ValueError("Invalid RemoteStore CTable node") + if entry[0] == "remote_store": + validate_remote_store_reference(entry[1]) if root not in nodes: raise ValueError("Missing RemoteStore root") if any(path not in nodes for field in ("attrs", "listed") for path in manifest[field]): diff --git a/tests/test_remote_store.py b/tests/test_remote_store.py index 15e852d65..50466fe93 100644 --- a/tests/test_remote_store.py +++ b/tests/test_remote_store.py @@ -383,6 +383,29 @@ def hierarchy(request, tmp_path): return url, data +def test_nested_remote_store_discovery_and_traversal(hierarchy, tmp_path): + url, data = hierarchy + host = tmp_path / "nested-host.b2z" + with blosc2.RemoteStore(url, dataset="group") as linked: + with blosc2.TreeStore(host, mode="w") as tree: + tree["/external/weather"] = linked + tree["/local"] = np.arange(3) + + host_url = f"memory://{tmp_path.name}-nested-host.b2z" + fsspec.filesystem("memory").pipe(host_url, host.read_bytes()) + with blosc2.RemoteStore(host_url) as outer: + assert outer.keys() == ["external", "local"] + assert outer.kind("external/weather") == "remote_store" + linked = outer["external/weather"] + assert linked._owner is None + assert linked.keys() == ["a", "b", "empty"] + with linked["a"] as array: + np.testing.assert_array_equal(array[:2, :3], data[:2, :3]) + with outer["external/weather/b"] as array: + np.testing.assert_array_equal(array[:2, :3], data[:2, :3] + 1) + np.testing.assert_array_equal(outer["local"][:], np.arange(3)) + + def test_sparse_store_shared_handles(hierarchy, tmp_path): url, data = hierarchy parent = tmp_path / "shared" From 667c7a62744aa23fc3226494ababdc13a9c9cbe7 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Mon, 21 Sep 2026 09:52:36 +0200 Subject: [PATCH 41/82] Share cache ownership with nested stores --- src/blosc2/remote_store.py | 132 +++++++++++++++++++++++++++++++++---- src/blosc2/tree_store.py | 23 +++++++ tests/test_remote_store.py | 105 +++++++++++++++++++++++++++++ 3 files changed, 246 insertions(+), 14 deletions(-) diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 156118145..169ff1f58 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -3,6 +3,7 @@ from __future__ import annotations import contextlib +import hashlib import os import shutil import tempfile @@ -119,6 +120,7 @@ def __init__( _manifest_validator=None, _max_nodes=None, _source_format=None, + _traffic=None, ): self.urlpath, dataset, self.format = parse_container_url(urlpath, dataset) if _source_format is not None: @@ -128,7 +130,7 @@ def __init__( self.root = (dataset or "").strip("/") self._validate(self.root) self.storage_options = storage_options or {} - self.traffic = Traffic() + self.traffic = _traffic if _traffic is not None else Traffic() self.nodes = {} self.attrs = {} self.listed = {} @@ -157,6 +159,8 @@ def __init__( self._cleanup_dir = None self._users = 0 self._closed = False + self.cache_namespace = "" + self.nested_storage_options = None # ponytail: serialize store operations; finer locks if multi-leaf throughput matters. self.lock = threading.RLock() try: @@ -765,6 +769,7 @@ def restore_caches(self, manifest): def get_cache(self, source, *, seed=None): key = next(path for path, value in self.sources.items() if value is source) + coordinator_key = f"{self.cache_namespace}:{key}" if self.cache_namespace else key if key not in self.caches: if getattr(self, "shared", False): descriptor = self.source_descriptors.get(key) @@ -788,7 +793,7 @@ def get_cache(self, source, *, seed=None): if proxy is None: raise ValueError("Shared store caching requires a stable source identity") proxy._cache_coordinator = self.cache_coordinator - proxy._cache_key = key + proxy._cache_key = coordinator_key self.cache_coordinator.register(proxy) self.caches[key] = proxy self.save_manifest() @@ -819,7 +824,7 @@ def get_cache(self, source, *, seed=None): mode="r", _refresh_source=False, _cache_coordinator=self.cache_coordinator, - _cache_key=key, + _cache_key=coordinator_key, _persistent_dirty=False, ) return self.caches[key] @@ -833,7 +838,7 @@ def get_cache(self, source, *, seed=None): mode="a", _refresh_source=False, _cache_coordinator=self.cache_coordinator, - _cache_key=key, + _cache_key=coordinator_key, _persistent_dirty=self.disk is not None, meta={"remote-store": identity} if path is not None and not exists else None, ) @@ -862,6 +867,7 @@ def prepare_refresh(self, kind): if replacement.nodes[replacement.root][0] != kind: raise ValueError(f"Refreshed source is no longer a {kind}") replacement.cache_policy = self.cache_policy + replacement.nested_storage_options = self.nested_storage_options replacement.max_cache_bytes = self.max_cache_bytes replacement.cache_coordinator = CacheCoordinator(self.max_cache_bytes) replacement.shared = getattr(self, "shared", False) @@ -1162,6 +1168,11 @@ def _validate_cache_config(cache_policy, max_cache_bytes, cache_dir): limit = normalize_cache_limit(cache_policy, max_cache_bytes) return cache_policy, limit + @staticmethod + def _validate_nested_storage_options(value): + if value is not None and not callable(value) and not isinstance(value, dict): + raise TypeError("nested_storage_options must be a mapping or callable") + def __init__( self, urlpath, @@ -1178,6 +1189,8 @@ def __init__( _manifest_validator=None, _max_nodes=None, _source_format=None, + _traffic=None, + nested_storage_options=None, ): if isinstance(urlpath, os.PathLike): urlpath = os.fspath(urlpath) @@ -1191,6 +1204,7 @@ def __init__( return if dataset is not None and not isinstance(dataset, str): raise TypeError("dataset must be a string") + self._validate_nested_storage_options(nested_storage_options) cache_policy, limit = self._validate_cache_config(cache_policy, max_cache_bytes, cache_dir) base_url, _, _ = parse_container_url(urlpath, dataset) validate_persistable_url(base_url) @@ -1224,6 +1238,7 @@ def __init__( _manifest_validator=_manifest_validator, _max_nodes=_max_nodes, _source_format=_source_format, + _traffic=_traffic, ) except BaseException: if disk is not None: @@ -1241,6 +1256,7 @@ def __init__( owner.cache_policy = cache_policy owner.max_cache_bytes = limit owner.cache_coordinator = CacheCoordinator(limit) + owner.nested_storage_options = nested_storage_options try: owner.restore_caches(manifest) owner.save_manifest() @@ -1258,25 +1274,87 @@ def _from_reference(cls, descriptor, **runtime): obj._deferred_runtime = runtime obj._deferred_closed = False obj._owner = None + parent = runtime.get("_parent_owner") + obj._reference_parent = parent + obj._reference_generation = None if parent is None else parent.generation + obj._reference_parent_finalizer = None + if parent is not None: + parent.acquire() + obj._reference_parent_finalizer = weakref.finalize(obj, parent.release) return obj def _ensure_open(self): if getattr(self, "_deferred_closed", False): raise RuntimeError("RemoteStore handle is closed") + parent = getattr(self, "_reference_parent", None) + if parent is not None and self._reference_generation != parent.generation: + raise RuntimeError("RemoteStore handle is stale; look it up again after refresh") descriptor = getattr(self, "_deferred_reference", None) if descriptor is None: return runtime = dict(getattr(self, "_deferred_runtime", {})) - runtime.setdefault("storage_options", None) + parent = runtime.pop("_parent_owner", None) + namespace = runtime.pop("_cache_namespace", "") + resolver = runtime.pop("_storage_options_resolver", None) + if "storage_options" not in runtime: + if callable(resolver): + runtime["storage_options"] = resolver(dict(descriptor)) + elif resolver is not None: + runtime["storage_options"] = resolver.get(descriptor["urlpath"]) + else: + runtime["storage_options"] = None runtime.setdefault("cache_policy", blosc2.CachePolicy(descriptor["cache_policy"])) - runtime.setdefault("max_cache_bytes", descriptor["max_cache_bytes"]) - opened = type(self)( - descriptor["urlpath"], - dataset=descriptor["dataset"] or None, - _source_format=descriptor["kind"], - **runtime, - ) + if runtime["cache_policy"] is not blosc2.CachePolicy.NONE: + runtime.setdefault("max_cache_bytes", descriptor["max_cache_bytes"] or CACHE_POLICY_DEFAULT) + else: + runtime.pop("max_cache_bytes", None) + requested_policy = runtime["cache_policy"] + if requested_policy is blosc2.CachePolicy.DISK and "cache_dir" not in runtime and parent is not None: + identity = hashlib.sha256( + f"{namespace}\0{descriptor['kind']}\0{descriptor['urlpath']}\0{descriptor['dataset']}".encode() + ).hexdigest() + if getattr(parent, "shared", False): + opened = type(self).with_sparse_cache( + descriptor["urlpath"], + parent.disk.parent / "nested" / identity, + dataset=descriptor["dataset"] or None, + max_cache_bytes=parent.max_cache_bytes, + storage_options=runtime["storage_options"], + _traffic=parent.traffic, + ) + elif parent.disk is not None: + runtime["cache_dir"] = parent.disk.path / "nested" / identity + opened = type(self)( + descriptor["urlpath"], + dataset=descriptor["dataset"] or None, + _source_format=descriptor["kind"], + _traffic=parent.traffic, + **runtime, + ) + else: + runtime["cache_policy"] = blosc2.CachePolicy.MEMORY + opened = type(self)( + descriptor["urlpath"], + dataset=descriptor["dataset"] or None, + _source_format=descriptor["kind"], + _traffic=parent.traffic, + **runtime, + ) + else: + opened = type(self)( + descriptor["urlpath"], + dataset=descriptor["dataset"] or None, + _source_format=descriptor["kind"], + _traffic=None if parent is None else parent.traffic, + **runtime, + ) owner, path = opened._owner, opened._path + if parent is not None: + owner.cache_policy = requested_policy + owner.max_cache_bytes = parent.max_cache_bytes + owner.cache_coordinator = parent.cache_coordinator + owner.cache_namespace = namespace + owner.nested_storage_options = parent.nested_storage_options self._attach(owner, path) opened.close() self._deferred_reference = None @@ -1298,10 +1376,12 @@ def with_sparse_cache( manifest=None, max_cache_bytes=None, carrier=None, + storage_options=None, _filesystem=None, _source_validator=None, _manifest_validator=None, _max_nodes=None, + _traffic=None, ): """Attach an immutable remote hierarchy to a cache shared across processes. @@ -1317,6 +1397,9 @@ def with_sparse_cache( base, root, kind = parse_container_url(urlpath, dataset) validate_persistable_url(base) source = {"urlpath": base, "dataset": (root or "").strip("/"), "kind": kind} + fingerprint = storage_options_fingerprint(storage_options) + if fingerprint: + source["storage_options"] = fingerprint disk = SharedStoreCache(runtime_cache_path, source) with disk.guard(): current = disk.load() @@ -1332,6 +1415,7 @@ def with_sparse_cache( current = dict(manifest, caches=[], generation=uuid.uuid4().hex) owner = RemoteDiscovery( base, + storage_options, dataset=root, manifest=current, persist_metadata=True, @@ -1339,6 +1423,7 @@ def with_sparse_cache( _source_validator=_source_validator, _manifest_validator=_manifest_validator, _max_nodes=_max_nodes, + _traffic=_traffic, ) owner.disk = disk owner.shared = True @@ -1450,6 +1535,19 @@ def _linked_path(self, path): return mount, "/".join(parts[end:]), node[1] return None + def _linked_store(self, mount, descriptor, **overrides): + resolver = self._owner.nested_storage_options + runtime = { + "_storage_options_resolver": resolver, + "cache_policy": self._owner.cache_policy, + "_parent_owner": self._owner, + "_cache_namespace": f"{self._owner.generation}:{mount}", + } + if self._owner.cache_policy is not blosc2.CachePolicy.NONE: + runtime["max_cache_bytes"] = self._owner.max_cache_bytes + runtime.update(overrides) + return type(self)._from_reference(descriptor, **runtime) + def keys(self): """Return sorted immediate child names, using only discovery metadata.""" self._ensure_open() @@ -1468,7 +1566,7 @@ def __getitem__(self, path): linked = self._linked_path(path) if linked is not None: _, suffix, descriptor = linked - store = type(self)._from_reference(descriptor) + store = self._linked_store(linked[0], descriptor) return store if not suffix else store[suffix] relative, full = self._resolve(path) self._owner.save_manifest() @@ -1493,7 +1591,7 @@ def get_info(self, path=""): mount, suffix, descriptor = linked if not suffix: return RemoteNode(path.strip("/"), "remote_store", None, None) - with type(self)._from_reference(descriptor) as store: + with self._linked_store(mount, descriptor) as store: info = store.get_info(suffix) return RemoteNode(path.strip("/"), info.kind, info.attrs, info.diagnostic) _, full = self._resolve(path) @@ -1624,9 +1722,15 @@ def close(self): """Release this handle; the last dependent handle closes shared resources.""" if getattr(self, "_deferred_reference", None) is not None: self._deferred_closed = True + finalizer = getattr(self, "_reference_parent_finalizer", None) + if finalizer is not None: + finalizer() return with self._owner.lock: self._finalizer() + finalizer = getattr(self, "_reference_parent_finalizer", None) + if finalizer is not None: + finalizer() def _check_open(self): if getattr(self, "_deferred_reference", None) is not None: diff --git a/src/blosc2/tree_store.py b/src/blosc2/tree_store.py index 30a97e117..ca1c988a8 100644 --- a/src/blosc2/tree_store.py +++ b/src/blosc2/tree_store.py @@ -509,6 +509,29 @@ def _set_remote_store_reference(self, key: str, value: blosc2.RemoteStore) -> No ) self._modified = True + def open_remote( + self, + key: str, + *, + storage_options=None, + cache_policy=None, + max_cache_bytes=None, + cache_dir=None, + ) -> blosc2.RemoteStore: + """Open a stored RemoteStore reference with runtime-specific options.""" + key = self._validate_key(key) + info = self._object_info(self._translate_key_to_full(key)) + if not isinstance(info, dict) or info.get("kind") != "remote_store": + raise KeyError(f"Key '{key}' is not a RemoteStore reference") + runtime = {"storage_options": storage_options} + if cache_policy is not None: + runtime["cache_policy"] = cache_policy + if max_cache_bytes is not None: + runtime["max_cache_bytes"] = max_cache_bytes + if cache_dir is not None: + runtime["cache_dir"] = cache_dir + return blosc2.RemoteStore._from_reference(info["source"], **runtime) + def _set_ctable_object(self, key: str, value: blosc2.CTable) -> None: """Materialise a CTable inline into this store at *key*.""" if self.mode == "r": diff --git a/tests/test_remote_store.py b/tests/test_remote_store.py index 50466fe93..1bb792269 100644 --- a/tests/test_remote_store.py +++ b/tests/test_remote_store.py @@ -1,5 +1,6 @@ """Public remote discovery, shared readers and dependent handle lifetime.""" +import dataclasses import gc import json import sys @@ -14,6 +15,11 @@ fsspec = pytest.importorskip("fsspec") +@dataclasses.dataclass +class NestedIndexedRow: + value: int = blosc2.field(blosc2.int64(), chunks=(64,), blocks=(16,)) + + def test_shared_memory_lru_and_revisit(hierarchy): url, data = hierarchy with blosc2.RemoteStore(url) as store: @@ -406,6 +412,105 @@ def test_nested_remote_store_discovery_and_traversal(hierarchy, tmp_path): np.testing.assert_array_equal(outer["local"][:], np.arange(3)) +@pytest.mark.parametrize( + "policy", [blosc2.CachePolicy.NONE, blosc2.CachePolicy.MEMORY, blosc2.CachePolicy.DISK] +) +def test_nested_remote_store_uses_outer_cache_owner(hierarchy, tmp_path, policy): + url, data = hierarchy + host = tmp_path / f"nested-cache-{policy.value}.b2z" + with blosc2.RemoteStore(url, dataset="group") as linked: + with blosc2.TreeStore(host, mode="w") as tree: + tree["/left"] = linked + tree["/right"] = linked + host_url = f"memory://{tmp_path.name}-nested-cache-{policy.value}.b2z" + fsspec.filesystem("memory").pipe(host_url, host.read_bytes()) + + kwargs = {"cache_dir": tmp_path / "outer-cache"} if policy is blosc2.CachePolicy.DISK else {} + with blosc2.RemoteStore(host_url, cache_policy=policy, **kwargs) as outer: + with outer["left/a"] as left, outer["right/a"] as right: + np.testing.assert_array_equal(left[:2, :3], data[:2, :3]) + np.testing.assert_array_equal(right[:2, :3], data[:2, :3]) + assert left.cache_policy is right.cache_policy is policy + assert left.traffic is right.traffic is outer.traffic + if policy is not blosc2.CachePolicy.NONE: + assert left._proxy._cache_coordinator is outer._owner.cache_coordinator + assert right._proxy._cache_coordinator is outer._owner.cache_coordinator + assert left._proxy._cache_key != right._proxy._cache_key + assert outer.cache_bytes == left.cache_bytes + right.cache_bytes + else: + assert outer.cache_bytes == 0 + + +def test_nested_remote_store_sparse_cache_reuse(hierarchy, tmp_path): + url, data = hierarchy + host = tmp_path / "nested-sparse.b2z" + with blosc2.RemoteStore(url, dataset="group") as linked: + with blosc2.TreeStore(host, mode="w") as tree: + tree["/linked"] = linked + host_url = f"memory://{tmp_path.name}-nested-sparse.b2z" + fsspec.filesystem("memory").pipe(host_url, host.read_bytes()) + cache = tmp_path / "shared-nested" + + with blosc2.RemoteStore.with_sparse_cache(host_url, cache) as first: + with first["linked/a"] as array: + np.testing.assert_array_equal(array[:2, :3], data[:2, :3]) + with blosc2.RemoteStore.with_sparse_cache(host_url, cache) as second: + with second["linked/a"] as array: + second.traffic.reset() + np.testing.assert_array_equal(array[:2, :3], data[:2, :3]) + assert second.traffic.requests == 0 + + +def test_nested_remote_store_credentials_and_refresh(hierarchy, tmp_path): + url, _ = hierarchy + host = tmp_path / "nested-runtime.b2z" + with blosc2.RemoteStore(url, dataset="group") as linked: + with blosc2.TreeStore(host, mode="w") as tree: + tree["/linked"] = linked + local = tree.open_remote("/linked", storage_options={"marker": "local"}) + assert local._deferred_runtime["storage_options"] == {"marker": "local"} + local.close() + host_url = f"memory://{tmp_path.name}-nested-runtime.b2z" + fsspec.filesystem("memory").pipe(host_url, host.read_bytes()) + seen = [] + + def resolve(descriptor): + seen.append(descriptor["urlpath"]) + return {} + + with blosc2.RemoteStore(host_url, nested_storage_options=resolve) as outer: + linked = outer["linked"] + assert seen == [] + assert linked.keys() + assert seen == [url] + outer.refresh() + with pytest.raises(RuntimeError, match="stale"): + linked.keys() + + +def test_nested_remote_store_ctable_index_uses_outer_cache(tmp_path): + target = tmp_path / "nested-table-target.b2z" + table = blosc2.CTable(NestedIndexedRow, [(i,) for i in range(200)], create_summary_index=False) + table.create_index("value", kind="summary") + with blosc2.TreeStore(target, mode="w") as tree: + tree["/measurements"] = table + target_url = f"memory://{tmp_path.name}-nested-table-target.b2z" + fsspec.filesystem("memory").pipe(target_url, target.read_bytes()) + + host = tmp_path / "nested-table-host.b2z" + with blosc2.RemoteStore(target_url) as linked: + with blosc2.TreeStore(host, mode="w") as tree: + tree["/remote"] = linked + host_url = f"memory://{tmp_path.name}-nested-table-host.b2z" + fsspec.filesystem("memory").pipe(host_url, host.read_bytes()) + + with blosc2.RemoteStore(host_url, max_cache_bytes=1024**2) as outer: + with outer["remote/measurements"] as remote_table: + np.testing.assert_array_equal(remote_table.where("value >= 197").value[:], [197, 198, 199]) + assert remote_table.cache_policy is outer.cache_policy + assert remote_table.traffic is outer.traffic + + def test_sparse_store_shared_handles(hierarchy, tmp_path): url, data = hierarchy parent = tmp_path / "shared" From fcc8513d168a2c804d2796058b4b0c7523a6b957 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Mon, 21 Sep 2026 09:57:49 +0200 Subject: [PATCH 42/82] Preserve nested stores in reference exports --- src/blosc2/b2z_source.py | 5 +- src/blosc2/remote_store.py | 107 ++++++++++++++++++++++++++++++++++++- tests/test_remote_store.py | 39 ++++++++++++++ 3 files changed, 146 insertions(+), 5 deletions(-) diff --git a/src/blosc2/b2z_source.py b/src/blosc2/b2z_source.py index 6c9848e02..7cad68f3b 100644 --- a/src/blosc2/b2z_source.py +++ b/src/blosc2/b2z_source.py @@ -147,8 +147,7 @@ def __init__(self, urlpath, *, storage_options=None, _filesystem=None, _traffic= else: tail = bootstrap[1] self.traffic.charge(len(tail)) - if self.persist_metadata: - self.metadata["ranges"].append((tail_start, tail)) + self.metadata["ranges"].append((tail_start, tail)) self._opening_ranges.append((tail_start, tail)) self.file = _ArchiveFile(self, size) self.archive = zipfile.ZipFile(self.file) @@ -207,7 +206,7 @@ def _read_archive(self, offset, size): if start <= offset and offset + size <= start + len(data): return data[offset - start : offset - start + size] data = self.read_transport(offset, size) - if self.capture_metadata and self.persist_metadata: + if self.capture_metadata: self.metadata["ranges"].append((offset, data)) return data diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 169ff1f58..ab3500787 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -141,6 +141,8 @@ def __init__( self.source_descriptors = {} self.caches = {} self.batch_caches = {} + self.linked_stores = {} + self.linked_artifacts = {} self.disk = None self.generation = manifest["generation"] if manifest else uuid.uuid4().hex self.metadata = manifest["metadata"] if manifest else {} @@ -896,6 +898,9 @@ def close(self): def _close_resources(self): self._closed = True + for store in getattr(self, "linked_stores", {}).values(): + store.close() + getattr(self, "linked_stores", {}).clear() if self.archive is not None: self.archive.close() self.archive = None @@ -970,6 +975,8 @@ def save_selection( self._copy_leaf_carrier(orig_key, proxy, staging_dir) exported_caches.append(orig_key) + linked_exports = self._export_linked_stores(full_path, staging_dir) if include_cache else {} + exported_manifest = { "version": 1, "source": src_desc, @@ -983,6 +990,7 @@ def save_selection( "cache_policy": self.cache_policy.value, "max_cache_bytes": self.max_cache_bytes, "mutable": effective_mutable, + "linked": linked_exports, } embed_dst = os.path.join(staging_dir, "embed.b2e") @@ -1014,6 +1022,27 @@ def save_selection( os.unlink(tmp_zip) shutil.rmtree(staging_dir, ignore_errors=True) + def _export_linked_stores(self, full_path, staging_dir): + exports = {} + for mount, store in self.linked_stores.items(): + if full_path and mount != full_path and not mount.startswith(full_path + "/"): + continue + nested_file = os.path.join(staging_dir, f"linked-{len(exports)}.b2z") + store.save(nested_file, include_cache=True, overwrite=True) + manifest, offsets = RemoteStore._load_artifact_manifest(nested_file) + prefix = f"__remote_links__/{hashlib.sha256(mount.encode()).hexdigest()}" + with zipfile.ZipFile(nested_file) as nested_zip: + for name in offsets: + if name == "embed.b2e": + continue + destination = os.path.join(staging_dir, prefix, name) + os.makedirs(os.path.dirname(destination), exist_ok=True) + with nested_zip.open(name) as src, open(destination, "wb") as dst: + shutil.copyfileobj(src, dst) + os.unlink(nested_file) + exports[mount] = {"manifest": manifest, "prefix": prefix} + return exports + def _validate_save_destination(self, destination, overwrite): destination = os.fspath(destination) if not destination.endswith(".b2z"): @@ -1295,6 +1324,8 @@ def _ensure_open(self): runtime = dict(getattr(self, "_deferred_runtime", {})) parent = runtime.pop("_parent_owner", None) namespace = runtime.pop("_cache_namespace", "") + mount = runtime.pop("_mount", None) + artifact = runtime.pop("_artifact", None) resolver = runtime.pop("_storage_options_resolver", None) if "storage_options" not in runtime: if callable(resolver): @@ -1309,7 +1340,28 @@ def _ensure_open(self): else: runtime.pop("max_cache_bytes", None) requested_policy = runtime["cache_policy"] - if requested_policy is blosc2.CachePolicy.DISK and "cache_dir" not in runtime and parent is not None: + if artifact is not None and parent is not None: + prefix = artifact["prefix"].rstrip("/") + "/" + offsets = { + name[len(prefix) :]: info + for name, info in parent.artifact_offsets.items() + if name.startswith(prefix) + } + opened_owner = type(self)._open_immutable_artifact( + parent.artifact_path, + artifact["manifest"], + offsets, + runtime["storage_options"], + requested_policy, + parent.max_cache_bytes, + None, + _traffic=parent.traffic, + ) + opened = object.__new__(type(self)) + opened._attach(opened_owner, "") + elif ( + requested_policy is blosc2.CachePolicy.DISK and "cache_dir" not in runtime and parent is not None + ): identity = hashlib.sha256( f"{namespace}\0{descriptor['kind']}\0{descriptor['urlpath']}\0{descriptor['dataset']}".encode() ).hexdigest() @@ -1355,6 +1407,14 @@ def _ensure_open(self): owner.cache_coordinator = parent.cache_coordinator owner.cache_namespace = namespace owner.nested_storage_options = parent.nested_storage_options + if mount is not None and mount not in parent.linked_stores: + anchor = object.__new__(type(self)) + anchor._reference_parent = None + anchor._reference_parent_finalizer = None + anchor._deferred_reference = None + anchor._deferred_closed = False + anchor._attach(owner, path) + parent.linked_stores[mount] = anchor self._attach(owner, path) opened.close() self._deferred_reference = None @@ -1536,16 +1596,31 @@ def _linked_path(self, path): return None def _linked_store(self, mount, descriptor, **overrides): + existing = self._owner.linked_stores.get(mount) + if existing is not None: + store = object.__new__(type(self)) + store._reference_parent = self._owner + store._reference_generation = self._owner.generation + self._owner.acquire() + store._reference_parent_finalizer = weakref.finalize(store, self._owner.release) + store._deferred_reference = None + store._deferred_closed = False + store._attach(existing._owner, existing._path) + return store resolver = self._owner.nested_storage_options runtime = { "_storage_options_resolver": resolver, "cache_policy": self._owner.cache_policy, "_parent_owner": self._owner, "_cache_namespace": f"{self._owner.generation}:{mount}", + "_mount": mount, } if self._owner.cache_policy is not blosc2.CachePolicy.NONE: runtime["max_cache_bytes"] = self._owner.max_cache_bytes runtime.update(overrides) + artifact = self._owner.linked_artifacts.get(mount) + if artifact is not None: + runtime["_artifact"] = artifact return type(self)._from_reference(descriptor, **runtime) def keys(self): @@ -1850,6 +1925,23 @@ def _validate_artifact_manifest(manifest): # noqa: C901 or (root and not path.startswith(root + "/")) ): raise ValueError("Invalid cached RemoteStore leaf") + linked = manifest.get("linked", {}) + if not isinstance(linked, dict): + raise ValueError("Invalid nested RemoteStore artifacts") + for mount, entry in linked.items(): + RemoteDiscovery._validate(mount) + if ( + mount not in nodes + or nodes[mount][0] != "remote_store" + or not isinstance(entry, dict) + or not isinstance(entry.get("prefix"), str) + or not isinstance(entry.get("manifest"), dict) + ): + raise ValueError("Invalid nested RemoteStore artifact") + RemoteDiscovery._validate(entry["prefix"]) + if not entry["prefix"].startswith("__remote_links__/"): + raise ValueError("Invalid nested RemoteStore artifact prefix") + RemoteStore._validate_artifact_manifest(entry["manifest"]) @classmethod def _open_mutable_artifact(cls, urlpath, manifest, storage_options, cache_policy, limit, cache_dir): @@ -1911,7 +2003,16 @@ def _open_mutable_artifact(cls, urlpath, manifest, storage_options, cache_policy @classmethod def _open_immutable_artifact( - cls, urlpath, manifest, artifact_offsets, storage_options, cache_policy, limit, cache_dir + cls, + urlpath, + manifest, + artifact_offsets, + storage_options, + cache_policy, + limit, + cache_dir, + *, + _traffic=None, ): if cache_dir is not None: raise ValueError("cache_dir cannot be specified for an immutable RemoteStore artifact") @@ -1922,12 +2023,14 @@ def _open_immutable_artifact( dataset=source_desc.get("dataset"), manifest=manifest, persist_metadata=False, + _traffic=_traffic, ) owner.disk = None owner.is_mutable = False owner.mutable = False owner.artifact_path = os.path.abspath(urlpath) owner.artifact_offsets = artifact_offsets + owner.linked_artifacts = manifest.get("linked", {}) owner.cache_policy = cache_policy owner.max_cache_bytes = limit owner.cache_coordinator = CacheCoordinator(None) diff --git a/tests/test_remote_store.py b/tests/test_remote_store.py index 1bb792269..858bf01a4 100644 --- a/tests/test_remote_store.py +++ b/tests/test_remote_store.py @@ -511,6 +511,45 @@ def test_nested_remote_store_ctable_index_uses_outer_cache(tmp_path): assert remote_table.traffic is outer.traffic +def test_nested_remote_store_reference_artifact_cold_and_warm(tmp_path): + data = np.arange(200, dtype="int32") + target = tmp_path / "artifact-target.b2z" + with blosc2.TreeStore(target, mode="w") as tree: + tree["/group/data"] = blosc2.asarray(data, chunks=(50,), blocks=(10,)) + target_url = f"memory://{tmp_path.name}-artifact-target.b2z" + fs = fsspec.filesystem("memory") + fs.pipe(target_url, target.read_bytes()) + + host = tmp_path / "artifact-host.b2z" + with blosc2.RemoteStore(target_url, dataset="group") as linked: + with blosc2.TreeStore(host, mode="w") as tree: + tree["/linked"] = linked + host_url = f"memory://{tmp_path.name}-artifact-host.b2z" + fs.pipe(host_url, host.read_bytes()) + + cold = tmp_path / "nested-cold.b2z" + warm = tmp_path / "nested-warm.b2z" + with blosc2.RemoteStore(host_url) as outer: + outer.traffic.reset() + outer.save(cold, include_cache=False) + assert outer.traffic.requests == 0 + with outer["linked/data"] as array: + np.testing.assert_array_equal(array[:20], data[:20]) + outer.traffic.reset() + outer.save(warm, include_cache=True) + assert outer.traffic.requests == 0 + + with blosc2.open(cold) as reopened: + np.testing.assert_array_equal(reopened["linked/data"][:20], data[:20]) + + fs.rm(target_url) + with blosc2.open(warm) as reopened: + with reopened["linked/data"] as array: + reopened.traffic.reset() + np.testing.assert_array_equal(array[:20], data[:20]) + assert reopened.traffic.requests == 0 + + def test_sparse_store_shared_handles(hierarchy, tmp_path): url, data = hierarchy parent = tmp_path / "shared" From 2aed755c36778d6987f5267874b77cd2210f8cf1 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Mon, 21 Sep 2026 10:12:18 +0200 Subject: [PATCH 43/82] Materialize nested RemoteStore trees --- src/blosc2/b2z_source.py | 9 +- src/blosc2/ctable_storage.py | 3 + src/blosc2/remote_store.py | 35 +++++-- src/blosc2/store_materialize.py | 160 ++++++++++++++++++++++++++++++++ src/blosc2/tree_store.py | 6 ++ tests/test_remote_store.py | 120 +++++++++++++++++++++++- 6 files changed, 323 insertions(+), 10 deletions(-) create mode 100644 src/blosc2/store_materialize.py diff --git a/src/blosc2/b2z_source.py b/src/blosc2/b2z_source.py index 7cad68f3b..db5775769 100644 --- a/src/blosc2/b2z_source.py +++ b/src/blosc2/b2z_source.py @@ -138,6 +138,7 @@ def __init__(self, urlpath, *, storage_options=None, _filesystem=None, _traffic= ): raise ValueError("Invalid cached B2Z metadata range") self.capture_metadata = True + self._captured_ranges = [] self._opening_ranges = [] self._batch_ranges = [] # ponytail: small directories fit in 8 KiB; larger ones use exact reads. @@ -147,7 +148,9 @@ def __init__(self, urlpath, *, storage_options=None, _filesystem=None, _traffic= else: tail = bootstrap[1] self.traffic.charge(len(tail)) - self.metadata["ranges"].append((tail_start, tail)) + self._captured_ranges.append((tail_start, tail)) + if self.persist_metadata: + self.metadata["ranges"].append((tail_start, tail)) self._opening_ranges.append((tail_start, tail)) self.file = _ArchiveFile(self, size) self.archive = zipfile.ZipFile(self.file) @@ -207,7 +210,9 @@ def _read_archive(self, offset, size): return data[offset - start : offset - start + size] data = self.read_transport(offset, size) if self.capture_metadata: - self.metadata["ranges"].append((offset, data)) + self._captured_ranges.append((offset, data)) + if self.persist_metadata: + self.metadata["ranges"].append((offset, data)) return data def read_transport(self, offset, size): diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index 9ed1ace19..d03280709 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -878,6 +878,9 @@ def load_index_catalog(self) -> dict: if os.path.isabs(path) or ".." in parts or not path.endswith(".b2nd"): raise ValueError(f"Unsafe remote index path for column {name!r}") logical = path[:-5].strip("/") + prefix = self._root_key + "/" if self._root_key else "" + if prefix and logical.startswith(prefix): + logical = logical[len(prefix) :] if not self._has_array(logical): raise ValueError(f"Missing remote index sidecar for column {name!r}") remote_path = f"remote-index://{id(self)}/{path}" diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index ab3500787..6b04273b7 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -928,6 +928,26 @@ def _close_resources(self): close(self.filesystem.loop, session) self.filesystem = None + def _export_metadata(self): + if self.format == "hdf5": + self._validate_hdf5_index() + return self.hdf5_index + if self.format != "b2z": + return self.metadata + metadata = dict(self.archive.metadata) + ranges = list(metadata.get("ranges", ())) + for offset, data in ( + *self.archive._captured_ranges, + *self.archive._opening_ranges, + *self.archive._batch_ranges, + ): + if not any( + start <= offset and offset + len(data) <= start + len(saved) for start, saved in ranges + ): + ranges.append((offset, data)) + metadata["ranges"] = ranges + return metadata + def save_selection( self, full_path, @@ -953,13 +973,7 @@ def save_selection( f"Retained cache ({retained} bytes) exceeds max_cache_bytes ({self.max_cache_bytes})" ) - if self.format == "hdf5": - self._validate_hdf5_index() - metadata = self.hdf5_index - elif self.format == "b2z": - metadata = self.archive.metadata - else: - metadata = self.metadata + metadata = self._export_metadata() src_desc, nodes, attrs, listed, candidates = self._collect_export_nodes(full_path, include_cache) @@ -1830,6 +1844,13 @@ def save( full, destination, include_cache=include_cache, mutable=mutable, overwrite=overwrite ) + def materialize(self, destination, *, overwrite=False): + """Write this hierarchy and reachable store references as one local TreeStore.""" + from blosc2.store_materialize import materialize_store + + self._ensure_open() + return materialize_store(self, destination, overwrite=overwrite) + @classmethod def _load_artifact_manifest(cls, urlpath): if os.path.isdir(urlpath): diff --git a/src/blosc2/store_materialize.py b/src/blosc2/store_materialize.py new file mode 100644 index 000000000..ebbee38df --- /dev/null +++ b/src/blosc2/store_materialize.py @@ -0,0 +1,160 @@ +"""Recursive materialization for local and remote hierarchy references.""" + +from __future__ import annotations + +import contextlib +import os +import tempfile +import uuid + +import blosc2 + + +def materialize_store(source, destination, *, overwrite=False): + """Materialize *source* and reachable RemoteStore links into one TreeStore.""" + destination = os.path.abspath(os.fspath(destination)) + if not destination.endswith((".b2z", ".b2d")): + raise ValueError("materialize destination must end in .b2z or .b2d") + if os.path.exists(destination) and not overwrite: + raise FileExistsError(f"'{destination}' already exists. Use overwrite=True to overwrite.") + parent = os.path.dirname(destination) + if not os.path.isdir(parent): + raise FileNotFoundError(f"Destination directory '{parent}' does not exist") + + with tempfile.TemporaryDirectory(prefix="materialize-", dir=parent) as staging: + working = os.path.join(staging, "tree.b2d") + with blosc2.TreeStore(working, mode="w", threshold=0) as target: + if isinstance(source, blosc2.RemoteStore): + _copy_remote_group(source, target, "/", set(), 0, staging) + else: + _copy_local_group(source, target, "/", set(), 0, staging) + if destination.endswith(".b2d"): + staged = working + else: + staged = os.path.join(staging, "tree.b2z") + with blosc2.TreeStore(working, mode="r") as packed: + packed.to_b2z(filename=staged) + _publish(staged, destination, overwrite, staging) + return destination + + +def _publish(staged, destination, overwrite, staging): + previous = None + if os.path.exists(destination): + if not overwrite: + raise FileExistsError(destination) + previous = os.path.join(staging, "previous") + os.replace(destination, previous) + try: + os.replace(staged, destination) + except BaseException: + if previous is not None: + os.replace(previous, destination) + raise + + +def _join(base, name): + return "/" + name if base == "/" else base.rstrip("/") + "/" + name + + +def _copy_attrs(attrs, target, path): + if attrs is None: + return + destination = target.attrs if path == "/" else target.get_subtree(path).attrs + for key, value in attrs.items(): + destination[key] = value + + +def _identity(store): + source = store.source + return source["kind"], source["urlpath"], source.get("dataset", "") + + +def _copy_remote_group(store, target, path, active, depth, staging): + if depth > 64: + raise ValueError("RemoteStore materialization exceeds the reference depth limit") + identity = _identity(store) + if identity in active: + raise ValueError(f"RemoteStore reference cycle at {path}: {identity[1]}::{identity[2]}") + active.add(identity) + try: + _copy_attrs(store.attrs, target, path) + for name in store: + child_path = _join(path, name) + info = store.get_info(name) + if info.kind in {"group", "remote_store"}: + with store[name] as child: + _copy_remote_group(child, target, child_path, active, depth + 1, staging) + elif info.kind == "ctable": + with store[name] as table: + _copy_table(table, target, child_path) + elif info.kind == "ndarray": + with store[name] as array: + _copy_array(array, target, child_path, staging) + else: + raise NotImplementedError( + f"Cannot materialize {child_path!r}: {info.diagnostic or info.kind}" + ) + finally: + active.remove(identity) + + +def _copy_local_group(store, target, path, active, depth, staging): + _copy_attrs(store.attrs, target, path) + for child in store.get_children("/"): + name = child.rsplit("/", 1)[-1] + child_path = _join(path, name) + full = store._translate_key_to_full(child) + info = store._object_info(full) + if isinstance(info, dict) and info.get("kind") == "remote_store": + with store[child] as linked: + _copy_remote_group(linked, target, child_path, active, depth + 1, staging) + elif isinstance(info, dict) and info.get("kind") == "ctable": + with contextlib.closing(store[child]) as table: + _copy_table(table, target, child_path) + elif store.get_descendants(child): + _copy_local_group(store.get_subtree(child), target, child_path, active, depth, staging) + else: + value = store[child] + if isinstance(value, blosc2.RemoteArray): + with value: + _copy_array(value, target, child_path, staging) + else: + target[child_path] = value + + +def _copy_array(source, target, path, staging): + local_path = os.path.join(staging, f"array-{uuid.uuid4().hex}.b2nd") + local = blosc2.empty( + source.shape, + source.dtype, + chunks=source.chunks, + blocks=source.blocks, + cparams=source.cparams, + urlpath=local_path, + mode="w", + ) + if not source.shape: + local[()] = source[()] + else: + step = source.chunks[0] + tail = (slice(None),) * (len(source.shape) - 1) + for start in range(0, source.shape[0], step): + item = (slice(start, min(start + step, source.shape[0])), *tail) + local[item] = source[item] + target[path] = local + + +def _copy_table(table, target, path): + indexes = {name: descriptor["kind"] for name, descriptor in table._get_index_catalog().items()} + local = table.copy(compact=True) + local._source_bound = False + local._source_columns = set() + target[path] = local + local.close() + materialized = target[path] + for name, kind in indexes.items(): + materialized.create_index(name, kind=kind) + if indexes and not materialized._get_index_catalog(): + raise RuntimeError(f"Failed to rebuild CTable indexes at {path!r}") + materialized.close() diff --git a/src/blosc2/tree_store.py b/src/blosc2/tree_store.py index ca1c988a8..1f6a29ff1 100644 --- a/src/blosc2/tree_store.py +++ b/src/blosc2/tree_store.py @@ -190,6 +190,12 @@ def __init__(self, *args, _from_parent_store=None, **kwargs): self._known_object_roots_cache: set[str] | None = None self._effective_object_roots_cache: tuple[str, set[str]] | None = None + def materialize(self, destination, *, overwrite=False): + """Write this tree and all RemoteStore references as one local TreeStore.""" + from blosc2.store_materialize import materialize_store + + return materialize_store(self, destination, overwrite=overwrite) + # ------------------------------------------------------------------ # Object registry helpers # ------------------------------------------------------------------ diff --git a/tests/test_remote_store.py b/tests/test_remote_store.py index 858bf01a4..72441928b 100644 --- a/tests/test_remote_store.py +++ b/tests/test_remote_store.py @@ -491,9 +491,11 @@ def resolve(descriptor): def test_nested_remote_store_ctable_index_uses_outer_cache(tmp_path): target = tmp_path / "nested-table-target.b2z" table = blosc2.CTable(NestedIndexedRow, [(i,) for i in range(200)], create_summary_index=False) - table.create_index("value", kind="summary") with blosc2.TreeStore(target, mode="w") as tree: tree["/measurements"] = table + inline = tree["/measurements"] + inline.create_index("value", kind="summary") + inline.close() target_url = f"memory://{tmp_path.name}-nested-table-target.b2z" fsspec.filesystem("memory").pipe(target_url, target.read_bytes()) @@ -509,6 +511,12 @@ def test_nested_remote_store_ctable_index_uses_outer_cache(tmp_path): np.testing.assert_array_equal(remote_table.where("value >= 197").value[:], [197, 198, 199]) assert remote_table.cache_policy is outer.cache_policy assert remote_table.traffic is outer.traffic + materialized = tmp_path / "nested-table-materialized.b2z" + outer.materialize(materialized) + with blosc2.open(materialized) as local: + table = local["/remote/measurements"] + assert table._get_index_catalog()["value"]["kind"] == "summary" + np.testing.assert_array_equal(table.where("value >= 197").value[:], [197, 198, 199]) def test_nested_remote_store_reference_artifact_cold_and_warm(tmp_path): @@ -550,6 +558,116 @@ def test_nested_remote_store_reference_artifact_cold_and_warm(tmp_path): assert reopened.traffic.requests == 0 +def test_nested_remote_store_materialize_mixed_source(hierarchy, tmp_path): + url, data = hierarchy + host = tmp_path / "materialize-host.b2z" + with blosc2.RemoteStore(url, dataset="group") as linked: + with blosc2.TreeStore(host, mode="w") as tree: + tree["/remote"] = linked + tree["/repeat"] = linked + tree["/local"] = np.arange(4) + host_url = f"memory://{tmp_path.name}-materialize-host.b2z" + fs = fsspec.filesystem("memory") + fs.pipe(host_url, host.read_bytes()) + + destination = tmp_path / "materialized.b2z" + with blosc2.RemoteStore(host_url) as outer: + outer.materialize(destination) + fs.rm(url, recursive=True) + fs.rm(host_url) + + with blosc2.open(destination) as local: + assert all(info.get("kind") != "remote_store" for info in local._objects_registry().values()) + assert local.get_subtree("/remote").attrs["title"] == "child" + np.testing.assert_array_equal(local["/remote/a"][:2, :3], data[:2, :3]) + np.testing.assert_array_equal(local["/remote/b"][:2, :3], data[:2, :3] + 1) + np.testing.assert_array_equal(local["/repeat/a"][:2, :3], data[:2, :3]) + np.testing.assert_array_equal(local["/local"][:], np.arange(4)) + + +def test_local_tree_materialize_remote_reference_to_b2d(tmp_path): + remote = _nested_materialize_source(tmp_path) + catalog = tmp_path / "local-catalog.b2z" + with remote: + with blosc2.TreeStore(catalog, mode="w") as tree: + tree["/mount"] = remote + destination = tmp_path / "local-materialized.b2d" + with blosc2.TreeStore(catalog, mode="r") as tree: + tree.materialize(destination) + with blosc2.open(destination) as local: + np.testing.assert_array_equal(local["/mount/data"][:], np.arange(20)) + + +def test_nested_remote_store_materialize_multiple_levels(tmp_path): + source = _nested_materialize_source(tmp_path) + middle_path = tmp_path / "middle.b2z" + with source: + with blosc2.TreeStore(middle_path, mode="w") as middle: + middle["/inner"] = source + middle_url = f"memory://{tmp_path.name}-middle.b2z" + fsspec.filesystem("memory").pipe(middle_url, middle_path.read_bytes()) + + outer_path = tmp_path / "outer.b2z" + with blosc2.RemoteStore(middle_url) as middle: + with blosc2.TreeStore(outer_path, mode="w") as outer: + outer["/middle"] = middle + outer_url = f"memory://{tmp_path.name}-outer.b2z" + fsspec.filesystem("memory").pipe(outer_url, outer_path.read_bytes()) + + destination = tmp_path / "multiple-levels.b2z" + with blosc2.RemoteStore(outer_url) as outer: + outer.materialize(destination) + with blosc2.open(destination) as local: + np.testing.assert_array_equal(local["/middle/inner/data"][:], np.arange(20)) + + +def _nested_materialize_source(tmp_path): + target = tmp_path / "materialize-source.b2z" + with blosc2.TreeStore(target, mode="w") as tree: + tree["/data"] = np.arange(20) + url = f"memory://{tmp_path.name}-materialize-source.b2z" + fsspec.filesystem("memory").pipe(url, target.read_bytes()) + return blosc2.RemoteStore(url) + + +def test_nested_remote_store_materialize_cycle_preserves_destination(tmp_path): + fs = fsspec.filesystem("memory") + url_a = f"memory://{tmp_path.name}-cycle-a.b2z" + url_b = f"memory://{tmp_path.name}-cycle-b.b2z" + + def descriptor(url): + return { + "kind": "b2z", + "version": 1, + "urlpath": url, + "dataset": "", + "assume_immutable": True, + } + + for name, mount, linked_url, remote_url in ( + ("a", "/b", url_b, url_a), + ("b", "/a", url_a, url_b), + ): + path = tmp_path / f"cycle-{name}.b2z" + with blosc2.TreeStore(path, mode="w") as tree: + tree._register_object( + mount, + kind="remote_store", + version=1, + layout="reference", + source=descriptor(linked_url), + strict=True, + ) + fs.pipe(remote_url, path.read_bytes()) + + destination = tmp_path / "existing.b2z" + destination.write_bytes(b"existing") + with blosc2.RemoteStore(url_a) as root: + with pytest.raises(ValueError, match="cycle"): + root.materialize(destination, overwrite=True) + assert destination.read_bytes() == b"existing" + + def test_sparse_store_shared_handles(hierarchy, tmp_path): url, data = hierarchy parent = tmp_path / "shared" From 31ea5af760c02c5a8773d8f726cac38b1856e26d Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Mon, 21 Sep 2026 10:15:08 +0200 Subject: [PATCH 44/82] Document nested RemoteStore references --- doc/guides/remote_objects.md | 51 +++++++++++++++++++++++++++++---- doc/reference/remotestore.rst | 30 ++++++++++++++++++- examples/remote/nested-store.py | 29 +++++++++++++++++++ 3 files changed, 104 insertions(+), 6 deletions(-) create mode 100644 examples/remote/nested-store.py diff --git a/doc/guides/remote_objects.md b/doc/guides/remote_objects.md index a1540f095..85f228633 100644 --- a/doc/guides/remote_objects.md +++ b/doc/guides/remote_objects.md @@ -47,7 +47,7 @@ with blosc2.RemoteStore("https://datasets.example.org/data.h5") as store: # 2. Inspect a node's kind and attributes info = store.get_info("experiment") - print(info.kind) # "group", "ndarray", "ctable", or "unsupported" + print(info.kind) # "group", "ndarray", "ctable", "remote_store", or "unsupported" # 3. Read user metadata on groups or arrays print(store["experiment"].attrs[:]) @@ -63,11 +63,40 @@ with blosc2.RemoteStore("https://datasets.example.org/data.h5") as store: - **Relative paths**: Lookups can use slash paths or chained indexing interchangeably (`store["experiment/temperature"]` is equivalent to `store["experiment"]["temperature"]`). Leaves return a {ref}`RemoteArray` or {ref}`RemoteCTable` according to their kind. - **Node inspection with `RemoteNode`**: Call `store.get_info(name)` to inspect a node without creating leaf readers or allocating cache memory. A `RemoteNode` provides: - `path`: relative dataset path. - - `kind`: `"group"`, `"ndarray"`, `"ctable"`, or `"unsupported"`. + - `kind`: `"group"`, `"ndarray"`, `"ctable"`, `"remote_store"`, or `"unsupported"`. - `attrs`: user metadata mapping (or `None` if array attributes require opening the leaf). - `diagnostic`: explanation for unsupported nodes (e.g. non-array objects or unsupported codecs). - **Graceful degradation**: Unsupported nodes remain visible during discovery and raise an informative `NotImplementedError` only when selected as arrays, allowing you to browse mixed containers without errors. +### Mount one remote hierarchy inside another + +A `TreeStore` can persist a `RemoteStore` reference at an explicit path. The path +is always chosen by the application; it is not derived from the remote filename. +The reference may select a complete B2Z, HDF5, or Zarr hierarchy, or a subgroup +selected with `dataset=`: + +```python +with blosc2.RemoteStore("s3://weather/europe.zarr", dataset="spain") as weather: + with blosc2.TreeStore("catalog.b2z", mode="w") as catalog: + catalog["/external/weather"] = weather + +with blosc2.RemoteStore("catalog.b2z") as catalog: + print(catalog.get_info("external/weather").kind) # "remote_store" + values = catalog["external/weather/temperature"][:100] +``` + +Opening the catalog discovers the mount descriptor without opening its target. +The target opens on the first lookup below the mount. Direct slash lookup and +chained lookup have the same result, and nested mounts can contain further mounts. + +The outer `RemoteStore` owns the cache policy, aggregate byte allowance, traffic +counter, and eviction across local and mounted leaves. Defaults saved in the +reference apply only when opening it directly from a local `TreeStore`. For +authenticated mounts, pass `nested_storage_options` to the outer store as either +a mapping keyed by URL or a callable receiving each credential-free descriptor. +Credentials are never persisted. A local catalog can instead use +`tree.open_remote(path, storage_options=...)` for one mount. + ### Persistent disk caching with `cache_dir` Specify `cache_dir` when creating a `RemoteStore` to persist discovery metadata and downloaded chunks to local disk: @@ -258,9 +287,10 @@ For a `RemoteStore`, `store.traffic` reports cumulative traffic across discovery the selected cache. It does not fetch missing payload. `include_cache=False` writes a cold reference containing only the source and bootstrap metadata. -Arrays and tables also provide `materialize()`, which reads everything required -for an independent local object. Stores have no recursive materialization API; -navigate to an array or table leaf first. +Arrays and tables provide `materialize()`, which reads everything required for +an independent local object. `RemoteStore.materialize()` and +`TreeStore.materialize()` recursively expand mounted stores into one local `.b2z` +or `.b2d` tree. ```python # Table example: .b2z references and local materialization @@ -272,6 +302,9 @@ table.to_b2d("local.b2d") # Array example: .b2nd reference and selected materialization array.save("reference.b2nd") subset = array.materialize(item=slice(0, 100)) + +# Store example: expand local and mounted leaves into one independent hierarchy +store.materialize("complete-tree.b2z") ``` An array reports its own retained payload. Stores and tables report their shared @@ -302,6 +335,14 @@ with blosc2.RemoteStore("https://datasets.example.org/data.h5") as store: - **`include_cache=False`**: Omits cached payload chunks, producing a minimal reference archive for remote streaming. - **Subtree export**: Calling `save()` on a group view exports that subtree with relative child keys and the appropriate source root. +Mounted stores remain references in a saved snapshot. Warm data already retained +by an opened mount is included when `include_cache=True`; saving does not open an +unvisited mount or fetch missing payload. Materialization follows every reachable +mount, copies arrays in chunk-sized slabs, and rebuilds persisted CTable indexes. +Repeated targets are copied at each explicit mount. A reference cycle or a chain +deeper than 64 mounts raises `ValueError`, and a failed materialization leaves an +existing destination unchanged. + ### Reopen reference files with `blosc2.open()` Array `.b2nd` carriers and store or table `.b2z` references can be reopened directly with `blosc2.open()`: diff --git a/doc/reference/remotestore.rst b/doc/reference/remotestore.rst index d24b0b8ee..975f30462 100644 --- a/doc/reference/remotestore.rst +++ b/doc/reference/remotestore.rst @@ -42,7 +42,8 @@ An array root must be opened with ``RemoteArray`` instead. ``keys()`` and ``get_info()`` do not construct leaf readers or payload caches. Discovery can read archive prefixes, attributes and small HDF5 inline values. ``get_info()`` returns a ``RemoteNode`` with a relative path, a kind (``group``, -``ndarray``, ``ctable`` or ``unsupported``), known attributes and a diagnostic. +``ndarray``, ``ctable``, ``remote_store`` or ``unsupported``), known attributes +and a diagnostic. Unknown array attributes are ``None``; open the array to retrieve them. Unsupported nodes stay discoverable and raise ``NotImplementedError`` when selected. Missing paths raise ``KeyError``. @@ -65,6 +66,33 @@ the owned archive/store wrappers and private HTTP/S3 transport sessions. Operati Standalone ``RemoteArray`` exports remain self-contained references, including the native HDF5 index when applicable. +Nested stores +------------- + +Assigning a ``RemoteStore`` to a ``TreeStore`` persists a lazy, credential-free +reference at the exact path supplied by the caller:: + + with blosc2.RemoteStore("s3://weather/europe.zarr", dataset="spain") as remote: + with blosc2.TreeStore("catalog.b2z", mode="w") as tree: + tree["/external/weather"] = remote + +Remote B2Z discovery reports that object root as ``remote_store`` without opening +the linked source. Lookup through the mount supports direct and chained paths. +B2Z, HDF5, Zarr v2 and Zarr v3 targets and subgroup references are supported. + +The outer ``RemoteStore`` overrides saved cache defaults and owns one policy, +traffic counter and aggregate allowance across mounted sources. Pass +``nested_storage_options`` as a URL-to-options mapping or a callable accepting a +source descriptor when mounted sources need different credentials. A local +``TreeStore`` can open one reference with runtime options through +``open_remote()``. + +``save()`` preserves nested references and includes only already-retained warm +payload from mounts that were opened. It never opens an unvisited mount to save +it. ``materialize(destination)`` follows all reachable mounts and writes one +independent local ``.b2z`` or ``.b2d`` TreeStore. Materialization detects cycles, +limits nesting to 64 levels, and publishes the destination atomically. + ``b2view`` uses ``RemoteStore`` for remote hierarchies with one 64 MiB MEMORY allowance, and ``RemoteArray`` for selected or directly opened leaves. Switching selection releases the UI handle while retaining the store's warm chunks. diff --git a/examples/remote/nested-store.py b/examples/remote/nested-store.py new file mode 100644 index 000000000..62e8c650a --- /dev/null +++ b/examples/remote/nested-store.py @@ -0,0 +1,29 @@ +"""Persist a named remote-store mount, read through it, and materialize it locally. + +Usage: python examples/remote/nested-store.py URL DATASET LEAF +""" + +import argparse + +import blosc2 + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("url", help="Remote B2Z, HDF5, or Zarr URL") + parser.add_argument("dataset", help="Remote group to mount") + parser.add_argument("leaf", help="Array below the mounted group") + args = parser.parse_args() + + with blosc2.RemoteStore(args.url, dataset=args.dataset) as remote: + with blosc2.TreeStore("catalog.b2z", mode="w") as tree: + tree["/external/weather"] = remote + + with blosc2.RemoteStore("catalog.b2z") as catalog: + with catalog[f"external/weather/{args.leaf}"] as array: + print(array[:10]) + catalog.materialize("weather-local.b2z") + + +if __name__ == "__main__": + main() From 93967091008bfb15b4b289a52853dcc0c51a2dc2 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Mon, 21 Sep 2026 10:17:03 +0200 Subject: [PATCH 45/82] Record nested RemoteStore implementation --- plans/remote-nested-store.md | 266 +++++++++++++++++++++++++++++++++++ 1 file changed, 266 insertions(+) create mode 100644 plans/remote-nested-store.md diff --git a/plans/remote-nested-store.md b/plans/remote-nested-store.md new file mode 100644 index 000000000..bc8a6c1c9 --- /dev/null +++ b/plans/remote-nested-store.md @@ -0,0 +1,266 @@ +# RemoteStore references inside TreeStore + +Status: implemented and verified, 2026-09-21. + +## Goal and agreed semantics + +Allow a TreeStore to contain references to independent RemoteStore sources: +B2Z, HDF5 and Zarr, including a selected subgroup. The hosting TreeStore may +later be opened locally or hosted remotely and opened through RemoteStore. +Reuse the existing format readers; this feature connects stores together. + +Names are always explicit: + +```python +with blosc2.RemoteStore("https://example.org/weather.zarr") as weather: + with blosc2.TreeStore("catalog.b2z", mode="w") as tree: + tree["/external/weather"] = weather +``` + +The assignment persists a reference, not the remote contents or a borrowed live +Python object. Closing the original handle must not invalidate the reference. +There is no automatic naming or extension stripping. The mount path remains +`/external/weather` in reference exports and materialized outputs. + +Opening and listing the local tree must not contact referenced sources. Access +to the reference returns a RemoteStore; discovery is deferred until an operation +needs its metadata or payload. The linked store remains read-only even when the +hosting TreeStore is writable. Deleting the local reference never deletes remote +data. Replacing an object root follows TreeStore's explicit-delete convention. + +## Reference representation and integration + +Use TreeStore's existing object registry and object-root boundary handling for a +versioned `remote_store` object. Persist the descriptor in local metadata, readable +without constructing a child reader. Choose one authoritative descriptor location +and avoid duplicating it across registry and carrier metadata. Registry writes +for references must propagate failures rather than use best-effort registration. + +The descriptor contains the source kind, credential-free URL, selected dataset, +format version and immutable-source assumption. Persist portable cache-policy +preferences for standalone local reopening, but no credentials, filesystem +objects, callbacks or machine-specific cache directories. Cache payload, when +included, uses the existing remote artifact machinery with source-scoped keys. + +Validate descriptors and logical paths before publishing or opening them. Reject +unsupported versions, invalid source kinds, traversal and credentials embedded +in URLs according to existing remote-source rules. Preserve ordinary groups, +CTables and RemoteArray leaves. Links target groups; array and table roots retain +their existing object types. + +Integration points: + +- `tree_store.py`: assignment, registry persistence, object-root lookup, + containment, listing, subtree views, deletion and handle cleanup. +- `remote_store.py`: discovery, lazy linked handles, path routing, ownership, + manifest validation, refresh and portable export. Its current node kinds and + single-source owner require explicit extension. +- `remote_store_cache.py`: existing aggregate cache coordination, sparse-cache + identity and process-safe operations. +- `schunk.py`: artifact dispatch only where the new manifest requires it. +- Existing B2Z, HDF5 and Zarr readers remain responsible for format access. + +The hosting B2Z discovery must recognize the reference boundary from metadata +and hide its storage internals, as it does for CTable roots. Do not mistake a +reference carrier for an ordinary group or array. Unknown reference versions +must report a clear unsupported-node diagnostic. + +## Opening locally and traversing remotely + +Local use: + +```python +with blosc2.TreeStore("catalog.b2z", mode="r") as tree: + with tree["/external/weather"] as weather: + with weather["temperature"] as temperature: + values = temperature[:100] +``` + +Root and immediate-child enumeration expose the mount without expanding it. +RemoteStore discovery reports `kind="remote_store"` for the reference, while +ordinary subgroups continue to report `group`. Looking up the mount returns a +RemoteStore. Opening a child or asking the linked store for its keys may contact +that source. Failures at one target must leave unrelated nodes usable. + +Support both chained lookup and paths crossing a mount, including mounts under +subgroup views. Resolve the remaining suffix relative to the linked dataset, +never relative to the source container root. Keep local structural walking from +implicitly following links; recursive materialization explicitly follows them. +Update consumers of node kinds, including browser presentation, so references +are browsable without eager expansion. + +The current RemoteStore constructor performs discovery immediately. Introduce +only the deferred initialization needed for persisted references; do not claim +network-free reference lookup while constructing a normal eager RemoteStore. + +## Cache policy, credentials and lifetime + +When opened through a local TreeStore, a linked RemoteStore uses its persisted +standalone policy, defaulting to MEMORY. Provide a runtime override and runtime +credentials through an explicit reference-opening path. Proposed API: +`tree.open_remote(path, *, storage_options=None, cache_policy=None, +max_cache_bytes=None, cache_dir=None)`. Ordinary indexing uses the saved defaults. +A saved DISK preference requires a runtime cache location or the existing safe +temporary-directory convention; never reuse a persisted absolute directory. + +When reached through an outer RemoteStore, the outer owner determines cache +policy and the aggregate byte budget for all linked sources and descendants. +Inner saved policies do not override it. NONE, MEMORY, DISK and shared sparse +caching must work across mounts, including CTables and their index sidecars. + +Keep discovery and refresh state per source while reusing aggregate cache +coordination and traffic accounting. Namescope cache entries by source identity, +dataset and generation, not just relative leaf path. Equal leaf names in +different stores must never collide. Retained compressed data shares the outer +budget; metadata, decoded working buffers and output allocations remain separately +accounted, following existing conventions. + +Credentials are runtime-only and source-specific. Add a minimal resolver keyed +by the validated source descriptor for nested access, and thread it through +reference reopening. Do not forward parent credentials or filesystem instances +to an unrelated endpoint. A resolver is not serialized. Reuse existing source +authorization hooks before contacting linked targets. + +Use existing dependent-handle ownership rules: closing a parent handle does not +close resources still held by children. Release linked owners when their last +dependent closes. An outer refresh invalidates previously issued linked handles +and cached discovery consistently, without eagerly opening undiscovered targets. +Route refresh of borrowed linked handles through the outer root, matching tables. +Standalone links opened from a local tree can refresh their own remote session. + +## Reference saving + +`RemoteStore.save()` retains its reference-export meaning. Preserve the root +source plus every known linked-source descriptor and its mount path. A cold export +must reopen links later without losing dataset selection or source identity. +`include_cache=True` includes only already retained payload; it must not expand +unvisited links or download missing chunks. Include known linked discovery needed +to use warm caches offline. `include_cache=False` omits payload at every depth. + +TreeStore assignment stores a cold reference by default; it does not implicitly +copy the assigned handle's potentially large cache. Local TreeStore packing and +format conversion preserve references. They must not silently acquire recursive +materialization semantics. + +Version manifests when required by the additional source graph. Keep old artifacts +readable; reject unsupported new manifests clearly. Saved subtree exports include +only reachable mounts and their retained caches, without unrelated siblings. + +## Explicit materialization + +Add `materialize(destination, *, overwrite=False)` to RemoteStore and TreeStore +as an explicit operation. A `.b2z` destination produces a local archive; a `.b2d` +destination produces a local directory tree. Use existing export machinery for +ordinary leaves and CTables. This proposed API avoids changing reference saving +or TreeStore's existing format-conversion behavior. + +Materialization recursively downloads the hosting tree and every reachable linked +store. Replace each reference with an ordinary structural subtree at the same +explicit path, preserving group attributes, supported arrays, tables and their +schema. External RemoteArray references must also become local data. Convert HDF5 +and Zarr contents to supported Blosc2 objects; do not copy their native containers +as opaque files. The result opens as a self-contained TreeStore with networking +disabled. CTable indexes must be preserved through valid local export or rebuilt +consistently; never carry stale source-bound descriptors into the output. + +Detect cycles using source kind, canonical source identity and dataset along the +active expansion chain. Repeated references in separate branches are valid and +materialize separately. Catch cycles through subgroup selection as traversal +reaches them; do not globally reject every repeated source URL. Apply a finite +depth guard as protection against aliases that cannot be canonicalized reliably. +Errors identify the mount chain. Unsupported nodes fail explicitly rather than +silently produce incomplete output. + +Stream leaves with existing buffer limits instead of collecting the complete +tree in memory. Stage output and publish only after success. Errors, missing +credentials, cycles and interrupted downloads must leave existing destinations +intact and clean up owned staging resources. + +## Implementation sequence and verification + +1. **Persist explicit references in TreeStore.** Implement descriptor validation, + object boundaries, assignment, deletion and local reopening for `.b2d` and + `.b2z`. Test explicit names, subgroup targets, replacement rules, malformed + descriptors and original-handle closure. Assert local open/list/lookup of a + deferred handle performs no linked-source requests. +2. **Discover and traverse remote mounts.** Add the node kind, deferred discovery + and path routing. Cover B2Z, HDF5 and Zarr targets, nested mounts, subgroup + views, empty groups, direct/chained lookup, missing targets and unaffected + siblings. Check browser and metadata APIs without eager target expansion. +3. **Integrate ownership and shared caching.** Add runtime option resolution, + per-source discovery and outer aggregate cache ownership. Verify all cache + policies, eviction across sources, identity collisions, sparse reuse, distinct + credentials, CTable/index leaves, close ordering and refresh invalidation. + Test local saved defaults separately from outer-owner overrides. +4. **Extend portable reference exports.** Round-trip cold and warm nested + references, subtree exports and policy overrides. Assert saving performs no + missing-payload fetches and unvisited links remain unopened. Test offline warm + reads, runtime credential injection, artifact compatibility and malformed + multi-source manifests. +5. **Implement recursive materialization.** Cover mixed-format links, multiple + levels, subgroup mounts, tables, RemoteArray leaves, attributes, repeated + targets, cycles and unsupported nodes. Reopen results with networking forbidden + and compare hierarchy and values. Inject failures to verify destination safety + and bounded streaming. +6. **Document and verify the complete feature.** Explain explicit naming, local + versus remote cache ownership, credentials, lazy discovery, save versus + materialize and cycle handling. Add one small example. Run focused TreeStore, + RemoteStore, RemoteArray and CTable tests, Ruff and documentation checks, then + the full default suite in the `blosc2` conda environment. Record actual commands, + results and remaining limitations in this plan. + +Use deterministic local fixtures and instrumented transport for request assertions; +do not depend on public remote services. Reuse existing optional HDF5 and Zarr test +fixtures. Keep format-reader regressions and ordinary same-source subgroup behavior +covered while adding cross-source ownership. + +## Implementation record + +The six implementation points were completed in order: + +1. `f4dc723c` — Persist RemoteStore references in TreeStore. +2. `b55b70ef` — Traverse nested RemoteStore references. +3. `667c7a62` — Share cache ownership with nested stores. +4. `fcc8513d` — Preserve nested stores in reference exports. +5. `2aed755c` — Materialize nested RemoteStore trees. +6. `31ea5af7` — Document nested RemoteStore references. + +The implementation stores validated, credential-free descriptors in the existing +TreeStore object registry. Local lookup creates a deferred handle, while remote +B2Z discovery exposes a `remote_store` boundary and routes path suffixes through +the mounted owner. B2Z, HDF5, Zarr v2 and Zarr v3 mounts use their existing +readers. Nested owners share the outer cache policy, coordinator, traffic counter +and aggregate allowance; source-specific namespaces prevent equal leaf paths from +colliding. Runtime credentials can be supplied per source. + +Cold exports retain descriptors without opening targets. Warm exports recursively +include only metadata and payload already retained by opened targets. Recursive +materialization writes `.b2z` or `.b2d`, copies array data by first-axis chunk +slabs, rebuilds CTable indexes, accepts repeated targets in separate branches, +detects active-chain cycles, enforces a depth limit of 64 and publishes only after +the staged tree succeeds. + +Verification in the `blosc2` conda environment: + +- `pytest -q tests/test_remote_store.py tests/test_tree_store.py + tests/ctable/test_remote_ctable.py tests/ctable/test_ctable_indexing.py`: + **496 passed, 6 skipped**. +- Ruff on every changed Python file: **passed**. +- `python -m sphinx -b html doc /tmp/python-blosc2-nested-store-docs`: + **passed**; the build retained the repository's existing documentation warnings. +- `pytest -q`: **10,392 passed, 36 skipped**. + +Remaining limits are deliberate: references are read-only and assume immutable +sources; credentials remain runtime-only; cold artifacts still need their remote +sources; a saved DISK preference opened directly from a local TreeStore needs an +explicit runtime `cache_dir`; repeated targets are materialized independently; +and materialization supports only node kinds already readable as Blosc2 arrays, +CTables or groups. There is no automatic mount naming, remote transaction layer, +content deduplication or source-change monitor. + +## Out of scope + +Archives physically embedded inside other archives, automatic mount naming, +remote writes, automatic source-change detection, cross-store transactions and +content deduplication of materialized repeated targets. No new format readers, +general virtual-filesystem framework or remote query service is required. From 3de1681f6067c51269b1c04800dbd5f21ac7aaa4 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Mon, 21 Sep 2026 11:29:52 +0200 Subject: [PATCH 46/82] Fix PyTables HDF5 metadata decoding --- src/blosc2/hdf5_source.py | 16 ++++++++++++---- tests/test_hdf5_source.py | 11 +++++++++++ 2 files changed, 23 insertions(+), 4 deletions(-) diff --git a/src/blosc2/hdf5_source.py b/src/blosc2/hdf5_source.py index 30d22f4cc..73ca24391 100644 --- a/src/blosc2/hdf5_source.py +++ b/src/blosc2/hdf5_source.py @@ -67,7 +67,13 @@ def _dtype_field_from_json(field): # dtype.descr writes a titled field as (title, name); JSON and msgpack # round trips turn that tuple into a list again. name = tuple(name) - if isinstance(spec, list): + if isinstance(spec, list) and len(spec) == 2 and isinstance(spec[1], dict): + # h5py annotates fixed-width HDF5 strings as (dtype, metadata). + # NumPy includes that pair in dtype.descr, but does not accept it when + # reconstructing a structured dtype. The storage dtype is the first + # item; encoding metadata does not change the bytes on disk. + spec = spec[0] + elif isinstance(spec, list): spec = [_dtype_field_from_json(item) for item in spec] return (name, spec, tuple(shape[0])) if shape else (name, spec) @@ -112,9 +118,11 @@ def _from_json_value(value): if "__float__" in value: return float(value["__float__"]) if "__scalar__" in value: - return np.frombuffer(base64.b64decode(value["__scalar__"]), dtype=dtype_from_value(value["dtype"]))[ - 0 - ] + data = base64.b64decode(value["__scalar__"]) + dtype = dtype_from_value(value["dtype"]) + if dtype.itemsize == 0: + return np.array(data, dtype=dtype)[()] + return np.frombuffer(data, dtype=dtype)[0] if "__object_ndarray__" in value: items = [_from_json_value(item) for item in value["__object_ndarray__"]] result = np.empty(value["shape"], dtype=object) diff --git a/tests/test_hdf5_source.py b/tests/test_hdf5_source.py index 2b6fa6d95..224f3829b 100644 --- a/tests/test_hdf5_source.py +++ b/tests/test_hdf5_source.py @@ -92,6 +92,17 @@ def test_hdf5_titled_dtype_roundtrip(): assert dtype_from_value(json.loads(json.dumps(dtype_value(dtype)))) == dtype +def test_hdf5_fixed_string_metadata_and_empty_scalar_roundtrip(): + import h5py + + from blosc2.hdf5_source import _from_json_value, _json_value, dtype_from_value, dtype_value + + dtype = np.dtype([("name", h5py.string_dtype("ascii", 8)), ("value", " Date: Mon, 21 Sep 2026 11:34:01 +0200 Subject: [PATCH 47/82] Add remote PyTables table access --- src/blosc2/ctable_storage.py | 70 ++++++++++++++++++++++++++++++ src/blosc2/hdf5_source.py | 45 ++++++++++++++++++- src/blosc2/remote_store.py | 4 +- tests/ctable/test_remote_ctable.py | 34 +++++++++++++++ 4 files changed, 149 insertions(+), 4 deletions(-) diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index d03280709..6fc65b306 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -644,6 +644,53 @@ def index_anchor_path(self, col_name: str) -> str | None: return None +class _RemoteHDF5Field(blosc2.Operand): + """One field of a shared remote HDF5 compound dataset.""" + + def __init__(self, records, name): + self.records = records + self.field = name + self._dtype = np.dtype(records.dtype.fields[name][0]) + self._shape = records.shape + self.chunks = records.chunks + self.blocks = records.blocks + + dtype = property(lambda self: self._dtype) + shape = property(lambda self: self._shape) + ndim = property(lambda self: len(self._shape)) + + def __len__(self): + return self.shape[0] + + def __getitem__(self, key): + return self.records[key][self.field] + + +class _AllValidRows(blosc2.Operand): + """Virtual validity column for immutable row-complete sources.""" + + def __init__(self, size, source_chunks): + chunk = source_chunks[0] if source_chunks else max(1, min(size, 1 << 16)) + self._shape = (size,) + self.chunks = (chunk,) + self.blocks = (chunk,) + self._dtype = np.dtype(np.bool_) + + dtype = property(lambda self: self._dtype) + shape = property(lambda self: self._shape) + ndim = property(lambda self: 1) + + def __len__(self): + return self.shape[0] + + def __getitem__(self, key): + if isinstance(key, (int, np.integer)): + if not -self.shape[0] <= key < self.shape[0]: + raise IndexError("row index out of range") + return np.bool_(True) + return np.ones(self.shape[0], dtype=bool)[key] + + class RemoteTableStorage(TableStorage): """Read-only CTable storage over a shared RemoteStore owner.""" @@ -666,9 +713,16 @@ def __init__( self._arrays: list[blosc2.RemoteArray] = [] self._registered_index_paths: list[str] = [] self._closed = False + self._hdf5_records = None owner.acquire() def open_columns(self, table, names, load): + if self._owner.format == "hdf5": + with self._owner.lock: + self._check_open() + for name in names: + load(name) + return None from blosc2.ctable_remote_read import open_columns return open_columns(self, table, names, load) @@ -703,6 +757,8 @@ def _metadata(self) -> dict: def _has_array(self, logical_key: str) -> bool: self._check_open() + if self._owner.format == "hdf5": + return False member = self._full_key(logical_key) + ".b2nd" return sum(info.filename == member for info in self._owner.archive.members) == 1 @@ -711,6 +767,11 @@ def _not_supported(*args, **kwargs): raise RuntimeError("RemoteTableStorage is read-only") def open_column(self, name: str) -> blosc2.RemoteArray: + if self._owner.format == "hdf5": + if self._hdf5_records is None: + self._hdf5_records = self._owner.remote_array(self._root_key) + self._arrays.append(self._hdf5_records) + return _RemoteHDF5Field(self._hdf5_records, name) source_columns = set(self.load_schema().get("source_columns", ())) if name in source_columns: logical_key = f"{_COLS_DIR}/{_column_name_to_relpath(name)}" @@ -783,6 +844,9 @@ def open_dictionary_column(self, name: str, spec) -> DictionaryColumn: raise def open_valid_rows(self) -> blosc2.RemoteArray: + if self._owner.format == "hdf5": + metadata = self._metadata() + return _AllValidRows(metadata["shape"][0], metadata["chunks"]) return self._open_array("_valid_rows") def open_null_mask(self, name: str) -> blosc2.RemoteArray: @@ -811,6 +875,12 @@ def load_user_attrs(self) -> dict: if hasattr(self, "_user_attrs"): return dict(self._user_attrs) self._user_attrs = self._owner.load_ctable_attrs(self._root_key) + if self._owner.format == "hdf5": + self._user_attrs = { + key: value + for key, value in self._user_attrs.items() + if key not in {"CLASS", "VERSION", "NROWS"} and not key.startswith("FIELD_") + } return dict(self._user_attrs) def table_exists(self) -> bool: diff --git a/src/blosc2/hdf5_source.py b/src/blosc2/hdf5_source.py index 73ca24391..1a38b4b0a 100644 --- a/src/blosc2/hdf5_source.py +++ b/src/blosc2/hdf5_source.py @@ -27,6 +27,42 @@ HDF5_INDEX_FORMAT = "blosc2-hdf5-index" HDF5_INDEX_VERSION = 1 + + +def _pytables_table_schema(dtype, shape): + """Return a source-bound CTable schema for a supported PyTables Table.""" + dtype = np.dtype(dtype) + if len(shape) != 1 or dtype.names is None: + raise TypeError("PyTables tables require a one-dimensional compound dataset") + columns = [] + for name in dtype.names: + field = dtype.fields[name][0] + if field.fields is not None or field.subdtype is not None: + raise TypeError(f"PyTables field {name!r} must be a scalar") + if field.kind == "S": + spec = {"kind": "bytes", "max_length": field.itemsize} + elif field.kind == "b": + spec = {"kind": "bool"} + elif field.kind in "iu": + prefix = "int" if field.kind == "i" else "uint" + spec = {"kind": f"{prefix}{field.itemsize * 8}"} + elif field.kind in "fc": + prefix = "float" if field.kind == "f" else "complex" + spec = {"kind": f"{prefix}{field.itemsize * 8}"} + else: + raise TypeError(f"Unsupported PyTables field {name!r} with dtype {field}") + columns.append({"name": name, **spec}) + return { + "version": 1, + "columns": columns, + "source_bindings_version": 1, + "source_columns": list(dtype.names), + "n_rows": int(shape[0]), + "create_summary_index": False, + "summary_indexes_built": True, + } + + _DIRECT_FILTERS = {1, 2, 32026} # deflate, shuffle, Blosc2 @@ -67,7 +103,7 @@ def _dtype_field_from_json(field): # dtype.descr writes a titled field as (title, name); JSON and msgpack # round trips turn that tuple into a list again. name = tuple(name) - if isinstance(spec, list) and len(spec) == 2 and isinstance(spec[1], dict): + if isinstance(spec, (list, tuple)) and len(spec) == 2 and isinstance(spec[1], dict): # h5py annotates fixed-width HDF5 strings as (dtype, metadata). # NumPy includes that pair in dtype.descr, but does not accept it when # reconstructing a structured dtype. The storage dtype is the first @@ -214,7 +250,7 @@ def _dataset_metadata(dataset): "size": int(info.size), } ) - return { + metadata = { "shape": [int(v) for v in dataset.shape], "dtype": dtype_value(dataset.dtype), "chunks": chunks, @@ -224,6 +260,11 @@ def _dataset_metadata(dataset): "direct": direct, "allocated": allocated, } + table_class = dataset.attrs.get("CLASS") + if table_class in {"TABLE", b"TABLE", np.bytes_(b"TABLE")}: + metadata["kind"] = "ctable" + metadata["schema"] = json.dumps(_pytables_table_schema(dataset.dtype, dataset.shape)) + return metadata def scan_hdf5_index(urlpath, storage_options=None, *, unsupported=None, traffic=None, _filesystem=None): diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 6b04273b7..8974b1cc5 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -459,7 +459,7 @@ def _open_hdf5(self): key: decode_hdf5_value(value) for key, value in metadata.get("attrs", {}).items() } for path, metadata in self.hdf5_index["datasets"].items(): - self._add(path, "ndarray", metadata) + self._add(path, "ctable" if metadata.get("kind") == "ctable" else "ndarray", metadata) self.attrs[path] = { key: decode_hdf5_value(value) for key, value in metadata.get("attrs", {}).items() } @@ -578,7 +578,7 @@ def open_source(self, path): if full in self.sources: return self.sources[full] kind, value = self.nodes[full] - if kind != "ndarray": + if kind != "ndarray" and not (kind == "ctable" and self.format == "hdf5"): raise NotImplementedError( value if isinstance(value, str) else "Array access is unavailable for this node" ) diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index a09256ece..fff036bcb 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -3,6 +3,7 @@ from __future__ import annotations import dataclasses +import io import itertools import os import zipfile @@ -38,6 +39,39 @@ def remote_table_url(tmp_path, table, name="table"): return url +def pytables_hdf5_url(name="pytables-table.h5"): + h5py = pytest.importorskip("h5py") + data = np.array( + [(1, 1.5, b"one"), (2, 2.5, b"two"), (3, 3.5, b"three")], + dtype=[("id", "= 2").label[:], data["label"][1:]) + assert table.attrs["owner"] == "test" + assert table.attrs["TITLE"] == b"example" + assert table._cols["id"].records is table._cols["label"].records + assert sum(isinstance(array, blosc2.RemoteArray) for array in table._storage._arrays) == 1 + + def indexed_remote_table_url(tmp_path, kind, *, name=None, rows=1000, **kwargs): name = name or f"indexed-{kind}" path = tmp_path / f"{name}.b2z" From d5c1ffab948441226260dfa8a24d228e6c2af387 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Mon, 21 Sep 2026 11:37:15 +0200 Subject: [PATCH 48/82] Import PyTables full indexes as OPSI --- src/blosc2/ctable_storage.py | 69 ++++++++++++++++++++++++++++++ src/blosc2/hdf5_source.py | 43 +++++++++++++++++++ src/blosc2/remote_store.py | 21 ++++++++- tests/ctable/test_remote_ctable.py | 54 ++++++++++++++++++++--- 4 files changed, 180 insertions(+), 7 deletions(-) diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index 6fc65b306..c0761bc7f 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -928,6 +928,8 @@ def close(self) -> None: def load_index_catalog(self) -> dict: self._check_open() + if self._owner.format == "hdf5": + return self._load_pytables_index_catalog() raw = self._metadata().get("index_catalog") if not isinstance(raw, dict): return {} @@ -972,6 +974,73 @@ def load_index_catalog(self) -> dict: catalog[name] = copy.deepcopy(descriptor) return catalog + def _load_pytables_index_catalog(self) -> dict: + from blosc2.indexing import _build_descriptor, _field_target_descriptor, _store_array_sidecar + + catalog = {} + for name, source in self._metadata().get("pytables_indexes", {}).items(): + column = self.open_column(name) + opened = [] + try: + for key in ("sorted", "indices", "sortedLR", "indicesLR"): + array = self._owner.remote_array(source[key]) + self._arrays.append(array) + opened.append(array) + tail = int(source["tail"]) + values = np.concatenate((opened[0][:].reshape(-1), opened[2][:tail])) + raw_positions = np.concatenate((opened[1][:].reshape(-1), opened[3][:tail])) + if raw_positions.dtype.kind != "u" or raw_positions.itemsize != 8: + continue + if len(raw_positions) and int(raw_positions.max()) >= len(column): + continue + positions = raw_positions.astype(np.int64) + chunk_len = max(1, int(source["slicesize"])) + last = np.minimum(np.arange(chunk_len, len(values) + chunk_len, chunk_len), len(values)) - 1 + mins = values[::chunk_len] + maxs = values[last] + target = _field_target_descriptor(None) + token = name + opsi = { + "chunk_len": chunk_len, + "block_len": chunk_len, + "chunk_multiplier": 1, + "nblocks": len(mins), + "cycles": int(source["optlevel"]), + "attempted_cycles": int(source["optlevel"]), + "max_cycles": int(source["optlevel"]), + "is_csi": bool(source["is_csi"]), + } + for category, sidecar_name, data in ( + ("opsi", "values", values), + ("opsi", "positions", positions), + ("opsi_nav", "mins", mins), + ("opsi_nav", "maxs", maxs), + ): + geometry = {"chunks": (chunk_len,), "blocks": (chunk_len,)} if category == "opsi" else {} + info = _store_array_sidecar( + column, token, "opsi", category, sidecar_name, data, False, **geometry + ) + opsi[f"{sidecar_name}_path"] = info["path"] + catalog[name] = _build_descriptor( + column, + target, + token, + "opsi", + int(source["optlevel"]), + False, + False, + None, + column.dtype, + {}, + None, + None, + None, + opsi=opsi, + ) + except (KeyError, OSError, TypeError, ValueError): + continue + return catalog + def open_membership_postings(self, name: str, descriptor: dict, indexes: list[int]): """Read only the compressed posting-list batches needed by a query.""" from blosc2.remote_batch import _RemoteBatchArray diff --git a/src/blosc2/hdf5_source.py b/src/blosc2/hdf5_source.py index 1a38b4b0a..5fba4eedc 100644 --- a/src/blosc2/hdf5_source.py +++ b/src/blosc2/hdf5_source.py @@ -63,6 +63,48 @@ def _pytables_table_schema(dtype, shape): } +def _decoded_attr(metadata, name, default=None): + value = metadata.get("attrs", {}).get(name, default) + return _from_json_value(value) + + +def _pytables_full_indexes(datasets, groups, table_path, nrows, dtype): + parent, _, table_name = table_path.rpartition("/") + index_root = "/".join(part for part in (parent, f"_i_{table_name}") if part) + indexes = {} + for name in dtype.names or (): + group_path = f"{index_root}/{name}" + group = groups.get(group_path) + paths = {leaf: f"{group_path}/{leaf}" for leaf in ("sorted", "indices", "sortedLR", "indicesLR")} + if group is None or any(path not in datasets for path in paths.values()): + continue + indices = datasets[paths["indices"]] + if dtype_from_value(indices["dtype"]).itemsize != 8 or bool(_decoded_attr(group, "DIRTY", 1)): + continue + tail = int(_decoded_attr(datasets[paths["indicesLR"]], "nelements", 0)) + regular = math.prod(datasets[paths["indices"]]["shape"]) + if regular + tail != nrows: + continue + indexes[name] = { + **paths, + "tail": tail, + "slicesize": int(_decoded_attr(group, "slicesize", datasets[paths["indices"]]["shape"][-1])), + "optlevel": int(_decoded_attr(group, "optlevel", 0)), + "is_csi": bool(_decoded_attr(group, "is_csi", 0)), + } + return indexes + + +def _attach_pytables_indexes(datasets, groups): + for table_path, metadata in datasets.items(): + if metadata.get("kind") != "ctable": + continue + dtype = dtype_from_value(metadata["dtype"]) + metadata["pytables_indexes"] = _pytables_full_indexes( + datasets, groups, table_path, metadata["shape"][0], dtype + ) + + _DIRECT_FILTERS = {1, 2, 32026} # deflate, shuffle, Blosc2 @@ -314,6 +356,7 @@ def visit(name, obj): unsupported[name] = f"{type(exc).__name__}: {exc}" h5file.visititems(visit) + _attach_pytables_indexes(datasets, groups) with contextlib.suppress(Exception): size = os.path.getsize(path) if local else int(fs.info(path)["size"]) finally: diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 8974b1cc5..ce8a9ffa4 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -715,8 +715,25 @@ def _open_b2z_source(self, full): def remote_array(self, full): """Return a RemoteArray sharing this discovery owner's resources.""" - relative = full[len(self.root) + 1 :] if self.root else full - source = self.open_source(relative) if full in self.nodes else None + if self.format == "hdf5" and full in self.nodes: + from blosc2.hdf5_source import HDF5NDSource + + source = self.sources.get(full) + if source is None: + source = HDF5NDSource( + self.urlpath, + full, + hdf5_index=self.hdf5_index, + storage_options=self.storage_options, + _traffic=self.traffic, + _filesystem=self.filesystem, + ) + if self.source_validator is not None: + self.source_validator(source) + self.sources[full] = source + else: + relative = full[len(self.root) + 1 :] if self.root else full + source = self.open_source(relative) if full in self.nodes else None if source is None: raise KeyError(full) descriptor = { diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index fff036bcb..1b60fad40 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -39,12 +39,14 @@ def remote_table_url(tmp_path, table, name="table"): return url -def pytables_hdf5_url(name="pytables-table.h5"): +def pytables_hdf5_url(name="pytables-table.h5", *, indexed=False, indexed_field="id", index_dtype="u8"): h5py = pytest.importorskip("h5py") - data = np.array( - [(1, 1.5, b"one"), (2, 2.5, b"two"), (3, 3.5, b"three")], - dtype=[("id", "= 5) & (data["id"] < 9)] + np.testing.assert_array_equal(table.where("(id >= 5) & (id < 9)").label[:], expected) + + +def test_remote_pytables_light_index_falls_back_to_scan(): + url, data = pytables_hdf5_url("pytables-light.h5", indexed=True, index_dtype="u1") + with blosc2.RemoteCTable(url, dataset="table") as table: + assert table._get_index_catalog() == {} + np.testing.assert_array_equal(table.where("id < 3").id[:], data["id"][data["id"] < 3]) + + +def test_remote_pytables_fixed_string_full_index(): + url, data = pytables_hdf5_url("pytables-string-index.h5", indexed=True, indexed_field="label") + with blosc2.RemoteCTable(url, dataset="table") as table: + assert table._get_index_catalog()["label"]["kind"] == "opsi" + np.testing.assert_array_equal( + table.where(table.label == b"5").id[:], data["id"][data["label"] == b"5"] + ) + + def indexed_remote_table_url(tmp_path, kind, *, name=None, rows=1000, **kwargs): name = name or f"indexed-{kind}" path = tmp_path / f"{name}.b2z" From 27bd0ef0d8db399cc2f3ff209d37f25610f5e9e5 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Mon, 21 Sep 2026 11:48:31 +0200 Subject: [PATCH 49/82] Cache imported PyTables indexes --- src/blosc2/ctable_storage.py | 85 +++++++++++++++++++++++++++--- src/blosc2/hdf5_source.py | 19 +++++++ src/blosc2/msgpack_utils.py | 4 ++ src/blosc2/remote_store.py | 29 +++++++++- tests/ctable/test_remote_ctable.py | 75 ++++++++++++++++++++++---- 5 files changed, 193 insertions(+), 19 deletions(-) diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index c0761bc7f..cc9d8916f 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -23,9 +23,11 @@ import contextlib import copy +import hashlib import json import os import pathlib +import uuid from typing import Any import numpy as np @@ -980,6 +982,10 @@ def _load_pytables_index_catalog(self) -> dict: catalog = {} for name, source in self._metadata().get("pytables_indexes", {}).items(): column = self.open_column(name) + cached = self._load_cached_pytables_index(name, column) + if cached is not None: + catalog[name] = cached + continue opened = [] try: for key in ("sorted", "indices", "sortedLR", "indicesLR"): @@ -1010,24 +1016,30 @@ def _load_pytables_index_catalog(self) -> dict: "max_cycles": int(source["optlevel"]), "is_csi": bool(source["is_csi"]), } - for category, sidecar_name, data in ( + sidecars = ( ("opsi", "values", values), ("opsi", "positions", positions), ("opsi_nav", "mins", mins), ("opsi_nav", "maxs", maxs), - ): + ) + paths = self._pytables_index_paths(name) + for category, sidecar_name, data in sidecars: geometry = {"chunks": (chunk_len,), "blocks": (chunk_len,)} if category == "opsi" else {} - info = _store_array_sidecar( - column, token, "opsi", category, sidecar_name, data, False, **geometry - ) - opsi[f"{sidecar_name}_path"] = info["path"] - catalog[name] = _build_descriptor( + if paths is None: + info = _store_array_sidecar( + column, token, "opsi", category, sidecar_name, data, False, **geometry + ) + opsi[f"{sidecar_name}_path"] = info["path"] + else: + self._write_pytables_sidecar(paths[sidecar_name], data, **geometry) + opsi[f"{sidecar_name}_path"] = str(paths[sidecar_name]) + descriptor = _build_descriptor( column, target, token, "opsi", int(source["optlevel"]), - False, + paths is not None, False, None, column.dtype, @@ -1037,10 +1049,67 @@ def _load_pytables_index_catalog(self) -> dict: None, opsi=opsi, ) + if paths is not None: + from blosc2.remote_store_cache import atomic_write + + marker = { + "version": 1, + "root": self._root_key, + "column": name, + "descriptor": descriptor, + } + atomic_write(paths["marker"], json.dumps(marker).encode()) + catalog[name] = descriptor except (KeyError, OSError, TypeError, ValueError): continue return catalog + def _pytables_index_paths(self, name): + if self._owner.disk is None: + return None + digest = hashlib.sha256(f"{self._root_key}\0{name}".encode()).hexdigest() + prefix = f"_pytables_indexes/{digest}" + paths = { + key: self._owner.disk.payload_path(self._generation, f"{prefix}/{key}") + for key in ("values", "positions", "mins", "maxs") + } + paths["marker"] = paths["values"].parent / "complete.json" + return paths + + def _load_cached_pytables_index(self, name, column): + paths = self._pytables_index_paths(name) + if paths is None or not paths["marker"].is_file(): + return None + try: + marker = json.loads(paths["marker"].read_text()) + descriptor = marker["descriptor"] + opsi = descriptor["opsi"] + valid = ( + marker.get("version") == 1 + and marker.get("root") == self._root_key + and marker.get("column") == name + and descriptor.get("kind") == "opsi" + and tuple(descriptor.get("shape", ())) == tuple(column.shape) + and tuple(descriptor.get("chunks", ())) == tuple(column.chunks) + and all( + opsi.get(f"{key}_path") == str(paths[key]) and paths[key].is_file() + for key in ("values", "positions", "mins", "maxs") + ) + ) + return descriptor if valid else None + except (KeyError, TypeError, ValueError, json.JSONDecodeError, OSError): + return None + + @staticmethod + def _write_pytables_sidecar(path, data, **geometry): + temporary = path.with_name(f".{path.name}.{uuid.uuid4().hex}.b2nd") + try: + array = blosc2.asarray(np.asarray(data), urlpath=str(temporary), mode="w", **geometry) + del array + os.replace(temporary, path) + finally: + blosc2.remove_urlpath(str(temporary)) + def open_membership_postings(self, name: str, descriptor: dict, indexes: list[int]): """Read only the compressed posting-list batches needed by a query.""" from blosc2.remote_batch import _RemoteBatchArray diff --git a/src/blosc2/hdf5_source.py b/src/blosc2/hdf5_source.py index 5fba4eedc..30c5cff6f 100644 --- a/src/blosc2/hdf5_source.py +++ b/src/blosc2/hdf5_source.py @@ -257,6 +257,23 @@ def _filesystem_and_path(urlpath, storage_options=None, filesystem=None): return fsspec.core.url_to_fs(urlpath, **options) +def hdf5_source_state(urlpath, storage_options=None, filesystem=None): + """Return the stable fsspec identity used to invalidate cached discovery.""" + fs, path = _filesystem_and_path(urlpath, storage_options, filesystem) + try: + info = fs.info(path) + state = {"size": int(info["size"])} + with contextlib.suppress(Exception): + state["ukey"] = str(fs.ukey(path)) + for key in ("etag", "version_id", "mtime"): + if info.get(key) is not None: + state[key] = str(info[key]) + return state + finally: + if filesystem is None: + _close_owned_filesystem(fs) + + def _close_owned_filesystem(filesystem): """Release the async session of an fsspec filesystem this code created.""" close = getattr(filesystem, "close_session", None) @@ -359,6 +376,7 @@ def visit(name, obj): _attach_pytables_indexes(datasets, groups) with contextlib.suppress(Exception): size = os.path.getsize(path) if local else int(fs.info(path)["size"]) + source_state = None if local else hdf5_source_state(urlpath, storage_options, fs) finally: if fs is not None and _filesystem is None: _close_owned_filesystem(fs) @@ -369,6 +387,7 @@ def visit(name, obj): "version": HDF5_INDEX_VERSION, "urlpath": os.fspath(urlpath), "size": size, + "source_state": source_state, "groups": groups, "datasets": datasets, } diff --git a/src/blosc2/msgpack_utils.py b/src/blosc2/msgpack_utils.py index 692489872..3a6bc358c 100644 --- a/src/blosc2/msgpack_utils.py +++ b/src/blosc2/msgpack_utils.py @@ -119,6 +119,10 @@ def _encode_msgpack_ext(obj): return float(obj) if isinstance(obj, np.bool_): return bool(obj) + if isinstance(obj, np.bytes_): + return bytes(obj) + if isinstance(obj, np.str_): + return str(obj) return blosc2_ext.encode_tuple(obj) diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index ce8a9ffa4..0aa2d23bc 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -172,6 +172,15 @@ def __init__( self.filesystem = _filesystem if self.filesystem is None: self.filesystem, _ = fsspec.core.url_to_fs(self.urlpath, **options) + if manifest and self.format == "hdf5": + from blosc2.hdf5_source import hdf5_source_state + + current_state = hdf5_source_state(self.urlpath, self.storage_options, self.filesystem) + if manifest["metadata"].get("source_state") != current_state: + manifest = None + self.generation = uuid.uuid4().hex + self.metadata = {} + self.restored_manifest = manifest if manifest: self._restore_manifest(manifest) elif self.format == "b2z": @@ -778,8 +787,25 @@ def restore_caches(self, manifest): # persisted carrier is opened. get_cache() adopts this # artifact leaf after that source has been authorized. continue - if path not in self.nodes or self.nodes[path][0] != "ndarray": + if path not in self.nodes or ( + self.nodes[path][0] != "ndarray" + and not (self.format == "hdf5" and self.nodes[path][0] == "ctable") + ): raise ValueError("Invalid cached RemoteStore leaf") + if self.format == "hdf5": + from blosc2.hdf5_source import HDF5NDSource + + source = HDF5NDSource( + self.urlpath, + path, + hdf5_index=self.hdf5_index, + storage_options=self.storage_options, + _traffic=self.traffic, + _filesystem=self.filesystem, + ) + self.sources[path] = source + self.get_cache(source) + continue relative = path[len(self.root) + 1 :] if self.root else path self.resolve(relative) self.get_cache(self.open_source(relative)) @@ -1300,6 +1326,7 @@ def __init__( _source_format=_source_format, _traffic=_traffic, ) + manifest = owner.restored_manifest except BaseException: if disk is not None: disk.close() diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 1b60fad40..79a4de742 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -6,6 +6,7 @@ import io import itertools import os +import pathlib import zipfile import numpy as np @@ -39,12 +40,15 @@ def remote_table_url(tmp_path, table, name="table"): return url -def pytables_hdf5_url(name="pytables-table.h5", *, indexed=False, indexed_field="id", index_dtype="u8"): +def pytables_hdf5_url( + name="pytables-table.h5", *, indexed=False, indexed_field="id", index_dtype="u8", indexed_rows=21 +): h5py = pytest.importorskip("h5py") rows = [(1, 1.5, b"one"), (2, 2.5, b"two"), (3, 3.5, b"three")] if indexed: rows = [ - (value, value + 0.5, str(value).encode()) for value in np.random.default_rng(4).permutation(21) + (value, value + 0.5, str(value).encode()) + for value in np.random.default_rng(4).permutation(indexed_rows) ] data = np.array(rows, dtype=[("id", "= 22").id[:], data["id"][data["id"] >= 22]) + + def indexed_remote_table_url(tmp_path, kind, *, name=None, rows=1000, **kwargs): name = name or f"indexed-{kind}" path = tmp_path / f"{name}.b2z" From 36602d09eb1b487aa3edf702412c356223780d57 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Mon, 21 Sep 2026 11:56:57 +0200 Subject: [PATCH 50/82] Support HDF5 tables in portable stores --- src/blosc2/ctable_storage.py | 15 +++++++++++++++ src/blosc2/remote_store.py | 17 ++++++++++++++--- tests/ctable/test_remote_ctable.py | 4 ++++ 3 files changed, 33 insertions(+), 3 deletions(-) diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index cc9d8916f..ea20a14e3 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -25,6 +25,7 @@ import copy import hashlib import json +import math import os import pathlib import uuid @@ -660,6 +661,8 @@ def __init__(self, records, name): dtype = property(lambda self: self._dtype) shape = property(lambda self: self._shape) ndim = property(lambda self: len(self._shape)) + nbytes = property(lambda self: math.prod(self._shape) * self._dtype.itemsize) + cbytes = property(lambda self: 0) def __len__(self): return self.shape[0] @@ -667,6 +670,11 @@ def __len__(self): def __getitem__(self, key): return self.records[key][self.field] + def _take_numpy(self, indices, /, *, axis=None): + if axis not in (None, 0, -1): + raise ValueError("axis is out of bounds for a one-dimensional column") + return np.ascontiguousarray(self[indices]) + class _AllValidRows(blosc2.Operand): """Virtual validity column for immutable row-complete sources.""" @@ -681,6 +689,8 @@ def __init__(self, size, source_chunks): dtype = property(lambda self: self._dtype) shape = property(lambda self: self._shape) ndim = property(lambda self: 1) + nbytes = property(lambda self: self._shape[0]) + cbytes = property(lambda self: 0) def __len__(self): return self.shape[0] @@ -692,6 +702,11 @@ def __getitem__(self, key): return np.bool_(True) return np.ones(self.shape[0], dtype=bool)[key] + def _take_numpy(self, indices, /, *, axis=None): + if axis not in (None, 0, -1): + raise ValueError("axis is out of bounds for a one-dimensional column") + return np.ones(np.asarray(indices).shape, dtype=bool) + class RemoteTableStorage(TableStorage): """Read-only CTable storage over a shared RemoteStore owner.""" diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 0aa2d23bc..6596fe33b 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -1959,10 +1959,18 @@ def _validate_artifact_manifest(manifest): # noqa: C901 if entry[0] == "ctable": metadata = entry[1] if ( - source.get("kind") != "b2z" + source.get("kind") not in {"b2z", "hdf5"} or not isinstance(metadata, dict) or metadata.get("kind") not in {"ctable", b"ctable"} or not isinstance(metadata.get("schema"), (str, bytes)) + or ( + source.get("kind") == "hdf5" + and ( + not isinstance(metadata.get("shape"), (list, tuple)) + or len(metadata["shape"]) != 1 + or not isinstance(metadata.get("dtype"), dict) + ) + ) ): raise ValueError("Invalid RemoteStore CTable node") if entry[0] == "remote_store": @@ -1986,8 +1994,11 @@ def _validate_artifact_manifest(manifest): # noqa: C901 continue if ( path not in nodes - or nodes[path][0] != "ndarray" - or (root and not path.startswith(root + "/")) + or ( + nodes[path][0] != "ndarray" + and not (source.get("kind") == "hdf5" and nodes[path][0] == "ctable") + ) + or (source.get("kind") != "hdf5" and root and not path.startswith(root + "/")) ): raise ValueError("Invalid cached RemoteStore leaf") linked = manifest.get("linked", {}) diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 79a4de742..81fc1dfd3 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -97,6 +97,10 @@ def test_remote_pytables_table_scan_and_shared_records(): assert table.attrs["TITLE"] == b"example" assert table._cols["id"].records is table._cols["label"].records assert sum(isinstance(array, blosc2.RemoteArray) for array in table._storage._arrays) == 1 + assert table.nbytes == data.nbytes + len(data) + assert table.cbytes == 0 + sliced = table.slice(1, 3) + np.testing.assert_array_equal(sliced.label[:], data["label"][1:3]) def test_remote_pytables_full_index_is_native_opsi(): From 669603c40010cb3d7f2ecffa327fd473585fdb17 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Mon, 21 Sep 2026 11:58:10 +0200 Subject: [PATCH 51/82] Document remote PyTables implementation --- plans/remote-pytables-table.md | 279 +++++++++++++++++++++++++++++++++ 1 file changed, 279 insertions(+) create mode 100644 plans/remote-pytables-table.md diff --git a/plans/remote-pytables-table.md b/plans/remote-pytables-table.md new file mode 100644 index 000000000..0a31c2f26 --- /dev/null +++ b/plans/remote-pytables-table.md @@ -0,0 +1,279 @@ +# Remote PyTables tables through RemoteCTable + +## Objective and agreed scope + +Make PyTables `Table` datasets in remote HDF5 files accessible as read-only +`RemoteCTable` objects through the existing fsspec infrastructure and, subsequently, +Caterva2 integration. + +The remote file remains in its original PyTables/HDF5 format. Local cached table +data and imported indexes use native Blosc2 representations. PyTables format +knowledge belongs in discovery and conversion code, not in the query planner or +index-search implementation. + +Table data is fetched and cached on demand; opening a table must not require +downloading all its records. The initial index importer eagerly converts each +selected supported index into reusable native sidecars. This does not imply +importing every index automatically at table open. + +Supported index imports are limited to **clean PyTables `full` indexes covering +every table row**, including both partially sorted full indexes and completely +sorted indexes (CSI). An incomplete final sorted slice is normal and supported. +Dirty indexes and indexes with incomplete row coverage are outside the initial +import scope. + +## Fallback behavior + +PyTables `medium`, `light`, and `ultralight` indexes are ignored. Their presence +does not prevent table access. + +Queries without a usable imported index use normal Blosc2 predicate evaluation +over the remotely backed table. Required HDF5 record chunks are fetched on demand +and their data cached locally in native Blosc2 format. Later queries can reuse +that cache. Another supported index in the same query may still narrow the rows +that need scanning, where the existing planner supports that predicate. + +Do not automatically build replacement indexes for unsupported index kinds: +doing so would require a column scan and additional local storage. Encountering +a dirty or incompletely covering index likewise leaves the table readable through +the scan path. Malformed metadata is a separate validation concern and must not +be treated as a trustworthy index. + +## Table representation and access + +A PyTables Table is a one-dimensional HDF5 compound dataset identified by +`CLASS="TABLE"`. Its dtype describes fields, while attributes supply additional +table and field metadata. + +Extend HDF5 discovery to recognize table nodes and derive a CTable schema. Expose +fields as lazy columns over a shared structured-record source, provide the +all-valid row representation required by CTable, and preserve appropriate user +attributes. Distinguish user metadata from PyTables implementation metadata and +keep index datasets available internally without presenting them as table columns. + +Start with scalar numeric, Boolean, and fixed-width byte-string fields. Define +explicit behavior for nested records, shaped fields, enums, and PyTables time +types before claiming support. In particular, PyTables `Time64` has conversion +semantics that cannot be inferred by treating its storage as an ordinary number. + +Fixed-width strings are exposed as native CTable `bytes(max_length=N)` columns +backed by NumPy `S` fields. Values remain bytes, comparisons use byte literals, +and no UTF-8 decoding is implied. This preserves the HDF5 fixed-width storage +contract, including its maximum width, while avoiding h5py's separate string +metadata from leaking into the CTable schema. + +Reuse HDF5NDSource transport, supported-filter decoding, cache accounting, and +RemoteStore ownership where possible. RemoteTableStorage and its batched reader +currently assume B2Z members and separate column arrays; recognizing a table node +alone does not remove those assumptions. + +### Shared physical reads + +PyTables stores records together. Selecting a field from a compressed dataset +generally requires fetching the compressed chunk containing all fields. Multiple +column accesses must share physical fetches and decoding rather than independently +downloading the same HDF5 chunk. + +Keep transient record buffers bounded, group row gathers by physical chunk, and +account for shared cache usage and concurrent requests. Persisted cache data must +remain Blosc2-native. Preserve read-only, close, refresh, and stale-handle behavior. + +## Native OPSI index conversion + +PyTables indexes reside in groups such as `/group/_i_table/column`. Full indexes +retain every indexed value and its absolute 64-bit row position; their sorted +values are not reduced. Consequently, the sorting and OPSI optimization already +performed by PyTables can be reused. + +| PyTables representation | Native Blosc2 representation | +| --- | --- | +| `sorted`, a two-dimensional array of sorted slices | One-dimensional OPSI values sidecar | +| Corresponding `indices` | One-dimensional `int64` positions sidecar | +| Valid prefix of `sortedLR` | Final entries of the values sidecar | +| Valid prefix of `indicesLR` | Final entries of the positions sidecar | +| `abounds` / `zbounds`, or imported block endpoints | OPSI minimum/maximum navigation sidecars | +| Index geometry and table binding | Native index descriptor | + +Validate supported format versions, clean state, full row coverage, dataset +shapes, entry counts, dtypes, and row-position range before publishing the native +descriptor. Normalize byte order and validate unsigned positions before converting +them to signed `int64`. Trim last-slice arrays using their valid lengths; exclude +padding and the bounds stored after the valid `sortedLR` payload. Older index +versions encode last-slice lengths differently and need explicit handling or an +unsupported-version diagnostic. + +### Preserve sorted-run boundaries + +Every native OPSI chunk must be sorted. Do not let a native chunk straddle two +independently sorted PyTables slices: Blosc2 can binary-search adjacent blocks as +one span, so independently sorted blocks alone are insufficient. + +A simple initial geometry maps one PyTables index chunk to one native Blosc2 +chunk with one compression block. Larger chunks are possible when their boundaries +respect sorted slices. The final incomplete slice follows the same rule. + +Flatten and repack the existing value/position pairs without re-sorting, merging, +or running OPSI optimization cycles. Generate navigation bounds for the chosen +native geometry, reusing stored bounds only when their semantics and geometry +match. The imported index retains the source ordering quality, although query +performance can change with navigation and compression geometry. + +Register the result as an ordinary native OPSI index. The planner and query +executor must not dispatch to a PyTables-specific search implementation. CSI is +included in this path; a separate conversion to native FULL is not required for +the initial implementation. + +### Original-row summaries + +Blosc2's summaries over original row segments are distinct from bounds over +sorted index blocks. Do not substitute PyTables sorted bounds for these summaries. + +Derive required summaries from imported value/position pairs, or explicitly +support their absence in the relevant native paths. Avoid accidentally invoking +the normal index builder and scanning the table payload merely to populate +summaries. Any omitted summaries must leave fallback planning correct. + +### Cost and persistence + +Eager conversion is a streaming O(N) repack: read/decompress the existing index, +normalize and reshape it, then compress native sidecars. It avoids table-record +reads and index sorting, but still transfers the selected index's entire payload. +For scale, one billion float64 values plus 64-bit positions occupy approximately +16 GB before compression. + +Reuse converted sidecars across queries and opens. Bind them to the remote source +identity/version, dataset, index identity, and conversion format version. Publish +them only after a successful conversion; partial or stale imports must not become +queryable. The original HDF5 file remains unchanged. + +Compressed-byte copying is not the general conversion path: PyTables indexes can +use zlib/shuffle or other HDF5 filters. Any future compatible Blosc2 chunk-copy +optimization requires separate validation. + +## Why lighter indexes are not imported initially + +Medium, light, and ultralight indexes discard exact within-bucket row positions. +Some also retain only sampled sorted values. Metadata translation cannot recover +the missing information needed by native exact OPSI indexes. + +Native BUCKET indexes are not a direct substitute: they use a different layout +organized around source chunks and their own value-reduction scheme. Converting +approximate indexes into that representation is a separate investigation, not a +requirement for this plan. The agreed fallback is scanning, not reconstruction. + +## fsspec and Caterva2 + +Implement and validate the storage/conversion path with fsspec first. For Caterva2, +reuse the same native representation and keep PyTables interpretation at the +source/preparation boundary. Server-side preparation could convert indexes once +near the HDF5 file and serve reusable native sidecars to clients. + +Caterva2 table discovery, metadata, and serving integration need explicit work; +existing HDF5 array support does not establish end-to-end table support. A second +PyTables predicate engine on the server is not part of this proposal. + +## Implementation sequence + +1. [x] Fix HDF5 metadata compatibility prerequisites and add representative fixtures. +2. [x] Recognize PyTables tables and implement lazy shared-record access with native + local caching; verify scan queries and read-only lifecycle behavior. +3. [x] Import clean, fully covering full/CSI indexes into native OPSI sidecars and + register them with the existing planner. Keep unsupported-index scan fallback. +4. [x] Add conversion reuse, source-version invalidation, and interrupted-import checks. +5. [x] Integrate the same representation into Caterva2 and measure cold/warm behavior. + +## Implementation status + +Implemented on 2026-09-21 in these sequence commits: + +1. `3de1681f` — fixed HDF5 metadata decoding for h5py fixed strings and empty + fixed-width scalar attributes. +2. `3dab45c6` — added PyTables table discovery and lazy shared-record + `RemoteCTable` access. +3. `d5c1ffab` — imported clean, fully covering 64-bit full indexes into native + OPSI sidecars, including fixed-width byte-string indexes. +4. `27bd0ef0` — persisted converted sidecars, added completion-marker recovery, + and invalidated generations using the fsspec source identity. +5. `36602d09` in Python-Blosc2 and `501c108` in Caterva2 — enabled portable + HDF5 CTable stores and Caterva2 metadata/filter/fetch handling through + `RemoteCTable`. + +The table fields share one remote structured-record array, so projecting several +columns reuses the same HDF5 payload cache. Imported indexes are ordinary native +OPSI descriptors; neither the planner nor Caterva2 contains a PyTables-specific +search engine. Disk-cache imports write each Blosc2 sidecar atomically and publish +`complete.json` last. A missing, corrupt, or incomplete publication is rebuilt. + +Caterva2 integration uses its portable `RemoteStore` path: a remote HDF5 table is +reported as a CTable, filtered through `RemoteCTable.where()`, and returned as a +normal CTable cframe. The direct local-HDF5 adapter remains an array-oriented +`HDF5Proxy`; converting that separate upload/unfold path was not required for +remote first-class access. + +## Validation and findings from the feasibility analysis + +Read-only probes were run in the `blosc2` conda environment using h5py and +PyTables' checked-in fixtures. PyTables itself was not installed in that environment. + +- RemoteArray successfully read structured records from `bug-idx.h5` through + fsspec's memory filesystem and matched h5py. +- Remote reads of `sorted`, `indices`, and `sortedLR` from `indexes_2_1.h5` + matched h5py. +- HDF5 table discovery exposed an empty-byte-string attribute decoding failure + (`itemsize cannot be zero in type`). +- A compound dtype containing h5py string metadata failed reconstruction + (`invalid shape in fixed-type tuple`). +- An in-memory conversion of the full `var4` index in `indexes_2_1.h5` passed + 924 range cases through the native OPSI reader and public indexed expressions, + including inclusive/exclusive bounds and the incomplete final slice. It did + not sort or call `create_index()`. This small CSI fixture demonstrates the + mapping, not broad compatibility or performance. +- Tiny multi-block native arrays exposed an incorrect span read in the current + environment. The successful conversion probe used single-block chunks. Resolve + this separately before claiming arbitrary native geometry support. + +Implemented automated checks cover lazy shared record access, scan fallback, +numeric and fixed-width byte-string OPSI queries, rejection of light indexes, +disk conversion reuse, incomplete-publication rebuild, source replacement, +portable-store validation, Caterva2 metadata/fetch integration, and sliced CTable +materialization. The Caterva2 cold/warm check records HDF5 chunk reads for the +first indexed request and verifies that repeating the request adds zero HDF5 +chunk reads. + +Further compatibility checks should include non-CSI full indexes with overlapping +sorted slices, nontrivial row permutations, duplicate values, numeric boundaries, +NaNs, supported strings, byte order, empty tables, and incomplete final slices. +Compare imported-index queries with forced scans and, where available, PyTables +results. Verify that unsupported index kinds remain readable through scanning. + +Broader scale checks should confirm that conversion does not read table payload to rebuild the index, that +multi-column selections share physical HDF5 reads, and that warm caches avoid +unnecessary refetches. Exercise cache limits, close/refresh behavior, interrupted +conversion, and source changes. + +Benchmark converted-index size, conversion time, transferred bytes, request +count, and cold/warm query latency. Include selective predicates and wide-record, +narrow-projection scans: indexing can reduce record reads, but cannot make the +remote row-oriented layout columnar. + +## Deferred work + +- Lazy conversion of index payload chunks behind native sidecar arrays. This + could reduce first-query traffic while keeping the query planner format-agnostic, + but retains PyTables layout knowledge in a source adapter. +- Import of lighter index kinds, dirty-index recovery, and indexed-prefix plus + unindexed-tail execution. +- Additional field types and geometry/compressed-copy optimizations beyond the + validated initial scope. + +## References and implementation entry points + +- [OPSI paper](https://blosc.org/docs/OPSI-indexes.pdf), also available from + [PyTables](https://www.pytables.org/docs/OPSI-indexes.pdf). +- `src/blosc2/hdf5_source.py`: metadata, transport, and HDF5 chunk decoding. +- `src/blosc2/remote_store.py`: discovery and shared remote ownership. +- `src/blosc2/remote_ctable.py`, `ctable_storage.py`, and `ctable_remote_read.py`: + table construction, storage assumptions, and batched reads. +- `src/blosc2/indexing.py`: native OPSI sidecars, navigation, and query planning. +- `/Users/faltet/blosc/PyTables/tables/index.py` and `idxutils.py`: source index + layout, last-slice handling, position encoding, and reduction rules. From f81cb28cf71aed2d8342c493540b618851223a00 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Mon, 21 Sep 2026 13:24:37 +0200 Subject: [PATCH 52/82] Fix HDF5 sparse cache restoration --- src/blosc2/remote_store.py | 20 +++++++++++--------- 1 file changed, 11 insertions(+), 9 deletions(-) diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 6596fe33b..12fd12d9d 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -795,15 +795,17 @@ def restore_caches(self, manifest): if self.format == "hdf5": from blosc2.hdf5_source import HDF5NDSource - source = HDF5NDSource( - self.urlpath, - path, - hdf5_index=self.hdf5_index, - storage_options=self.storage_options, - _traffic=self.traffic, - _filesystem=self.filesystem, - ) - self.sources[path] = source + source = self.sources.get(path) + if source is None: + source = HDF5NDSource( + self.urlpath, + path, + hdf5_index=self.hdf5_index, + storage_options=self.storage_options, + _traffic=self.traffic, + _filesystem=self.filesystem, + ) + self.sources[path] = source self.get_cache(source) continue relative = path[len(self.root) + 1 :] if self.root else path From 3c8ed53ee50b847a61fe7221515c54e969d3e657 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Mon, 21 Sep 2026 13:24:49 +0200 Subject: [PATCH 53/82] Test native PyTables interoperability --- pyproject.toml | 2 + tests/ctable/test_remote_pytables_interop.py | 73 ++++++++++++++++++++ 2 files changed, 75 insertions(+) create mode 100644 tests/ctable/test_remote_pytables_interop.py diff --git a/pyproject.toml b/pyproject.toml index 25c4c8ae1..e1ce9b786 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -100,6 +100,8 @@ test = [ # tests/test_zarr_source.py and tests/test_hdf5_source.py. "zarr>=3.0.9; platform_machine != 'wasm32'", "h5py; platform_machine != 'wasm32'", + # Exercise interoperability against files and indexes written by PyTables itself. + "tables; platform_machine != 'wasm32'", # tests/test_hdf5_source.py exercises the optional Blosc2 HDF5 filter through # it; without the plugin those paths skip instead of running in the default job. "hdf5plugin; platform_machine != 'wasm32'", diff --git a/tests/ctable/test_remote_pytables_interop.py b/tests/ctable/test_remote_pytables_interop.py new file mode 100644 index 000000000..d67dc42aa --- /dev/null +++ b/tests/ctable/test_remote_pytables_interop.py @@ -0,0 +1,73 @@ +"""Optional interoperability checks using files written by PyTables itself.""" + +from __future__ import annotations + +import numpy as np +import pytest + +import blosc2 + +fsspec = pytest.importorskip("fsspec") +tables = pytest.importorskip("tables", reason="PyTables is optional") + + +class NativeRow(tables.IsDescription): + id = tables.Int32Col(pos=0) + value = tables.Float64Col(pos=1) + active = tables.BoolCol(pos=2) + label = tables.StringCol(8, pos=3) + + +def native_pytables_url(tmp_path, name, *, index_kind=None, csi=False): + path = tmp_path / name + values = np.random.default_rng(4).permutation(2049) + with tables.open_file(path, mode="w") as h5file: + table = h5file.create_table("/", "table", NativeRow, expectedrows=len(values)) + data = np.empty(len(values), dtype=table.dtype) + data["id"] = values + data["value"] = values + 0.5 + data["active"] = values % 2 == 0 + data["label"] = [str(value).encode() for value in values] + table.append(data) + table.flush() + table.attrs.owner = "pytables" + if csi: + table.cols.id.create_csindex() + elif index_kind is not None: + table.cols.id.create_index(kind=index_kind, optlevel=6) + table.cols.label.create_index(kind=index_kind, optlevel=6) + expected = table.read_where("(id >= 15) & (id < 25)") + is_csi = table.cols.id.index.is_csi if table.cols.id.is_indexed else False + + url = f"memory://{tmp_path.name}/{name}" + fsspec.filesystem("memory").pipe(url, path.read_bytes()) + return url, data, expected, is_csi + + +def test_native_pytables_table_and_full_indexes(tmp_path): + url, data, expected, _ = native_pytables_url(tmp_path, "native-full.h5", index_kind="full") + with blosc2.RemoteCTable(url, dataset="table") as table: + catalog = table._get_index_catalog() + assert catalog["id"]["kind"] == catalog["label"]["kind"] == "opsi" + assert table.attrs["owner"] == b"pytables" + np.testing.assert_array_equal(table.where("(id >= 15) & (id < 25)").id[:], expected["id"]) + np.testing.assert_array_equal( + table.where(table.label == b"5").id[:], data["id"][data["label"] == b"5"] + ) + + +def test_native_pytables_csi(tmp_path): + url, _, expected, is_csi = native_pytables_url(tmp_path, "native-csi.h5", csi=True) + assert is_csi + with blosc2.RemoteCTable(url, dataset="table") as table: + descriptor = table._get_index_catalog()["id"] + assert descriptor["kind"] == "opsi" + assert descriptor["opsi"]["is_csi"] + np.testing.assert_array_equal(table.where("(id >= 15) & (id < 25)").id[:], expected["id"]) + + +def test_native_pytables_light_index_falls_back_to_scan(tmp_path): + url, _, expected, _ = native_pytables_url(tmp_path, "native-light.h5", index_kind="light") + with blosc2.RemoteCTable(url, dataset="table") as table: + assert table._get_index_catalog() == {} + np.testing.assert_array_equal(table.where("(id >= 15) & (id < 25)").id[:], expected["id"]) From 6c39f5c8456397ec1209f12942c0c2aa79dfbaf0 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Tue, 22 Sep 2026 07:42:54 +0200 Subject: [PATCH 54/82] Fix PyTables boolean and view handling --- src/blosc2/ctable_remote_read.py | 12 ++++++++++-- src/blosc2/ctable_storage.py | 10 ++++++---- src/blosc2/hdf5_source.py | 16 +++++++++++++--- tests/ctable/test_remote_pytables_interop.py | 5 +++++ 4 files changed, 34 insertions(+), 9 deletions(-) diff --git a/src/blosc2/ctable_remote_read.py b/src/blosc2/ctable_remote_read.py index 9de43bab7..9ad757894 100644 --- a/src/blosc2/ctable_remote_read.py +++ b/src/blosc2/ctable_remote_read.py @@ -114,6 +114,7 @@ def open_columns(storage, table, names, load): # noqa: C901 ) owner = storage._owner + source_columns = _source_columns(table) with owner.lock: storage._check_open() archive = owner.archive @@ -122,7 +123,7 @@ def open_columns(storage, table, names, load): # noqa: C901 members.setdefault(info.filename, []).append(info) def ranges_for(name): - if name in getattr(table, "_source_columns", ()): + if name in source_columns: return [] key = storage._full_key(f"_cols/{_column_name_to_relpath(name)}") spec = table._schema.columns_by_name[name].spec @@ -236,6 +237,12 @@ def _array_items(source, positions): yield (selected, *spans), (positions[selected], *spans) +def _source_columns(table): + while table.base is not None: + table = table.base + return getattr(table, "_source_columns", ()) + + def _array_values(array, positions): # noqa: C901 """Fetch and decode one chunk's selected rows before cache eviction can run.""" source = array.src @@ -330,6 +337,7 @@ def column_values(table, names, positions, *, null_masks=None): # noqa: C901 ) storage = table._remote_read_storage() + source_columns = _source_columns(table) with storage._owner.lock: storage._check_open() stored = [name for name in names if name not in table._computed_cols] @@ -348,7 +356,7 @@ def column_values(table, names, positions, *, null_masks=None): # noqa: C901 def reader(name): col = table._cols[name] spec = table._schema.columns_by_name[name].spec - if name in getattr(table, "_source_columns", ()): + if name in source_columns: return col[positions] if isinstance(col, UTF8Array): values = np.empty(len(positions), dtype=col.dtype) diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index ea20a14e3..5d1357a79 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -650,10 +650,11 @@ def index_anchor_path(self, col_name: str) -> str | None: class _RemoteHDF5Field(blosc2.Operand): """One field of a shared remote HDF5 compound dataset.""" - def __init__(self, records, name): + def __init__(self, records, name, dtype=None): self.records = records self.field = name - self._dtype = np.dtype(records.dtype.fields[name][0]) + self._storage_dtype = np.dtype(records.dtype.fields[name][0]) + self._dtype = self._storage_dtype if dtype is None else np.dtype(dtype) self._shape = records.shape self.chunks = records.chunks self.blocks = records.blocks @@ -668,7 +669,8 @@ def __len__(self): return self.shape[0] def __getitem__(self, key): - return self.records[key][self.field] + values = self.records[key][self.field] + return values if self._dtype == self._storage_dtype else values.astype(self._dtype, copy=False) def _take_numpy(self, indices, /, *, axis=None): if axis not in (None, 0, -1): @@ -788,7 +790,7 @@ def open_column(self, name: str) -> blosc2.RemoteArray: if self._hdf5_records is None: self._hdf5_records = self._owner.remote_array(self._root_key) self._arrays.append(self._hdf5_records) - return _RemoteHDF5Field(self._hdf5_records, name) + return _RemoteHDF5Field(self._hdf5_records, name, self._schema_spec(name).dtype) source_columns = set(self.load_schema().get("source_columns", ())) if name in source_columns: logical_key = f"{_COLS_DIR}/{_column_name_to_relpath(name)}" diff --git a/src/blosc2/hdf5_source.py b/src/blosc2/hdf5_source.py index 30c5cff6f..195bf2019 100644 --- a/src/blosc2/hdf5_source.py +++ b/src/blosc2/hdf5_source.py @@ -29,7 +29,7 @@ HDF5_INDEX_VERSION = 1 -def _pytables_table_schema(dtype, shape): +def _pytables_table_schema(dtype, shape, boolean_fields=()): """Return a source-bound CTable schema for a supported PyTables Table.""" dtype = np.dtype(dtype) if len(shape) != 1 or dtype.names is None: @@ -39,7 +39,9 @@ def _pytables_table_schema(dtype, shape): field = dtype.fields[name][0] if field.fields is not None or field.subdtype is not None: raise TypeError(f"PyTables field {name!r} must be a scalar") - if field.kind == "S": + if name in boolean_fields: + spec = {"kind": "bool"} + elif field.kind == "S": spec = {"kind": "bytes", "max_length": field.itemsize} elif field.kind == "b": spec = {"kind": "bool"} @@ -321,8 +323,16 @@ def _dataset_metadata(dataset): } table_class = dataset.attrs.get("CLASS") if table_class in {"TABLE", b"TABLE", np.bytes_(b"TABLE")}: + from h5py import h5t + + hdf5_type = dataset.id.get_type() + boolean_fields = { + dataset.dtype.names[index] + for index in range(hdf5_type.get_nmembers()) + if hdf5_type.get_member_type(index).get_class() == h5t.BITFIELD + } metadata["kind"] = "ctable" - metadata["schema"] = json.dumps(_pytables_table_schema(dataset.dtype, dataset.shape)) + metadata["schema"] = json.dumps(_pytables_table_schema(dataset.dtype, dataset.shape, boolean_fields)) return metadata diff --git a/tests/ctable/test_remote_pytables_interop.py b/tests/ctable/test_remote_pytables_interop.py index d67dc42aa..b7070d252 100644 --- a/tests/ctable/test_remote_pytables_interop.py +++ b/tests/ctable/test_remote_pytables_interop.py @@ -49,8 +49,13 @@ def test_native_pytables_table_and_full_indexes(tmp_path): with blosc2.RemoteCTable(url, dataset="table") as table: catalog = table._get_index_catalog() assert catalog["id"]["kind"] == catalog["label"]["kind"] == "opsi" + assert table.schema_dict()["columns"][2]["kind"] == "bool" assert table.attrs["owner"] == b"pytables" + assert "label" in str(table[:3]) np.testing.assert_array_equal(table.where("(id >= 15) & (id < 25)").id[:], expected["id"]) + np.testing.assert_array_equal( + table.where("(id >= 15) & active").id[:], data["id"][(data["id"] >= 15) & data["active"]] + ) np.testing.assert_array_equal( table.where(table.label == b"5").id[:], data["id"][data["label"] == b"5"] ) From 131a041ae65e2c59b3b55a1091579679a11227b6 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Tue, 22 Sep 2026 07:43:06 +0200 Subject: [PATCH 55/82] Add indexed remote table examples --- examples/ctable/remote_handling.py | 38 +++++- examples/ctable/remote_pytables.py | 193 +++++++++++++++++++++++++++++ 2 files changed, 229 insertions(+), 2 deletions(-) create mode 100644 examples/ctable/remote_pytables.py diff --git a/examples/ctable/remote_handling.py b/examples/ctable/remote_handling.py index c2e00d0a6..71355e385 100644 --- a/examples/ctable/remote_handling.py +++ b/examples/ctable/remote_handling.py @@ -12,7 +12,7 @@ import pprint import sys import time -from dataclasses import dataclass +from dataclasses import dataclass, fields from pathlib import Path import numpy as np @@ -39,6 +39,9 @@ class Reading: region: str = blosc2.field(blosc2.dictionary(nullable=True)) +FULL_INDEX_UNSUPPORTED = {"message", "tags"} + + def make_notes(ids): notes = np.array(["", "café", "東京の観測", "🌦️ weather improving"], dtype=object)[ids % 4] notes = np.array( @@ -56,6 +59,22 @@ def write_table(args) -> None: raise ValueError("--rows and --batch-size must be positive") if output.exists() and not args.overwrite: raise FileExistsError(f"{output} already exists; pass --overwrite to replace it") + column_names = [field.name for field in fields(Reading)] + indexed_columns = [] + if args.full is not None: + indexed_columns = ( + [name for name in column_names if name not in FULL_INDEX_UNSUPPORTED] + if args.full == "*" + else [name.strip() for name in args.full.split(",")] + ) + unknown = set(indexed_columns) - set(column_names) + unsupported = set(indexed_columns) & FULL_INDEX_UNSUPPORTED + if not all(indexed_columns) or unknown: + raise ValueError(f"invalid --full columns: {', '.join(sorted(unknown)) or args.full!r}") + if unsupported: + raise ValueError(f"FULL indexes are not supported for: {', '.join(sorted(unsupported))}") + if len(indexed_columns) != len(set(indexed_columns)): + raise ValueError("--full columns must not contain duplicates") output.parent.mkdir(parents=True, exist_ok=True) rng = np.random.default_rng(42) @@ -105,7 +124,11 @@ def write_table(args) -> None: }, validate=False, ) - table.create_index("tags", kind="membership") + if indexed_columns: + started = time.perf_counter() + for name in indexed_columns: + table.create_index(name, kind="full") + print(f"Created FULL indexes in {time.perf_counter() - started:.2f} s") with blosc2.CTable.open(str(output)) as table: assert len(table) == args.rows @@ -118,8 +141,10 @@ def write_table(args) -> None: expected[sample_ids % 43 == 0] = "" np.testing.assert_array_equal(table["note"][: len(sample_ids)], expected) assert null_counts["note"] == (args.rows + 42) // 43 + assert all(table._get_index_catalog()[name]["kind"] == "full" for name in indexed_columns) print(f"Created {output} ({output.stat().st_size / 1_000_000:.1f} MB, {args.rows:,} rows)") + print(f"FULL indexes: {', '.join(indexed_columns) if indexed_columns else 'none'}") print(f"Mask-backed null counts: {null_counts}") print(f"Now upload {output} to your cloud object storage.") @@ -299,6 +324,13 @@ def main() -> int: "url", nargs="?", help="Local .b2z CTable path or remote URL (s3://, http://, https://)" ) parser.add_argument("--write", type=Path, metavar="FILE", help="Create a local .b2z CTable instead") + parser.add_argument( + "--full", + nargs="?", + const="*", + metavar="COL1,COL2", + help="Create FULL indexes for every supported column, or only the comma-separated columns", + ) parser.add_argument( "--cache-dir", type=Path, metavar="DIR", help="Persist remote data in DIR (default: in-memory cache)" ) @@ -313,6 +345,8 @@ def main() -> int: parser.error("URL cannot be combined with --write") if args.write is None and args.url is None: parser.error("provide a local path or remote URL, or use --write FILE.b2z") + if args.full is not None and args.write is None: + parser.error("--full requires --write") try: write_table(args) if args.write is not None else access_table(args) diff --git a/examples/ctable/remote_pytables.py b/examples/ctable/remote_pytables.py new file mode 100644 index 000000000..7df42b2e5 --- /dev/null +++ b/examples/ctable/remote_pytables.py @@ -0,0 +1,193 @@ +#!/usr/bin/env python3 +####################################################################### +# Copyright (c) 2019-present, Blosc Development Team +# All rights reserved. +# +# SPDX-License-Identifier: BSD-3-Clause +####################################################################### + +"""Create a PyTables table locally or access it remotely through RemoteCTable.""" + +import argparse +import pprint +import sys +import time +from pathlib import Path + +import numpy as np +import tables + +import blosc2 + +DEFAULT_PROFILE = "blosc2" +DEFAULT_ENDPOINT_URL = "https://s3.us-west-001.backblazeb2.com" +TABLE_NAME = "readings" + + +class Reading(tables.IsDescription): + id = tables.Int64Col(pos=0) + station_id = tables.Int32Col(pos=1) + temperature = tables.Float32Col(pos=2) + humidity = tables.Int16Col(pos=3) + status = tables.StringCol(8, pos=4) + active = tables.BoolCol(pos=5) + note = tables.StringCol(32, pos=6) + + +def write_table(args) -> None: + output = args.write + if output.suffix != ".h5": + raise ValueError("output must end in .h5") + if args.rows < 1 or args.batch_size < 1: + raise ValueError("--rows and --batch-size must be positive") + if output.exists() and not args.overwrite: + raise FileExistsError(f"{output} already exists; pass --overwrite to replace it") + indexed_columns = [] + if args.full is not None: + indexed_columns = ( + list(Reading.columns) if args.full == "*" else [name.strip() for name in args.full.split(",")] + ) + unknown = set(indexed_columns) - set(Reading.columns) + if not all(indexed_columns) or unknown: + raise ValueError(f"invalid --full columns: {', '.join(sorted(unknown)) or args.full!r}") + if len(indexed_columns) != len(set(indexed_columns)): + raise ValueError("--full columns must not contain duplicates") + + output.parent.mkdir(parents=True, exist_ok=True) + rng = np.random.default_rng(42) + statuses = np.array([b"ok", b"warning", b"offline", b""], dtype="S8") + notes = np.array([b"", b"clear", b"cloudy", b"rain"], dtype="S32") + + with tables.open_file(output, mode="w") as h5file: + filters = tables.Filters(complevel=5, complib="blosc2:zstd", shuffle=True) + table = h5file.create_table( + "/", + TABLE_NAME, + Reading, + title="Synthetic weather-station readings", + filters=filters, + expectedrows=args.rows, + ) + table.attrs.version = 1 + table.attrs.sampling_interval = 0.5 + table.attrs.description = "Synthetic weather-station readings" + + for start in range(0, args.rows, args.batch_size): + stop = min(start + args.batch_size, args.rows) + ids = np.arange(start, stop, dtype=np.int64) + data = np.empty(len(ids), dtype=table.dtype) + data["id"] = ids + data["station_id"] = ids % 100 + data["temperature"] = rng.normal(18, 10, len(ids)).astype(np.float32) + data["humidity"] = rng.integers(0, 101, len(ids), dtype=np.int16) + data["status"] = statuses[(ids // 7) % len(statuses)] + data["active"] = ids % 5 != 0 + data["note"] = notes[ids % len(notes)] + table.append(data) + table.flush() + + if indexed_columns: + started = time.perf_counter() + for name in indexed_columns: + getattr(table.cols, name).create_csindex(filters=filters) + print(f"Created FULL (CSI) indexes in {time.perf_counter() - started:.2f} s") + + with tables.open_file(output) as h5file: + table = h5file.root.readings + assert table.nrows == args.rows + assert all(getattr(table.cols, name).index.is_csi for name in indexed_columns) + + print(f"Created {output} ({output.stat().st_size / 1_000_000:.1f} MB, {args.rows:,} rows)") + print(f"FULL (CSI) indexes: {', '.join(indexed_columns) if indexed_columns else 'none'}") + print(f"Now upload {output} to your cloud object storage.") + + +def access_table(args) -> None: + storage_options = None + if args.url.startswith("s3://"): + storage_options = { + "profile": args.profile, + "client_kwargs": {"endpoint_url": args.endpoint_url}, + } + + print(f"Accessing: {args.url}::{TABLE_NAME}") + started = time.perf_counter() + cache_options = {"cache_dir": args.cache_dir} if args.cache_dir is not None else {} + with blosc2.RemoteCTable( + args.url, dataset=TABLE_NAME, storage_options=storage_options, **cache_options + ) as table: + metadata_time = time.perf_counter() - started + metadata_bytes = table.traffic.nbytes + metadata_requests = table.traffic.requests + metadata = { + "type": type(table).__name__, + "rows": table.nrows, + "columns": table.col_names, + "schema": table.schema_dict(), + "attrs": dict(table.attrs), + "indexes": sorted(table._get_index_catalog()), + } + print("\n[Format: PyTables/HDF5]") + for name, value in metadata.items(): + rendered = pprint.pformat(value) if isinstance(value, (dict, list)) else value + print(f"{name:<9}: {rendered}") + + sample_start = max(0, table.nrows // 2 - 2) + sample_stop = min(sample_start + 5, table.nrows) + print(f"\nSample rows [{sample_start}:{sample_stop}]:") + print(table[sample_start:sample_stop]) + + started = time.perf_counter() + ids = table.where("(station_id == 42) & active").id[:5] + query_time = time.perf_counter() - started + print("\nQuery: (station_id == 42) & active") + print(f"first ids: {ids}") + print( + f"metadata: {metadata_time * 1000:.1f} ms, " + f"{metadata_requests} requests, {metadata_bytes / 1024:.2f} KiB" + ) + print( + f"query: {query_time * 1000:.1f} ms, " + f"{table.traffic.requests - metadata_requests} requests, " + f"{(table.traffic.nbytes - metadata_bytes) / 1024:.2f} KiB" + ) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("url", nargs="?", help="Remote .h5 URL (s3://, http://, or https://)") + parser.add_argument("--write", type=Path, metavar="FILE", help="Create a local .h5 table instead") + parser.add_argument( + "--full", + nargs="?", + const="*", + metavar="COL1,COL2", + help="Create FULL (CSI) indexes for every column, or only the comma-separated columns", + ) + parser.add_argument( + "--cache-dir", type=Path, metavar="DIR", help="Persist remote data in DIR (default: memory)" + ) + parser.add_argument("--rows", type=int, default=1_000_000) + parser.add_argument("--batch-size", type=int, default=100_000) + parser.add_argument("--overwrite", action="store_true") + parser.add_argument("--profile", default=DEFAULT_PROFILE) + parser.add_argument("--endpoint-url", default=DEFAULT_ENDPOINT_URL) + args = parser.parse_args() + + if args.write is not None and args.url is not None: + parser.error("URL cannot be combined with --write") + if args.write is None and args.url is None: + parser.error("provide a remote URL or use --write FILE.h5") + if args.full is not None and args.write is None: + parser.error("--full requires --write") + + try: + write_table(args) if args.write is not None else access_table(args) + except Exception as exc: + print(f"Error: {exc}", file=sys.stderr) + return 1 + return 0 + + +if __name__ == "__main__": + sys.exit(main()) From 69dc2490eeb317dbaa3dd45aaf3e40adb0ee8053 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Tue, 22 Sep 2026 07:43:18 +0200 Subject: [PATCH 56/82] Promote membership indexes in the public API --- src/blosc2/__init__.py | 2 ++ src/blosc2/ctable.py | 4 ++-- src/blosc2/ctable_indexing.py | 10 +++++++++- src/blosc2/indexing.py | 19 ++++++++++++++++--- src/blosc2/ndarray.py | 2 ++ tests/ctable/test_column.py | 21 +++++++++++++++++---- tests/ndarray/test_indexing.py | 6 ++++++ 7 files changed, 54 insertions(+), 10 deletions(-) diff --git a/src/blosc2/__init__.py b/src/blosc2/__init__.py index 72cc28ab8..3a1b36ee3 100644 --- a/src/blosc2/__init__.py +++ b/src/blosc2/__init__.py @@ -228,6 +228,8 @@ class IndexKind(Enum): FULL = "full" #: Tunable iterative-ordering payloads for exact filtering; not a full/CSI index. OPSI = "opsi" + #: Value-to-row postings for CTable list columns; not supported on NDArray. + MEMBERSHIP = "membership" class CachePolicy(Enum): diff --git a/src/blosc2/ctable.py b/src/blosc2/ctable.py index 049650bc0..b4f9837cf 100644 --- a/src/blosc2/ctable.py +++ b/src/blosc2/ctable.py @@ -15072,8 +15072,8 @@ def info_items(self) -> list[tuple[str, object]]: if stats is None: suffix = "(size=n/a, sidecars not directly addressable)" else: - _, cbytes, _ = stats - suffix = f"({format_nbytes_human(cbytes)})" + _, cbytes, cratio = stats + suffix = f"(cbytes: {format_nbytes_human(cbytes)}, cratio: {cratio:.2f}x)" index_summary[idx.col_name] = f"[{idx.kind}{stale}{label}] {suffix}" items = [ diff --git a/src/blosc2/ctable_indexing.py b/src/blosc2/ctable_indexing.py index c7337b1bf..1a81ada91 100644 --- a/src/blosc2/ctable_indexing.py +++ b/src/blosc2/ctable_indexing.py @@ -984,6 +984,14 @@ def create_index( # noqa: C901 lightest kind; it may still skip segments for broad range queries but cannot accelerate ``sort_by``. + ``MEMBERSHIP`` is specific to stored ``list()`` columns. It maps each + distinct scalar list item to the rows containing it and accelerates + :meth:`Column.contains` and :meth:`Column.overlaps` predicates. Nested + lists and struct items are not supported. For example:: + + table.create_index("tags", kind=blosc2.IndexKind.MEMBERSHIP) + selected = table[table["tags"].overlaps(["python", "numpy"])] + When *kind* is omitted it defaults to ``BUCKET``, except on ``utf8()`` and ``dictionary()`` columns, which are indexed by alphabetical rank and only ever consulted through a ``FULL`` index — there the default is @@ -1010,7 +1018,7 @@ def create_index( # noqa: C901 if col_name is not None: col_name = self._logical_to_physical_name(col_name) - if kind == "membership": + if kind in ("membership", blosc2.IndexKind.MEMBERSHIP): if expression is not None or col_name is None: raise ValueError("Membership indexes require a stored list column") if col_name not in self._cols: diff --git a/src/blosc2/indexing.py b/src/blosc2/indexing.py index 26771852f..4aa19df81 100644 --- a/src/blosc2/indexing.py +++ b/src/blosc2/indexing.py @@ -4154,6 +4154,8 @@ def create_index( raise TypeError(f"unexpected keyword argument(s): {unexpected}") if not isinstance(kind, blosc2.IndexKind): raise TypeError("kind must be a blosc2.IndexKind") + if kind is blosc2.IndexKind.MEMBERSHIP: + raise ValueError("IndexKind.MEMBERSHIP is only supported by CTable list columns") kind = _normalize_index_kind(kind) build = _normalize_build_mode(build) if opsi_max_cycles_arg is None: @@ -4300,10 +4302,17 @@ def _resolve_index_token(store: dict, field: str | None, name: str | None) -> st def iter_index_components(array: blosc2.NDArray, descriptor: dict): - for level in descriptor["levels"]: - level_info = descriptor["levels"][level] + levels = descriptor.get("levels") or {} + for level in levels: + level_info = levels[level] yield IndexComponent(f"summary.{level}", "summary", level, level_info.get("path")) + membership = descriptor.get("membership") + if membership is not None: + yield IndexComponent( + "membership.postings", "membership", "postings", membership.get("postings_path") + ) + bucket = descriptor.get("bucket") if bucket is not None: yield IndexComponent("bucket.values", "bucket", "values", bucket.get("values_path")) @@ -4353,6 +4362,8 @@ def iter_index_components(array: blosc2.NDArray, descriptor: dict): def _component_nbytes(array: blosc2.NDArray, descriptor: dict, component: IndexComponent) -> int: if component.path is not None: + if component.category == "membership": + return int(blosc2.BatchArray(urlpath=component.path, mode="r").nbytes) return int(_open_sidecar_file(component.path, _INDEX_MMAP_MODE).nbytes) token = descriptor["token"] return int(_load_array_sidecar(array, token, component.category, component.name, component.path).nbytes) @@ -4360,6 +4371,8 @@ def _component_nbytes(array: blosc2.NDArray, descriptor: dict, component: IndexC def _component_cbytes(array: blosc2.NDArray, descriptor: dict, component: IndexComponent) -> int: if component.path is not None: + if component.category == "membership": + return int(blosc2.BatchArray(urlpath=component.path, mode="r").cbytes) return int(_open_sidecar_file(component.path, _INDEX_MMAP_MODE).cbytes) token = descriptor["token"] sidecar = _load_array_sidecar(array, token, component.category, component.name, component.path) @@ -4531,7 +4544,7 @@ def _component_store_key(path: str) -> str: if idx < 0: raise KeyError(f"Cannot resolve index component path {path!r} inside table store.") relpath = normalized[idx:] - for suffix in (".b2nd", ".b2f"): + for suffix in (".b2nd", ".b2f", ".b2b"): if relpath.endswith(suffix): relpath = relpath[: -len(suffix)] break diff --git a/src/blosc2/ndarray.py b/src/blosc2/ndarray.py index eac313882..6a9765613 100644 --- a/src/blosc2/ndarray.py +++ b/src/blosc2/ndarray.py @@ -5150,6 +5150,8 @@ def create_index( separate exact-filtering index kind; it incrementally improves physical ordering but does not try to produce a completely sorted full/CSI payload. + ``MEMBERSHIP`` is reserved for CTable list columns and is not + supported by NDArray. optlevel : int, optional Optimization level for index payload construction. For ``kind=OPSI``, this controls the default number of iterative OPSI diff --git a/tests/ctable/test_column.py b/tests/ctable/test_column.py index fa9bdd1d6..290a231be 100644 --- a/tests/ctable/test_column.py +++ b/tests/ctable/test_column.py @@ -938,7 +938,7 @@ def test_info_schema_expands_unicode_dtype_labels(): assert "U16 (Unicode)" in info -def test_info_indexes_only_report_cbytes(tmp_path): +def test_info_indexes_report_cbytes_and_cratio(tmp_path): @dataclass class IndexedRow: id: int = blosc2.field(blosc2.int32()) @@ -951,11 +951,24 @@ class IndexedRow: info = repr(t.info) index_block = re.split(r"\nindexes\s+:", info, maxsplit=1)[1] - # Indexes report only their (compressed) on-disk size, no nbytes/cratio. assert "[full]" in index_block - assert re.search(r"\[full\] \([\d.]+ (?:B|KiB|MiB|GiB)\)", index_block) + assert re.search(r"\[full\] \(cbytes: [\d.]+ (?:B|KiB|MiB|GiB), cratio: \d+\.\d{2}x\)", index_block) assert "nbytes" not in index_block - assert "cratio" not in index_block + + +def test_info_reports_membership_index_size(tmp_path): + @dataclass + class MembershipRow: + tags: list[int] = blosc2.field(blosc2.list(blosc2.int32(), batch_rows=2)) # noqa: RUF009 + + path = str(tmp_path / "membership.b2z") + with CTable(MembershipRow, urlpath=path, mode="w") as table: + table.extend({"tags": [[1, 2], [2], [3, 4]]}) + table.create_index("tags", kind=blosc2.IndexKind.MEMBERSHIP) + + with CTable.open(path) as table: + info = repr(table.info) + assert re.search(r"\[membership\] \(cbytes: [\d.]+ (?:B|KiB|MiB|GiB), cratio: \d+\.\d{2}x\)", info) def test_info_cratio_uses_two_decimals_with_suffix(): diff --git a/tests/ndarray/test_indexing.py b/tests/ndarray/test_indexing.py index 554800129..23aab884e 100644 --- a/tests/ndarray/test_indexing.py +++ b/tests/ndarray/test_indexing.py @@ -47,6 +47,12 @@ def test_scalar_index_matches_scan(kind): np.testing.assert_array_equal(indexed, data[(data >= 120_000) & (data < 125_000)]) +def test_membership_index_is_ctable_only(): + arr = blosc2.arange(10) + with pytest.raises(ValueError, match="only supported by CTable list columns"): + arr.create_index(kind=blosc2.IndexKind.MEMBERSHIP) + + def test_opsi_accepts_non_multiple_chunk_block(): rng = np.random.default_rng(42) data = rng.random(5_000, dtype=np.float64) From ae19766025702483a0910ec0c6a6d3667e5dfe6e Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Tue, 22 Sep 2026 11:01:29 +0200 Subject: [PATCH 57/82] Add HDF5 table dispatch to open --- plans/remote-pytables-table.md | 9 +++-- src/blosc2/ctable.py | 3 ++ src/blosc2/ctable_storage.py | 2 + src/blosc2/hdf5_source.py | 19 --------- src/blosc2/remote_store.py | 33 ++++++++------- src/blosc2/schunk.py | 64 +++++++++++++++++++++++++++--- tests/ctable/test_remote_ctable.py | 22 +++++++++- tests/test_hdf5_source.py | 7 ++-- tests/test_remote_store.py | 3 +- 9 files changed, 110 insertions(+), 52 deletions(-) diff --git a/plans/remote-pytables-table.md b/plans/remote-pytables-table.md index 0a31c2f26..f551075a9 100644 --- a/plans/remote-pytables-table.md +++ b/plans/remote-pytables-table.md @@ -179,7 +179,7 @@ PyTables predicate engine on the server is not part of this proposal. local caching; verify scan queries and read-only lifecycle behavior. 3. [x] Import clean, fully covering full/CSI indexes into native OPSI sidecars and register them with the existing planner. Keep unsupported-index scan fallback. -4. [x] Add conversion reuse, source-version invalidation, and interrupted-import checks. +4. [x] Add conversion reuse, explicit-refresh invalidation, and interrupted-import checks. 5. [x] Integrate the same representation into Caterva2 and measure cold/warm behavior. ## Implementation status @@ -192,8 +192,9 @@ Implemented on 2026-09-21 in these sequence commits: `RemoteCTable` access. 3. `d5c1ffab` — imported clean, fully covering 64-bit full indexes into native OPSI sidecars, including fixed-width byte-string indexes. -4. `27bd0ef0` — persisted converted sidecars, added completion-marker recovery, - and invalidated generations using the fsspec source identity. +4. `27bd0ef0` — persisted converted sidecars and added completion-marker recovery. + Cached generations now remain valid until explicit refresh, matching the + immutable-source contract. 5. `36602d09` in Python-Blosc2 and `501c108` in Caterva2 — enabled portable HDF5 CTable stores and Caterva2 metadata/filter/fetch handling through `RemoteCTable`. @@ -234,7 +235,7 @@ PyTables' checked-in fixtures. PyTables itself was not installed in that environ Implemented automated checks cover lazy shared record access, scan fallback, numeric and fixed-width byte-string OPSI queries, rejection of light indexes, -disk conversion reuse, incomplete-publication rebuild, source replacement, +disk conversion reuse, incomplete-publication rebuild, explicit source refresh, portable-store validation, Caterva2 metadata/fetch integration, and sliced CTable materialization. The Caterva2 cold/warm check records HDF5 chunk reads for the first indexed request and verifies that repeating the request adds zero HDF5 diff --git a/src/blosc2/ctable.py b/src/blosc2/ctable.py index b4f9837cf..aff637738 100644 --- a/src/blosc2/ctable.py +++ b/src/blosc2/ctable.py @@ -6023,6 +6023,9 @@ def _resolve_last_pos(self) -> int: return self._last_pos arr = self._valid_rows + if getattr(arr, "_all_valid", False): + self._last_pos = arr.shape[0] + return self._last_pos chunk_size = arr.chunks[0] last_true_pos = -1 diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index 5d1357a79..b9a649525 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -681,6 +681,8 @@ def _take_numpy(self, indices, /, *, axis=None): class _AllValidRows(blosc2.Operand): """Virtual validity column for immutable row-complete sources.""" + _all_valid = True + def __init__(self, size, source_chunks): chunk = source_chunks[0] if source_chunks else max(1, min(size, 1 << 16)) self._shape = (size,) diff --git a/src/blosc2/hdf5_source.py b/src/blosc2/hdf5_source.py index 195bf2019..bdd6f4b3c 100644 --- a/src/blosc2/hdf5_source.py +++ b/src/blosc2/hdf5_source.py @@ -259,23 +259,6 @@ def _filesystem_and_path(urlpath, storage_options=None, filesystem=None): return fsspec.core.url_to_fs(urlpath, **options) -def hdf5_source_state(urlpath, storage_options=None, filesystem=None): - """Return the stable fsspec identity used to invalidate cached discovery.""" - fs, path = _filesystem_and_path(urlpath, storage_options, filesystem) - try: - info = fs.info(path) - state = {"size": int(info["size"])} - with contextlib.suppress(Exception): - state["ukey"] = str(fs.ukey(path)) - for key in ("etag", "version_id", "mtime"): - if info.get(key) is not None: - state[key] = str(info[key]) - return state - finally: - if filesystem is None: - _close_owned_filesystem(fs) - - def _close_owned_filesystem(filesystem): """Release the async session of an fsspec filesystem this code created.""" close = getattr(filesystem, "close_session", None) @@ -386,7 +369,6 @@ def visit(name, obj): _attach_pytables_indexes(datasets, groups) with contextlib.suppress(Exception): size = os.path.getsize(path) if local else int(fs.info(path)["size"]) - source_state = None if local else hdf5_source_state(urlpath, storage_options, fs) finally: if fs is not None and _filesystem is None: _close_owned_filesystem(fs) @@ -397,7 +379,6 @@ def visit(name, obj): "version": HDF5_INDEX_VERSION, "urlpath": os.fspath(urlpath), "size": size, - "source_state": source_state, "groups": groups, "datasets": datasets, } diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 12fd12d9d..35e23669e 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -120,6 +120,7 @@ def __init__( _manifest_validator=None, _max_nodes=None, _source_format=None, + _hdf5_index=None, _traffic=None, ): self.urlpath, dataset, self.format = parse_container_url(urlpath, dataset) @@ -172,21 +173,13 @@ def __init__( self.filesystem = _filesystem if self.filesystem is None: self.filesystem, _ = fsspec.core.url_to_fs(self.urlpath, **options) - if manifest and self.format == "hdf5": - from blosc2.hdf5_source import hdf5_source_state - - current_state = hdf5_source_state(self.urlpath, self.storage_options, self.filesystem) - if manifest["metadata"].get("source_state") != current_state: - manifest = None - self.generation = uuid.uuid4().hex - self.metadata = {} self.restored_manifest = manifest if manifest: self._restore_manifest(manifest) elif self.format == "b2z": self._open_b2z() elif self.format == "hdf5": - self._open_hdf5() + self._open_hdf5(_hdf5_index) else: self._open_zarr() self._check_node_limit() @@ -451,17 +444,21 @@ def _open_b2z(self): self.archive._opening_ranges.clear() self.archive.capture_metadata = False - def _open_hdf5(self): + def _open_hdf5(self, hdf5_index=None): from blosc2.hdf5_source import decode_hdf5_value, scan_hdf5_index unsupported = {} - self.hdf5_index = scan_hdf5_index( - self.urlpath, - self.storage_options, - unsupported=unsupported, - traffic=self.traffic, - _filesystem=self.filesystem, - ) + if hdf5_index is None: + self.hdf5_index = scan_hdf5_index( + self.urlpath, + self.storage_options, + unsupported=unsupported, + traffic=self.traffic, + _filesystem=self.filesystem, + ) + else: + self.hdf5_index = hdf5_index + self._validate_hdf5_index() for path, metadata in self.hdf5_index["groups"].items(): self._add(path, "group") self.attrs[path] = { @@ -1277,6 +1274,7 @@ def __init__( _manifest_validator=None, _max_nodes=None, _source_format=None, + _hdf5_index=None, _traffic=None, nested_storage_options=None, ): @@ -1326,6 +1324,7 @@ def __init__( _manifest_validator=_manifest_validator, _max_nodes=_max_nodes, _source_format=_source_format, + _hdf5_index=_hdf5_index, _traffic=_traffic, ) manifest = owner.restored_manifest diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index 9b682095b..fc4bf34fc 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2230,6 +2230,14 @@ def _resolve_fsspec_format(urlpath, dataset, source_format, hdf5_index): return urlpath, dataset, source_format +def _open_lazy_fsspec(urlpath, source_format, options): + if source_format == "b2z": + return _open_remote_b2z(urlpath, options) + if source_format == "hdf5": + return _open_remote_hdf5(urlpath, options) + return blosc2.RemoteArray(urlpath, **options) + + def _open_fsspec_url(urlpath: str, mode: str, offset: int, kwargs: dict): """Open a container living behind an fsspec URL. @@ -2284,9 +2292,7 @@ def _open_fsspec_url(urlpath: str, mode: str, offset: int, kwargs: dict): requested = [k for k, v in kwargs.items() if v is not None] if requested: raise NotImplementedError(f"{', '.join(requested)} is not supported with lazy=True") - if source_format == "b2z": - return _open_remote_b2z(urlpath, remote_array_options) - return blosc2.RemoteArray(urlpath, **remote_array_options) + return _open_lazy_fsspec(urlpath, source_format, remote_array_options) _validate_non_lazy_fsspec_options(immutable_present, remote_array_options, cache_path, max_concurrency) @@ -2362,6 +2368,53 @@ def _open_remote_b2z(urlpath, options): raise +def _open_remote_hdf5(urlpath, options): + """Discover HDF5 groups and PyTables tables while retaining array-only options.""" + if ( + not is_fsspec_url(urlpath) + or options["cache_path"] is not None + or "hdf5_index" in options + or options["assume_immutable"] is not True + ): + return blosc2.RemoteArray(urlpath, **options) + dataset = options.get("dataset") + hdf5_index = None + traffic = None + if dataset: + array = blosc2.RemoteArray(urlpath, **options) + metadata = array.src._hdf5_index["datasets"][array.dataset] + if metadata.get("kind") != "ctable": + return array + hdf5_index = array.src._hdf5_index + traffic = array.traffic + array.close() + store_options = { + key: value + for key, value in options.items() + if key in {"dataset", "storage_options", "cache_dir", "cache_policy", "max_cache_bytes"} + } + with blosc2.RemoteStore( + urlpath, + _allow_array_root=True, + _source_format="hdf5", + _hdf5_index=hdf5_index, + _traffic=traffic, + **store_options, + ) as store: + _, full = store._resolve("") + kind = store._owner.nodes[full][0] + max_concurrency = options["max_concurrency"] + if max_concurrency is not None: + if kind != "ctable": + raise NotImplementedError("max_concurrency is only supported for remote arrays and tables") + return blosc2.RemoteCTable._from_owner( + store._owner, + full, + max_concurrency=max_concurrency, + ) + return store[""] + + def _is_hdf5_open_request(urlpath: str, kwargs: dict) -> bool: if kwargs.get("source_format") == "hdf5" or "hdf5_index" in kwargs: return True @@ -2510,7 +2563,8 @@ def open( ``lazy=True`` and reject explicit ``False``. For an fsspec URL or a Caterva2 :ref:`URLPath`, return a :ref:`RemoteArray` over the remote array dataset and read the byte ranges a slice touches. - B2Z table and group nodes return :class:`RemoteCTable` and :class:`RemoteStore` instead. + B2Z and HDF5 table and group nodes return :class:`RemoteCTable` and + :class:`RemoteStore` instead. A slice landing in a small part of a large chunk costs only the *blocks* it touches when ranges are available; chunks small enough to be one cheap request are still fetched whole. @@ -2578,7 +2632,7 @@ def open( an fsspec URL (for instance credentials, endpoint URL, token, client_kwargs, etc.). dataset: str, optional Array path within HDF5, Zarr, or B2Z containers (e.g. ``dataset="d0/d1/a2"``). - B2Z also supports table and group paths in immutable archives. + B2Z and HDF5 also support table and group paths in immutable containers. Requires ``lazy=True``. hdf5_index: dict | str | PathLike, optional Pre-computed native HDF5 index or path to a JSON index file. diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 81fc1dfd3..4ff3eff81 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -120,6 +120,21 @@ def test_remote_pytables_light_index_falls_back_to_scan(): np.testing.assert_array_equal(table.where("id < 3").id[:], data["id"][data["id"] < 3]) +def test_open_dispatches_remote_pytables_table(tmp_path): + url, data = pytables_hdf5_url(f"{tmp_path.name}-open-pytables.h5") + cache = tmp_path / "cache" + + with blosc2.open(url, lazy=True, cache_dir=cache) as store: + assert isinstance(store, blosc2.RemoteStore) + assert "table" in store + + for target, options in ((url, {"dataset": "table"}), (url + "::table", {})): + with blosc2.open(target, lazy=True, cache_dir=cache, **options) as table: + assert isinstance(table, blosc2.RemoteCTable) + np.testing.assert_array_equal(table["id"][:], data["id"]) + assert "RemoteCTable" in str(table.info) + + def test_remote_pytables_fixed_string_full_index(): url, data = pytables_hdf5_url("pytables-string-index.h5", indexed=True, indexed_field="label") with blosc2.RemoteCTable(url, dataset="table") as table: @@ -160,7 +175,7 @@ def test_remote_pytables_incomplete_index_import_is_rebuilt(tmp_path): np.testing.assert_array_equal(table.where("id < 3").id[:], data["id"][data["id"] < 3]) -def test_remote_pytables_source_change_invalidates_indexes(tmp_path): +def test_remote_pytables_source_change_requires_refresh(tmp_path): name = "pytables-changing-index.h5" url, _ = pytables_hdf5_url(name, indexed=True) options = {"dataset": "table", "cache_policy": blosc2.CachePolicy.DISK, "cache_dir": tmp_path} @@ -170,6 +185,9 @@ def test_remote_pytables_source_change_invalidates_indexes(tmp_path): url, data = pytables_hdf5_url(name, indexed=True, indexed_rows=25) with blosc2.RemoteCTable(url, **options) as table: + assert table._storage._owner.generation == generation + assert len(table) != 25 + table.refresh() assert table._storage._owner.generation != generation assert len(table) == 25 np.testing.assert_array_equal(table.where("id >= 22").id[:], data["id"][data["id"] >= 22]) @@ -406,7 +424,7 @@ def test_remote_example_batch_columns(tmp_path, capsys): script = Path(__file__).resolve().parents[2] / "examples/ctable/remote_handling.py" example = runpy.run_path(str(script)) path = tmp_path / "example-batches.b2z" - example["write_table"](SimpleNamespace(write=path, rows=100, batch_size=37, overwrite=False)) + example["write_table"](SimpleNamespace(write=path, rows=100, batch_size=37, overwrite=False, full=None)) with blosc2.open(path) as local: assert local["message"][47] is None assert local["tags"][53] is None diff --git a/tests/test_hdf5_source.py b/tests/test_hdf5_source.py index 224f3829b..78d458a37 100644 --- a/tests/test_hdf5_source.py +++ b/tests/test_hdf5_source.py @@ -606,10 +606,11 @@ def test_hdf5_auto_detection(tmp_path): assert mem_proxy.source["kind"] == "hdf5" -def test_hdf5_requires_dataset(): +def test_hdf5_without_dataset_opens_store(): url = make_memory_h5("no_ds.h5", data=np.arange(10)) - with pytest.raises(ValueError, match="HDF5 sources require a dataset path"): - blosc2.open(url, lazy=True) + with blosc2.open(url, lazy=True) as store: + assert isinstance(store, blosc2.RemoteStore) + assert store.keys() == ["data"] def test_hdf5_url_syntax_variants(tmp_path): diff --git a/tests/test_remote_store.py b/tests/test_remote_store.py index 72441928b..63e858949 100644 --- a/tests/test_remote_store.py +++ b/tests/test_remote_store.py @@ -179,8 +179,7 @@ def forbidden(*args, **kwargs): raise AssertionError("reopen fetched remote bytes") patch.setattr(type(fsspec.filesystem("memory")), "cat_file", forbidden) - if url.endswith(".b2z"): - patch.setattr(type(fsspec.filesystem("memory")), "info", forbidden) + patch.setattr(type(fsspec.filesystem("memory")), "info", forbidden) reopened = blosc2.RemoteStore(url, cache_dir=parent) reopened.close() with blosc2.RemoteStore(url, cache_dir=parent) as store: From 2e769e4cbbadbe23390e2861eae470cc7a255c8d Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Tue, 22 Sep 2026 12:30:29 +0200 Subject: [PATCH 58/82] Optimize remote HDF5 cold opens --- doc/guides/remote_arrays.md | 46 +++- doc/reference/hdf5ndsource.rst | 4 + doc/reference/remotearray.rst | 7 +- doc/reference/remotectable.rst | 3 + doc/reference/remotestore.rst | 5 + plans/remote-hdf5-opts.md | 318 +++++++++++++++++++++++ src/blosc2/__init__.py | 4 +- src/blosc2/ctable_storage.py | 1 + src/blosc2/hdf5_source.py | 389 +++++++++++++++++++++++------ src/blosc2/remote_array.py | 7 +- src/blosc2/remote_ctable.py | 5 + src/blosc2/remote_store.py | 96 ++++++- src/blosc2/schunk.py | 24 +- tests/ctable/test_remote_ctable.py | 29 +++ tests/test_fsspec.py | 15 +- tests/test_hdf5_source.py | 67 ++++- tests/test_remote_store.py | 5 +- 17 files changed, 914 insertions(+), 111 deletions(-) create mode 100644 plans/remote-hdf5-opts.md diff --git a/doc/guides/remote_arrays.md b/doc/guides/remote_arrays.md index f86db9a85..d40e4bb9b 100644 --- a/doc/guides/remote_arrays.md +++ b/doc/guides/remote_arrays.md @@ -71,11 +71,53 @@ Datasets can be specified using standard slash syntax (`file.h5/d0/d1/a2`), the Remote pre-indexing uses h5py to record dataset metadata and allocated chunk byte ranges. When opening a single {ref}`RemoteArray`, the native index is cached inside the array carrier (`schunk.vlmeta["hdf5-index"]`). When using {ref}`RemoteStore`, indexing is performed once for the entire container and shared across all leaves and sessions. Uncompressed, deflate, shuffle, and Blosc2 pipelines are decoded directly after fsspec range reads. Other pipelines use a retained h5py reader, including filters registered by `hdf5plugin`. Use `blosc2.available_datasets(url)` to inspect datasets in an HDF5 container. +An explicitly selected dataset is discovered directly; unrelated siblings and +PyTables index groups are deferred. Complete hierarchy opens discover all nodes, +but allocated-chunk maps are built only when a leaf is first read. Remote HDF5 +objects up to 8 MiB are fetched once and retained for the source session, because +one bounded transfer is cheaper than many metadata ranges. A warm disk cache +restores discovery metadata without downloading the complete source again. + +For published immutable data, a native index can be generated once and served as +an explicit JSON sidecar. The sidecar is Blosc2 metadata; the HDF5 file is not +modified, and the filename has no required convention: + +```python +import json + +import blosc2 + +source = "https://example.com/readings.h5" +index = blosc2.scan_hdf5_index(source) +with open("readings.h5.b2index.json", "w") as file: + json.dump(index, file) + +table = blosc2.open( + source + "::readings", + hdf5_index="https://example.com/readings.h5.b2index.json", +) +``` + +`hdf5_index=` accepts a dictionary, local JSON path, or remote fsspec URL and +works for arrays, PyTables tables, and hierarchy stores. `scan_hdf5_index()` +creates a complete container index by default; pass `dataset=` to create an index +scoped to one dataset or group subtree. A scoped index can only open that exact +dataset, or a store rooted at that group. The +recorded source URL must exactly match the URL being opened. Regenerate the +sidecar whenever the source object changes; remote HDF5 sources otherwise follow +the same immutable-URL contract as the cache. `storage_options` are used for +both source and sidecar URLs. + +Sidecars are never probed automatically: an explicit `hdf5_index=` avoids adding +a failed metadata request to sources that do not publish one. Version-1 native +indexes remain readable; legacy Kerchunk/reference maps are not native indexes +and are rejected. + Local HDF5 files use h5py directly, without pre-indexing or an fsspec dependency. For example, `blosc2.open("hierarchy.h5::/d0/a2")` reads the selected dataset through h5py and caches converted Blosc2 chunks in memory. Explicit -`hdf5_index=` accepts a native HDF5 index, including for local files. Legacy -HDF5 reference maps are rejected; omit `hdf5_index=` to regenerate the native index. +`hdf5_index=` also accepts a native HDF5 index for local files. Legacy HDF5 +reference maps are rejected; omit it to regenerate the native index. `RemoteArray` assumes remote sources are immutable by default, avoiding a metadata request before every read. For a replaceable `.b2nd` or Caterva2 source, pass `assume_immutable=False` to refresh its identity and invalidate stale cached chunks before each operation. diff --git a/doc/reference/hdf5ndsource.rst b/doc/reference/hdf5ndsource.rst index 969db5606..6ca20a93c 100644 --- a/doc/reference/hdf5ndsource.rst +++ b/doc/reference/hdf5ndsource.rst @@ -28,6 +28,10 @@ garbage-collected. .. autofunction:: blosc2.available_datasets +.. autofunction:: blosc2.scan_hdf5_index + +.. autofunction:: blosc2.validate_hdf5_index + .. autoclass:: blosc2.HDF5NDSource .. automethod:: __init__ diff --git a/doc/reference/remotearray.rst b/doc/reference/remotearray.rst index 257e87ff7..462802199 100644 --- a/doc/reference/remotearray.rst +++ b/doc/reference/remotearray.rst @@ -60,8 +60,11 @@ or ``dataset="dataset"``). HDF5 datasets on a local path are read directly with ``h5py``; remote HDF5 files use native metadata pre-indexing with ``h5py`` and byte ranges through fsspec. Like Zarr, HDF5 sources are assumed immutable (``assume_immutable=True``); mutable HDF5 sources are not supported. -Pre-computed native HDF5 indexes can be supplied via ``hdf5_index`` to avoid remote scanning -(including when opening a local file through the indexed reader). +Pre-computed native HDF5 indexes can be supplied via ``hdf5_index`` to avoid +remote scanning (including when opening a local file through the indexed +reader). It accepts a dictionary, local JSON path, or remote fsspec URL. The +index must match the source URL and selected dataset scope. See +:doc:`../guides/remote_arrays` for generation and publication. .. code-block:: python diff --git a/doc/reference/remotectable.rst b/doc/reference/remotectable.rst index 854086443..d5eb13650 100644 --- a/doc/reference/remotectable.rst +++ b/doc/reference/remotectable.rst @@ -8,6 +8,9 @@ archive. Fixed-width, shaped, nullable, UTF-8, batch-backed variable-length, batch-backed list, struct/object, and dictionary columns are fetched on demand. Standalone tables can be opened directly; tables inside a hierarchy can be selected with ``dataset=`` or through :class:`blosc2.RemoteStore`. +PyTables/HDF5 sources may supply ``hdf5_index=`` as a native index dictionary, +local JSON path, or remote fsspec URL. This skips source discovery and does not +modify the HDF5 file; see :doc:`../guides/remote_arrays`. Batch-backed reads transfer and decode whole compressed batches. Dictionary codes remain selective, while the full vocabulary is loaded on first use. diff --git a/doc/reference/remotestore.rst b/doc/reference/remotestore.rst index 975f30462..4b8c02e2d 100644 --- a/doc/reference/remotestore.rst +++ b/doc/reference/remotestore.rst @@ -16,6 +16,11 @@ DISK accepts ``max_cache_bytes=None`` for unbounded retention. Sources must be immutable. Generic ``blosc2.open(..., lazy=True, dataset=...)`` continues to open a single array. +For HDF5, ``hdf5_index=`` accepts a native index dictionary, local JSON path, or +remote fsspec URL. An explicit index skips hierarchy discovery and must match the +source URL and selected scope. See :doc:`../guides/remote_arrays` for the +sidecar-generation workflow. + .. code-block:: python with blosc2.RemoteStore( diff --git a/plans/remote-hdf5-opts.md b/plans/remote-hdf5-opts.md new file mode 100644 index 000000000..96605c434 --- /dev/null +++ b/plans/remote-hdf5-opts.md @@ -0,0 +1,318 @@ +# Remote HDF5 Cold-Open Optimizations and Explicit Index Sidecars + +Status: implemented and verified on 2026-09-22. + +## Objective + +Reduce first-open latency for remote HDF5 and PyTables sources while preserving +bounded memory use and on-demand payload reads. The motivating cold open is a +1.49 MiB indexed PyTables file on Backblaze B2: + +| Operation | Time | Data requests | Bytes | +| --- | ---: | ---: | ---: | +| Open indexed table | 10.38 s | 51 | 38 KiB | +| Render its head/tail | 1.18 s | 2 | 255 KiB | +| Load/convert its PyTables index | about 10 s | 392 | 554 KiB | +| Fetch the complete HDF5 file | 1.61 s | 1 | 1.49 MiB | + +The current scanner opens the remote object with `block_size=1` and +`cache_type="none"`, then asks h5py to walk every node and enumerate every +allocated chunk. HDF5 metadata is pointer-rich, so this translates small, +sequential h5py reads into many network round trips. PyTables makes the effect +larger by adding hidden `_i_` groups and their chunked index datasets. + +Implement the following four improvements together: + +1. Fetch and retain small remote HDF5 files in one request. +2. Discover an explicitly selected dataset directly and defer its PyTables index. +3. Build allocated-chunk maps only when a dataset needs direct reads. +4. Complete the existing explicit `hdf5_index=` sidecar workflow, including + RemoteCTable/RemoteStore support and public documentation. + +This is an HDF5 optimization. B2Z already uses its ZIP tail and bounded member +prefixes; Zarr has a different metadata-object/listing problem and is outside +this plan. + +## Constraints and invariants + +- Keep remote sources immutable. Replacing an HDF5 object at the same URL still + requires cache invalidation/refresh or publication at a new URL. +- Preserve exact `Traffic` accounting at the transport boundary. A one-request + prefetch is one request carrying the complete object, rather than dozens of + logical h5py reads against an in-memory buffer. +- Keep the memory fast path bounded by one internal constant. Do not add a public + tuning option until measurements show that one is needed. +- A selected dataset must retain the same public type dispatch: arrays produce + `RemoteArray`, PyTables tables produce `RemoteCTable`, and group roots produce + `RemoteStore`. +- Root/group discovery must still provide a complete hierarchy. Dataset-targeted + opens may remain partial and must not claim completeness. +- A missing PyTables index, unsupported index kind, dirty index, or malformed + sidecar must retain the existing safe fallback behavior. +- Do not change the source HDF5 file. Native indexes and cache manifests remain + Blosc2 metadata external to the HDF5 object. + +## Phase 1: retained small-file bootstrap + +Add a private small-object threshold in `src/blosc2/hdf5_source.py`, initially +8 MiB. During an uncached remote scan: + +1. Obtain the source size from the existing filesystem object. +2. If the size is at or below the threshold, fetch the object once with + `cat_file()`, charge that transfer once, and open h5py over `io.BytesIO`. +3. Retain the immutable bytes on the shared `RemoteDiscovery` owner for the + lifetime of that source session. +4. Let every `HDF5NDSource` created by that owner slice compressed chunk ranges + from the retained bytes. Its h5py fallback reader must use the same bytes. + +The retained object is a source-level read buffer, not cache payload. It is +bounded independently of `max_cache_bytes`, just like existing B2Z opening +buffers. A DISK cache persists the generated HDF5 index and ordinary converted +chunks; it need not duplicate the complete source object. A warm reopen that +already has discovery metadata must not download the full HDF5 file merely +because it is small. + +Files above the threshold keep exact range reads initially. Do not introduce an +unbounded fsspec block cache as part of this phase. Targeted discovery and lazy +allocation maps below reduce their request count without speculative payload +downloads. + +Failure handling remains simple: a short or failed full-object response aborts +the open and releases the owned filesystem session. Local files and explicit +indexes do not enter the bootstrap path. + +## Phase 2: targeted dataset discovery + +Extend the internal scan path with an optional normalized dataset scope. When +the caller supplies `dataset=` or `::dataset`, open that HDF5 object directly +instead of calling `h5file.visititems()` across the container. + +For an ordinary array, record only the selected dataset and the ancestor groups +needed to represent its path. For a PyTables table, inspect the selected compound +dataset's `CLASS`, dtype and attributes and build its existing CTable schema. +Do not inspect `_i_
` during this initial open. + +Update `_open_remote_hdf5()` so it performs one targeted discovery and uses the +result for type dispatch. Avoid the current pattern in which a temporary +`RemoteArray` performs a full-container scan before handing the same index to a +`RemoteStore`. The partial discovery object must be shared by the returned leaf, +including its filesystem, retained small-file bytes, traffic counters and cache +owner. + +Root or group opens still need complete hierarchy discovery and may use +`visititems()`. Keep these two contracts explicit: + +- **Targeted index:** sufficient for one selected dataset/table and its deferred + PyTables index metadata. +- **Container index:** sufficient for hierarchy listing and arbitrary leaf + resolution. + +When `RemoteTableStorage.load_index_catalog()` is first called for a PyTables +table whose index metadata has not been discovered, inspect only the predictable +`_i_
/` paths for that table. Check the four required leaves +(`sorted`, `indices`, `sortedLR`, `indicesLR`) and the existing `DIRTY`, row-count, +dtype and tail invariants. Merge valid metadata into the shared discovery index +under its owner lock and persist the updated manifest when a DISK cache exists. +Repeated catalog access and warm reopens must not repeat this metadata scan. + +This keeps `print(table)` independent of PyTables indexes. `table.indexes`, +`table.info`, or an index-planned query may pay the deferred index cost. + +## Phase 3: lazy allocated-chunk maps + +Separate cheap dataset metadata from the allocated-chunk table currently built +inside `_dataset_metadata()`. Shape, dtype, chunk geometry, fill value, +attributes and filter descriptions are cheap enough for discovery. The loop over +`get_num_chunks()` / `get_chunk_info()` is deferred unless the dataset is the +selected read target. + +Introduce HDF5 index format version 2 with an unambiguous allocation state: + +- `allocated: null` means the allocation map has not been scanned. +- `allocated: []` means it was scanned and the dataset has no allocated chunks. +- A list retains the current validated records. +- A top-level scope/completeness field distinguishes a targeted index from a + complete container index. + +Continue accepting version-1 indexes, where `allocated` is always a list. +Do not rewrite user-supplied v1 dictionaries or JSON files on disk. New public +full-index generation should produce a complete v2 index by default; internal +targeted discovery and container browsing may leave non-selected maps deferred. + +Add one shared `ensure_hdf5_allocations(path)` operation on the discovery owner. +It opens only the named dataset, builds and validates its records, updates the +in-memory index, and persists the manifest when appropriate. Guard it with the +existing owner lock so concurrent first reads do the work once. A direct +`HDF5NDSource` without a discovery owner uses the same targeted helper locally. + +Call this operation immediately before constructing `_chunk_records` for a +direct-range source. Do not build maps for unsupported filter pipelines, because +they use the retained h5py fallback. PyTables index discovery records its four +leaf datasets cheaply; their maps are populated only when the index is actually +converted or queried. + +Validation must reject malformed allocation states, invalid byte ranges and a +targeted sidecar used outside its declared scope. Existing sparse-fill behavior +must remain unchanged after a deferred map is installed. + +## Phase 4: complete explicit `hdf5_index=` sidecars + +The native index is a Blosc2 JSON-compatible catalog, not an HDF5, h5py or +fsspec standard. Keep sidecars explicit: do not probe an adjacent filename and +do not establish a mandatory `.b2index.json` naming convention. + +### Public API + +- Export `scan_hdf5_index` and `validate_hdf5_index` from `blosc2` so users do + not have to import an internal module to create a supported sidecar. +- Keep `hdf5_index=` accepting a dictionary, local path, or remote fsspec URL. +- Centralize dictionary/path/URL loading and validation so `HDF5NDSource`, + `RemoteArray`, `RemoteStore`, `RemoteCTable`, and `blosc2.open()` behave alike + and fetch a remote sidecar only once. +- Add public `hdf5_index=` parameters to `RemoteStore` and `RemoteCTable` and + pass them through `blosc2.open()`. +- Remove the dispatch shortcut that forces an explicit indexed PyTables dataset + into `RemoteArray`. Inspect the supplied catalog and return `RemoteCTable` or + `RemoteStore` where its metadata requires it. +- Use the source's `storage_options` for the sidecar URL. Separate credentials + for source and sidecar are outside this plan. + +The public creation workflow should be: + +```python +import json + +import blosc2 + +source = "https://example.com/readings.h5" +index = blosc2.scan_hdf5_index(source) +with open("readings.h5.b2index.json", "w") as file: + json.dump(index, file) + +table = blosc2.open( + source + "::readings", + hdf5_index="https://example.com/readings.h5.b2index.json", +) +``` + +The example filename is a user convention only. Document that generation scans +the source once, that the recorded source URL must exactly match the opened HDF5 +URL, and that replacing the source requires regenerating the sidecar. Supplying +the sidecar skips discovery; it does not alter or embed data in the HDF5 file. + +### Documentation + +Update: + +- `doc/guides/remote_arrays.md` with generation, serialization and remote URL + examples, the immutable-source contract, scope semantics and table support. +- `doc/reference/remotearray.rst`, `doc/reference/remotestore.rst`, + `doc/reference/remotectable.rst`, and `doc/reference/hdf5ndsource.rst` with the + accepted input forms and ownership/lifetime behavior. +- The `blosc2.open`, `RemoteArray`, `RemoteStore`, `RemoteCTable`, + `scan_hdf5_index`, and `validate_hdf5_index` API documentation. + +State clearly that automatic adjacent-sidecar discovery is intentionally absent, +because a failed probe would add latency to every source without a sidecar. + +## Implementation order + +1. Extract shared HDF5 index loading/validation and add v1/v2 compatibility. +2. Add targeted metadata and allocation-map helpers with local/memory filesystem + tests before changing dispatch. +3. Route selected datasets through targeted discovery and lazy PyTables index + discovery. +4. Add the retained small-file bootstrap and share it with every leaf source. +5. Expose explicit sidecars through tables/stores and fix type dispatch. +6. Export and document the public index-generation workflow. +7. Measure the motivating HTTP case and retain the results in this plan. + +This order keeps each behavioral change testable. The small-file fast path comes +after targeted helpers so its retained buffer plugs into one read abstraction +instead of creating a second scanner. + +## Verification results + +The motivating indexed Backblaze table now opens with one request for the +1,489,409-byte source. On the same URL, a cold run measured 2.386 s to open, +effectively 0 s for ``t.info``, and 0.019 s for ``str(t)``. The affected test +set passes with 618 tests and 5 skips; Ruff and the normal Sphinx build also +pass. Sphinx ``-W`` remains blocked by pre-existing repository-wide warnings. + +## Tests + +Use the `blosc2` conda environment. Extend the existing HDF5, fsspec, +RemoteStore and PyTables interoperability tests rather than adding a separate +framework. + +### Small-file bootstrap + +- Use the deterministic ranged HTTP server to verify that a file below the + threshold crosses the wire once, is charged once, and serves discovery, + head/tail reads and PyTables index conversion from retained bytes. +- Verify exact-threshold behavior, short responses, cleanup after failures, and + that a larger file is not downloaded wholesale. +- Verify that a warm DISK reopen uses its persisted index without repeating the + full-file bootstrap. + +### Targeted discovery and lazy allocation + +- Create an HDF5 container with unrelated groups, many sibling datasets and a + PyTables index. Opening one selected dataset must omit sibling and hidden-index + metadata work. +- Confirm array/table/group dispatch, attributes, ancestor paths, unsupported + sibling isolation and missing-dataset errors. +- Confirm `print(table)` does not discover or load `_i_
`, while + `table.indexes` discovers it once and a warm reopen performs no new scan. +- Verify allocation maps are built for only the first accessed leaf, are shared + by sibling handles, survive DISK reopen, and preserve sparse fill chunks. +- Exercise concurrent first access to the same map and failure cleanup. + +### Explicit sidecars + +- Cover dictionaries, local JSON paths, `memory://` URLs and the HTTP test + server for arrays, PyTables tables and group stores. +- Monkeypatch `scan_hdf5_index` to prove a valid supplied sidecar performs no + source discovery. +- Reject URL mismatches, out-of-scope targeted indexes, malformed JSON, unknown + versions, bad ranges and legacy reference maps with actionable errors. +- Verify version-1 compatibility and version-2 round trips through JSON, array + carriers, RemoteStore manifests and exported remote-reference artifacts. +- Verify a remote sidecar is fetched once and shares the source filesystem + session and traffic accounting where applicable. + +### Regression and quality checks + +- Run focused `tests/test_hdf5_source.py`, `tests/test_fsspec.py`, + `tests/test_remote_array.py`, `tests/test_remote_store.py`, and + `tests/ctable/test_remote_pytables_interop.py` coverage, followed by the + default suite. +- Run Ruff on changed Python files and build the documentation with warnings as + errors. +- No C/Cython change or new dependency is expected. + +## Acceptance criteria + +- A cold open plus `print(table)` for the motivating 1.49 MiB file requires one + source data request on the small-file path; incidental identity requests are + reported separately. +- `table.indexes` against that retained file adds no network request, while + producing the same OPSI results as today. +- A selected dataset in a large multi-dataset HDF5 file does not traverse + unrelated siblings or build their allocation maps. +- Full RemoteStore hierarchy browsing remains correct, with leaf allocation maps + populated only on first read. +- An explicit remote JSON index opens arrays, PyTables tables and HDF5 stores + without scanning the source and is fully documented as a public workflow. +- Warm DISK behavior remains zero-traffic for persisted discovery metadata and + cached payloads, subject to the existing cache limits. + +## Non-goals + +- Automatic sidecar filename probing or publication/upload APIs. +- Mutable remote HDF5/SWMR semantics. +- A general HDF5 metadata server or Kerchunk/reference-map compatibility. +- Changing B2Z or Zarr discovery. +- Coalescing large PyTables index payload ranges beyond what the retained + small-file path provides. That remains a follow-up if large indexed files are + still request-bound after these changes. diff --git a/src/blosc2/__init__.py b/src/blosc2/__init__.py index 3a1b36ee3..355df9969 100644 --- a/src/blosc2/__init__.py +++ b/src/blosc2/__init__.py @@ -607,7 +607,7 @@ def _raise(exc): ) from .zarr_source import ZarrNDSource from .b2z_source import B2ZNDSource -from .hdf5_source import HDF5NDSource, available_datasets +from .hdf5_source import HDF5NDSource, available_datasets, scan_hdf5_index, validate_hdf5_index from .indexing import Index from .schunk import SChunk, load, open @@ -947,6 +947,8 @@ def _raise(exc): "arange", "array", "available_datasets", + "scan_hdf5_index", + "validate_hdf5_index", "arccos", "arccosh", "arcsin", diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index b9a649525..c5284d767 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -998,6 +998,7 @@ def load_index_catalog(self) -> dict: def _load_pytables_index_catalog(self) -> dict: from blosc2.indexing import _build_descriptor, _field_target_descriptor, _store_array_sidecar + self._owner.ensure_pytables_indexes(self._root_key) catalog = {} for name, source in self._metadata().get("pytables_indexes", {}).items(): column = self.open_column(name) diff --git a/src/blosc2/hdf5_source.py b/src/blosc2/hdf5_source.py index bdd6f4b3c..22968a4de 100644 --- a/src/blosc2/hdf5_source.py +++ b/src/blosc2/hdf5_source.py @@ -26,7 +26,9 @@ from blosc2.proxy_source import REMOTE_MAX_CONCURRENCY, ProxyNDSource, Traffic HDF5_INDEX_FORMAT = "blosc2-hdf5-index" -HDF5_INDEX_VERSION = 1 +HDF5_INDEX_VERSION = 2 +_HDF5_INDEX_VERSIONS = {1, HDF5_INDEX_VERSION} +_SMALL_REMOTE_FILE = 8 << 20 def _pytables_table_schema(dtype, shape, boolean_fields=()): @@ -245,6 +247,23 @@ def tell(self): return self.file.tell() +@contextlib.contextmanager +def _open_hdf5_file(path, *, local=False, filesystem=None, traffic=None, blob=None): + """Open one HDF5 file from local storage, retained bytes, or exact ranges.""" + import h5py + + with contextlib.ExitStack() as stack: + if blob is not None: + raw = stack.enter_context(io.BytesIO(blob)) + elif local: + raw = stack.enter_context(open(path, "rb")) + else: + raw = stack.enter_context(filesystem.open(path, "rb", block_size=1, cache_type="none")) + if traffic is not None: + raw = _CountingFile(raw, traffic) + yield stack.enter_context(h5py.File(raw, "r")) + + def _filesystem_and_path(urlpath, storage_options=None, filesystem=None): import fsspec @@ -267,7 +286,22 @@ def _close_owned_filesystem(filesystem): close(filesystem.loop, session) -def _dataset_metadata(dataset): +def _allocated_chunks(dataset): + allocated = [] + for index in range(dataset.id.get_num_chunks()): + info = dataset.id.get_chunk_info(index) + allocated.append( + { + "offset": [int(v) for v in info.chunk_offset], + "filter_mask": int(info.filter_mask), + "byte_offset": int(info.byte_offset), + "size": int(info.size), + } + ) + return allocated + + +def _dataset_metadata(dataset, *, include_allocated=True): dcpl = dataset.id.get_create_plist() filters = [] for index in range(dcpl.get_nfilters()): @@ -282,18 +316,7 @@ def _dataset_metadata(dataset): ) chunks = None if dataset.chunks is None else [int(v) for v in dataset.chunks] direct = chunks is not None and all(item["id"] in _DIRECT_FILTERS for item in filters) - allocated = [] - if direct: - for index in range(dataset.id.get_num_chunks()): - info = dataset.id.get_chunk_info(index) - allocated.append( - { - "offset": [int(v) for v in info.chunk_offset], - "filter_mask": int(info.filter_mask), - "byte_offset": int(info.byte_offset), - "size": int(info.size), - } - ) + allocated = _allocated_chunks(dataset) if direct and include_allocated else None if direct else [] metadata = { "shape": [int(v) for v in dataset.shape], "dtype": dtype_value(dataset.dtype), @@ -319,11 +342,109 @@ def _dataset_metadata(dataset): return metadata -def scan_hdf5_index(urlpath, storage_options=None, *, unsupported=None, traffic=None, _filesystem=None): - """Build a versioned native index for one local or remote HDF5 container.""" +def _record_hdf5_object(name, obj, groups, datasets, unsupported, *, include_allocated): import h5py + try: + if isinstance(obj, h5py.Group): + groups[name] = {"attrs": {key: _json_value(value) for key, value in obj.attrs.items()}} + return + if not isinstance(obj, h5py.Dataset): + return + if obj.is_virtual: + raise TypeError("HDF5 virtual datasets are not supported") + if obj.external: + raise TypeError("HDF5 externally stored datasets are not supported") + if obj.shape is None: + raise TypeError("HDF5 null datasets are not supported") + dtype = np.dtype(obj.dtype) + if dtype.hasobject or dtype.itemsize == 0: + raise TypeError(f"HDF5NDSource only supports fixed-size dtypes, got {dtype}") + datasets[name] = _dataset_metadata(obj, include_allocated=include_allocated) + except Exception as exc: + if unsupported is None: + raise + unsupported[name] = f"{type(exc).__name__}: {exc}" + + +def _scan_hdf5_objects(h5file, dataset, groups, datasets, unsupported, lazy_allocations): + import h5py + + if dataset is None: + h5file.visititems( + lambda name, obj: _record_hdf5_object( + name, + obj, + groups, + datasets, + unsupported, + include_allocated=not lazy_allocations, + ) + ) + _attach_pytables_indexes(datasets, groups) + return + if dataset not in h5file: + raise ValueError(f"dataset {dataset!r} not found") + obj = h5file[dataset] + parent = dataset.rpartition("/")[0] + ancestors = [] + while parent: + ancestors.append(parent) + parent = parent.rpartition("/")[0] + for name in reversed(ancestors): + _record_hdf5_object(name, h5file[name], groups, datasets, unsupported, include_allocated=False) + if isinstance(obj, h5py.Group): + _record_hdf5_object(dataset, obj, groups, datasets, unsupported, include_allocated=False) + obj.visititems( + lambda name, child: _record_hdf5_object( + f"{dataset}/{name}", + child, + groups, + datasets, + unsupported, + include_allocated=not lazy_allocations, + ) + ) + if not lazy_allocations: + _attach_pytables_indexes(datasets, groups) + return + _record_hdf5_object(dataset, obj, groups, datasets, unsupported, include_allocated=True) + + +def scan_hdf5_index( + urlpath, + storage_options=None, + *, + dataset=None, + unsupported=None, + traffic=None, + _filesystem=None, + _lazy_allocations=False, + _return_blob=False, +): + """Build a native byte-range index for a local or remote HDF5 source. + + ``dataset`` limits discovery to one dataset, or one group subtree, plus its ancestor groups. The + returned dictionary is JSON-compatible and can be supplied via + ``hdf5_index=`` on later opens. + + Parameters + ---------- + urlpath: str or path-like + Local path or fsspec URL of the immutable HDF5 source. + storage_options: dict, optional + Options passed to the fsspec filesystem. + dataset: str, optional + Build a scoped index for this dataset or group subtree. By default, + index the complete container. + + Returns + ------- + dict + A JSON-compatible native HDF5 index. + """ urlpath = blosc2.core.normalize_urlpath(os.fspath(urlpath)) + dataset = None if dataset is None else str(dataset).strip("/") local = _filesystem is None and (not urlsplit(urlpath).scheme or os.path.isabs(urlpath)) if local: _check_h5py_dependencies() @@ -333,59 +454,58 @@ def scan_hdf5_index(urlpath, storage_options=None, *, unsupported=None, traffic= check_hdf5_dependencies() fs, path = _filesystem_and_path(urlpath, storage_options, _filesystem) groups, datasets = {"": {"attrs": {}}}, {} + blob = None try: - with contextlib.ExitStack() as stack: - if local: - raw = stack.enter_context(open(path, "rb")) - else: - raw = stack.enter_context(fs.open(path, "rb", block_size=1, cache_type="none")) - fileobj = _CountingFile(raw, traffic) if traffic is not None else raw - with h5py.File(fileobj, "r") as h5file: - groups[""]["attrs"] = {key: _json_value(value) for key, value in h5file.attrs.items()} - - def visit(name, obj): - try: - if isinstance(obj, h5py.Group): - groups[name] = { - "attrs": {key: _json_value(value) for key, value in obj.attrs.items()} - } - elif isinstance(obj, h5py.Dataset): - if obj.is_virtual: - raise TypeError("HDF5 virtual datasets are not supported") - if obj.external: - raise TypeError("HDF5 externally stored datasets are not supported") - if obj.shape is None: - raise TypeError("HDF5 null datasets are not supported") - dtype = np.dtype(obj.dtype) - if dtype.hasobject or dtype.itemsize == 0: - raise TypeError(f"HDF5NDSource only supports fixed-size dtypes, got {dtype}") - datasets[name] = _dataset_metadata(obj) - except Exception as exc: - if unsupported is None: - raise - unsupported[name] = f"{type(exc).__name__}: {exc}" - - h5file.visititems(visit) - _attach_pytables_indexes(datasets, groups) - with contextlib.suppress(Exception): - size = os.path.getsize(path) if local else int(fs.info(path)["size"]) + size = os.path.getsize(path) if local else int(fs.info(path)["size"]) + if not local and size <= _SMALL_REMOTE_FILE: + blob = fs.cat_file(path) + if len(blob) != size: + raise OSError(f"Short HDF5 read: expected {size} bytes, got {len(blob)}") + if traffic is not None: + traffic.charge(len(blob)) + with _open_hdf5_file(path, local=local, filesystem=fs, traffic=traffic, blob=blob) as h5file: + groups[""]["attrs"] = {key: _json_value(value) for key, value in h5file.attrs.items()} + _scan_hdf5_objects(h5file, dataset, groups, datasets, unsupported, _lazy_allocations) finally: if fs is not None and _filesystem is None: _close_owned_filesystem(fs) - if "size" not in locals(): - size = None - return { + index = { "format": HDF5_INDEX_FORMAT, "version": HDF5_INDEX_VERSION, "urlpath": os.fspath(urlpath), "size": size, + "complete": dataset is None, + "scope": dataset, "groups": groups, "datasets": datasets, } + return (index, blob) if _return_blob else index + + +def load_hdf5_index(index, urlpath, storage_options=None, *, filesystem=None, dataset=None): + """Load a native HDF5 index dictionary or JSON path and validate its source.""" + if isinstance(index, (str, os.PathLike)): + index_path = os.fspath(index) + if urlsplit(index_path).scheme: + import fsspec + + with fsspec.open(index_path, "r", **(storage_options or {})) as file: + index = json.load(file) + else: + with open(index_path) as file: + index = json.load(file) + if not isinstance(index, dict): + raise TypeError("hdf5_index must be a dict, string, or path-like object") + return validate_hdf5_index(index, urlpath, dataset=dataset) -def validate_hdf5_index(index, urlpath=None): - """Validate and return a native HDF5 index.""" +def validate_hdf5_index(index, urlpath=None, *, dataset=None): + """Validate and return a native HDF5 index. + + ``urlpath`` checks the recorded source URL. ``dataset`` additionally checks + that a scoped index describes the requested dataset. Version-1 and version-2 + native indexes are accepted. + """ if not isinstance(index, dict): raise ValueError("Invalid HDF5 index") if index.get("format") != HDF5_INDEX_FORMAT: @@ -397,19 +517,29 @@ def validate_hdf5_index(index, urlpath=None): "Legacy HDF5 reference maps are unsupported; omit hdf5_index and rescan the source" ) raise ValueError("Invalid HDF5 index format") - if index.get("version") != HDF5_INDEX_VERSION: + version = index.get("version") + if version not in _HDF5_INDEX_VERSIONS: raise ValueError(f"Unsupported HDF5 index version {index.get('version')!r}") if urlpath is not None and index.get("urlpath") != os.fspath(urlpath): raise ValueError("HDF5 index specification does not match the requested URL") if not isinstance(index.get("groups"), dict) or not isinstance(index.get("datasets"), dict): raise ValueError("Invalid HDF5 index contents") + if version == 2: + complete, scope = index.get("complete"), index.get("scope") + if not isinstance(complete, bool) or (scope is not None and not isinstance(scope, str)): + raise ValueError("Invalid HDF5 index scope") + if complete != (scope is None): + raise ValueError("Invalid HDF5 index completeness") + requested = None if dataset is None else str(dataset).strip("/") + if not complete and ((requested is None and urlpath is not None) or requested not in {None, scope}): + raise ValueError(f"HDF5 index is scoped to dataset {scope!r}") size = index.get("size") for path, meta in index["datasets"].items(): - _validate_dataset_entry(path, meta, size) + _validate_dataset_entry(path, meta, size, version) return index -def _validate_dataset_entry(path, meta, file_size): +def _validate_dataset_entry(path, meta, file_size, version=HDF5_INDEX_VERSION): """Validate one dataset entry in a native index.""" if not isinstance(path, str) or not isinstance(meta, dict): raise ValueError("Invalid HDF5 dataset entry") @@ -444,6 +574,8 @@ def _validate_dataset_entry(path, meta, file_size): if meta["direct"]: _validate_direct_filters(path, chunks, filters) allocated = meta.get("allocated") + if allocated is None and version == 2 and meta["direct"]: + return if not isinstance(allocated, list): raise ValueError(f"Invalid HDF5 allocation table for {path!r}") _validate_allocated_records(path, allocated, shape, chunks, filters, file_size) @@ -490,6 +622,71 @@ def _validate_allocated_records(path, allocated, shape, chunks, filters, file_si raise ValueError(f"HDF5 chunk range exceeds the file for {path!r}") +def scan_hdf5_allocations( + urlpath, dataset, storage_options=None, *, traffic=None, _filesystem=None, _blob=None +): + """Return the allocated-chunk records for one remote HDF5 dataset.""" + urlpath = blosc2.core.normalize_urlpath(os.fspath(urlpath)) + local = _filesystem is None and (not urlsplit(urlpath).scheme or os.path.isabs(urlpath)) + fs = None + path = urlpath + try: + if not local: + check_hdf5_dependencies() + fs, path = _filesystem_and_path(urlpath, storage_options, _filesystem) + with _open_hdf5_file(path, local=local, filesystem=fs, traffic=traffic, blob=_blob) as h5file: + obj = h5file[str(dataset).strip("/")] + return _allocated_chunks(obj) + finally: + if fs is not None and _filesystem is None: + _close_owned_filesystem(fs) + + +def scan_pytables_indexes( + urlpath, table_path, table_metadata, storage_options=None, *, traffic=None, _filesystem=None, _blob=None +): + """Discover only the PyTables index nodes belonging to one table.""" + import h5py + + urlpath = blosc2.core.normalize_urlpath(os.fspath(urlpath)) + local = _filesystem is None and (not urlsplit(urlpath).scheme or os.path.isabs(urlpath)) + fs = None + path = urlpath + groups, datasets = {}, {} + try: + if not local: + check_hdf5_dependencies() + fs, path = _filesystem_and_path(urlpath, storage_options, _filesystem) + with _open_hdf5_file(path, local=local, filesystem=fs, traffic=traffic, blob=_blob) as h5file: + parent, _, table_name = table_path.rpartition("/") + root = "/".join(part for part in (parent, f"_i_{table_name}") if part) + dtype = dtype_from_value(table_metadata["dtype"]) + for name in dtype.names or (): + group_path = f"{root}/{name}" + if group_path not in h5file or not isinstance(h5file[group_path], h5py.Group): + continue + group = h5file[group_path] + groups[group_path] = { + "attrs": {key: _json_value(value) for key, value in group.attrs.items()} + } + for leaf in ("sorted", "indices", "sortedLR", "indicesLR"): + leaf_path = f"{group_path}/{leaf}" + if leaf_path not in h5file or not isinstance(h5file[leaf_path], h5py.Dataset): + break + datasets[leaf_path] = _dataset_metadata(h5file[leaf_path], include_allocated=False) + indexes = _pytables_full_indexes( + datasets, + groups, + table_path, + table_metadata["shape"][0], + dtype_from_value(table_metadata["dtype"]), + ) + return groups, datasets, indexes + finally: + if fs is not None and _filesystem is None: + _close_owned_filesystem(fs) + + # Kept while callers migrate from the old internal name. def available_datasets(url, storage_options: dict | None = None) -> list[str]: """Return all dataset paths in an HDF5 file or native index.""" @@ -609,10 +806,14 @@ def __init__( cparams=None, _traffic: Traffic | None = None, _filesystem=None, + _blob=None, + _ensure_allocations=None, ): urlpath, dataset = self._parse_url(urlpath, dataset) self.urlpath, self.dataset, self.max_concurrency = urlpath, dataset, max_concurrency self._storage_options, self._external_filesystem = storage_options, _filesystem + self._blob = _blob + self._ensure_allocations = _ensure_allocations self._fallback_lock = threading.RLock() self._fallback_h5 = self._fallback_file = None self._lifecycle = threading.Condition() @@ -640,6 +841,18 @@ def __init__( self._hdf5_index = self._load_or_scan_index(hdf5_index) self._validate_dataset_presence(dataset) self._metadata = self._hdf5_index["datasets"][self.dataset] + if self._metadata["direct"] and self._metadata["allocated"] is None: + if self._ensure_allocations is not None: + self._metadata = self._ensure_allocations(self.dataset) + else: + self._metadata["allocated"] = scan_hdf5_allocations( + self.urlpath, + self.dataset, + self._storage_options, + traffic=self.traffic, + _filesystem=self._filesystem, + _blob=self._blob, + ) shape, physical_chunks = tuple(self._metadata["shape"]), self._metadata["chunks"] dtype = dtype_from_value(self._metadata["dtype"]) self._chunk_records = {tuple(item["offset"]): item for item in self._metadata["allocated"]} @@ -702,22 +915,27 @@ def _open_local(self): def _load_or_scan_index(self, hdf5_index): if hdf5_index is None: - return scan_hdf5_index( - self.urlpath, self._storage_options, traffic=self.traffic, _filesystem=self._filesystem + result = scan_hdf5_index( + self.urlpath, + self._storage_options, + dataset=self.dataset, + traffic=self.traffic, + _filesystem=self._filesystem, + _return_blob=True, ) - if isinstance(hdf5_index, (str, os.PathLike)): - hdf5_index_str = os.fspath(hdf5_index) - if urlsplit(hdf5_index_str).scheme: - import fsspec - - with fsspec.open(hdf5_index_str, "r", **(self._storage_options or {})) as file: - hdf5_index = json.load(file) - else: - with open(hdf5_index_str) as file: - hdf5_index = json.load(file) - if not isinstance(hdf5_index, dict): - raise TypeError("hdf5_index must be a dict, string, or path-like object") - return validate_hdf5_index(hdf5_index, self.urlpath) + hdf5_index, blob = result + if self._blob is None: + self._blob = blob + return hdf5_index + if self._ensure_allocations is not None and isinstance(hdf5_index, dict): + return hdf5_index # The shared discovery owner already validated it. + return load_hdf5_index( + hdf5_index, + self.urlpath, + self._storage_options, + filesystem=getattr(self, "_filesystem", None), + dataset=self.dataset, + ) def _validate_dataset_presence(self, raw_dataset): if self.dataset in self._hdf5_index["groups"]: @@ -771,8 +989,14 @@ def _open_fallback(self): if self._closed: raise RuntimeError("HDF5 source is closed") - raw = self._filesystem.open(self._path, "rb", block_size=1, cache_type="none") - fileobj = _CountingFile(raw, self.traffic) if self.traffic is not None else raw + raw = ( + io.BytesIO(self._blob) + if self._blob is not None + else self._filesystem.open(self._path, "rb", block_size=1, cache_type="none") + ) + fileobj = ( + _CountingFile(raw, self.traffic) if self._blob is None and self.traffic is not None else raw + ) try: h5file = h5py.File(fileobj, "r") except Exception: @@ -794,16 +1018,21 @@ def _direct_values(self, offsets, selection): if record is None: return np.full(valid_shape, _from_json_value(self._metadata["fill_value"]), dtype=self.dtype) filesystem = self._filesystem + blob = self._blob self._active_reads += 1 try: - data = filesystem.cat_file( - self._path, start=record["byte_offset"], end=record["byte_offset"] + record["size"] - ) + if blob is not None: + start = record["byte_offset"] + data = blob[start : start + record["size"]] + else: + data = filesystem.cat_file( + self._path, start=record["byte_offset"], end=record["byte_offset"] + record["size"] + ) finally: with self._lifecycle: self._active_reads -= 1 self._lifecycle.notify_all() - if self.traffic is not None: + if blob is None and self.traffic is not None: self.traffic.charge(len(data)) if len(data) != record["size"]: raise OSError(f"Short HDF5 chunk read for {self.dataset!r} at {offsets}") diff --git a/src/blosc2/remote_array.py b/src/blosc2/remote_array.py index b59074470..aa3242e3a 100644 --- a/src/blosc2/remote_array.py +++ b/src/blosc2/remote_array.py @@ -574,9 +574,10 @@ class RemoteArray(RemoteObject, blosc2.Operand): Array path within an HDF5, Zarr, or B2Z container. B2Z supports external NDArray leaves in immutable archives, e.g. ``dataset="d0/a3"``. hdf5_index: dict, str, or path-like, optional - Pre-computed native HDF5 index for the dataset, or the path to a JSON - encoding of one. It must match the source URL. Legacy HDF5 reference - maps are rejected; omit it to rescan and build a native index. + Pre-computed native HDF5 index for the dataset, or a local or remote + fsspec URL to its JSON encoding. It must match the source URL and dataset + scope. Legacy HDF5 reference maps are rejected; omit it to scan the + source and build a native index. assume_immutable: bool, optional Skip remote identity checks before reads. Defaults to ``True``. Set to ``False`` when the object at the URL may be replaced. diff --git a/src/blosc2/remote_ctable.py b/src/blosc2/remote_ctable.py index 5f0aa7945..2bbc6a8ee 100644 --- a/src/blosc2/remote_ctable.py +++ b/src/blosc2/remote_ctable.py @@ -66,6 +66,9 @@ class RemoteCTable(RemoteObject, CTable): including columns backed by independent RemoteArray carriers. Those columns share the table owner's cache budget and traffic accounting; their persisted standalone policies are not used or modified by the table. + + ``hdf5_index`` accepts a native index dictionary or a local/remote JSON + path for PyTables/HDF5 sources. Supplying one skips HDF5 discovery. """ def __new__( @@ -77,6 +80,7 @@ def __new__( cache_policy=CACHE_POLICY_DEFAULT, max_cache_bytes=CACHE_POLICY_DEFAULT, cache_dir=None, + hdf5_index=None, max_concurrency=8, metadata_buffer_bytes=8 << 20, row_buffer_bytes=64 << 20, @@ -102,6 +106,7 @@ def __new__( cache_policy=cache_policy, max_cache_bytes=max_cache_bytes, cache_dir=cache_dir, + hdf5_index=hdf5_index, _allow_array_root=True, _filesystem=_filesystem, ) diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 35e23669e..5b19825e8 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -104,6 +104,13 @@ class RemoteNode: diagnostic: str | None = None +def _resolve_hdf5_options(hdf5_index, private_index, source_format): + if hdf5_index is not None and private_index is not None: + raise TypeError("hdf5_index was supplied twice") + index = hdf5_index if hdf5_index is not None else private_index + return index, "hdf5" if index is not None and source_format is None else source_format + + class RemoteDiscovery: """Shared metadata and source resources, independent of browser presentation.""" @@ -121,6 +128,7 @@ def __init__( _max_nodes=None, _source_format=None, _hdf5_index=None, + _hdf5_blob=None, _traffic=None, ): self.urlpath, dataset, self.format = parse_container_url(urlpath, dataset) @@ -138,6 +146,9 @@ def __init__( self.notice = None self.archive = None self.zstore = None + self.hdf5_blob = None + if _hdf5_blob is not None: + self.hdf5_blob = _hdf5_blob self.sources = {} self.source_descriptors = {} self.caches = {} @@ -234,7 +245,7 @@ def _restore_manifest(self, manifest): def _validate_hdf5_index(self): from blosc2.hdf5_source import validate_hdf5_index - validate_hdf5_index(self.hdf5_index, self.urlpath) + validate_hdf5_index(self.hdf5_index, self.urlpath, dataset=self.root or None) def save_manifest(self): if self.disk is None or self.restoring or not self.is_mutable: @@ -445,20 +456,28 @@ def _open_b2z(self): self.archive.capture_metadata = False def _open_hdf5(self, hdf5_index=None): - from blosc2.hdf5_source import decode_hdf5_value, scan_hdf5_index + from blosc2.hdf5_source import decode_hdf5_value, load_hdf5_index, scan_hdf5_index unsupported = {} if hdf5_index is None: - self.hdf5_index = scan_hdf5_index( + self.hdf5_index, self.hdf5_blob = scan_hdf5_index( self.urlpath, self.storage_options, + dataset=self.root or None, unsupported=unsupported, traffic=self.traffic, _filesystem=self.filesystem, + _lazy_allocations=True, + _return_blob=True, ) else: - self.hdf5_index = hdf5_index - self._validate_hdf5_index() + self.hdf5_index = load_hdf5_index( + hdf5_index, + self.urlpath, + self.storage_options, + filesystem=self.filesystem, + dataset=self.root or None, + ) for path, metadata in self.hdf5_index["groups"].items(): self._add(path, "group") self.attrs[path] = { @@ -474,6 +493,58 @@ def _open_hdf5(self, hdf5_index=None): self.nodes[path] = ("unsupported", message) self.notice = "HDF5 view includes indexed groups and datasets; external and soft links are omitted." + def ensure_hdf5_allocations(self, path): + """Populate one deferred HDF5 allocation map and return its metadata.""" + from blosc2.hdf5_source import scan_hdf5_allocations + + with self.lock: + metadata = self.hdf5_index["datasets"][path] + if metadata["allocated"] is None: + metadata["allocated"] = scan_hdf5_allocations( + self.urlpath, + path, + self.storage_options, + traffic=self.traffic, + _filesystem=self.filesystem, + _blob=self.hdf5_blob, + ) + self.save_manifest() + return metadata + + def ensure_pytables_indexes(self, table_path): + """Discover one table's hidden PyTables index nodes on first use.""" + from blosc2.hdf5_source import decode_hdf5_value, scan_pytables_indexes + + with self.lock: + metadata = self.hdf5_index["datasets"][table_path] + if "pytables_indexes" in metadata: + return + groups, datasets, indexes = scan_pytables_indexes( + self.urlpath, + table_path, + metadata, + self.storage_options, + traffic=self.traffic, + _filesystem=self.filesystem, + _blob=self.hdf5_blob, + ) + self.hdf5_index["groups"].update(groups) + self.hdf5_index["datasets"].update(datasets) + metadata["pytables_indexes"] = indexes + for path, item in groups.items(): + if path not in self.nodes: + self._add(path, "group") + self.attrs[path] = { + key: decode_hdf5_value(value) for key, value in item.get("attrs", {}).items() + } + for path, item in datasets.items(): + if path not in self.nodes: + self._add(path, "ndarray") + self.attrs[path] = { + key: decode_hdf5_value(value) for key, value in item.get("attrs", {}).items() + } + self.save_manifest() + def _open_zarr(self): import zarr @@ -600,6 +671,8 @@ def open_source(self, path): storage_options=self.storage_options, _traffic=self.traffic, _filesystem=self.filesystem, + _blob=self.hdf5_blob, + _ensure_allocations=self.ensure_hdf5_allocations, ) else: from blosc2.zarr_source import ZarrNDSource @@ -733,6 +806,8 @@ def remote_array(self, full): storage_options=self.storage_options, _traffic=self.traffic, _filesystem=self.filesystem, + _blob=self.hdf5_blob, + _ensure_allocations=self.ensure_hdf5_allocations, ) if self.source_validator is not None: self.source_validator(source) @@ -801,6 +876,8 @@ def restore_caches(self, manifest): storage_options=self.storage_options, _traffic=self.traffic, _filesystem=self.filesystem, + _blob=self.hdf5_blob, + _ensure_allocations=self.ensure_hdf5_allocations, ) self.sources[path] = source self.get_cache(source) @@ -1219,6 +1296,9 @@ class RemoteStore(RemoteObject): ``keys()`` lists immediate children; ``get_info()`` inspects metadata without creating an array cache. Paths are relative to this group. Closing a handle leaves its previously returned arrays and group handles usable. + + ``hdf5_index`` accepts a native index dictionary or a local/remote JSON path + for HDF5 sources. Supplying one skips HDF5 discovery. """ @classmethod @@ -1267,6 +1347,7 @@ def __init__( cache_policy=CACHE_POLICY_DEFAULT, max_cache_bytes=CACHE_POLICY_DEFAULT, cache_dir=None, + hdf5_index=None, _allow_array_root=False, _filesystem=None, _manifest=None, @@ -1275,6 +1356,7 @@ def __init__( _max_nodes=None, _source_format=None, _hdf5_index=None, + _hdf5_blob=None, _traffic=None, nested_storage_options=None, ): @@ -1282,6 +1364,7 @@ def __init__( urlpath = os.fspath(urlpath) if not isinstance(urlpath, str): raise TypeError("RemoteStore requires a remote URL string") + hdf5_index, _source_format = _resolve_hdf5_options(hdf5_index, _hdf5_index, _source_format) artifact = self._try_open_artifact( urlpath, dataset, storage_options, cache_policy, max_cache_bytes, cache_dir, _allow_array_root ) @@ -1324,7 +1407,8 @@ def __init__( _manifest_validator=_manifest_validator, _max_nodes=_max_nodes, _source_format=_source_format, - _hdf5_index=_hdf5_index, + _hdf5_index=hdf5_index, + _hdf5_blob=_hdf5_blob, _traffic=_traffic, ) manifest = owner.restored_manifest diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index fc4bf34fc..773510fec 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2373,12 +2373,12 @@ def _open_remote_hdf5(urlpath, options): if ( not is_fsspec_url(urlpath) or options["cache_path"] is not None - or "hdf5_index" in options or options["assume_immutable"] is not True ): return blosc2.RemoteArray(urlpath, **options) dataset = options.get("dataset") - hdf5_index = None + hdf5_index = options.get("hdf5_index") + hdf5_blob = None traffic = None if dataset: array = blosc2.RemoteArray(urlpath, **options) @@ -2386,6 +2386,7 @@ def _open_remote_hdf5(urlpath, options): if metadata.get("kind") != "ctable": return array hdf5_index = array.src._hdf5_index + hdf5_blob = array.src._blob traffic = array.traffic array.close() store_options = { @@ -2398,21 +2399,26 @@ def _open_remote_hdf5(urlpath, options): _allow_array_root=True, _source_format="hdf5", _hdf5_index=hdf5_index, + _hdf5_blob=hdf5_blob, _traffic=traffic, **store_options, ) as store: _, full = store._resolve("") kind = store._owner.nodes[full][0] max_concurrency = options["max_concurrency"] - if max_concurrency is not None: - if kind != "ctable": - raise NotImplementedError("max_concurrency is only supported for remote arrays and tables") + if kind == "ctable": return blosc2.RemoteCTable._from_owner( store._owner, full, - max_concurrency=max_concurrency, + **({} if max_concurrency is None else {"max_concurrency": max_concurrency}), ) - return store[""] + result = store[""] + if max_concurrency is not None: + if not isinstance(result, blosc2.RemoteArray): + result.close() + raise NotImplementedError("max_concurrency is only supported for remote arrays and tables") + result.src.max_concurrency = max_concurrency + return result def _is_hdf5_open_request(urlpath: str, kwargs: dict) -> bool: @@ -2635,7 +2641,9 @@ def open( B2Z and HDF5 also support table and group paths in immutable containers. Requires ``lazy=True``. hdf5_index: dict | str | PathLike, optional - Pre-computed native HDF5 index or path to a JSON index file. + Pre-computed native HDF5 index, or a local path or remote fsspec URL + to its JSON encoding. It must match the source HDF5 URL and dataset + scope. Arrays, PyTables tables, and hierarchy stores are supported. source_format: {None, "blosc2", "zarr", "hdf5", "b2z"}, optional Format of a lazy remote source. A ``.zarr`` URL path component selects Zarr automatically; a ``.h5`` or ``.hdf5`` path selects HDF5 automatically; diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 4ff3eff81..40d6917bf 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -5,6 +5,7 @@ import dataclasses import io import itertools +import json import os import pathlib import zipfile @@ -135,6 +136,34 @@ def test_open_dispatches_remote_pytables_table(tmp_path): assert "RemoteCTable" in str(table.info) +def test_open_dispatches_remote_pytables_table_with_json_sidecar(monkeypatch): + import blosc2.hdf5_source as hdf5_source + + url, data = pytables_hdf5_url("pytables-sidecar.h5", indexed=True) + sidecar = "memory://pytables-sidecar.json" + fsspec.filesystem("memory").pipe(sidecar, json.dumps(blosc2.scan_hdf5_index(url)).encode()) + monkeypatch.setattr( + hdf5_source, + "scan_hdf5_index", + lambda *args, **kwargs: pytest.fail("explicit sidecar must skip source discovery"), + ) + with blosc2.open(url + "::table", hdf5_index=sidecar) as table: + assert isinstance(table, blosc2.RemoteCTable) + np.testing.assert_array_equal(table.id[:], data["id"]) + assert table._get_index_catalog()["id"]["kind"] == "opsi" + with blosc2.RemoteCTable(url, dataset="table", hdf5_index=sidecar) as table: + np.testing.assert_array_equal(table.id[:], data["id"]) + + +def test_small_remote_pytables_file_is_retained(): + url, _ = pytables_hdf5_url("pytables-retained.h5", indexed=True, indexed_rows=2049) + with blosc2.open(url + "::table") as table: + assert table.traffic.requests == 1 + str(table) + table._get_index_catalog() + assert table.traffic.requests == 1 + + def test_remote_pytables_fixed_string_full_index(): url, data = pytables_hdf5_url("pytables-string-index.h5", indexed=True, indexed_field="label") with blosc2.RemoteCTable(url, dataset="table") as table: diff --git a/tests/test_fsspec.py b/tests/test_fsspec.py index a3d1ba8e1..1ae7cc8e6 100644 --- a/tests/test_fsspec.py +++ b/tests/test_fsspec.py @@ -778,13 +778,24 @@ def test_http_hdf5_scan_and_warm_slice(tmp_path): with _ranged_server(tmp_path) as (urlbase, requests): remote = blosc2.open(f"{urlbase}/{path.name}", lazy=True, dataset="data") np.testing.assert_array_equal(remote[:10], data[:10]) - assert requests - assert all(requests) # Metadata and payload both use bounded ranges. + assert requests == [None] # The small source is retained by one full GET. count = len(requests) np.testing.assert_array_equal(remote[:10], data[:10]) assert len(requests) == count +def test_http_large_hdf5_keeps_range_reads(tmp_path): + h5py = pytest.importorskip("h5py") + path = tmp_path / "large-seekable.h5" + with h5py.File(path, "w") as file: + file.create_dataset("data", data=np.zeros(9 << 20, dtype="u1"), chunks=(1 << 20,)) + with _ranged_server(tmp_path) as (urlbase, requests): + remote = blosc2.open(f"{urlbase}/{path.name}", lazy=True, dataset="data") + assert remote[0] == 0 + assert requests + assert all(request is not None for request in requests) + + def test_http_hdf5_source_close_closes_session(tmp_path): h5py = pytest.importorskip("h5py") data = np.arange(10_000, dtype="int32") diff --git a/tests/test_hdf5_source.py b/tests/test_hdf5_source.py index 78d458a37..f4dbdd26e 100644 --- a/tests/test_hdf5_source.py +++ b/tests/test_hdf5_source.py @@ -49,6 +49,9 @@ def test_hdf5_native_index(): assert index["format"] == HDF5_INDEX_FORMAT assert index["datasets"]["data"]["direct"] is True assert len(index["datasets"]["data"]["allocated"]) == 3 + assert index["version"] == 2 + assert index["complete"] is True + assert index["scope"] is None malformed = json.loads(json.dumps(index)) malformed["datasets"]["data"]["allocated"] = "not-a-list" @@ -64,6 +67,37 @@ def test_hdf5_native_index(): with pytest.raises(ValueError, match="Incomplete"): validate_hdf5_index(incomplete) + legacy = dict(index, version=1) + legacy.pop("complete") + legacy.pop("scope") + assert validate_hdf5_index(legacy) is legacy + + +def test_hdf5_targeted_index_scope_and_lazy_allocations(): + from blosc2.hdf5_source import scan_hdf5_index, validate_hdf5_index + + url = make_memory_h5( + "targeted-index.h5", + a=(np.arange(12, dtype="i4"), (4,)), + b=(np.arange(8, dtype="i4"), (4,)), + ) + targeted = scan_hdf5_index(url, dataset="a") + assert targeted["complete"] is False + assert targeted["scope"] == "a" + assert set(targeted["datasets"]) == {"a"} + validate_hdf5_index(targeted, url, dataset="a") + with pytest.raises(ValueError, match="scoped to dataset 'a'"): + validate_hdf5_index(targeted, url, dataset="b") + + store = blosc2.open(url, cache_policy=blosc2.CachePolicy.MEMORY) + assert store._owner.hdf5_index["datasets"]["a"]["allocated"] is None + assert store._owner.hdf5_index["datasets"]["b"]["allocated"] is None + with store["a"] as array: + np.testing.assert_array_equal(array[:], np.arange(12, dtype="i4")) + assert store._owner.hdf5_index["datasets"]["a"]["allocated"] is not None + assert store._owner.hdf5_index["datasets"]["b"]["allocated"] is None + store.close() + def test_hdf5_object_ndarray_reconstruction(): from blosc2.hdf5_source import _from_json_value, _json_value @@ -743,7 +777,7 @@ def test_publish_hdf5_index_skips_carriers_without_a_snapshot(tmp_path): @pytest.mark.parametrize("snapshot", ["new", "legacy", "damaged", "invalid"]) -def test_hdf5_disk_cache_shares_index_between_leaves(tmp_path, monkeypatch, snapshot): +def test_hdf5_disk_cache_scopes_index_per_leaf(tmp_path, monkeypatch, snapshot): import blosc2.hdf5_source as hdf5_source data = np.arange(40, dtype=np.int32) @@ -770,14 +804,15 @@ def counting_scan(*args, **kwargs): shared.write_bytes(blosc2.compress(json.dumps({"format": "bad"}).encode(), typesize=1)) with blosc2.open(url + "::b", cache_dir=tmp_path) as sibling: np.testing.assert_array_equal(sibling[:], data + 1) - rescan = snapshot in {"damaged", "invalid"} - assert len(scans) == (2 if rescan else 1) + # Dataset-targeted snapshots deliberately omit siblings, so opening b scans + # b directly even when a's snapshot is intact. + assert len(scans) == 2 assert "b" in json.loads(blosc2.decompress(shared.read_bytes()))["datasets"] # Different access configurations must not share a container snapshot. with blosc2.open(url + "::b", cache_dir=tmp_path, storage_options={"skip_instance_cache": True}): pass - assert len(scans) == (3 if rescan else 2) + assert len(scans) == 3 def test_hdf5_blosc2_filter_decodes_super_chunk(): @@ -798,20 +833,40 @@ def test_hdf5_traffic_accounting(): url = make_memory_h5("traffic.h5", data=(data, (20,))) proxy = blosc2.open(url, lazy=True, dataset="data") - # Initial traffic should only be metadata scanning + # Small sources are retained in one request and serve later chunks locally. initial_traffic = proxy.traffic.nbytes assert initial_traffic > 0 + assert proxy.traffic.requests == 1 # Cold chunk fetch _ = proxy[20:40] after_chunk = proxy.traffic.nbytes - assert after_chunk > initial_traffic + assert after_chunk == initial_traffic # Warm chunk hit _ = proxy[20:40] assert proxy.traffic.nbytes == after_chunk +def test_remote_hdf5_json_sidecar(monkeypatch): + import blosc2.hdf5_source as hdf5_source + + data = np.arange(20, dtype=np.int32) + url = make_memory_h5("remote-sidecar.h5", data=(data, (5,))) + sidecar = "memory://remote-sidecar.json" + fsspec.filesystem("memory").pipe(sidecar, json.dumps(blosc2.scan_hdf5_index(url)).encode()) + monkeypatch.setattr( + hdf5_source, + "scan_hdf5_index", + lambda *args, **kwargs: pytest.fail("explicit sidecar must skip source discovery"), + ) + with blosc2.open(url + "::data", hdf5_index=sidecar) as array: + np.testing.assert_array_equal(array[:], data) + with blosc2.RemoteStore(url, hdf5_index=sidecar) as store: + with store["data"] as array: + np.testing.assert_array_equal(array[:], data) + + # --------------------------------------------------------------------------- # Persistence tests # --------------------------------------------------------------------------- diff --git a/tests/test_remote_store.py b/tests/test_remote_store.py index 63e858949..41230c90b 100644 --- a/tests/test_remote_store.py +++ b/tests/test_remote_store.py @@ -890,7 +890,10 @@ def forbidden(*args, **kwargs): np.testing.assert_array_equal(leaf[:2, :3], data[:2, :3]) before = root.traffic.nbytes np.testing.assert_array_equal(alias[:2, :3], data[:2, :3]) - assert root.traffic.nbytes > before # NONE fetches again. + if owner.format == "hdf5": + assert root.traffic.nbytes == before # Retained small-file bytes serve later reads. + else: + assert root.traffic.nbytes > before # NONE fetches again. assert leaf.cache is None assert leaf.cache_bytes == root.cache_bytes == 0 np.testing.assert_array_equal((leaf + 2)[:], data + 2) From ae1bb0a394d8b1df665117808ad894f93226217b Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Tue, 22 Sep 2026 13:34:59 +0200 Subject: [PATCH 59/82] Persist small remote HDF5 sources across sessions --- bench/remote_hdf5_source_cache.py | 114 ++++++++ doc/development/index.rst | 1 + doc/development/remote_cache_design.md | 229 +++++++++++++++ doc/guides/remote_arrays.md | 16 +- plans/remote-hdf5-opts2.md | 377 +++++++++++++++++++++++++ src/blosc2/hdf5_source.py | 164 ++++++++++- src/blosc2/remote_array.py | 39 +++ src/blosc2/remote_ctable.py | 1 + src/blosc2/remote_store.py | 58 +++- src/blosc2/remote_store_cache.py | 29 +- tests/ctable/test_remote_ctable.py | 29 ++ tests/test_fsspec.py | 38 +++ tests/test_hdf5_source.py | 193 +++++++++++++ tests/test_remote_store.py | 2 +- 14 files changed, 1264 insertions(+), 26 deletions(-) create mode 100644 bench/remote_hdf5_source_cache.py create mode 100644 doc/development/remote_cache_design.md create mode 100644 plans/remote-hdf5-opts2.md diff --git a/bench/remote_hdf5_source_cache.py b/bench/remote_hdf5_source_cache.py new file mode 100644 index 000000000..933abd28e --- /dev/null +++ b/bench/remote_hdf5_source_cache.py @@ -0,0 +1,114 @@ +"""Compare metadata-only and persisted-source HDF5 caches across fresh processes. + +Run with the blosc2 environment, e.g.: + conda run -n blosc2 python bench/remote_hdf5_source_cache.py --repeats 3 +""" + +import argparse +import json +import statistics +import subprocess +import sys +import tempfile +import time + + +def worker(args): + import blosc2 + import blosc2.hdf5_source as hdf5_source + + if args.mode == "metadata-only": + # Reproduce the previous cache policy while keeping discovery identical. + hdf5_source.publish_hdf5_source_cache = lambda *a, **kw: None + start = time.perf_counter() + with blosc2.open(args.url, cache_dir=args.cache) as table: + opened = time.perf_counter() + repr(table.info if args.operation == "info" else table) + rendered = time.perf_counter() + result = { + "open_s": opened - start, + "render_s": rendered - opened, + "total_s": rendered - start, + "requests": table.traffic.requests, + "bytes": table.traffic.nbytes, + } + print(json.dumps(result)) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--repeats", type=int, default=3) + parser.add_argument("--worker", action="store_true") + parser.add_argument("--mode", choices=("metadata-only", "persisted")) + parser.add_argument("--cache") + parser.add_argument("--operation", choices=("info", "preview")) + parser.add_argument("--url") + args = parser.parse_args() + if args.worker: + worker(args) + return + if args.repeats < 1: + parser.error("--repeats must be positive") + urls = ( + [args.url] + if args.url + else [ + f"https://f001.backblazeb2.com/file/blosc2/{name}.h5::readings" + for name in ("pt-readings", "pt-readings-idx") + ] + ) + for url in urls: + samples = {} + for repeat in range(args.repeats): + # Alternate order to reduce systematic network warm-up bias. + modes = ("metadata-only", "persisted") if repeat % 2 == 0 else ("persisted", "metadata-only") + for mode in modes: + with tempfile.TemporaryDirectory(prefix="blosc2-hdf5-bench-") as cache: + for phase, operation in ( + ("cold-info", "info"), + ("next-preview", "preview"), + ("warm-preview", "preview"), + ): + start = time.perf_counter() + result = subprocess.run( + [ + sys.executable, + __file__, + "--worker", + "--mode", + mode, + "--cache", + cache, + "--operation", + operation, + "--url", + url, + ], + check=True, + capture_output=True, + text=True, + ) + elapsed = time.perf_counter() - start + sample = json.loads(result.stdout) + sample["process_s"] = elapsed + samples.setdefault((mode, phase), []).append(sample) + for (mode, phase), values in samples.items(): + print( + json.dumps( + { + "url": url, + "mode": mode, + "phase": phase, + "repeats": args.repeats, + **{ + key: round(statistics.median(value[key] for value in values), 4) + for key in values[0] + }, + } + ), + flush=True, + ) + + +if __name__ == "__main__": + main() diff --git a/doc/development/index.rst b/doc/development/index.rst index 10395c3db..3e3553a08 100644 --- a/doc/development/index.rst +++ b/doc/development/index.rst @@ -7,3 +7,4 @@ Development contributing code-of-conduct roadmap + remote_cache_design diff --git a/doc/development/remote_cache_design.md b/doc/development/remote_cache_design.md new file mode 100644 index 000000000..35884a030 --- /dev/null +++ b/doc/development/remote_cache_design.md @@ -0,0 +1,229 @@ +# Remote cache design + +This note describes the implemented cache model for remote native Blosc2 arrays, +B2Z containers, HDF5 files, and Zarr stores. For usage, see +{doc}`../guides/remote_objects`, {doc}`../guides/remote_arrays`, and +{doc}`../guides/remote_tables`. Cache layouts and metadata fields described here +are private implementation details, not portable format contracts. + +## Shared model + +Opening a remote object discovers enough metadata to describe it. Payload reads +are deferred until indexing or computation needs them, except where bounded +prefetch also retrieves payload. Three kinds of cached information must be +distinguished: + +| Layer | Purpose | Current implementation | +| --- | --- | --- | +| Discovery metadata | Reconstruct readers, locate data, and navigate containers | Array carriers and store manifests; format-specific headers, indexes, or metadata objects | +| Compressed payload | Reuse fetched or converted data | Native Blosc2 chunks/blocks, or chunks converted from HDF5/Zarr | +| Complete source object | Reuse original bytes across dataset scopes | Small HDF5 files only | + +A warm metadata cache does not imply a warm payload cache. Reopening may avoid +discovery requests but still fetch data for a preview. HDF5's complete-source +cache bridges that gap for small files; the other routes do not provide a +general persistent copy of the remote source. + +### Identity and ownership + +Managed cache paths distinguish source URLs and non-reversible +`storage_options` fingerprints. Dataset-scoped entries also distinguish the +selected leaf or subtree. Source stamps and recorded geometry determine whether +existing payload can be reused; their inputs differ by format. A stamp is not +a universal content checksum or an automatic remote freshness check. + +A standalone `RemoteArray` owns its cache carrier. A `RemoteStore` discovery +owner shares transport, metadata, traffic, and a cache coordinator across its +leaves and table columns. Store disk caches retain a manifest and payload under +an active generation. Ordinary store disk caches have exclusive ownership; +the separate sparse shared-cache path uses operation-scoped locking. +Dependent handles keep the owner alive until its last user closes. + +### Policies, accounting, and persistence + +`MEMORY` retains payload in RAM; `DISK` retains it across sessions; `NONE` does +not retain payload between operations. NONE does not mean that readers discard +all discovery state or temporary buffers. Array disk caching supports an +explicit `cache_path` or a managed `cache_dir`; stores use `cache_dir`. + +`max_cache_bytes` bounds retained compressed payload, with LRU enforcement after +operations. Store-owned readers share the owner's allowance. The default bound +is 256 MiB; DISK also accepts `None` for unbounded retention. This is not a limit +on decompressed results, temporary conversion buffers, discovery metadata, or +total filesystem usage. The HDF5 source-copy exclusion is described below. + +`Traffic` measures transport activity at the reader's instrumentation points, +not local cache reads. Discovery and payload reads share an owner's counter. +It is not a packet-level HTTP trace: backend metadata probes, retries, and +batched operations need not map one-to-one to its request count. + +Portable reference exports are distinct from disposable runtime caches. They +record source locators and optionally warm payload, not a promise that all source +data is local. Immutable references do not grow an on-disk cache; mutable +references use a writable runtime cache without modifying the original archive. + +### Freshness and refresh + +Remote reads assume immutable sources by default. For standalone `.b2nd` and +Caterva2 sources, `assume_immutable=False` enables identity checking and cache +invalidation. The B2Z, HDF5, and Zarr adapters do not support that mutable-source +mode; publishing new data at a new URL remains the simplest safe contract. + +Explicit table/store refresh prepares new discovery before replacing the active +generation and invalidating derived caches. Failure during preparation leaves +the previous generation usable. Existing child handles of a refreshed store +become stale and must be reacquired. Other independently opened scopes are not +automatically refreshed; HDF5's shared-source version check adds the specific +reopen behavior described below. Immutable reference snapshots cannot refresh. + +## Native Blosc2 arrays and B2Z containers + +### Standalone `.b2nd` + +`FsspecNDSource` uses the native frame reader in `proxy_source.py`. Opening reads +the frame header; chunk offsets and payload ranges are obtained as needed. +Native compressed chunks and, where supported, individual blocks can be fetched +without decoding an entire array or translating it to another storage format. +The cache retains that compressed payload, not a copy of the source file. + +Source identity uses transport information, including HTTP validators when +available. Reopening a disk carrier does not guarantee zero bootstrap traffic +for this route. The immutable-source policy avoids repeated identity checks +before each data operation; it is not the HDF5 whole-file persistence mechanism. + +### `.b2z` archives + +`B2ZArchive` discovers ZIP members through bounded byte ranges. Lazy NDArray +reads address external `ZIP_STORED` members and translate native frame offsets +to archive offsets. Table/store discovery shares the archive reader and metadata +across leaves, rather than opening each member as an unrelated remote file. + +Persisted archive metadata records object identity and captured discovery +ranges. Leaf carriers can also store a `b2z-frame` bootstrap seed, allowing a +reader to be reconstructed without fetching its header again. A populated +table/store cache trusts the saved archive identity on reopen; older caches +may need an identity lookup when upgrading their metadata. + +Bounded opening prefetch can happen to contain a complete small member. Its +payload is transferred to the normal chunk cache, while the persistent leaf +bootstrap keeps the metadata it needs. This is not an archive-wide source cache: +uncached members or chunks still require remote reads. Replacing an archive at +the same URL requires explicit refresh or cache replacement. + +## Zarr stores + +`ZarrNDSource` delegates Zarr v2/v3 metadata and codec handling to Zarr. A +counting store wrapper retains metadata objects such as `.zarray`, `.zgroup`, +`.zattrs`, `.zmetadata`, and `zarr.json`, including remembered missing metadata +keys. Standalone carriers can restore this metadata; store discovery shares and +persists it in its manifest. Resolving a selected array directly avoids an +unnecessary parent listing, but unseen hierarchy paths can still require reads. + +On a payload miss, Zarr reads and decodes the required source chunk data. The +adapter converts the resulting values into a Blosc2-compressed logical chunk, +which enters the ordinary payload cache. Repeated reads can therefore bypass +both remote transfer and Zarr decoding. The persisted payload is not a mirror +of the original Zarr objects. Sharded layouts and codec details remain Zarr's +responsibility, so a logical chunk read need not equal one whole-object GET. + +The source stamp includes location, geometry, dtype, conversion layout, and the +storage-options fingerprint. It does not detect changed values under unchanged +metadata. Source immutability and explicit refresh/cache replacement are thus +essential. There is no whole-store download or shared raw-source cache, and +peak memory can exceed the retained-payload budget during decode/conversion. + +## HDF5 files + +The native index and source cache are Blosc2 mechanisms, not HDF5 conventions +or fsspec cache formats. + +### Discovery and reads + +Remote discovery uses h5py to record dataset metadata and allocated chunk byte +ranges. Selecting a dataset avoids traversing unrelated siblings; PyTables index +groups are discovered lazily. Opening a hierarchy discovers its nodes, but +allocated-chunk maps are deferred until a leaf is read. + +A standalone `RemoteArray` stores its native index in the carrier's +`schunk.vlmeta["hdf5-index"]`. A `RemoteStore` shares discovery metadata across +its leaves and persists it in a manifest. Supported filter pipelines decode +fsspec range reads directly; other pipelines use a retained h5py reader. + +For objects up to 8 MiB, discovery fetches the complete file once: one bounded +transfer avoids many latency-bound metadata reads. Retaining these bytes lets +later reads and PyTables index conversion reuse the same download. + +### Shared complete-source cache + +The complete source bytes, discovery metadata, and converted Blosc2 chunks have +different lifetimes. Source bytes are shared across dataset scopes; metadata and +converted payload remain in their existing array/store caches. + +With an explicit `cache_dir` and `CachePolicy.DISK`, source bytes persist under +`cache_dir/hdf5-sources/`. The source identity reuses `fsspec_cache_path()`: +normalized base URL plus the non-reversible `storage_options` fingerprint, +without the dataset scope. Different datasets in one file share a single copy; +different cache directories, URLs, or credential fingerprints remain isolated. + +Each source has a `.hdf5-source` file, a JSON marker, and a publication lock. +The marker records its schema version, byte size, SHA-256, and version token. +These names and fields are private implementation details. + +Source copies do not count against `max_cache_bytes`. The 8 MiB ceiling is per +file; aggregate source-cache disk usage is unbounded. They survive individual +dataset-generation cleanup. Close active cache users before removing +`cache_dir/hdf5-sources/` to clear source copies, or the whole cache directory to +reset all caches. There is no automatic source eviction or freshness check. + +MEMORY/NONE policies, `cache_path`-only opens, and portable references without +an explicit shared cache root retain their previous behavior. B2Z and Zarr do +not use this source cache. + +### Publication and reuse + +Normal carrier/manifest initialization precedes source publication. Existing +atomic-write helpers publish the blob and then its marker. A short source-level +file lock serializes version checks and publication, not downloads. No +store/array lock is acquired while holding the source lock. A stale opener must +retry rather than overwrite a newer refreshed source. + +Readers validate marker schema, file size, bounded reads, and SHA-256. Missing, +corrupt, or interrupted file/marker pairs are cache misses, so ordinary remote +reads remain available. Local source reads do not contribute to `Traffic`. +Valid bytes can also supply discovery for a new dataset scope without a remote +size lookup or download. + +### Version compatibility and refresh + +Generated indexes record `source_sha256` when complete bytes are available. +An index incompatible with the shared source must be rebuilt, and its derived +payload invalidated. Generated legacy indexes without a digest are rebound by +rebuilding once when a shared source becomes available. + +Table/store refresh bypasses the retained bytes and prepares fresh discovery. +A preparation failure leaves the old live generation usable. Successful refresh +publishes replacement metadata and source bytes; other scopes check compatibility +on their next open. Existing handles may retain their old immutable snapshot. + +If the refreshed file exceeds 8 MiB, publication replaces the old marker with a +new version token and null checksum, then removes the old blob. Generated indexes +record this token as `source_cache_version`, so sibling scopes invalidate old +metadata and payload even though there is no replacement cached blob. + +Explicit `hdf5_index=` never causes a complete download just to populate this +cache. It may reuse verified cached bytes; a recorded digest mismatch is an +error requiring a regenerated sidecar, not silent replacement of the supplied +index. Legacy explicit indexes without a digest retain the immutable-URL trust +contract: matching sizes alone cannot prove version compatibility. + +## Boundaries and future work + +The common model does not imply identical bootstrap costs or invalidation +mechanisms across formats. In particular, the 8 MiB complete-source threshold, +checksum marker, and cross-scope source-version reconciliation are HDF5-only. + +A persistent small-archive source cache for B2Z remains deferred pending +measurements. Zarr would require an object-level policy rather than copying +HDF5's single-file design. Automatic freshness checks for containers, aggregate +source-cache eviction, and a global disk-space budget are not implemented by +this design. diff --git a/doc/guides/remote_arrays.md b/doc/guides/remote_arrays.md index d40e4bb9b..7c08d5bdd 100644 --- a/doc/guides/remote_arrays.md +++ b/doc/guides/remote_arrays.md @@ -68,15 +68,17 @@ Converted Blosc2 chunks are cached under an immutable source contract, so publis Remote HDF5 needs `pip install "blosc2[hdf5,fsspec]"`. HTTP/HTTPS works directly; cloud stores require their protocol driver (`s3fs` for S3, etc.). Datasets can be specified using standard slash syntax (`file.h5/d0/d1/a2`), the double-colon separator (`file.h5::d0/d1/a2`), or the `dataset="d0/d1/a2"` parameter. -Remote pre-indexing uses h5py to record dataset metadata and allocated chunk byte ranges. When opening a single {ref}`RemoteArray`, the native index is cached inside the array carrier (`schunk.vlmeta["hdf5-index"]`). When using {ref}`RemoteStore`, indexing is performed once for the entire container and shared across all leaves and sessions. Uncompressed, deflate, shuffle, and Blosc2 pipelines are decoded directly after fsspec range reads. Other pipelines use a retained h5py reader, including filters registered by `hdf5plugin`. +Remote HDF5 reads cache discovery metadata and fetch dataset chunks as needed. +Uncompressed, deflate, shuffle, and Blosc2 pipelines are supported directly; +other pipelines use h5py, including filters registered by `hdf5plugin`. Use `blosc2.available_datasets(url)` to inspect datasets in an HDF5 container. -An explicitly selected dataset is discovered directly; unrelated siblings and -PyTables index groups are deferred. Complete hierarchy opens discover all nodes, -but allocated-chunk maps are built only when a leaf is first read. Remote HDF5 -objects up to 8 MiB are fetched once and retained for the source session, because -one bounded transfer is cheaper than many metadata ranges. A warm disk cache -restores discovery metadata without downloading the complete source again. +Small remote HDF5 files (up to 8 MiB) are downloaded once during discovery. +With `cache_dir=` and `CachePolicy.DISK`, that copy is shared across datasets +and later processes, so a subsequent table preview needs no further download. +These source copies are excluded from `max_cache_bytes`, which is not a total +disk-space limit. For implementation details and cleanup, see +{doc}`../development/remote_cache_design`. For published immutable data, a native index can be generated once and served as an explicit JSON sidecar. The sidecar is Blosc2 metadata; the HDF5 file is not diff --git a/plans/remote-hdf5-opts2.md b/plans/remote-hdf5-opts2.md new file mode 100644 index 000000000..c56d510ee --- /dev/null +++ b/plans/remote-hdf5-opts2.md @@ -0,0 +1,377 @@ +# Persist Small Remote HDF5 Source Objects + +Status: implemented on 2026-09-22; measured results below. + +## Objective + +When a remote HDF5 object is at most 8 MiB, discovery already downloads the +complete object in one request and retains it in memory. With ``cache_dir=`` the +generated HDF5 index and converted Blosc2 chunks survive process exit, but the +downloaded source bytes do not. A later process can therefore restore metadata +without network traffic and still need remote range requests for table previews +or uncached dataset chunks. + +Persist that already-downloaded small HDF5 object once per source URL so all +dataset-scoped caches and later processes can reuse it. + +The motivating sequence currently behaves as follows: + +| Operation | Requests | Bytes | +| --- | ---: | ---: | +| Cold ``print(t.info)`` | 1 | complete 886 KiB HDF5 file | +| Next-process ``print(t)`` | 2 | about 255 KiB of head/tail data | + +After this change, the second operation should issue no source requests. + +## Scope + +Implement this for remote HDF5 sources using ``CachePolicy.DISK`` with an +explicit ``cache_dir=``. Cover ``RemoteArray``, +``RemoteCTable``, ``RemoteStore``, and ``blosc2.open()``. + +Use that explicit directory as the shared source-cache root. Store-owned leaves +inherit their owner's root. A standalone ``cache_path=`` or a restored reference +artifact with only an internal runtime cache has no explicit shared root: keep +its current behavior in this first version. Do not infer a root from a carrier's +parent directory or persist machine-local source-cache paths in portable +artifacts. Restored references may participate when their supported opening API +explicitly supplies a runtime ``cache_dir``; otherwise defer this integration. + +Do not change MEMORY or NONE cache behavior. Do not fetch a complete object when +an explicit ``hdf5_index=`` avoids source discovery, and do not expand the +existing 8 MiB threshold. + +B2Z may reuse the source-object mechanism later if benchmarks justify complete +prefetch for small archives. Zarr remains outside this design because it is a +multi-object store rather than one source object. + +## Cache model + +Keep three cache categories separate: + +1. **Source object**: the immutable original HDF5 bytes, bounded by the existing + 8 MiB bootstrap threshold and shared across dataset scopes. +2. **Discovery metadata**: native HDF5 indexes, hierarchy manifests, and + PyTables index descriptions. +3. **Converted payload**: cached Blosc2 chunks and generated PyTables index + sidecars, governed by ``max_cache_bytes``. + +The source object is not a converted payload and must not be stored as a +``.b2nd`` leaf. It is cache infrastructure like the discovery manifest. It does +not count against ``max_cache_bytes``. The 8 MiB ceiling applies to each source +object, not the total cache directory: many distinct sources can accumulate +unbounded disk usage. Document this accounting policy and manual cleanup +explicitly; ``max_cache_bytes`` is not a total disk-space guarantee. + +## Source-level identity and layout + +Reuse ``fsspec_cache_path()`` so identity is based on the normalized base URL +and the non-reversible ``storage_options`` fingerprint. Do not include the HDF5 +dataset scope. + +For example: + +```text +cache_dir/ +├── hdf5-sources/ +│ └── pt-readings.h5--/ +│ ├── pt-readings.h5.hdf5-source +│ ├── pt-readings.h5.hdf5-source.json +│ └── pt-readings.h5.hdf5-source.lock +└── ... existing array and dataset-scoped store caches ... +``` + +The dedicated namespace avoids collisions with existing dataset-scoped cache +directories; their layout does not change. + +The exact suffix is private. The important invariant is that these calls resolve +to the same source-object path: + +```python +blosc2.open(url + "::readings", cache_dir=cache_dir) +blosc2.open(url + "::stations", cache_dir=cache_dir) +blosc2.open(url, cache_dir=cache_dir) +``` + +Calls using another URL, storage-options fingerprint, or cache directory remain +isolated. Repeating an identical call reuses the same file. + +Store a small JSON marker beside the source object: + +```json +{ + "version": 1, + "size": 906738, + "token": "...same SHA-256 digest...", + "sha256": "..." +} +``` + +The URL and credentials fingerprint are already part of the parent-directory +identity and need not be duplicated in the marker. + +## Publication and validation + +Use the existing ``atomic_write()`` helper for both files: + +1. Compute the digest from the downloaded bytes and attach it to generated + discovery metadata before that metadata is persisted. +2. Initialize the normal carrier/store cache successfully. +3. Write the source object atomically, then write its marker atomically. + +The source artifact is optional, so carriers/manifests may be published before +it. A crash between those steps leaves an ordinary metadata/chunk cache that +can still use remote reads. Use this ordering consistently for arrays and stores. + +Concurrent dataset opens may both download the same immutable source and race to +publish it. Each replacement is atomic, but the two files are not one atomic +transaction. Readers must treat missing or mismatched pairs as cache misses. +Publication reuses the existing platform file-lock helper for a short +source-specific lock around the version check and replacement. It holds no lock +during downloads, and acquires no store/array locks while holding this lock. +An opener whose expected version has changed fails with a retry message instead +of overwriting a newer refresh. Readers still validate the two-file pair. + +On reuse: + +1. Read and validate the marker schema. +2. Validate that the marker size is a positive integer at most 8 MiB, and check + the opened file's size before allocating its contents. Bound the read as well + so a concurrent file change cannot bypass the limit. +3. Require the actual size to match the marker, and the index ``size`` when an + index is available. A new dataset scope may have no index yet. +4. Verify SHA-256 and check index/source compatibility as specified below. +5. Pass the verified bytes as ``_blob``/``_hdf5_blob``. + +An absent, truncated, corrupt, or unrecognized source cache is disposable. Ignore +it and retain the existing metadata-index plus remote-range fallback; do not make +a valid discovery cache unusable. A later complete bootstrap may replace the bad +source artifact. + +A valid artifact whose digest disagrees with a generated index is a different +case: it signals inconsistent source versions, not simple artifact corruption. +Never use that index's byte offsets against either the replacement blob or +remote ranges without rebuilding discovery and invalidating its derived caches. + +Do not count reads from the local source-object cache in ``Traffic``. Traffic +continues to represent remote transport only. + +## Integration + +### Shared helpers + +Add minimal private helpers near the existing HDF5 cache/bootstrap code: + +- Resolve the source-object and marker paths from base URL, ``cache_dir``, and + ``storage_options``. +- Load and validate a cached object, returning ``bytes`` or ``None``. +- Atomically publish a newly fetched object. + +Extend the internal ``scan_hdf5_index()`` path to accept an already validated +``_blob``. In that case derive the source size from ``len(_blob)`` and open h5py +over those bytes without calling ``filesystem.info()`` or ``cat_file()``. This +is required when a new dataset scope has no cached index yet but the shared +source object is already present. + +Reuse ``fsspec_cache_path()``, ``atomic_write()``, ``hashlib``, and ``json``. +Do not introduce a general cache class or a new dependency. + +### RemoteArray path + +``RemoteArray`` already computes a source-level shared HDF5 index path before +opening the source. Extend that setup for ``CachePolicy.DISK``: + +1. Try the cached source object before constructing ``HDF5NDSource``, including + when an explicit index is supplied. Lookup alone must not fetch the source. +2. Pass valid bytes into ``HDF5NDSource`` and its index scanner so discovery and + reads use local bytes without contacting the remote filesystem. +3. If scanning instead downloads a small object, publish ``src._blob`` after the + disk carrier has been initialized successfully. +4. Persist the source digest with generated indexes and enforce compatibility + before reusing converted chunks. + +The source-object lookup is independent of the selected dataset, whereas the +carrier remains dataset-specific. + +### RemoteStore and RemoteCTable path + +``StoreDiskCache`` is constructed before ``RemoteDiscovery``, but currently the +owner receives ``owner.disk`` only after discovery. Avoid a broad lifecycle +refactor: + +1. Resolve/load the source object from ``cache_dir`` before creating + ``RemoteDiscovery`` and pass it as ``_hdf5_blob``. +2. On a cold discovery, attach the source digest to generated metadata. After + normal cache initialization and manifest publication succeed, publish the + retained ``owner.hdf5_blob`` as an optional source artifact. +3. For the table dispatch path, reuse the blob transferred from the temporary + targeted ``RemoteArray`` instead of downloading or writing it twice. + +This is a post-discovery persistence step, not a new owner responsibility. + +### Explicit indexes + +If ``hdf5_index=`` is supplied and no complete source bytes are fetched, do not +download the HDF5 file merely to populate the source cache. A previously cached +source object may still be used after it passes size and integrity validation. +If the explicit index records a source digest, require it to match; reject a +mismatch with an actionable error asking for a regenerated sidecar. Do not +silently replace a user-supplied index. Legacy explicit sidecars without a +digest retain their existing immutable-URL trust contract; size alone cannot +prove they describe the same source version. + +## Source versions and refresh + +Record an optional ``source_sha256`` in native indexes generated from complete +source bytes, retaining compatibility with existing indexes without that field. +The marker's checksum establishes file integrity; matching the index's digest +establishes that its offsets describe those bytes. Validate the new field when +present and preserve it through carriers, manifests, and JSON round trips. + +``RemoteStore.refresh()`` and ``RemoteCTable.refresh()`` already exist. Integrate +with their current refresh lifecycle in this change: + +- Explicit refresh bypasses the retained source artifact and performs fresh + discovery. Keep the old live generation usable if preparing refresh fails. +- On successful refresh, publish the replacement metadata and source artifact, + invalidating derived column chunks and converted PyTables indexes through the + existing generation replacement mechanism. +- Other dataset scopes compare their recorded digest with the shared artifact + on their next open. On mismatch, rebuild their scoped metadata from the new + bytes and discard their derived caches before serving data. Existing live + handles can retain their old immutable snapshot until refreshed or reopened. +- A pre-existing generated cache without a digest cannot establish compatibility + with a newly shared source artifact. Rebuild its metadata and invalidate its + derived payload once before binding it to the artifact. Keep older caches + working as before when no source artifact is available. + +If refresh produces a file above the threshold, publish a marker with its new +size, a random version token, and null SHA-256, then remove the previous small +artifact. Generated indexes record this token as ``source_cache_version``. +Other scopes compare tokens before reusing metadata/payload; a missing blob +alone is not evidence that old metadata remains valid. + +Concurrent refresh and opens must either observe compatible metadata and bytes +or retry/fail clearly. Atomic writes alone do not establish this compatibility. +New URLs remain the preferred way to publish a new immutable source version. + +## Cache lifecycle + +The source object is shared across dataset-specific cache entries, so it cannot +belong to one dataset generation or be removed by +``discard_old_generations()``. It remains under the source-level directory until +the user removes that source cache or the entire ``cache_dir``. + +This matches the current immutable-URL model. Automatic source-object eviction, +TTL handling, and a global cache-size budget are outside this change. Add them +only if accumulated small sources become a demonstrated problem. + +The explicit refresh behavior above is in scope even though ordinary reads +assume immutable URLs. Automatic remote freshness checks remain outside scope. + +## Tests + +Use the ``blosc2`` conda environment and the existing ranged HTTP test server. + +1. Cold open of a sub-threshold HDF5 source performs one complete GET and writes + the source object plus marker. +2. Run an actual subprocess regression: process 1 formats ``t.info`` and exits; + process 2 formats ``t`` using the same ``cache_dir``. The parent HTTP server + must observe zero requests from process 2, including HEAD requests. Also + verify array slicing and PyTables index discovery with new owners. Use the + active blosc2 environment's interpreter for subprocesses. +3. Different dataset scopes reuse the same source-object path and do not create + duplicate raw files. +4. Dataset-specific carriers/manifests remain distinct and contain the correct + scoped metadata. +5. Files above 8 MiB are not stored as source objects and retain range reads. +6. Explicit ``hdf5_index=`` does not trigger a complete source download. +7. Missing source artifact falls back to remote ranges while retained metadata + remains usable. +8. Truncated data, wrong size, oversized files/markers, malformed marker, unknown + marker version, and SHA-256 corruption all fall back safely. Oversized cached + files must be rejected before a whole-file allocation. +9. Concurrent publication of identical bytes leaves a valid file/marker pair. + Missing or mismatched pairs during publication produce safe cache misses. +10. ``Traffic`` counts the initial remote transfer once and counts no local + source-cache reads. +11. Existing DISK cache reopening, cache limits, source isolation, and cleanup + tests remain unchanged. +12. Replace an HDF5 source with different contents/layout but exactly the same + size, then explicitly refresh. Verify correct rows, index conversion, and + invalidation of another scope's cached metadata and chunks on reopen. +13. Cover refresh failure, interrupted publication, concurrent refresh/open, + and a refresh that crosses above the 8 MiB threshold. +14. Cover one-time migration of generated indexes without digests and explicit + sidecars with matching, mismatched, and absent digests. +15. Verify cache_path-only and artifact-only opens preserve their documented + behavior, and that eligible store-owned leaves share the explicit root. + +Run the focused HDF5/fsspec/remote-table/store suites, Ruff, and the normal +Sphinx build, followed by the default test suite. + +## Acceptance benchmark + +With a new ``cache_dir``: + +```text +process 1: print(t.info) -> one complete-file request +process 2: print(t) -> zero requests +process 3: table slice -> zero requests +``` + +Opening another dataset from the same HDF5 URL must also use zero source +requests once its targeted metadata can be derived from the shared cached file. +Record wall time, request count, and transferred bytes for the motivating +Backblaze files before marking the plan implemented. + +## Deliberate non-goals + +- No public threshold or source-cache switch. +- No complete download for files above 8 MiB. +- No adjacent-sidecar probing. +- No B2Z implementation without measurements. +- No whole-store Zarr cache. +- No shared-cache garbage collector or new lock manager. + +## Implementation results (2026-09-22) + +Implemented for explicit DISK cache directories, reusing existing cache identity, +atomic writes, and platform file locking; no new dependencies or general cache +framework. Also fixed restored HDF5 table nodes to share their canonical index +metadata, so lazy PyTables index discovery remains visible after reopening. + +Reproduce with: + +```sh +conda run -n blosc2 python bench/remote_hdf5_source_cache.py --repeats 3 +``` + +The baseline disables only source-object publication to reproduce the previous +metadata-only cache policy. Each sequence uses a fresh temporary cache and three +separate Python processes; modes alternate order. These are medians of three +sequences per mode against the public Backblaze files, without concurrent tests +or builds. Process times include interpreter startup/imports; operation times +measure opening and rendering only. + +| File / phase | Baseline process | Persisted process | Baseline operation | Persisted operation | Requests before → after | +| --- | ---: | ---: | ---: | ---: | ---: | +| `pt-readings.h5`: cold info | 2.086 s | 2.082 s | 1.936 s | 1.930 s | 1 → 1 | +| next-process preview | 1.427 s | 0.283 s | 1.259 s | 0.138 s | 2 → 0 | +| warm preview | 0.277 s | 0.278 s | 0.129 s | 0.133 s | 0 → 0 | +| `pt-readings-idx.h5`: cold info | 2.321 s | 2.354 s | 2.171 s | 2.208 s | 1 → 1 | +| next-process preview | 1.518 s | 0.293 s | 1.357 s | 0.148 s | 2 → 0 | +| warm preview | 0.284 s | 0.286 s | 0.138 s | 0.141 s | 0 → 0 | + +The next-process preview is **5.0–5.2× faster including startup**, or about +**9.2× faster for open/render**. Both files avoid 261,074 transferred bytes and +two source requests on that preview. Cold info still transfers the complete +906,738-byte or 1,489,409-byte file once; warm previews remain essentially +unchanged. Small cold/warm timing differences are not evidence of a speedup or +regression with only three network runs. + +Validation: default suite **10,427 passed, 36 skipped**; Ruff checks and formatting +pass; Sphinx builds successfully with existing cross-reference warnings. +Regression coverage includes a second process with socket connections disabled, +shared dataset scopes, corruption and interrupted publication, explicit indexes, +refresh failure, stale publishers, same-size replacements, and growth above 8 MiB. diff --git a/src/blosc2/hdf5_source.py b/src/blosc2/hdf5_source.py index 22968a4de..6a8f51624 100644 --- a/src/blosc2/hdf5_source.py +++ b/src/blosc2/hdf5_source.py @@ -18,6 +18,7 @@ import threading import weakref import zlib +from pathlib import Path from urllib.parse import urlsplit import numpy as np @@ -31,6 +32,125 @@ _SMALL_REMOTE_FILE = 8 << 20 +def hdf5_source_cache_path(urlpath, cache_dir, storage_options=None): + """Resolve a source artifact independently of dataset-scoped carriers.""" + return Path( + blosc2.core.fsspec_cache_path( + urlpath, Path(cache_dir) / "hdf5-sources", ".hdf5-source", storage_options=storage_options + ) + ) + + +def _source_marker(path): + try: + with path.with_suffix(path.suffix + ".json").open("rb") as file: + marker = json.loads(file.read(4097)) + size, token = marker["size"], marker["token"] + if marker["version"] != 1 or type(size) is not int or size <= 0: + return None + if not isinstance(token, str) or len(token) != 64 or any(c not in "0123456789abcdef" for c in token): + return None + if marker["sha256"] != (token if size <= _SMALL_REMOTE_FILE else None): + return None + return marker + except (OSError, ValueError, KeyError, TypeError): + return None + + +def load_hdf5_source_cache(path): + """Return verified bytes and version metadata, including large-source tombstones.""" + marker = _source_marker(path) + if marker is None or marker["size"] > _SMALL_REMOTE_FILE: + return None, marker + try: + with path.open("rb") as file: + if os.fstat(file.fileno()).st_size != marker["size"]: + return None, marker + blob = file.read(_SMALL_REMOTE_FILE + 1) + if len(blob) == marker["size"] and hashlib.sha256(blob).hexdigest() == marker["sha256"]: + return blob, marker + except OSError: + pass + return None, marker + + +def hdf5_source_version(index): + return index.get("source_cache_version", index.get("source_sha256")) + + +def reconcile_hdf5_index(index, urlpath, dataset, storage_options, blob, marker, *, explicit): + """Reject stale explicit indexes; rescan disposable generated indexes.""" + if index is None: + return None + index = load_hdf5_index(index, urlpath, storage_options, dataset=dataset) + if marker is None: + return index + version = hdf5_source_version(index) + mismatch = index.get("size") != marker["size"] or version != marker["token"] + if explicit: + if index.get("size") != marker["size"] or (version is not None and mismatch): + raise ValueError("HDF5 source version does not match hdf5_index; regenerate the sidecar") + return index + return None if mismatch else index + + +def publish_hdf5_source_cache(path, blob, index, *, expected=None, refresh=False): + """Publish an optional complete source, or an invalidation marker after refresh.""" + from blosc2.remote_store_cache import lock_cache_file + + # Serialize the version check and replacement against concurrent refresh. + # Readers use checksums; no store/array locks are acquired under this lock. + with path.with_suffix(path.suffix + ".lock").open("a+b") as lock: + lock_cache_file(lock, blocking=True) + _publish_source_cache(path, blob, index, expected, refresh) + + +def _publish_source_cache(path, blob, index, expected, refresh): + from blosc2.remote_store_cache import atomic_write + + current = _source_marker(path) + token = hdf5_source_version(index) + if not refresh and current != expected and (current or {}).get("token") != token: + raise RuntimeError("HDF5 source cache changed during open; retry the operation") + if blob is None and not refresh: + return + size = index["size"] + marker = {"version": 1, "size": size, "token": token, "sha256": index.get("source_sha256")} + if blob is not None: + if len(blob) != size or size > _SMALL_REMOTE_FILE or hashlib.sha256(blob).hexdigest() != token: + raise ValueError("Invalid HDF5 source-cache bytes") + if current == marker: + existing, _ = load_hdf5_source_cache(path) + if existing is not None: + return + atomic_write(path, blob) + atomic_write(path.with_suffix(path.suffix + ".json"), json.dumps(marker).encode()) + if blob is None: + path.unlink(missing_ok=True) + + +def prepare_hdf5_source_cache( + urlpath, dataset, cache_dir, storage_options, index, manifest, blob, *, explicit +): + """Resolve source bytes and discard a stale store manifest before opening leaves.""" + path = hdf5_source_cache_path(urlpath, cache_dir, storage_options) + cached_blob, marker = load_hdf5_source_cache(path) + if cached_blob is not None: + blob = cached_blob + elif blob is not None and marker is not None and hashlib.sha256(blob).hexdigest() != marker["token"]: + blob = None + index = reconcile_hdf5_index(index, urlpath, dataset, storage_options, blob, marker, explicit=explicit) + if ( + manifest is not None + and reconcile_hdf5_index( + manifest["metadata"], urlpath, dataset, storage_options, blob, marker, explicit=False + ) + is None + ): + manifest = None + return path, marker, blob, index, manifest + + def _pytables_table_schema(dtype, shape, boolean_fields=()): """Return a source-bound CTable schema for a supported PyTables Table.""" dtype = np.dtype(dtype) @@ -421,6 +541,7 @@ def scan_hdf5_index( _filesystem=None, _lazy_allocations=False, _return_blob=False, + _blob=None, ): """Build a native byte-range index for a local or remote HDF5 source. @@ -454,10 +575,12 @@ def scan_hdf5_index( check_hdf5_dependencies() fs, path = _filesystem_and_path(urlpath, storage_options, _filesystem) groups, datasets = {"": {"attrs": {}}}, {} - blob = None + blob = _blob try: - size = os.path.getsize(path) if local else int(fs.info(path)["size"]) - if not local and size <= _SMALL_REMOTE_FILE: + size = ( + len(blob) if blob is not None else os.path.getsize(path) if local else int(fs.info(path)["size"]) + ) + if blob is None and not local and size <= _SMALL_REMOTE_FILE: blob = fs.cat_file(path) if len(blob) != size: raise OSError(f"Short HDF5 read: expected {size} bytes, got {len(blob)}") @@ -479,6 +602,8 @@ def scan_hdf5_index( "groups": groups, "datasets": datasets, } + if blob is not None: + index["source_sha256"] = hashlib.sha256(blob).hexdigest() return (index, blob) if _return_blob else index @@ -518,6 +643,12 @@ def validate_hdf5_index(index, urlpath=None, *, dataset=None): ) raise ValueError("Invalid HDF5 index format") version = index.get("version") + for field in ("source_sha256", "source_cache_version"): + value = index.get(field) + if value is not None and ( + not isinstance(value, str) or len(value) != 64 or any(c not in "0123456789abcdef" for c in value) + ): + raise ValueError(f"Invalid HDF5 {field}") if version not in _HDF5_INDEX_VERSIONS: raise ValueError(f"Unsupported HDF5 index version {index.get('version')!r}") if urlpath is not None and index.get("urlpath") != os.fspath(urlpath): @@ -808,6 +939,8 @@ def __init__( _filesystem=None, _blob=None, _ensure_allocations=None, + _source_cache_dir=None, + _index_explicit=True, ): urlpath, dataset = self._parse_url(urlpath, dataset) self.urlpath, self.dataset, self.max_concurrency = urlpath, dataset, max_concurrency @@ -823,6 +956,8 @@ def __init__( self.traffic = _traffic if _traffic is not None else Traffic() if remote else None self._local = (not remote or os.path.isabs(urlpath)) and _filesystem is None self._hdf5_index = None + self._source_cache_path = None + self._source_cache_marker = None try: if self._local: self._open_local() @@ -835,10 +970,30 @@ def __init__( else: check_hdf5_dependencies() self._filesystem, self._path = _filesystem_and_path(urlpath, storage_options, _filesystem) + if _source_cache_dir is not None: + self._source_cache_path = hdf5_source_cache_path( + urlpath, _source_cache_dir, storage_options + ) + self._blob, self._source_cache_marker = load_hdf5_source_cache(self._source_cache_path) + hdf5_index = reconcile_hdf5_index( + hdf5_index, + urlpath, + dataset, + storage_options, + self._blob, + self._source_cache_marker, + explicit=_index_explicit, + ) # No extra session finalizer here: fsspec's HTTP and S3 filesystems # already register one when they create their session, and a second # close makes s3fs/aiobotocore assert on the already-exited client. self._hdf5_index = self._load_or_scan_index(hdf5_index) + if self._source_cache_path is not None and self._blob is not None: + self._hdf5_index = dict(self._hdf5_index) + self._hdf5_index["source_sha256"] = hashlib.sha256(self._blob).hexdigest() + if self._source_cache_marker is not None and self._blob is None: + self._hdf5_index = dict(self._hdf5_index) + self._hdf5_index["source_cache_version"] = self._source_cache_marker["token"] self._validate_dataset_presence(dataset) self._metadata = self._hdf5_index["datasets"][self.dataset] if self._metadata["direct"] and self._metadata["allocated"] is None: @@ -866,6 +1021,8 @@ def __init__( "blocks": self._blocks, "dtype": self._dtype.descr if self._dtype.fields else self._dtype.str, } + if self._hdf5_index is not None and hdf5_source_version(self._hdf5_index) is not None: + identity["source_version"] = hdf5_source_version(self._hdf5_index) self.stamp = hashlib.sha256( json.dumps(identity, sort_keys=True, separators=(",", ":")).encode() ).hexdigest() @@ -922,6 +1079,7 @@ def _load_or_scan_index(self, hdf5_index): traffic=self.traffic, _filesystem=self._filesystem, _return_blob=True, + _blob=self._blob, ) hdf5_index, blob = result if self._blob is None: diff --git a/src/blosc2/remote_array.py b/src/blosc2/remote_array.py index aa3242e3a..6a885e6bb 100644 --- a/src/blosc2/remote_array.py +++ b/src/blosc2/remote_array.py @@ -344,6 +344,8 @@ def _open_url_source( seed=None, blocks=None, cparams=None, + hdf5_cache_dir=None, + hdf5_index_explicit=True, ): if persistable: validate_persistable_url(urlpath) @@ -372,6 +374,8 @@ def _open_url_source( urlpath, dataset, hdf5_index=hdf5_index, + _source_cache_dir=hdf5_cache_dir, + _index_explicit=hdf5_index_explicit, _traffic=traffic, blocks=blocks, cparams=cparams, @@ -620,6 +624,7 @@ def __init__( urlpath, dataset, source_format, hdf5_index ) self._authorized_source = _source_descriptor is not None + hdf5_index_explicit = hdf5_index is not None shared_index_path = None if ( not self._authorized_source @@ -678,6 +683,8 @@ def __init__( seed=seed, blocks=_source_blocks, cparams=_source_cparams, + hdf5_cache_dir=cache_dir if cache_policy is blosc2.CachePolicy.DISK else None, + hdf5_index_explicit=hdf5_index_explicit, ) self._assume_immutable = assume_immutable self._storage_options = storage_options @@ -701,6 +708,8 @@ def __init__( self._initialize_runtime_cache(cache_dir, cache_path, _runtime_cache_path) + self._publish_hdf5_source() + _publish_hdf5_index(shared_index_path, self._carrier, scanned=hdf5_index is None) if self._carrier is not None: @@ -714,6 +723,22 @@ def __init__( self._store_owner = _store_owner self._store_finalizer = weakref.finalize(self, _store_owner.release) + def _publish_hdf5_source(self): + if ( + not self._authorized_source + and isinstance(self.src, blosc2.HDF5NDSource) + and self.src._source_cache_path is not None + ): + from blosc2.hdf5_source import publish_hdf5_source_cache + + _store_hdf5_index(self._carrier, self.src._hdf5_index) + publish_hdf5_source_cache( + self.src._source_cache_path, + self.src._blob, + self.src._hdf5_index, + expected=self.src._source_cache_marker, + ) + def _initialize_runtime_cache(self, cache_dir, cache_path, _runtime_cache_path): if self._store_owner is not None and self.cache_policy is not blosc2.CachePolicy.NONE: self._attach_carrier_cache() @@ -829,6 +854,16 @@ def _open_or_create_carrier(self, cache_dir, cache_path): if stored is not None and current is not None and stored != current else "reused" ) + if ( + status == "invalidated/rebuilt" + and isinstance(self.src, blosc2.HDF5NDSource) + and self.src._source_cache_path is not None + and self._geometry(carrier) != self._geometry(self.src) + ): + del carrier + return self._to_b2object_carrier( + urlpath=path, contiguous=True, mode="w", mutable=True + ), status if status == "reused": if self._cached_meta is None: self._cached_meta = self._meta_from_carrier(carrier) @@ -1114,6 +1149,8 @@ def _open_source( seed=None, blocks=None, cparams=None, + hdf5_cache_dir=None, + hdf5_index_explicit=True, ): if isinstance(urlpath, blosc2.C2Array): if source_format not in {None, "blosc2"}: @@ -1162,6 +1199,8 @@ def _open_source( seed=seed, blocks=blocks, cparams=cparams, + hdf5_cache_dir=hdf5_cache_dir, + hdf5_index_explicit=hdf5_index_explicit, ) else: raise TypeError("RemoteArray requires a URL string, URLPath, or C2Array") diff --git a/src/blosc2/remote_ctable.py b/src/blosc2/remote_ctable.py index 2bbc6a8ee..992687c54 100644 --- a/src/blosc2/remote_ctable.py +++ b/src/blosc2/remote_ctable.py @@ -232,6 +232,7 @@ def refresh(self) -> None: ) replacement.restoring = False replacement.save_manifest() + replacement.publish_hdf5_source(refresh=True) except BaseException: replacement.disk = None if fresh is not None: diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 5b19825e8..c9923dfd7 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -147,6 +147,7 @@ def __init__( self.archive = None self.zstore = None self.hdf5_blob = None + self.hdf5_source_cache_path = None if _hdf5_blob is not None: self.hdf5_blob = _hdf5_blob self.sources = {} @@ -238,6 +239,10 @@ def _restore_manifest(self, manifest): elif self.format == "hdf5": self.hdf5_index = self.metadata self._validate_hdf5_index() + # JSON round trips lose the shared table metadata used by lazy index discovery. + for path, (kind, _) in self.nodes.items(): + if kind == "ctable": + self.nodes[path] = (kind, self.hdf5_index["datasets"][path]) elif self.format == "zarr" and self.zstore is None: self._open_zarr() self._check_node_limit() @@ -469,6 +474,7 @@ def _open_hdf5(self, hdf5_index=None): _filesystem=self.filesystem, _lazy_allocations=True, _return_blob=True, + _blob=self.hdf5_blob, ) else: self.hdf5_index = load_hdf5_index( @@ -511,6 +517,27 @@ def ensure_hdf5_allocations(self, path): self.save_manifest() return metadata + def attach_hdf5_source_cache(self, path, marker): + self.hdf5_source_cache_path = path + self.hdf5_source_cache_marker = marker + if path is not None and self.hdf5_blob is not None: + self.hdf5_index = dict(self.hdf5_index) + self.hdf5_index["source_sha256"] = hashlib.sha256(self.hdf5_blob).hexdigest() + elif marker is not None: + self.hdf5_index["source_cache_version"] = marker["token"] + + def publish_hdf5_source(self, *, refresh=False): + if self.hdf5_source_cache_path is not None: + from blosc2.hdf5_source import publish_hdf5_source_cache + + publish_hdf5_source_cache( + self.hdf5_source_cache_path, + self.hdf5_blob, + self.hdf5_index, + expected=getattr(self, "hdf5_source_cache_marker", None), + refresh=refresh, + ) + def ensure_pytables_indexes(self, table_path): """Discover one table's hidden PyTables index nodes on first use.""" from blosc2.hdf5_source import decode_hdf5_value, scan_pytables_indexes @@ -995,6 +1022,13 @@ def prepare_refresh(self, kind): replacement.mutable = self.mutable replacement.restoring = True replacement.disk = self.disk + replacement.hdf5_source_cache_path = self.hdf5_source_cache_path + if self.hdf5_source_cache_path is not None and replacement.hdf5_blob is None: + # A large replacement has no retained bytes, but must invalidate + # sibling scopes that still describe the previous small source. + replacement.hdf5_index["source_cache_version"] = hashlib.sha256( + uuid.uuid4().bytes + ).hexdigest() return replacement except BaseException: replacement.close() @@ -1360,10 +1394,9 @@ def __init__( _traffic=None, nested_storage_options=None, ): - if isinstance(urlpath, os.PathLike): - urlpath = os.fspath(urlpath) - if not isinstance(urlpath, str): + if not isinstance(urlpath, (str, os.PathLike)): raise TypeError("RemoteStore requires a remote URL string") + urlpath = os.fspath(urlpath) hdf5_index, _source_format = _resolve_hdf5_options(hdf5_index, _hdf5_index, _source_format) artifact = self._try_open_artifact( urlpath, dataset, storage_options, cache_policy, max_cache_bytes, cache_dir, _allow_array_root @@ -1378,6 +1411,7 @@ def __init__( base_url, _, _ = parse_container_url(urlpath, dataset) validate_persistable_url(base_url) disk = None + source_cache_path = source_cache_marker = None manifest = _manifest if cache_policy is blosc2.CachePolicy.DISK: from blosc2.remote_store_cache import StoreDiskCache @@ -1396,6 +1430,21 @@ def __init__( disk = StoreDiskCache(cache_dir, source) try: manifest = disk.load() if disk is not None else manifest + if disk is not None and source["kind"] == "hdf5" and cache_dir is not None: + from blosc2.hdf5_source import prepare_hdf5_source_cache + + source_cache_path, source_cache_marker, _hdf5_blob, hdf5_index, manifest = ( + prepare_hdf5_source_cache( + base_url, + root or None, + cache_dir, + storage_options, + hdf5_index, + manifest, + _hdf5_blob, + explicit=_hdf5_index is None, + ) + ) owner = RemoteDiscovery( urlpath, storage_options, @@ -1412,6 +1461,7 @@ def __init__( _traffic=_traffic, ) manifest = owner.restored_manifest + owner.attach_hdf5_source_cache(source_cache_path, source_cache_marker) except BaseException: if disk is not None: disk.close() @@ -1432,6 +1482,7 @@ def __init__( try: owner.restore_caches(manifest) owner.save_manifest() + owner.publish_hdf5_source() if disk is not None: disk.discard_old_generations(owner.generation) except BaseException: @@ -1920,6 +1971,7 @@ def refresh(self): try: replacement.restoring = False replacement.save_manifest() + replacement.publish_hdf5_source(refresh=True) if replacement.disk is not None: replacement.disk.discard_old_generations(replacement.generation) except BaseException: diff --git a/src/blosc2/remote_store_cache.py b/src/blosc2/remote_store_cache.py index 020502a1b..3b217e77b 100644 --- a/src/blosc2/remote_store_cache.py +++ b/src/blosc2/remote_store_cache.py @@ -25,6 +25,22 @@ def validate_generation(generation): raise ValueError("Invalid RemoteStore generation") +def lock_cache_file(file, *, blocking=False): + """Lock an open cache lockfile until it is closed.""" + if os.name == "nt": + import msvcrt + + if not file.tell(): + file.write(b"\0") + file.flush() + file.seek(0) + msvcrt.locking(file.fileno(), msvcrt.LK_LOCK if blocking else msvcrt.LK_NBLCK, 1) + else: + import fcntl + + fcntl.flock(file.fileno(), fcntl.LOCK_EX | (0 if blocking else fcntl.LOCK_NB)) + + class StoreDiskCache: def __init__(self, parent, source, *, blocking=False): self.source = source @@ -33,18 +49,7 @@ def __init__(self, parent, source, *, blocking=False): self.path.mkdir(parents=True, exist_ok=True) self.file = (self.path / "owner.lock").open("a+b") try: - if os.name == "nt": - import msvcrt - - if not self.file.tell(): - self.file.write(b"\0") - self.file.flush() - self.file.seek(0) - msvcrt.locking(self.file.fileno(), msvcrt.LK_LOCK if blocking else msvcrt.LK_NBLCK, 1) - else: - import fcntl - - fcntl.flock(self.file, fcntl.LOCK_EX | (0 if blocking else fcntl.LOCK_NB)) + lock_cache_file(self.file, blocking=blocking) if (self.path / "manifest.msgpack").exists(): raise ValueError( diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 40d6917bf..3cd443d23 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -164,6 +164,35 @@ def test_small_remote_pytables_file_is_retained(): assert table.traffic.requests == 1 +def test_pytables_source_cache_survives_reopen_and_failed_refresh(tmp_path, monkeypatch): + import blosc2.hdf5_source as hs + + url, data = pytables_hdf5_url("pytables-disk-source.h5", indexed=True) + with blosc2.open(url + "::table", cache_dir=tmp_path) as table: + # Discovery alone must leave the source available for later index conversion. + assert table.traffic.requests == 1 + with blosc2.RemoteCTable(url, dataset="table", cache_dir=tmp_path) as table: + repr(table.info) + assert table._get_index_catalog()["id"]["kind"] == "opsi" + np.testing.assert_array_equal(table.id[:], data["id"]) + assert table.traffic.requests == 0 + path = hs.hdf5_source_cache_path(url, tmp_path) + before = hs.load_hdf5_source_cache(path) + owner = table._storage._owner + generation = owner.generation + + def fail(*args, **kwargs): + raise OSError("refresh failed") + + with monkeypatch.context() as patch: + patch.setattr(hs, "scan_hdf5_index", fail) + with pytest.raises(OSError, match="refresh failed"): + table.refresh() + assert owner.generation == generation + assert hs.load_hdf5_source_cache(path) == before + np.testing.assert_array_equal(table.id[:], data["id"]) + + def test_remote_pytables_fixed_string_full_index(): url, data = pytables_hdf5_url("pytables-string-index.h5", indexed=True, indexed_field="label") with blosc2.RemoteCTable(url, dataset="table") as table: diff --git a/tests/test_fsspec.py b/tests/test_fsspec.py index 1ae7cc8e6..a717809f0 100644 --- a/tests/test_fsspec.py +++ b/tests/test_fsspec.py @@ -13,6 +13,7 @@ import http.server import os import pathlib +import subprocess import sys import threading @@ -751,6 +752,7 @@ def do_GET(self): return None def do_HEAD(self): + self.server.head_requests.append(self.path) body = (root / self.path.lstrip("/")).read_bytes() self.send_response(200) self.send_header("Accept-Ranges", "bytes") @@ -761,6 +763,7 @@ def do_HEAD(self): handler = functools.partial(Ranged, directory=str(root)) server = http.server.ThreadingHTTPServer(("127.0.0.1", 0), handler) server.requests = [] + server.head_requests = [] threading.Thread(target=server.serve_forever, daemon=True).start() try: yield f"http://127.0.0.1:{server.server_address[1]}", server.requests @@ -784,6 +787,41 @@ def test_http_hdf5_scan_and_warm_slice(tmp_path): assert len(requests) == count +def test_http_hdf5_source_cache_across_processes(tmp_path): + h5py = pytest.importorskip("h5py") + data = np.zeros(100, dtype=[("id", "i4"), ("value", "f8")]) + data["id"] = np.arange(len(data)) + path = tmp_path / "table.h5" + with h5py.File(path, "w") as file: + table = file.create_dataset("table", data=data, chunks=(10,)) + table.attrs["CLASS"] = np.bytes_(b"TABLE") + cache = tmp_path / "cache" + with _ranged_server(tmp_path) as (urlbase, requests): + script = ( + "import blosc2, sys; " + "t=blosc2.open(sys.argv[1], cache_dir=sys.argv[2]); " + "print(t.info if sys.argv[3] == 'info' else t); " + "print('requests', t.traffic.requests); t.close()" + ) + url = f"{urlbase}/{path.name}::table" + subprocess.run( + [sys.executable, "-c", script, url, str(cache), "info"], check=True, capture_output=True + ) + assert requests == [None] + requests.clear() + # Reject all HTTP activity, including metadata HEADs, in the second process. + offline = "import socket; socket.socket.connect=lambda *a, **k: (_ for _ in ()).throw(RuntimeError('network')); " + result = subprocess.run( + [sys.executable, "-c", offline + script, url, str(cache), "table"], + check=True, + capture_output=True, + text=True, + ) + assert "requests 0" in result.stdout + assert "100 rows" in result.stdout + assert requests == [] + + def test_http_large_hdf5_keeps_range_reads(tmp_path): h5py = pytest.importorskip("h5py") path = tmp_path / "large-seekable.h5" diff --git a/tests/test_hdf5_source.py b/tests/test_hdf5_source.py index f4dbdd26e..ddc6673d5 100644 --- a/tests/test_hdf5_source.py +++ b/tests/test_hdf5_source.py @@ -73,6 +73,199 @@ def test_hdf5_native_index(): assert validate_hdf5_index(legacy) is legacy +def test_shared_source_cache_scopes_and_explicit_index(tmp_path, monkeypatch): + import blosc2.hdf5_source as hs + + data = np.arange(24, dtype="i4") + url = make_memory_h5("shared-source.h5", a=(data, (4,)), b=(data + 1, (4,))) + with blosc2.open(url + "::a", cache_dir=tmp_path) as array: + index = array.src._hdf5_index + assert array.traffic.requests == 1 + path = hs.hdf5_source_cache_path(url, tmp_path) + assert path.exists() + + def forbidden(*args, **kwargs): + pytest.fail("warm source must not contact the filesystem") + + monkeypatch.setattr(fsspec.implementations.memory.MemoryFileSystem, "info", forbidden) + monkeypatch.setattr(fsspec.implementations.memory.MemoryFileSystem, "cat_file", forbidden) + for target in ("a", "b"): + with blosc2.open(url + "::" + target, cache_dir=tmp_path) as array: + np.testing.assert_array_equal(array[:], data + (target == "b")) + assert array.traffic.requests == 0 + with blosc2.RemoteStore(url, cache_dir=tmp_path) as store: + with store["b"] as array: + np.testing.assert_array_equal(array[:], data + 1) + assert store.traffic.requests == 0 + with blosc2.open(url + "::a", cache_dir=tmp_path, hdf5_index=index) as array: + np.testing.assert_array_equal(array[:], data) + assert array.traffic.requests == 0 + assert len(list((tmp_path / "hdf5-sources").rglob("*.hdf5-source"))) == 1 + bad = dict(index, source_sha256="0" * 64) + with pytest.raises(ValueError, match="regenerate the sidecar"): + blosc2.open(url + "::a", cache_dir=tmp_path, hdf5_index=bad) + + +@pytest.mark.parametrize("damage", ["missing", "truncated", "digest", "json", "version", "oversize"]) +def test_source_cache_damage_falls_back(tmp_path, damage): + import blosc2.hdf5_source as hs + + data = np.arange(24, dtype="i4") + url = make_memory_h5("source-damage.h5", a=(data, (4,))) + with blosc2.open(url + "::a", cache_dir=tmp_path): + pass + path = hs.hdf5_source_cache_path(url, tmp_path) + marker_path = path.with_suffix(path.suffix + ".json") + if damage == "missing": + path.unlink() + elif damage == "truncated": + path.write_bytes(path.read_bytes()[:100]) + elif damage == "digest": + blob = bytearray(path.read_bytes()) + blob[-1] ^= 1 + path.write_bytes(blob) + elif damage == "json": + marker_path.write_text("invalid") + elif damage == "version": + marker = json.loads(marker_path.read_text()) + marker["version"] = 2 + marker_path.write_text(json.dumps(marker)) + else: + with path.open("ab") as file: + file.truncate((8 << 20) + 1) + assert hs.load_hdf5_source_cache(path)[0] is None + with blosc2.open(url + "::a", cache_dir=tmp_path) as array: + np.testing.assert_array_equal(array[:], data) + assert array.traffic.requests > 0 + + +@pytest.mark.parametrize("large", [False, True]) +def test_source_refresh_invalidates_sibling_scopes(tmp_path, large): + import blosc2.hdf5_source as hs + + data = np.arange(24, dtype="i4") + url = make_memory_h5("source-refresh.h5", a=(data, (4,)), b=(data + 1, (4,))) + fs = fsspec.filesystem("memory") + old_size = fs.info(url)["size"] + with blosc2.open(url + "::a", cache_dir=tmp_path) as array: + np.testing.assert_array_equal(array[:], data) + with blosc2.RemoteStore(url, cache_dir=tmp_path) as store: + with store["b"] as array: + np.testing.assert_array_equal(array[:], data + 1) + extra = {"padding": np.zeros(9 << 20, dtype="u1")} if large else {} + make_memory_h5("source-refresh.h5", a=(data + 100, (4,)), b=(data + 101, (4,)), **extra) + if not large: + assert fs.info(url)["size"] == old_size + store.refresh() + with store["b"] as array: + np.testing.assert_array_equal(array[:], data + 101) + with blosc2.open(url + "::a", cache_dir=tmp_path) as array: + np.testing.assert_array_equal(array[:], data + 100) + with blosc2.open(url + "::a", cache_dir=tmp_path) as array: + np.testing.assert_array_equal(array[:], data + 100) + blob, marker = hs.load_hdf5_source_cache(hs.hdf5_source_cache_path(url, tmp_path)) + assert (blob is None) == large + assert (marker["sha256"] is None) == large + + +def test_source_cache_concurrent_publication(tmp_path): + from concurrent.futures import ThreadPoolExecutor + + import blosc2.hdf5_source as hs + + url = make_memory_h5("concurrent-source.h5", a=(np.arange(12), (4,))) + index, blob = hs.scan_hdf5_index(url, _return_blob=True) + path = hs.hdf5_source_cache_path(url, tmp_path) + with ThreadPoolExecutor(max_workers=4) as pool: + list(pool.map(lambda _: hs.publish_hdf5_source_cache(path, blob, index), range(8))) + assert hs.load_hdf5_source_cache(path)[0] == blob + + +def test_source_cache_refresh_rejects_stale_publisher(tmp_path): + import blosc2.hdf5_source as hs + + url = make_memory_h5("source-publisher.h5", a=(np.arange(12), (4,))) + old_index, old_blob = hs.scan_hdf5_index(url, _return_blob=True) + path = hs.hdf5_source_cache_path(url, tmp_path) + hs.publish_hdf5_source_cache(path, old_blob, old_index) + _, old_marker = hs.load_hdf5_source_cache(path) + make_memory_h5("source-publisher.h5", a=(np.arange(12) + 1, (4,))) + index, blob = hs.scan_hdf5_index(url, _return_blob=True) + hs.publish_hdf5_source_cache(path, blob, index, refresh=True) + with pytest.raises(RuntimeError, match="retry"): + hs.publish_hdf5_source_cache(path, old_blob, old_index, expected=old_marker) + assert hs.load_hdf5_source_cache(path)[0] == blob + + +def test_source_cache_legacy_index_migration_and_geometry_change(tmp_path): + import blosc2.hdf5_source as hs + from blosc2.remote_array import _hdf5_index_from_carrier, _store_hdf5_index + + data = np.arange(24, dtype="i4") + url = make_memory_h5("source-migrate.h5", a=(data, (4,))) + with blosc2.open(url + "::a", cache_dir=tmp_path) as array: + np.testing.assert_array_equal(array[:], data) + carrier_path = blosc2.core.fsspec_cache_path(url, tmp_path, ".b2nd", dataset="a") + carrier = blosc2.blosc2_ext.open(carrier_path, "a", 0) + index = _hdf5_index_from_carrier(carrier) + index.pop("source_sha256") + _store_hdf5_index(carrier, index) + del carrier + with blosc2.open(url + "::a", cache_dir=tmp_path) as array: + assert "source_sha256" in array.src._hdf5_index + assert array.traffic.requests == 0 + np.testing.assert_array_equal(array[:], data) + # Legacy explicit sidecars retain the immutable-URL trust contract. + with blosc2.open(url + "::a", cache_dir=tmp_path, hdf5_index=index) as array: + np.testing.assert_array_equal(array[:], data) + with blosc2.RemoteStore(url, cache_dir=tmp_path) as store: + make_memory_h5("source-migrate.h5", a=(np.arange(40, dtype="i4"), (8,))) + store.refresh() + with blosc2.open(url + "::a", cache_dir=tmp_path) as array: + assert array.shape == (40,) + assert array.chunks == (8,) + np.testing.assert_array_equal(array[:], np.arange(40)) + assert array.traffic.requests == 0 + assert hs.load_hdf5_source_cache(hs.hdf5_source_cache_path(url, tmp_path))[0] is not None + + +def test_source_cache_is_opt_in_via_disk_directory(tmp_path): + data = np.arange(24, dtype="i4") + url = make_memory_h5("source-policy.h5", a=(data, (4,))) + for options in ( + {"cache_policy": blosc2.CachePolicy.NONE}, + {"cache_policy": blosc2.CachePolicy.MEMORY}, + {"cache_path": tmp_path / "array.b2nd", "cache_policy": blosc2.CachePolicy.DISK}, + ): + with blosc2.RemoteArray(url, dataset="a", **options) as array: + np.testing.assert_array_equal(array[:], data) + assert array.src._source_cache_path is None + assert not (tmp_path / "hdf5-sources").exists() + + +def test_source_cache_interrupted_publication(tmp_path, monkeypatch): + import blosc2.hdf5_source as hs + import blosc2.remote_store_cache as cache + + url = make_memory_h5("source-interrupted.h5", a=(np.arange(12), (4,))) + index, blob = hs.scan_hdf5_index(url, _return_blob=True) + path = hs.hdf5_source_cache_path(url, tmp_path) + write = cache.atomic_write + + def fail_marker(target, data): + if target.suffix == ".json": + raise OSError("interrupted marker publication") + write(target, data) + + with monkeypatch.context() as patch: + patch.setattr(cache, "atomic_write", fail_marker) + with pytest.raises(OSError, match="interrupted"): + hs.publish_hdf5_source_cache(path, blob, index) + assert hs.load_hdf5_source_cache(path) == (None, None) + hs.publish_hdf5_source_cache(path, blob, index) + assert hs.load_hdf5_source_cache(path)[0] == blob + + def test_hdf5_targeted_index_scope_and_lazy_allocations(): from blosc2.hdf5_source import scan_hdf5_index, validate_hdf5_index diff --git a/tests/test_remote_store.py b/tests/test_remote_store.py index 41230c90b..5e62319bf 100644 --- a/tests/test_remote_store.py +++ b/tests/test_remote_store.py @@ -219,7 +219,7 @@ def test_disk_cache_partitions_storage_options(hierarchy, tmp_path): np.testing.assert_array_equal(a[:2, :2], data[:2, :2]) # Different backends must not share a manifest or its leaf payloads. - assert len([path for path in cache.iterdir() if path.is_dir()]) == 2 + assert len([path for path in cache.iterdir() if path.is_dir() and path.name != "hdf5-sources"]) == 2 def test_artifact_keeps_storage_options_identity(hierarchy, tmp_path): From 4be2c66d711288370f32195fc4f917b8f962801d Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Tue, 22 Sep 2026 13:50:33 +0200 Subject: [PATCH 60/82] Fix Windows remote HDF5 cache regression test --- tests/test_fsspec.py | 22 +++++++++++++--------- 1 file changed, 13 insertions(+), 9 deletions(-) diff --git a/tests/test_fsspec.py b/tests/test_fsspec.py index a717809f0..912edec80 100644 --- a/tests/test_fsspec.py +++ b/tests/test_fsspec.py @@ -724,7 +724,7 @@ def test_http_stamp_prefers_etag_and_falls_back_to_modified_size(): @contextlib.contextmanager -def _ranged_server(root): +def _ranged_server(root, *, head_requests=None): """A web server over *root* that honours `Range`, which the stock one does not.""" class Ranged(http.server.SimpleHTTPRequestHandler): @@ -763,7 +763,7 @@ def do_HEAD(self): handler = functools.partial(Ranged, directory=str(root)) server = http.server.ThreadingHTTPServer(("127.0.0.1", 0), handler) server.requests = [] - server.head_requests = [] + server.head_requests = [] if head_requests is None else head_requests threading.Thread(target=server.serve_forever, daemon=True).start() try: yield f"http://127.0.0.1:{server.server_address[1]}", server.requests @@ -796,7 +796,8 @@ def test_http_hdf5_source_cache_across_processes(tmp_path): table = file.create_dataset("table", data=data, chunks=(10,)) table.attrs["CLASS"] = np.bytes_(b"TABLE") cache = tmp_path / "cache" - with _ranged_server(tmp_path) as (urlbase, requests): + head_requests = [] + with _ranged_server(tmp_path, head_requests=head_requests) as (urlbase, requests): script = ( "import blosc2, sys; " "t=blosc2.open(sys.argv[1], cache_dir=sys.argv[2]); " @@ -804,22 +805,25 @@ def test_http_hdf5_source_cache_across_processes(tmp_path): "print('requests', t.traffic.requests); t.close()" ) url = f"{urlbase}/{path.name}::table" - subprocess.run( - [sys.executable, "-c", script, url, str(cache), "info"], check=True, capture_output=True + result = subprocess.run( + [sys.executable, "-c", script, url, str(cache), "info"], capture_output=True, text=True ) + assert result.returncode == 0, result.stderr assert requests == [None] requests.clear() - # Reject all HTTP activity, including metadata HEADs, in the second process. - offline = "import socket; socket.socket.connect=lambda *a, **k: (_ for _ in ()).throw(RuntimeError('network')); " + head_requests.clear() + # Observe HTTP requests instead of blocking socket.connect: Windows asyncio + # needs internal loopback connections to create its wakeup socket pair. result = subprocess.run( - [sys.executable, "-c", offline + script, url, str(cache), "table"], - check=True, + [sys.executable, "-c", script, url, str(cache), "table"], capture_output=True, text=True, ) + assert result.returncode == 0, result.stderr assert "requests 0" in result.stdout assert "100 rows" in result.stdout assert requests == [] + assert head_requests == [] def test_http_large_hdf5_keeps_range_reads(tmp_path): From eba810a42c8780bb454f49625f1300fcbf5c8484 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Tue, 22 Sep 2026 17:48:56 +0200 Subject: [PATCH 61/82] Hide unavailable HDF5 table compression statistics --- src/blosc2/ctable.py | 25 ++++++++++++++++--------- src/blosc2/ctable_storage.py | 5 +++++ tests/ctable/test_remote_ctable.py | 20 ++++++++++++++++++++ 3 files changed, 41 insertions(+), 9 deletions(-) diff --git a/src/blosc2/ctable.py b/src/blosc2/ctable.py index aff637738..71896eb99 100644 --- a/src/blosc2/ctable.py +++ b/src/blosc2/ctable.py @@ -54,6 +54,7 @@ InMemoryTableStorage, TableStorage, TreeStoreTableStorage, + column_cbytes_for_info, join_field_path, split_field_path, ) @@ -2056,7 +2057,7 @@ def summary(self) -> str: f" dtype : {self.dtype}", f" storage : NDArray shape={getattr(raw, 'shape', None)}, chunks={getattr(raw, 'chunks', None)}, blocks={getattr(raw, 'blocks', None)}", ] - cbytes = getattr(raw, "cbytes", None) + cbytes = column_cbytes_for_info(raw) if cbytes is not None: lines.append(f" cbytes : {format_nbytes_info(cbytes)}") if rows and self.dtype is not None and self.dtype.kind in "biufc": @@ -2149,8 +2150,8 @@ def info_items(self) -> list[tuple[str, object]]: items.append(("blocks", blocks)) nbytes = getattr(raw, "nbytes", None) - cbytes = getattr(raw, "cbytes", None) - cratio = getattr(raw, "cratio", None) + cbytes = column_cbytes_for_info(raw) + cratio = getattr(raw, "cratio", None) if cbytes is not None else None if nbytes is not None: items.append(("nbytes", format_nbytes_info(nbytes))) if cbytes is not None: @@ -4286,7 +4287,7 @@ def info_items(self) -> list[tuple[str, object]]: dtype_label = table._dtype_info_label( getattr(table._cols[name], "dtype", None), spec ) + table._null_info_tag(spec) - cbytes = getattr(table._cols[name], "cbytes", None) + cbytes = column_cbytes_for_info(table._cols[name]) if cbytes is not None: nbytes = getattr(table._cols[name], "nbytes", None) detail = f"cbytes: {format_nbytes_human(cbytes)}" @@ -4297,6 +4298,11 @@ def info_items(self) -> list[tuple[str, object]]: column_summary[rel_name] = _InfoLiteral(dtype_label) descendant = set(self._descendant_col_names()) + compression_available = all( + column_cbytes_for_info(table._cols[name]) is not None + for name in descendant + if name in table._cols + ) index_summary = {} for idx in table.indexes: if idx.col_name not in descendant: @@ -4316,8 +4322,8 @@ def info_items(self) -> list[tuple[str, object]]: ("storage", storage_type), ("nrows", self.nrows), ("nbytes", format_nbytes_info(self.nbytes)), - ("cbytes", format_nbytes_info(self.cbytes)), - ("cratio", f"{self.cratio:.2f}x"), + ("cbytes", format_nbytes_info(self.cbytes) if compression_available else "n/a"), + ("cratio", f"{self.cratio:.2f}x" if compression_available else "n/a"), ("columns", column_summary), ("indexes", index_summary if index_summary else "none"), ] @@ -15057,7 +15063,7 @@ def info_items(self) -> list[tuple[str, object]]: dtype_label = self._dtype_info_label( getattr(self._cols[name], "dtype", None), spec ) + self._null_info_tag(spec) - cbytes = getattr(self._cols[name], "cbytes", None) + cbytes = column_cbytes_for_info(self._cols[name]) if cbytes is not None: nbytes = getattr(self._cols[name], "nbytes", None) detail = f"cbytes: {format_nbytes_human(cbytes)}" @@ -15079,6 +15085,7 @@ def info_items(self) -> list[tuple[str, object]]: suffix = f"(cbytes: {format_nbytes_human(cbytes)}, cratio: {cratio:.2f}x)" index_summary[idx.col_name] = f"[{idx.kind}{stale}{label}] {suffix}" + compression_available = all(column_cbytes_for_info(col) is not None for col in self._cols.values()) items = [ ("type", self.__class__.__name__), ("storage", storage_type), @@ -15088,8 +15095,8 @@ def info_items(self) -> list[tuple[str, object]]: ("chunks", self.chunks if self.chunks is not None else "none (no fixed-size columns)"), ("blocks", self.blocks if self.blocks is not None else "none (no fixed-size columns)"), ("nbytes", format_nbytes_info(self.nbytes)), - ("cbytes", format_nbytes_info(self.cbytes)), - ("cratio", f"{self.cratio:.2f}x"), + ("cbytes", format_nbytes_info(self.cbytes) if compression_available else "n/a"), + ("cratio", f"{self.cratio:.2f}x" if compression_available else "n/a"), ("columns", column_summary), ("indexes", index_summary if index_summary else "none"), ] diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index c5284d767..4380ef7bb 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -647,6 +647,11 @@ def index_anchor_path(self, col_name: str) -> str | None: return None +def column_cbytes_for_info(column): + """Do not present shared HDF5 record storage as a per-field compressed size.""" + return None if isinstance(column, _RemoteHDF5Field) else getattr(column, "cbytes", None) + + class _RemoteHDF5Field(blosc2.Operand): """One field of a shared remote HDF5 compound dataset.""" diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 3cd443d23..89621fd8f 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -104,6 +104,26 @@ def test_remote_pytables_table_scan_and_shared_records(): np.testing.assert_array_equal(sliced.label[:], data["label"][1:3]) +def test_remote_pytables_info_omits_shared_column_sizes(tmp_path): + url, data = pytables_hdf5_url("pytables-info.h5", indexed=True) + with blosc2.RemoteCTable(url, dataset="table", cache_dir=tmp_path) as table: + for read in (False, True): + if read: + np.testing.assert_array_equal(table.id[:], data["id"]) + info = dict(table.info_items) + assert info["cbytes"] == info["cratio"] == "n/a" + assert "nbytes" in info + for name, summary in info["columns"].items(): + assert "cbytes" not in repr(summary) + assert "cratio" not in repr(summary) + column_info = dict(table[name].info_items) + assert "cbytes" not in column_info + assert "cratio" not in column_info + assert "nbytes" in column_info + assert "[opsi]" in info["indexes"]["id"] + assert "cbytes:" in info["indexes"]["id"] + + def test_remote_pytables_full_index_is_native_opsi(): url, data = pytables_hdf5_url("pytables-indexed.h5", indexed=True) with blosc2.RemoteCTable(url, dataset="table", cache_policy=blosc2.CachePolicy.MEMORY) as table: From 2edbfde79b69fb4f3ed4dda75707160aa3da8294 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Tue, 22 Sep 2026 17:55:24 +0200 Subject: [PATCH 62/82] Handle remote validity masks when resolving table extent --- src/blosc2/ctable.py | 12 +++++++++++- tests/ctable/test_remote_ctable.py | 25 +++++++++++++++++++++++++ 2 files changed, 36 insertions(+), 1 deletion(-) diff --git a/src/blosc2/ctable.py b/src/blosc2/ctable.py index 71896eb99..8a63a5939 100644 --- a/src/blosc2/ctable.py +++ b/src/blosc2/ctable.py @@ -6023,7 +6023,8 @@ def _resolve_last_pos(self) -> int: Returns the cached ``_last_pos`` when available. After a deletion ``_last_pos`` is ``None``; this method then walks chunk metadata of ``_valid_rows`` from the end (no full decompression) to find the last - ``True`` position, caches the result, and returns it. + ``True`` position, caches the result, and returns it. Readers without + chunk metadata are scanned backwards one chunk at a time instead. """ if self._last_pos is not None: return self._last_pos @@ -6033,6 +6034,15 @@ def _resolve_last_pos(self) -> int: self._last_pos = arr.shape[0] return self._last_pos chunk_size = arr.chunks[0] + if not hasattr(arr, "iterchunks_info"): + last_pos = 0 + for start in reversed(range(0, arr.shape[0], chunk_size)): + nonzero = np.flatnonzero(arr[start : min(start + chunk_size, arr.shape[0])]) + if len(nonzero): + last_pos = start + int(nonzero[-1]) + 1 + break + self._last_pos = last_pos + return self._last_pos last_true_pos = -1 for info in reversed(list(arr.iterchunks_info())): diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 89621fd8f..80bc9ceb0 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -1376,6 +1376,31 @@ class TextRow: read() +@pytest.mark.parametrize("deleted", [[], [1, 4], list(range(3, 11)), list(range(11))]) +def test_remote_ctable_info_with_validity_mask(tmp_path, deleted): + local = blosc2.CTable( + Row, urlpath=str(tmp_path / "source.b2d"), mode="w", expected_size=11, create_summary_index=False + ) + # Exercise multiple mask chunks, including a partial tail chunk. + local._valid_rows = local._storage.create_valid_rows(shape=(11,), chunks=(4,), blocks=(2,)) + local.extend([(i, [i, i + 1], str(i)) for i in range(11)]) + if deleted: + local.delete(deleted) + live = np.flatnonzero(local._valid_rows[:]) + expected = int(live[-1]) + 1 if len(live) else 0 + url = remote_table_url(tmp_path, local) + local.close() + with blosc2.open(url, cache_dir=tmp_path / "cache") as remote: + assert isinstance(remote._valid_rows, blosc2.RemoteArray) + assert remote._resolve_last_pos() == expected + assert remote._last_pos == expected + remote._last_pos = None # Exercise resolution through the report, too. + info = dict(remote.info_items) + assert ("valid_rows" in info) == (expected > len(live)) + assert info["nrows"] == len(live) + assert "RemoteCTable" in repr(remote.info) + + def test_remote_ctable_deleted_rows_and_disk_cache(tmp_path): source = str(tmp_path / "deleted.b2d") table = blosc2.CTable(Row, urlpath=source, mode="w", expected_size=8, create_summary_index=False) From 7039715f8d8b96e32dfd76a13fb4b3226bcb743c Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Tue, 22 Sep 2026 18:25:09 +0200 Subject: [PATCH 63/82] Cache small remote B2Z archives eagerly --- bench/remote_hdf5_source_cache.py | 6 +- doc/development/remote_cache_design.md | 63 +++++++----- doc/guides/remote_arrays.md | 7 +- src/blosc2/b2z_source.py | 136 +++++++++++++++++++++---- src/blosc2/hdf5_source.py | 94 +++-------------- src/blosc2/remote_array.py | 25 +++-- src/blosc2/remote_ctable.py | 2 +- src/blosc2/remote_source_cache.py | 90 ++++++++++++++++ src/blosc2/remote_store.py | 49 ++++++++- src/blosc2/schunk.py | 6 +- tests/b2view/test_hierarchy.py | 2 + tests/conftest.py | 8 ++ tests/ctable/test_remote_ctable.py | 7 ++ tests/test_b2z_source.py | 129 +++++++++++++++++++++++ tests/test_fsspec.py | 35 +++++-- tests/test_remote_store.py | 12 ++- 16 files changed, 517 insertions(+), 154 deletions(-) create mode 100644 src/blosc2/remote_source_cache.py diff --git a/bench/remote_hdf5_source_cache.py b/bench/remote_hdf5_source_cache.py index 933abd28e..f34597eaa 100644 --- a/bench/remote_hdf5_source_cache.py +++ b/bench/remote_hdf5_source_cache.py @@ -1,4 +1,4 @@ -"""Compare metadata-only and persisted-source HDF5 caches across fresh processes. +"""Compare metadata-only and persisted-source container caches across fresh processes. Run with the blosc2 environment, e.g.: conda run -n blosc2 python bench/remote_hdf5_source_cache.py --repeats 3 @@ -20,6 +20,10 @@ def worker(args): if args.mode == "metadata-only": # Reproduce the previous cache policy while keeping discovery identical. hdf5_source.publish_hdf5_source_cache = lambda *a, **kw: None + # B2Z previously used range reads, even for a small archive. + import blosc2.b2z_source as b2z_source + + b2z_source.SMALL_REMOTE_FILE = 0 start = time.perf_counter() with blosc2.open(args.url, cache_dir=args.cache) as table: opened = time.perf_counter() diff --git a/doc/development/remote_cache_design.md b/doc/development/remote_cache_design.md index 35884a030..894782453 100644 --- a/doc/development/remote_cache_design.md +++ b/doc/development/remote_cache_design.md @@ -17,11 +17,11 @@ distinguished: | --- | --- | --- | | Discovery metadata | Reconstruct readers, locate data, and navigate containers | Array carriers and store manifests; format-specific headers, indexes, or metadata objects | | Compressed payload | Reuse fetched or converted data | Native Blosc2 chunks/blocks, or chunks converted from HDF5/Zarr | -| Complete source object | Reuse original bytes across dataset scopes | Small HDF5 files only | +| Complete source object | Reuse original bytes across dataset scopes | HDF5 files and B2Z archives up to 8 MiB | A warm metadata cache does not imply a warm payload cache. Reopening may avoid -discovery requests but still fetch data for a preview. HDF5's complete-source -cache bridges that gap for small files; the other routes do not provide a +discovery requests but still fetch data for a preview. HDF5 and B2Z complete-source +caches bridge that gap for small files; the other routes do not provide a general persistent copy of the remote source. ### Identity and ownership @@ -73,7 +73,7 @@ Explicit table/store refresh prepares new discovery before replacing the active generation and invalidating derived caches. Failure during preparation leaves the previous generation usable. Existing child handles of a refreshed store become stale and must be reacquired. Other independently opened scopes are not -automatically refreshed; HDF5's shared-source version check adds the specific +automatically refreshed; HDF5/B2Z shared-source version checks add the specific reopen behavior described below. Immutable reference snapshots cannot refresh. ## Native Blosc2 arrays and B2Z containers @@ -104,11 +104,17 @@ reader to be reconstructed without fetching its header again. A populated table/store cache trusts the saved archive identity on reopen; older caches may need an identity lookup when upgrading their metadata. -Bounded opening prefetch can happen to contain a complete small member. Its -payload is transferred to the normal chunk cache, while the persistent leaf -bootstrap keeps the metadata it needs. This is not an archive-wide source cache: -uncached members or chunks still require remote reads. Replacing an archive at -the same URL requires explicit refresh or cache replacement. +Cold discovery eagerly downloads archives up to 8 MiB and retains the bytes for +all members. HTTP discovery first requests the ZIP tail to learn the object size; +if the tail contains the whole archive it is reused, otherwise a second request +fetches the complete small archive. Larger archives keep the range-read path. +With an explicit DISK cache directory, the shared source copy survives reopening. +All archive read paths, including parallel table reads, can use these bytes. + +Bounded member prefetch can also contain a complete small member. Its payload is +transferred to the normal chunk cache, while the persistent leaf bootstrap keeps +the metadata it needs. This remains useful for large archives without a source +copy. Replacing an archive at the same URL requires refresh or cache replacement. ## Zarr stores @@ -153,31 +159,36 @@ For objects up to 8 MiB, discovery fetches the complete file once: one bounded transfer avoids many latency-bound metadata reads. Retaining these bytes lets later reads and PyTables index conversion reuse the same download. -### Shared complete-source cache +## Shared complete-source cache (HDF5 and B2Z) The complete source bytes, discovery metadata, and converted Blosc2 chunks have different lifetimes. Source bytes are shared across dataset scopes; metadata and converted payload remain in their existing array/store caches. With an explicit `cache_dir` and `CachePolicy.DISK`, source bytes persist under -`cache_dir/hdf5-sources/`. The source identity reuses `fsspec_cache_path()`: +`cache_dir/hdf5-sources/` or `cache_dir/b2z-sources/`. Both readers reuse the +integrity, publication, and locking helpers in `remote_source_cache.py`. +The source identity reuses `fsspec_cache_path()`: normalized base URL plus the non-reversible `storage_options` fingerprint, without the dataset scope. Different datasets in one file share a single copy; different cache directories, URLs, or credential fingerprints remain isolated. -Each source has a `.hdf5-source` file, a JSON marker, and a publication lock. +Each source has a `.hdf5-source` or `.b2z-source` file, a JSON marker, and a publication lock. The marker records its schema version, byte size, SHA-256, and version token. These names and fields are private implementation details. Source copies do not count against `max_cache_bytes`. The 8 MiB ceiling is per file; aggregate source-cache disk usage is unbounded. They survive individual dataset-generation cleanup. Close active cache users before removing -`cache_dir/hdf5-sources/` to clear source copies, or the whole cache directory to +the corresponding `hdf5-sources/` or `b2z-sources/` directory to clear source copies, +or the whole cache directory to reset all caches. There is no automatic source eviction or freshness check. -MEMORY/NONE policies, `cache_path`-only opens, and portable references without -an explicit shared cache root retain their previous behavior. B2Z and Zarr do -not use this source cache. +MEMORY/NONE policies can retain prefetched source bytes for the current session, +but do not persist them. `cache_path`-only opens and portable references without +an explicit shared root do not gain a persistent source copy. Saved metadata +alone does not force a whole-archive download on portable B2Z reference reopen. +Zarr does not use this source cache. ### Publication and reuse @@ -195,10 +206,12 @@ size lookup or download. ### Version compatibility and refresh -Generated indexes record `source_sha256` when complete bytes are available. +Generated HDF5 indexes and B2Z discovery metadata record `source_sha256` when +complete bytes are available. Small B2Z member stamps derive from that content +identity rather than transport-dependent header fields. An index incompatible with the shared source must be rebuilt, and its derived -payload invalidated. Generated legacy indexes without a digest are rebound by -rebuilding once when a shared source becomes available. +payload invalidated. Legacy small-source disk metadata without a digest is +rebound by rebuilding discovery and invalidating derived payload once. Table/store refresh bypasses the retained bytes and prepares fresh discovery. A preparation failure leaves the old live generation usable. Successful refresh @@ -206,7 +219,7 @@ publishes replacement metadata and source bytes; other scopes check compatibilit on their next open. Existing handles may retain their old immutable snapshot. If the refreshed file exceeds 8 MiB, publication replaces the old marker with a -new version token and null checksum, then removes the old blob. Generated indexes +new version token and null checksum, then removes the old blob. Discovery metadata record this token as `source_cache_version`, so sibling scopes invalidate old metadata and payload even though there is no replacement cached blob. @@ -219,11 +232,11 @@ contract: matching sizes alone cannot prove version compatibility. ## Boundaries and future work The common model does not imply identical bootstrap costs or invalidation -mechanisms across formats. In particular, the 8 MiB complete-source threshold, -checksum marker, and cross-scope source-version reconciliation are HDF5-only. +mechanisms across formats. The 8 MiB complete-source threshold, checksum marker, +and cross-scope source-version reconciliation apply to HDF5 and B2Z, not to +standalone `.b2nd` files or Zarr stores. -A persistent small-archive source cache for B2Z remains deferred pending -measurements. Zarr would require an object-level policy rather than copying -HDF5's single-file design. Automatic freshness checks for containers, aggregate +Zarr would require an object-level policy rather than this single-file design. +Automatic freshness checks for containers, aggregate source-cache eviction, and a global disk-space budget are not implemented by this design. diff --git a/doc/guides/remote_arrays.md b/doc/guides/remote_arrays.md index 7c08d5bdd..bd635e32b 100644 --- a/doc/guides/remote_arrays.md +++ b/doc/guides/remote_arrays.md @@ -55,7 +55,12 @@ a1[100:110, :50] # data is fetched now Remote B2Z needs `pip install "blosc2[fsspec]"`. HTTP and HTTPS URLs work out of the box; cloud object stores need their respective protocol driver (such as `s3fs` for S3, `gcsfs` for GCS, or `adlfs` for Azure). -It accesses external `ZIP_STORED` NDArray members in `.b2z` archives using native Blosc2 chunk and block range reads without decompressing or downloading the archive. +It accesses external `ZIP_STORED` NDArray members using native Blosc2 chunk and +block reads. Archives up to 8 MiB are downloaded eagerly; larger archives use +range reads. With `cache_dir=` and `CachePolicy.DISK`, the small-archive copy is +shared across members and later processes. Source copies are excluded from +`max_cache_bytes`; see {doc}`../development/remote_cache_design` for accounting +and cleanup details. For a suffix-free URL, pass `source_format="b2z"`. Embedded leaves and CTable columns are not supported as lazy NDArrays. diff --git a/src/blosc2/b2z_source.py b/src/blosc2/b2z_source.py index db5775769..0594103a0 100644 --- a/src/blosc2/b2z_source.py +++ b/src/blosc2/b2z_source.py @@ -6,14 +6,23 @@ """Native range reads of external NDArray members in immutable B2Z archives.""" +import hashlib import io import operator import re +import uuid import zipfile from contextlib import contextmanager from blosc2.core import _import_fsspec from blosc2.proxy_source import REMOTE_MAX_CONCURRENCY, ByteRangeNDSource, Traffic +from blosc2.remote_source_cache import ( + SMALL_REMOTE_FILE, + load_source_cache, + publish_source_cache, + source_cache_path, + source_version, +) # One request can carry a small member whole -- local header, frame header and the # trailing vlmeta -- sparing a round trip each against the object store. @@ -21,6 +30,14 @@ _LOCAL_HEADER_HEADROOM = 4096 +def b2z_source_cache(urlpath, cache_dir, storage_options=None, *, refresh=False): + if cache_dir is None: + return None, None, None + path = source_cache_path(urlpath, cache_dir, storage_options, kind="b2z") + blob, marker = load_source_cache(path) + return path, marker, None if refresh else blob + + async def _http_tail(fs, path): """Obtain HTTP object identity and the ZIP tail in one bounded request.""" kwargs = fs.kwargs.copy() @@ -87,7 +104,17 @@ def read(self, size=-1): class B2ZArchive: """Session directory and bounded range access shared by discovery and leaves.""" - def __init__(self, urlpath, *, storage_options=None, _filesystem=None, _traffic=None, _metadata=None): + def __init__( + self, + urlpath, + *, + storage_options=None, + _filesystem=None, + _traffic=None, + _metadata=None, + _source_cache=(None, None, None), + _refresh=False, + ): self.storage_options = storage_options or {} self.urlpath = urlpath fsspec = _import_fsspec(urlpath) @@ -96,28 +123,21 @@ def __init__(self, urlpath, *, storage_options=None, _filesystem=None, _traffic= else: self._fs, self._path = _filesystem, _filesystem._strip_protocol(urlpath) self.traffic = _traffic if _traffic is not None else Traffic() - bootstrap = None - if self._fs.protocol in ("http", "https", ("http", "https")) and not _metadata: - from fsspec.asyn import sync - from fsspec.implementations.http import HTTPFileSystem - - if isinstance(self._fs, HTTPFileSystem): - bootstrap = sync(self._fs.loop, _http_tail, self._fs, self._path) - legacy_metadata = bool(_metadata) and "object_info" not in _metadata - object_info = (_metadata or {}).get("object_info") - if object_info is None: - # Old manifests need one identity lookup to acquire this bootstrap. - object_info = self._fs.info(self._path) if bootstrap is None else bootstrap[0] - size = object_info["size"] - if not isinstance(size, int) or isinstance(size, bool) or size < 0: - raise ValueError("Invalid cached B2Z archive size") - self.size = size + self.source_cache_path, self.source_cache_marker, self.blob = _source_cache + object_info, bootstrap, legacy_metadata = self._bootstrap_source(_metadata) + size = self.size self.metadata = _metadata if _metadata is not None else {} self.persist_metadata = _metadata is not None identity = repr(sorted((key, str(value)) for key, value in object_info.items())) if self.metadata and self.metadata.get("identity") != identity: raise ValueError("B2Z source changed; refresh the store cache") self.metadata.setdefault("identity", identity) + if self.blob is not None: + self.metadata["source_sha256"] = object_info["source_sha256"] + elif _refresh and self.source_cache_path is not None: + self.metadata["source_cache_version"] = hashlib.sha256(uuid.uuid4().bytes).hexdigest() + elif self.source_cache_marker is not None: + self.metadata["source_cache_version"] = self.source_cache_marker["token"] # Backend info can contain datetime or other non-msgpack values. Only # size needs its native type; retain per-member stamps before normalizing. self.metadata.setdefault( @@ -157,6 +177,43 @@ def __init__(self, urlpath, *, storage_options=None, _filesystem=None, _traffic= self.members = self.archive.infolist() self.capture_metadata = False + def _bootstrap_source(self, _metadata): + bootstrap = None + if self.blob is None and self._fs.protocol in ("http", "https", ("http", "https")) and not _metadata: + from fsspec.asyn import sync + from fsspec.implementations.http import HTTPFileSystem + + if isinstance(self._fs, HTTPFileSystem): + bootstrap = sync(self._fs.loop, _http_tail, self._fs, self._path) + legacy_metadata = bool(_metadata) and "object_info" not in _metadata + object_info = (_metadata or {}).get("object_info") + if self.blob is not None: + object_info = {"size": len(self.blob), "source_sha256": hashlib.sha256(self.blob).hexdigest()} + if object_info is None: + # Old manifests need one identity lookup to acquire this bootstrap. + object_info = self._fs.info(self._path) if bootstrap is None else bootstrap[0] + size = object_info["size"] + if not isinstance(size, int) or isinstance(size, bool) or size < 0: + raise ValueError("Invalid cached B2Z archive size") + self.size = size + if ( + self.blob is None + and size <= SMALL_REMOTE_FILE + and (not _metadata or self.source_cache_path is not None) + ): + if bootstrap is not None: + self.traffic.charge(len(bootstrap[1])) + if bootstrap is not None and len(bootstrap[1]) == size: + self.blob = bootstrap[1] + else: + self.blob = self.read_transport(0, size) + bootstrap = None + # A content identity is stable across transports and cached reopens. + object_info = {"size": size, "source_sha256": hashlib.sha256(self.blob).hexdigest()} + if _metadata and not source_version(_metadata): + _metadata.clear() # Upgrade legacy discovery against the downloaded source. + return object_info, bootstrap, legacy_metadata + def member_window(self, info, *, prefetch=False): size = self.size file, archive = self.file, self.archive @@ -219,12 +276,24 @@ def read_transport(self, offset, size): """Read immutable bytes only; safe while the owner parses other responses.""" if not 0 <= offset <= self.size or not 0 <= size <= self.size - offset: raise ValueError("B2Z range exceeds archive bounds") + if self.blob is not None: + return self.blob[offset : offset + size] data = self._fs.cat_file(self._path, start=offset, end=offset + size) if len(data) != size: raise ValueError("B2Z transport did not honor the requested byte range") self.traffic.charge(len(data)) return data + def publish_source(self, *, refresh=False): + if self.source_cache_path is not None: + publish_source_cache( + self.source_cache_path, + self.blob, + {**self.metadata, "size": self.size}, + expected=self.source_cache_marker, + refresh=refresh, + ) + @contextmanager def buffered_ranges(self, ranges): """Owner-thread-only prefix reuse while opening a batch of members.""" @@ -290,9 +359,10 @@ def read_transport(self, offset, size): class B2ZArrayNotFoundError(ValueError): """The selected archive path is not an external NDArray member.""" - def __init__(self, dataset, traffic): + def __init__(self, dataset, traffic, blob=None): super().__init__(f"No supported external NDArray at {dataset!r}; specify an external array leaf") self.traffic = traffic + self.blob = blob class B2ZNDSource(ByteRangeNDSource): @@ -315,6 +385,7 @@ def __init__( _traffic=None, _archive=None, _seed=None, + _source_cache_dir=None, ): if not isinstance(dataset, str) or not dataset.strip("/"): raise ValueError("B2Z sources require a dataset path (e.g. dataset='d0/a3')") @@ -324,6 +395,13 @@ def __init__( ): raise ValueError("invalid B2Z dataset path") self.dataset = dataset + cache = b2z_source_cache(urlpath, _source_cache_dir, storage_options) + if _source_cache_dir is not None and _seed is not None and "source_cache_version" not in _seed: + _seed = None # One-time migration from a metadata-only carrier. + if cache[1] is not None and _seed is not None and source_version(_seed) != cache[1]["token"]: + _seed = None + if cache[2] is not None: + _seed = None if _seed is not None: if _archive is not None: raise ValueError("a cached B2Z seed cannot be combined with an archive") @@ -332,7 +410,11 @@ def __init__( if _archive is not None and _archive.urlpath != urlpath: raise ValueError("B2Z source URL does not match its archive") archive = _archive or B2ZArchive( - urlpath, storage_options=storage_options, _filesystem=_filesystem, _traffic=_traffic + urlpath, + storage_options=storage_options, + _filesystem=_filesystem, + _traffic=_traffic, + _source_cache=cache, ) try: archive.capture_metadata = True @@ -344,7 +426,7 @@ def __init__( object_info = archive.object_info matches = [info for info in archive.members if info.filename == dataset + ".b2nd"] if not matches: - raise B2ZArrayNotFoundError(dataset, self.traffic) + raise B2ZArrayNotFoundError(dataset, self.traffic, archive.blob) if len(matches) != 1: raise ValueError("duplicate B2Z array member") self.member_offset, self.member_length = archive.member_window(matches[0], prefetch=True) @@ -356,7 +438,12 @@ def __init__( self.stamp = archive.metadata.setdefault("member_stamps", {}).setdefault( dataset, tokenize( - urlpath, sorted(object_info.items()), dataset, self.member_offset, self.member_length + urlpath, + sorted(object_info.items()), + dataset, + self.member_offset, + self.member_length, + *([source_version(archive.metadata)] if source_version(archive.metadata) else []), ), ) super().__init__(urlpath, max_concurrency, traffic=self.traffic) @@ -376,6 +463,7 @@ def __init__( def _init_seeded(self, urlpath, max_concurrency, storage_options, filesystem, traffic, seed): """Rebuild this source from a carrier's cached bootstrap, without network.""" + self._seed = seed self.member_offset = int(seed["member_offset"]) self.member_length = int(seed["member_length"]) self.stamp = str(seed["stamp"]) @@ -423,6 +511,12 @@ def _prepare_prefetch(self, prefix_start, prefix): "prefix_start": self.member_offset, "prefix": self._raw_header, "stamp": self.stamp, + "source_sha256": self._archive.metadata.get("source_sha256") + if hasattr(self._archive, "metadata") + else self._seed.get("source_sha256"), + "source_cache_version": source_version(self._archive.metadata) + if hasattr(self._archive, "metadata") + else source_version(self._seed), } def _take_prefetched_array(self): diff --git a/src/blosc2/hdf5_source.py b/src/blosc2/hdf5_source.py index 6a8f51624..6af70670b 100644 --- a/src/blosc2/hdf5_source.py +++ b/src/blosc2/hdf5_source.py @@ -18,64 +18,33 @@ import threading import weakref import zlib -from pathlib import Path from urllib.parse import urlsplit import numpy as np import blosc2 from blosc2.proxy_source import REMOTE_MAX_CONCURRENCY, ProxyNDSource, Traffic +from blosc2.remote_source_cache import ( + SMALL_REMOTE_FILE as _SMALL_REMOTE_FILE, +) +from blosc2.remote_source_cache import ( + load_source_cache as load_hdf5_source_cache, +) +from blosc2.remote_source_cache import publish_source_cache as publish_hdf5_source_cache # noqa: F401 +from blosc2.remote_source_cache import ( + source_cache_path, +) +from blosc2.remote_source_cache import ( + source_version as hdf5_source_version, +) HDF5_INDEX_FORMAT = "blosc2-hdf5-index" HDF5_INDEX_VERSION = 2 _HDF5_INDEX_VERSIONS = {1, HDF5_INDEX_VERSION} -_SMALL_REMOTE_FILE = 8 << 20 def hdf5_source_cache_path(urlpath, cache_dir, storage_options=None): - """Resolve a source artifact independently of dataset-scoped carriers.""" - return Path( - blosc2.core.fsspec_cache_path( - urlpath, Path(cache_dir) / "hdf5-sources", ".hdf5-source", storage_options=storage_options - ) - ) - - -def _source_marker(path): - try: - with path.with_suffix(path.suffix + ".json").open("rb") as file: - marker = json.loads(file.read(4097)) - size, token = marker["size"], marker["token"] - if marker["version"] != 1 or type(size) is not int or size <= 0: - return None - if not isinstance(token, str) or len(token) != 64 or any(c not in "0123456789abcdef" for c in token): - return None - if marker["sha256"] != (token if size <= _SMALL_REMOTE_FILE else None): - return None - return marker - except (OSError, ValueError, KeyError, TypeError): - return None - - -def load_hdf5_source_cache(path): - """Return verified bytes and version metadata, including large-source tombstones.""" - marker = _source_marker(path) - if marker is None or marker["size"] > _SMALL_REMOTE_FILE: - return None, marker - try: - with path.open("rb") as file: - if os.fstat(file.fileno()).st_size != marker["size"]: - return None, marker - blob = file.read(_SMALL_REMOTE_FILE + 1) - if len(blob) == marker["size"] and hashlib.sha256(blob).hexdigest() == marker["sha256"]: - return blob, marker - except OSError: - pass - return None, marker - - -def hdf5_source_version(index): - return index.get("source_cache_version", index.get("source_sha256")) + return source_cache_path(urlpath, cache_dir, storage_options, kind="hdf5") def reconcile_hdf5_index(index, urlpath, dataset, storage_options, blob, marker, *, explicit): @@ -94,41 +63,6 @@ def reconcile_hdf5_index(index, urlpath, dataset, storage_options, blob, marker, return None if mismatch else index -def publish_hdf5_source_cache(path, blob, index, *, expected=None, refresh=False): - """Publish an optional complete source, or an invalidation marker after refresh.""" - from blosc2.remote_store_cache import lock_cache_file - - # Serialize the version check and replacement against concurrent refresh. - # Readers use checksums; no store/array locks are acquired under this lock. - with path.with_suffix(path.suffix + ".lock").open("a+b") as lock: - lock_cache_file(lock, blocking=True) - _publish_source_cache(path, blob, index, expected, refresh) - - -def _publish_source_cache(path, blob, index, expected, refresh): - from blosc2.remote_store_cache import atomic_write - - current = _source_marker(path) - token = hdf5_source_version(index) - if not refresh and current != expected and (current or {}).get("token") != token: - raise RuntimeError("HDF5 source cache changed during open; retry the operation") - if blob is None and not refresh: - return - size = index["size"] - marker = {"version": 1, "size": size, "token": token, "sha256": index.get("source_sha256")} - if blob is not None: - if len(blob) != size or size > _SMALL_REMOTE_FILE or hashlib.sha256(blob).hexdigest() != token: - raise ValueError("Invalid HDF5 source-cache bytes") - if current == marker: - existing, _ = load_hdf5_source_cache(path) - if existing is not None: - return - atomic_write(path, blob) - atomic_write(path.with_suffix(path.suffix + ".json"), json.dumps(marker).encode()) - if blob is None: - path.unlink(missing_ok=True) - - def prepare_hdf5_source_cache( urlpath, dataset, cache_dir, storage_options, index, manifest, blob, *, explicit ): diff --git a/src/blosc2/remote_array.py b/src/blosc2/remote_array.py index 6a885e6bb..a7add89ed 100644 --- a/src/blosc2/remote_array.py +++ b/src/blosc2/remote_array.py @@ -344,7 +344,7 @@ def _open_url_source( seed=None, blocks=None, cparams=None, - hdf5_cache_dir=None, + source_cache_dir=None, hdf5_index_explicit=True, ): if persistable: @@ -374,7 +374,7 @@ def _open_url_source( urlpath, dataset, hdf5_index=hdf5_index, - _source_cache_dir=hdf5_cache_dir, + _source_cache_dir=source_cache_dir, _index_explicit=hdf5_index_explicit, _traffic=traffic, blocks=blocks, @@ -391,7 +391,9 @@ def _open_url_source( elif source_format == "b2z": if not assume_immutable: raise NotImplementedError("mutable B2Z sources are not supported") - src = blosc2.B2ZNDSource(urlpath, dataset, _traffic=traffic, _seed=seed, **kwargs) + src = blosc2.B2ZNDSource( + urlpath, dataset, _traffic=traffic, _seed=seed, _source_cache_dir=source_cache_dir, **kwargs + ) source = { "kind": "b2z", "version": 1, @@ -683,7 +685,7 @@ def __init__( seed=seed, blocks=_source_blocks, cparams=_source_cparams, - hdf5_cache_dir=cache_dir if cache_policy is blosc2.CachePolicy.DISK else None, + source_cache_dir=cache_dir if cache_policy is blosc2.CachePolicy.DISK else None, hdf5_index_explicit=hdf5_index_explicit, ) self._assume_immutable = assume_immutable @@ -708,7 +710,7 @@ def __init__( self._initialize_runtime_cache(cache_dir, cache_path, _runtime_cache_path) - self._publish_hdf5_source() + self._publish_source_cache() _publish_hdf5_index(shared_index_path, self._carrier, scanned=hdf5_index is None) @@ -723,7 +725,11 @@ def __init__( self._store_owner = _store_owner self._store_finalizer = weakref.finalize(self, _store_owner.release) - def _publish_hdf5_source(self): + def _publish_source_cache(self): + if not self._authorized_source and isinstance(self.src, blosc2.B2ZNDSource): + publish = getattr(self.src._archive, "publish_source", None) + if publish is not None: + publish() if ( not self._authorized_source and isinstance(self.src, blosc2.HDF5NDSource) @@ -856,8 +862,7 @@ def _open_or_create_carrier(self, cache_dir, cache_path): ) if ( status == "invalidated/rebuilt" - and isinstance(self.src, blosc2.HDF5NDSource) - and self.src._source_cache_path is not None + and isinstance(self.src, (blosc2.HDF5NDSource, blosc2.B2ZNDSource)) and self._geometry(carrier) != self._geometry(self.src) ): del carrier @@ -1149,7 +1154,7 @@ def _open_source( seed=None, blocks=None, cparams=None, - hdf5_cache_dir=None, + source_cache_dir=None, hdf5_index_explicit=True, ): if isinstance(urlpath, blosc2.C2Array): @@ -1199,7 +1204,7 @@ def _open_source( seed=seed, blocks=blocks, cparams=cparams, - hdf5_cache_dir=hdf5_cache_dir, + source_cache_dir=source_cache_dir, hdf5_index_explicit=hdf5_index_explicit, ) else: diff --git a/src/blosc2/remote_ctable.py b/src/blosc2/remote_ctable.py index 992687c54..ce77af03b 100644 --- a/src/blosc2/remote_ctable.py +++ b/src/blosc2/remote_ctable.py @@ -232,7 +232,7 @@ def refresh(self) -> None: ) replacement.restoring = False replacement.save_manifest() - replacement.publish_hdf5_source(refresh=True) + replacement.publish_source_cache(refresh=True) except BaseException: replacement.disk = None if fresh is not None: diff --git a/src/blosc2/remote_source_cache.py b/src/blosc2/remote_source_cache.py new file mode 100644 index 000000000..a5699ffc1 --- /dev/null +++ b/src/blosc2/remote_source_cache.py @@ -0,0 +1,90 @@ +"""Bounded whole-source artifacts shared by remote container readers.""" + +import hashlib +import json +import os +from pathlib import Path + +import blosc2 + +SMALL_REMOTE_FILE = 8 << 20 + + +def source_cache_path(urlpath, cache_dir, storage_options=None, *, kind): + return Path( + blosc2.core.fsspec_cache_path( + urlpath, Path(cache_dir) / f"{kind}-sources", f".{kind}-source", storage_options=storage_options + ) + ) + + +def _source_marker(path): + try: + with path.with_suffix(path.suffix + ".json").open("rb") as file: + marker = json.loads(file.read(4097)) + size, token = marker["size"], marker["token"] + if marker["version"] != 1 or type(size) is not int or size <= 0: + return None + if not isinstance(token, str) or len(token) != 64 or any(c not in "0123456789abcdef" for c in token): + return None + if marker["sha256"] != (token if size <= SMALL_REMOTE_FILE else None): + return None + return marker + except (OSError, ValueError, KeyError, TypeError): + return None + + +def load_source_cache(path): + """Return verified bytes and version metadata, including large-source tombstones.""" + marker = _source_marker(path) + if marker is None or marker["size"] > SMALL_REMOTE_FILE: + return None, marker + try: + with path.open("rb") as file: + if os.fstat(file.fileno()).st_size != marker["size"]: + return None, marker + blob = file.read(SMALL_REMOTE_FILE + 1) + if len(blob) == marker["size"] and hashlib.sha256(blob).hexdigest() == marker["sha256"]: + return blob, marker + except OSError: + pass + return None, marker + + +def source_version(index): + return index.get("source_cache_version", index.get("source_sha256")) + + +def publish_source_cache(path, blob, index, *, expected=None, refresh=False): + """Publish an optional complete source, or an invalidation marker after refresh.""" + from blosc2.remote_store_cache import lock_cache_file + + # Serialize the version check and replacement against concurrent refresh. + # Readers use checksums; no store/array locks are acquired under this lock. + with path.with_suffix(path.suffix + ".lock").open("a+b") as lock: + lock_cache_file(lock, blocking=True) + _publish_source_cache(path, blob, index, expected, refresh) + + +def _publish_source_cache(path, blob, index, expected, refresh): + from blosc2.remote_store_cache import atomic_write + + current = _source_marker(path) + token = source_version(index) + if not refresh and current != expected and (current or {}).get("token") != token: + raise RuntimeError("Remote source cache changed during open; retry the operation") + if blob is None and not refresh: + return + size = index["size"] + marker = {"version": 1, "size": size, "token": token, "sha256": index.get("source_sha256")} + if blob is not None: + if len(blob) != size or size > SMALL_REMOTE_FILE or hashlib.sha256(blob).hexdigest() != token: + raise ValueError("Invalid Remote source-cache bytes") + if current == marker: + existing, _ = load_source_cache(path) + if existing is not None: + return + atomic_write(path, blob) + atomic_write(path.with_suffix(path.suffix + ".json"), json.dumps(marker).encode()) + if blob is None: + path.unlink(missing_ok=True) diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index c9923dfd7..7e6c71bc0 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -130,6 +130,9 @@ def __init__( _hdf5_index=None, _hdf5_blob=None, _traffic=None, + _source_cache_dir=None, + _refresh_source=False, + _b2z_blob=None, ): self.urlpath, dataset, self.format = parse_container_url(urlpath, dataset) if _source_format is not None: @@ -139,6 +142,33 @@ def __init__( self.root = (dataset or "").strip("/") self._validate(self.root) self.storage_options = storage_options or {} + self.source_cache_dir = _source_cache_dir + self.refresh_source = _refresh_source + self.b2z_source_cache = (None, None, None) + if self.format == "b2z": + from blosc2.b2z_source import SMALL_REMOTE_FILE, b2z_source_cache + from blosc2.remote_source_cache import source_version + + self.b2z_source_cache = b2z_source_cache( + self.urlpath, _source_cache_dir, self.storage_options, refresh=_refresh_source + ) + marker = self.b2z_source_cache[1] + if ( + self.b2z_source_cache[2] is None + and _b2z_blob is not None + and (marker is None or hashlib.sha256(_b2z_blob).hexdigest() == marker["token"]) + ): + self.b2z_source_cache = (*self.b2z_source_cache[:2], _b2z_blob) + if manifest and marker and source_version(manifest["metadata"]) != marker["token"]: + manifest = None + if ( + manifest + and _source_cache_dir is not None + and not source_version(manifest["metadata"]) + and SMALL_REMOTE_FILE + and manifest["metadata"].get("object_info", {}).get("size", 0) <= SMALL_REMOTE_FILE + ): + manifest = None # Rebind legacy metadata/payload to the source digest once. self.traffic = _traffic if _traffic is not None else Traffic() self.nodes = {} self.attrs = {} @@ -235,6 +265,8 @@ def _restore_manifest(self, manifest): _traffic=self.traffic, _metadata=self.metadata, _filesystem=self.filesystem, + _source_cache=self.b2z_source_cache, + _refresh=self.refresh_source, ) elif self.format == "hdf5": self.hdf5_index = self.metadata @@ -412,6 +444,8 @@ def _open_b2z(self): _traffic=self.traffic, _metadata=self.metadata if self.persist_metadata else None, _filesystem=self.filesystem, + _source_cache=self.b2z_source_cache, + _refresh=self.refresh_source, ) self.archive.capture_metadata = True members = {} @@ -526,7 +560,9 @@ def attach_hdf5_source_cache(self, path, marker): elif marker is not None: self.hdf5_index["source_cache_version"] = marker["token"] - def publish_hdf5_source(self, *, refresh=False): + def publish_source_cache(self, *, refresh=False): + if self.format == "b2z" and self.archive is not None: + self.archive.publish_source(refresh=refresh) if self.hdf5_source_cache_path is not None: from blosc2.hdf5_source import publish_hdf5_source_cache @@ -802,7 +838,7 @@ def _open_b2z_source(self, full): # Immutable archives trust their persisted identity when restoring leaves. seeds = self.archive.metadata.get("ctable_seeds", {}) - if full in seeds: + if full in seeds and self.archive.blob is None: return B2ZNDSource( self.urlpath, full, @@ -1010,6 +1046,8 @@ def prepare_refresh(self, kind): _manifest_validator=self.manifest_validator, _max_nodes=self.max_nodes, _source_format=self.format, + _source_cache_dir=self.source_cache_dir, + _refresh_source=True, ) try: if replacement.nodes[replacement.root][0] != kind: @@ -1393,6 +1431,7 @@ def __init__( _hdf5_blob=None, _traffic=None, nested_storage_options=None, + _b2z_blob=None, ): if not isinstance(urlpath, (str, os.PathLike)): raise TypeError("RemoteStore requires a remote URL string") @@ -1459,6 +1498,8 @@ def __init__( _hdf5_index=hdf5_index, _hdf5_blob=_hdf5_blob, _traffic=_traffic, + _source_cache_dir=cache_dir if disk is not None else None, + _b2z_blob=_b2z_blob, ) manifest = owner.restored_manifest owner.attach_hdf5_source_cache(source_cache_path, source_cache_marker) @@ -1482,7 +1523,7 @@ def __init__( try: owner.restore_caches(manifest) owner.save_manifest() - owner.publish_hdf5_source() + owner.publish_source_cache() if disk is not None: disk.discard_old_generations(owner.generation) except BaseException: @@ -1971,7 +2012,7 @@ def refresh(self): try: replacement.restoring = False replacement.save_manifest() - replacement.publish_hdf5_source(refresh=True) + replacement.publish_source_cache(refresh=True) if replacement.disk is not None: replacement.disk.discard_old_generations(replacement.generation) except BaseException: diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index 773510fec..ff2c70e48 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2344,7 +2344,11 @@ def _open_remote_b2z(urlpath, options): } try: with blosc2.RemoteStore( - urlpath, _allow_array_root=True, _source_format="b2z", **store_options + urlpath, + _allow_array_root=True, + _source_format="b2z", + _b2z_blob=array_error.blob if array_error is not None else None, + **store_options, ) as store: if array_error is not None: # Include the initial array lookup in shared transfer accounting. diff --git a/tests/b2view/test_hierarchy.py b/tests/b2view/test_hierarchy.py index 239bd7086..b154aec33 100644 --- a/tests/b2view/test_hierarchy.py +++ b/tests/b2view/test_hierarchy.py @@ -39,6 +39,7 @@ def b2z_url(tmp_path, *, threshold=0): return url, data +@pytest.mark.usefixtures("b2z_range_reads") def test_b2z_discovery_and_dispatch(tmp_path, monkeypatch): url, data = b2z_url(tmp_path) fs = fsspec.filesystem("memory") @@ -256,6 +257,7 @@ def slow(self, path, **kwargs): await pilot.press("q") +@pytest.mark.usefixtures("b2z_range_reads") def test_embedded_b2z_index_is_bounded(tmp_path, monkeypatch): url, data = b2z_url(tmp_path, threshold=10**9) fs = fsspec.filesystem("memory") diff --git a/tests/conftest.py b/tests/conftest.py index 8a33ceb18..fa5abc290 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -38,6 +38,14 @@ def expected_nthreads(nthreads: int) -> int: return 1 if blosc2.IS_WASM else nthreads +@pytest.fixture +def b2z_range_reads(monkeypatch): + """Exercise the large-archive range path using small, inexpensive fixtures.""" + import blosc2.b2z_source + + monkeypatch.setattr(blosc2.b2z_source, "SMALL_REMOTE_FILE", 0) + + @pytest.fixture(autouse=True, scope="session") def _fast_textual_idle(): """Speed up the b2view (tui) tests by shrinking Textual's idle poll. diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 80bc9ceb0..c3e74793d 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -317,6 +317,7 @@ def test_remote_summary_index_prunes_queries(tmp_path, granularity): @pytest.mark.parametrize("policy", [blosc2.CachePolicy.NONE, blosc2.CachePolicy.MEMORY]) +@pytest.mark.usefixtures("b2z_range_reads") def test_remote_summary_single_payload_request_and_cache(tmp_path, policy): from blosc2.indexing import _open_level_summary_handle @@ -547,6 +548,7 @@ def changed_info(self, path, **kwargs): @pytest.mark.parametrize("max_concurrency", [1, 8]) @pytest.mark.parametrize("legacy", [False, True]) +@pytest.mark.usefixtures("b2z_range_reads") def test_disk_cache_reuses_ctable_bootstrap(tmp_path, monkeypatch, max_concurrency, legacy): schema = dataclasses.make_dataclass("Sample", [("x", float), ("note", str, blosc2.field(blosc2.utf8()))]) values = np.random.default_rng(42).random(20_000) @@ -885,6 +887,7 @@ class Lists: assert remote[remote["values"].overlaps([[3], [4]])]["values"][:] == [[[3]]] +@pytest.mark.usefixtures("b2z_range_reads") def test_remote_ctable_membership_index_avoids_list_payload(tmp_path, monkeypatch): @dataclasses.dataclass class Rows: @@ -969,6 +972,7 @@ class Mixed: remote["category"][:] +@pytest.mark.usefixtures("b2z_range_reads") def test_remote_batch_reads_mutation_lifetime_and_copy(tmp_path, monkeypatch): @dataclasses.dataclass class Mixed: @@ -1135,6 +1139,7 @@ def test_remote_store_returns_table_with_independent_lifetime(tmp_path): np.testing.assert_array_equal(sparse["x"][:], [1, 2]) +@pytest.mark.usefixtures("b2z_range_reads") def test_remote_ctable_reference_save_roundtrip(tmp_path): @dataclasses.dataclass class TextRow: @@ -1508,6 +1513,7 @@ class TextRow: @pytest.mark.parametrize("policy", list(blosc2.CachePolicy)) @pytest.mark.parametrize("blocks", [False, True]) +@pytest.mark.usefixtures("b2z_range_reads") def test_remote_utf8_bounded_transfer(tmp_path, monkeypatch, policy, blocks): from blosc2 import proxy_source from blosc2._utf8_array import UTF8Array @@ -1672,6 +1678,7 @@ def counted(self, path, start=None, end=None, **kwargs): assert store["group/table"]["x"][0] == 1 +@pytest.mark.usefixtures("b2z_range_reads") def test_parallel_metadata_benchmark(tmp_path, monkeypatch): import runpy import time diff --git a/tests/test_b2z_source.py b/tests/test_b2z_source.py index fcbc39eee..89fd3bddc 100644 --- a/tests/test_b2z_source.py +++ b/tests/test_b2z_source.py @@ -29,6 +29,129 @@ def memory_archive(data=None, *, compression=zipfile.ZIP_STORED, zip64=False): return "memory://v10.b2z", data +def test_small_archive_source_cache_shared_across_scopes(tmp_path, monkeypatch): + from blosc2.b2z_source import b2z_source_cache + + url, data = memory_archive() + with blosc2.open(url + "::d0/a", cache_dir=tmp_path) as first: + assert first.traffic.requests == 1 + assert first.traffic.nbytes == len(fsspec.filesystem("memory").cat_file(url)) + path, marker, blob = b2z_source_cache(url, tmp_path) + assert blob is not None + assert marker["size"] == len(blob) + assert list(tmp_path.rglob("*.b2z-source")) == [path] + + def no_network(*args, **kwargs): + pytest.fail("shared source must serve discovery and payload without network") + + fs = fsspec.filesystem("memory") + monkeypatch.setattr(type(fs), "cat_file", no_network) + monkeypatch.setattr(type(fs), "info", no_network) + for dataset, expected in (("d0/a", data), ("d0/b", data[::-1])): + with blosc2.open(url, dataset=dataset, cache_dir=tmp_path) as array: + np.testing.assert_array_equal(array[:], expected) + assert array.traffic.requests == 0 + with blosc2.RemoteStore(url, cache_dir=tmp_path) as store: + np.testing.assert_array_equal(store["d0/a"][:], data) + assert store.traffic.requests == 0 + + +@pytest.mark.parametrize("large", [False, True]) +def test_small_archive_refresh_invalidates_other_scopes(tmp_path, large): + from blosc2.b2z_source import b2z_source_cache + + url, data = memory_archive() + with blosc2.open(url, dataset="d0/a", cache_dir=tmp_path) as array: + np.testing.assert_array_equal(array[:], data) + with blosc2.RemoteStore(url, cache_dir=tmp_path) as store: + changed = data ^ np.uint8(1) + memory_archive(changed) + if large: + fs = fsspec.filesystem("memory") + buffer = io.BytesIO(fs.cat_file(url)) + with zipfile.ZipFile(buffer, "a") as archive: + archive.writestr("padding", bytes(8 << 20)) + fs.pipe_file(url, buffer.getvalue()) + store.refresh() + np.testing.assert_array_equal(store["d0/a"][:], changed) + path, marker, blob = b2z_source_cache(url, tmp_path) + assert (blob is None) == large + assert path.exists() != large + assert (marker["sha256"] is None) == large + with blosc2.open(url, dataset="d0/a", cache_dir=tmp_path) as array: + np.testing.assert_array_equal(array[:], changed) + + +@pytest.mark.parametrize("store", [False, True]) +def test_small_archive_legacy_cache_migration(tmp_path, monkeypatch, store): + import blosc2.b2z_source as bs + + url, data = memory_archive() + with monkeypatch.context() as patch: + patch.setattr(bs, "SMALL_REMOTE_FILE", 0) + if store: + with blosc2.RemoteStore(url, cache_dir=tmp_path) as remote: + np.testing.assert_array_equal(remote["d0/a"][:], data) + else: + with blosc2.open(url, dataset="d0/a", cache_dir=tmp_path) as array: + np.testing.assert_array_equal(array[:], data) + seed = array._carrier.schunk.vlmeta["b2z-frame"] + seed.pop("source_sha256", None) + seed.pop("source_cache_version", None) + array._carrier.schunk.vlmeta["b2z-frame"] = seed + if store: + with blosc2.RemoteStore(url, cache_dir=tmp_path) as remote: + np.testing.assert_array_equal(remote["d0/a"][:], data) + else: + with blosc2.open(url, dataset="d0/a", cache_dir=tmp_path) as array: + np.testing.assert_array_equal(array[:], data) + assert bs.b2z_source_cache(url, tmp_path)[2] is not None + + +@pytest.mark.parametrize("damage", ["missing", "corrupt"]) +def test_small_archive_source_cache_recovers(tmp_path, damage): + from blosc2.b2z_source import b2z_source_cache + + url, data = memory_archive() + with blosc2.RemoteStore(url, cache_dir=tmp_path): + pass + path, _, _ = b2z_source_cache(url, tmp_path) + if damage == "missing": + path.unlink() + else: + path.write_bytes(b"damaged") + with blosc2.RemoteStore(url, cache_dir=tmp_path) as store: + np.testing.assert_array_equal(store["d0/a"][:], data) + assert store.traffic.requests == 1 + assert b2z_source_cache(url, tmp_path)[2] is not None + + +@pytest.mark.parametrize("extra", [0, 1]) +def test_archive_eager_download_threshold(tmp_path, extra): + from blosc2.b2z_source import SMALL_REMOTE_FILE, b2z_source_cache + + url, data = memory_archive() + fs = fsspec.filesystem("memory") + original = fs.cat_file(url) + buffer = io.BytesIO(original) + # ZIP_STORED adds a 30-byte local header and a 46-byte directory entry. + padding = SMALL_REMOTE_FILE + extra - len(original) - 76 - 2 * len("padding") + with zipfile.ZipFile(buffer, "a") as archive: + archive.writestr("padding", bytes(padding)) + assert len(buffer.getvalue()) == SMALL_REMOTE_FILE + extra + fs.pipe_file(url, buffer.getvalue()) + with blosc2.open(url, dataset="d0/a", cache_dir=tmp_path) as array: + assert (array.src._archive.blob is None) == bool(extra) + if not extra: + assert array.traffic.requests == 1 + assert array.traffic.nbytes == SMALL_REMOTE_FILE + else: + assert array.traffic.nbytes < SMALL_REMOTE_FILE + np.testing.assert_array_equal(array[:], data) + assert (b2z_source_cache(url, tmp_path)[2] is None) == bool(extra) + + +@pytest.mark.usefixtures("b2z_range_reads") def test_remote_batch_member_range_experiment(tmp_path, monkeypatch): """A distant BatchArray chunk can be decoded without fetching its member.""" from blosc2.b2z_source import B2ZArchive, member_vlmeta @@ -110,6 +233,7 @@ def test_internal_remote_batch_reader(tmp_path): @pytest.mark.parametrize("address", ["::/d0/a", "/d0/a", "keyword"]) +@pytest.mark.usefixtures("b2z_range_reads") def test_addressing_and_hits(address, monkeypatch): url, data = memory_archive() fs = fsspec.filesystem("memory") @@ -145,6 +269,7 @@ def counted(self, path, start=None, end=None, **kwargs): np.testing.assert_array_equal(arr[-3:, -4:], data[-3:, -4:]) +@pytest.mark.usefixtures("b2z_range_reads") def test_small_member_prefetch_carries_vlmeta(monkeypatch): data = np.random.default_rng(7).integers(0, 256, (200, 200), dtype="uint8") array = blosc2.asarray(data) @@ -173,6 +298,7 @@ def counted(self, path, start=None, end=None, **kwargs): @pytest.mark.parametrize("suffix", ["supported", "ignored", "rejected", "malformed"]) @pytest.mark.parametrize("small", [False, True]) @pytest.mark.parametrize("ctable", [False, True]) +@pytest.mark.usefixtures("b2z_range_reads") def test_http_tail_bootstrap(suffix, small, ctable, tmp_path): import http.server import threading @@ -269,6 +395,7 @@ def do_GET(self): @pytest.mark.parametrize("reopen", ["url", "carrier", "cframe"]) +@pytest.mark.usefixtures("b2z_range_reads") def test_disk_cache_reopen_replays_b2z_bootstrap(tmp_path, monkeypatch, reopen): data = np.random.default_rng(3).integers(0, 256, (1000, 1000), dtype="uint8") array = blosc2.asarray(data, chunks=(200, 250), blocks=(50, 50)) @@ -335,6 +462,7 @@ def test_disk_persistence_and_eviction(tmp_path, limit): assert restored.dataset == "d0/a" +@pytest.mark.usefixtures("b2z_range_reads") def test_policies_exports_and_identity(tmp_path): url, data = memory_archive() none = blosc2.RemoteArray(url, dataset="d0/a") @@ -660,6 +788,7 @@ def test_https_b2z_slice(): @pytest.mark.parametrize("policy", list(blosc2.CachePolicy)) @pytest.mark.parametrize("limit", [1000, 100_000]) +@pytest.mark.usefixtures("b2z_range_reads") def test_whole_member_prefetch_counts_and_obeys_budget(tmp_path, policy, limit): data = np.random.default_rng(123).integers(0, 256, (200, 200), dtype="uint8") array = blosc2.asarray(data, chunks=(100, 100), blocks=(50, 50)) diff --git a/tests/test_fsspec.py b/tests/test_fsspec.py index 912edec80..08a55d784 100644 --- a/tests/test_fsspec.py +++ b/tests/test_fsspec.py @@ -7,6 +7,7 @@ ####################################################################### import contextlib +import dataclasses import functools import gc import hashlib @@ -740,7 +741,10 @@ def do_GET(self): return super().do_GET() body = (root / self.path.lstrip("/")).read_bytes() first, _, last = span.removeprefix("bytes=").partition("-") - first, last = int(first), int(last) if last else len(body) - 1 + if not first: + first, last = max(0, len(body) - int(last)), len(body) - 1 + else: + first, last = int(first), int(last) if last else len(body) - 1 part = body[first : last + 1] self.send_response(206) self.send_header("Content-Range", f"bytes {first}-{last}/{len(body)}") @@ -787,14 +791,25 @@ def test_http_hdf5_scan_and_warm_slice(tmp_path): assert len(requests) == count -def test_http_hdf5_source_cache_across_processes(tmp_path): - h5py = pytest.importorskip("h5py") +@pytest.mark.parametrize("format", ["hdf5", "b2z"]) +def test_http_source_cache_across_processes(tmp_path, format): data = np.zeros(100, dtype=[("id", "i4"), ("value", "f8")]) data["id"] = np.arange(len(data)) - path = tmp_path / "table.h5" - with h5py.File(path, "w") as file: - table = file.create_dataset("table", data=data, chunks=(10,)) - table.attrs["CLASS"] = np.bytes_(b"TABLE") + if format == "hdf5": + h5py = pytest.importorskip("h5py") + path = tmp_path / "table.h5" + with h5py.File(path, "w") as file: + table = file.create_dataset("table", data=data, chunks=(10,)) + table.attrs["CLASS"] = np.bytes_(b"TABLE") + else: + path = tmp_path / "table.b2z" + schema = dataclasses.make_dataclass( + "Row", [("id", int), ("value", float), ("note", str, blosc2.field(blosc2.utf8()))] + ) + rng = np.random.default_rng(42) + rows = [(i, 0.0, rng.bytes(1500).hex()) for i in range(100)] + blosc2.CTable(schema, rows, create_summary_index=False).to_b2z(path) + assert path.stat().st_size > 64 * 1024 cache = tmp_path / "cache" head_requests = [] with _ranged_server(tmp_path, head_requests=head_requests) as (urlbase, requests): @@ -804,12 +819,14 @@ def test_http_hdf5_source_cache_across_processes(tmp_path): "print(t.info if sys.argv[3] == 'info' else t); " "print('requests', t.traffic.requests); t.close()" ) - url = f"{urlbase}/{path.name}::table" + url = f"{urlbase}/{path.name}" + ("::table" if format == "hdf5" else "") result = subprocess.run( [sys.executable, "-c", script, url, str(cache), "info"], capture_output=True, text=True ) assert result.returncode == 0, result.stderr - assert requests == [None] + assert requests == ( + [None] if format == "hdf5" else ["bytes=-8192", f"bytes=0-{path.stat().st_size - 1}"] + ) requests.clear() head_requests.clear() # Observe HTTP requests instead of blocking socket.connect: Windows asyncio diff --git a/tests/test_remote_store.py b/tests/test_remote_store.py index 5e62319bf..6e71ac7e4 100644 --- a/tests/test_remote_store.py +++ b/tests/test_remote_store.py @@ -219,7 +219,16 @@ def test_disk_cache_partitions_storage_options(hierarchy, tmp_path): np.testing.assert_array_equal(a[:2, :2], data[:2, :2]) # Different backends must not share a manifest or its leaf payloads. - assert len([path for path in cache.iterdir() if path.is_dir() and path.name != "hdf5-sources"]) == 2 + assert ( + len( + [ + path + for path in cache.iterdir() + if path.is_dir() and path.name not in {"hdf5-sources", "b2z-sources"} + ] + ) + == 2 + ) def test_artifact_keeps_storage_options_identity(hierarchy, tmp_path): @@ -820,6 +829,7 @@ def test_sparse_store_processes_and_crash(tmp_path): np.testing.assert_array_equal(array[:], np.arange(10000, dtype="i4")) +@pytest.mark.usefixtures("b2z_range_reads") def test_discovery_aliases_sources_and_lifetime(hierarchy, tmp_path, monkeypatch): url, data = hierarchy translations = [] From 8f614d2cbca4fc13b217d0914f6c25350a8c0b3b Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Tue, 22 Sep 2026 18:30:51 +0200 Subject: [PATCH 64/82] Speed up remote CTable tests --- tests/ctable/test_remote_ctable.py | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index c3e74793d..01fee5a50 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -1698,7 +1698,7 @@ class TextRow: original = filesystem_type.cat_file def delayed(self, *args, **kwargs): - time.sleep(0.01) + time.sleep(0.002) return original(self, *args, **kwargs) monkeypatch.setattr(filesystem_type, "cat_file", delayed) @@ -1768,12 +1768,11 @@ class Mixed: y: float = blosc2.field(blosc2.float64()) text: str = blosc2.field(blosc2.utf8(null_storage="mask")) + # Only the first 16 live rows are selected; keep export after deletion cheap. + # Uncompressed columns still exceed the 128-byte buffer and 1 KiB cache budgets. local = blosc2.CTable( Mixed, - [ - (None if i % 7 == 0 else i, i / 3, None if i % 5 == 0 else f"🌦 café 東京 {i}") - for i in range(20000) - ], + [(None if i % 7 == 0 else i, i / 3, None if i % 5 == 0 else f"🌦 café 東京 {i}") for i in range(128)], cparams={"clevel": 0}, create_summary_index=False, ) From 4110e84b2905706c4943e3c7577aaf95f5ff8ab8 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Tue, 22 Sep 2026 18:42:37 +0200 Subject: [PATCH 65/82] Add path alias for dataset selection --- doc/guides/remote_arrays.md | 22 +++-- doc/guides/remote_objects.md | 4 +- doc/guides/remote_tables.md | 2 +- examples/remote/nested-store.py | 2 +- src/blosc2/b2z_source.py | 9 +- src/blosc2/core.py | 19 +++++ src/blosc2/hdf5_source.py | 23 ++++-- src/blosc2/remote_array.py | 11 ++- src/blosc2/remote_ctable.py | 14 +++- src/blosc2/remote_store.py | 10 +++ src/blosc2/schunk.py | 14 +++- tests/test_dataset_path.py | 142 ++++++++++++++++++++++++++++++++ todo/change-dataset-path.md | 40 +++++++++ 13 files changed, 288 insertions(+), 24 deletions(-) create mode 100644 tests/test_dataset_path.py create mode 100644 todo/change-dataset-path.md diff --git a/doc/guides/remote_arrays.md b/doc/guides/remote_arrays.md index bd635e32b..b87df81fb 100644 --- a/doc/guides/remote_arrays.md +++ b/doc/guides/remote_arrays.md @@ -35,11 +35,11 @@ b = blosc2.open( ) # Open individual datasets inside containers (B2Z, Zarr, or HDF5) -# Datasets can be named using slashes (/), container separators (::), or dataset=: +# Datasets can be named using slashes (/), container separators (::), or path=: b1 = blosc2.open("s3://bucket/hierarchy.b2z/d0/d1/a2", lazy=True) c1 = blosc2.open("https://datasets.example.org/hierarchy.zarr::d0/d1/a2", lazy=True) h1 = blosc2.open( - "https://datasets.example.org/hierarchy.h5", lazy=True, dataset="d0/d1/a2" + "https://datasets.example.org/hierarchy.h5", lazy=True, path="d0/d1/a2" ) # Open whole hierarchies with RemoteStore to discover and navigate containers: @@ -53,6 +53,18 @@ a1.shape, a1.dtype # metadata is available immediately a1[100:110, :50] # data is fetched now ``` +`path=` selects a node **inside** the source, not the source URL or local filename. +It is keyword-only and works in `blosc2.open()` (including local containers), +`RemoteArray`, `RemoteStore`, `RemoteCTable`, and the store/table +`with_sparse_cache()` factories. `dataset=` remains supported without deprecation. +If both names are supplied, they must agree after stripping leading and trailing +slashes. `None` leaves selection unspecified; `""` and `"/"` select the container +root (which must match the requested object type). Do not combine a selector +keyword with one embedded in the URL, even if they name the same node. +Both spellings share cache identities and portable artifacts. +The exported `B2ZNDSource`, `HDF5NDSource`, `scan_hdf5_index()` and +`validate_hdf5_index()` APIs also accept `path=` alongside `dataset=`. + Remote B2Z needs `pip install "blosc2[fsspec]"`. HTTP and HTTPS URLs work out of the box; cloud object stores need their respective protocol driver (such as `s3fs` for S3, `gcsfs` for GCS, or `adlfs` for Azure). It accesses external `ZIP_STORED` NDArray members using native Blosc2 chunk and @@ -66,13 +78,13 @@ Embedded leaves and CTable columns are not supported as lazy NDArrays. Remote Zarr needs `pip install "blosc2[zarr,fsspec]"`. HTTP/HTTPS works directly; cloud stores require their protocol driver (`s3fs` for S3, etc.). -Datasets can be named directly by path (`/sub/arr`), with the `::sub/arr` separator, or via `dataset="sub/arr"`. +Datasets can be named directly by path (`/sub/arr`), with the `::sub/arr` separator, or via `path="sub/arr"`. For a suffix-free URL, pass `source_format="zarr"`. Converted Blosc2 chunks are cached under an immutable source contract, so publish changed data at a new URL or replace its cache. Remote HDF5 needs `pip install "blosc2[hdf5,fsspec]"`. HTTP/HTTPS works directly; cloud stores require their protocol driver (`s3fs` for S3, etc.). -Datasets can be specified using standard slash syntax (`file.h5/d0/d1/a2`), the double-colon separator (`file.h5::d0/d1/a2`), or the `dataset="d0/d1/a2"` parameter. +Datasets can be specified using standard slash syntax (`file.h5/d0/d1/a2`), the double-colon separator (`file.h5::d0/d1/a2`), or the `path="d0/d1/a2"` parameter. Remote HDF5 reads cache discovery metadata and fetch dataset chunks as needed. Uncompressed, deflate, shuffle, and Blosc2 pipelines are supported directly; other pipelines use h5py, including filters registered by `hdf5plugin`. @@ -107,7 +119,7 @@ table = blosc2.open( `hdf5_index=` accepts a dictionary, local JSON path, or remote fsspec URL and works for arrays, PyTables tables, and hierarchy stores. `scan_hdf5_index()` -creates a complete container index by default; pass `dataset=` to create an index +creates a complete container index by default; pass `path=` to create an index scoped to one dataset or group subtree. A scoped index can only open that exact dataset, or a store rooted at that group. The recorded source URL must exactly match the URL being opened. Regenerate the diff --git a/doc/guides/remote_objects.md b/doc/guides/remote_objects.md index 85f228633..9acc90b1d 100644 --- a/doc/guides/remote_objects.md +++ b/doc/guides/remote_objects.md @@ -73,10 +73,10 @@ with blosc2.RemoteStore("https://datasets.example.org/data.h5") as store: A `TreeStore` can persist a `RemoteStore` reference at an explicit path. The path is always chosen by the application; it is not derived from the remote filename. The reference may select a complete B2Z, HDF5, or Zarr hierarchy, or a subgroup -selected with `dataset=`: +selected with `path=` (`dataset=` remains a supported alias): ```python -with blosc2.RemoteStore("s3://weather/europe.zarr", dataset="spain") as weather: +with blosc2.RemoteStore("s3://weather/europe.zarr", path="spain") as weather: with blosc2.TreeStore("catalog.b2z", mode="w") as catalog: catalog["/external/weather"] = weather diff --git a/doc/guides/remote_tables.md b/doc/guides/remote_tables.md index 8fa951f65..550dfaaf7 100644 --- a/doc/guides/remote_tables.md +++ b/doc/guides/remote_tables.md @@ -7,7 +7,7 @@ inside a hierarchy can also be opened through `RemoteStore`. `blosc2.open()` dispatches local table archives to `CTable` and remote table archives to `RemoteCTable`. Remote `.b2z` groups return `RemoteStore` by default; -array leaves retain their `RemoteArray` behavior. Use `dataset="group/table"` +array leaves retain their `RemoteArray` behavior. Use `path="group/table"` or a `::group/table` URL suffix to select a nested table. For a complete local download instead, pass `lazy=False, cache_dir="download-cache"`. diff --git a/examples/remote/nested-store.py b/examples/remote/nested-store.py index 62e8c650a..18e8c70b9 100644 --- a/examples/remote/nested-store.py +++ b/examples/remote/nested-store.py @@ -15,7 +15,7 @@ def main(): parser.add_argument("leaf", help="Array below the mounted group") args = parser.parse_args() - with blosc2.RemoteStore(args.url, dataset=args.dataset) as remote: + with blosc2.RemoteStore(args.url, path=args.dataset) as remote: with blosc2.TreeStore("catalog.b2z", mode="w") as tree: tree["/external/weather"] = remote diff --git a/src/blosc2/b2z_source.py b/src/blosc2/b2z_source.py index 0594103a0..3c3094426 100644 --- a/src/blosc2/b2z_source.py +++ b/src/blosc2/b2z_source.py @@ -14,7 +14,7 @@ import zipfile from contextlib import contextmanager -from blosc2.core import _import_fsspec +from blosc2.core import _import_fsspec, resolve_dataset_path from blosc2.proxy_source import REMOTE_MAX_CONCURRENCY, ByteRangeNDSource, Traffic from blosc2.remote_source_cache import ( SMALL_REMOTE_FILE, @@ -368,7 +368,8 @@ def __init__(self, dataset, traffic, blob=None): class B2ZNDSource(ByteRangeNDSource): """Read a stored external NDArray from an immutable B2Z archive via fsspec. - ``dataset`` is a logical tree key, e.g. ``d0/a3``, without the member's + ``path`` (or its supported alias ``dataset``) is a logical tree key, + e.g. ``d0/a3``, without the member's ``.b2nd`` suffix. Embedded leaves and ZIP-compressed members are unsupported. Opening uses bounded metadata prefetch; native chunks and blocks are fetched on demand. Replacing the archive requires replacing its cache. @@ -377,9 +378,10 @@ class B2ZNDSource(ByteRangeNDSource): def __init__( self, urlpath, - dataset, + dataset=None, max_concurrency=REMOTE_MAX_CONCURRENCY, *, + path=None, storage_options=None, _filesystem=None, _traffic=None, @@ -387,6 +389,7 @@ def __init__( _seed=None, _source_cache_dir=None, ): + dataset = resolve_dataset_path(dataset, path) if not isinstance(dataset, str) or not dataset.strip("/"): raise ValueError("B2Z sources require a dataset path (e.g. dataset='d0/a3')") dataset = dataset.strip("/") diff --git a/src/blosc2/core.py b/src/blosc2/core.py index 8acf3ca77..61160cb72 100644 --- a/src/blosc2/core.py +++ b/src/blosc2/core.py @@ -708,6 +708,25 @@ def _parse_b2z_url(urlpath, dataset): return None +def resolve_dataset_path(dataset, path): + """Resolve the public path alias without changing persisted dataset names. + + None means unspecified. Empty strings and slash-only strings select the + root; retain them until URL parsing so duplicate URL selectors still fail. + """ + if path is None: + return dataset + if not isinstance(path, str): + raise TypeError("path must be a string or None") + if dataset is not None: + if not isinstance(dataset, str): + raise TypeError("dataset must be a string or None") + if dataset.strip("/") != path.strip("/"): + raise ValueError("Conflicting dataset and path parameters") + return dataset + return path + + def parse_container_url( urlpath: object, dataset: str | None = None, diff --git a/src/blosc2/hdf5_source.py b/src/blosc2/hdf5_source.py index 6af70670b..971afedc3 100644 --- a/src/blosc2/hdf5_source.py +++ b/src/blosc2/hdf5_source.py @@ -470,6 +470,7 @@ def scan_hdf5_index( storage_options=None, *, dataset=None, + path=None, unsupported=None, traffic=None, _filesystem=None, @@ -479,7 +480,7 @@ def scan_hdf5_index( ): """Build a native byte-range index for a local or remote HDF5 source. - ``dataset`` limits discovery to one dataset, or one group subtree, plus its ancestor groups. The + ``path`` limits discovery to one dataset, or one group subtree, plus its ancestor groups. The returned dictionary is JSON-compatible and can be supplied via ``hdf5_index=`` on later opens. @@ -489,15 +490,18 @@ def scan_hdf5_index( Local path or fsspec URL of the immutable HDF5 source. storage_options: dict, optional Options passed to the fsspec filesystem. - dataset: str, optional + path: str, optional Build a scoped index for this dataset or group subtree. By default, index the complete container. + dataset: str, optional + Supported alias of ``path``; both must agree after stripping outer slashes. Returns ------- dict A JSON-compatible native HDF5 index. """ + dataset = blosc2.core.resolve_dataset_path(dataset, path) urlpath = blosc2.core.normalize_urlpath(os.fspath(urlpath)) dataset = None if dataset is None else str(dataset).strip("/") local = _filesystem is None and (not urlsplit(urlpath).scheme or os.path.isabs(urlpath)) @@ -558,13 +562,14 @@ def load_hdf5_index(index, urlpath, storage_options=None, *, filesystem=None, da return validate_hdf5_index(index, urlpath, dataset=dataset) -def validate_hdf5_index(index, urlpath=None, *, dataset=None): +def validate_hdf5_index(index, urlpath=None, *, dataset=None, path=None): """Validate and return a native HDF5 index. - ``urlpath`` checks the recorded source URL. ``dataset`` additionally checks + ``urlpath`` checks the recorded source URL. ``path`` (or ``dataset``) additionally checks that a scoped index describes the requested dataset. Version-1 and version-2 native indexes are accepted. """ + dataset = blosc2.core.resolve_dataset_path(dataset, path) if not isinstance(index, dict): raise ValueError("Invalid HDF5 index") if index.get("format") != HDF5_INDEX_FORMAT: @@ -852,7 +857,11 @@ def _close_hdf5_file(h5file, raw): class HDF5NDSource(ProxyNDSource): - """Read one immutable HDF5 dataset as Blosc2-compressed logical chunks.""" + """Read one immutable HDF5 dataset as Blosc2-compressed logical chunks. + + Select it with keyword-only ``path`` or the supported ``dataset`` alias. + If both are supplied they must agree after stripping outer slashes. + """ serves_blocks = False # The logical Blosc2 cache chunks are unchanged from the former reader, so @@ -862,8 +871,9 @@ class HDF5NDSource(ProxyNDSource): def __init__( self, urlpath, - dataset: str, + dataset: str | None = None, *, + path: str | None = None, hdf5_index=None, storage_options=None, max_concurrency=REMOTE_MAX_CONCURRENCY, @@ -876,6 +886,7 @@ def __init__( _source_cache_dir=None, _index_explicit=True, ): + dataset = blosc2.core.resolve_dataset_path(dataset, path) urlpath, dataset = self._parse_url(urlpath, dataset) self.urlpath, self.dataset, self.max_concurrency = urlpath, dataset, max_concurrency self._storage_options, self._external_filesystem = storage_options, _filesystem diff --git a/src/blosc2/remote_array.py b/src/blosc2/remote_array.py index a7add89ed..cf9bff007 100644 --- a/src/blosc2/remote_array.py +++ b/src/blosc2/remote_array.py @@ -576,9 +576,14 @@ class RemoteArray(RemoteObject, blosc2.Operand): an fsspec URL. source_format: {None, "blosc2", "zarr", "hdf5", "b2z"}, optional Format of a URL source, inferred from its container suffix when omitted. - dataset: str, optional + path: str, optional Array path within an HDF5, Zarr, or B2Z container. B2Z supports external - NDArray leaves in immutable archives, e.g. ``dataset="d0/a3"``. + NDArray leaves in immutable archives, e.g. ``path="d0/a3"``. + None leaves selection unspecified; ``""`` and ``"/"`` select the root. + Do not combine with a selector embedded in the URL. + dataset: str, optional + Supported alias of ``path``. If both are given, they must agree after + stripping leading/trailing slashes. hdf5_index: dict, str, or path-like, optional Pre-computed native HDF5 index for the dataset, or a local or remote fsspec URL to its JSON encoding. It must match the source URL and dataset @@ -602,6 +607,7 @@ def __init__( source_format: str | None = None, assume_immutable: bool = True, dataset: str | None = None, + path: str | None = None, hdf5_index=None, _carrier=None, _runtime_cache_path=None, @@ -611,6 +617,7 @@ def __init__( _store_owner=None, _runtime_is_mutable: bool = True, ): + dataset = blosc2.core.resolve_dataset_path(dataset, path) if not isinstance(cache_policy, blosc2.CachePolicy): raise TypeError("cache_policy must be a blosc2.CachePolicy instance") assume_immutable = _validate_assume_immutable(assume_immutable) diff --git a/src/blosc2/remote_ctable.py b/src/blosc2/remote_ctable.py index ce77af03b..38fbb824d 100644 --- a/src/blosc2/remote_ctable.py +++ b/src/blosc2/remote_ctable.py @@ -69,6 +69,11 @@ class RemoteCTable(RemoteObject, CTable): ``hdf5_index`` accepts a native index dictionary or a local/remote JSON path for PyTables/HDF5 sources. Supplying one skips HDF5 discovery. + + ``path`` selects the table within the source. ``dataset`` remains a supported + alias; when both are supplied they must agree after stripping outer slashes. + None leaves selection unspecified; an empty string or slash selects the root. + Selector keywords cannot be combined with a selector embedded in the URL. """ def __new__( @@ -76,6 +81,7 @@ def __new__( urlpath=None, *, dataset=None, + path=None, storage_options=None, cache_policy=CACHE_POLICY_DEFAULT, max_cache_bytes=CACHE_POLICY_DEFAULT, @@ -102,6 +108,7 @@ def __new__( store = RemoteStore( urlpath, dataset=dataset, + path=path, storage_options=storage_options, cache_policy=cache_policy, max_cache_bytes=max_cache_bytes, @@ -132,6 +139,7 @@ def with_sparse_cache( runtime_cache_path, *, dataset=None, + path=None, manifest=None, max_cache_bytes=None, carrier=None, @@ -143,7 +151,10 @@ def with_sparse_cache( _manifest_validator=None, _max_nodes=None, ): - """Attach a remote CTable to a sparse disk cache shared across processes.""" + """Attach a remote CTable to a sparse disk cache shared across processes. + + ``path`` and ``dataset`` select the table as in the ordinary constructor. + """ settings = { name: _positive_integer(name, value) for name, value in { @@ -159,6 +170,7 @@ def with_sparse_cache( urlpath, runtime_cache_path, dataset=dataset, + path=path, manifest=manifest, max_cache_bytes=max_cache_bytes, carrier=carrier, diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 7e6c71bc0..7b7a8ca60 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -1371,6 +1371,11 @@ class RemoteStore(RemoteObject): ``hdf5_index`` accepts a native index dictionary or a local/remote JSON path for HDF5 sources. Supplying one skips HDF5 discovery. + + ``path`` selects a group within the source. ``dataset`` remains a supported + alias; when both are supplied they must agree after stripping outer slashes. + None leaves selection unspecified; an empty string or slash selects the root. + Selector keywords cannot be combined with a selector embedded in the URL. """ @classmethod @@ -1415,6 +1420,7 @@ def __init__( urlpath, *, dataset=None, + path=None, storage_options=None, cache_policy=CACHE_POLICY_DEFAULT, max_cache_bytes=CACHE_POLICY_DEFAULT, @@ -1433,6 +1439,7 @@ def __init__( nested_storage_options=None, _b2z_blob=None, ): + dataset = blosc2.core.resolve_dataset_path(dataset, path) if not isinstance(urlpath, (str, os.PathLike)): raise TypeError("RemoteStore requires a remote URL string") urlpath = os.fspath(urlpath) @@ -1668,6 +1675,7 @@ def with_sparse_cache( runtime_cache_path, *, dataset=None, + path=None, manifest=None, max_cache_bytes=None, carrier=None, @@ -1685,9 +1693,11 @@ def with_sparse_cache( allowance. The caller authorizes the supplied filesystem and manifest; no credentials or filesystem objects are persisted. Portable artifacts are exported with ``save`` rather than opened as mutable runtime storage. + ``path`` and ``dataset`` select the group as in the ordinary constructor. """ from blosc2.remote_store_cache import SharedStoreCache, SharedStoreOperation + dataset = blosc2.core.resolve_dataset_path(dataset, path) limit = normalize_cache_limit(blosc2.CachePolicy.DISK, max_cache_bytes) base, root, kind = parse_container_url(urlpath, dataset) validate_persistable_url(base) diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index ff2c70e48..168b6979a 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2509,6 +2509,8 @@ def open( offset: int = 0, dataset: str | None = None, hdf5_index: dict | str | os.PathLike | None = None, + *, + path: str | None = None, **kwargs: dict, ) -> ( blosc2.SChunk @@ -2640,10 +2642,15 @@ def open( storage_options: dict, optional Parameters passed to the underlying ``fsspec`` filesystem when opening an fsspec URL (for instance credentials, endpoint URL, token, client_kwargs, etc.). - dataset: str, optional - Array path within HDF5, Zarr, or B2Z containers (e.g. ``dataset="d0/d1/a2"``). + path: str, optional + Node path within HDF5, Zarr, or B2Z containers (e.g. ``path="d0/d1/a2"``). B2Z and HDF5 also support table and group paths in immutable containers. Requires ``lazy=True``. + None leaves selection unspecified; ``""`` and ``"/"`` select the root. + Do not combine with a selector embedded in the URL. + dataset: str, optional + Supported alias of ``path``. Both may be supplied if they agree after + stripping leading/trailing slashes; conflicting values raise ValueError. hdf5_index: dict | str | PathLike, optional Pre-computed native HDF5 index, or a local path or remote fsspec URL to its JSON encoding. It must match the source HDF5 URL and dataset @@ -2694,7 +2701,7 @@ def open( covers ``.b2nd``, ``.b2f`` and ``.b2e`` only. Remote ``.b2z`` archives default to lazy discovery: table nodes return :class:`RemoteCTable`, groups return :class:`RemoteStore`, and external array leaves return - :ref:`RemoteArray`. Select a nested node with ``dataset=`` or ``::path``. + :ref:`RemoteArray`. Select a nested node with ``path=`` or ``::path``. Tables and groups use ``cache_dir`` for DISK caching and default to MEMORY otherwise; array leaves additionally support ``cache_path``. Use ``lazy=False, cache_dir=...`` to download a complete archive and open @@ -2753,6 +2760,7 @@ def open( >>> all(sc_open.decompress_chunk(i, dest1) == sc_open_mmap.decompress_chunk(i, dest1) for i in range(nchunks)) True """ + dataset = blosc2.core.resolve_dataset_path(dataset, path) _reject_table_buffer_options(kwargs) if isinstance(urlpath, blosc2.URLPath): return _open_c2_urlpath(urlpath, mode, offset, kwargs) diff --git a/tests/test_dataset_path.py b/tests/test_dataset_path.py new file mode 100644 index 000000000..01a3e1047 --- /dev/null +++ b/tests/test_dataset_path.py @@ -0,0 +1,142 @@ +"""Public node selectors remain aliases, not new source identities.""" + +import dataclasses + +import numpy as np +import pytest + +import blosc2 + +fsspec = pytest.importorskip("fsspec") + + +@pytest.fixture +def container(tmp_path): + @dataclasses.dataclass + class Row: + x: int + + filename = tmp_path / "selectors.b2z" + with blosc2.TreeStore(filename, mode="w", threshold=0) as tree: + tree["group/array"] = blosc2.arange(8) + tree["group/table"] = blosc2.CTable(Row, [(1,), (2,)], create_summary_index=False) + url = f"memory://{tmp_path.name}/selectors.b2z" + fsspec.filesystem("memory").pipe(url, filename.read_bytes()) + return filename, url + + +@pytest.mark.parametrize("api", ["open", "array", "store", "table", "sparse_store", "sparse_table"]) +def test_path_alias(container, tmp_path, api): + filename, url = container + target = "group/array" if api in {"open", "array"} else "group/table" if "table" in api else "group" + opener = { + "open": blosc2.open, + "array": blosc2.RemoteArray, + "store": blosc2.RemoteStore, + "table": blosc2.RemoteCTable, + "sparse_store": blosc2.RemoteStore.with_sparse_cache, + "sparse_table": blosc2.RemoteCTable.with_sparse_cache, + }[api] + args = (url, tmp_path / "sparse") if api.startswith("sparse") else (url,) + for options in ({"path": target}, {"dataset": target}, {"path": f"/{target}/", "dataset": target}): + with opener(*args, **options) as obj: + if api in {"open", "array"}: + np.testing.assert_array_equal(obj[:], np.arange(8)) + assert obj.dataset == target + elif "table" in api: + np.testing.assert_array_equal(obj["x"][:], [1, 2]) + else: + assert obj.keys() == ["array", "table"] + with pytest.raises(ValueError, match="Conflicting dataset and path"): + opener(*args, path=target, dataset="different") + with pytest.raises(TypeError, match="path must be a string"): + opener(*args, path=123) + for path in (target, "", "/"): + with pytest.raises(ValueError, match="both URL path and dataset"): + opener(url + "::" + target, *args[1:], path=path) + if api == "open": + with blosc2.open(filename, path=target) as obj: + np.testing.assert_array_equal(obj[:], np.arange(8)) + # The old positional dataset argument keeps its meaning. + with blosc2.open(url, "r", 0, target) as obj: + np.testing.assert_array_equal(obj[:], np.arange(8)) + + +@pytest.mark.parametrize( + "options", [{"path": None}, {"path": ""}, {"path": "/"}, {"path": "/", "dataset": ""}] +) +def test_path_root(container, options): + _, url = container + for opener in (blosc2.open, blosc2.RemoteStore): + with opener(url, **options) as store: + assert store.keys() == ["group"] + + +def test_path_cache_and_artifact_compatibility(container, tmp_path): + _, url = container + cache = tmp_path / "cache" + artifact = tmp_path / "reference.b2nd" + with blosc2.open(url, path="group/array", cache_dir=cache) as array: + np.testing.assert_array_equal(array[:], np.arange(8)) + source = array._source.copy() + array.save(artifact) + with blosc2.open(url, dataset="group/array", cache_dir=cache) as array: + assert array._source == source + np.testing.assert_array_equal(array[:], np.arange(8)) + assert array.traffic.requests == 0 + with blosc2.open(artifact) as array: + assert array._source == source + np.testing.assert_array_equal(array[:], np.arange(8)) + + +def test_path_b2z_source(container): + _, url = container + source = blosc2.B2ZNDSource(url, path="/group/array/", dataset="group/array") + np.testing.assert_array_equal(blosc2.Proxy(source)[:], np.arange(8)) + with pytest.raises(ValueError, match="Conflicting dataset and path"): + blosc2.B2ZNDSource(url, path="group/array", dataset="different") + + +@pytest.mark.parametrize("format", ["hdf5", "zarr2", "zarr3"]) +def test_path_other_formats(tmp_path, format): + data = np.arange(8) + if format == "hdf5": + h5py = pytest.importorskip("h5py") + filename = tmp_path / "selectors.h5" + with h5py.File(filename, "w") as file: + file.create_dataset("group/array", data=data, chunks=(4,)) + index = blosc2.scan_hdf5_index(filename, path="group") + assert index == blosc2.scan_hdf5_index(filename, dataset="group") + assert blosc2.validate_hdf5_index(index, path="group") is index + source = blosc2.HDF5NDSource(filename, path="group/array") + try: + np.testing.assert_array_equal(blosc2.Proxy(source)[:], data) + finally: + source.close() + for opener, args in ( + (blosc2.HDF5NDSource, (filename,)), + (blosc2.scan_hdf5_index, (filename,)), + (blosc2.validate_hdf5_index, (index,)), + ): + with pytest.raises(ValueError, match="Conflicting dataset and path"): + opener(*args, path="group/array", dataset="different") + else: + zarr = pytest.importorskip("zarr") + filename = tmp_path / "selectors.zarr" + group = zarr.open_group(filename, mode="w", zarr_format=int(format[-1])) + group.create_group("group").create_array("array", data=data, chunks=(4,)) + with blosc2.open(filename, path="group/array") as array: + np.testing.assert_array_equal(array[:], data) + url = f"memory://{tmp_path.name}/{filename.name}" + fs = fsspec.filesystem("memory") + if filename.is_file(): + fs.pipe(url, filename.read_bytes()) + else: + for entry in filename.rglob("*"): + if entry.is_file(): + fs.pipe(url + "/" + entry.relative_to(filename).as_posix(), entry.read_bytes()) + for opener in (blosc2.open, blosc2.RemoteArray): + with opener(url, path="/group/array/", dataset="group/array") as array: + np.testing.assert_array_equal(array[:], data) + with blosc2.RemoteStore(url, path="group") as store: + assert store.keys() == ["array"] diff --git a/todo/change-dataset-path.md b/todo/change-dataset-path.md new file mode 100644 index 000000000..f31b546e8 --- /dev/null +++ b/todo/change-dataset-path.md @@ -0,0 +1,40 @@ +# Add path= as an alias for dataset= + +Status: implemented. + +Add `path=` as a backward-compatible alias for `dataset=` in remote opening +APIs. The selected node can be a group, table or array; `path=` describes this +more clearly than `dataset=`. + +Example: + +```python +store = blosc2.RemoteStore( + "https://example.org/weather.zarr", + path="europe/spain", +) +``` + +- Keep existing `dataset=` calls working; no removal or deprecation is planned. +- Apply the alias consistently across relevant constructors, opening helpers + and sparse-cache factories. +- Define and test behavior when both names are supplied; reject conflicting + values with a clear error. +- Normalize to the existing internal representation so persisted source + descriptors and artifacts remain compatible. +- Prefer `path=` in documentation and examples, explaining that it selects a + node within the source rather than the source URL itself. + +Implementation details: + +- `path` is keyword-only; existing positional `dataset` arguments are unchanged. +- `None` is unspecified; `""` and `"/"` explicitly select the root. Both names + may be supplied when equal after stripping outer slashes. +- URL-selector conflict rules remain unchanged. +- Supported by `open` (including local containers), remote array/store/table + constructors, store/table sparse-cache factories, B2Z/HDF5 source constructors, + and HDF5 index scanning/validation helpers. +- One boundary resolver preserves internal `dataset` names, cache identities, + and persisted artifacts. Regression tests are in `tests/test_dataset_path.py`. + +Related plan: [Nested remote stores](../plans/remote-nested-store.md). From f90c83ee3a2e86757242a08c2567788c228bce98 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Wed, 23 Sep 2026 07:30:41 +0200 Subject: [PATCH 66/82] Clarify shared remote cache usage --- doc/guides/remote_objects.md | 33 +++++++++++++++++++++++++++++++-- 1 file changed, 31 insertions(+), 2 deletions(-) diff --git a/doc/guides/remote_objects.md b/doc/guides/remote_objects.md index 9acc90b1d..ef7982822 100644 --- a/doc/guides/remote_objects.md +++ b/doc/guides/remote_objects.md @@ -105,16 +105,45 @@ Specify `cache_dir` when creating a `RemoteStore` to persist discovery metadata with blosc2.RemoteStore( "https://datasets.example.org/data.h5", cache_dir="./b2store_cache", - max_cache_bytes=512 * 2**20, # 512 MiB shared disk limit + max_cache_bytes=512 * 2**20, # 512 MiB aggregate payload limit ) as store: temp = store["experiment/temperature"] values = temp[:100] ``` When reopening the same store later with the same `cache_dir`: + - Discovery metadata (such as B2Z member offsets or native HDF5 indexes) is restored from local disk, avoiding repeated remote scans. `store.metadata_bytes` reports the encoded manifest size. - Retained leaf chunks are available immediately from disk without network transfers. -- Single-owner locks ensure that concurrent processes do not corrupt the shared cache. + +Ordinary `RemoteStore` and `RemoteCTable` disk caches have **exclusive ownership**. +Another process can reuse the same cache entry after its owner and dependent +handles close, but opening that entry while it is still owned raises +`RuntimeError: RemoteStore cache is already owned`. This also applies to stores +and tables opened through `blosc2.open(..., cache_dir=...)`. + +### Sharing a cache between simultaneous processes + +Use `with_sparse_cache()` when multiple processes need to keep the same store +or table open: + +```python +with blosc2.RemoteCTable.with_sparse_cache( + "https://datasets.example.org/data.h5", + "./shared-table-cache", + path="readings", +) as table: + print(table.info) +``` + +For hierarchies, use `RemoteStore.with_sparse_cache()` instead. These constructors +use operation-scoped locks: handles can coexist across processes, but operations +on the same store serialize. Every process using that cache must use the shared +constructor; do not mix it with ordinary `cache_dir=` access. Use a separate +directory for the shared cache. + +See {doc}`../reference/remotestore` ("Shared sparse runtime caches") for locking +and refresh details, and {doc}`remote_tables` for table usage. ### Lifetime and clean shutdown From 39436f2804f506aaab4c60673f660aa00cdc0c4c Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Wed, 23 Sep 2026 08:03:53 +0200 Subject: [PATCH 67/82] Speed up remote HDF5 table info --- src/blosc2/ctable.py | 27 ++++++++++++++++----------- tests/ctable/test_remote_ctable.py | 5 +++-- 2 files changed, 19 insertions(+), 13 deletions(-) diff --git a/src/blosc2/ctable.py b/src/blosc2/ctable.py index 8a63a5939..a9d2c6514 100644 --- a/src/blosc2/ctable.py +++ b/src/blosc2/ctable.py @@ -15083,17 +15083,22 @@ def info_items(self) -> list[tuple[str, object]]: else: column_summary[name] = _InfoLiteral(dtype_label) - index_summary = {} - for idx in self.indexes: - stale = " stale" if idx.stale else "" - label = f" name={idx.name!r}" if idx.name and idx.name != "__self__" else "" - stats = idx.storage_stats() - if stats is None: - suffix = "(size=n/a, sidecars not directly addressable)" - else: - _, cbytes, cratio = stats - suffix = f"(cbytes: {format_nbytes_human(cbytes)}, cratio: {cratio:.2f}x)" - index_summary[idx.col_name] = f"[{idx.kind}{stale}{label}] {suffix}" + remote_storage = self._remote_read_storage() + if remote_storage is not None and remote_storage._owner.format == "hdf5": + remote_storage._owner.ensure_pytables_indexes(remote_storage._root_key) + index_summary = dict.fromkeys(remote_storage._metadata().get("pytables_indexes", {}), "[opsi]") + else: + index_summary = {} + for idx in self.indexes: + stale = " stale" if idx.stale else "" + label = f" name={idx.name!r}" if idx.name and idx.name != "__self__" else "" + stats = idx.storage_stats() + if stats is None: + suffix = "(size=n/a, sidecars not directly addressable)" + else: + _, cbytes, cratio = stats + suffix = f"(cbytes: {format_nbytes_human(cbytes)}, cratio: {cratio:.2f}x)" + index_summary[idx.col_name] = f"[{idx.kind}{stale}{label}] {suffix}" compression_available = all(column_cbytes_for_info(col) is not None for col in self._cols.values()) items = [ diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 01fee5a50..61ecd7a49 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -120,8 +120,9 @@ def test_remote_pytables_info_omits_shared_column_sizes(tmp_path): assert "cbytes" not in column_info assert "cratio" not in column_info assert "nbytes" in column_info - assert "[opsi]" in info["indexes"]["id"] - assert "cbytes:" in info["indexes"]["id"] + assert info["indexes"]["id"] == "[opsi]" + assert getattr(table, "_cached_index_catalog", None) is None + np.testing.assert_array_equal(table.where("id < 3").id[:], data["id"][data["id"] < 3]) def test_remote_pytables_full_index_is_native_opsi(): From 95723bed91280ef26ecce652448bc74600271bf7 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Wed, 23 Sep 2026 09:55:30 +0200 Subject: [PATCH 68/82] Merge remote PyTables index reads --- src/blosc2/ctable_storage.py | 8 +++- src/blosc2/hdf5_source.py | 74 ++++++++++++++++++++++++++---- tests/ctable/test_remote_ctable.py | 44 ++++++++++++++++-- 3 files changed, 110 insertions(+), 16 deletions(-) diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index 4380ef7bb..57db8467e 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -1001,6 +1001,7 @@ def load_index_catalog(self) -> dict: return catalog def _load_pytables_index_catalog(self) -> dict: + from blosc2.hdf5_source import read_pytables_index_arrays from blosc2.indexing import _build_descriptor, _field_target_descriptor, _store_array_sidecar self._owner.ensure_pytables_indexes(self._root_key) @@ -1018,8 +1019,11 @@ def _load_pytables_index_catalog(self) -> dict: self._arrays.append(array) opened.append(array) tail = int(source["tail"]) - values = np.concatenate((opened[0][:].reshape(-1), opened[2][:tail])) - raw_positions = np.concatenate((opened[1][:].reshape(-1), opened[3][:tail])) + sorted_values, sorted_positions, tail_values, tail_positions = read_pytables_index_arrays( + opened, tail + ) + values = np.concatenate((sorted_values.reshape(-1), tail_values)) + raw_positions = np.concatenate((sorted_positions.reshape(-1), tail_positions)) if raw_positions.dtype.kind != "u" or raw_positions.itemsize != 8: continue if len(raw_positions) and int(raw_positions.max()) >= len(column): diff --git a/src/blosc2/hdf5_source.py b/src/blosc2/hdf5_source.py index 971afedc3..42a46db9b 100644 --- a/src/blosc2/hdf5_source.py +++ b/src/blosc2/hdf5_source.py @@ -757,6 +757,60 @@ def scan_pytables_indexes( _close_owned_filesystem(fs) +def read_pytables_index_arrays(arrays, tail): + """Read a PyTables index with bounded, merged source ranges.""" + sources = [array.src for array in arrays] + if any( + source._local or source._blob is not None or not source._metadata["direct"] for source in sources + ): + return [arrays[0][:], arrays[1][:], arrays[2][:tail], arrays[3][:tail]] + + shapes = [source.shape if i < 2 else (tail,) for i, source in enumerate(sources)] + outputs = [ + np.full(shape, _from_json_value(source._metadata["fill_value"]), dtype=source.dtype) + for source, shape in zip(sources, shapes, strict=True) + ] + chunks = [] + for source, output in zip(sources, outputs, strict=True): + for record in source._metadata["allocated"]: + offsets = tuple(record["offset"]) + if offsets[0] >= output.shape[0]: + continue + selection = tuple( + slice(offset, min(offset + length, extent)) + for offset, length, extent in zip(offsets, source.chunks, output.shape, strict=True) + ) + start = record["byte_offset"] + chunks.append((start, start + record["size"], source, output, offsets, selection)) + chunks.sort(key=lambda chunk: chunk[0]) + + def read_group(start, end, members): + data = sources[0]._filesystem.cat_file(sources[0]._path, start=start, end=end) + if len(data) != end - start: + raise OSError(f"Short PyTables index read: expected {end - start} bytes, got {len(data)}") + if sources[0].traffic is not None: + sources[0].traffic.charge(len(data)) + for chunk_start, chunk_end, source, output, offsets, selection in members: + payload = data[chunk_start - start : chunk_end - start] + output[selection] = source._direct_values(offsets, selection, data=payload) + + start = end = None + members = [] + for chunk in chunks: + chunk_start, chunk_end = chunk[:2] + if members and (chunk_start - end > 64 << 10 or max(end, chunk_end) - start > 8 << 20): + read_group(start, end, members) + members = [] + if not members: + start, end = chunk_start, chunk_end + else: + end = max(end, chunk_end) + members.append(chunk) + if members: + read_group(start, end, members) + return outputs + + # Kept while callers migrate from the old internal name. def available_datasets(url, storage_options: dict | None = None) -> list[str]: """Return all dataset paths in an HDF5 file or native index.""" @@ -1109,8 +1163,9 @@ def _open_fallback(self): self._fallback_finalizer = weakref.finalize(self, _close_hdf5_file, h5file, raw) return h5file[self.dataset or "/"] - def _direct_values(self, offsets, selection): + def _direct_values(self, offsets, selection, data=None): valid_shape = tuple(item.stop - item.start for item in selection) + prefetched = data is not None # Count in-flight reads so close() can wait for them without serializing # independent direct fetches against each other. The closed check comes # first so sparse fill chunks obey the same contract as allocated ones. @@ -1124,18 +1179,19 @@ def _direct_values(self, offsets, selection): blob = self._blob self._active_reads += 1 try: - if blob is not None: - start = record["byte_offset"] - data = blob[start : start + record["size"]] - else: - data = filesystem.cat_file( - self._path, start=record["byte_offset"], end=record["byte_offset"] + record["size"] - ) + if not prefetched: + if blob is not None: + start = record["byte_offset"] + data = blob[start : start + record["size"]] + else: + data = filesystem.cat_file( + self._path, start=record["byte_offset"], end=record["byte_offset"] + record["size"] + ) finally: with self._lifecycle: self._active_reads -= 1 self._lifecycle.notify_all() - if blob is None and self.traffic is not None: + if not prefetched and blob is None and self.traffic is not None: self.traffic.charge(len(data)) if len(data) != record["size"]: raise OSError(f"Short HDF5 chunk read for {self.dataset!r} at {offsets}") diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 61ecd7a49..d9cfab486 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -42,7 +42,13 @@ def remote_table_url(tmp_path, table, name="table"): def pytables_hdf5_url( - name="pytables-table.h5", *, indexed=False, indexed_field="id", index_dtype="u8", indexed_rows=21 + name="pytables-table.h5", + *, + indexed=False, + indexed_field="id", + index_dtype="u8", + indexed_rows=21, + padding=0, ): h5py = pytest.importorskip("h5py") rows = [(1, 1.5, b"one"), (2, 2.5, b"two"), (3, 3.5, b"three")] @@ -77,12 +83,24 @@ def pytables_hdf5_url( ) offsets = np.arange(0, regular, 16)[:, None] tail_order = np.argsort(data[indexed_field][regular:], kind="stable") - group.create_dataset("sorted", data=regular_values) - group.create_dataset("indices", data=(regular_order + offsets).astype(index_dtype)) - sorted_lr = group.create_dataset("sortedLR", data=data[indexed_field][regular:][tail_order]) - indices_lr = group.create_dataset("indicesLR", data=(tail_order + regular).astype(index_dtype)) + group.create_dataset("sorted", data=regular_values, chunks=(1, 16) if padding else None) + group.create_dataset( + "indices", + data=(regular_order + offsets).astype(index_dtype), + chunks=(1, 16) if padding else None, + ) + sorted_lr = group.create_dataset( + "sortedLR", data=data[indexed_field][regular:][tail_order], chunks=(1,) if padding else None + ) + indices_lr = group.create_dataset( + "indicesLR", + data=(tail_order + regular).astype(index_dtype), + chunks=(1,) if padding else None, + ) sorted_lr.attrs["nelements"] = np.int32(len(tail_order)) indices_lr.attrs["nelements"] = np.int32(len(tail_order)) + if padding: + h5file.create_dataset("padding", data=np.zeros(padding, dtype="u1")) url = f"memory://{name}" fsspec.filesystem("memory").pipe(url, buffer.getvalue()) return url, data @@ -135,6 +153,22 @@ def test_remote_pytables_full_index_is_native_opsi(): np.testing.assert_array_equal(table.where("(id >= 5) & (id < 9)").label[:], expected) +def test_remote_pytables_index_merged_ranges(): + url, data = pytables_hdf5_url( + "pytables-merged-index.h5", indexed=True, indexed_rows=2049, padding=9 << 20 + ) + with blosc2.RemoteCTable(url, dataset="table") as table: + owner = table._storage._owner + owner.ensure_pytables_indexes("table") + for path in owner.hdf5_index["datasets"]["table"]["pytables_indexes"]["id"].values(): + if isinstance(path, str): + owner.ensure_hdf5_allocations(path) + table.traffic.reset() + assert table._get_index_catalog()["id"]["kind"] == "opsi" + assert table.traffic.requests <= 2 + np.testing.assert_array_equal(table.where("id < 3").id[:], data["id"][data["id"] < 3]) + + def test_remote_pytables_light_index_falls_back_to_scan(): url, data = pytables_hdf5_url("pytables-light.h5", indexed=True, index_dtype="u1") with blosc2.RemoteCTable(url, dataset="table") as table: From 7a1449bda840a17806413b878187fbdf45c6e993 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Wed, 23 Sep 2026 10:43:02 +0200 Subject: [PATCH 69/82] Batch remote PyTables allocation scans --- src/blosc2/ctable_storage.py | 4 ++- src/blosc2/hdf5_source.py | 44 +++++++++++++++++++++++++++--- src/blosc2/remote_store.py | 19 +++++++++++++ tests/ctable/test_remote_ctable.py | 18 ++++++++++-- 4 files changed, 78 insertions(+), 7 deletions(-) diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index 57db8467e..66bfd711a 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -1012,9 +1012,11 @@ def _load_pytables_index_catalog(self) -> dict: if cached is not None: catalog[name] = cached continue + keys = ("sorted", "indices", "sortedLR", "indicesLR") + self._owner.ensure_hdf5_allocations_many(source[key] for key in keys) opened = [] try: - for key in ("sorted", "indices", "sortedLR", "indicesLR"): + for key in keys: array = self._owner.remote_array(source[key]) self._arrays.append(array) opened.append(array) diff --git a/src/blosc2/hdf5_source.py b/src/blosc2/hdf5_source.py index 42a46db9b..054f38b86 100644 --- a/src/blosc2/hdf5_source.py +++ b/src/blosc2/hdf5_source.py @@ -302,8 +302,8 @@ def tell(self): @contextlib.contextmanager -def _open_hdf5_file(path, *, local=False, filesystem=None, traffic=None, blob=None): - """Open one HDF5 file from local storage, retained bytes, or exact ranges.""" +def _open_hdf5_file(path, *, local=False, filesystem=None, traffic=None, blob=None, buffered=False): + """Open one HDF5 file from local storage, retained bytes, or remote ranges.""" import h5py with contextlib.ExitStack() as stack: @@ -312,8 +312,23 @@ def _open_hdf5_file(path, *, local=False, filesystem=None, traffic=None, blob=No elif local: raw = stack.enter_context(open(path, "rb")) else: - raw = stack.enter_context(filesystem.open(path, "rb", block_size=1, cache_type="none")) - if traffic is not None: + options = ( + {"block_size": 64 << 10, "cache_type": "blockcache", "cache_options": {"maxblocks": 32}} + if buffered + else {"block_size": 1, "cache_type": "none"} + ) + raw = stack.enter_context(filesystem.open(path, "rb", **options)) + cache = getattr(raw, "cache", None) if buffered else None + if traffic is not None and cache is not None and hasattr(cache, "fetcher"): + fetcher = cache.fetcher + + def counted_fetch(start, end): + data = fetcher(start, end) + traffic.charge(len(data)) + return data + + cache.fetcher = counted_fetch + elif traffic is not None: raw = _CountingFile(raw, traffic) yield stack.enter_context(h5py.File(raw, "r")) @@ -712,6 +727,27 @@ def scan_hdf5_allocations( _close_owned_filesystem(fs) +def scan_hdf5_allocations_many( + urlpath, datasets, storage_options=None, *, traffic=None, _filesystem=None, _blob=None +): + """Scan several HDF5 allocation maps through one bounded metadata cache.""" + urlpath = blosc2.core.normalize_urlpath(os.fspath(urlpath)) + local = _filesystem is None and (not urlsplit(urlpath).scheme or os.path.isabs(urlpath)) + fs = None + path = urlpath + try: + if not local: + check_hdf5_dependencies() + fs, path = _filesystem_and_path(urlpath, storage_options, _filesystem) + with _open_hdf5_file( + path, local=local, filesystem=fs, traffic=traffic, blob=_blob, buffered=True + ) as h5file: + return {dataset: _allocated_chunks(h5file[dataset]) for dataset in datasets} + finally: + if fs is not None and _filesystem is None: + _close_owned_filesystem(fs) + + def scan_pytables_indexes( urlpath, table_path, table_metadata, storage_options=None, *, traffic=None, _filesystem=None, _blob=None ): diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 7b7a8ca60..5b7f37ee6 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -551,6 +551,25 @@ def ensure_hdf5_allocations(self, path): self.save_manifest() return metadata + def ensure_hdf5_allocations_many(self, paths): + """Populate deferred HDF5 allocation maps in one scan.""" + from blosc2.hdf5_source import scan_hdf5_allocations_many + + with self.lock: + missing = [path for path in paths if self.hdf5_index["datasets"][path]["allocated"] is None] + if missing: + allocations = scan_hdf5_allocations_many( + self.urlpath, + missing, + self.storage_options, + traffic=self.traffic, + _filesystem=self.filesystem, + _blob=self.hdf5_blob, + ) + for path, allocated in allocations.items(): + self.hdf5_index["datasets"][path]["allocated"] = allocated + self.save_manifest() + def attach_hdf5_source_cache(self, path, marker): self.hdf5_source_cache_path = path self.hdf5_source_cache_marker = marker diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index d9cfab486..51229b599 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -143,12 +143,26 @@ def test_remote_pytables_info_omits_shared_column_sizes(tmp_path): np.testing.assert_array_equal(table.where("id < 3").id[:], data["id"][data["id"] < 3]) -def test_remote_pytables_full_index_is_native_opsi(): - url, data = pytables_hdf5_url("pytables-indexed.h5", indexed=True) +def test_remote_pytables_full_index_is_native_opsi(monkeypatch): + from blosc2 import hdf5_source + + scans = [] + scan = hdf5_source.scan_hdf5_allocations_many + + def count_scan(*args, **kwargs): + scans.append(args[1]) + return scan(*args, **kwargs) + + monkeypatch.setattr(hdf5_source, "scan_hdf5_allocations_many", count_scan) + url, data = pytables_hdf5_url( + "pytables-batched-allocations.h5", indexed=True, indexed_rows=2049, padding=9 << 20 + ) with blosc2.RemoteCTable(url, dataset="table", cache_policy=blosc2.CachePolicy.MEMORY) as table: descriptor = table._get_index_catalog()["id"] assert descriptor["kind"] == "opsi" assert descriptor["opsi"]["values_path"] is None + paths = table._storage._owner.hdf5_index["datasets"]["table"]["pytables_indexes"]["id"] + assert scans == [[paths[key] for key in ("sorted", "indices", "sortedLR", "indicesLR")]] expected = data["label"][(data["id"] >= 5) & (data["id"] < 9)] np.testing.assert_array_equal(table.where("(id >= 5) & (id < 9)").label[:], expected) From f06ce727369925f549ae4c935bed496a1d26cc28 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Wed, 23 Sep 2026 11:12:08 +0200 Subject: [PATCH 70/82] Avoid duplicate HDF5 table cache directory --- src/blosc2/core.py | 4 ++- src/blosc2/remote_array.py | 54 +++++++++++++++++++++++------- src/blosc2/remote_store.py | 23 ++++++------- src/blosc2/remote_store_cache.py | 8 +++-- src/blosc2/schunk.py | 15 +++++++-- tests/ctable/test_remote_ctable.py | 12 +++++++ 6 files changed, 86 insertions(+), 30 deletions(-) diff --git a/src/blosc2/core.py b/src/blosc2/core.py index 61160cb72..a9abeda71 100644 --- a/src/blosc2/core.py +++ b/src/blosc2/core.py @@ -824,6 +824,7 @@ def fsspec_cache_path( *, dataset: str | None = None, storage_options: dict | None = None, + create_parent: bool = True, ) -> str: """Readable source directory and optional dataset path, creating parent directories.""" identity = urlpath @@ -841,7 +842,8 @@ def fsspec_cache_path( path = directory.joinpath(*parts) else: path = directory - path.parent.mkdir(parents=True, exist_ok=True) + if create_parent: + path.parent.mkdir(parents=True, exist_ok=True) return str(path) diff --git a/src/blosc2/remote_array.py b/src/blosc2/remote_array.py index cf9bff007..7d9ba47e3 100644 --- a/src/blosc2/remote_array.py +++ b/src/blosc2/remote_array.py @@ -616,6 +616,7 @@ def __init__( _source_cparams=None, _store_owner=None, _runtime_is_mutable: bool = True, + _defer_cache: bool = False, ): dataset = blosc2.core.resolve_dataset_path(dataset, path) if not isinstance(cache_policy, blosc2.CachePolicy): @@ -643,7 +644,13 @@ def __init__( and hdf5_index is None ): shared_index_path = Path( - fsspec_cache_path(urlpath, cache_dir, ".hdf5-index.b2", storage_options=storage_options) + fsspec_cache_path( + urlpath, + cache_dir, + ".hdf5-index.b2", + storage_options=storage_options, + create_parent=not _defer_cache, + ) ) if self._authorized_source: self.src, self._source = _validate_authorized_source( @@ -665,7 +672,9 @@ def __init__( ): # A DISK carrier already holds the container bootstrap from a # previous run; reuse it rather than redoing the remote discovery. - path = self._carrier_path(cache_dir, cache_path, urlpath, storage_options) + path = self._carrier_path( + cache_dir, cache_path, urlpath, storage_options, create_parent=not _defer_cache + ) if os.path.exists(path): with contextlib.suppress(Exception): cached = blosc2.blosc2_ext.open(path, "r", 0, dparams=blosc2.DParams(nthreads=1)) @@ -715,22 +724,34 @@ def __init__( self._runtime_is_mutable = _runtime_is_mutable self._mutable = False - self._initialize_runtime_cache(cache_dir, cache_path, _runtime_cache_path) - - self._publish_source_cache() + self._deferred_cache = ( + cache_dir, + cache_path, + _runtime_cache_path, + shared_index_path, + hdf5_index is None, + ) + if not _defer_cache: + self._complete_deferred_cache() - _publish_hdf5_index(shared_index_path, self._carrier, scanned=hdf5_index is None) + if _store_owner is not None: + _store_owner.acquire() + self._store_owner = _store_owner + self._store_finalizer = weakref.finalize(self, _store_owner.release) + def _complete_deferred_cache(self): + if self._deferred_cache is None: + return + cache_dir, cache_path, runtime_cache_path, shared_index_path, scanned = self._deferred_cache + self._initialize_runtime_cache(cache_dir, cache_path, runtime_cache_path) + self._publish_source_cache() + _publish_hdf5_index(shared_index_path, self._carrier, scanned=scanned) if self._carrier is not None: if self._cached_meta is None: self._cached_meta = self._meta_from_carrier(self._carrier) if self._cached_vlmeta is None: self._cached_vlmeta = read_b2object_user_vlmeta(self._carrier) - - if _store_owner is not None: - _store_owner.acquire() - self._store_owner = _store_owner - self._store_finalizer = weakref.finalize(self, _store_owner.release) + self._deferred_cache = None def _publish_source_cache(self): if not self._authorized_source and isinstance(self.src, blosc2.B2ZNDSource): @@ -828,7 +849,9 @@ def _geometry(src): tuple(src.blocks), ) - def _carrier_path(self, cache_dir, cache_path, urlpath=None, storage_options=None): + def _carrier_path( + self, cache_dir, cache_path, urlpath=None, storage_options=None, *, create_parent=True + ): if cache_path is not None: path = os.fspath(cache_path) if os.path.isdir(path): @@ -841,7 +864,12 @@ def _carrier_path(self, cache_dir, cache_path, urlpath=None, storage_options=Non parsed = urlsplit(urlpath) urlpath = urlunsplit(parsed._replace(path=parsed.path.rstrip("/")[: -len(self._dataset) - 1])) return fsspec_cache_path( - urlpath, cache_dir, ".b2nd", dataset=self._dataset, storage_options=storage_options + urlpath, + cache_dir, + ".b2nd", + dataset=self._dataset, + storage_options=storage_options, + create_parent=create_parent, ) def _open_or_create_carrier(self, cache_dir, cache_path): diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 5b7f37ee6..9e7b650ea 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -1418,6 +1418,15 @@ def _try_open_artifact( } return cls._open_artifact(urlpath, **kwargs) + @staticmethod + def _cache_source(urlpath, dataset, source_format, storage_options): + base_url, root, kind = parse_container_url(urlpath, dataset) + source = {"urlpath": base_url, "dataset": (root or "").strip("/"), "kind": source_format or kind} + fingerprint = storage_options_fingerprint(storage_options) + if fingerprint: + source["storage_options"] = fingerprint + return source + @staticmethod def _validate_cache_config(cache_policy, max_cache_bytes, cache_dir): if cache_policy is CACHE_POLICY_DEFAULT: @@ -1481,17 +1490,7 @@ def __init__( if cache_policy is blosc2.CachePolicy.DISK: from blosc2.remote_store_cache import StoreDiskCache - base_url, root, kind = parse_container_url(urlpath, dataset) - source = { - "urlpath": base_url, - "dataset": (root or "").strip("/"), - "kind": _source_format or kind, - } - fingerprint = storage_options_fingerprint(storage_options) - if fingerprint: - # The same URL through another endpoint or account must not - # reuse this manifest and its leaf payloads. - source["storage_options"] = fingerprint + source = self._cache_source(urlpath, dataset, _source_format, storage_options) disk = StoreDiskCache(cache_dir, source) try: manifest = disk.load() if disk is not None else manifest @@ -1501,7 +1500,7 @@ def __init__( source_cache_path, source_cache_marker, _hdf5_blob, hdf5_index, manifest = ( prepare_hdf5_source_cache( base_url, - root or None, + source["dataset"] or None, cache_dir, storage_options, hdf5_index, diff --git a/src/blosc2/remote_store_cache.py b/src/blosc2/remote_store_cache.py index 3b217e77b..aa4f36246 100644 --- a/src/blosc2/remote_store_cache.py +++ b/src/blosc2/remote_store_cache.py @@ -42,10 +42,14 @@ def lock_cache_file(file, *, blocking=False): class StoreDiskCache: + @staticmethod + def path_for(parent, source): + identity = msgpack.packb(source, use_bin_type=True) + return Path(parent) / cache_directory_name(source["urlpath"], identity) + def __init__(self, parent, source, *, blocking=False): self.source = source - identity = msgpack.packb(source, use_bin_type=True) - self.path = Path(parent) / cache_directory_name(source["urlpath"], identity) + self.path = self.path_for(parent, source) self.path.mkdir(parents=True, exist_ok=True) self.file = (self.path / "owner.lock").open("a+b") try: diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index 168b6979a..b2a093b74 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2384,10 +2384,19 @@ def _open_remote_hdf5(urlpath, options): hdf5_index = options.get("hdf5_index") hdf5_blob = None traffic = None - if dataset: - array = blosc2.RemoteArray(urlpath, **options) + cached_store = False + if dataset and options["cache_dir"] is not None and hdf5_index is None: + from blosc2.remote_store_cache import StoreDiskCache + + source = blosc2.RemoteStore._cache_source(urlpath, dataset, "hdf5", options.get("storage_options")) + cached_store = ( + StoreDiskCache.path_for(options["cache_dir"], source) / "active_generation.json" + ).exists() + if dataset and not cached_store: + array = blosc2.RemoteArray(urlpath, _defer_cache=True, **options) metadata = array.src._hdf5_index["datasets"][array.dataset] if metadata.get("kind") != "ctable": + array._complete_deferred_cache() return array hdf5_index = array.src._hdf5_index hdf5_blob = array.src._blob @@ -2416,6 +2425,8 @@ def _open_remote_hdf5(urlpath, options): full, **({} if max_concurrency is None else {"max_concurrency": max_concurrency}), ) + if cached_store and kind == "ndarray": + return blosc2.RemoteArray(urlpath, **options) result = store[""] if max_concurrency is not None: if not isinstance(result, blosc2.RemoteArray): diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 51229b599..6fea41aad 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -205,6 +205,18 @@ def test_open_dispatches_remote_pytables_table(tmp_path): assert "RemoteCTable" in str(table.info) +def test_open_remote_pytables_table_uses_one_cache_directory(tmp_path): + name = f"{tmp_path.name}-one-cache.h5" + url, data = pytables_hdf5_url(name, indexed=True) + cache = tmp_path / "cache" + + with blosc2.open(url + "::table", cache_dir=cache) as table: + np.testing.assert_array_equal(table.where("id < 3").id[:], data["id"][data["id"] < 3]) + with blosc2.open(url + "::table", cache_dir=cache) as table: + assert table.traffic.requests == 0 + assert len(list(cache.glob(f"{name}--*"))) == 1 + + def test_open_dispatches_remote_pytables_table_with_json_sidecar(monkeypatch): import blosc2.hdf5_source as hdf5_source From d0ea9b63ab11469d755eafa184a1fd463a4c7e01 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Wed, 23 Sep 2026 11:44:54 +0200 Subject: [PATCH 71/82] Add shared remote caches to open and align cache budgets --- RELEASE_NOTES.md | 9 ++++ doc/guides/remote_objects.md | 38 ++++++++++++---- doc/guides/remote_tables.md | 14 ++++-- doc/reference/remotectable.rst | 14 ++++-- doc/reference/remotestore.rst | 18 ++++++-- src/blosc2/remote_array.py | 3 ++ src/blosc2/remote_ctable.py | 5 +- src/blosc2/remote_store.py | 11 ++++- src/blosc2/schunk.py | 73 ++++++++++++++++++++++++++++-- tests/ctable/test_remote_ctable.py | 38 ++++++++++++---- tests/test_remote_array.py | 7 +++ tests/test_remote_store.py | 71 +++++++++++++++++++++++++++-- 12 files changed, 262 insertions(+), 39 deletions(-) diff --git a/RELEASE_NOTES.md b/RELEASE_NOTES.md index be2d493ac..4c68f79a8 100644 --- a/RELEASE_NOTES.md +++ b/RELEASE_NOTES.md @@ -19,6 +19,10 @@ XXX version-specific blurb XXX #### Common remote-object API +- Added `blosc2.open(url, cache_dir=..., shared_cache=True)` for process-shared + sparse caches of remote B2Z, HDF5, and Zarr tables, groups, and array leaves. + This is the preferred entry point for ordinary shared caching; + `with_sparse_cache()` remains available for advanced attachment. - Added the public `RemoteObject` base for `RemoteArray`, `RemoteStore`, and `RemoteCTable`. It documents their shared source, attributes, traffic, cache accounting, export mutability, reference saving, and lifetime contract. @@ -31,6 +35,11 @@ XXX version-specific blurb XXX ### Compatibility notes +- `RemoteCTable.with_sparse_cache()` and `RemoteStore.with_sparse_cache()` now + default to a 256 MiB aggregate compressed-payload budget, matching `open()` + and `RemoteArray.with_sparse_cache()`. Explicitly pass `max_cache_bytes=None` + to retain unlimited caching. Internal leaf caches still share one aggregate + allowance rather than receiving independent 256 MiB limits. - The ListArray construction default changed from caller-managed batches to 2048 rows per batch. Pass `batch_rows=None` to retain the previous behavior. Existing arrays are not rewritten and keep their stored boundaries. diff --git a/doc/guides/remote_objects.md b/doc/guides/remote_objects.md index ef7982822..00e71fa28 100644 --- a/doc/guides/remote_objects.md +++ b/doc/guides/remote_objects.md @@ -120,27 +120,45 @@ Ordinary `RemoteStore` and `RemoteCTable` disk caches have **exclusive ownership Another process can reuse the same cache entry after its owner and dependent handles close, but opening that entry while it is still owned raises `RuntimeError: RemoteStore cache is already owned`. This also applies to stores -and tables opened through `blosc2.open(..., cache_dir=...)`. +and tables opened through `blosc2.open(..., cache_dir=...)` with the default +`shared_cache=False`. ### Sharing a cache between simultaneous processes -Use `with_sparse_cache()` when multiple processes need to keep the same store -or table open: +Use `blosc2.open(..., cache_dir=..., shared_cache=True)` when multiple processes +need to keep the same store or table open: ```python -with blosc2.RemoteCTable.with_sparse_cache( +with blosc2.open( "https://datasets.example.org/data.h5", - "./shared-table-cache", + cache_dir="./shared-table-cache", + shared_cache=True, path="readings", ) as table: print(table.info) ``` -For hierarchies, use `RemoteStore.with_sparse_cache()` instead. These constructors -use operation-scoped locks: handles can coexist across processes, but operations -on the same store serialize. Every process using that cache must use the shared -constructor; do not mix it with ordinary `cache_dir=` access. Use a separate -directory for the shared cache. +The same opener returns groups and array leaves selected within remote B2Z, +HDF5, or Zarr containers. Sharing requires lazy access, `CachePolicy.DISK`, and +an immutable source (`assume_immutable=True`). Standalone `.b2nd` and Caterva2 +sources are not supported by this option. + +Shared caches use sparse frames (separate chunk files) and operation-scoped +locks: handles can coexist across processes, but operations on the same store +serialize. Ordinary table/store disk caches use contiguous array cache files +and lifetime ownership locks. **Both fetch data on demand.** Every process using +the shared directory must enable sharing; do not mix it with ordinary exclusive +`cache_dir=` access. Use a separate directory for the shared cache. + +Both modes default to a **256 MiB aggregate compressed-payload budget** across +the owner's leaves. Set `max_cache_bytes` to a positive byte count to change it, +or explicitly pass `None` for unlimited retention. The budget applies after +operations; it does not bound metadata, total disk footprint, or peak RAM. + +`RemoteStore.with_sparse_cache()` and `RemoteCTable.with_sparse_cache()` remain +available for advanced attachment with manifests, seed carriers, or authorized +filesystems. They use the same 256 MiB default; explicit `max_cache_bytes=None` +keeps the previous unlimited behavior. See {doc}`../reference/remotestore` ("Shared sparse runtime caches") for locking and refresh details, and {doc}`remote_tables` for table usage. diff --git a/doc/guides/remote_tables.md b/doc/guides/remote_tables.md index 550dfaaf7..756e8dbf8 100644 --- a/doc/guides/remote_tables.md +++ b/doc/guides/remote_tables.md @@ -139,16 +139,22 @@ refresh. Refresh a table obtained from a {ref}`RemoteStore` through the root store, then retrieve the table again. Immutable reference artifacts reject `refresh()`. -For a cache shared by multiple server processes, use the sparse constructor: +For a cache shared by multiple processes, enable `shared_cache` in the opener: ```python -with blosc2.RemoteCTable.with_sparse_cache(url, "shared-table-cache") as table: +with blosc2.open(url, cache_dir="shared-table-cache", shared_cache=True) as table: print(table[:5]) ``` The outer table owns one aggregate cache budget for its ordinary and referenced -`RemoteArray` columns. Every process using the cache directory must use this -constructor. +`RemoteArray` columns: 256 MiB of retained compressed payload by default. +Pass `max_cache_bytes=None` for unlimited retention. The budget does not bound +total disk usage or peak RAM. Every process using the cache directory must +enable sharing; use a separate directory from ordinary exclusive caches. +Handles can coexist, but operations on the same store serialize. + +`RemoteCTable.with_sparse_cache()` remains available for advanced attachment +with a manifest or seed carrier, and now has the same 256 MiB default. ## See also diff --git a/doc/reference/remotectable.rst b/doc/reference/remotectable.rst index d5eb13650..4cfb5b4b7 100644 --- a/doc/reference/remotectable.rst +++ b/doc/reference/remotectable.rst @@ -60,10 +60,16 @@ views remain borrowed from that table. Refresh a nested table through its root store; a standalone table can call ``refresh()``. Server processes can share one bounded sparse disk cache with -``RemoteCTable.with_sparse_cache(url, runtime_cache_path)``. The constructor -uses the same process-safe cache and aggregate byte limit as -``RemoteStore.with_sparse_cache()``; all processes using that directory must -open it through the sparse-cache constructor. +``blosc2.open(url, cache_dir="shared-cache", shared_cache=True)``. All processes +using that directory must enable sharing. Operations serialize per store, +and the aggregate compressed-payload budget defaults to 256 MiB; explicitly +pass ``max_cache_bytes=None`` for unlimited retention. Use a separate directory +from ordinary exclusive caches. The budget does not bound total disk usage or +peak RAM. + +``RemoteCTable.with_sparse_cache(url, runtime_cache_path)`` remains available +for advanced attachment with manifests or seed carriers, with the same default +budget and shared-cache implementation. See :doc:`Working with Remote Tables <../guides/remote_tables>` for column access, filtering, buffering, reference saving, and materialization examples. diff --git a/doc/reference/remotestore.rst b/doc/reference/remotestore.rst index 4b8c02e2d..904f3d11f 100644 --- a/doc/reference/remotestore.rst +++ b/doc/reference/remotestore.rst @@ -143,13 +143,16 @@ uses POSIX flock or Windows byte locking; Windows execution remains a CI check. Shared sparse runtime caches ---------------------------- -Services and multiple local processes can use ``RemoteStore.with_sparse_cache`` +Services and multiple local processes can use ``blosc2.open(..., shared_cache=True)`` to keep simultaneous handles to the same private runtime cache: .. code-block:: python - with blosc2.RemoteStore.with_sparse_cache( - "https://host/data.b2z", "shared-runtime", max_cache_bytes=64 << 20 + with blosc2.open( + "https://host/data.b2z", + cache_dir="shared-runtime", + shared_cache=True, + max_cache_bytes=64 << 20, ) as store: with store["experiment/temperature"] as array: values = array[:100] @@ -159,9 +162,16 @@ This mode stores leaf payload in sparse RemoteArray caches. Each operation acquires a store-wide OS lock, reloads discovery and leaf accounting, and applies one aggregate payload allowance. Handles may coexist across processes, while operations within a store serialize. All users of that directory must use the -shared constructor. A process-local memory cache or the ordinary exclusive +shared mode. A process-local memory cache or the ordinary exclusive ``cache_dir`` constructor must not write to it. +The default aggregate compressed-payload budget is 256 MiB. Explicitly pass +``max_cache_bytes=None`` to disable eviction. This is a post-operation payload +bound, not a bound on metadata, total disk usage, or peak RAM. +``RemoteStore.with_sparse_cache()`` remains available for advanced attachment +with manifests, seed carriers, and authorized filesystems; it uses the same +default budget. Use a separate directory from ordinary exclusive caches. + Manifests and generation pointers are published atomically. A process that dies during an operation causes the next owner to discard the disposable payload generation; remote sources are not contacted by offline trimming or manifest recovery. diff --git a/src/blosc2/remote_array.py b/src/blosc2/remote_array.py index 7d9ba47e3..8951c746b 100644 --- a/src/blosc2/remote_array.py +++ b/src/blosc2/remote_array.py @@ -1020,6 +1020,9 @@ def with_sparse_cache( directory must construct it through this method so frame locking and interrupted-mutation recovery remain enabled. + The compressed-payload budget defaults to 256 MiB; pass + ``max_cache_bytes=None`` for unlimited retention. + ``carrier`` is the portable RemoteArray carrier. If it contains valid warm chunks when the sparse runtime cache is first created, those chunks are copied into the runtime cache. Both copies continue to exist until diff --git a/src/blosc2/remote_ctable.py b/src/blosc2/remote_ctable.py index 38fbb824d..defcb9c62 100644 --- a/src/blosc2/remote_ctable.py +++ b/src/blosc2/remote_ctable.py @@ -141,7 +141,7 @@ def with_sparse_cache( dataset=None, path=None, manifest=None, - max_cache_bytes=None, + max_cache_bytes=CACHE_POLICY_DEFAULT, carrier=None, max_concurrency=8, metadata_buffer_bytes=8 << 20, @@ -154,6 +154,9 @@ def with_sparse_cache( """Attach a remote CTable to a sparse disk cache shared across processes. ``path`` and ``dataset`` select the table as in the ordinary constructor. + The aggregate compressed-payload budget defaults to 256 MiB; pass + ``max_cache_bytes=None`` for unlimited retention. For ordinary shared + caching, prefer ``blosc2.open(url, cache_dir=..., shared_cache=True)``. """ settings = { name: _positive_integer(name, value) diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 9e7b650ea..1995956e6 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -1695,7 +1695,7 @@ def with_sparse_cache( dataset=None, path=None, manifest=None, - max_cache_bytes=None, + max_cache_bytes=CACHE_POLICY_DEFAULT, carrier=None, storage_options=None, _filesystem=None, @@ -1703,6 +1703,8 @@ def with_sparse_cache( _manifest_validator=None, _max_nodes=None, _traffic=None, + _source_format=None, + _hdf5_index=None, ): """Attach an immutable remote hierarchy to a cache shared across processes. @@ -1712,12 +1714,17 @@ def with_sparse_cache( no credentials or filesystem objects are persisted. Portable artifacts are exported with ``save`` rather than opened as mutable runtime storage. ``path`` and ``dataset`` select the group as in the ordinary constructor. + The aggregate compressed-payload budget defaults to 256 MiB; pass + ``max_cache_bytes=None`` for unlimited retention. For ordinary shared + caching, prefer ``blosc2.open(url, cache_dir=..., shared_cache=True)``. """ from blosc2.remote_store_cache import SharedStoreCache, SharedStoreOperation dataset = blosc2.core.resolve_dataset_path(dataset, path) limit = normalize_cache_limit(blosc2.CachePolicy.DISK, max_cache_bytes) base, root, kind = parse_container_url(urlpath, dataset) + if _source_format is not None: + kind = _source_format validate_persistable_url(base) source = {"urlpath": base, "dataset": (root or "").strip("/"), "kind": kind} fingerprint = storage_options_fingerprint(storage_options) @@ -1747,6 +1754,8 @@ def with_sparse_cache( _manifest_validator=_manifest_validator, _max_nodes=_max_nodes, _traffic=_traffic, + _source_format=kind, + _hdf5_index=_hdf5_index, ) owner.disk = disk owner.shared = True diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index b2a093b74..2a8429684 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2172,7 +2172,11 @@ def _validate_fsspec_lazy_options(urlpath: str, source_format, dataset, lazy: bo raise ValueError("HDF5 sources require lazy=True") -def _validate_non_lazy_fsspec_options(immutable_present, remote_array_options, cache_path, max_concurrency): +def _validate_non_lazy_fsspec_options( + immutable_present, remote_array_options, cache_path, max_concurrency, shared_cache +): + if shared_cache: + raise ValueError("shared_cache=True requires lazy=True") if immutable_present: raise NotImplementedError("assume_immutable requires lazy=True") if remote_array_options is not None: @@ -2230,7 +2234,44 @@ def _resolve_fsspec_format(urlpath, dataset, source_format, hdf5_index): return urlpath, dataset, source_format -def _open_lazy_fsspec(urlpath, source_format, options): +def _open_shared_remote(urlpath, source_format, options): + """Discover a container through its process-shared sparse cache.""" + if options["cache_dir"] is None: + raise ValueError("shared_cache=True requires cache_dir") + if source_format not in {"b2z", "hdf5", "zarr"}: + raise NotImplementedError("shared_cache=True requires a remote B2Z, HDF5, or Zarr container") + if options["cache_policy"] is not blosc2.CachePolicy.DISK: + raise ValueError("shared_cache=True requires cache_policy=CachePolicy.DISK") + if options["assume_immutable"] is not True: + raise ValueError("shared_cache=True requires assume_immutable=True") + store_options = { + key: value + for key, value in options.items() + if key in {"dataset", "storage_options", "max_cache_bytes"} + } + with blosc2.RemoteStore.with_sparse_cache( + urlpath, + options["cache_dir"], + _source_format=source_format, + _hdf5_index=options.get("hdf5_index"), + **store_options, + ) as store: + result = store[""] + concurrency = options["max_concurrency"] + if concurrency is not None: + if isinstance(result, blosc2.RemoteCTable): + result.max_concurrency = concurrency + elif isinstance(result, blosc2.RemoteArray): + result.src.max_concurrency = concurrency + else: + result.close() + raise NotImplementedError("max_concurrency is only supported for remote arrays and tables") + return result + + +def _open_lazy_fsspec(urlpath, source_format, options, shared_cache=False): + if shared_cache: + return _open_shared_remote(urlpath, source_format, options) if source_format == "b2z": return _open_remote_b2z(urlpath, options) if source_format == "hdf5": @@ -2251,6 +2292,7 @@ def _open_fsspec_url(urlpath: str, mode: str, offset: int, kwargs: dict): raise NotImplementedError(f"fsspec URLs can only be opened with mode='r', not {mode!r}") cache_dir, cache_path = _remote_cache_options(kwargs) + shared_cache = kwargs.pop("shared_cache", False) storage_options = kwargs.pop("storage_options", None) source_format = kwargs.pop("source_format", None) dataset = kwargs.pop("dataset", None) @@ -2292,9 +2334,11 @@ def _open_fsspec_url(urlpath: str, mode: str, offset: int, kwargs: dict): requested = [k for k, v in kwargs.items() if v is not None] if requested: raise NotImplementedError(f"{', '.join(requested)} is not supported with lazy=True") - return _open_lazy_fsspec(urlpath, source_format, remote_array_options) + return _open_lazy_fsspec(urlpath, source_format, remote_array_options, shared_cache) - _validate_non_lazy_fsspec_options(immutable_present, remote_array_options, cache_path, max_concurrency) + _validate_non_lazy_fsspec_options( + immutable_present, remote_array_options, cache_path, max_concurrency, shared_cache + ) if cache_dir is not None: localized = localize_fsspec_url(urlpath, cache_dir, storage_options=storage_options) @@ -2514,6 +2558,15 @@ def _reject_table_buffer_options(kwargs): ) +def _validate_shared_cache_request(urlpath, shared_cache, kwargs): + if not isinstance(shared_cache, bool): + raise TypeError("shared_cache must be a bool") + if shared_cache: + if not isinstance(urlpath, str) or not is_fsspec_url(urlpath): + raise ValueError("shared_cache=True requires a remote container URL") + kwargs["shared_cache"] = True + + def open( urlpath: str | pathlib.Path | blosc2.URLPath, mode: str = "r", @@ -2522,6 +2575,7 @@ def open( hdf5_index: dict | str | os.PathLike | None = None, *, path: str | None = None, + shared_cache: bool = False, **kwargs: dict, ) -> ( blosc2.SChunk @@ -2575,6 +2629,16 @@ def open( (e.g. in a file containing several such objects). A nonzero offset in a local file opens the embedded Blosc2 frame directly, bypassing filename-based container format detection. + shared_cache: bool, optional + Share an on-demand disk cache between processes. Defaults to False. + Requires ``cache_dir`` and a remote B2Z, HDF5, or Zarr container with + lazy access, ``CachePolicy.DISK``, and ``assume_immutable=True``. + Returns the selected table, group, or array using sparse cache storage + and operation-scoped locks. Operations on the same store serialize. + All processes using this cache must enable sharing; use a separate + directory from ordinary exclusive caches. The aggregate retained + compressed-payload budget defaults to 256 MiB; ``max_cache_bytes=None`` + disables eviction. This does not bound total disk usage or peak RAM. kwargs: dict, optional lazy: bool or None, optional ``None`` (the default) automatically selects the access mode. ``True`` @@ -2773,6 +2837,7 @@ def open( """ dataset = blosc2.core.resolve_dataset_path(dataset, path) _reject_table_buffer_options(kwargs) + _validate_shared_cache_request(urlpath, shared_cache, kwargs) if isinstance(urlpath, blosc2.URLPath): return _open_c2_urlpath(urlpath, mode, offset, kwargs) diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 6fea41aad..9d00950e0 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -190,16 +190,17 @@ def test_remote_pytables_light_index_falls_back_to_scan(): np.testing.assert_array_equal(table.where("id < 3").id[:], data["id"][data["id"] < 3]) -def test_open_dispatches_remote_pytables_table(tmp_path): +@pytest.mark.parametrize("shared_cache", [False, True]) +def test_open_dispatches_remote_pytables_table(tmp_path, shared_cache): url, data = pytables_hdf5_url(f"{tmp_path.name}-open-pytables.h5") cache = tmp_path / "cache" - with blosc2.open(url, lazy=True, cache_dir=cache) as store: + with blosc2.open(url, lazy=True, cache_dir=cache, shared_cache=shared_cache) as store: assert isinstance(store, blosc2.RemoteStore) assert "table" in store for target, options in ((url, {"dataset": "table"}), (url + "::table", {})): - with blosc2.open(target, lazy=True, cache_dir=cache, **options) as table: + with blosc2.open(target, lazy=True, cache_dir=cache, shared_cache=shared_cache, **options) as table: assert isinstance(table, blosc2.RemoteCTable) np.testing.assert_array_equal(table["id"][:], data["id"]) assert "RemoteCTable" in str(table.info) @@ -217,7 +218,8 @@ def test_open_remote_pytables_table_uses_one_cache_directory(tmp_path): assert len(list(cache.glob(f"{name}--*"))) == 1 -def test_open_dispatches_remote_pytables_table_with_json_sidecar(monkeypatch): +@pytest.mark.parametrize("shared_cache", [False, True]) +def test_open_dispatches_remote_pytables_table_with_json_sidecar(monkeypatch, tmp_path, shared_cache): import blosc2.hdf5_source as hdf5_source url, data = pytables_hdf5_url("pytables-sidecar.h5", indexed=True) @@ -228,7 +230,9 @@ def test_open_dispatches_remote_pytables_table_with_json_sidecar(monkeypatch): "scan_hdf5_index", lambda *args, **kwargs: pytest.fail("explicit sidecar must skip source discovery"), ) - with blosc2.open(url + "::table", hdf5_index=sidecar) as table: + with blosc2.open( + url + "::table", hdf5_index=sidecar, cache_dir=tmp_path / "cache", shared_cache=shared_cache + ) as table: assert isinstance(table, blosc2.RemoteCTable) np.testing.assert_array_equal(table.id[:], data["id"]) assert table._get_index_catalog()["id"]["kind"] == "opsi" @@ -487,13 +491,21 @@ def test_remote_full_index_merges_incremental_runs(tmp_path): assert len(remote._get_index_catalog()["x"]["full"]["runs"]) == 1 -def test_sparse_cache_shared_handles_and_refresh(tmp_path): +@pytest.mark.parametrize("api", ["factory", "open"]) +def test_sparse_cache_shared_handles_and_refresh(tmp_path, api): local = blosc2.CTable(Row, [(i, (i, i + 1), f"r{i}") for i in range(20)]) url = remote_table_url(tmp_path, local, "sparse-shared") cache = tmp_path / "sparse-cache" - with blosc2.RemoteCTable.with_sparse_cache(url, cache) as first: - with blosc2.RemoteCTable.with_sparse_cache(url, cache) as second: + def open_shared(): + if api == "open": + return blosc2.open(url, cache_dir=cache, shared_cache=True, max_concurrency=2) + return blosc2.RemoteCTable.with_sparse_cache(url, cache, max_concurrency=2) + + with open_shared() as first: + assert first.max_cache_bytes == 256 << 20 + assert first.max_concurrency == 2 + with open_shared() as second: np.testing.assert_array_equal(first.x[:], np.arange(20)) second.traffic.reset() np.testing.assert_array_equal(second.x[:], np.arange(20)) @@ -506,6 +518,16 @@ def test_sparse_cache_shared_handles_and_refresh(tmp_path): first.x[:] +@pytest.mark.parametrize("limit", [None, 1]) +def test_sparse_table_cache_budget(tmp_path, limit): + local = blosc2.CTable(Row, [(i, (i, i + 1), f"r{i}") for i in range(20)]) + url = remote_table_url(tmp_path, local) + with blosc2.RemoteCTable.with_sparse_cache(url, tmp_path / "cache", max_cache_bytes=limit) as table: + assert table.max_cache_bytes == limit + np.testing.assert_array_equal(table.x[:], np.arange(20)) + assert (table.cache_bytes > 0) if limit is None else (table.cache_bytes <= limit) + + @pytest.mark.parametrize("include_note", [False, True]) @pytest.mark.parametrize("nrows", [0, 2, 20, 21]) def test_remote_example_total_timing(tmp_path, capsys, include_note, nrows): diff --git a/tests/test_remote_array.py b/tests/test_remote_array.py index b5c357393..a9eb90126 100644 --- a/tests/test_remote_array.py +++ b/tests/test_remote_array.py @@ -266,6 +266,13 @@ def test_disk_bound_shrinks_self_caching_carrier(tmp_path): assert reopened.cache_bytes <= 120_000 +@pytest.mark.parametrize(("options", "expected"), [({}, 256 << 20), ({"max_cache_bytes": None}, None)]) +def test_sparse_cache_default_budget(tmp_path, options, expected): + url, _ = _remote_array("sparse-default.b2nd") + with blosc2.RemoteArray.with_sparse_cache(url, tmp_path / "cache", **options) as array: + assert array.max_cache_bytes == expected + + def test_server_sparse_cache_reopens_and_exports_portable_carriers(tmp_path): url, data = _remote_array("server-sparse.b2nd", nchunks=3, chunk_size=100_000) runtime_path = tmp_path / "private-runtime" diff --git a/tests/test_remote_store.py b/tests/test_remote_store.py index 6e71ac7e4..0add00680 100644 --- a/tests/test_remote_store.py +++ b/tests/test_remote_store.py @@ -676,11 +676,19 @@ def descriptor(url): assert destination.read_bytes() == b"existing" -def test_sparse_store_shared_handles(hierarchy, tmp_path): +@pytest.mark.parametrize("api", ["factory", "open"]) +def test_sparse_store_shared_handles(hierarchy, tmp_path, api): url, data = hierarchy parent = tmp_path / "shared" - with blosc2.RemoteStore.with_sparse_cache(url, parent) as first: - with blosc2.RemoteStore.with_sparse_cache(url, parent) as second: + + def open_shared(): + if api == "open": + return blosc2.open(url, cache_dir=parent, shared_cache=True) + return blosc2.RemoteStore.with_sparse_cache(url, parent) + + with open_shared() as first: + assert first.max_cache_bytes == 256 << 20 + with open_shared() as second: with first["group/a"] as a, second["group/a"] as b: np.testing.assert_array_equal(a[:], data) second.traffic.reset() @@ -693,6 +701,63 @@ def test_sparse_store_shared_handles(hierarchy, tmp_path): first.keys() +@pytest.mark.parametrize("limit", [None, 1]) +def test_open_shared_cache_budget_and_selected_array(hierarchy, tmp_path, limit): + url, data = hierarchy + with blosc2.open(url, cache_dir=tmp_path / "cache", shared_cache=True, max_cache_bytes=limit) as store: + assert store.max_cache_bytes == limit + with store["group/a"] as array: + np.testing.assert_array_equal(array[:], data) + assert (store.cache_bytes > 0) if limit is None else (store.cache_bytes <= limit) + with blosc2.open( + url, path="group/a", cache_dir=tmp_path / "selected", shared_cache=True, max_concurrency=2 + ) as array: + assert isinstance(array, blosc2.RemoteArray) + np.testing.assert_array_equal(array[:], data) + assert array._proxy._cache.schunk.contiguous is False + assert array.src.max_concurrency == 2 + + +def test_open_shared_cache_explicit_source_format(tmp_path): + path = tmp_path / "source.b2z" + with blosc2.TreeStore(path, mode="w") as store: + store["a"] = blosc2.arange(10) + url = f"memory://{tmp_path.name}/no-suffix" + fsspec.filesystem("memory").pipe(url, path.read_bytes()) + for _ in range(2): + with blosc2.open(url, source_format="b2z", cache_dir=tmp_path / "cache", shared_cache=True) as store: + with store["a"] as array: + np.testing.assert_array_equal(array[:], np.arange(10)) + + +@pytest.mark.parametrize( + ("url", "options", "error", "message"), + [ + ("memory://table.b2z", {"shared_cache": 1}, TypeError, "shared_cache must be a bool"), + ("local.b2z", {}, ValueError, "remote container URL"), + ("memory://table.b2z", {"cache_dir": None}, ValueError, "requires cache_dir"), + ("memory://table.b2z", {"lazy": False}, ValueError, "requires lazy=True"), + ("memory://table.b2z", {"assume_immutable": False}, ValueError, "assume_immutable=True"), + ( + "memory://table.b2z", + {"cache_policy": blosc2.CachePolicy.MEMORY}, + ValueError, + "CachePolicy.DISK", + ), + ("memory://array.b2nd", {}, NotImplementedError, "B2Z, HDF5, or Zarr"), + ("memory://table.b2z", {"cache_path": "cache.b2nd"}, ValueError, "mutually exclusive"), + ("memory://table.b2z", {"mode": "a"}, NotImplementedError, "mode='r'"), + ("memory://table.b2z", {"offset": 1, "lazy": True}, NotImplementedError, "offset"), + ("memory://table.b2z", {"mmap_mode": "r"}, ValueError, "requires lazy=True"), + ], +) +def test_open_shared_cache_invalid_options(tmp_path, url, options, error, message): + kwargs = {"cache_dir": tmp_path / "cache", "shared_cache": True, **options} + with pytest.raises(error, match=message): + blosc2.open(url, **kwargs) + assert not (tmp_path / "cache").exists() + + def test_sparse_store_trim_export_recovery(hierarchy, tmp_path): url, data = hierarchy parent = tmp_path / "shared" From 529ea29da83218edb7227b95bdfe16941a540689 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Wed, 23 Sep 2026 12:06:44 +0200 Subject: [PATCH 72/82] Extend shared caching to remote arrays with consistent lazy access --- RELEASE_NOTES.md | 8 ++- doc/guides/remote_arrays.md | 36 ++++++++++- doc/guides/remote_objects.md | 37 ++++++++--- doc/reference/remotearray.rst | 14 +++- src/blosc2/blosc2_ext.pyx | 31 +++++---- src/blosc2/remote_array.py | 59 +++++++++++------ src/blosc2/schunk.py | 52 ++++++++------- tests/ndarray/test_c2array_blocks.py | 96 ++++++++++++++++++++++++++++ tests/test_fsspec.py | 32 ++++++++++ tests/test_locking.py | 30 +++++++++ tests/test_remote_array.py | 83 ++++++++++++++++++++++++ tests/test_remote_store.py | 5 +- 12 files changed, 411 insertions(+), 72 deletions(-) diff --git a/RELEASE_NOTES.md b/RELEASE_NOTES.md index 4c68f79a8..c895da442 100644 --- a/RELEASE_NOTES.md +++ b/RELEASE_NOTES.md @@ -20,9 +20,15 @@ XXX version-specific blurb XXX #### Common remote-object API - Added `blosc2.open(url, cache_dir=..., shared_cache=True)` for process-shared - sparse caches of remote B2Z, HDF5, and Zarr tables, groups, and array leaves. + sparse caches of standalone `.b2nd` URLs, Caterva2 `URLPath` sources, and + remote B2Z, HDF5, and Zarr tables, groups, and array leaves. This is the preferred entry point for ordinary shared caching; `with_sparse_cache()` remains available for advanced attachment. +- Shared caches select lazy access for every remote source when `lazy` is + omitted or `None`, including suffix-free fsspec URLs. Explicit `lazy=False` + is rejected. Sparse array cache initialization is + serialized so simultaneous first openers cannot overwrite each other's cache. + Locked frame opens release the GIL so another Python thread can finish its read. - Added the public `RemoteObject` base for `RemoteArray`, `RemoteStore`, and `RemoteCTable`. It documents their shared source, attributes, traffic, cache accounting, export mutability, reference saving, and lifetime contract. diff --git a/doc/guides/remote_arrays.md b/doc/guides/remote_arrays.md index b87df81fb..93d7e5a4b 100644 --- a/doc/guides/remote_arrays.md +++ b/doc/guides/remote_arrays.md @@ -146,9 +146,43 @@ A `URLPath` always means Caterva2. If its `urlbase` is omitted, the server comes from {func}`blosc2.c2context` or `BLOSC_C2URLBASE`. Other transports can be added with a custom {ref}`ByteRangeNDSource`; see [Use your own transport](#use-your-own-transport). +## Share a disk cache between processes + +Use `shared_cache=True` to share downloaded chunks of a standalone `.b2nd` +array, a Caterva2 dataset, or an array inside a remote container: + +```python +a = blosc2.open( + "https://datasets.example.org/big.b2nd", + cache_dir="shared-array-cache", + shared_cache=True, +) +b = blosc2.open( + blosc2.URLPath("@public/big.b2nd", urlbase="https://cat2.cloud/demo"), + cache_dir="shared-caterva-cache", + shared_cache=True, +) +``` + +The cache uses sparse storage, with locks protecting initialization and +operations across processes. Missing chunks or blocks are fetched on demand. +The default retained compressed-payload limit is **256 MiB**; pass +`max_cache_bytes=None` for unlimited retention. A table/store shares one budget +across its leaves; independently opened arrays each have their own budget. +Metadata, total disk footprint, and peak RAM are outside this limit. + +Sharing requires `cache_dir`, disk caching, and `assume_immutable=True`. +Shared caching selects lazy access for every remote source when `lazy` is +omitted or `None`; explicit `lazy=False` is rejected. +Authentication from `URLPath` or `c2context` remains +in process memory. Authenticated users must use separate cache directories. +All handles using a shared cache must enable sharing. For advanced attachment +with seed carriers or authorized sources, `RemoteArray.with_sparse_cache()` +remains available. + ## Operating with remote arrays -Remote arrays can be operands in lazy expressions. Opening an array and building the expression only reads metadata; data is fetched when the expression is sliced or computed. A shared `cache_dir` keeps downloaded chunks locally between runs: +Remote arrays can be operands in lazy expressions. Opening an array and building the expression only reads metadata; data is fetched when the expression is sliced or computed. A persistent `cache_dir` keeps downloaded chunks locally between runs: ![Remote hierarchical compute across HDF5, Zarr, and B2Z datasets with tiered caching and lazy evaluation.](../tutorials/images/remote_lazy_expr_architecture.png) diff --git a/doc/guides/remote_objects.md b/doc/guides/remote_objects.md index 00e71fa28..493b98434 100644 --- a/doc/guides/remote_objects.md +++ b/doc/guides/remote_objects.md @@ -126,7 +126,7 @@ and tables opened through `blosc2.open(..., cache_dir=...)` with the default ### Sharing a cache between simultaneous processes Use `blosc2.open(..., cache_dir=..., shared_cache=True)` when multiple processes -need to keep the same store or table open: +need to keep the same array, store, or table open: ```python with blosc2.open( @@ -138,24 +138,43 @@ with blosc2.open( print(table.info) ``` -The same opener returns groups and array leaves selected within remote B2Z, -HDF5, or Zarr containers. Sharing requires lazy access, `CachePolicy.DISK`, and -an immutable source (`assume_immutable=True`). Standalone `.b2nd` and Caterva2 -sources are not supported by this option. +The same opener supports standalone `.b2nd` URLs, Caterva2 `URLPath` sources, +and groups or array leaves selected within remote B2Z, HDF5, or Zarr containers. +Sharing requires lazy access, `CachePolicy.DISK`, and an immutable source +(`assume_immutable=True`). For every remote source, `shared_cache=True` selects +lazy access when `lazy` is omitted or `None`; explicit `lazy=False` is rejected. +This also applies to fsspec URLs without a recognizable filename suffix. + +```python +array = blosc2.open( + "https://datasets.example.org/array.b2nd", + cache_dir="shared-array-cache", + shared_cache=True, +) +caterva_array = blosc2.open( + blosc2.URLPath("@public/array.b2nd", urlbase="https://cat2.cloud/demo"), + cache_dir="shared-caterva-cache", + shared_cache=True, +) +``` + +Caterva2 honors explicit authentication tokens and `c2context` settings without +persisting credentials. Authenticated users must use separate cache directories. Shared caches use sparse frames (separate chunk files) and operation-scoped -locks: handles can coexist across processes, but operations on the same store +locks: handles can coexist across processes, but operations on the same cache serialize. Ordinary table/store disk caches use contiguous array cache files and lifetime ownership locks. **Both fetch data on demand.** Every process using the shared directory must enable sharing; do not mix it with ordinary exclusive `cache_dir=` access. Use a separate directory for the shared cache. -Both modes default to a **256 MiB aggregate compressed-payload budget** across -the owner's leaves. Set `max_cache_bytes` to a positive byte count to change it, +Both modes default to a **256 MiB compressed-payload budget** per standalone +array, or in aggregate across a table/store owner's leaves. +Set `max_cache_bytes` to a positive byte count to change it, or explicitly pass `None` for unlimited retention. The budget applies after operations; it does not bound metadata, total disk footprint, or peak RAM. -`RemoteStore.with_sparse_cache()` and `RemoteCTable.with_sparse_cache()` remain +The `RemoteArray`, `RemoteStore`, and `RemoteCTable` `with_sparse_cache()` factories remain available for advanced attachment with manifests, seed carriers, or authorized filesystems. They use the same 256 MiB default; explicit `max_cache_bytes=None` keeps the previous unlimited behavior. diff --git a/doc/reference/remotearray.rst b/doc/reference/remotearray.rst index 462802199..a0d8c2376 100644 --- a/doc/reference/remotearray.rst +++ b/doc/reference/remotearray.rst @@ -145,6 +145,17 @@ is always returned: specifying ``cache_dir`` or ``cache_path`` configures it wit :attr:`blosc2.CachePolicy.DISK`, while omitting them configures it with :attr:`blosc2.CachePolicy.MEMORY`. +For concurrent processes, prefer +``blosc2.open(url, cache_dir="shared-cache", shared_cache=True)``. This supports +standalone ``.b2nd`` URLs, array leaves in remote containers, and Caterva2 +``URLPath`` sources. It uses sparse storage with process-safe initialization +and operation locks, and the same default 256 MiB payload budget. All remote +sources enter lazy mode when ``lazy`` is omitted or None; explicit ``lazy=False`` +is rejected, including for suffix-free fsspec URLs. +All users of the cache must enable sharing. Authenticated Caterva2 users must +use separate directories; tokens are not persisted. +``RemoteArray.with_sparse_cache()`` remains available for advanced attachment. + By default, :meth:`RemoteArray.save ` and :meth:`RemoteArray.to_cframe ` include valid warm chunks already retained by DISK or MEMORY proxies. Pass ``include_cache=False`` @@ -165,7 +176,8 @@ The raw ``cache`` is incomplete storage for inspection, not a materialized array Reads and exports on one handle are serialized. Async methods use worker threads; cancelling an await does not stop a running operation. Separate handles and -processes sharing a carrier need external locking. Unreadable cache files are +processes sharing an ordinary carrier need external locking; ``shared_cache=True`` +handles synchronization for sparse runtime caches. Unreadable cache files are preserved and their opening errors are propagated. Authentication supplied to a live Caterva2 source is deliberately omitted from diff --git a/src/blosc2/blosc2_ext.pyx b/src/blosc2/blosc2_ext.pyx index 4b257c686..aa6db5658 100644 --- a/src/blosc2/blosc2_ext.pyx +++ b/src/blosc2/blosc2_ext.pyx @@ -478,8 +478,8 @@ cdef extern from "blosc2.h": blosc2_schunk *blosc2_schunk_new(blosc2_storage *storage) blosc2_schunk *blosc2_schunk_copy(blosc2_schunk *schunk, blosc2_storage *storage) blosc2_schunk *blosc2_schunk_from_buffer(uint8_t *cframe, int64_t len, c_bool copy) - blosc2_schunk *blosc2_schunk_open_offset(const char* urlpath, int64_t offset) - blosc2_schunk* blosc2_schunk_open_offset_udio(const char* urlpath, int64_t offset, const blosc2_io *udio) + blosc2_schunk *blosc2_schunk_open_offset(const char* urlpath, int64_t offset) nogil + blosc2_schunk* blosc2_schunk_open_offset_udio(const char* urlpath, int64_t offset, const blosc2_io *udio) nogil int64_t blosc2_schunk_to_buffer(blosc2_schunk* schunk, uint8_t** cframe, c_bool* needs_free) nogil void blosc2_schunk_avoid_cframe_free(blosc2_schunk *schunk, c_bool avoid_cframe_free) @@ -1741,12 +1741,12 @@ cdef class SChunk: create_storage(&storage, kwargs) if self.mode == "r": - offset = 0 - if storage.io != NULL: - # mmap or locking: open through the user-defined io - self.schunk = blosc2_schunk_open_offset_udio(storage.urlpath, offset, storage.io) - else: - self.schunk = blosc2_schunk_open_offset(storage.urlpath, offset) + with nogil: # A lock holder in another thread must be able to resume. + if storage.io != NULL: + # mmap or locking: open through the user-defined io + self.schunk = blosc2_schunk_open_offset_udio(storage.urlpath, 0, storage.io) + else: + self.schunk = blosc2_schunk_open_offset(storage.urlpath, 0) if kwargs is not None: check_schunk_params(self.schunk, kwargs) @@ -3384,6 +3384,8 @@ def meta_keys(self): def open(urlpath, mode, offset, **kwargs): urlpath_ = urlpath.encode("utf-8") if isinstance(urlpath, str) else urlpath + cdef const char* path = urlpath_ + cdef int64_t frame_offset = offset cdef blosc2_schunk* schunk cdef blosc2_stdio_mmap* mmap_file cdef blosc2_io* io @@ -3407,10 +3409,12 @@ def open(urlpath, mode, offset, **kwargs): raise ValueError("initial_mapping_size can only be used with writing modes (r+, c)") if mmap_mode is None: - if locking: - schunk = blosc2_schunk_open_offset_udio(urlpath_, offset, &_locking_io) - else: - schunk = blosc2_schunk_open_offset(urlpath_, offset) + io = &_locking_io if locking else NULL + with nogil: # Opening can wait for another Python thread's frame lock. + if io != NULL: + schunk = blosc2_schunk_open_offset_udio(path, frame_offset, io) + else: + schunk = blosc2_schunk_open_offset(path, frame_offset) else: mmap_file = malloc(sizeof(BLOSC2_STDIO_MMAP_DEFAULTS)) memcpy(mmap_file, &BLOSC2_STDIO_MMAP_DEFAULTS, sizeof(BLOSC2_STDIO_MMAP_DEFAULTS)) @@ -3424,7 +3428,8 @@ def open(urlpath, mode, offset, **kwargs): io = malloc(sizeof(blosc2_io)) io.id = BLOSC2_IO_FILESYSTEM_MMAP io.params = mmap_file - schunk = blosc2_schunk_open_offset_udio(urlpath_, offset, io) + with nogil: + schunk = blosc2_schunk_open_offset_udio(path, frame_offset, io) if schunk == NULL: if mmap_mode is not None: diff --git a/src/blosc2/remote_array.py b/src/blosc2/remote_array.py index 8951c746b..5c2075b3f 100644 --- a/src/blosc2/remote_array.py +++ b/src/blosc2/remote_array.py @@ -534,10 +534,11 @@ class RemoteArray(RemoteObject, blosc2.Operand): """A persistable, optionally self-caching reference to a remote array. With :attr:`CachePolicy.DISK`, the public constructor uses the persisted - B2ND carrier itself as the bounded cache. Server code can instead use - :meth:`with_sparse_cache` to keep a private directory-backed runtime cache - beside a portable carrier. With :attr:`CachePolicy.MEMORY`, chunks are - retained in process memory up to a bounded size. With + B2ND carrier itself as the bounded cache. Use + ``blosc2.open(url, cache_dir=..., shared_cache=True)`` for a process-shared + sparse runtime cache. Server code can also use :meth:`with_sparse_cache` + for advanced attachment beside a portable carrier. With + :attr:`CachePolicy.MEMORY`, chunks are retained in process memory up to a bounded size. With :attr:`CachePolicy.NONE`, reads retain no data. .. note:: @@ -617,6 +618,7 @@ def __init__( _store_owner=None, _runtime_is_mutable: bool = True, _defer_cache: bool = False, + _shared_cache: bool = False, ): dataset = blosc2.core.resolve_dataset_path(dataset, path) if not isinstance(cache_policy, blosc2.CachePolicy): @@ -706,6 +708,8 @@ def __init__( ) self._assume_immutable = assume_immutable self._storage_options = storage_options + if _shared_cache: + _runtime_cache_path = self._carrier_path(cache_dir, None) + ".cache" self._runtime_urlpath = self._runtime_source(urlpath) self._expected_geometry = self._geometry(self.src) self._expected_cparams = self.src.cparams @@ -919,28 +923,39 @@ def _open_or_create_sparse_cache(self, cache_path): This is deliberately separate from ``cache_path`` in the public constructor: portable RemoteArray carriers remain contiguous files. """ + from blosc2.remote_store_cache import lock_cache_file + + # The frame lock cannot protect a directory that does not exist yet. + lock_path = Path(os.fspath(cache_path) + ".init.lock") + lock_path.parent.mkdir(parents=True, exist_ok=True) + with lock_path.open("a+b") as lock: + lock_cache_file(lock, blocking=True) + return self._open_sparse_cache(cache_path) + + def _open_sparse_cache(self, cache_path): path = os.fspath(cache_path) if os.path.exists(path): if not os.path.isdir(path): raise ValueError("runtime_cache_path must name a sparse frame directory") runtime = blosc2.blosc2_ext.open(path, "a", 0, dparams=blosc2.DParams(nthreads=1), locking=True) - if runtime.schunk.vlmeta.get("b2o") != self._payload(mutable=True): - raise ValueError(f"the sparse runtime cache at {path} has a different specification") - self._validate_geometry( - (runtime.shape, runtime.dtype, runtime.chunks, runtime.blocks), src=self.src - ) - stored = runtime.schunk.vlmeta.get("proxy-stamp") - current = getattr(self.src, "stamp", None) - status = ( - "invalidated/rebuilt" - if stored is not None and current is not None and stored != current - else "reused" - ) - if status == "reused": - if self._cached_meta is None: - self._cached_meta = self._meta_from_carrier(runtime) - if self._cached_vlmeta is None: - self._cached_vlmeta = read_b2object_user_vlmeta(runtime) + with runtime.holding_lock(): + if runtime.schunk.vlmeta.get("b2o") != self._payload(mutable=True): + raise ValueError(f"the sparse runtime cache at {path} has a different specification") + self._validate_geometry( + (runtime.shape, runtime.dtype, runtime.chunks, runtime.blocks), src=self.src + ) + stored = runtime.schunk.vlmeta.get("proxy-stamp") + current = getattr(self.src, "stamp", None) + status = ( + "invalidated/rebuilt" + if stored is not None and current is not None and stored != current + else "reused" + ) + if status == "reused": + if self._cached_meta is None: + self._cached_meta = self._meta_from_carrier(runtime) + if self._cached_vlmeta is None: + self._cached_vlmeta = read_b2object_user_vlmeta(runtime) return runtime, status if self._carrier is not None: @@ -1022,6 +1037,8 @@ def with_sparse_cache( The compressed-payload budget defaults to 256 MiB; pass ``max_cache_bytes=None`` for unlimited retention. + For ordinary shared caching, prefer + ``blosc2.open(url, cache_dir=..., shared_cache=True)``. ``carrier`` is the portable RemoteArray carrier. If it contains valid warm chunks when the sparse runtime cache is first created, those chunks diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index 2a8429684..403089e0e 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2139,6 +2139,7 @@ def _open_c2_urlpath(urlpath: blosc2.URLPath, mode: str, offset: int, kwargs: di raise NotImplementedError("offset is not supported for Caterva2 arrays") cache_dir, cache_path = _remote_cache_options(kwargs) + shared_cache = kwargs.pop("shared_cache", False) max_concurrency = kwargs.pop("max_concurrency", None) immutable_present = "assume_immutable" in kwargs assume_immutable = kwargs.pop("assume_immutable", True) @@ -2157,7 +2158,7 @@ def _open_c2_urlpath(urlpath: blosc2.URLPath, mode: str, offset: int, kwargs: di urlpath, immutable_present, remote_array_options, cache_dir, cache_path, max_concurrency ) - return blosc2.RemoteArray(urlpath, **remote_array_options) + return _open_lazy_remote(urlpath, None, remote_array_options, shared_cache) def _validate_fsspec_lazy_options(urlpath: str, source_format, dataset, lazy: bool): @@ -2172,11 +2173,7 @@ def _validate_fsspec_lazy_options(urlpath: str, source_format, dataset, lazy: bo raise ValueError("HDF5 sources require lazy=True") -def _validate_non_lazy_fsspec_options( - immutable_present, remote_array_options, cache_path, max_concurrency, shared_cache -): - if shared_cache: - raise ValueError("shared_cache=True requires lazy=True") +def _validate_non_lazy_fsspec_options(immutable_present, remote_array_options, cache_path, max_concurrency): if immutable_present: raise NotImplementedError("assume_immutable requires lazy=True") if remote_array_options is not None: @@ -2235,15 +2232,15 @@ def _resolve_fsspec_format(urlpath, dataset, source_format, hdf5_index): def _open_shared_remote(urlpath, source_format, options): - """Discover a container through its process-shared sparse cache.""" + """Open remote arrays or containers through a process-shared sparse cache.""" if options["cache_dir"] is None: raise ValueError("shared_cache=True requires cache_dir") - if source_format not in {"b2z", "hdf5", "zarr"}: - raise NotImplementedError("shared_cache=True requires a remote B2Z, HDF5, or Zarr container") if options["cache_policy"] is not blosc2.CachePolicy.DISK: raise ValueError("shared_cache=True requires cache_policy=CachePolicy.DISK") if options["assume_immutable"] is not True: raise ValueError("shared_cache=True requires assume_immutable=True") + if source_format not in {"b2z", "hdf5", "zarr"}: + return blosc2.RemoteArray(urlpath, _shared_cache=True, **options) store_options = { key: value for key, value in options.items() @@ -2269,7 +2266,7 @@ def _open_shared_remote(urlpath, source_format, options): return result -def _open_lazy_fsspec(urlpath, source_format, options, shared_cache=False): +def _open_lazy_remote(urlpath, source_format, options, shared_cache=False): if shared_cache: return _open_shared_remote(urlpath, source_format, options) if source_format == "b2z": @@ -2334,11 +2331,9 @@ def _open_fsspec_url(urlpath: str, mode: str, offset: int, kwargs: dict): requested = [k for k, v in kwargs.items() if v is not None] if requested: raise NotImplementedError(f"{', '.join(requested)} is not supported with lazy=True") - return _open_lazy_fsspec(urlpath, source_format, remote_array_options, shared_cache) + return _open_lazy_remote(urlpath, source_format, remote_array_options, shared_cache) - _validate_non_lazy_fsspec_options( - immutable_present, remote_array_options, cache_path, max_concurrency, shared_cache - ) + _validate_non_lazy_fsspec_options(immutable_present, remote_array_options, cache_path, max_concurrency) if cache_dir is not None: localized = localize_fsspec_url(urlpath, cache_dir, storage_options=storage_options) @@ -2562,9 +2557,15 @@ def _validate_shared_cache_request(urlpath, shared_cache, kwargs): if not isinstance(shared_cache, bool): raise TypeError("shared_cache must be a bool") if shared_cache: - if not isinstance(urlpath, str) or not is_fsspec_url(urlpath): - raise ValueError("shared_cache=True requires a remote container URL") + if not isinstance(urlpath, blosc2.URLPath) and ( + not isinstance(urlpath, str) or not is_fsspec_url(urlpath) + ): + raise ValueError("shared_cache=True requires a remote URL or Caterva2 URLPath") kwargs["shared_cache"] = True + if kwargs.get("lazy") is False: + raise ValueError("shared_cache=True requires lazy=True") + if kwargs.get("lazy") is None: + kwargs["lazy"] = True def open( @@ -2631,14 +2632,19 @@ def open( directly, bypassing filename-based container format detection. shared_cache: bool, optional Share an on-demand disk cache between processes. Defaults to False. - Requires ``cache_dir`` and a remote B2Z, HDF5, or Zarr container with - lazy access, ``CachePolicy.DISK``, and ``assume_immutable=True``. + Requires ``cache_dir`` and a remote source: a standalone ``.b2nd`` URL, + a B2Z, HDF5, or Zarr container, or a Caterva2 :ref:`URLPath`. + Requires lazy access, ``CachePolicy.DISK``, and ``assume_immutable=True``. + Enables lazy access for every remote source when ``lazy`` is omitted or None; + explicit ``lazy=False`` is rejected. Returns the selected table, group, or array using sparse cache storage - and operation-scoped locks. Operations on the same store serialize. + and operation-scoped locks. Operations on the same cache serialize. All processes using this cache must enable sharing; use a separate directory from ordinary exclusive caches. The aggregate retained - compressed-payload budget defaults to 256 MiB; ``max_cache_bytes=None`` - disables eviction. This does not bound total disk usage or peak RAM. + compressed-payload budget defaults to 256 MiB per standalone array or + table/store owner; ``max_cache_bytes=None`` disables eviction. + This does not bound total disk usage or peak RAM. Authenticated Caterva2 + users must use separate cache directories; tokens are not persisted. kwargs: dict, optional lazy: bool or None, optional ``None`` (the default) automatically selects the access mode. ``True`` @@ -2761,8 +2767,8 @@ def open( ownership when closed; other handles from that store remain usable. * If :paramref:`urlpath` is a :ref:`URLPath` instance, :paramref:`mode` - must be 'r' and :paramref:`offset` must be 0. Without ``lazy=True`` it - returns a :ref:`C2Array`. With ``lazy=True``, it returns a :ref:`RemoteArray` + must be 'r' and :paramref:`offset` must be 0. By default it returns a + :ref:`C2Array`. With ``lazy=True`` or ``shared_cache=True``, it returns a :ref:`RemoteArray` (defaulting to ``CachePolicy.DISK`` when ``cache_dir`` or ``cache_path`` is provided, and ``CachePolicy.MEMORY`` otherwise). Authenticated users sharing a machine must use separate caches. diff --git a/tests/ndarray/test_c2array_blocks.py b/tests/ndarray/test_c2array_blocks.py index 12385c353..8b7679294 100644 --- a/tests/ndarray/test_c2array_blocks.py +++ b/tests/ndarray/test_c2array_blocks.py @@ -444,6 +444,102 @@ def test_open_urlpath_cache_options_need_lazy(tmp_path, server): blosc2.open(urlpath, max_concurrency=2) +@pytest.mark.parametrize("context_auth", [False, True]) +@pytest.mark.parametrize("options", [{}, {"lazy": None}, {"lazy": True}]) +def test_open_shared_caterva_cache(tmp_path, server, any_chunk_wants_blocks, context_auth, options): + token = "session=shared-secret" + data = _incompressible((200, 200)) + array, srv = server(data, chunks=(100, 200), blocks=(10, 20), cookie=token) + urlpath = blosc2.URLPath(array.path, urlbase=array.urlbase, auth_token=token) + context = contextlib.nullcontext() + if context_auth: + urlpath = blosc2.URLPath(array.path) + context = blosc2.c2context(urlbase=array.urlbase, auth_token=token) + with context: + first = blosc2.open( + urlpath, cache_dir=tmp_path / "cache", shared_cache=True, max_concurrency=2, **options + ) + assert first.max_cache_bytes == 256 << 20 + assert first.schunk.contiguous is False + assert first.src.max_concurrency == 2 + np.testing.assert_array_equal(first[:5, :10], data[:5, :10]) + second = blosc2.open(urlpath, cache_dir=tmp_path / "cache", shared_cache=True) + srv.log.clear() + np.testing.assert_array_equal(second[:5, :10], data[:5, :10]) + assert srv.log == [] + np.testing.assert_array_equal(second[100:105, :10], data[100:105, :10]) + srv.log.clear() + np.testing.assert_array_equal(first[100:105, :10], data[100:105, :10]) + assert srv.log == [] + assert token not in repr(first.source) + for path in (tmp_path / "cache").rglob("*"): + assert token not in str(path) + if path.is_file(): + assert token.encode() not in path.read_bytes() + + +def _shared_caterva_reader(urlpath, cache, barrier, results): + try: + barrier.wait(timeout=30) + with blosc2.open(urlpath, cache_dir=cache, shared_cache=True) as array: + array.traffic.reset() + np.testing.assert_array_equal(array[:100], np.arange(100)) + requests = array.traffic.requests + barrier.wait(timeout=30) + array.traffic.reset() + np.testing.assert_array_equal(array[:100], np.arange(100)) + results.put((requests, array.traffic.requests)) + except BaseException as exc: + results.put(repr(exc)) + + +def test_open_shared_caterva_processes(tmp_path, server): + import multiprocessing + + array, _ = server(np.arange(200), chunks=(100,), blocks=(100,), accept_ranges="none", cookie="key=1") + urlpath = blosc2.URLPath(array.path, urlbase=array.urlbase, auth_token="key=1") + ctx = multiprocessing.get_context("spawn") + barrier, results = ctx.Barrier(4), ctx.Queue() + workers = [ + ctx.Process(target=_shared_caterva_reader, args=(urlpath, tmp_path / "cache", barrier, results)) + for _ in range(4) + ] + try: + for worker in workers: + worker.start() + reports = [results.get(timeout=45) for _ in workers] + for worker in workers: + worker.join(timeout=10) + assert worker.exitcode == 0 + assert all(isinstance(report, tuple) for report in reports), reports + assert sum(cold for cold, _ in reports) == 1 + assert all(warm == 0 for _, warm in reports) + finally: + for worker in workers: + if worker.is_alive(): + worker.terminate() + worker.join() + + +@pytest.mark.parametrize( + ("options", "error", "message"), + [ + ({"cache_dir": None}, ValueError, "requires cache_dir"), + ({"lazy": False}, ValueError, "requires lazy=True"), + ({"cache_policy": blosc2.CachePolicy.NONE}, ValueError, "CachePolicy.DISK"), + ({"assume_immutable": False}, ValueError, "assume_immutable=True"), + ({"cache_path": "cache.b2nd"}, ValueError, "mutually exclusive"), + ({"mode": "a"}, NotImplementedError, "mode='r'"), + ({"offset": 1}, NotImplementedError, "offset"), + ], +) +def test_open_shared_caterva_invalid_options(tmp_path, options, error, message): + urlpath = blosc2.URLPath("@public/array.b2nd", urlbase="https://example.org") + with pytest.raises(error, match=message): + blosc2.open(urlpath, **{"cache_dir": tmp_path / "cache", "shared_cache": True, **options}) + assert not (tmp_path / "cache").exists() + + def test_blocks_are_read_over_ranges(server, any_chunk_wants_blocks): data = _incompressible((200, 200)) array, srv = server(data, chunks=(100, 200), blocks=(10, 20)) diff --git a/tests/test_fsspec.py b/tests/test_fsspec.py index 08a55d784..cd6600060 100644 --- a/tests/test_fsspec.py +++ b/tests/test_fsspec.py @@ -666,6 +666,38 @@ def test_http_url_is_read_through_fsspec(tmp_path): assert np.array_equal(lazy[3:5, 100:120], data[3:5, 100:120]) +@_http_server_skip +@pytest.mark.skipif(blosc2.IS_WASM, reason="no listening sockets on wasm32") +def test_http_shared_b2nd_cache_across_processes(tmp_path): + pytest.importorskip("aiohttp") + blosc2.asarray( + np.arange(20000, dtype="i4"), + chunks=(10000,), + blocks=(10000,), + urlpath=tmp_path / "array.b2nd", + ) + script = """ +import sys +import blosc2 +import numpy as np +with blosc2.open(sys.argv[1], cache_dir=sys.argv[2], shared_cache=True) as array: + assert not array.schunk.contiguous + array.traffic.reset() + np.testing.assert_array_equal(array[:10000], np.arange(10000)) + print(array.traffic.requests) +""" + with _ranged_server(tmp_path) as (urlbase, _): + for repeat in range(2): + result = subprocess.run( + [sys.executable, "-c", script, f"{urlbase}/array.b2nd", str(tmp_path / "cache")], + capture_output=True, + text=True, + timeout=30, + ) + assert result.returncode == 0, result.stderr + assert int(result.stdout) == 0 if repeat else int(result.stdout) > 0 + + @_http_server_skip def test_http_lazy_cache_rebuilt_when_remote_changes(tmp_path): pytest.importorskip("aiohttp") diff --git a/tests/test_locking.py b/tests/test_locking.py index ffde52eab..6f2757acc 100644 --- a/tests/test_locking.py +++ b/tests/test_locking.py @@ -70,6 +70,36 @@ def test_no_sidecar_by_default(tmp_path, contiguous): blosc2.remove_urlpath(str(urlpath)) +@pytest.mark.parametrize("contiguous", [False, True]) +@pytest.mark.parametrize("opener", ["open", "SChunk"]) +def test_open_locked_frame_releases_gil(tmp_path, contiguous, opener): + path = tmp_path / "threaded-open.b2frame" + create_schunk(path, contiguous=contiguous, locking=True) + script = """ +import sys +import threading +import time +from concurrent.futures import ThreadPoolExecutor +import blosc2 + +holder = blosc2.open(sys.argv[1], locking=True) +started = threading.Event() +def open_frame(): + started.set() + return getattr(blosc2, sys.argv[2])(urlpath=sys.argv[1], mode="r", locking=True) + +with ThreadPoolExecutor(max_workers=1) as pool: + with holder.holding_lock(): + future = pool.submit(open_frame) + assert started.wait(timeout=5) + time.sleep(0.05) + assert not future.done() + assert future.result(timeout=5).nchunks == holder.nchunks +""" + # A GIL deadlock must fail this test rather than hang the pytest worker. + subprocess.run([sys.executable, "-c", script, str(path), opener], check=True, timeout=15) + + def test_two_handles_coherent(tmp_path): # The Python twin of c-blosc2's examples/file-locking.c: a mutation through # one locked handle is picked up coherently by another one diff --git a/tests/test_remote_array.py b/tests/test_remote_array.py index a9eb90126..9ba797ea2 100644 --- a/tests/test_remote_array.py +++ b/tests/test_remote_array.py @@ -273,6 +273,89 @@ def test_sparse_cache_default_budget(tmp_path, options, expected): assert array.max_cache_bytes == expected +@pytest.mark.parametrize( + ("options", "limit"), [({}, 256 << 20), ({"max_cache_bytes": None}, None), ({"max_cache_bytes": 1}, 1)] +) +def test_open_shared_b2nd(tmp_path, options, limit): + from pathlib import Path + + url, data = _remote_array("shared-open.b2nd", nchunks=2, chunk_size=1000) + cache_dir = tmp_path / "cache" + with blosc2.open(url, cache_dir=cache_dir, shared_cache=True, **options) as first: + assert first.max_cache_bytes == limit + assert first.schunk.contiguous is False + assert Path(first.runtime_cache_path).is_dir() + np.testing.assert_array_equal(first[:1000], data[:1000]) + with blosc2.open(url, cache_dir=cache_dir, shared_cache=True, **options) as second: + second.traffic.reset() + np.testing.assert_array_equal(second[:1000], data[:1000]) + assert (second.traffic.requests > 0) if limit == 1 else (second.traffic.requests == 0) + np.testing.assert_array_equal(second[1000:], data[1000:]) + first.traffic.reset() + np.testing.assert_array_equal(first[1000:], data[1000:]) + assert (first.traffic.requests > 0) if limit == 1 else (first.traffic.requests == 0) + assert first.cache_bytes <= limit if limit is not None else first.cache_bytes > 0 + + +def test_open_shared_b2nd_storage_options(tmp_path): + url, data = _remote_array("shared-options.b2nd", nchunks=1, chunk_size=100) + paths = [] + for account in ("one", "two", "one"): + with blosc2.open( + url, cache_dir=tmp_path, shared_cache=True, storage_options={"account": account} + ) as array: + paths.append(array.runtime_cache_path) + array.traffic.reset() + np.testing.assert_array_equal(array[:], data) + assert array.traffic.requests == 0 if len(paths) == 3 else array.traffic.requests > 0 + assert paths[0] == paths[2] != paths[1] + + +@pytest.mark.parametrize("options", [{}, {"lazy": None}, {"lazy": True}]) +def test_open_shared_suffix_free_url(tmp_path, options): + url, data = _remote_array("shared-no-suffix", nchunks=1, chunk_size=100) + with blosc2.open(url, cache_dir=tmp_path, shared_cache=True, **options) as array: + assert isinstance(array, blosc2.RemoteArray) + assert not array.schunk.contiguous + np.testing.assert_array_equal(array[:], data) + with pytest.raises(ValueError, match="shared_cache=True requires lazy=True"): + blosc2.open(url, cache_dir=tmp_path, shared_cache=True, lazy=False) + # Ordinary suffix-free URLs retain their existing eager default. + assert isinstance(blosc2.open(url), blosc2.NDArray) + + +@pytest.mark.parametrize("api", ["open", "factory"]) +def test_sparse_cache_simultaneous_creation(tmp_path, monkeypatch, api): + import threading + import time + from concurrent.futures import ThreadPoolExecutor + + url, data = _remote_array("shared-creation.b2nd", nchunks=1, chunk_size=100) + original = blosc2.RemoteArray._to_b2object_carrier + creations = [] + + def slow_create(self, *args, **kwargs): + creations.append(kwargs["urlpath"]) + time.sleep(0.05) # Expose a second creator after the existence check. + return original(self, *args, **kwargs) + + monkeypatch.setattr(blosc2.RemoteArray, "_to_b2object_carrier", slow_create) + barrier = threading.Barrier(4) + + def read(): + barrier.wait(timeout=10) + array = ( + blosc2.open(url, cache_dir=tmp_path, shared_cache=True) + if api == "open" + else blosc2.RemoteArray.with_sparse_cache(url, tmp_path / "runtime") + ) + np.testing.assert_array_equal(array[:], data) + + with ThreadPoolExecutor(max_workers=4) as pool: + list(pool.map(lambda _: read(), range(4))) + assert len(creations) == 1 + + def test_server_sparse_cache_reopens_and_exports_portable_carriers(tmp_path): url, data = _remote_array("server-sparse.b2nd", nchunks=3, chunk_size=100_000) runtime_path = tmp_path / "private-runtime" diff --git a/tests/test_remote_store.py b/tests/test_remote_store.py index 0add00680..39e1ab031 100644 --- a/tests/test_remote_store.py +++ b/tests/test_remote_store.py @@ -734,7 +734,7 @@ def test_open_shared_cache_explicit_source_format(tmp_path): ("url", "options", "error", "message"), [ ("memory://table.b2z", {"shared_cache": 1}, TypeError, "shared_cache must be a bool"), - ("local.b2z", {}, ValueError, "remote container URL"), + ("local.b2z", {}, ValueError, "remote URL or Caterva2 URLPath"), ("memory://table.b2z", {"cache_dir": None}, ValueError, "requires cache_dir"), ("memory://table.b2z", {"lazy": False}, ValueError, "requires lazy=True"), ("memory://table.b2z", {"assume_immutable": False}, ValueError, "assume_immutable=True"), @@ -744,11 +744,10 @@ def test_open_shared_cache_explicit_source_format(tmp_path): ValueError, "CachePolicy.DISK", ), - ("memory://array.b2nd", {}, NotImplementedError, "B2Z, HDF5, or Zarr"), ("memory://table.b2z", {"cache_path": "cache.b2nd"}, ValueError, "mutually exclusive"), ("memory://table.b2z", {"mode": "a"}, NotImplementedError, "mode='r'"), ("memory://table.b2z", {"offset": 1, "lazy": True}, NotImplementedError, "offset"), - ("memory://table.b2z", {"mmap_mode": "r"}, ValueError, "requires lazy=True"), + ("memory://table.b2z", {"mmap_mode": "r"}, NotImplementedError, "mmap_mode"), ], ) def test_open_shared_cache_invalid_options(tmp_path, url, options, error, message): From daf89c362faa057eeee8b1683ede817ea373bbc9 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Thu, 24 Sep 2026 05:10:32 +0200 Subject: [PATCH 73/82] Open local PyTables tables as RemoteCTable --- doc/guides/remote_arrays.md | 11 +++--- doc/guides/remote_tables.md | 5 +-- doc/reference/remotectable.rst | 5 ++- src/blosc2/remote_array.py | 4 ++- src/blosc2/remote_store.py | 8 ++--- src/blosc2/schunk.py | 64 ++++++++++++++++++++++++++++++---- tests/test_hdf5_source.py | 22 ++++++++++++ 7 files changed, 101 insertions(+), 18 deletions(-) diff --git a/doc/guides/remote_arrays.md b/doc/guides/remote_arrays.md index 93d7e5a4b..82da45f3b 100644 --- a/doc/guides/remote_arrays.md +++ b/doc/guides/remote_arrays.md @@ -132,11 +132,14 @@ a failed metadata request to sources that do not publish one. Version-1 native indexes remain readable; legacy Kerchunk/reference maps are not native indexes and are rejected. -Local HDF5 files use h5py directly, without pre-indexing or an fsspec +Local HDF5 datasets use h5py directly, without pre-indexing or an fsspec dependency. For example, `blosc2.open("hierarchy.h5::/d0/a2")` reads the selected -dataset through h5py and caches converted Blosc2 chunks in memory. Explicit -`hdf5_index=` also accepts a native HDF5 index for local files. Legacy HDF5 -reference maps are rejected; omit it to regenerate the native index. +dataset through h5py and caches converted Blosc2 chunks in memory. A selected +PyTables table returns `RemoteCTable`, so `blosc2.open("readings.h5", path="readings").where("humidity < 10")` +uses the same query API as a remote table. Local tables also use h5py and support +MEMORY or NONE caching; disk caches are not supported. Explicit `hdf5_index=` +accepts a native HDF5 index for local files. Legacy HDF5 reference maps are +rejected; omit it to regenerate the native index. `RemoteArray` assumes remote sources are immutable by default, avoiding a metadata request before every read. For a replaceable `.b2nd` or Caterva2 source, pass `assume_immutable=False` to refresh its identity and invalidate stale cached chunks before each operation. diff --git a/doc/guides/remote_tables.md b/doc/guides/remote_tables.md index 756e8dbf8..5f4d027d9 100644 --- a/doc/guides/remote_tables.md +++ b/doc/guides/remote_tables.md @@ -5,8 +5,9 @@ Fixed-width, `blosc2.utf8()`, batch-backed variable-length, list, struct/object, and dictionary columns are fetched on demand, including their null masks. A table inside a hierarchy can also be opened through `RemoteStore`. -`blosc2.open()` dispatches local table archives to `CTable` and remote table -archives to `RemoteCTable`. Remote `.b2z` groups return `RemoteStore` by default; +`blosc2.open()` dispatches local B2Z table archives to `CTable`, and remote B2Z +archives and selected local or remote PyTables tables to `RemoteCTable`. +Remote `.b2z` groups return `RemoteStore` by default; array leaves retain their `RemoteArray` behavior. Use `path="group/table"` or a `::group/table` URL suffix to select a nested table. For a complete local download instead, pass `lazy=False, cache_dir="download-cache"`. diff --git a/doc/reference/remotectable.rst b/doc/reference/remotectable.rst index 4cfb5b4b7..22ea13dda 100644 --- a/doc/reference/remotectable.rst +++ b/doc/reference/remotectable.rst @@ -4,8 +4,11 @@ RemoteCTable ============ ``RemoteCTable`` is a read-only :class:`blosc2.CTable` backed by a remote B2Z -archive. Fixed-width, shaped, nullable, UTF-8, batch-backed variable-length, +archive or a local or remote PyTables/HDF5 table. Fixed-width, shaped, nullable, +UTF-8, batch-backed variable-length, batch-backed list, struct/object, and dictionary columns are fetched on demand. +Open local PyTables tables through :func:`blosc2.open` with ``path=`` or a +``::table`` selector. Standalone tables can be opened directly; tables inside a hierarchy can be selected with ``dataset=`` or through :class:`blosc2.RemoteStore`. PyTables/HDF5 sources may supply ``hdf5_index=`` as a native index dictionary, diff --git a/src/blosc2/remote_array.py b/src/blosc2/remote_array.py index 5c2075b3f..b0c35c61d 100644 --- a/src/blosc2/remote_array.py +++ b/src/blosc2/remote_array.py @@ -325,7 +325,9 @@ def _validate_authorized_source(urlpath, storage_options, source_descriptor, *, if source_descriptor != expected: raise ValueError("source_descriptor does not match the supplied source") persisted_url = urlpath.urlbase if isinstance(urlpath, blosc2.C2Array) else urlpath.urlpath - if persisted_url is not None: + if persisted_url is not None and not ( + store_attachment and isinstance(urlpath, hdf5_cls) and urlpath._local + ): validate_persistable_url(persisted_url) return urlpath, dict(expected) diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index 1995956e6..d6cf811d7 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -209,11 +209,11 @@ def __init__( # ponytail: serialize store operations; finer locks if multi-leaf throughput matters. self.lock = threading.RLock() try: - import fsspec - - options = {**self.storage_options, "skip_instance_cache": True} self.filesystem = _filesystem - if self.filesystem is None: + if self.filesystem is None and not (self.format == "hdf5" and os.path.isfile(self.urlpath)): + import fsspec + + options = {**self.storage_options, "skip_instance_cache": True} self.filesystem, _ = fsspec.core.url_to_fs(self.urlpath, **options) self.restored_manifest = manifest if manifest: diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index 403089e0e..a783c28b9 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2411,13 +2411,64 @@ def _open_remote_b2z(urlpath, options): raise +def _open_local_hdf5(urlpath, options): + """Use table discovery only for PyTables nodes; keep ordinary datasets on h5py.""" + import h5py + + dataset = options.get("dataset") + if dataset: + with h5py.File(urlpath, "r") as h5file: + node = h5file.get(dataset.strip("/")) + is_table = isinstance(node, h5py.Dataset) and node.attrs.get("CLASS") in {"TABLE", b"TABLE"} + if is_table: + if options["cache_dir"] is not None or options["cache_path"] is not None: + raise NotImplementedError("Local HDF5 tables do not support disk caches") + if options["assume_immutable"] is not True: + raise NotImplementedError("Local HDF5 tables require assume_immutable=True") + if options.get("storage_options") is not None: + raise ValueError("storage_options is only supported for fsspec URLs") + from blosc2.proxy import CacheCoordinator + from blosc2.remote_array import CACHE_POLICY_DEFAULT, normalize_cache_limit + from blosc2.remote_store import RemoteDiscovery + + policy = options["cache_policy"] + if policy is CACHE_POLICY_DEFAULT: + policy = blosc2.CachePolicy.MEMORY + if policy is blosc2.CachePolicy.DISK: + raise NotImplementedError("Local HDF5 tables do not support disk caches") + if not isinstance(policy, blosc2.CachePolicy): + raise TypeError("cache_policy must be a blosc2.CachePolicy instance") + limit = normalize_cache_limit(policy, options.get("max_cache_bytes", CACHE_POLICY_DEFAULT)) + owner = RemoteDiscovery( + urlpath, + dataset=dataset, + _source_format="hdf5", + _hdf5_index=options.get("hdf5_index"), + ) + owner.cache_policy = policy + owner.max_cache_bytes = limit + owner.cache_coordinator = CacheCoordinator(limit) + try: + return blosc2.RemoteCTable._from_owner( + owner, + dataset.strip("/"), + **( + {} + if options["max_concurrency"] is None + else {"max_concurrency": options["max_concurrency"]} + ), + ) + except BaseException: + owner.close() + raise + return blosc2.RemoteArray(urlpath, **options) + + def _open_remote_hdf5(urlpath, options): """Discover HDF5 groups and PyTables tables while retaining array-only options.""" - if ( - not is_fsspec_url(urlpath) - or options["cache_path"] is not None - or options["assume_immutable"] is not True - ): + if not is_fsspec_url(urlpath): + return _open_local_hdf5(urlpath, options) + if options["cache_path"] is not None or options["assume_immutable"] is not True: return blosc2.RemoteArray(urlpath, **options) dataset = options.get("dataset") hdf5_index = options.get("hdf5_index") @@ -2657,7 +2708,8 @@ def open( For an fsspec URL or a Caterva2 :ref:`URLPath`, return a :ref:`RemoteArray` over the remote array dataset and read the byte ranges a slice touches. B2Z and HDF5 table and group nodes return :class:`RemoteCTable` and - :class:`RemoteStore` instead. + :class:`RemoteStore` instead. Selected local PyTables/HDF5 tables also + return :class:`RemoteCTable`; local ordinary datasets remain :class:`RemoteArray`. A slice landing in a small part of a large chunk costs only the *blocks* it touches when ranges are available; chunks small enough to be one cheap request are still fetched whole. diff --git a/tests/test_hdf5_source.py b/tests/test_hdf5_source.py index ddc6673d5..41682cbe3 100644 --- a/tests/test_hdf5_source.py +++ b/tests/test_hdf5_source.py @@ -465,6 +465,28 @@ def blocked_import(name, *args, **kwargs): assert not file_id.valid +@pytest.mark.parametrize("syntax", ["path", "separator"]) +def test_local_pytables_table_opens_as_ctable(tmp_path, monkeypatch, syntax): + path = tmp_path / "table.h5" + data = np.array([(0, 3), (1, 12), (2, 7)], dtype=[("id", "i4"), ("humidity", "i4")]) + with h5py.File(path, "w") as file: + table = file.create_dataset("readings", data=data, chunks=(2,)) + table.attrs["CLASS"] = "TABLE" + + real_import = builtins.__import__ + + def blocked_import(name, *args, **kwargs): + if name.split(".")[0] == "fsspec": + raise AssertionError("Local HDF5 must not import fsspec") + return real_import(name, *args, **kwargs) + + monkeypatch.setattr(builtins, "__import__", blocked_import) + table = blosc2.open(path, path="readings") if syntax == "path" else blosc2.open(f"{path}::readings") + assert isinstance(table, blosc2.RemoteCTable) + assert table.where("humidity < 10")["id"][:].tolist() == [0, 2] + table.close() + + def test_local_hdf5_closed_source_rejects_reads(tmp_path): path = tmp_path / "closed-local.h5" with h5py.File(path, "w") as file: From 2f5a4d7e1d9d1b76c4814d3eba42e22097ab8259 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Thu, 24 Sep 2026 05:20:18 +0200 Subject: [PATCH 74/82] Cache local PyTables tables across runs --- doc/guides/remote_arrays.md | 5 ++- src/blosc2/remote_store.py | 27 +++++++++++- src/blosc2/schunk.py | 43 ++++++++------------ tests/ctable/test_remote_pytables_interop.py | 29 +++++++++++++ 4 files changed, 73 insertions(+), 31 deletions(-) diff --git a/doc/guides/remote_arrays.md b/doc/guides/remote_arrays.md index 82da45f3b..e631ff87a 100644 --- a/doc/guides/remote_arrays.md +++ b/doc/guides/remote_arrays.md @@ -136,8 +136,9 @@ Local HDF5 datasets use h5py directly, without pre-indexing or an fsspec dependency. For example, `blosc2.open("hierarchy.h5::/d0/a2")` reads the selected dataset through h5py and caches converted Blosc2 chunks in memory. A selected PyTables table returns `RemoteCTable`, so `blosc2.open("readings.h5", path="readings").where("humidity < 10")` -uses the same query API as a remote table. Local tables also use h5py and support -MEMORY or NONE caching; disk caches are not supported. Explicit `hdf5_index=` +uses the same query API as a remote table. Local tables also use h5py. Add +`cache_dir="table-cache"` to reuse converted chunks and PyTables index sidecars +across processes; the cache is rebuilt when the local file changes. Explicit `hdf5_index=` accepts a native HDF5 index for local files. Legacy HDF5 reference maps are rejected; omit it to regenerate the native index. diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index d6cf811d7..c4447b7ff 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -111,6 +111,24 @@ def _resolve_hdf5_options(hdf5_index, private_index, source_format): return index, "hdf5" if index is not None and source_format is None else source_format +def _local_hdf5_stat(urlpath, source_format, allow_local): + if not (allow_local and source_format == "hdf5" and os.path.isfile(urlpath)): + validate_persistable_url(urlpath) + return None + stat = os.stat(urlpath) + return (stat.st_dev, stat.st_ino, stat.st_size, stat.st_mtime_ns, stat.st_ctime_ns) + + +def _reuse_local_hdf5_manifest(manifest, source_stat): + if ( + source_stat is not None + and manifest is not None + and tuple(manifest["metadata"].get("local_source_stat", ())) != source_stat + ): + return None + return manifest + + class RemoteDiscovery: """Shared metadata and source resources, independent of browser presentation.""" @@ -1466,6 +1484,7 @@ def __init__( _traffic=None, nested_storage_options=None, _b2z_blob=None, + _allow_local_hdf5=False, ): dataset = blosc2.core.resolve_dataset_path(dataset, path) if not isinstance(urlpath, (str, os.PathLike)): @@ -1483,7 +1502,8 @@ def __init__( self._validate_nested_storage_options(nested_storage_options) cache_policy, limit = self._validate_cache_config(cache_policy, max_cache_bytes, cache_dir) base_url, _, _ = parse_container_url(urlpath, dataset) - validate_persistable_url(base_url) + local_source_stat = _local_hdf5_stat(base_url, _source_format, _allow_local_hdf5) + local_hdf5 = local_source_stat is not None disk = None source_cache_path = source_cache_marker = None manifest = _manifest @@ -1494,7 +1514,8 @@ def __init__( disk = StoreDiskCache(cache_dir, source) try: manifest = disk.load() if disk is not None else manifest - if disk is not None and source["kind"] == "hdf5" and cache_dir is not None: + manifest = _reuse_local_hdf5_manifest(manifest, local_source_stat) + if disk is not None and source["kind"] == "hdf5" and not local_hdf5: from blosc2.hdf5_source import prepare_hdf5_source_cache source_cache_path, source_cache_marker, _hdf5_blob, hdf5_index, manifest = ( @@ -1527,6 +1548,8 @@ def __init__( _b2z_blob=_b2z_blob, ) manifest = owner.restored_manifest + if local_hdf5: + owner.hdf5_index["local_source_stat"] = local_source_stat owner.attach_hdf5_source_cache(source_cache_path, source_cache_marker) except BaseException: if disk is not None: diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index a783c28b9..091f63aeb 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2421,36 +2421,26 @@ def _open_local_hdf5(urlpath, options): node = h5file.get(dataset.strip("/")) is_table = isinstance(node, h5py.Dataset) and node.attrs.get("CLASS") in {"TABLE", b"TABLE"} if is_table: - if options["cache_dir"] is not None or options["cache_path"] is not None: - raise NotImplementedError("Local HDF5 tables do not support disk caches") + if options["cache_path"] is not None: + raise NotImplementedError("Local HDF5 tables use cache_dir, not cache_path") if options["assume_immutable"] is not True: raise NotImplementedError("Local HDF5 tables require assume_immutable=True") if options.get("storage_options") is not None: raise ValueError("storage_options is only supported for fsspec URLs") - from blosc2.proxy import CacheCoordinator - from blosc2.remote_array import CACHE_POLICY_DEFAULT, normalize_cache_limit - from blosc2.remote_store import RemoteDiscovery - - policy = options["cache_policy"] - if policy is CACHE_POLICY_DEFAULT: - policy = blosc2.CachePolicy.MEMORY - if policy is blosc2.CachePolicy.DISK: - raise NotImplementedError("Local HDF5 tables do not support disk caches") - if not isinstance(policy, blosc2.CachePolicy): - raise TypeError("cache_policy must be a blosc2.CachePolicy instance") - limit = normalize_cache_limit(policy, options.get("max_cache_bytes", CACHE_POLICY_DEFAULT)) - owner = RemoteDiscovery( + store_options = { + key: value + for key, value in options.items() + if key in {"dataset", "cache_dir", "cache_policy", "max_cache_bytes", "hdf5_index"} + } + with blosc2.RemoteStore( urlpath, - dataset=dataset, + _allow_array_root=True, + _allow_local_hdf5=True, _source_format="hdf5", - _hdf5_index=options.get("hdf5_index"), - ) - owner.cache_policy = policy - owner.max_cache_bytes = limit - owner.cache_coordinator = CacheCoordinator(limit) - try: + **store_options, + ) as store: return blosc2.RemoteCTable._from_owner( - owner, + store._owner, dataset.strip("/"), **( {} @@ -2458,9 +2448,6 @@ def _open_local_hdf5(urlpath, options): else {"max_concurrency": options["max_concurrency"]} ), ) - except BaseException: - owner.close() - raise return blosc2.RemoteArray(urlpath, **options) @@ -2730,7 +2717,9 @@ def open( local copy — either the whole thing, or just the chunks and blocks ``lazy`` has fetched so far (as a persistent :ref:`RemoteArray` with :attr:`CachePolicy.DISK`). Either way a later run starts from what is already there, and the copy is discarded when the remote no longer matches - it. There is no default on purpose, so nothing writes to a disk you did not name. + it. For selected local PyTables/HDF5 tables, retains converted chunks and index sidecars + across runs and rebuilds them when the source file changes. There is no default on purpose, + so nothing writes to a disk you did not name. cache_path: str | pathlib.Path, optional With ``lazy=True``, the exact file to use for the remote array's persistent :ref:`RemoteArray` cache (:attr:`CachePolicy.DISK`). Mutually exclusive with diff --git a/tests/ctable/test_remote_pytables_interop.py b/tests/ctable/test_remote_pytables_interop.py index b7070d252..3057ea7d6 100644 --- a/tests/ctable/test_remote_pytables_interop.py +++ b/tests/ctable/test_remote_pytables_interop.py @@ -71,6 +71,35 @@ def test_native_pytables_csi(tmp_path): np.testing.assert_array_equal(table.where("(id >= 15) & (id < 25)").id[:], expected["id"]) +def test_local_pytables_disk_cache_reuse_and_invalidation(tmp_path): + native_pytables_url(tmp_path, "local-csi.h5", csi=True) + path = tmp_path / "local-csi.h5" + cache_dir = tmp_path / "cache" + options = {"path": "table", "cache_dir": cache_dir} + + with blosc2.open(path, **options) as table: + assert sorted(table.where("id < 10").id[:].tolist()) == list(range(10)) + generation = table._storage._owner.generation + descriptor = table._get_index_catalog()["id"] + assert descriptor["persistent"] + + with blosc2.open(path, **options) as table: + assert table._storage._owner.generation == generation + assert table._get_index_catalog()["id"]["opsi"]["values_path"] == descriptor["opsi"]["values_path"] + + import h5py + + with h5py.File(path, "r+") as h5file: + row = h5file["table"][0] + row["value"] = 999 + h5file["table"][0] = row + + with blosc2.open(path, **options) as table: + assert table._storage._owner.generation != generation + assert table["value"][0] == 999 + assert sorted(table.where("id < 10").id[:].tolist()) == list(range(10)) + + def test_native_pytables_light_index_falls_back_to_scan(tmp_path): url, _, expected, _ = native_pytables_url(tmp_path, "native-light.h5", index_kind="light") with blosc2.RemoteCTable(url, dataset="table") as table: From e3c797cb473ae960e669f81903b5974e254d3ff5 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Thu, 24 Sep 2026 06:20:08 +0200 Subject: [PATCH 75/82] Support persistent caches for local data sources --- doc/guides/remote_arrays.md | 36 +++- doc/guides/remote_tables.md | 14 +- plans/local-cache-dir.md | 189 +++++++++++++++++++ src/blosc2/remote_array.py | 58 +++++- src/blosc2/remote_store.py | 58 +++--- src/blosc2/schunk.py | 145 ++++++++++---- tests/ctable/test_remote_ctable.py | 16 ++ tests/ctable/test_remote_pytables_interop.py | 6 +- tests/test_hdf5_source.py | 27 ++- tests/test_remote_array.py | 98 ++++++++++ tests/test_zarr_source.py | 17 ++ 11 files changed, 588 insertions(+), 76 deletions(-) create mode 100644 plans/local-cache-dir.md diff --git a/doc/guides/remote_arrays.md b/doc/guides/remote_arrays.md index e631ff87a..ef5423a0d 100644 --- a/doc/guides/remote_arrays.md +++ b/doc/guides/remote_arrays.md @@ -65,6 +65,35 @@ Both spellings share cache identities and portable artifacts. The exported `B2ZNDSource`, `HDF5NDSource`, `scan_hdf5_index()` and `validate_hdf5_index()` APIs also accept `path=` alongside `dataset=`. +## Persistent caches for local sources + +Passing `cache_dir=` or `cache_path=` to `blosc2.open()` also opts local sources +into read-only, on-demand caching. This supports standalone `.b2nd` arrays, +array leaves in `.b2z`, HDF5 datasets, and Zarr arrays. Tables and groups use +`cache_dir=` and return `RemoteCTable` or `RemoteStore`; `cache_path=` is only +for arrays. Without a cache option, local files keep their normal native open +behavior. + +```python +array = blosc2.open("local-data.h5", path="measurements", cache_dir="cache") +same_array = blosc2.open( + "local-data.h5", path="measurements", cache_path="measurements.b2nd" +) +``` + +Local caches assume the source is immutable and are keyed by its absolute path +and selected node. Replacing data at that path violates the assumption and can +mix cached values with uncached reads. For a local `RemoteCTable` or root +`RemoteStore`, call `refresh()` after changing the source; this rebuilds source +metadata, cached payload, and table indexes together. For a standalone +`RemoteArray`, use a fresh cache location or remove its cache before reopening. +Local-source cache files cannot be exported as portable remote references. + +Local cached opens require `mode="r"` and `assume_immutable=True`; write modes, +`lazy=False`, `mmap_mode`, embedded-frame offsets, and `shared_cache=True` are +not supported. Native array sources must be contiguous files, not sparse frame +directories. Close other cached handles before refreshing or replacing a cache. + Remote B2Z needs `pip install "blosc2[fsspec]"`. HTTP and HTTPS URLs work out of the box; cloud object stores need their respective protocol driver (such as `s3fs` for S3, `gcsfs` for GCS, or `adlfs` for Azure). It accesses external `ZIP_STORED` NDArray members using native Blosc2 chunk and @@ -138,11 +167,14 @@ dataset through h5py and caches converted Blosc2 chunks in memory. A selected PyTables table returns `RemoteCTable`, so `blosc2.open("readings.h5", path="readings").where("humidity < 10")` uses the same query API as a remote table. Local tables also use h5py. Add `cache_dir="table-cache"` to reuse converted chunks and PyTables index sidecars -across processes; the cache is rebuilt when the local file changes. Explicit `hdf5_index=` +across processes. Local caches assume the source is immutable: a changed file at +the same path is not detected automatically. Call `refresh()` on the root +`RemoteCTable`/`RemoteStore`, or clear the array cache before reopening. +Explicit `hdf5_index=` accepts a native HDF5 index for local files. Legacy HDF5 reference maps are rejected; omit it to regenerate the native index. -`RemoteArray` assumes remote sources are immutable by default, avoiding a metadata request before every read. +`RemoteArray` assumes sources are immutable by default, avoiding a metadata request before every read. For a replaceable `.b2nd` or Caterva2 source, pass `assume_immutable=False` to refresh its identity and invalidate stale cached chunks before each operation. Mutable B2Z, Zarr, and HDF5 sources are not supported. diff --git a/doc/guides/remote_tables.md b/doc/guides/remote_tables.md index 5f4d027d9..2918cab78 100644 --- a/doc/guides/remote_tables.md +++ b/doc/guides/remote_tables.md @@ -5,12 +5,16 @@ Fixed-width, `blosc2.utf8()`, batch-backed variable-length, list, struct/object, and dictionary columns are fetched on demand, including their null masks. A table inside a hierarchy can also be opened through `RemoteStore`. -`blosc2.open()` dispatches local B2Z table archives to `CTable`, and remote B2Z -archives and selected local or remote PyTables tables to `RemoteCTable`. +`blosc2.open()` dispatches uncached local B2Z table archives to `CTable`, and +remote B2Z archives and selected local or remote PyTables tables to +`RemoteCTable`. Supplying `cache_dir=` also selects `RemoteCTable` for a local +B2Z table and retains its accessed chunks and indexes on disk. Remote `.b2z` groups return `RemoteStore` by default; array leaves retain their `RemoteArray` behavior. Use `path="group/table"` or a `::group/table` URL suffix to select a nested table. For a complete local download instead, pass `lazy=False, cache_dir="download-cache"`. +Passing `cache_dir` for a local `.b2z` table or group opts into a persistent, +read-only cache, just as it does for remote sources. ```python with blosc2.open("https://example.org/readings.b2z") as table: @@ -125,9 +129,9 @@ costs separately. ## Refresh a remote table -Remote containers are assumed immutable. A standalone table with a writable -cache can call `refresh()` to rediscover its schema and replace the cache -generation while preserving cache limits and parallel-read settings: +Cached sources, local or remote, are assumed immutable. A standalone table with +a writable cache can call `refresh()` to rediscover its schema and replace the +cache generation while preserving cache limits and parallel-read settings: ```python with blosc2.RemoteCTable(url, cache_dir="table-cache") as table: diff --git a/plans/local-cache-dir.md b/plans/local-cache-dir.md new file mode 100644 index 000000000..c01146e1d --- /dev/null +++ b/plans/local-cache-dir.md @@ -0,0 +1,189 @@ +# Consistent caching for local and remote sources + +## Objective and agreed contract + +An explicit `cache_dir=` requests a real persistent cache, whether the source is +local or remote and whether its format is native Blosc2, HDF5, or Zarr. A local +source may live on slower storage than the cache. The library must honor the +request without deciding whether caching is worthwhile on the user's machine. + +`cache_path=` follows the same caching and validity model. Its distinction is +placement: it names the exact cache file for one array. `cache_dir=` names a +parent directory under which source-specific array, table, or store caches are +managed. Do not introduce a single-file cache representation for tables/stores. + +Implemented using the existing array carriers and store caches. Local runtime +descriptors remain separate from portable remote references. The behavior below +is the agreed contract; the next section records the original starting point. + +## Original behavior to replace + +Verified against the current working tree: + +| Local source | Current behavior with `cache_dir=` | +| --- | --- | +| Selected PyTables table | Persistent chunks and imported index sidecars; file-stat invalidation | +| Ordinary HDF5 dataset | Rejected by remote URL validation | +| Standalone or nested Zarr array | Rejected by remote URL validation | +| Native `.b2nd` | Returns the native array and silently ignores caching | +| Native `.b2z` root | Uncommitted patch accepts and ignores caching | +| Array selected inside `.b2z` | Rejected by remote URL validation | + +Replace the uncommitted `.b2z` no-op code, its documentation, and its test that +expects no cache directory. Remove local PyTables automatic file-stat +invalidation in favor of the immutable-source contract below. + +## Public behavior + +### Cache placement and returned objects + +| Selected object | `cache_dir=` | `cache_path=` | Cached handle | +| --- | --- | --- | --- | +| Standalone `.b2nd` array | Supported | Supported | `RemoteArray` | +| Array inside `.b2z` | Supported | Supported | `RemoteArray` | +| HDF5 dataset | Supported | Supported | `RemoteArray` | +| Zarr array, standalone or nested | Supported | Supported | `RemoteArray` | +| Native B2Z or PyTables table | Supported | Reject; direct users to `cache_dir` | `RemoteCTable` | +| Supported container group | Supported | Reject; direct users to `cache_dir` | `RemoteStore` | + +The existing wrapper names also apply to local sources. Document that choosing +caching can change a native `NDArray`/`CTable` into its read-only cached wrapper. +Without cache options, preserve current dispatch and native writable access. +Discover a selected node's kind before applying array-only cache options; +`cache_path` must not accidentally turn a PyTables table into a record array. + +Both placement options imply `CachePolicy.DISK` when policy is omitted. They are +mutually exclusive. An explicit incompatible policy raises a clear error. +Retain the existing 256 MiB default compressed-payload budget and +`max_cache_bytes=None` behavior. Table/store leaves share their owner's budget. +Metadata and derived index sidecars retain their existing budget treatment; +do not describe the payload limit as a bound on total disk consumption. + +Omitted or `None` `lazy` selects on-demand access for the new local caching +paths. Opening may create cache metadata; reads populate missing payload. +Reuse compressed native chunks where possible. Convert HDF5/Zarr data through +existing source adapters. Cache indexes using the existing table machinery. +Do not add decoded-data or query-result caching. + +Initially support local cached opens in `mode="r"`. Reject write modes and +incompatible `mmap_mode`, embedded-frame offsets, and explicit `lazy=False` +with actionable messages. Preserve existing remote eager-localization behavior +where already supported; do not silently reinterpret `lazy=False`. + +Keep the deprecated `cache_storage` alias on the same normalization path, with +its existing warning and mutual-exclusion checks. + +### Source validity + +Default to `assume_immutable=True` for local cached sources, as for remote +sources. Reopening an unchanged source identity reuses the cache; it does not +check modification times, hash payload, scan directories, or revalidate cached +chunks against the source. + +Changing a source at the same identity violates that assumption. Cached and +uncached reads can otherwise mix versions. Users must close other handles and +explicitly refresh/rebuild or clear the cache before reading a changed source. +Do not promise offline access merely because some chunks have been cached. + +Refresh/rebuild must discard stale source metadata, payload, and imported index +sidecars together, then publish a fresh generation. Audit the existing array, +table, and store refresh APIs and document the supported procedure for each; +reuse them rather than adding a competing invalidation API. + +Reject `assume_immutable=False` for new local paths until change detection is +implemented. Preserve remote cases that already support it. Removing local +PyTables stat invalidation must not remove structural, geometry, descriptor, +cache-identity, or corruption checks. + +### Identity and persistence + +Local runtime cache identities must include a normalized absolute source path, +the selected node, source format, and relevant representation settings. Resolve +relative paths at open time so another working directory cannot reuse an +unrelated source's cache. Test equivalent relative/absolute paths and `file://` +inputs, and distinguish local sources from remote URLs in a shared cache parent. +Do not include file timestamps or content digests in the immutable identity. + +An explicit `cache_path` already associated with another source or incompatible +representation must fail rather than overwrite or reuse that cache. Reject a +cache destination that would overwrite its source. + +Treat local runtime cache descriptors separately from portable remote +references. Do not globally relax `validate_persistable_url()`: a loaded remote +reference must not acquire arbitrary local-file access. Ensure supported local +cache reopening retains an unambiguous absolute source identity. Keep portable +remote export restrictions unless local-reference export is explicitly designed. + +## Implementation sequence + +1. **Unify dispatch and option validation in `src/blosc2/schunk.py`.** + Detect an explicit cache request before the native local fast path. Route + local arrays/tables/groups to the existing cached readers. Remove the B2Z + no-op and ensure `cache_path` follows the same node discovery as `cache_dir`. + Keep ordinary uncached native opens on their current path. + +2. **Generalize local source attachment.** + Reuse `remote_array.py`, `remote_store.py`, and the existing source adapters. + Replace the HDF5-table-only local exception with narrowly scoped support for + authorized local runtime sources. Keep direct h5py reads for local HDF5 and + preserve their lack of a required fsspec dependency. Support native contiguous + arrays first through the existing chunk reader; explicitly check sparse + `.b2nd` layouts and reject unsupported layouts clearly rather than ignoring + caching. Use existing B2Z and Zarr discovery/readers for those formats. + +3. **Reuse persistent cache storage.** + Attach local arrays to the existing carrier/proxy cache and local tables and + groups to `StoreDiskCache`. Preserve resource ownership, cache budgets, + incomplete-cache recovery, and source separation. Retain HDF5 converted + indexes and native B2Z index access through the current table reader. + Avoid a second local-only cache backend or full-file copies on every open. + +4. **Apply the immutable-source rule end to end.** + Remove `_local_hdf5_stat`, `_reuse_local_hdf5_manifest`, and their automatic + invalidation behavior. Treat old local stat metadata as obsolete rather than + requiring it for reuse. Audit adapter stamps, small-file bootstrap caches, + and cache reopening for implicit checks that would undermine this contract. + Verify explicit refresh rebuilds all relevant metadata and payload together. + +5. **Document and verify the public contract.** + Update `open()` and wrapper documentation plus the remote array/table guides. + Explain local cache placement, wrapper return types, the array-only meaning + of `cache_path`, immutable sources, and refresh procedures. Replace statements + that local native caches are ignored or PyTables caches automatically detect + file changes. + +## Sharing boundary + +Reuse existing ownership and locking rules. An ordinary cache is not implicitly +safe for concurrent processes. `shared_cache=True` remains an explicit opt-in. +This implementation must not silently ignore it for local sources. Keep the +current clear rejection until local shared attachment is implemented and tested +using the existing sparse shared backend. Do not extend `cache_path` to shared +table/store caches as part of this work. + +## Validation and acceptance + +Use the `blosc2` conda environment for all checks. Add focused tests in the +existing HDF5, Zarr, B2Z, remote array/store, and CTable suites: + +- Each supported local format creates real reusable payload after a read; + directory existence alone is insufficient evidence. +- A fresh process reuses cached payload and PyTables index sidecars without + rereading/reconverting those source payloads. Metadata access may still occur. +- Array `cache_dir` and `cache_path` produce equivalent values and cache policy. + Cover standalone arrays and nested selectors using `path=` and `::`. +- Native uncached opens retain their existing types and behavior. Cached native + table queries preserve null semantics and index correctness. +- Mutating a source does not automatically invalidate cached data under the + immutable contract. The documented explicit rebuild procedure observes the + new data and indexes without mixing generations. +- Test path identity, mismatched explicit cache files, invalid option + combinations, unsupported mutable access, and table/group `cache_path` errors. +- Preserve local HDF5 operation without fsspec, and run remote cache regression + tests to catch accidental changes to remote descriptors or dispatch. + +Run the relevant existing suites, Ruff, and diff checks. The moto S3 HDF5 tests +previously stalled in this environment; report any exclusion explicitly. +Benchmark the user's large B2Z and indexed HDF5 examples in fresh processes with +cold and warm caches. Report timing without requiring a speedup over uncached +native reads: honoring explicit cache placement is the acceptance criterion. diff --git a/src/blosc2/remote_array.py b/src/blosc2/remote_array.py index b0c35c61d..ea7b8ee38 100644 --- a/src/blosc2/remote_array.py +++ b/src/blosc2/remote_array.py @@ -269,6 +269,15 @@ def _validate_max_concurrency(value: int | None) -> int | None: return value +def _local_source_path(urlpath, enabled): + if not enabled: + return None, urlpath + if not isinstance(urlpath, str) or (urlsplit(urlpath).scheme and not os.path.splitdrive(urlpath)[0]): + raise ValueError("local cache sources must use a local filesystem path") + urlpath = os.path.abspath(urlpath) + return urlpath, urlpath + + def _validate_payload_limit(policy: blosc2.CachePolicy, limit) -> None: if policy is blosc2.CachePolicy.NONE: if limit is not None: @@ -280,7 +289,9 @@ def _validate_payload_limit(policy: blosc2.CachePolicy, limit) -> None: raise ValueError(f"persisted {policy.name} RemoteArray requires positive max_cache_bytes") -def _validate_authorized_source(urlpath, storage_options, source_descriptor, *, store_attachment=False): +def _validate_authorized_source( + urlpath, storage_options, source_descriptor, *, store_attachment=False, allow_local_source=False +): if storage_options is not None: raise ValueError("storage_options cannot be used with an authorized source") hdf5_cls = getattr(blosc2, "HDF5NDSource", ()) @@ -326,7 +337,9 @@ def _validate_authorized_source(urlpath, storage_options, source_descriptor, *, raise ValueError("source_descriptor does not match the supplied source") persisted_url = urlpath.urlbase if isinstance(urlpath, blosc2.C2Array) else urlpath.urlpath if persisted_url is not None and not ( - store_attachment and isinstance(urlpath, hdf5_cls) and urlpath._local + store_attachment + and allow_local_source + and (not urlsplit(persisted_url).scheme or os.path.splitdrive(persisted_url)[0]) ): validate_persistable_url(persisted_url) return urlpath, dict(expected) @@ -348,8 +361,9 @@ def _open_url_source( cparams=None, source_cache_dir=None, hdf5_index_explicit=True, + local_source=False, ): - if persistable: + if persistable and not local_source: validate_persistable_url(urlpath) kwargs = {} if max_concurrency is None else {"max_concurrency": max_concurrency} if storage_options is not None: @@ -535,6 +549,9 @@ def _resolve_init_dataset_and_url(urlpath, dataset, source_format, hdf5_index=No class RemoteArray(RemoteObject, blosc2.Operand): """A persistable, optionally self-caching reference to a remote array. + ``blosc2.open(local_path, cache_dir=...)`` also returns this read-only + wrapper for local arrays; these local references cannot be exported. + With :attr:`CachePolicy.DISK`, the public constructor uses the persisted B2ND carrier itself as the bounded cache. Use ``blosc2.open(url, cache_dir=..., shared_cache=True)`` for a process-shared @@ -621,6 +638,7 @@ def __init__( _runtime_is_mutable: bool = True, _defer_cache: bool = False, _shared_cache: bool = False, + _local_source: bool = False, ): dataset = blosc2.core.resolve_dataset_path(dataset, path) if not isinstance(cache_policy, blosc2.CachePolicy): @@ -637,6 +655,9 @@ def __init__( urlpath, self._dataset, self._source_format = _resolve_init_dataset_and_url( urlpath, dataset, source_format, hdf5_index ) + _local_source = _local_source and cache_policy is blosc2.CachePolicy.DISK + self._local_source = _local_source + self._local_source_path, urlpath = _local_source_path(urlpath, _local_source) self._authorized_source = _source_descriptor is not None hdf5_index_explicit = hdf5_index is not None shared_index_path = None @@ -658,7 +679,11 @@ def __init__( ) if self._authorized_source: self.src, self._source = _validate_authorized_source( - urlpath, storage_options, _source_descriptor, store_attachment=_store_owner is not None + urlpath, + storage_options, + _source_descriptor, + store_attachment=_store_owner is not None, + allow_local_source=getattr(_store_owner, "local_source", False), ) else: read_seed = ( @@ -707,7 +732,10 @@ def __init__( cparams=_source_cparams, source_cache_dir=cache_dir if cache_policy is blosc2.CachePolicy.DISK else None, hdf5_index_explicit=hdf5_index_explicit, + local_source=_local_source, ) + if _local_source: + self.src.stamp = ("local", urlpath, self._dataset, self._source_format) self._assume_immutable = assume_immutable self._storage_options = storage_options if _shared_cache: @@ -862,6 +890,11 @@ def _carrier_path( path = os.fspath(cache_path) if os.path.isdir(path): raise ValueError("cache_path must name a file, not a directory") + if self._local_source and ( + os.path.abspath(path) == self._local_source_path + or (os.path.exists(path) and os.path.samefile(path, self._local_source_path)) + ): + raise ValueError("cache_path cannot overwrite the local source") return path if urlpath is None: urlpath = self._source.get("urlpath", self._source_identity()) @@ -869,11 +902,14 @@ def _carrier_path( if self._source_format == "zarr" and self._dataset: parsed = urlsplit(urlpath) urlpath = urlunsplit(parsed._replace(path=parsed.path.rstrip("/")[: -len(self._dataset) - 1])) + cache_dataset = self._dataset + if self._local_source: + cache_dataset = f"{self._source_format}::{self._dataset or ''}" return fsspec_cache_path( urlpath, cache_dir, ".b2nd", - dataset=self._dataset, + dataset=cache_dataset, storage_options=storage_options, create_parent=create_parent, ) @@ -1213,6 +1249,7 @@ def _open_source( cparams=None, source_cache_dir=None, hdf5_index_explicit=True, + local_source=False, ): if isinstance(urlpath, blosc2.C2Array): if source_format not in {None, "blosc2"}: @@ -1263,6 +1300,7 @@ def _open_source( cparams=cparams, source_cache_dir=source_cache_dir, hdf5_index_explicit=hdf5_index_explicit, + local_source=local_source, ) else: raise TypeError("RemoteArray requires a URL string, URLPath, or C2Array") @@ -1488,6 +1526,7 @@ def info_items(self) -> list[tuple[str, object]]: @property def source(self) -> dict: """A copy of the credential-free source descriptor.""" + self._ensure_exportable_source() self._payload() # Runtime-only URLs must not escape as portable descriptors. return dict(self._source) @@ -1724,12 +1763,12 @@ async def aget_chunk(self, nchunk: int) -> bytes: def _payload(self, mutable=None): url = self._source.get("urlpath", self._source.get("urlbase")) - if url is not None: + if url is not None and not self._local_source: validate_persistable_url(url) return { "kind": "remote_array", "version": 1, - "source": dict(self._source), + "source": {**self._source, **({"local": True} if self._local_source else {})}, "cache_policy": self.cache_policy.value, "max_cache_bytes": self.max_cache_bytes, "mutable": self.mutable if mutable is None else mutable, @@ -1801,7 +1840,12 @@ def _export_carrier_with_policy(self, cache_policy, effective_mutable): write_b2object_payload(carrier, payload) return carrier + def _ensure_exportable_source(self): + if self._local_source: + raise ValueError("local-source caches cannot be exported as portable RemoteArray references") + def _export_carrier(self, include_cache: bool, cache_policy=None, mutable=None): + self._ensure_exportable_source() if not isinstance(include_cache, bool): raise TypeError("include_cache must be a boolean") effective_mutable = self.mutable if mutable is None else mutable diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index c4447b7ff..a25a98f1d 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -111,24 +111,6 @@ def _resolve_hdf5_options(hdf5_index, private_index, source_format): return index, "hdf5" if index is not None and source_format is None else source_format -def _local_hdf5_stat(urlpath, source_format, allow_local): - if not (allow_local and source_format == "hdf5" and os.path.isfile(urlpath)): - validate_persistable_url(urlpath) - return None - stat = os.stat(urlpath) - return (stat.st_dev, stat.st_ino, stat.st_size, stat.st_mtime_ns, stat.st_ctime_ns) - - -def _reuse_local_hdf5_manifest(manifest, source_stat): - if ( - source_stat is not None - and manifest is not None - and tuple(manifest["metadata"].get("local_source_stat", ())) != source_stat - ): - return None - return manifest - - class RemoteDiscovery: """Shared metadata and source resources, independent of browser presentation.""" @@ -151,6 +133,7 @@ def __init__( _source_cache_dir=None, _refresh_source=False, _b2z_blob=None, + _local_source=False, ): self.urlpath, dataset, self.format = parse_container_url(urlpath, dataset) if _source_format is not None: @@ -160,6 +143,7 @@ def __init__( self.root = (dataset or "").strip("/") self._validate(self.root) self.storage_options = storage_options or {} + self.local_source = _local_source self.source_cache_dir = _source_cache_dir self.refresh_source = _refresh_source self.b2z_source_cache = (None, None, None) @@ -1085,6 +1069,7 @@ def prepare_refresh(self, kind): _source_format=self.format, _source_cache_dir=self.source_cache_dir, _refresh_source=True, + _local_source=self.local_source, ) try: if replacement.nodes[replacement.root][0] != kind: @@ -1186,6 +1171,8 @@ def save_selection( overwrite: bool = False, ) -> str: """Export the current store or subtree to a portable .b2z reference archive.""" + if self.local_source: + raise ValueError("local-source caches cannot be exported as portable RemoteStore references") if not isinstance(include_cache, bool): raise TypeError("include_cache must be a boolean") if mutable is not None and not isinstance(mutable, bool): @@ -1397,6 +1384,8 @@ def _copy_leaf_carrier(self, orig_key, proxy, staging_dir): class RemoteStore(RemoteObject): """Read-only remote B2Z, Zarr or HDF5 hierarchy. + Also used for local hierarchies opened with ``blosc2.open(..., cache_dir=...)``. + Discovery and returned array handles share source resources and traffic. MEMORY shares one bounded cache across all leaves; NONE retains no payload. DISK retains payload and discovery under an exclusively owned cache directory. @@ -1439,7 +1428,16 @@ def _try_open_artifact( @staticmethod def _cache_source(urlpath, dataset, source_format, storage_options): base_url, root, kind = parse_container_url(urlpath, dataset) - source = {"urlpath": base_url, "dataset": (root or "").strip("/"), "kind": source_format or kind} + local = not urlsplit(base_url).scheme or bool(os.path.splitdrive(base_url)[0]) + if local: + base_url = os.path.abspath(base_url) + source = { + "urlpath": base_url, + "dataset": (root or "").strip("/"), + "kind": source_format or kind, + } + if local: + source["local"] = True fingerprint = storage_options_fingerprint(storage_options) if fingerprint: source["storage_options"] = fingerprint @@ -1484,7 +1482,7 @@ def __init__( _traffic=None, nested_storage_options=None, _b2z_blob=None, - _allow_local_hdf5=False, + _allow_local_source=False, ): dataset = blosc2.core.resolve_dataset_path(dataset, path) if not isinstance(urlpath, (str, os.PathLike)): @@ -1501,9 +1499,19 @@ def __init__( raise TypeError("dataset must be a string") self._validate_nested_storage_options(nested_storage_options) cache_policy, limit = self._validate_cache_config(cache_policy, max_cache_bytes, cache_dir) - base_url, _, _ = parse_container_url(urlpath, dataset) - local_source_stat = _local_hdf5_stat(base_url, _source_format, _allow_local_hdf5) - local_hdf5 = local_source_stat is not None + base_url, dataset, _ = parse_container_url(urlpath, dataset) + urlpath = base_url + local_source = bool( + _allow_local_source + and (not urlsplit(base_url).scheme or os.path.splitdrive(base_url)[0]) + and os.path.exists(base_url) + ) + if not local_source: + validate_persistable_url(base_url) + else: + base_url = os.path.abspath(base_url) + urlpath = base_url + local_hdf5 = local_source and _source_format == "hdf5" disk = None source_cache_path = source_cache_marker = None manifest = _manifest @@ -1514,7 +1522,6 @@ def __init__( disk = StoreDiskCache(cache_dir, source) try: manifest = disk.load() if disk is not None else manifest - manifest = _reuse_local_hdf5_manifest(manifest, local_source_stat) if disk is not None and source["kind"] == "hdf5" and not local_hdf5: from blosc2.hdf5_source import prepare_hdf5_source_cache @@ -1546,10 +1553,9 @@ def __init__( _traffic=_traffic, _source_cache_dir=cache_dir if disk is not None else None, _b2z_blob=_b2z_blob, + _local_source=local_source, ) manifest = owner.restored_manifest - if local_hdf5: - owner.hdf5_index["local_source_stat"] = local_source_stat owner.attach_hdf5_source_cache(source_cache_path, source_cache_marker) except BaseException: if disk is not None: diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index 091f63aeb..ed9ece47f 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2070,6 +2070,7 @@ def _remote_array_options( assume_immutable=True, dataset=None, hdf5_index=None, + local_source=False, ): """Return the explicit RemoteArray options, or None when remote access was not requested.""" policy_present = "cache_policy" in kwargs @@ -2102,6 +2103,8 @@ def _remote_array_options( options["dataset"] = dataset if hdf5_index is not None: options["hdf5_index"] = hdf5_index + if local_source: + options["_local_source"] = True return options @@ -2273,9 +2276,62 @@ def _open_lazy_remote(urlpath, source_format, options, shared_cache=False): return _open_remote_b2z(urlpath, options) if source_format == "hdf5": return _open_remote_hdf5(urlpath, options) + if source_format == "zarr" and options.get("_local_source"): + if options["cache_path"] is not None: + try: + return blosc2.RemoteArray(urlpath, **options) + except ValueError as exc: + if "is a Zarr group" in str(exc): + raise NotImplementedError("Zarr groups use cache_dir, not cache_path") from exc + raise + store_options = { + key: value + for key, value in options.items() + if key in {"dataset", "cache_dir", "cache_policy", "max_cache_bytes"} + } + with blosc2.RemoteStore( + urlpath, + _allow_array_root=True, + _allow_local_source=True, + _source_format="zarr", + **store_options, + ) as store: + result = store[""] + if options["max_concurrency"] is not None: + if not isinstance(result, blosc2.RemoteArray): + result.close() + raise NotImplementedError( + "max_concurrency is only supported for remote arrays and tables" + ) + result.src.max_concurrency = options["max_concurrency"] + return result return blosc2.RemoteArray(urlpath, **options) +def _normalize_open_source_path(urlpath): + parsed = urlsplit(os.fspath(urlpath)) if isinstance(urlpath, os.PathLike) else urlsplit(urlpath) + local_source = not is_fsspec_url(urlpath) or parsed.scheme == "file" + if parsed.scheme == "file": + from urllib.parse import unquote + + urlpath = os.path.abspath(unquote(parsed.path)) + if local_source and os.path.isdir(urlpath) and urlpath.endswith((".b2nd", ".b2frame", ".b2d")): + raise NotImplementedError( + "local source caching requires a contiguous native frame, not a sparse directory" + ) + return urlpath, local_source + + +def _resolve_local_cache_lazy(local_source, cache_dir, cache_path, lazy, assume_immutable): + if not (local_source and (cache_dir is not None or cache_path is not None)): + return lazy + if lazy is False: + raise NotImplementedError("local source caches require on-demand access; omit lazy=False") + if assume_immutable is not True: + raise NotImplementedError("local source caches require assume_immutable=True") + return True if lazy is None else lazy + + def _open_fsspec_url(urlpath: str, mode: str, offset: int, kwargs: dict): """Open a container living behind an fsspec URL. @@ -2285,12 +2341,16 @@ def _open_fsspec_url(urlpath: str, mode: str, offset: int, kwargs: dict): lazy access, nothing is fetched up front and each slice pulls just the chunks it needs, retaining them under `cache_dir` when supplied. """ + urlpath, local_source = _normalize_open_source_path(urlpath) if mode != "r": - raise NotImplementedError(f"fsspec URLs can only be opened with mode='r', not {mode!r}") + opener = "local cached sources" if local_source else "fsspec URLs" + raise NotImplementedError(f"{opener} can only be opened with mode='r', not {mode!r}") cache_dir, cache_path = _remote_cache_options(kwargs) shared_cache = kwargs.pop("shared_cache", False) storage_options = kwargs.pop("storage_options", None) + if local_source and storage_options is not None: + raise ValueError("storage_options is only supported for fsspec URLs") source_format = kwargs.pop("source_format", None) dataset = kwargs.pop("dataset", None) hdf5_index = kwargs.pop("hdf5_index", None) @@ -2309,6 +2369,7 @@ def _open_fsspec_url(urlpath: str, mode: str, offset: int, kwargs: dict): raise NotImplementedError("hdf5_index is only supported with lazy=True") if lazy is None and source_format == "b2z": lazy = True + lazy = _resolve_local_cache_lazy(local_source, cache_dir, cache_path, lazy, assume_immutable) # Auto-infer lazy=True only when the caller left the choice unspecified. lazy = _resolve_lazy(lazy, dataset, source_format, urlpath) @@ -2324,6 +2385,7 @@ def _open_fsspec_url(urlpath: str, mode: str, offset: int, kwargs: dict): assume_immutable=assume_immutable, dataset=dataset, hdf5_index=hdf5_index, + local_source=local_source, ) if lazy: if offset != 0: @@ -2385,6 +2447,7 @@ def _open_remote_b2z(urlpath, options): with blosc2.RemoteStore( urlpath, _allow_array_root=True, + _allow_local_source=options.get("_local_source", False), _source_format="b2z", _b2z_blob=array_error.blob if array_error is not None else None, **store_options, @@ -2416,29 +2479,33 @@ def _open_local_hdf5(urlpath, options): import h5py dataset = options.get("dataset") + is_table = False + is_group = False if dataset: with h5py.File(urlpath, "r") as h5file: node = h5file.get(dataset.strip("/")) + is_group = isinstance(node, h5py.Group) is_table = isinstance(node, h5py.Dataset) and node.attrs.get("CLASS") in {"TABLE", b"TABLE"} - if is_table: - if options["cache_path"] is not None: - raise NotImplementedError("Local HDF5 tables use cache_dir, not cache_path") - if options["assume_immutable"] is not True: - raise NotImplementedError("Local HDF5 tables require assume_immutable=True") - if options.get("storage_options") is not None: - raise ValueError("storage_options is only supported for fsspec URLs") - store_options = { - key: value - for key, value in options.items() - if key in {"dataset", "cache_dir", "cache_policy", "max_cache_bytes", "hdf5_index"} - } - with blosc2.RemoteStore( - urlpath, - _allow_array_root=True, - _allow_local_hdf5=True, - _source_format="hdf5", - **store_options, - ) as store: + if is_table or is_group or dataset is None: + if not is_table and options["max_concurrency"] is not None: + raise NotImplementedError("max_concurrency is only supported for remote arrays and tables") + if options["cache_path"] is not None: + raise NotImplementedError("HDF5 tables and groups use cache_dir, not cache_path") + if options["assume_immutable"] is not True: + raise NotImplementedError("Local HDF5 caches require assume_immutable=True") + store_options = { + key: value + for key, value in options.items() + if key in {"dataset", "cache_dir", "cache_policy", "max_cache_bytes", "hdf5_index"} + } + with blosc2.RemoteStore( + urlpath, + _allow_array_root=True, + _allow_local_source=options.get("_local_source", False), + _source_format="hdf5", + **store_options, + ) as store: + if is_table: return blosc2.RemoteCTable._from_owner( store._owner, dataset.strip("/"), @@ -2448,6 +2515,7 @@ def _open_local_hdf5(urlpath, options): else {"max_concurrency": options["max_concurrency"]} ), ) + return store[""] return blosc2.RemoteArray(urlpath, **options) @@ -2541,6 +2609,13 @@ def _is_container_open_request(urlpath: str, kwargs: dict) -> bool: ) +def _should_use_fsspec_opener(urlpath, kwargs): + local_cache_requested = any( + kwargs.get(key) is not None for key in ("cache_dir", "cache_path", "cache_storage") + ) + return is_fsspec_url(urlpath) or _is_container_open_request(urlpath, kwargs) or local_cache_requested + + def _try_open_special_store(urlpath: str, mode: str, offset: int, kwargs: dict): if urlpath.endswith((".b2d", ".b2z", ".b2e")): special = _open_special_store(urlpath, mode, offset, **kwargs) @@ -2713,17 +2788,16 @@ def open( their table-specific temporary buffer settings are available through ``RemoteCTable``, not through this general opener. cache_dir: str | pathlib.Path, optional - For fsspec URLs and lazy Caterva2 :ref:`URLPath` objects, a directory holding this container's - local copy — either the whole thing, or just the chunks and blocks ``lazy`` has fetched so far - (as a persistent :ref:`RemoteArray` with :attr:`CachePolicy.DISK`). Either way a later run - starts from what is already there, and the copy is discarded when the remote no longer matches - it. For selected local PyTables/HDF5 tables, retains converted chunks and index sidecars - across runs and rebuilds them when the source file changes. There is no default on purpose, - so nothing writes to a disk you did not name. + Parent directory for a persistent :ref:`RemoteArray`, :class:`RemoteCTable`, or + :class:`RemoteStore` cache. For remote inputs it stores fetched chunks and metadata; + for local Blosc2, HDF5, and Zarr inputs it stores accessed chunks and converted data. + Supplying it selects on-demand, read-only access for local inputs. Local caches assume + the source is immutable: replacing it at the same path requires clearing/rebuilding the + cache. No cache is created unless you name one. cache_path: str | pathlib.Path, optional - With ``lazy=True``, the exact file to use for the remote array's - persistent :ref:`RemoteArray` cache (:attr:`CachePolicy.DISK`). Mutually exclusive with - ``cache_dir``. + Exact file for a persistent array cache (:attr:`CachePolicy.DISK`), for local or remote + standalone arrays and array leaves. Tables and groups require ``cache_dir``. Mutually + exclusive with ``cache_dir``. cache_storage: str | pathlib.Path, optional Deprecated alias for ``cache_dir``. Mutually exclusive with ``cache_dir`` and ``cache_path``. @@ -2785,7 +2859,8 @@ def open( ``lazy=True``. assume_immutable: bool, optional With ``lazy=True``, skip remote identity checks before reads. Defaults - to ``True``; set to ``False`` when the remote object may be replaced. + to ``True``. Local disk caches always assume immutable sources; set to a new path or + explicitly rebuild the cache after changing a local source. Returns ------- @@ -2888,7 +2963,11 @@ def open( if isinstance(urlpath, blosc2.URLPath): return _open_c2_urlpath(urlpath, mode, offset, kwargs) - if offset != 0 and not is_fsspec_url(urlpath): + if ( + offset != 0 + and not is_fsspec_url(urlpath) + and not any(kwargs.get(key) is not None for key in ("cache_dir", "cache_path", "cache_storage")) + ): local_path = normalize_urlpath(os.fspath(urlpath)) if os.path.isfile(local_path): if dataset is not None or hdf5_index is not None: @@ -2898,7 +2977,7 @@ def open( urlpath = _normalize_open_target(urlpath, kwargs, dataset, hdf5_index) - if is_fsspec_url(urlpath) or _is_container_open_request(urlpath, kwargs): + if _should_use_fsspec_opener(urlpath, kwargs): return _open_fsspec_url(urlpath, mode, offset, kwargs) # The native local opener does not consume the public lazy option. diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 9d00950e0..7d0bacad6 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -1710,6 +1710,22 @@ class TextRow: assert table.cache_policy == blosc2.CachePolicy.MEMORY +def test_open_local_b2z_with_cache_dir(tmp_path): + local = blosc2.CTable(Row, [(1, [1, 2], "one")], create_summary_index=False) + path = tmp_path / "table.b2z" + local.to_b2z(path) + cache_dir = tmp_path / "cache" + + with blosc2.open(path, cache_dir=cache_dir) as table: + assert isinstance(table, blosc2.RemoteCTable) + assert table.where("x == 1").x[:].tolist() == [1] + assert cache_dir.exists() + assert list(cache_dir.rglob("active_generation.json")) + assert list(cache_dir.rglob("*.b2nd")) + with pytest.raises(NotImplementedError, match="cache_dir, not cache_path"): + blosc2.open(path, cache_path=tmp_path / "table-cache.b2nd") + + @pytest.mark.parametrize("suffix", [".b2z", ""]) def test_open_dispatches_remote_table_hierarchy(tmp_path, suffix, monkeypatch): local = blosc2.CTable(Row, [(1, [1, 2], "one")], create_summary_index=False) diff --git a/tests/ctable/test_remote_pytables_interop.py b/tests/ctable/test_remote_pytables_interop.py index 3057ea7d6..d58396b5b 100644 --- a/tests/ctable/test_remote_pytables_interop.py +++ b/tests/ctable/test_remote_pytables_interop.py @@ -71,7 +71,7 @@ def test_native_pytables_csi(tmp_path): np.testing.assert_array_equal(table.where("(id >= 15) & (id < 25)").id[:], expected["id"]) -def test_local_pytables_disk_cache_reuse_and_invalidation(tmp_path): +def test_local_pytables_disk_cache_reuse_and_refresh(tmp_path): native_pytables_url(tmp_path, "local-csi.h5", csi=True) path = tmp_path / "local-csi.h5" cache_dir = tmp_path / "cache" @@ -79,6 +79,7 @@ def test_local_pytables_disk_cache_reuse_and_invalidation(tmp_path): with blosc2.open(path, **options) as table: assert sorted(table.where("id < 10").id[:].tolist()) == list(range(10)) + table["value"][:] generation = table._storage._owner.generation descriptor = table._get_index_catalog()["id"] assert descriptor["persistent"] @@ -95,6 +96,9 @@ def test_local_pytables_disk_cache_reuse_and_invalidation(tmp_path): h5file["table"][0] = row with blosc2.open(path, **options) as table: + assert table._storage._owner.generation == generation + assert table["value"][0] != 999 + table.refresh() assert table._storage._owner.generation != generation assert table["value"][0] == 999 assert sorted(table.where("id < 10").id[:].tolist()) == list(range(10)) diff --git a/tests/test_hdf5_source.py b/tests/test_hdf5_source.py index 41682cbe3..4421c130f 100644 --- a/tests/test_hdf5_source.py +++ b/tests/test_hdf5_source.py @@ -570,6 +570,8 @@ def blocked_import(name, *args, **kwargs): monkeypatch.setattr(builtins, "__import__", blocked_import) index = scan_hdf5_index(str(path)) assert "data" in validate_hdf5_index(index)["datasets"] + with blosc2.open(path, path="data", cache_dir=tmp_path / "cache") as cached: + np.testing.assert_array_equal(cached[:], np.arange(8, dtype="i4")) def test_local_hdf5_explicit_index(tmp_path): @@ -848,13 +850,34 @@ def test_hdf5_auto_detection(tmp_path): assert isinstance(proxy.src, blosc2.HDF5NDSource) assert proxy.dataset == "data" np.testing.assert_array_equal(proxy[:], data) - - # With remote URL, proxy.source also works + # With remote URL, proxy.source also works. mem_url = make_memory_h5("auto_detect_mem.h5", data=(data, (10,))) mem_proxy = blosc2.open(mem_url, lazy=True, dataset="data") assert mem_proxy.source["kind"] == "hdf5" +def test_local_hdf5_disk_cache_dir_and_path(tmp_path): + path = tmp_path / "local-cache.h5" + data = np.arange(32, dtype=np.int32) + with h5py.File(path, "w") as h5file: + h5file.create_dataset("data", data=data, chunks=(8,)) + + with blosc2.open(path, path="data", cache_dir=tmp_path / "cache") as cached: + assert isinstance(cached, blosc2.RemoteArray) + np.testing.assert_array_equal(cached[::2], data[::2]) + carrier_path = cached.cache_path + assert carrier_path is not None + + explicit_path = tmp_path / "explicit.b2nd" + with blosc2.open(path, dataset="data", cache_path=explicit_path) as cached: + np.testing.assert_array_equal(cached[:], data) + assert explicit_path.exists() + + with blosc2.open(path.resolve(), path="data", cache_path=explicit_path) as reopened: + reopened.src.get_chunk = lambda nchunk: (_ for _ in ()).throw(AssertionError("cache miss")) + np.testing.assert_array_equal(reopened[:], data) + + def test_hdf5_without_dataset_opens_store(): url = make_memory_h5("no_ds.h5", data=np.arange(10)) with blosc2.open(url, lazy=True) as store: diff --git a/tests/test_remote_array.py b/tests/test_remote_array.py index 9ba797ea2..cad8e8285 100644 --- a/tests/test_remote_array.py +++ b/tests/test_remote_array.py @@ -8,6 +8,9 @@ from __future__ import annotations import asyncio +import subprocess +import sys +from pathlib import Path import numpy as np import pytest @@ -19,6 +22,101 @@ fsspec = pytest.importorskip("fsspec") +def test_local_b2nd_disk_cache(tmp_path): + source = tmp_path / "source.b2nd" + cache_dir = tmp_path / "cache" + data = np.arange(40, dtype=np.int32) + blosc2.asarray(data, urlpath=source) + + with blosc2.open(source, cache_dir=cache_dir) as cached: + assert isinstance(cached, blosc2.RemoteArray) + np.testing.assert_array_equal(cached[::3], data[::3]) + cache_path = cached.cache_path + assert cache_path is not None + with pytest.raises(ValueError, match="cannot be exported"): + cached.to_cframe() + with pytest.raises(ValueError, match="cannot be exported"): + blosc2.Ref.from_object(cached) + assert Path(cache_path).exists() + with pytest.raises(ValueError, match="source descriptor"): + blosc2.open(cache_path) + + with blosc2.open(source.resolve(), cache_dir=cache_dir) as reopened: + assert reopened.cache_path == cache_path + reopened.src.get_chunk = lambda nchunk: (_ for _ in ()).throw(AssertionError("cache miss")) + np.testing.assert_array_equal(reopened[:], data) + + with blosc2.open(source.as_uri(), cache_dir=cache_dir) as file_url: + assert file_url.cache_path == cache_path + np.testing.assert_array_equal(file_url[:], data) + + subprocess.run( + [ + sys.executable, + "-c", + "import blosc2, sys, numpy as np\n" + "with blosc2.open(sys.argv[1], cache_dir=sys.argv[2]) as array:\n" + " array.src.get_chunk = lambda n: sys.exit('unexpected source read')\n" + " np.testing.assert_array_equal(array[:], np.arange(40, dtype=np.int32))\n", + str(source), + str(cache_dir), + ], + check=True, + ) + + with pytest.raises(ValueError, match="cannot overwrite the local source"): + blosc2.open(source, cache_path=source) + with pytest.raises(NotImplementedError, match="omit lazy=False"): + blosc2.open(source, cache_dir=cache_dir, lazy=False) + with pytest.raises(NotImplementedError, match="assume_immutable=True"): + blosc2.open(source, cache_dir=cache_dir, assume_immutable=False) + for options in ({"mode": "a"}, {"offset": 1}, {"mmap_mode": "r"}, {"shared_cache": True}): + with pytest.raises((ValueError, NotImplementedError)): + blosc2.open(source, cache_dir=cache_dir, **options) + + other = tmp_path / "other.b2nd" + blosc2.asarray(data + 1, urlpath=other) + with pytest.raises(ValueError, match="different specification"): + blosc2.open(other, cache_path=cache_path) + + sparse = tmp_path / "sparse.b2nd" + blosc2.asarray(data, urlpath=sparse, contiguous=False) + with pytest.raises(NotImplementedError, match="contiguous native frame"): + blosc2.open(sparse, cache_dir=cache_dir) + + +@pytest.mark.parametrize("format", ["b2z", "hdf5", "zarr"]) +def test_local_container_cache_selection(tmp_path, format): + source = tmp_path / f"source.{format}" + data = np.arange(24, dtype=np.int32) + if format == "b2z": + with blosc2.TreeStore(source, mode="w", threshold=0) as store: + store["values"] = blosc2.asarray(data) + elif format == "hdf5": + h5py = pytest.importorskip("h5py") + with h5py.File(source, "w") as store: + store.create_dataset("values", data=data, chunks=(8,)) + else: + zarr = pytest.importorskip("zarr") + store = zarr.open_group(source, mode="w") + store.create_array("values", data=data, chunks=(8,)) + + with blosc2.open(source, cache_dir=tmp_path / "group-cache") as group: + assert isinstance(group, blosc2.RemoteStore) + with group["values"] as array: + np.testing.assert_array_equal(array[:], data) + with pytest.raises(NotImplementedError, match="cache_dir"): + blosc2.open(source, cache_path=tmp_path / "group.b2nd") + + for placement in ("cache_dir", "cache_path"): + options = {placement: tmp_path / f"{placement}.b2nd"} + with blosc2.open(source, path="values", **options) as array: + np.testing.assert_array_equal(array[:], data) + with blosc2.open(f"{source}::values", **options) as array: + array.src.get_chunk = lambda nchunk: (_ for _ in ()).throw(AssertionError("cache miss")) + np.testing.assert_array_equal(array[:], data) + + def test_bounded_unbounded_cache_accounting_transition(tmp_path): url, data = _remote_array("accounting-transition.b2nd", nchunks=3, chunk_size=10_000) path = tmp_path / "transition.b2nd" diff --git a/tests/test_zarr_source.py b/tests/test_zarr_source.py index cf363f542..0c8600008 100644 --- a/tests/test_zarr_source.py +++ b/tests/test_zarr_source.py @@ -268,6 +268,23 @@ def test_direct_proxy_zarr_cache_reopens(tmp_path, zarr): np.testing.assert_array_equal(reopened[:3, :4], data[:3, :4]) +def test_local_zarr_open_uses_disk_cache(tmp_path, zarr): + source = tmp_path / "cached.zarr" + data = np.arange(35, dtype=np.int32).reshape(5, 7) + array = zarr.create_array(source, shape=data.shape, chunks=(3, 4), dtype=data.dtype) + array[:] = data + cache_dir = tmp_path / "cache" + + with blosc2.open(source, path="", cache_dir=cache_dir) as cached: + assert isinstance(cached, blosc2.RemoteArray) + np.testing.assert_array_equal(cached[:], data) + assert cached.cache_bytes > 0 + assert list(cache_dir.rglob("active_generation.json")) + + with blosc2.open(source.resolve(), cache_dir=cache_dir) as reopened: + np.testing.assert_array_equal(reopened[:], data) + + def test_zarr_v3_shards_are_decoded_as_logical_chunks(tmp_path, zarr): path = tmp_path / "sharded.zarr" data = np.arange(64, dtype=np.float32).reshape(8, 8) From 1b77cad9e6de8162629b9dd1c377e4336f4d018e Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Thu, 24 Sep 2026 06:38:54 +0200 Subject: [PATCH 76/82] Add refresh support to standalone remote arrays --- doc/guides/remote_arrays.md | 58 +++++++++--------- src/blosc2/remote_array.py | 95 +++++++++++++++++++++++++++++ src/blosc2/schunk.py | 16 ++--- tests/test_remote_array.py | 118 ++++++++++++++++++++++++++++++++++++ tests/test_zarr_source.py | 2 +- 5 files changed, 249 insertions(+), 40 deletions(-) diff --git a/doc/guides/remote_arrays.md b/doc/guides/remote_arrays.md index ef5423a0d..69b4df777 100644 --- a/doc/guides/remote_arrays.md +++ b/doc/guides/remote_arrays.md @@ -81,18 +81,15 @@ same_array = blosc2.open( ) ``` -Local caches assume the source is immutable and are keyed by its absolute path -and selected node. Replacing data at that path violates the assumption and can -mix cached values with uncached reads. For a local `RemoteCTable` or root -`RemoteStore`, call `refresh()` after changing the source; this rebuilds source -metadata, cached payload, and table indexes together. For a standalone -`RemoteArray`, use a fresh cache location or remove its cache before reopening. +Local caches are keyed by the source's absolute path and selected node. If the +source changes at that path, [refresh the cached handle](#handle-source-changes) +before reading it again. Local-source cache files cannot be exported as portable remote references. Local cached opens require `mode="r"` and `assume_immutable=True`; write modes, `lazy=False`, `mmap_mode`, embedded-frame offsets, and `shared_cache=True` are not supported. Native array sources must be contiguous files, not sparse frame -directories. Close other cached handles before refreshing or replacing a cache. +directories. Remote B2Z needs `pip install "blosc2[fsspec]"`. HTTP and HTTPS URLs work out of the box; cloud object stores need their respective protocol driver (such as `s3fs` for S3, `gcsfs` for GCS, or `adlfs` for Azure). @@ -109,7 +106,6 @@ Remote Zarr needs `pip install "blosc2[zarr,fsspec]"`. HTTP/HTTPS works directly; cloud stores require their protocol driver (`s3fs` for S3, etc.). Datasets can be named directly by path (`/sub/arr`), with the `::sub/arr` separator, or via `path="sub/arr"`. For a suffix-free URL, pass `source_format="zarr"`. -Converted Blosc2 chunks are cached under an immutable source contract, so publish changed data at a new URL or replace its cache. Remote HDF5 needs `pip install "blosc2[hdf5,fsspec]"`. HTTP/HTTPS works directly; cloud stores require their protocol driver (`s3fs` for S3, etc.). @@ -167,16 +163,9 @@ dataset through h5py and caches converted Blosc2 chunks in memory. A selected PyTables table returns `RemoteCTable`, so `blosc2.open("readings.h5", path="readings").where("humidity < 10")` uses the same query API as a remote table. Local tables also use h5py. Add `cache_dir="table-cache"` to reuse converted chunks and PyTables index sidecars -across processes. Local caches assume the source is immutable: a changed file at -the same path is not detected automatically. Call `refresh()` on the root -`RemoteCTable`/`RemoteStore`, or clear the array cache before reopening. -Explicit `hdf5_index=` -accepts a native HDF5 index for local files. Legacy HDF5 reference maps are -rejected; omit it to regenerate the native index. - -`RemoteArray` assumes sources are immutable by default, avoiding a metadata request before every read. -For a replaceable `.b2nd` or Caterva2 source, pass `assume_immutable=False` to refresh its identity and invalidate stale cached chunks before each operation. -Mutable B2Z, Zarr, and HDF5 sources are not supported. +across processes. Explicit `hdf5_index=` accepts a native HDF5 index for local +files. Legacy HDF5 reference maps are rejected; omit it to regenerate the native +index. A `URLPath` always means Caterva2. If its `urlbase` is omitted, the server comes from {func}`blosc2.c2context` or `BLOSC_C2URLBASE`. @@ -383,28 +372,35 @@ Prefer direct `C2Array` indexing for sparse, one-off point retrieval; prefer a { Remote CTable access, filtering, buffering, saving, and materialization now live in {doc}`remote_tables`. This heading remains as a pointer for existing links. -## Handle remote changes +## Handle source changes + +Local and remote caches assume the source is immutable by default. After +replacing data at the same path or URL, close other cache handles and refresh: + +- `array.refresh()` for a standalone `RemoteArray` replaces its metadata and + cached chunks, even if its shape changed. +- `table.refresh()` for a standalone `RemoteCTable` also rebuilds its indexes. +- `store.refresh()` for a `RemoteStore` rebuilds the hierarchy. Retrieve child + arrays and tables again afterward; they cannot refresh themselves. -### Standalone arrays and Caterva2 sources +Shared sparse `RemoteArray` caches cannot be refreshed directly; use a new URL +or cache directory. See {doc}`remote_objects` and {doc}`remote_tables` for store +and table details. -A persistent cache records the source identity when one is available. -On a later `blosc2.open()` with the same `cache_dir` or `cache_path`, a mismatched cache is discarded and rebuilt automatically. +For a replaceable `.b2nd` or Caterva2 source, `assume_immutable=False` checks +its identity before each read and invalidates stale chunks automatically. +Other source formats require explicit refresh when data changes at the same +location. -When constructing a proxy directly in append mode, a mismatch is reported instead: +A direct `Proxy` cache reports a source mismatch in append mode: ```python p = blosc2.Proxy(source, urlpath="cache.b2nd", mode="a") # ValueError if cache.b2nd belongs to different remote bytes ``` -Use `mode="w"` to start that cache again. -If a source cannot provide an identity, compatibility is checked only from shape, dtype, chunks, and blocks. -Use a fresh cache when such a source may have changed without changing its geometry. - -For a replaceable `.b2nd` or Caterva2 source, pass `assume_immutable=False` to check for updates and invalidate stale cached chunks before each operation. - -RemoteStore and RemoteCTable refresh behavior is documented in -{doc}`remote_objects` and {doc}`remote_tables`. +Use `mode="w"` to start that cache again. If a source cannot provide an +identity, compatibility is checked only from shape, dtype, chunks, and blocks. ## Fill a Caterva2 array concurrently diff --git a/src/blosc2/remote_array.py b/src/blosc2/remote_array.py index ea7b8ee38..8e888323f 100644 --- a/src/blosc2/remote_array.py +++ b/src/blosc2/remote_array.py @@ -738,6 +738,9 @@ def __init__( self.src.stamp = ("local", urlpath, self._dataset, self._source_format) self._assume_immutable = assume_immutable self._storage_options = storage_options + self._refresh_cache_dir = cache_dir + self._source_blocks = _source_blocks + self._source_cparams = _source_cparams if _shared_cache: _runtime_cache_path = self._carrier_path(cache_dir, None) + ".cache" self._runtime_urlpath = self._runtime_source(urlpath) @@ -864,6 +867,98 @@ def close(self): self._closed = True hdf5_source.close() + def refresh(self) -> None: + """Reload a standalone source and discard its cached payload and metadata. + + Refresh arrays obtained from a RemoteStore through the root store instead. + Ordinary disk caches must not be used concurrently by other processes. + """ + with self._operation_lock: + self._check_open() + if self._store_owner is not None: + raise ValueError("Refresh the root RemoteStore, then retrieve this array again") + if self._shared_runtime_cache: + raise NotImplementedError("Shared sparse array caches cannot be refreshed directly") + if not self.is_cache_mutable: + raise ValueError("Cannot refresh an immutable RemoteArray carrier") + + urlpath = self._runtime_urlpath + if isinstance(urlpath, str) and self._source_format == "zarr" and self._dataset: + parsed = urlsplit(urlpath) + suffix = f"/{self._dataset}" + if parsed.path.endswith(suffix): + urlpath = urlunsplit(parsed._replace(path=parsed.path[: -len(suffix)])) + options = { + "cache_policy": self.cache_policy, + "max_concurrency": self._max_concurrency, + "storage_options": self._storage_options, + "assume_immutable": self._assume_immutable, + "dataset": self._dataset, + "_local_source": self._local_source, + "_source_blocks": self._source_blocks, + "_source_cparams": self._source_cparams, + } + if isinstance(urlpath, str): + options["source_format"] = self._source_format + if self.cache_policy is not blosc2.CachePolicy.NONE: + options["max_cache_bytes"] = self.max_cache_bytes + + cache_path = self.cache_path if self.cache_policy is blosc2.CachePolicy.DISK else None + if self.cache_policy is blosc2.CachePolicy.DISK and cache_path is None: + raise ValueError("Refresh requires a writable disk cache path") + temporary = None + fresh = None + try: + if cache_path is not None: + fd, temporary = tempfile.mkstemp( + prefix=".refresh-", suffix=".b2nd", dir=Path(cache_path).parent + ) + os.close(fd) + os.unlink(temporary) + options["cache_path"] = temporary + fresh = type(self)(urlpath, **options) + if temporary is not None: + self._discard_source_snapshots() + os.replace(temporary, cache_path) + carrier = blosc2.blosc2_ext.open(cache_path, "a", 0, dparams=blosc2.DParams(nthreads=1)) + fresh._carrier = fresh._runtime_cache = carrier + fresh._attach_carrier_cache() + fresh._cache_status = "refreshed" + old_source = self.src + state = fresh.__dict__.copy() + state["_operation_lock"] = self._operation_lock + state["_refresh_lock"] = self._refresh_lock + state["_refresh_cache_dir"] = self._refresh_cache_dir + self.__dict__ = state + if isinstance(old_source, blosc2.HDF5NDSource): + old_source.close() + finally: + if temporary is not None: + Path(temporary).unlink(missing_ok=True) + + def _discard_source_snapshots(self): + if self._refresh_cache_dir is None or self._local_source: + return + if self._source_format in {"hdf5", "b2z"}: + from blosc2.remote_source_cache import source_cache_path + + path = source_cache_path( + self._source["urlpath"], + self._refresh_cache_dir, + self._storage_options, + kind=self._source_format, + ) + path.unlink(missing_ok=True) + path.with_suffix(path.suffix + ".json").unlink(missing_ok=True) + if self._source_format == "hdf5": + index = fsspec_cache_path( + self._source["urlpath"], + self._refresh_cache_dir, + ".hdf5-index.b2", + storage_options=self._storage_options, + ) + Path(index).unlink(missing_ok=True) + def _runtime_source(self, original): """Keep credentials in live process state, outside the descriptor.""" if isinstance(self.src, blosc2.C2Array): diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index ed9ece47f..bdd47ec7a 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2277,13 +2277,13 @@ def _open_lazy_remote(urlpath, source_format, options, shared_cache=False): if source_format == "hdf5": return _open_remote_hdf5(urlpath, options) if source_format == "zarr" and options.get("_local_source"): - if options["cache_path"] is not None: - try: - return blosc2.RemoteArray(urlpath, **options) - except ValueError as exc: - if "is a Zarr group" in str(exc): - raise NotImplementedError("Zarr groups use cache_dir, not cache_path") from exc + try: + return blosc2.RemoteArray(urlpath, **options) + except ValueError as exc: + if "is a Zarr group" not in str(exc): raise + if options["cache_path"] is not None: + raise NotImplementedError("Zarr groups use cache_dir, not cache_path") from exc store_options = { key: value for key, value in options.items() @@ -2792,8 +2792,8 @@ def open( :class:`RemoteStore` cache. For remote inputs it stores fetched chunks and metadata; for local Blosc2, HDF5, and Zarr inputs it stores accessed chunks and converted data. Supplying it selects on-demand, read-only access for local inputs. Local caches assume - the source is immutable: replacing it at the same path requires clearing/rebuilding the - cache. No cache is created unless you name one. + the source is immutable: call ``refresh()`` on the cached handle after replacing it at + the same path. No cache is created unless you name one. cache_path: str | pathlib.Path, optional Exact file for a persistent array cache (:attr:`CachePolicy.DISK`), for local or remote standalone arrays and array leaves. Tables and groups require ``cache_dir``. Mutually diff --git a/tests/test_remote_array.py b/tests/test_remote_array.py index cad8e8285..6c3f70bbb 100644 --- a/tests/test_remote_array.py +++ b/tests/test_remote_array.py @@ -8,6 +8,8 @@ from __future__ import annotations import asyncio +import os +import shutil import subprocess import sys from pathlib import Path @@ -85,6 +87,122 @@ def test_local_b2nd_disk_cache(tmp_path): blosc2.open(sparse, cache_dir=cache_dir) +@pytest.mark.parametrize("format", ["b2nd", "b2z", "h5", "zarr"]) +@pytest.mark.parametrize("placement", ["cache_dir", "cache_path"]) +def test_local_array_refresh(tmp_path, format, placement, monkeypatch): + source = tmp_path / f"source.{format}" + replacement = tmp_path / f"replacement.{format}" + dataset = "values" if format in {"b2z", "h5", "zarr"} else None + + def write(path, data): + if format == "b2nd": + blosc2.asarray(data, urlpath=path) + elif format == "b2z": + with blosc2.TreeStore(path, mode="w", threshold=0) as store: + store["values"] = blosc2.asarray(data) + elif format == "h5": + h5py = pytest.importorskip("h5py") + with h5py.File(path, "w") as store: + store.create_dataset("values", data=data, chunks=(4,)) + else: + zarr = pytest.importorskip("zarr") + store = zarr.open_group(path, mode="w") + store.create_array("values", data=data, chunks=(4,)) + + old = np.arange(8, dtype="i4") + new = np.arange(12, dtype="i4") + 100 + write(source, old) + options = {placement: tmp_path / ("cache" if placement == "cache_dir" else "cache.b2nd")} + with blosc2.open(source, path=dataset, **options) as array: + np.testing.assert_array_equal(array[:], old) + carrier = array.cache_path + write(replacement, new) + if source.is_dir(): + shutil.rmtree(source) + os.replace(replacement, source) + np.testing.assert_array_equal(array[:], old) + + with monkeypatch.context() as patch: + patch.setattr( + blosc2.RemoteArray, "_open_source", lambda *a, **k: (_ for _ in ()).throw(OSError("offline")) + ) + with pytest.raises(OSError, match="offline"): + array.refresh() + np.testing.assert_array_equal(array[:], old) + assert array.cache_path == carrier + + array.refresh() + assert array.cache_status == "refreshed" + assert array.cache_path == carrier + np.testing.assert_array_equal(array[:], new) + with blosc2.open(source, path=dataset, **options) as reopened: + reopened.src.get_chunk = lambda n: (_ for _ in ()).throw(AssertionError("cache miss")) + np.testing.assert_array_equal(reopened[:], new) + + +def test_remote_array_refresh_policies_and_store_owner(tmp_path): + url, old = _remote_array("refresh-array.b2nd", nchunks=1, chunk_size=8) + fs = fsspec.filesystem("memory") + new = np.arange(12, dtype="u1") + for policy in (blosc2.CachePolicy.NONE, blosc2.CachePolicy.MEMORY, blosc2.CachePolicy.DISK): + options = {"cache_policy": policy} + if policy is blosc2.CachePolicy.DISK: + options["cache_dir"] = tmp_path / "remote-cache" + array = blosc2.RemoteArray(url, **options) + np.testing.assert_array_equal(array[:], old) + fs.pipe_file("refresh-array.b2nd", blosc2.asarray(new).to_cframe()) + array.refresh() + np.testing.assert_array_equal(array[:], new) + fs.pipe_file("refresh-array.b2nd", blosc2.asarray(old).to_cframe()) + + source = tmp_path / "source.b2z" + with blosc2.TreeStore(source, mode="w", threshold=0) as tree: + tree["values"] = blosc2.asarray(new) + with blosc2.open(source, cache_dir=tmp_path / "store-cache") as store: + with store["values"] as array: + with pytest.raises(ValueError, match="root RemoteStore"): + array.refresh() + + shared = blosc2.RemoteArray.with_sparse_cache(url, tmp_path / "shared-cache") + with pytest.raises(NotImplementedError, match="Shared sparse"): + shared.refresh() + + readonly = blosc2.from_cframe(shared.to_cframe(mutable=False)) + with pytest.raises(ValueError, match="immutable RemoteArray"): + readonly.refresh() + + +@pytest.mark.parametrize("format", ["b2z", "h5"]) +def test_remote_container_array_refresh_discards_source_snapshot(tmp_path, format): + source = tmp_path / f"source.{format}" + url = f"memory://refresh-container.{format}" + fs = fsspec.filesystem("memory") + + def upload(data): + if format == "b2z": + with blosc2.TreeStore(source, mode="w", threshold=0) as store: + store["values"] = blosc2.asarray(data) + else: + h5py = pytest.importorskip("h5py") + with h5py.File(source, "w") as store: + store.create_dataset("values", data=data, chunks=(4,)) + fs.pipe_file(f"refresh-container.{format}", source.read_bytes()) + + old = np.arange(8, dtype="i4") + new = np.arange(12, dtype="i4") + 100 + upload(old) + cache_dir = tmp_path / "cache" + with blosc2.open(url, path="values", cache_dir=cache_dir) as array: + np.testing.assert_array_equal(array[:], old) + upload(new) + np.testing.assert_array_equal(array[:], old) + array.refresh() + np.testing.assert_array_equal(array[:], new) + with blosc2.open(url, path="values", cache_dir=cache_dir) as reopened: + reopened.src.get_chunk = lambda n: (_ for _ in ()).throw(AssertionError("cache miss")) + np.testing.assert_array_equal(reopened[:], new) + + @pytest.mark.parametrize("format", ["b2z", "hdf5", "zarr"]) def test_local_container_cache_selection(tmp_path, format): source = tmp_path / f"source.{format}" diff --git a/tests/test_zarr_source.py b/tests/test_zarr_source.py index 0c8600008..4d23ef975 100644 --- a/tests/test_zarr_source.py +++ b/tests/test_zarr_source.py @@ -279,7 +279,7 @@ def test_local_zarr_open_uses_disk_cache(tmp_path, zarr): assert isinstance(cached, blosc2.RemoteArray) np.testing.assert_array_equal(cached[:], data) assert cached.cache_bytes > 0 - assert list(cache_dir.rglob("active_generation.json")) + assert cached.cache_path is not None with blosc2.open(source.resolve(), cache_dir=cache_dir) as reopened: np.testing.assert_array_equal(reopened[:], data) From 5f3f7fe4af371421d19289f87e0bb08d6759b77e Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Thu, 24 Sep 2026 06:53:57 +0200 Subject: [PATCH 77/82] Reuse cached remote CTable column prefixes --- src/blosc2/ctable_remote_read.py | 10 +++++++++- tests/ctable/test_remote_ctable.py | 31 ++++++++++++++++++++++++++++++ 2 files changed, 40 insertions(+), 1 deletion(-) diff --git a/src/blosc2/ctable_remote_read.py b/src/blosc2/ctable_remote_read.py index 9ad757894..493366436 100644 --- a/src/blosc2/ctable_remote_read.py +++ b/src/blosc2/ctable_remote_read.py @@ -159,7 +159,15 @@ def ranges_for(name): return ranges def reader(offset, size): - return (yield archive.read_transport, (offset, size), size) + for start, data in archive.metadata["ranges"]: + if start <= offset and offset + size <= start + len(data): + return data[offset - start : offset - start + size] + data = yield archive.read_transport, (offset, size), size + # Preserve prefetched prefixes on the owner thread: buffered reads + # during column opening otherwise bypass metadata capture. + if archive.persist_metadata: + archive.metadata["ranges"].append((offset, data)) + return data def consume(batch, ranges): prefixes, _ = run_reads( diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 7d0bacad6..24abfca1a 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -698,6 +698,37 @@ def unexpected_read(*args, **kwargs): assert table.traffic.requests > 0 +@pytest.mark.parametrize("max_concurrency", [1, 8]) +@pytest.mark.usefixtures("b2z_range_reads") +def test_disk_cache_reuses_batch_column_prefixes(tmp_path, monkeypatch, max_concurrency): + @dataclasses.dataclass + class Mixed: + x: int + message: str = blosc2.field(blosc2.vlstring(batch_rows=32)) + tags: list[int] = blosc2.field(blosc2.list(blosc2.int64(), batch_rows=32)) # noqa: RUF009 + region: str = blosc2.field(blosc2.dictionary()) + + local = blosc2.CTable( + Mixed, + [(i, f"message {i}", [i, i + 1], "east" if i % 2 else "west") for i in range(2000)], + cparams={"clevel": 0}, + create_summary_index=False, + ) + url = remote_table_url(tmp_path, local) + options = {"cache_dir": tmp_path / "cache", "max_concurrency": max_concurrency} + with blosc2.open(url, **options) as table: + expected = list(table.where("x < 3")) + + def unexpected_read(*args, **kwargs): + pytest.fail("Warm batch column metadata must not download archive bytes") + + monkeypatch.setattr(type(fsspec.filesystem("memory")), "cat_file", unexpected_read) + monkeypatch.setattr(type(fsspec.filesystem("memory")), "info", unexpected_read) + with blosc2.open(url, **options) as table: + assert list(table.where("x < 3")) == expected + assert table.traffic.requests == 0 + + @pytest.mark.parametrize( "policy", [blosc2.CachePolicy.NONE, blosc2.CachePolicy.MEMORY, blosc2.CachePolicy.DISK] ) From 400b50dad9bea87baaec9de92d7e89c0dd3d4087 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Thu, 24 Sep 2026 07:17:12 +0200 Subject: [PATCH 78/82] Fix remote object round trips and API consistency --- doc/guides/remote_tables.md | 4 +- doc/reference/remoteobject.rst | 4 +- src/blosc2/ctable_indexing.py | 29 +++++++++-- src/blosc2/ctable_storage.py | 2 +- src/blosc2/dict_store.py | 7 ++- src/blosc2/hdf5_source.py | 15 +++++- src/blosc2/list_array.py | 9 +++- src/blosc2/remote_batch.py | 34 +++++++++++-- src/blosc2/remote_ctable.py | 2 +- src/blosc2/remote_store.py | 76 ++++++++++++++++++++++++---- src/blosc2/schunk.py | 30 ++++++----- src/blosc2/store_materialize.py | 16 +++--- tests/ctable/test_remote_ctable.py | 80 ++++++++++++++++++++++++++++++ tests/test_dataset_path.py | 24 +++++++++ tests/test_list_array.py | 16 ++++++ tests/test_remote_store.py | 66 ++++++++++++++++++++++++ 16 files changed, 369 insertions(+), 45 deletions(-) diff --git a/doc/guides/remote_tables.md b/doc/guides/remote_tables.md index 2918cab78..2b1f8a3df 100644 --- a/doc/guides/remote_tables.md +++ b/doc/guides/remote_tables.md @@ -32,8 +32,8 @@ by the Blosc2 writers; the arrays inside remain Blosc2-compressed. The backing arrays share the table's cache budget and traffic counters. MEMORY, DISK (with `cache_dir`) and NONE policies are supported. Size reporting uses source metadata without scanning strings. Small-member metadata prefetch may also fetch -some payload. Repeated reads can reuse cached blocks; filtering scans the required -columns because persisted indexes are not used remotely. +some payload. Repeated reads can reuse cached blocks; filtering uses supported +persisted indexes when available and scans the required columns otherwise. Batch-backed columns transfer one whole compressed batch per required batch, then decode it locally. A small row selection can therefore fetch and allocate a large diff --git a/doc/reference/remoteobject.rst b/doc/reference/remoteobject.rst index 8dba8c333..008b95338 100644 --- a/doc/reference/remoteobject.rst +++ b/doc/reference/remoteobject.rst @@ -23,8 +23,8 @@ and tables report their shared owner's retained payload. ``RemoteObject`` is not a factory, storage backend, serialization format, or remote-write API. Data-specific operations remain on the concrete classes. -Arrays and tables provide ``materialize()``; stores are navigated to a leaf that -can be materialized. +Arrays and tables provide ``materialize()``; stores provide recursive +``materialize()`` to expand their contents and mounted stores into a local tree. See :doc:`Working with Remote Data <../guides/remote_objects>` for the shared cache, traffic, reference-saving, and lifetime behavior. diff --git a/src/blosc2/ctable_indexing.py b/src/blosc2/ctable_indexing.py index 1a81ada91..85dc407c9 100644 --- a/src/blosc2/ctable_indexing.py +++ b/src/blosc2/ctable_indexing.py @@ -466,6 +466,8 @@ def _index_create_kwargs_from_descriptor(self, descriptor: dict) -> dict[str, An kwargs["method"] = descriptor.get("full", {}).get("build_method", "global-sort") if descriptor.get("kind") == "opsi": kwargs["opsi_max_cycles"] = descriptor.get("opsi", {}).get("max_cycles") + if descriptor.get("kind") == "summary" and len(descriptor.get("levels", {})) == 1: + kwargs["granularity"] = next(iter(descriptor["levels"])) target = descriptor.get("target") or {} if target.get("source") == "expression": kwargs["expression"] = target.get("expression") @@ -678,7 +680,12 @@ def _index_target_array(self, lookup_key: str, descriptor: dict) -> blosc2.NDArr path = descriptor.get("expr_values_path") if path is None: raise KeyError(f"No backing array found for expression index {token!r}.") - arr = blosc2.open(path, mode="r" if root._read_only else "a") + if root._read_only: + from blosc2.indexing import _open_sidecar_file + + arr = _open_sidecar_file(path) + else: + arr = blosc2.open(path, mode="a") root._expr_index_arrays[token] = arr return arr @@ -1415,7 +1422,14 @@ def generic_visit(self, node): def _try_expression_index_where(self, expr_result: blosc2.LazyExpr, catalog: dict) -> np.ndarray | None: """Attempt to resolve *expr_result* via a direct table expression index.""" - from blosc2.indexing import evaluate_bucket_query, evaluate_segment_query, plan_query + from blosc2.indexing import ( + _clear_cached_data, + _load_store, + _register_descriptor_owner, + evaluate_bucket_query, + evaluate_segment_query, + plan_query, + ) expression = expr_result.expression operands = dict(expr_result.operands) @@ -1427,9 +1441,18 @@ def _try_expression_index_where(self, expr_result: blosc2.LazyExpr, catalog: dic if rewritten is None: continue expr_arr = self._index_target_array(lookup_key, descriptor) + # The table catalog rebases paths after moving or packing a store; + # the expression array's own metadata still names its original files. + store = _load_store(expr_arr) + if store["indexes"].get(lookup_key) is not descriptor: + _clear_cached_data(expr_arr, lookup_key) + store["indexes"][lookup_key] = descriptor + _register_descriptor_owner(expr_arr, lookup_key) where_dict = {"_where_x": expr_arr} merged_operands = {"_where_x": expr_arr} - plan = plan_query(rewritten, merged_operands, where_dict) + plan = plan_query( + rewritten, merged_operands, where_dict, array_to_col={id(expr_arr): lookup_key} + ) if not plan.usable: continue if plan.exact_positions is not None: diff --git a/src/blosc2/ctable_storage.py b/src/blosc2/ctable_storage.py index 66bfd711a..2fb2229fc 100644 --- a/src/blosc2/ctable_storage.py +++ b/src/blosc2/ctable_storage.py @@ -709,7 +709,7 @@ def __getitem__(self, key): if not -self.shape[0] <= key < self.shape[0]: raise IndexError("row index out of range") return np.bool_(True) - return np.ones(self.shape[0], dtype=bool)[key] + return np.broadcast_to(np.bool_(True), self.shape)[key] def _take_numpy(self, indices, /, *, axis=None): if axis not in (None, 0, -1): diff --git a/src/blosc2/dict_store.py b/src/blosc2/dict_store.py index 7f5721ce5..5edf73516 100644 --- a/src/blosc2/dict_store.py +++ b/src/blosc2/dict_store.py @@ -673,7 +673,12 @@ def __setitem__( source_path = ( value.cache_path if isinstance(value, blosc2.RemoteArray) else value.urlpath ) - shutil.copy2(source_path, tmp_path) + if zipfile.is_zipfile(source_path): + # A member's urlpath names the archive, not its own frame. + with open(tmp_path, "wb") as file: + file.write(value.to_cframe()) + else: + shutil.copy2(source_path, tmp_path) os.replace(tmp_path, dest_path) # Store relative path from tree directory diff --git a/src/blosc2/hdf5_source.py b/src/blosc2/hdf5_source.py index 054f38b86..f78dd83c4 100644 --- a/src/blosc2/hdf5_source.py +++ b/src/blosc2/hdf5_source.py @@ -43,6 +43,16 @@ _HDF5_INDEX_VERSIONS = {1, HDF5_INDEX_VERSION} +class HDF5GroupError(ValueError): + """An array lookup found a group; retain discovery for container dispatch.""" + + def __init__(self, message, source): + super().__init__(message) + self.index = source._hdf5_index + self.blob = source._blob + self.traffic = source.traffic + + def hdf5_source_cache_path(urlpath, cache_dir, storage_options=None): return source_cache_path(urlpath, cache_dir, storage_options, kind="hdf5") @@ -1132,8 +1142,9 @@ def _load_or_scan_index(self, hdf5_index): def _validate_dataset_presence(self, raw_dataset): if self.dataset in self._hdf5_index["groups"]: - raise ValueError( - f"{raw_dataset!r} is an HDF5 group; pass the path of a dataset. Available datasets: {available_datasets(self._hdf5_index)}" + raise HDF5GroupError( + f"{raw_dataset!r} is an HDF5 group; pass the path of a dataset. Available datasets: {available_datasets(self._hdf5_index)}", + self, ) if self.dataset not in self._hdf5_index["datasets"]: raise ValueError( diff --git a/src/blosc2/list_array.py b/src/blosc2/list_array.py index 05bac6bd9..2d089c8ad 100644 --- a/src/blosc2/list_array.py +++ b/src/blosc2/list_array.py @@ -366,7 +366,11 @@ def _from_batch_backend(cls, spec, backend): if spec.storage != "batch": raise NotImplementedError("Remote ListArray storage='vl' is not supported") stored = backend.meta.get("listarray") - if stored != spec.to_listarray_metadata(): + if ( + not isinstance(stored, dict) + or stored.get("version") != 1 + or ListSpec.from_metadata_dict(stored).to_listarray_metadata() != spec.to_listarray_metadata() + ): raise ValueError("Remote ListArray metadata does not match its table schema") obj = object.__new__(cls) obj.spec = spec @@ -524,11 +528,12 @@ def extend_arrow(self, arrow_array) -> None: # Persist pending rows first: chunks are appended straight to the # backend, which would otherwise reorder them ahead of pending cells. self.flush() + item = pa.field("item", self._arrow_item_type(), nullable=self.spec.item_spec.nullable) for chunk in chunks: step = self.batch_rows or len(chunk) or 1 for start in range(0, len(chunk), step): part = chunk.slice(start, step) - typed = self._typed_batch(part.to_pylist()) + typed = part.cast(pa.list_(item)) self._backend.append(typed) self._persisted_row_count += len(part) self._invalidate_batch_caches() diff --git a/src/blosc2/remote_batch.py b/src/blosc2/remote_batch.py index 6a3087087..61fc07b96 100644 --- a/src/blosc2/remote_batch.py +++ b/src/blosc2/remote_batch.py @@ -138,21 +138,30 @@ def cratio(self): class _RemoteBatchCache: """Compressed batch retention using the owner's aggregate cache budget.""" - def __init__(self, source, key, coordinator, path=None): + def __init__(self, source, key, coordinator, path=None, *, read_only=False, artifact=None): self._source = source self._cache_key = key self._cache_coordinator = coordinator self._path = None if path is None else Path(path) + self._read_only = read_only + self._artifact = artifact self._memory = {} self._cache_sizes = {} self._cache_lru = OrderedDict() if self._path is not None: - self._path.mkdir(parents=True, exist_ok=True) + if not read_only: + self._path.mkdir(parents=True, exist_ok=True) for file in sorted(self._path.glob("*.chunk"), key=lambda item: int(item.stem)): index = int(file.stem) - if index < len(source.offsets): + if 0 <= index < len(source.offsets): self._cache_sizes[index] = file.stat().st_size self._cache_lru[index] = None + if artifact is not None: + for index, info in artifact[1].items(): + if not 0 <= index < len(source.offsets): + raise ValueError("Invalid cached batch index") + self._cache_sizes[index] = info["length"] + self._cache_lru[index] = None coordinator.register(self) def __getattr__(self, name): @@ -161,12 +170,27 @@ def __getattr__(self, name): def _file(self, index): return self._path / f"{index}.chunk" + def read_cached_chunk(self, index): + """Read retained bytes without fetching or changing cache recency.""" + if self._artifact is not None: + path, offsets = self._artifact + info = offsets[index] + with open(path, "rb") as file: + file.seek(info["offset"]) + data = file.read(info["length"]) + if len(data) != info["length"]: + raise ValueError("Truncated cached batch") + return data + return self._file(index).read_bytes() if self._path is not None else self._memory[index] + def get_chunk(self, index): self._source._check() if index in self._cache_sizes: - chunk = self._file(index).read_bytes() if self._path is not None else self._memory[index] + chunk = self.read_cached_chunk(index) else: chunk = self._source.get_chunk(index) + if self._read_only: + return chunk if self._path is None: self._memory[index] = chunk else: @@ -187,6 +211,8 @@ def _retained_cache_bytes(self): return sum(self._cache_sizes.values()) def _trim_cache(self, target_bytes, *, max_chunks=None): + if self._read_only and self._retained_cache_bytes() > target_bytes: + raise ValueError("Cannot trim an immutable batch cache") removed = [] while self._retained_cache_bytes() > target_bytes and self._cache_lru: if max_chunks is not None and len(removed) >= max_chunks: diff --git a/src/blosc2/remote_ctable.py b/src/blosc2/remote_ctable.py index defcb9c62..6243c88c1 100644 --- a/src/blosc2/remote_ctable.py +++ b/src/blosc2/remote_ctable.py @@ -286,7 +286,7 @@ def attrs(self): def source(self): storage = self._remote_storage() return { - "kind": "b2z", + "kind": storage._owner.format, "version": 1, "urlpath": storage._owner.urlpath, "dataset": storage._root_key, diff --git a/src/blosc2/remote_store.py b/src/blosc2/remote_store.py index a25a98f1d..44e388a69 100644 --- a/src/blosc2/remote_store.py +++ b/src/blosc2/remote_store.py @@ -12,7 +12,7 @@ import weakref import zipfile from dataclasses import dataclass -from pathlib import PurePosixPath +from pathlib import Path, PurePosixPath from urllib.parse import urlsplit, urlunsplit import blosc2 @@ -309,6 +309,7 @@ def save_manifest(self): "notice": self.notice, "metadata": self.metadata, "caches": sorted(self.caches), + "batch_caches": sorted(self.batch_caches), "cache_policy": getattr(self, "cache_policy", blosc2.CachePolicy.DISK).value, "max_cache_bytes": getattr(self, "max_cache_bytes", None), "mutable": getattr(self, "mutable", False), @@ -807,6 +808,8 @@ def open_ctable_carrier(self, table_path, logical_key): def open_ctable_batch(self, full): """Open one external BatchArray member hidden below a CTable node.""" + if full in self.batch_caches: + return self.batch_caches[full] if self.format != "b2z": raise NotImplementedError("Remote CTable access currently requires a B2Z source") self._validate(full) @@ -828,13 +831,29 @@ def check_open(): self.archive._opening_ranges.clear() if self.cache_policy is blosc2.CachePolicy.NONE: return source - if full not in self.batch_caches: - from blosc2.remote_batch import _RemoteBatchCache + from blosc2.remote_batch import _RemoteBatchCache - path = None - if self.disk is not None: - path = self.disk.batch_payload_path(self.generation, full) - self.batch_caches[full] = _RemoteBatchCache(source, full, self.cache_coordinator, path) + path = None + artifact = None + if self.disk is not None: + path = self.disk.batch_payload_path(self.generation, full) + elif not self.is_mutable: + if self.artifact_offsets is not None: + prefix = f"{full}.b2b.cache/" + offsets = {} + for name, info in self.artifact_offsets.items(): + if name.startswith(prefix): + suffix = name[len(prefix) :] + if not suffix.endswith(".chunk") or not suffix[:-6].isdigit(): + raise ValueError("Invalid cached batch member") + offsets[int(suffix[:-6])] = info + artifact = self.artifact_path, offsets + elif self.artifact_path is not None: + path = os.path.join(self.artifact_path, f"{full}.b2b.cache") + key = f"{self.cache_namespace}:{full}" if self.cache_namespace else full + self.batch_caches[full] = _RemoteBatchCache( + source, key, self.cache_coordinator, path, read_only=not self.is_mutable, artifact=artifact + ) return self.batch_caches[full] def load_ctable_attrs(self, table_path): @@ -969,6 +988,8 @@ def restore_caches(self, manifest): relative = path[len(self.root) + 1 :] if self.root else path self.resolve(relative) self.get_cache(self.open_source(relative)) + for path in manifest.get("batch_caches", []): + self.open_ctable_batch(path) self.cache_coordinator.enforce() self.restoring = False @@ -1204,6 +1225,8 @@ def save_selection( self._copy_leaf_carrier(orig_key, proxy, staging_dir) exported_caches.append(orig_key) + exported_batches = self._export_batch_caches(full_path, staging_dir) if include_cache else [] + linked_exports = self._export_linked_stores(full_path, staging_dir) if include_cache else {} exported_manifest = { @@ -1216,6 +1239,7 @@ def save_selection( "notice": self.notice, "metadata": metadata, "caches": sorted(exported_caches), + "batch_caches": sorted(exported_batches), "cache_policy": self.cache_policy.value, "max_cache_bytes": self.max_cache_bytes, "mutable": effective_mutable, @@ -1251,6 +1275,20 @@ def save_selection( os.unlink(tmp_zip) shutil.rmtree(staging_dir, ignore_errors=True) + def _export_batch_caches(self, full_path, staging_dir): + exported = [] + for key, cache in self.batch_caches.items(): + if full_path and not key.startswith(full_path + "/"): + continue + if not cache._cache_sizes: + continue + folder = Path(staging_dir) / f"{key}.b2b.cache" + folder.mkdir(parents=True, exist_ok=True) + for index in cache._cache_sizes: + (folder / f"{index}.chunk").write_bytes(cache.read_cached_chunk(index)) + exported.append(key) + return exported + def _export_linked_stores(self, full_path, staging_dir): exports = {} for mount, store in self.linked_stores.items(): @@ -1601,7 +1639,7 @@ def _from_reference(cls, descriptor, **runtime): obj._reference_parent_finalizer = weakref.finalize(obj, parent.release) return obj - def _ensure_open(self): + def _ensure_open(self): # noqa: C901 if getattr(self, "_deferred_closed", False): raise RuntimeError("RemoteStore handle is closed") parent = getattr(self, "_reference_parent", None) @@ -1695,6 +1733,10 @@ def _ensure_open(self): owner.max_cache_bytes = parent.max_cache_bytes owner.cache_coordinator = parent.cache_coordinator owner.cache_namespace = namespace + for key, cache in (*owner.caches.items(), *owner.batch_caches.items()): + cache._cache_coordinator = parent.cache_coordinator + cache._cache_key = f"{namespace}:{key}" + parent.cache_coordinator.register(cache) owner.nested_storage_options = parent.nested_storage_options if mount is not None and mount not in parent.linked_stores: anchor = object.__new__(type(self)) @@ -2245,6 +2287,20 @@ def _validate_artifact_manifest(manifest): # noqa: C901 or (source.get("kind") != "hdf5" and root and not path.startswith(root + "/")) ): raise ValueError("Invalid cached RemoteStore leaf") + batches = manifest.get("batch_caches", []) + if not isinstance(batches, list): + raise ValueError("Invalid RemoteStore batch caches") + for path in batches: + RemoteDiscovery._validate(path) + if ( + source.get("kind") != "b2z" + or (root and not path.startswith(root + "/")) + or not any( + kind == "ctable" and (not table or path.startswith(table + "/")) + for table, (kind, _) in nodes.items() + ) + ): + raise ValueError("Invalid cached RemoteStore batch") linked = manifest.get("linked", {}) if not isinstance(linked, dict): raise ValueError("Invalid nested RemoteStore artifacts") @@ -2390,7 +2446,9 @@ def _open_artifact(cls, urlpath, mode="r", **kwargs): if cache_policy is not None: if not isinstance(cache_policy, blosc2.CachePolicy): raise TypeError("cache_policy must be a blosc2.CachePolicy instance") - if cache_policy is blosc2.CachePolicy.NONE and manifest.get("caches"): + if cache_policy is blosc2.CachePolicy.NONE and ( + manifest.get("caches") or manifest.get("batch_caches") + ): raise ValueError( "Cannot reopen a warm RemoteStore artifact with CachePolicy.NONE; " "use a cold export or a cache policy that permits retained payload." diff --git a/src/blosc2/schunk.py b/src/blosc2/schunk.py index bdd47ec7a..27a087368 100644 --- a/src/blosc2/schunk.py +++ b/src/blosc2/schunk.py @@ -2478,7 +2478,7 @@ def _open_local_hdf5(urlpath, options): """Use table discovery only for PyTables nodes; keep ordinary datasets on h5py.""" import h5py - dataset = options.get("dataset") + dataset = (options.get("dataset") or "").strip("/") is_table = False is_group = False if dataset: @@ -2486,7 +2486,7 @@ def _open_local_hdf5(urlpath, options): node = h5file.get(dataset.strip("/")) is_group = isinstance(node, h5py.Group) is_table = isinstance(node, h5py.Dataset) and node.attrs.get("CLASS") in {"TABLE", b"TABLE"} - if is_table or is_group or dataset is None: + if is_table or is_group or not dataset: if not is_table and options["max_concurrency"] is not None: raise NotImplementedError("max_concurrency is only supported for remote arrays and tables") if options["cache_path"] is not None: @@ -2521,11 +2521,13 @@ def _open_local_hdf5(urlpath, options): def _open_remote_hdf5(urlpath, options): """Discover HDF5 groups and PyTables tables while retaining array-only options.""" + from blosc2.hdf5_source import HDF5GroupError + if not is_fsspec_url(urlpath): return _open_local_hdf5(urlpath, options) if options["cache_path"] is not None or options["assume_immutable"] is not True: return blosc2.RemoteArray(urlpath, **options) - dataset = options.get("dataset") + dataset = (options.get("dataset") or "").strip("/") hdf5_index = options.get("hdf5_index") hdf5_blob = None traffic = None @@ -2538,15 +2540,19 @@ def _open_remote_hdf5(urlpath, options): StoreDiskCache.path_for(options["cache_dir"], source) / "active_generation.json" ).exists() if dataset and not cached_store: - array = blosc2.RemoteArray(urlpath, _defer_cache=True, **options) - metadata = array.src._hdf5_index["datasets"][array.dataset] - if metadata.get("kind") != "ctable": - array._complete_deferred_cache() - return array - hdf5_index = array.src._hdf5_index - hdf5_blob = array.src._blob - traffic = array.traffic - array.close() + try: + array = blosc2.RemoteArray(urlpath, _defer_cache=True, **options) + except HDF5GroupError as exc: + hdf5_index, hdf5_blob, traffic = exc.index, exc.blob, exc.traffic + else: + metadata = array.src._hdf5_index["datasets"][array.dataset] + if metadata.get("kind") != "ctable": + array._complete_deferred_cache() + return array + hdf5_index = array.src._hdf5_index + hdf5_blob = array.src._blob + traffic = array.traffic + array.close() store_options = { key: value for key, value in options.items() diff --git a/src/blosc2/store_materialize.py b/src/blosc2/store_materialize.py index ebbee38df..8d3cd962a 100644 --- a/src/blosc2/store_materialize.py +++ b/src/blosc2/store_materialize.py @@ -131,6 +131,7 @@ def _copy_array(source, target, path, staging): chunks=source.chunks, blocks=source.blocks, cparams=source.cparams, + meta={key: value for key, value in source.meta.items() if key != "b2nd"}, urlpath=local_path, mode="w", ) @@ -142,19 +143,22 @@ def _copy_array(source, target, path, staging): for start in range(0, source.shape[0], step): item = (slice(start, min(start + step, source.shape[0])), *tail) local[item] = source[item] + for key, value in source.attrs.items(): + local.schunk.vlmeta[key] = value target[path] = local def _copy_table(table, target, path): - indexes = {name: descriptor["kind"] for name, descriptor in table._get_index_catalog().items()} + indexes = dict(table._get_index_catalog()) local = table.copy(compact=True) local._source_bound = False local._source_columns = set() target[path] = local local.close() materialized = target[path] - for name, kind in indexes.items(): - materialized.create_index(name, kind=kind) - if indexes and not materialized._get_index_catalog(): - raise RuntimeError(f"Failed to rebuild CTable indexes at {path!r}") - materialized.close() + try: + for name, descriptor in indexes.items(): + options = table._index_create_kwargs_from_descriptor(descriptor) + materialized.create_index(None if "expression" in options else name, **options) + finally: + materialized.close() diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 24abfca1a..0a07c7a94 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -2076,3 +2076,83 @@ class Timed: assert list(view) == list(local[["text"]]) assert "_cols/time" not in table._remote_storage()._owner.sources assert "_cols/vec" not in table._remote_storage()._owner.sources + + +@pytest.mark.parametrize("policy", [blosc2.CachePolicy.MEMORY, blosc2.CachePolicy.DISK]) +@pytest.mark.parametrize("mutable", [False, True]) +def test_batch_reference_preserves_warm_payload(tmp_path, policy, mutable): + @dataclasses.dataclass + class Payload: + value: bytes = blosc2.field(blosc2.vlbytes(batch_rows=1)) + + values = [np.random.default_rng(i).bytes(128_000) for i in range(2)] + local = blosc2.CTable(Payload, [(value,) for value in values], create_summary_index=False) + url = remote_table_url(tmp_path, local, "warm-batches") + options = {"cache_dir": tmp_path / "cache"} if policy is blosc2.CachePolicy.DISK else {} + artifact = tmp_path / "warm.b2z" + cold = tmp_path / "cold.b2z" + with blosc2.RemoteCTable(url, cache_policy=policy, **options) as table: + assert table["value"][0] == values[0] + retained = table.cache_bytes + table.traffic.reset() + table.save(artifact, mutable=mutable) + table.save(cold, include_cache=False) + assert table.traffic.requests == 0 + with blosc2.open(cold) as table: + assert table.cache_bytes == 0 + with pytest.raises(ValueError, match="warm RemoteStore"): + blosc2.open(artifact, cache_policy=blosc2.CachePolicy.NONE) + if not mutable: + with pytest.raises(ValueError, match="smaller than retained"): + blosc2.open(artifact, max_cache_bytes=1) + fsspec.filesystem("memory").rm(url) + again = tmp_path / "again.b2z" + with blosc2.open(artifact) as table: + assert table.cache_bytes == retained + table.traffic.reset() + assert table["value"][0] == values[0] + assert table.traffic.requests == 0 + table.save(again) + with pytest.raises(FileNotFoundError): + table["value"][1] + with blosc2.open(again) as table: + assert table["value"][0] == values[0] + + +def test_remote_legacy_list_metadata(tmp_path, monkeypatch): + from blosc2.schema import ListSpec + + @dataclasses.dataclass + class Lists: + tags: list[int] = blosc2.field(blosc2.list(blosc2.int64(), batch_rows=None)) # noqa: RUF009 + + original = ListSpec.to_metadata_dict + + def legacy_metadata(spec): + result = original(spec) + if result.get("batch_rows") is None: + result.pop("batch_rows", None) + return result + + with monkeypatch.context() as patch: + patch.setattr(ListSpec, "to_metadata_dict", legacy_metadata) + local = blosc2.CTable(Lists, [([1, 2],), ([3],)], create_summary_index=False) + url = remote_table_url(tmp_path, local, "legacy-list") + with blosc2.open(url) as table: + assert table["tags"][:] == [[1, 2], [3]] + + +def test_hdf5_table_source_and_virtual_validity(): + from blosc2.ctable_storage import _AllValidRows + + url, _ = pytables_hdf5_url("source-kind.h5") + with blosc2.open(url, path="table") as table: + assert table.source["kind"] == "hdf5" + # Slicing a billion-row virtual mask must not allocate a billion-byte buffer. + valid = _AllValidRows(10**9, (1024,)) + selected = valid[10:20] + assert selected.strides == (0,) + assert selected.all() + np.testing.assert_array_equal(valid[[-1, 0]], [True, True]) + with pytest.raises(IndexError): + valid[[10**9]] diff --git a/tests/test_dataset_path.py b/tests/test_dataset_path.py index 01a3e1047..0b8faf245 100644 --- a/tests/test_dataset_path.py +++ b/tests/test_dataset_path.py @@ -140,3 +140,27 @@ def test_path_other_formats(tmp_path, format): np.testing.assert_array_equal(array[:], data) with blosc2.RemoteStore(url, path="group") as store: assert store.keys() == ["array"] + + +@pytest.mark.parametrize("cached", [False, True]) +def test_remote_hdf5_group_dispatch(tmp_path, cached): + h5py = pytest.importorskip("h5py") + source = tmp_path / "groups.h5" + with h5py.File(source, "w") as file: + file.create_dataset("group/data", data=np.arange(4)) + url = f"memory://{tmp_path.name}-groups.h5" + fsspec.filesystem("memory").pipe(url, source.read_bytes()) + options = {"cache_dir": tmp_path / "cache"} if cached else {} + for _ in range(2): + with blosc2.open(url, path="group", **options) as group: + assert isinstance(group, blosc2.RemoteStore) + assert group.keys() == ["data"] + with group["data"] as array: + np.testing.assert_array_equal(array[:], np.arange(4)) + for target in (url, source): + for root in (None, "", "/"): + with blosc2.open(target, path=root, **options) as group: + assert isinstance(group, blosc2.RemoteStore) + assert group.keys() == ["group"] + with pytest.raises(ValueError, match="not found"): + blosc2.open(url, path="missing") diff --git a/tests/test_list_array.py b/tests/test_list_array.py index 5de9ebd3b..437ed5bc5 100644 --- a/tests/test_list_array.py +++ b/tests/test_list_array.py @@ -340,3 +340,19 @@ def test_listarray_extend_arrow_flushes_pending_rows(): arr.extend_arrow(pa.array([[3, 4], [5, 6]], type=pa.list_(pa.int64()))) arr.flush() assert arr[:] == [[1, 2], [3, 4], [5, 6]] + + +def test_extend_arrow_preserves_typed_batches_without_python_cells(monkeypatch): + pa = pytest.importorskip("pyarrow") + arr = blosc2.ListArray( + item_spec=blosc2.int64(nullable=True), nullable=True, serializer="arrow", batch_rows=2 + ) + + def reject_python_cells(*args): + raise AssertionError("Arrow input must stay in Arrow") + + monkeypatch.setattr(arr, "_typed_batch", reject_python_cells) + values = [[1, None], [], None, [3]] + arr.extend_arrow(pa.chunked_array([pa.array(values, type=pa.list_(pa.int32()))])) + assert arr[:] == values + assert arr._backend._batch_lengths == [2, 2] diff --git a/tests/test_remote_store.py b/tests/test_remote_store.py index 39e1ab031..b479f1365 100644 --- a/tests/test_remote_store.py +++ b/tests/test_remote_store.py @@ -1626,3 +1626,69 @@ def do_HEAD(self): server.shutdown() server.server_close() assert int(result.stdout.strip()) == int(data[:10, :10].sum()) + + +@pytest.mark.parametrize("remote", [False, True]) +def test_materialize_preserves_metadata_and_expression_indexes(tmp_path, remote): + source = tmp_path / "materialize-metadata.b2z" + data = blosc2.arange(8, meta={"units": "metres"}) + data.attrs["description"] = "distance" + with blosc2.TreeStore(source, mode="w") as tree: + tree["array"] = data + tree["table"] = blosc2.CTable(NestedIndexedRow, [(i,) for i in range(8)], create_summary_index=False) + table = tree["table"] + table.create_index(expression="value * 2", kind="full", name="double") + table.create_index("value", kind="summary", granularity="chunk", name="values") + table.close() + if remote: + url = f"memory://{tmp_path.name}-materialize-metadata.b2z" + fsspec.filesystem("memory").pipe(url, source.read_bytes()) + store = blosc2.RemoteStore(url) + else: + store = blosc2.TreeStore(source, mode="r") + destination = tmp_path / "materialized.b2z" + with store: + store.materialize(destination) + with blosc2.open(destination) as tree: + array = tree["array"] + assert array.schunk.meta["units"] == "metres" + assert array.attrs["description"] == "distance" + np.testing.assert_array_equal(array[:], np.arange(8)) + table = tree["table"] + assert table.index(expression="value * 2").name == "double" + assert table.index("value").name == "values" + assert set(table.index("value").descriptor["levels"]) == {"chunk"} + np.testing.assert_array_equal(table.where("value * 2 >= 10")["value"][:], [5, 6, 7]) + table.close() + + +def test_nested_reference_preserves_batch_cache(tmp_path): + @dataclasses.dataclass + class Payload: + value: bytes = blosc2.field(blosc2.vlbytes(batch_rows=1)) + + value = np.random.default_rng(1).bytes(128_000) + source = tmp_path / "batch-source.b2z" + with blosc2.TreeStore(source, mode="w", threshold=0) as tree: + tree["table"] = blosc2.CTable(Payload, [(value,)], create_summary_index=False) + fs = fsspec.filesystem("memory") + source_url = f"memory://{tmp_path.name}-batch-source.b2z" + fs.pipe(source_url, source.read_bytes()) + host = tmp_path / "batch-host.b2z" + with blosc2.RemoteStore(source_url) as remote, blosc2.TreeStore(host, mode="w") as tree: + tree["linked"] = remote + host_url = f"memory://{tmp_path.name}-batch-host.b2z" + fs.pipe(host_url, host.read_bytes()) + snapshot = tmp_path / "batch-snapshot.b2z" + with blosc2.RemoteStore(host_url) as root: + with root["linked/table"] as table: + assert table["value"][0] == value + retained = root.cache_bytes + root.save(snapshot) + fs.rm(source_url) + fs.rm(host_url) + with blosc2.open(snapshot) as root: + with root["linked/table"] as table: + assert root.cache_bytes == retained + assert table["value"][0] == value + assert root.traffic.requests == 0 From b40bf7fad7dda85ccd4572265d8b73dafad16a70 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Thu, 24 Sep 2026 07:27:52 +0200 Subject: [PATCH 79/82] Reduce slow test setup and teardown costs --- tests/ndarray/test_proxy.py | 2 +- tests/test_fsspec.py | 3 ++- tests/test_python_blosc.py | 3 ++- tests/test_remote_store.py | 17 +++++++---------- 4 files changed, 12 insertions(+), 13 deletions(-) diff --git a/tests/ndarray/test_proxy.py b/tests/ndarray/test_proxy.py index d6153f63a..6903c3d8d 100644 --- a/tests/ndarray/test_proxy.py +++ b/tests/ndarray/test_proxy.py @@ -20,7 +20,7 @@ (None, [77, 134, 13], [31, 13, 5], [7, 8, 3], (slice(3, 7), slice(50, 100), 7), np.float64), ( "b2nd", - [12, 13, 14, 15, 16], + [6, 7, 8, 9, 11], # Keep multiple chunks and partial edges in every dimension. [5, 5, 5, 5, 5], [2, 2, 2, 2, 2], (slice(1, 3), ..., slice(3, 6)), diff --git a/tests/test_fsspec.py b/tests/test_fsspec.py index cd6600060..e10b08279 100644 --- a/tests/test_fsspec.py +++ b/tests/test_fsspec.py @@ -800,7 +800,8 @@ def do_HEAD(self): server = http.server.ThreadingHTTPServer(("127.0.0.1", 0), handler) server.requests = [] server.head_requests = [] if head_requests is None else head_requests - threading.Thread(target=server.serve_forever, daemon=True).start() + # shutdown() waits for the next poll; the default adds up to 0.5 s per test. + threading.Thread(target=server.serve_forever, kwargs={"poll_interval": 0.01}, daemon=True).start() try: yield f"http://127.0.0.1:{server.server_address[1]}", server.requests finally: diff --git a/tests/test_python_blosc.py b/tests/test_python_blosc.py index 55e2b7aa7..5d9e25f3a 100644 --- a/tests/test_python_blosc.py +++ b/tests/test_python_blosc.py @@ -220,7 +220,8 @@ def leaks(operation, repeats=3): for _ in range(2): for _ in range(repeats): operation() - gc.collect() + # These operations return bytes, freed by reference counting; + # repeated full collections only rescan unrelated suite objects. used_mem_after = process.memory_info().rss growth.append(used_mem_after - used_mem_before) used_mem_before = used_mem_after diff --git a/tests/test_remote_store.py b/tests/test_remote_store.py index b479f1365..a233d63c4 100644 --- a/tests/test_remote_store.py +++ b/tests/test_remote_store.py @@ -862,7 +862,10 @@ def test_sparse_store_processes_and_crash(tmp_path): ctx = multiprocessing.get_context("spawn") barrier, results = ctx.Barrier(4), ctx.Queue() workers = [ctx.Process(target=_shared_reader, args=(url, cache, barrier, results)) for _ in range(4)] - for worker in workers: + # The crash uses a separate cache, so overlap its interpreter startup. + crash_cache = str(tmp_path / "crash") + crash_worker = ctx.Process(target=_shared_crash, args=(url, crash_cache)) + for worker in [*workers, crash_worker]: worker.start() try: answers = [results.get(timeout=60) for _ in workers] @@ -872,20 +875,14 @@ def test_sparse_store_processes_and_crash(tmp_path): for worker in workers: worker.join(timeout=10) assert worker.exitcode == 0 + crash_worker.join(timeout=30) finally: - for worker in workers: + for worker in [*workers, crash_worker]: if worker.is_alive(): worker.terminate() worker.join() results.close() - crash_cache = str(tmp_path / "crash") - worker = ctx.Process(target=_shared_crash, args=(url, crash_cache)) - worker.start() - worker.join(timeout=30) - if worker.is_alive(): - worker.terminate() - worker.join() - assert worker.exitcode == 17 + assert crash_worker.exitcode == 17 with ( blosc2.RemoteStore.with_sparse_cache(url, crash_cache, _filesystem=_shared_fs()) as store, store["a"] as array, From 385b24e3eba7ce4222f65e0b10e256ebdc232d19 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Thu, 24 Sep 2026 07:42:41 +0200 Subject: [PATCH 80/82] Fix local array refresh on Windows --- src/blosc2/remote_array.py | 11 +++++++---- tests/test_remote_array.py | 3 +++ 2 files changed, 10 insertions(+), 4 deletions(-) diff --git a/src/blosc2/remote_array.py b/src/blosc2/remote_array.py index 8e888323f..955e2aed5 100644 --- a/src/blosc2/remote_array.py +++ b/src/blosc2/remote_array.py @@ -884,10 +884,13 @@ def refresh(self) -> None: urlpath = self._runtime_urlpath if isinstance(urlpath, str) and self._source_format == "zarr" and self._dataset: - parsed = urlsplit(urlpath) - suffix = f"/{self._dataset}" - if parsed.path.endswith(suffix): - urlpath = urlunsplit(parsed._replace(path=parsed.path[: -len(suffix)])) + if self._local_source: + urlpath = str(Path(urlpath).parents[len(self._dataset.split("/")) - 1]) + else: + parsed = urlsplit(urlpath) + suffix = f"/{self._dataset}" + if parsed.path.endswith(suffix): + urlpath = urlunsplit(parsed._replace(path=parsed.path[: -len(suffix)])) options = { "cache_policy": self.cache_policy, "max_concurrency": self._max_concurrency, diff --git a/tests/test_remote_array.py b/tests/test_remote_array.py index 6c3f70bbb..946098678 100644 --- a/tests/test_remote_array.py +++ b/tests/test_remote_array.py @@ -117,6 +117,9 @@ def write(path, data): np.testing.assert_array_equal(array[:], old) carrier = array.cache_path write(replacement, new) + if format == "h5": + # Release h5py's file handle before replacing the source on Windows. + array.src.close() if source.is_dir(): shutil.rmtree(source) os.replace(replacement, source) From c8a5a33b35c70cb5b2111b2a9b7dfe09e93cca7a Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Thu, 24 Sep 2026 08:04:45 +0200 Subject: [PATCH 81/82] Fix remote review findings --- src/blosc2/core.py | 4 +- src/blosc2/msgpack_utils.py | 6 ++- src/blosc2/remote_store_cache.py | 4 ++ src/blosc2/store_materialize.py | 15 ++++-- tests/test_batch_array.py | 9 +++- tests/test_remote_store.py | 79 ++++++++++++++++++++++++++++++++ 6 files changed, 108 insertions(+), 9 deletions(-) diff --git a/src/blosc2/core.py b/src/blosc2/core.py index a9abeda71..67b240def 100644 --- a/src/blosc2/core.py +++ b/src/blosc2/core.py @@ -714,13 +714,13 @@ def resolve_dataset_path(dataset, path): None means unspecified. Empty strings and slash-only strings select the root; retain them until URL parsing so duplicate URL selectors still fail. """ + if dataset is not None and not isinstance(dataset, str): + raise TypeError("dataset must be a string or None") if path is None: return dataset if not isinstance(path, str): raise TypeError("path must be a string or None") if dataset is not None: - if not isinstance(dataset, str): - raise TypeError("dataset must be a string or None") if dataset.strip("/") != path.strip("/"): raise ValueError("Conflicting dataset and path parameters") return dataset diff --git a/src/blosc2/msgpack_utils.py b/src/blosc2/msgpack_utils.py index 3a6bc358c..2ea7a4c83 100644 --- a/src/blosc2/msgpack_utils.py +++ b/src/blosc2/msgpack_utils.py @@ -7,6 +7,7 @@ from __future__ import annotations +import math import struct import numpy as np @@ -173,7 +174,10 @@ def decode_ext(code, data): isinstance(size, bool) or not isinstance(size, int) or size < 0 for size in shape ): raise ValueError("Unsafe remote NumPy extension shape") - count = int(np.prod(shape, dtype=np.int64)) + count = math.prod(shape) + limit = np.iinfo(np.intp).max + if any(size > limit for size in shape) or count > limit // np.dtype(object).itemsize: + raise ValueError("Unsafe remote NumPy extension shape") if "values" in value: if not isinstance(value["values"], list) or len(value["values"]) != count: raise ValueError("Invalid remote object-array extension") diff --git a/src/blosc2/remote_store_cache.py b/src/blosc2/remote_store_cache.py index aa4f36246..2e81ff9f7 100644 --- a/src/blosc2/remote_store_cache.py +++ b/src/blosc2/remote_store_cache.py @@ -246,6 +246,9 @@ def __enter__(self): if manifest is not None: if owner.generation != manifest["generation"]: # Child handles fail their generation check before using these resources. + for store in owner.linked_stores.values(): + store.close() + owner.linked_stores.clear() if owner.archive is not None: owner.archive.close() owner.archive = None @@ -254,6 +257,7 @@ def __enter__(self): owner.zstore = None owner.sources.clear() owner.source_descriptors.clear() + owner.batch_caches.clear() owner.generation = manifest["generation"] owner.metadata = manifest["metadata"] owner.nodes.clear() diff --git a/src/blosc2/store_materialize.py b/src/blosc2/store_materialize.py index 8d3cd962a..3c1a9e831 100644 --- a/src/blosc2/store_materialize.py +++ b/src/blosc2/store_materialize.py @@ -150,13 +150,18 @@ def _copy_array(source, target, path, staging): def _copy_table(table, target, path): indexes = dict(table._get_index_catalog()) - local = table.copy(compact=True) - local._source_bound = False - local._source_columns = set() - target[path] = local - local.close() + batch_rows = 2048 + seed = table._empty_copy(capacity=1) + seed._source_bound = False + seed._source_columns = set() + seed._create_summary_index = False + target[path] = seed materialized = target[path] try: + for start in range(0, len(table), batch_rows): + materialized.extend(table[start : start + batch_rows], validate=False) + for key, value in table.attrs.items(): + materialized.attrs[key] = value for name, descriptor in indexes.items(): options = table._index_create_kwargs_from_descriptor(descriptor) materialized.create_index(None if "expression" in options else name, **options) diff --git a/tests/test_batch_array.py b/tests/test_batch_array.py index 49981739d..bb778c562 100644 --- a/tests/test_batch_array.py +++ b/tests/test_batch_array.py @@ -6,9 +6,10 @@ ####################################################################### import pytest +from msgpack import ExtType, packb import blosc2 -from blosc2.msgpack_utils import msgpack_packb, msgpack_unpackb +from blosc2.msgpack_utils import _safe_msgpack_unpackb, msgpack_packb, msgpack_unpackb BATCHES = [ [b"bytes\x00payload", "plain text", 42], @@ -17,6 +18,12 @@ ] +def test_remote_ndarray_shape_overflow_rejected(): + extension = ExtType(46, packb({"shape": [2**32, 2**32], "values": []}, use_bin_type=True)) + with pytest.raises(ValueError, match="Unsafe remote NumPy extension shape"): + _safe_msgpack_unpackb(packb(extension, use_bin_type=True)) + + def _make_payload(seed, size): base = bytes((seed + i) % 251 for i in range(251)) reps = size // len(base) + 1 diff --git a/tests/test_remote_store.py b/tests/test_remote_store.py index a233d63c4..084993097 100644 --- a/tests/test_remote_store.py +++ b/tests/test_remote_store.py @@ -1082,6 +1082,8 @@ def test_zarr_unsupported_codec(): def test_store_validation(): with pytest.raises(TypeError, match="dataset must be a string"): blosc2.RemoteStore("memory://a.b2z", dataset=1) + with pytest.raises(TypeError, match="dataset must be a string"): + blosc2.open("memory://a.b2z", dataset=1) with pytest.raises(ValueError, match="cache_dir"): blosc2.RemoteStore("memory://a.b2z", cache_policy=blosc2.CachePolicy.DISK) with pytest.raises(TypeError, match="CachePolicy"): @@ -1659,6 +1661,38 @@ def test_materialize_preserves_metadata_and_expression_indexes(tmp_path, remote) table.close() +def test_materialize_table_reads_bounded_row_batches(tmp_path, monkeypatch): + @dataclasses.dataclass + class Payload: + value: bytes = blosc2.field(blosc2.vlbytes(batch_rows=128)) + + source = tmp_path / "large-table.b2z" + with blosc2.TreeStore(source, mode="w") as tree: + tree["table"] = blosc2.CTable( + NestedIndexedRow, [(i,) for i in range(5000)], create_summary_index=False + ) + tree["batch"] = blosc2.CTable( + Payload, [(bytes([i % 251]),) for i in range(3000)], create_summary_index=False + ) + url = f"memory://{tmp_path.name}-large-table.b2z" + fsspec.filesystem("memory").pipe(url, source.read_bytes()) + + def no_full_copy(*args, **kwargs): + raise AssertionError("materialization must not copy the whole table into memory") + + with blosc2.RemoteStore(url) as store: + with monkeypatch.context() as patch: + patch.setattr(blosc2.CTable, "copy", no_full_copy) + store.materialize(tmp_path / "bounded-table.b2d") + with blosc2.open(tmp_path / "bounded-table.b2d") as tree: + with tree["table"] as table: + np.testing.assert_array_equal(table["value"][:], np.arange(5000)) + with tree["batch"] as table: + assert len(table) == 3000 + assert table["value"][0] == b"\x00" + assert table["value"][2999] == bytes([2999 % 251]) + + def test_nested_reference_preserves_batch_cache(tmp_path): @dataclasses.dataclass class Payload: @@ -1689,3 +1723,48 @@ class Payload: assert root.cache_bytes == retained assert table["value"][0] == value assert root.traffic.requests == 0 + + +def test_shared_generation_reload_replaces_batch_and_linked_owners(tmp_path): + @dataclasses.dataclass + class Payload: + value: bytes = blosc2.field(blosc2.vlbytes(batch_rows=1)) + + source = tmp_path / "generation-source.b2z" + with blosc2.TreeStore(source, mode="w", threshold=0) as tree: + tree["table"] = blosc2.CTable(Payload, [(b"old",)], create_summary_index=False) + fs = fsspec.filesystem("memory") + source_url = f"memory://{tmp_path.name}-generation-source.b2z" + fs.pipe(source_url, source.read_bytes()) + with blosc2.open(source_url, cache_dir=tmp_path / "leaf-cache", shared_cache=True) as root: + with root["table"] as table: + assert table["value"][0] == b"old" + old_batch = next(iter(root._owner.batch_caches.values())) + with root._owner.disk.guard(): + manifest = root._owner.disk.load() + manifest["generation"] = "b" * 32 + root._owner.disk.publish(manifest) + with root._owner.lock: + pass + assert all(cache is not old_batch for cache in root._owner.batch_caches.values()) + + host = tmp_path / "generation-host.b2z" + with blosc2.RemoteStore(source_url) as linked, blosc2.TreeStore(host, mode="w") as tree: + tree["linked"] = linked + host_url = f"memory://{tmp_path.name}-generation-host.b2z" + fs.pipe(host_url, host.read_bytes()) + + with blosc2.open(host_url, cache_dir=tmp_path / "cache", shared_cache=True) as root: + with root["linked/table"] as table: + assert table["value"][0] == b"old" + old_linked = root._owner.linked_stores["linked"]._owner + old_batch = next(iter(old_linked.batch_caches.values())) + with root._owner.disk.guard(): + manifest = root._owner.disk.load() + manifest["generation"] = "a" * 32 + root._owner.disk.publish(manifest) + with root._owner.lock: + pass + assert not root._owner.linked_stores + assert old_linked._closed + assert all(cache is not old_batch for cache in old_linked.batch_caches.values()) From f3244feb734f81840d0405823b7f373b3a285216 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Thu, 24 Sep 2026 08:28:49 +0200 Subject: [PATCH 82/82] Forward sparse table storage options --- src/blosc2/remote_ctable.py | 2 ++ tests/ctable/test_remote_ctable.py | 13 +++++++++++++ 2 files changed, 15 insertions(+) diff --git a/src/blosc2/remote_ctable.py b/src/blosc2/remote_ctable.py index 6243c88c1..3deaef4c8 100644 --- a/src/blosc2/remote_ctable.py +++ b/src/blosc2/remote_ctable.py @@ -143,6 +143,7 @@ def with_sparse_cache( manifest=None, max_cache_bytes=CACHE_POLICY_DEFAULT, carrier=None, + storage_options=None, max_concurrency=8, metadata_buffer_bytes=8 << 20, row_buffer_bytes=64 << 20, @@ -177,6 +178,7 @@ def with_sparse_cache( manifest=manifest, max_cache_bytes=max_cache_bytes, carrier=carrier, + storage_options=storage_options, _filesystem=_filesystem, _source_validator=_source_validator, _manifest_validator=_manifest_validator, diff --git a/tests/ctable/test_remote_ctable.py b/tests/ctable/test_remote_ctable.py index 0a07c7a94..52b45256f 100644 --- a/tests/ctable/test_remote_ctable.py +++ b/tests/ctable/test_remote_ctable.py @@ -439,6 +439,19 @@ def test_remote_summary_warm_reference_and_sparse_cache(tmp_path): old_summary[:] +def test_sparse_cache_storage_options_partition(tmp_path): + url = remote_table_url( + tmp_path, blosc2.CTable(IndexedRow, [(1, 2)], create_summary_index=False), "options" + ) + cache = tmp_path / "cache" + for endpoint in ("one", "two"): + with blosc2.RemoteCTable.with_sparse_cache( + url, cache, storage_options={"endpoint": endpoint} + ) as table: + assert table["x"][0] == 1 + assert len([path for path in cache.iterdir() if path.is_dir()]) == 2 + + @pytest.mark.parametrize(("expression", "expected"), [("x == 7", [4993]), ("x == 5000", [])]) def test_remote_full_index_selective_lookup(tmp_path, expression, expected): url, _ = indexed_remote_table_url(tmp_path, "full", rows=5000)