From b81d80960c5113c257e51e5a87fc26df5d8b8ddb Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 4 Sep 2026 18:06:44 +0200 Subject: [PATCH 01/20] Add secure remote proxy support --- caterva2-server.sample.toml | 14 ++ caterva2/services/remote_proxy.py | 310 ++++++++++++++++++++++++++++ caterva2/services/server.py | 49 ++++- caterva2/services/srv_utils.py | 9 +- caterva2/tests/test_api.py | 42 ++++ caterva2/tests/test_remote_proxy.py | 155 ++++++++++++++ doc/utilities/cat2-server.md | 38 ++++ pyproject.toml | 1 + 8 files changed, 609 insertions(+), 9 deletions(-) create mode 100644 caterva2/services/remote_proxy.py create mode 100644 caterva2/tests/test_remote_proxy.py diff --git a/caterva2-server.sample.toml b/caterva2-server.sample.toml index dd35ac3d..d6cafbb4 100644 --- a/caterva2-server.sample.toml +++ b/caterva2-server.sample.toml @@ -30,6 +30,20 @@ register = true # allow users to register # publish_root = "s3://a-bucket/published" # peer_cache_quota = "1G" +# Persisted RemoteProxy objects are discoverable but cannot make outbound +# requests by default. The first opt-in backend is credential-free HTTPS. +# Every destination must be listed exactly; redirects, URL queries, private or +# otherwise non-public destination addresses, and embedded expression +# references are refused. +# [server.remote_proxy] +# enabled = true +# allowed_hosts = ["datasets.example.org", "objects.example.org:8443"] +# timeout = 30 +# max_nbytes = 1073741824 +# max_rank = 16 +# max_chunks = 10000000 +# max_concurrency = 8 + # Mount a remote Caterva2 server's locally owned @public root as @labb. This # activates the bundled C2Cache provider. Repeat the table to mount more peers. # An optional per-peer cache_quota further bounds this peer within the pool. diff --git a/caterva2/services/remote_proxy.py b/caterva2/services/remote_proxy.py new file mode 100644 index 00000000..9efeb60f --- /dev/null +++ b/caterva2/services/remote_proxy.py @@ -0,0 +1,310 @@ +############################################################################### +# Caterva2 - On demand access to remote Blosc2 data repositories +# +# Copyright (c) 2023 ironArray SLU +# https://www.blosc.org +# License: GNU Affero General Public License v3.0 +# See LICENSE.txt for details about copyright and rights to use. +############################################################################### + +"""Policy boundary for persisted remote-array references. + +The carrier is inspected without resolving it. Resolution is default-deny and +the first supported server backend is HTTPS with an explicit host allowlist, +publicly routable pinned addresses, and redirects disabled. +""" + +from __future__ import annotations + +import ipaddress +import math +import socket +from dataclasses import dataclass +from urllib.parse import urlsplit + +import aiohttp +import blosc2 +from fsspec.implementations.http import HTTPFileSystem + + +class RemoteProxyDenied(ValueError): + """The server policy refuses a remote reference.""" + + +@dataclass(frozen=True) +class Policy: + enabled: bool = False + allowed_hosts: tuple[str, ...] = () + timeout: float = 30.0 + max_nbytes: int = 1 << 30 + max_rank: int = 16 + max_chunks: int = 10_000_000 + max_concurrency: int = 8 + + +policy = Policy() + + +def configure(conf) -> None: + """Load the remote-reference policy from the server configuration.""" + global policy + + enabled = conf.get(".remote_proxy.enabled", False) + hosts = conf.get(".remote_proxy.allowed_hosts", ()) + timeout = conf.get(".remote_proxy.timeout", 30.0) + max_nbytes = conf.get(".remote_proxy.max_nbytes", 1 << 30) + max_rank = conf.get(".remote_proxy.max_rank", 16) + max_chunks = conf.get(".remote_proxy.max_chunks", 10_000_000) + max_concurrency = conf.get(".remote_proxy.max_concurrency", 8) + + if not isinstance(enabled, bool): + raise ValueError("remote_proxy.enabled must be true or false") + if enabled and not hasattr(blosc2, "RemoteProxy"): + raise ValueError("remote_proxy.enabled requires a Python-Blosc2 version with RemoteProxy support") + if not isinstance(hosts, list | tuple) or any(not isinstance(host, str) for host in hosts): + raise ValueError("remote_proxy.allowed_hosts must be a list of host names") + if not isinstance(timeout, int | float) or isinstance(timeout, bool) or timeout <= 0: + raise ValueError("remote_proxy.timeout must be positive") + for name, value in { + "max_nbytes": max_nbytes, + "max_rank": max_rank, + "max_chunks": max_chunks, + "max_concurrency": max_concurrency, + }.items(): + if not isinstance(value, int) or isinstance(value, bool) or value <= 0: + raise ValueError(f"remote_proxy.{name} must be a positive integer") + + policy = Policy( + enabled=enabled, + allowed_hosts=tuple(_normalize_allowed_host(host) for host in hosts), + timeout=float(timeout), + max_nbytes=max_nbytes, + max_rank=max_rank, + max_chunks=max_chunks, + max_concurrency=max_concurrency, + ) + + +def _normalize_allowed_host(value: str) -> str: + parsed = urlsplit(f"//{value}") + if parsed.username is not None or parsed.password is not None or parsed.path not in {"", "/"}: + raise ValueError(f"invalid remote_proxy allowed host: {value!r}") + if parsed.hostname is None: + raise ValueError(f"invalid remote_proxy allowed host: {value!r}") + try: + host = parsed.hostname.encode("idna").decode("ascii").lower() + port = parsed.port + except (UnicodeError, ValueError) as exc: + raise ValueError(f"invalid remote_proxy allowed host: {value!r}") from exc + return f"{host}:{port}" if port is not None else host + + +def raw_carrier(path): + """Open a local Blosc2 carrier without dispatching its B2 object.""" + kwargs = {"dparams": blosc2.DParams(nthreads=1)} + return blosc2.blosc2_ext.open(str(path), "r", 0, **kwargs) + + +def inspect(path): + """Return ``(raw carrier, payload)`` for a RemoteProxy, otherwise ``None``.""" + if not hasattr(blosc2, "RemoteProxy"): + return None + try: + carrier = raw_carrier(path) + except (RuntimeError, ValueError): + return None + schunk = getattr(carrier, "schunk", carrier) + marker = schunk.meta.get("b2o") + if not isinstance(marker, dict) or marker.get("kind") != "remote_proxy": + return None + payload = schunk.vlmeta.get("b2o") + if not isinstance(payload, dict): + raise RemoteProxyDenied("RemoteProxy carrier has no valid payload") + return carrier, payload + + +def guard_embedded(path) -> None: + """Reject remote references hidden in another persisted B2 object. + + Structured LazyExpr/LazyUDF decoding resolves operand references while the + object is opened. Until the server can inject this module's secure source + factory into that decoder, refusing those operands closes an otherwise + easy way around the direct-carrier policy check. + """ + if not hasattr(blosc2, "RemoteProxy"): + return + try: + carrier = raw_carrier(path) + except (RuntimeError, ValueError): + return + schunk = getattr(carrier, "schunk", carrier) + marker = schunk.meta.get("b2o") + if not isinstance(marker, dict) or marker.get("kind") not in {"lazyexpr", "lazyudf"}: + return + payload = schunk.vlmeta.get("b2o") + if _contains_remote_reference(payload): + raise RemoteProxyDenied( + "remote references embedded in persisted expressions are disabled by server policy" + ) + + +def _contains_remote_reference(value) -> bool: + if isinstance(value, dict): + if value.get("kind") in {"fsspec", "remote_proxy"}: + return True + return any(_contains_remote_reference(item) for item in value.values()) + if isinstance(value, list | tuple): + return any(_contains_remote_reference(item) for item in value) + return False + + +def is_metadata(meta) -> bool: + """Whether an api/info model describes a reference-only carrier.""" + vlmeta = getattr(getattr(meta, "schunk", None), "vlmeta", None) or {} + payload = vlmeta.get("b2o") + return isinstance(payload, dict) and payload.get("kind") == "remote_proxy" + + +def _validated_source(payload: dict) -> str: + if not policy.enabled: + raise RemoteProxyDenied("RemoteProxy resolution is disabled by server policy") + if set(payload) != {"kind", "version", "source", "cache_policy"}: + raise RemoteProxyDenied("RemoteProxy payload contains unsupported fields") + if payload.get("kind") != "remote_proxy" or payload.get("version") != 1: + raise RemoteProxyDenied("unsupported RemoteProxy payload") + if payload.get("cache_policy") != "none": + raise RemoteProxyDenied("server RemoteProxy carriers must use cache policy 'none'") + source = payload.get("source") + if not isinstance(source, dict) or set(source) != {"kind", "version", "urlpath"}: + raise RemoteProxyDenied("server RemoteProxy supports only a versioned fsspec URL source") + if source.get("kind") != "fsspec" or source.get("version") != 1: + raise RemoteProxyDenied("server RemoteProxy supports only fsspec source version 1") + url = source.get("urlpath") + if not isinstance(url, str): + raise RemoteProxyDenied("RemoteProxy source URL must be a string") + + parsed = urlsplit(url) + if parsed.scheme.lower() != "https": + raise RemoteProxyDenied("server RemoteProxy currently permits only HTTPS sources") + if parsed.username is not None or parsed.password is not None: + raise RemoteProxyDenied("RemoteProxy source URLs cannot contain user information") + if parsed.query or parsed.fragment: + raise RemoteProxyDenied("RemoteProxy source URLs cannot contain a query or fragment") + if parsed.hostname is None: + raise RemoteProxyDenied("RemoteProxy source URL has no host") + try: + host = parsed.hostname.encode("idna").decode("ascii").lower() + port = parsed.port or 443 + except (UnicodeError, ValueError) as exc: + raise RemoteProxyDenied("RemoteProxy source URL has an invalid host or port") from exc + authority = host if port == 443 else f"{host}:{port}" + if authority not in policy.allowed_hosts: + raise RemoteProxyDenied(f"RemoteProxy destination {authority!r} is not allowed") + return url + + +def _public_addresses(host: str, port: int) -> tuple[str, ...]: + try: + answers = socket.getaddrinfo(host, port, type=socket.SOCK_STREAM) + except OSError as exc: + raise RemoteProxyDenied(f"RemoteProxy destination {host!r} cannot be resolved") from exc + addresses = tuple(dict.fromkeys(answer[4][0] for answer in answers)) + if not addresses: + raise RemoteProxyDenied(f"RemoteProxy destination {host!r} has no addresses") + denied = [address for address in addresses if not ipaddress.ip_address(address).is_global] + if denied: + raise RemoteProxyDenied(f"RemoteProxy destination {host!r} resolves to a non-public address") + return addresses + + +class _PinnedResolver(aiohttp.abc.AbstractResolver): + def __init__(self, host: str, addresses: tuple[str, ...]): + self.host = host + self.addresses = addresses + + async def resolve(self, host, port=0, family=socket.AF_UNSPEC): + if host.encode("idna").decode("ascii").lower() != self.host: + raise OSError("redirected hosts are not allowed for RemoteProxy sources") + records = [] + for address in self.addresses: + address_family = socket.AF_INET6 if ":" in address else socket.AF_INET + if family not in {socket.AF_UNSPEC, address_family}: + continue + records.append( + { + "hostname": host, + "host": address, + "port": port, + "family": address_family, + "proto": 0, + "flags": 0, + } + ) + return records + + async def close(self): + return None + + +def _https_filesystem(host: str, addresses: tuple[str, ...]): + async def get_client(**kwargs): + connector = aiohttp.TCPConnector(resolver=_PinnedResolver(host, addresses)) + timeout = aiohttp.ClientTimeout(total=policy.timeout) + return aiohttp.ClientSession(connector=connector, timeout=timeout, **kwargs) + + return HTTPFileSystem( + get_client=get_client, + allow_redirects=False, + skip_instance_cache=True, + ) + + +def resolve(carrier, payload): + """Resolve one allowed carrier as a no-retention remote array.""" + url = _validated_source(payload) + parsed = urlsplit(url) + host = parsed.hostname.encode("idna").decode("ascii").lower() + addresses = _public_addresses(host, parsed.port or 443) + fs = _https_filesystem(host, addresses) + source = blosc2.FsspecNDSource( + url, + max_concurrency=policy.max_concurrency, + _filesystem=fs, + ) + + expected = (tuple(carrier.shape), carrier.dtype, tuple(carrier.chunks), tuple(carrier.blocks)) + actual = (tuple(source.shape), source.dtype, tuple(source.chunks), tuple(source.blocks)) + if actual != expected: + raise RemoteProxyDenied( + f"RemoteProxy source geometry does not match its carrier: carrier={expected}, source={actual}" + ) + if len(source.shape) > policy.max_rank: + raise RemoteProxyDenied(f"RemoteProxy rank exceeds the configured limit of {policy.max_rank}") + nbytes = math.prod(source.shape) * source.dtype.itemsize + if nbytes > policy.max_nbytes: + raise RemoteProxyDenied( + f"RemoteProxy logical size exceeds the configured limit of {policy.max_nbytes}" + ) + chunks = math.prod( + math.ceil(size / chunk) for size, chunk in zip(source.shape, source.chunks, strict=True) + ) + if chunks > policy.max_chunks: + raise RemoteProxyDenied( + f"RemoteProxy chunk count exceeds the configured limit of {policy.max_chunks}" + ) + return ServerRemoteProxy(source, expected) + + +class ServerRemoteProxy: + """Operation-scoped assembly over a server-authorized remote source.""" + + def __init__(self, source, geometry): + self.src = source + self.shape, self.dtype, self.chunks, self.blocks = geometry + self.cparams = source.cparams + + def __getitem__(self, item): + return blosc2.Proxy(self.src, _refresh_source=False)[item] + + def get_chunk(self, nchunk): + return self.src.get_chunk(nchunk) diff --git a/caterva2/services/server.py b/caterva2/services/server.py index 715e1dfb..86d8d32e 100644 --- a/caterva2/services/server.py +++ b/caterva2/services/server.py @@ -61,7 +61,7 @@ # Project from caterva2 import hdf5, models, utils -from caterva2.services import db, providers, schemas, settings, srv_utils, users +from caterva2.services import db, providers, remote_proxy, schemas, settings, srv_utils, users from caterva2.services.notebook import inject_pyodide_bootstrap_cell BASE_DIR = pathlib.Path(__file__).resolve().parent @@ -215,6 +215,18 @@ def open_b2(abspath, path): if root not in {"@personal", "@shared", "@public"}: raise ValueError(f"Unexpected root={root}") + reference = remote_proxy.inspect(abspath) + if reference is not None: + carrier, payload = reference + try: + return remote_proxy.resolve(carrier, payload) + except remote_proxy.RemoteProxyDenied as exc: + raise fastapi.HTTPException(status_code=403, detail=str(exc)) from exc + + try: + remote_proxy.guard_embedded(abspath) + except remote_proxy.RemoteProxyDenied as exc: + raise fastapi.HTTPException(status_code=403, detail=str(exc)) from exc container = blosc2.open(abspath) # CTable has its own storage and no table-level cparams/dparams; return early. if isinstance(container, blosc2.CTable): @@ -584,7 +596,10 @@ async def get_info( etag = dataset_etag(abspath) if etag: response.headers["ETag"] = etag - meta = srv_utils.read_metadata(abspath) + try: + meta = srv_utils.read_metadata(abspath) + except remote_proxy.RemoteProxyDenied as exc: + raise fastapi.HTTPException(status_code=403, detail=str(exc)) from exc # A dataset with a file of its own is served by `FileResponse`, which honours # a range. Only said where it is certain: a directory or a lazy expression # is not a stored frame, and a container member depends on whether its leaf @@ -597,8 +612,11 @@ async def get_info( # the file on disk is a proxy's chunks, not the array's. It 416s a range, # and a client that took "bytes" on trust would find that out on its first # block read rather than on a probe it could have made - if isinstance(meta, models.Metadata) and not srv_utils.is_hdf5_proxy_meta(meta): - meta.accept_ranges = "bytes" + if isinstance(meta, models.Metadata): + if remote_proxy.is_metadata(meta): + meta.accept_ranges = "none" + elif not srv_utils.is_hdf5_proxy_meta(meta): + meta.accept_ranges = "bytes" return meta @@ -1034,7 +1052,10 @@ async def fetch_data( headers=with_etag(abspath), ) - if isinstance(container, (blosc2.NDArray, blosc2.LazyArray, hdf5.HDF5Proxy, blosc2.NDField)): + if isinstance( + container, + (blosc2.NDArray, blosc2.LazyArray, hdf5.HDF5Proxy, blosc2.NDField, remote_proxy.ServerRemoteProxy), + ): array = container schunk = getattr(array, "schunk", None) # not really needed typesize = array.dtype.itemsize @@ -1066,7 +1087,16 @@ async def fetch_data( if ( whole - and (not isinstance(array, blosc2.LazyArray | hdf5.HDF5Proxy | blosc2.NDField | blosc2.CTable)) + and ( + not isinstance( + array, + blosc2.LazyArray + | hdf5.HDF5Proxy + | blosc2.NDField + | blosc2.CTable + | remote_proxy.ServerRemoteProxy, + ) + ) and (not filter) ): if inner_key is None: @@ -1095,7 +1125,7 @@ async def fetch_data( srv_utils.refuse_range(range_header, path) if indices is not None: - if not isinstance(array, blosc2.NDArray): + if not isinstance(array, blosc2.NDArray | remote_proxy.ServerRemoteProxy): srv_utils.raise_bad_request(f"{path} is not an array that can be indexed by coordinates") try: # `NDArray` reads scattered coordinates through its own sparse gather, @@ -1120,6 +1150,10 @@ async def fetch_data( data = array[() if slice_ is None else slice_] data = blosc2.asarray(data) data = data.to_cframe() + elif isinstance(array, remote_proxy.ServerRemoteProxy): + data = await concurrency.run_in_threadpool( + lambda: blosc2.asarray(array[() if slice_ is None else slice_]).to_cframe() + ) elif isinstance(array, blosc2.NDArray): # Using NDArray.slice() allows a fast path when it is aligned with the chunks # As we are going to serialize the slice right away, it is not clear in which @@ -4139,6 +4173,7 @@ def main(): args = parser.parse_args() conf = utils.get_server_conf(args.conf) utils.config_log(args, conf) + remote_proxy.configure(conf) # Directories statedir = args.statedir or pathlib.Path(conf.get(".statedir", "_caterva2/state")) diff --git a/caterva2/services/srv_utils.py b/caterva2/services/srv_utils.py index 3b7800aa..304e372c 100644 --- a/caterva2/services/srv_utils.py +++ b/caterva2/services/srv_utils.py @@ -30,7 +30,7 @@ # Project from caterva2 import hdf5, models -from caterva2.services import db, schemas, settings, users +from caterva2.services import db, remote_proxy, schemas, settings, users # Shared suffix constants BLOSC2_ARRAY_SUFFIXES = {".b2nd", ".b2frame"} @@ -476,7 +476,12 @@ def read_metadata(obj, mtime=None): assert path.suffix in BLOSC2_NATIVE_SUFFIXES try: - obj = blosc2.open(path) + reference = remote_proxy.inspect(path) + if reference is not None: + obj = reference[0] + else: + remote_proxy.guard_embedded(path) + obj = blosc2.open(path) except blosc2.exceptions.MissingOperands as exc: error = "Lazy expression with missing operands" missing_ops = {k: get_relpath(v) for k, v in exc.missing_ops.items()} diff --git a/caterva2/tests/test_api.py b/caterva2/tests/test_api.py index 5f81a7ab..f71a7d67 100644 --- a/caterva2/tests/test_api.py +++ b/caterva2/tests/test_api.py @@ -12,6 +12,7 @@ import pathlib import blosc2 +import fsspec import httpx import numexpr as ne import numpy as np @@ -132,6 +133,47 @@ def test_dataset_info(client, fill_public): assert data.chunks == tuple(info["chunks"]) +def test_remote_proxy_is_discovered_but_resolution_is_disabled(client): + source = blosc2.arange(20, dtype=np.int32, chunks=(10,), blocks=(5,)) + fsspec.filesystem("memory").pipe_file("caterva2-disabled-reference.b2nd", source.to_cframe()) + reference = blosc2.RemoteProxy("memory://caterva2-disabled-reference.b2nd") + path = pathlib.Path(TEST_STATE_DIR) / "server/public/disabled-reference.b2nd" + reference.save(path) + before = path.read_bytes() + + try: + response = httpx.get(f"{client.urlbase}/api/info/@public/{path.name}") + response.raise_for_status() + info = response.json() + assert info["shape"] == [20] + assert info["chunks"] == [10] + assert info["blocks"] == [5] + assert info["accept_ranges"] == "none" + + response = httpx.get(f"{client.urlbase}/api/fetch/@public/{path.name}", params={"slice_": "0:2"}) + assert response.status_code == 403 + assert response.json()["detail"] == "RemoteProxy resolution is disabled by server policy" + assert path.read_bytes() == before + finally: + path.unlink(missing_ok=True) + + +def test_remote_proxy_hidden_in_expression_is_denied_before_open(client): + source = blosc2.arange(20, dtype=np.int32, chunks=(10,), blocks=(5,)) + fsspec.filesystem("memory").pipe_file("caterva2-embedded-reference.b2nd", source.to_cframe()) + reference = blosc2.RemoteProxy("memory://caterva2-embedded-reference.b2nd") + expression = blosc2.lazyexpr("a + 1", operands={"a": reference}) + path = pathlib.Path(TEST_STATE_DIR) / "server/public/embedded-reference.b2nd" + expression.save(path) + + try: + response = httpx.get(f"{client.urlbase}/api/info/@public/{path.name}") + assert response.status_code == 403 + assert "embedded" in response.json()["detail"] + finally: + path.unlink(missing_ok=True) + + @pytest.mark.parametrize("dirpath", [None, "dir1", "dir2", "dir2/dir3/dir4"]) @pytest.mark.parametrize("final_dir", [True, False]) def test_move(auth_client, dirpath, final_dir, fill_auth): diff --git a/caterva2/tests/test_remote_proxy.py b/caterva2/tests/test_remote_proxy.py new file mode 100644 index 00000000..73f4606b --- /dev/null +++ b/caterva2/tests/test_remote_proxy.py @@ -0,0 +1,155 @@ +############################################################################### +# Caterva2 - On demand access to remote Blosc2 data repositories +# +# Copyright (c) 2023 ironArray SLU +# https://www.blosc.org +# License: GNU Affero General Public License v3.0 +# See LICENSE.txt for details about copyright and rights to use. +############################################################################### + +import asyncio +import socket + +import blosc2 +import numpy as np +import pytest + +from caterva2.services import remote_proxy + + +class _Conf: + def __init__(self, values): + self.values = values + + def get(self, key, default=None): + return self.values.get(key, default) + + +def _payload(url): + return { + "kind": "remote_proxy", + "version": 1, + "source": {"kind": "fsspec", "version": 1, "urlpath": url}, + "cache_policy": "none", + } + + +@pytest.fixture(autouse=True) +def reset_policy(): + previous = remote_proxy.policy + yield + remote_proxy.policy = previous + + +def test_resolution_is_default_deny(): + remote_proxy.policy = remote_proxy.Policy() + with pytest.raises(remote_proxy.RemoteProxyDenied, match="disabled"): + remote_proxy._validated_source(_payload("https://data.example/array.b2nd")) + + +@pytest.mark.parametrize( + "url", + [ + "http://data.example/array.b2nd", + "https://user@data.example/array.b2nd", + "https://data.example/array.b2nd?token=secret", + "https://data.example:bad/array.b2nd", + "https://other.example/array.b2nd", + ], +) +def test_enabled_policy_still_rejects_unsafe_destinations(url): + remote_proxy.policy = remote_proxy.Policy(enabled=True, allowed_hosts=("data.example",)) + with pytest.raises(remote_proxy.RemoteProxyDenied): + remote_proxy._validated_source(_payload(url)) + + +def test_configured_https_destination_is_accepted(): + remote_proxy.configure( + _Conf( + { + ".remote_proxy.enabled": True, + ".remote_proxy.allowed_hosts": ["DATA.example", "data.example:8443"], + } + ) + ) + assert remote_proxy._validated_source(_payload("https://data.example/array.b2nd")) + assert remote_proxy._validated_source(_payload("https://data.example:8443/array.b2nd")) + + +def test_private_resolution_is_rejected(monkeypatch): + monkeypatch.setattr( + socket, + "getaddrinfo", + lambda *args, **kwargs: [(socket.AF_INET, socket.SOCK_STREAM, 6, "", ("127.0.0.1", 443))], + ) + with pytest.raises(remote_proxy.RemoteProxyDenied, match="non-public"): + remote_proxy._public_addresses("data.example", 443) + + +def test_pinned_resolver_rejects_redirected_host(): + resolver = remote_proxy._PinnedResolver("data.example", ("203.0.113.10",)) + with pytest.raises(OSError, match="redirected hosts"): + asyncio.run(resolver.resolve("other.example", 443)) + + +def test_embedded_fsspec_reference_is_recognized(): + payload = { + "kind": "lazyexpr", + "operands": {"a": {"kind": "fsspec", "version": 1, "urlpath": "https://data.example/a"}}, + } + assert remote_proxy._contains_remote_reference(payload) + + +def test_allowed_source_is_resolved_with_the_secure_filesystem(monkeypatch): + class Carrier: + shape = (10,) + dtype = np.dtype(np.int32) + chunks = (5,) + blocks = (5,) + + class Source: + shape = Carrier.shape + dtype = Carrier.dtype + chunks = Carrier.chunks + blocks = Carrier.blocks + cparams = blosc2.CParams() + + filesystem = object() + seen = {} + + def fake_source(url, max_concurrency, *, _filesystem): + seen.update(url=url, max_concurrency=max_concurrency, filesystem=_filesystem) + return Source() + + remote_proxy.policy = remote_proxy.Policy( + enabled=True, + allowed_hosts=("data.example",), + max_concurrency=3, + ) + monkeypatch.setattr(remote_proxy, "_public_addresses", lambda host, port: ("93.184.216.34",)) + monkeypatch.setattr(remote_proxy, "_https_filesystem", lambda host, addresses: filesystem) + monkeypatch.setattr(blosc2, "FsspecNDSource", fake_source) + + resolved = remote_proxy.resolve(Carrier(), _payload("https://data.example/array.b2nd")) + assert isinstance(resolved, remote_proxy.ServerRemoteProxy) + assert seen == { + "url": "https://data.example/array.b2nd", + "max_concurrency": 3, + "filesystem": filesystem, + } + + +def test_https_filesystem_disables_redirects_and_pins_resolution(): + remote_proxy.policy = remote_proxy.Policy(timeout=7) + fs = remote_proxy._https_filesystem("data.example", ("93.184.216.34",)) + assert fs.kwargs["allow_redirects"] is False + + async def inspect_client(): + client = await fs.get_client() + try: + assert client.timeout.total == 7 + assert isinstance(client.connector._resolver, remote_proxy._PinnedResolver) + finally: + await client.close() + + asyncio.run(inspect_client()) diff --git a/doc/utilities/cat2-server.md b/doc/utilities/cat2-server.md index 1726165f..f45df19b 100644 --- a/doc/utilities/cat2-server.md +++ b/doc/utilities/cat2-server.md @@ -31,3 +31,41 @@ listen = "0.0.0.0:8080" ``` And then simply run `cat2-server` to start it on all network interfaces on port 8080. + +## Remote reference policy + +A persisted `blosc2.RemoteProxy` is a small B2ND carrier that asks Caterva2 to +read another dataset. Caterva2 can inspect and report the carrier's stored +shape, dtype, chunk, and block metadata without contacting that source. +Outbound resolution is disabled by default. + +The initial opt-in backend supports public, credential-free HTTPS sources: + +```toml +[server.remote_proxy] +enabled = true +allowed_hosts = ["datasets.example.org", "objects.example.org:8443"] +timeout = 30 +max_nbytes = 1073741824 +max_rank = 16 +max_chunks = 10000000 +max_concurrency = 8 +``` + +The allowlist is mandatory and matches normalized host names and explicit +non-default ports exactly. Before connecting, Caterva2 resolves every address, +rejects loopback, private, link-local, multicast, and other non-public results, +and pins the accepted addresses into the HTTP connector. Redirects are disabled. +Source URLs containing user information, query parameters, or fragments are +also rejected. These checks are applied by the server even when the carrier was +created by a client that performed its own validation. + +The limits bound each upstream request's time, source rank, logical +uncompressed size, chunk count, and concurrent range fetches. Set them for the +capacity of the installation; they are not inferred from untrusted carrier +metadata. + +S3, private-source credential selection, and remote references embedded inside +persisted expressions are not enabled yet. Server credentials must eventually +be selected from an administrator-controlled destination mapping and must never +be accepted from a carrier. diff --git a/pyproject.toml b/pyproject.toml index 1f20892e..c251457c 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -59,6 +59,7 @@ base-services = [ "uvicorn", ] server = [ + "aiohttp", "aiosqlite", "caterva2[base-services]", "caterva2[hdf5]", From 7478babd6e66dd32413ebe20691f931a8355215c Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sat, 5 Sep 2026 09:03:32 +0200 Subject: [PATCH 02/20] Add self-caching remote proxies --- caterva2-server.sample.toml | 3 +- caterva2/api_utils.py | 5 +- caterva2/client.py | 20 +++- caterva2/services/remote_proxy.py | 168 +++++++++++++++++++++++----- caterva2/services/server.py | 78 +++++++++++-- caterva2/services/srv_utils.py | 10 +- caterva2/tests/test_api.py | 35 ++++++ caterva2/tests/test_remote_proxy.py | 136 +++++++++++++++++++++- doc/utilities/cat2-server.md | 31 +++-- 9 files changed, 427 insertions(+), 59 deletions(-) diff --git a/caterva2-server.sample.toml b/caterva2-server.sample.toml index d6cafbb4..561c3bb6 100644 --- a/caterva2-server.sample.toml +++ b/caterva2-server.sample.toml @@ -34,7 +34,8 @@ register = true # allow users to register # requests by default. The first opt-in backend is credential-free HTTPS. # Every destination must be listed exactly; redirects, URL queries, private or # otherwise non-public destination addresses, and embedded expression -# references are refused. +# references are refused. A proxy may cache chunks inside its own carrier up to +# its persisted max_cache_bytes; that growth counts against this server's quota. # [server.remote_proxy] # enabled = true # allowed_hosts = ["datasets.example.org", "objects.example.org:8443"] diff --git a/caterva2/api_utils.py b/caterva2/api_utils.py index b8ba7ea8..63849343 100644 --- a/caterva2/api_utils.py +++ b/caterva2/api_utils.py @@ -152,8 +152,9 @@ def key_to_indices(key, ndim=None): return json.dumps(out, separators=(",", ":")) -def get_download_url(path, urlbase): - return f"{urlbase}/api/download/{path}" +def get_download_url(path, urlbase, *, include_cache=True): + url = f"{urlbase}/api/download/{path}" + return url if include_cache else f"{url}?include_cache=false" def get_handle_url(path, urlbase): diff --git a/caterva2/client.py b/caterva2/client.py index dd647edb..e918f3cf 100644 --- a/caterva2/client.py +++ b/caterva2/client.py @@ -300,7 +300,7 @@ def _toplevel_path(self): raise ValueError(f"Not supported for a member inside a container file: {self.path}") return self.path - def download(self, localpath=None): + def download(self, localpath=None, *, include_cache=True): """ Downloads the file to storage. @@ -309,6 +309,9 @@ def download(self, localpath=None): localpath : Path, optional The destination path for the downloaded file. If not specified, the file will be downloaded to the current working directory. + include_cache : bool, optional + For a RemoteProxy carrier, include its valid warm cache data. + Pass false to download a cold proxy without changing the server copy. Returns ------- @@ -329,6 +332,7 @@ def download(self, localpath=None): return self.client.download( self._toplevel_path(), localpath=localpath, + include_cache=include_cache, ) def unfold(self): @@ -528,10 +532,13 @@ def vlmeta(self): schunk_meta = self.meta.get("schunk", self.meta) return schunk_meta.get("vlmeta", {}) - def get_download_url(self): + def get_download_url(self, *, include_cache=True): """ Retrieves the download URL for the file. + ``include_cache=False`` requests a cold RemoteProxy carrier. It has no + effect on other file types. + Returns ------- str @@ -546,7 +553,7 @@ def get_download_url(self): >>> file.get_download_url() 'https://cat2.cloud/demo/api/fetch/example/ds-1d.b2nd' """ - return api_utils.get_download_url(self.path, self.urlbase) + return api_utils.get_download_url(self.path, self.urlbase, include_cache=include_cache) def __getitem__(self, item): """ @@ -1489,7 +1496,7 @@ def _download_url(self, url, localpath, auth_cookie=None): return localpath - def download(self, dataset, localpath=None): + def download(self, dataset, localpath=None, *, include_cache=True): """ Downloads a dataset to local storage. @@ -1504,6 +1511,9 @@ def download(self, dataset, localpath=None): localpath : Path, optional Local path to save the downloaded dataset. Defaults to the current working directory if not specified. + include_cache : bool, optional + For a RemoteProxy carrier, include its valid warm cache data. + Pass false to download a cold proxy without changing the server copy. Returns ------- @@ -1519,7 +1529,7 @@ def download(self, dataset, localpath=None): PosixPath('example/ds-2d-fields.b2nd') """ urlbase, dataset = _format_paths(self.urlbase, dataset) - url = api_utils.get_download_url(dataset, urlbase) + url = api_utils.get_download_url(dataset, urlbase, include_cache=include_cache) localpath = pathlib.Path(localpath) if localpath else None if localpath is None: path = "." / pathlib.Path(dataset) diff --git a/caterva2/services/remote_proxy.py b/caterva2/services/remote_proxy.py index 9efeb60f..d862238f 100644 --- a/caterva2/services/remote_proxy.py +++ b/caterva2/services/remote_proxy.py @@ -19,11 +19,15 @@ import ipaddress import math import socket +import threading +import weakref from dataclasses import dataclass from urllib.parse import urlsplit import aiohttp import blosc2 +import numpy as np +from blosc2.b2objects import make_b2object_carrier, write_b2object_payload from fsspec.implementations.http import HTTPFileSystem @@ -44,6 +48,20 @@ class Policy: policy = Policy() +_carrier_locks: weakref.WeakValueDictionary[str, threading.Lock] = weakref.WeakValueDictionary() +_carrier_locks_guard = threading.Lock() + + +def carrier_thread_lock(path) -> threading.Lock: + """Return the process-local guard paired with the carrier's file lock.""" + key = str(path) + with _carrier_locks_guard: + lock = _carrier_locks.get(key) + if lock is None: + lock = threading.Lock() + _carrier_locks[key] = lock + return lock + def configure(conf) -> None: """Load the remote-reference policy from the server configuration.""" @@ -99,28 +117,37 @@ def _normalize_allowed_host(value: str) -> str: return f"{host}:{port}" if port is not None else host -def raw_carrier(path): +def raw_carrier(path, mode="r", *, locking=False): """Open a local Blosc2 carrier without dispatching its B2 object.""" - kwargs = {"dparams": blosc2.DParams(nthreads=1)} - return blosc2.blosc2_ext.open(str(path), "r", 0, **kwargs) + kwargs = {"dparams": blosc2.DParams(nthreads=1), "locking": locking} + return blosc2.blosc2_ext.open(str(path), mode, 0, **kwargs) def inspect(path): """Return ``(raw carrier, payload)`` for a RemoteProxy, otherwise ``None``.""" if not hasattr(blosc2, "RemoteProxy"): return None - try: - carrier = raw_carrier(path) - except (RuntimeError, ValueError): - return None - schunk = getattr(carrier, "schunk", carrier) - marker = schunk.meta.get("b2o") - if not isinstance(marker, dict) or marker.get("kind") != "remote_proxy": - return None - payload = schunk.vlmeta.get("b2o") - if not isinstance(payload, dict): - raise RemoteProxyDenied("RemoteProxy carrier has no valid payload") - return carrier, payload + with carrier_thread_lock(path): + try: + carrier = raw_carrier(path) + except (RuntimeError, ValueError): + return None + schunk = getattr(carrier, "schunk", carrier) + marker = schunk.meta.get("b2o") + if not isinstance(marker, dict) or marker.get("kind") != "remote_proxy": + return None + # Only RemoteProxy carriers need a sidecar lock. Reopen after + # discrimination so inspecting ordinary datasets has no filesystem + # side effect, then re-read the marker and payload under that lock. + carrier = raw_carrier(path, locking=True) + schunk = getattr(carrier, "schunk", carrier) + marker = schunk.meta.get("b2o") + if not isinstance(marker, dict) or marker.get("kind") != "remote_proxy": + return None + payload = schunk.vlmeta.get("b2o") + if not isinstance(payload, dict): + raise RemoteProxyDenied("RemoteProxy carrier has no valid payload") + return carrier, payload def guard_embedded(path) -> None: @@ -159,7 +186,7 @@ def _contains_remote_reference(value) -> bool: def is_metadata(meta) -> bool: - """Whether an api/info model describes a reference-only carrier.""" + """Whether an api/info model describes a RemoteProxy carrier.""" vlmeta = getattr(getattr(meta, "schunk", None), "vlmeta", None) or {} payload = vlmeta.get("b2o") return isinstance(payload, dict) and payload.get("kind") == "remote_proxy" @@ -168,12 +195,20 @@ def is_metadata(meta) -> bool: def _validated_source(payload: dict) -> str: if not policy.enabled: raise RemoteProxyDenied("RemoteProxy resolution is disabled by server policy") - if set(payload) != {"kind", "version", "source", "cache_policy"}: + if set(payload) != {"kind", "version", "source", "cache_policy", "max_cache_bytes"}: raise RemoteProxyDenied("RemoteProxy payload contains unsupported fields") if payload.get("kind") != "remote_proxy" or payload.get("version") != 1: raise RemoteProxyDenied("unsupported RemoteProxy payload") - if payload.get("cache_policy") != "none": - raise RemoteProxyDenied("server RemoteProxy carriers must use cache policy 'none'") + cache_policy = payload.get("cache_policy") + max_cache_bytes = payload.get("max_cache_bytes") + if cache_policy == "none": + if max_cache_bytes is not None: + raise RemoteProxyDenied("RemoteProxy cache policy 'none' cannot have max_cache_bytes") + elif cache_policy == "disk": + if isinstance(max_cache_bytes, bool) or not isinstance(max_cache_bytes, int) or max_cache_bytes <= 0: + raise RemoteProxyDenied("RemoteProxy cache policy 'disk' requires positive max_cache_bytes") + else: + raise RemoteProxyDenied("server RemoteProxy supports only cache policies 'none' and 'disk'") source = payload.get("source") if not isinstance(source, dict) or set(source) != {"kind", "version", "urlpath"}: raise RemoteProxyDenied("server RemoteProxy supports only a versioned fsspec URL source") @@ -260,7 +295,7 @@ async def get_client(**kwargs): def resolve(carrier, payload): - """Resolve one allowed carrier as a no-retention remote array.""" + """Resolve one allowed carrier as a policy-limited remote array.""" url = _validated_source(payload) parsed = urlsplit(url) host = parsed.hostname.encode("idna").decode("ascii").lower() @@ -292,19 +327,100 @@ def resolve(carrier, payload): raise RemoteProxyDenied( f"RemoteProxy chunk count exceeds the configured limit of {policy.max_chunks}" ) - return ServerRemoteProxy(source, expected) + return ServerRemoteProxy(source, expected, carrier, payload) class ServerRemoteProxy: - """Operation-scoped assembly over a server-authorized remote source.""" + """Authorized remote source backed by its own carrier cache.""" - def __init__(self, source, geometry): + def __init__(self, source, geometry, carrier, payload): self.src = source self.shape, self.dtype, self.chunks, self.blocks = geometry self.cparams = source.cparams + self.path = carrier.schunk.urlpath + self.cache_policy = payload["cache_policy"] + self.max_cache_bytes = payload["max_cache_bytes"] + + def current_cache_bytes(self) -> int: + if self.cache_policy != "disk": + return 0 + with carrier_thread_lock(self.path): + carrier = raw_carrier(self.path, locking=True) + with carrier.schunk.holding_lock(): + sizes = carrier.schunk.vlmeta.get("proxy-cache-sizes", {}) + if not isinstance(sizes, dict): + return 0 + return sum(size for size in sizes.values() if isinstance(size, int) and size >= 0) + + def _backend(self, cache_limit=None, *, carrier=None): + if self.cache_policy != "disk" or cache_limit == 0: + return blosc2.Proxy(self.src, _refresh_source=False) + carrier = raw_carrier(self.path, mode="a", locking=True) if carrier is None else carrier + limit = self.max_cache_bytes if cache_limit is None else min(self.max_cache_bytes, cache_limit) + return blosc2.Proxy( + self.src, + _cache=carrier, + _refresh_source=False, + _max_cache_bytes=limit, + ) + + def read(self, item, *, cache_limit=None): + if self.cache_policy != "disk" or cache_limit == 0: + return self._backend(cache_limit)[item] + with carrier_thread_lock(self.path): + carrier = raw_carrier(self.path, mode="a", locking=True) + with carrier.schunk.holding_lock(): + backend = self._backend(cache_limit, carrier=carrier) + return backend[item] def __getitem__(self, item): - return blosc2.Proxy(self.src, _refresh_source=False)[item] + return self.read(item) + + def get_chunk(self, nchunk, *, cache_limit=None): + if self.cache_policy != "disk" or cache_limit == 0: + return self.src.get_chunk(nchunk) + item = tuple( + slice(coord * chunk, min((coord + 1) * chunk, size)) + for coord, chunk, size in zip( + np.unravel_index( + nchunk, + tuple( + math.ceil(size / chunk) for size, chunk in zip(self.shape, self.chunks, strict=True) + ), + ), + self.chunks, + self.shape, + strict=True, + ) + ) + with carrier_thread_lock(self.path): + carrier = raw_carrier(self.path, mode="a", locking=True) + with carrier.schunk.holding_lock(): + backend = self._backend(cache_limit, carrier=carrier) + backend.fetch(item) + chunk = backend.schunk.get_chunk(nchunk) + backend._enforce_cache_limit(item) + return chunk + + +def cold_cframe(carrier, payload) -> bytes: + """Return a cache-free carrier without resolving or mutating its source.""" + cold = make_b2object_carrier( + "remote_proxy", + carrier.shape, + carrier.dtype, + chunks=carrier.chunks, + blocks=carrier.blocks, + cparams=carrier.cparams, + ) + write_b2object_payload(cold, payload) + return cold.to_cframe() + - def get_chunk(self, nchunk): - return self.src.get_chunk(nchunk) +def export_cframe(carrier, payload, *, include_cache: bool) -> bytes: + """Snapshot a warm or cold carrier while excluding concurrent mutations.""" + path = carrier.schunk.urlpath + with carrier_thread_lock(path), carrier.schunk.holding_lock(): + if include_cache: + return carrier.to_cframe() + return cold_cframe(carrier, payload) diff --git a/caterva2/services/server.py b/caterva2/services/server.py index 86d8d32e..7a2d4f0b 100644 --- a/caterva2/services/server.py +++ b/caterva2/services/server.py @@ -137,7 +137,10 @@ def guess_type(path): def get_disk_usage(): exclude = {"db.json", "db.sqlite"} - return sum(path.stat().st_size for path, _ in srv_utils.walk_files(settings.statedir, exclude=exclude)) + return sum( + path.stat().st_size + for path, _ in srv_utils.walk_files(settings.statedir, exclude=exclude, include_internal=True) + ) DISK_USAGE_TTL = 10.0 @@ -172,6 +175,33 @@ def account_chunk_written(nbytes: int) -> None: _disk_usage["written"] += nbytes +def remote_proxy_cache_limit(proxy: remote_proxy.ServerRemoteProxy) -> int | None: + """Return the per-carrier cache bound after applying remaining customer quota.""" + if proxy.cache_policy != "disk": + return 0 + if not settings.quota: + return proxy.max_cache_bytes + retained = proxy.current_cache_bytes() + available = max(0, settings.quota - get_disk_usage_written(0)) + return min(proxy.max_cache_bytes, retained + available) + + +async def read_remote_proxy(proxy, item, abspath): + """Read one remote selection while serializing and accounting cache mutation.""" + lock = dataset_lock(abspath) + async with lock: + before = abspath.stat().st_size + cache_limit = remote_proxy_cache_limit(proxy) + data = await concurrency.run_in_threadpool( + lambda: blosc2.asarray(proxy.read(item, cache_limit=cache_limit)).to_cframe() + ) + if settings.quota: + growth = max(0, abspath.stat().st_size - before) + if growth: + account_chunk_written(growth) + return data + + def truncate_path(path, size=35): """ Smart truncation of a long path for display. @@ -615,6 +645,10 @@ async def get_info( if isinstance(meta, models.Metadata): if remote_proxy.is_metadata(meta): meta.accept_ranges = "none" + # Cache bookkeeping contains binary bitmaps and source stamps that + # are neither JSON metadata nor part of the public proxy contract. + # Expose only the portable descriptor through api/info. + meta.schunk.vlmeta = {"b2o": meta.schunk.vlmeta["b2o"]} elif not srv_utils.is_hdf5_proxy_meta(meta): meta.accept_ranges = "bytes" return meta @@ -1133,7 +1167,12 @@ async def fetch_data( # Off the event loop: bounded by `MAX_FETCH_COORDS` but not small, and # a gather that ran here would stall every other request for its # duration -- it reads, materializes and serializes, all blocking - data = await concurrency.run_in_threadpool(lambda: blosc2.asarray(array[indices]).to_cframe()) + if isinstance(array, remote_proxy.ServerRemoteProxy): + data = await read_remote_proxy(array, indices, abspath) + else: + data = await concurrency.run_in_threadpool( + lambda: blosc2.asarray(array[indices]).to_cframe() + ) except (IndexError, ValueError) as exc: srv_utils.raise_bad_request(str(exc)) elif isinstance(array, blosc2.CTable): @@ -1151,9 +1190,7 @@ async def fetch_data( data = blosc2.asarray(data) data = data.to_cframe() elif isinstance(array, remote_proxy.ServerRemoteProxy): - data = await concurrency.run_in_threadpool( - lambda: blosc2.asarray(array[() if slice_ is None else slice_]).to_cframe() - ) + data = await read_remote_proxy(array, () if slice_ is None else slice_, abspath) elif isinstance(array, blosc2.NDArray): # Using NDArray.slice() allows a fast path when it is aligned with the chunks # As we are going to serialize the slice right away, it is not clear in which @@ -1252,6 +1289,7 @@ async def post_fetch_data( async def download_data( path: pathlib.Path, user: db.User = Depends(optional_user), + include_cache: bool = True, accept_encoding: str | None = fastapi.Header(None), range_header: str | None = fastapi.Header(None, alias="Range"), ): @@ -1284,7 +1322,7 @@ async def download_data( decompress = accept_encoding != "blosc2" # Read before creating the response: a bad path must 404 up front, not # abort the stream after the 200 headers already went out. - content = await get_file_content(path, user, decompress=decompress) + content = await get_file_content(path, user, decompress=decompress, include_cache=include_cache) srv_utils.refuse_range(range_header, path) async def downloader(): @@ -1382,6 +1420,16 @@ async def get_chunk( if isinstance(container, blosc2.LazyArray): # In case we do, this would have to be changed. chunk = container.get_chunk(nchunk) + elif isinstance(container, remote_proxy.ServerRemoteProxy): + before = abspath.stat().st_size + cache_limit = remote_proxy_cache_limit(container) + chunk = await concurrency.run_in_threadpool( + lambda: container.get_chunk(nchunk, cache_limit=cache_limit) + ) + if settings.quota: + growth = max(0, abspath.stat().st_size - before) + if growth: + account_chunk_written(growth) else: schunk = getattr(container, "schunk", container) chunk = schunk.get_chunk(nchunk) @@ -2434,12 +2482,12 @@ async def remove( # Try to unlink the file. NotADirectoryError: a path descending into a # container file (e.g. foo.h5/g) names no real file of its own. try: - abspath.unlink() + srv_utils.unlink_with_b2lock(abspath) except (FileNotFoundError, NotADirectoryError): # Try adding a .b2 extension abspath = abspath.with_suffix(abspath.suffix + ".b2") try: - abspath.unlink() + srv_utils.unlink_with_b2lock(abspath) except (FileNotFoundError, NotADirectoryError) as exc: raise fastapi.HTTPException( status_code=404, # not found @@ -3884,7 +3932,7 @@ async def htmx_delete( if not abspath.exists(): return fastapi.HTTPException(status_code=404) - abspath.unlink() + srv_utils.unlink_with_b2lock(abspath) # Redirect to home url = make_url(request, "html_home") @@ -3896,7 +3944,7 @@ async def get_container(path, user): return open_b2(abspath, path) -async def get_file_content(path, user, decompress=True): +async def get_file_content(path, user, decompress=True, include_cache=True): """ This helper function returns the contents of the file at the given path, as a byte string (if the given user has acces to it). @@ -3916,6 +3964,16 @@ async def get_file_content(path, user, decompress=True): abspath = get_abspath(path, user) suffix = abspath.suffix + if suffix in {".b2frame", ".b2nd"}: + reference = remote_proxy.inspect(abspath) + if reference is not None: + lock = dataset_lock(abspath) + async with lock: + carrier, payload = remote_proxy.inspect(abspath) + return await concurrency.run_in_threadpool( + lambda: remote_proxy.export_cframe(carrier, payload, include_cache=include_cache) + ) + if suffix == ".b2": # Blosc2 compressed files are decompressed container = open_b2(abspath, path) diff --git a/caterva2/services/srv_utils.py b/caterva2/services/srv_utils.py index 304e372c..1f58c692 100644 --- a/caterva2/services/srv_utils.py +++ b/caterva2/services/srv_utils.py @@ -630,7 +630,7 @@ def iterdir(root): yield path, relpath -def walk_files(root, exclude=None): +def walk_files(root, exclude=None, *, include_internal=False): if exclude is None: exclude = set() @@ -638,10 +638,18 @@ def walk_files(root, exclude=None): for path in root.glob("**/*"): if path.is_file(): relpath = path.relative_to(root) + if not include_internal and path.name.endswith(".b2lock"): + continue if str(relpath) not in exclude: yield path, relpath +def unlink_with_b2lock(path): + """Remove a file and the advisory lock sidecar belonging to it.""" + path.unlink() + path.with_name(path.name + ".b2lock").unlink(missing_ok=True) + + # # HTTP server helpers # diff --git a/caterva2/tests/test_api.py b/caterva2/tests/test_api.py index f71a7d67..3079d92e 100644 --- a/caterva2/tests/test_api.py +++ b/caterva2/tests/test_api.py @@ -158,6 +158,41 @@ def test_remote_proxy_is_discovered_but_resolution_is_disabled(client): path.unlink(missing_ok=True) +def test_remote_proxy_download_can_omit_cache(client, tmp_path): + source = blosc2.arange(20, dtype=np.int32, chunks=(10,), blocks=(5,)) + fsspec.filesystem("memory").pipe_file("caterva2-download-reference.b2nd", source.to_cframe()) + reference = blosc2.RemoteProxy( + "memory://caterva2-download-reference.b2nd", + cache_policy=blosc2.CachePolicy.DISK, + cache_path=tmp_path / "warm-reference.b2nd", + max_cache_bytes=1_000_000, + ) + np.testing.assert_array_equal(reference[:10], np.arange(10, dtype=np.int32)) + path = pathlib.Path(TEST_STATE_DIR) / "server/public/download-reference.b2nd" + reference.save(path) + + try: + dataset = client.get(f"@public/{path.name}") + warm_url = dataset.get_download_url() + cold_url = dataset.get_download_url(include_cache=False) + assert warm_url == f"{client.urlbase}/api/download/@public/{path.name}" + assert cold_url == f"{warm_url}?include_cache=false" + + warm_response = httpx.get(warm_url) + cold_response = httpx.get(cold_url) + warm_response.raise_for_status() + cold_response.raise_for_status() + warm = blosc2.ndarray_from_cframe(warm_response.content) + cold = blosc2.ndarray_from_cframe(cold_response.content) + assert warm.schunk.vlmeta.get("proxy-cache-sizes") + assert not cold.schunk.vlmeta.get("proxy-cache-sizes", {}) + assert cold.schunk.vlmeta["b2o"] == warm.schunk.vlmeta["b2o"] + assert len(cold_response.content) < len(warm_response.content) + assert not any(name.endswith(".b2lock") for name in client.get_list("@public")) + finally: + path.unlink(missing_ok=True) + + def test_remote_proxy_hidden_in_expression_is_denied_before_open(client): source = blosc2.arange(20, dtype=np.int32, chunks=(10,), blocks=(5,)) fsspec.filesystem("memory").pipe_file("caterva2-embedded-reference.b2nd", source.to_cframe()) diff --git a/caterva2/tests/test_remote_proxy.py b/caterva2/tests/test_remote_proxy.py index 73f4606b..9b3fa036 100644 --- a/caterva2/tests/test_remote_proxy.py +++ b/caterva2/tests/test_remote_proxy.py @@ -8,9 +8,11 @@ ############################################################################### import asyncio +import concurrent.futures import socket import blosc2 +import fsspec import numpy as np import pytest @@ -25,12 +27,13 @@ def get(self, key, default=None): return self.values.get(key, default) -def _payload(url): +def _payload(url, *, cache_policy="none", max_cache_bytes=None): return { "kind": "remote_proxy", "version": 1, "source": {"kind": "fsspec", "version": 1, "urlpath": url}, - "cache_policy": "none", + "cache_policy": cache_policy, + "max_cache_bytes": max_cache_bytes, } @@ -76,6 +79,27 @@ def test_configured_https_destination_is_accepted(): assert remote_proxy._validated_source(_payload("https://data.example:8443/array.b2nd")) +@pytest.mark.parametrize( + ("payload", "match"), + [ + (_payload("https://data.example/a.b2nd", cache_policy="memory"), "only cache policies"), + ( + _payload("https://data.example/a.b2nd", cache_policy="none", max_cache_bytes=1), + "cannot have max_cache_bytes", + ), + (_payload("https://data.example/a.b2nd", cache_policy="disk"), "requires positive"), + ( + _payload("https://data.example/a.b2nd", cache_policy="disk", max_cache_bytes=True), + "requires positive", + ), + ], +) +def test_cache_specification_is_strict(payload, match): + remote_proxy.policy = remote_proxy.Policy(enabled=True, allowed_hosts=("data.example",)) + with pytest.raises(remote_proxy.RemoteProxyDenied, match=match): + remote_proxy._validated_source(payload) + + def test_private_resolution_is_rejected(monkeypatch): monkeypatch.setattr( socket, @@ -107,6 +131,9 @@ class Carrier: chunks = (5,) blocks = (5,) + class schunk: + urlpath = "/tmp/reference.b2nd" + class Source: shape = Carrier.shape dtype = Carrier.dtype @@ -139,6 +166,111 @@ def fake_source(url, max_concurrency, *, _filesystem): } +def test_cold_cframe_preserves_specification_but_not_cached_chunks(tmp_path): + source = blosc2.asarray(np.arange(20), chunks=(10,), blocks=(5,)) + source_url = "memory://cold-cframe-source.b2nd" + fsspec.filesystem("memory").pipe_file("cold-cframe-source.b2nd", source.to_cframe()) + carrier_path = tmp_path / "proxy.b2nd" + proxy = blosc2.RemoteProxy( + source_url, + cache_policy=blosc2.CachePolicy.DISK, + cache_path=carrier_path, + max_cache_bytes=1_000_000, + ) + np.testing.assert_array_equal(proxy[:10], np.arange(10)) + + carrier, payload = remote_proxy.inspect(carrier_path) + assert carrier.schunk.vlmeta.get("proxy-cache-sizes") + cold = blosc2.ndarray_from_cframe(remote_proxy.cold_cframe(carrier, payload)) + + assert cold.schunk.vlmeta["b2o"] == payload + assert not cold.schunk.vlmeta.get("proxy-cache-sizes", {}) + assert cold.to_cframe() != carrier.to_cframe() + + +def _server_proxy(tmp_path, name="server-cache"): + data = np.arange(40, dtype=np.int32) + source = blosc2.asarray(data, chunks=(10,), blocks=(5,)) + source_url = f"memory://{name}-source.b2nd" + fsspec.filesystem("memory").pipe_file(f"{name}-source.b2nd", source.to_cframe()) + carrier_path = tmp_path / f"{name}.b2nd" + creator = blosc2.RemoteProxy( + source_url, + cache_policy=blosc2.CachePolicy.DISK, + cache_path=carrier_path, + max_cache_bytes=1_000_000, + ) + carrier, payload = remote_proxy.inspect(carrier_path) + geometry = (creator.shape, creator.dtype, creator.chunks, creator.blocks) + return remote_proxy.ServerRemoteProxy(creator.src, geometry, carrier, payload), data, carrier_path + + +def test_server_proxy_reuses_its_carrier_cache(tmp_path): + proxy, data, carrier_path = _server_proxy(tmp_path) + proxy.src.traffic.reset() + np.testing.assert_array_equal(proxy.read(slice(0, 10)), data[:10]) + assert proxy.src.traffic.requests > 0 + + carrier, payload = remote_proxy.inspect(carrier_path) + fresh_source = blosc2.FsspecNDSource(payload["source"]["urlpath"]) + reopened = remote_proxy.ServerRemoteProxy( + fresh_source, + (proxy.shape, proxy.dtype, proxy.chunks, proxy.blocks), + carrier, + payload, + ) + fresh_source.traffic.reset() + np.testing.assert_array_equal(reopened.read(slice(0, 10)), data[:10]) + assert fresh_source.traffic.requests == 0 + + +def test_zero_effective_quota_reads_without_retaining(tmp_path): + proxy, data, carrier_path = _server_proxy(tmp_path, "zero-quota") + before = carrier_path.read_bytes() + np.testing.assert_array_equal(proxy.read(slice(0, 10), cache_limit=0), data[:10]) + assert carrier_path.read_bytes() == before + + +def test_customer_quota_reduces_the_proxy_cache_limit(monkeypatch): + monkeypatch.setenv("CATERVA2_SECRET", "remote-proxy-test-secret") + from caterva2.services import server + + class Proxy: + cache_policy = "disk" + max_cache_bytes = 500 + + @staticmethod + def current_cache_bytes(): + return 200 + + monkeypatch.setattr(server.settings, "quota", 1_000) + monkeypatch.setattr(server, "get_disk_usage_written", lambda pending: 900) + assert server.remote_proxy_cache_limit(Proxy()) == 300 + + monkeypatch.setattr(server, "get_disk_usage_written", lambda pending: 1_000) + assert server.remote_proxy_cache_limit(Proxy()) == 200 + + +def test_concurrent_server_proxy_fills_do_not_corrupt_carrier(tmp_path): + first, data, carrier_path = _server_proxy(tmp_path, "concurrent") + carrier, payload = remote_proxy.inspect(carrier_path) + second_source = blosc2.FsspecNDSource(payload["source"]["urlpath"]) + second = remote_proxy.ServerRemoteProxy( + second_source, + (first.shape, first.dtype, first.chunks, first.blocks), + carrier, + payload, + ) + with concurrent.futures.ThreadPoolExecutor(max_workers=2) as pool: + left = pool.submit(first.read, slice(0, 20)) + right = pool.submit(second.read, slice(20, 40)) + np.testing.assert_array_equal(left.result(timeout=10), data[:20]) + np.testing.assert_array_equal(right.result(timeout=10), data[20:]) + + reopened = blosc2.open(carrier_path, mode="r") + np.testing.assert_array_equal(reopened[:], data) + + def test_https_filesystem_disables_redirects_and_pins_resolution(): remote_proxy.policy = remote_proxy.Policy(timeout=7) fs = remote_proxy._https_filesystem("data.example", ("93.184.216.34",)) diff --git a/doc/utilities/cat2-server.md b/doc/utilities/cat2-server.md index f45df19b..27ada0c7 100644 --- a/doc/utilities/cat2-server.md +++ b/doc/utilities/cat2-server.md @@ -34,10 +34,12 @@ And then simply run `cat2-server` to start it on all network interfaces on port ## Remote reference policy -A persisted `blosc2.RemoteProxy` is a small B2ND carrier that asks Caterva2 to -read another dataset. Caterva2 can inspect and report the carrier's stored -shape, dtype, chunk, and block metadata without contacting that source. -Outbound resolution is disabled by default. +A persisted `blosc2.RemoteProxy` is a B2ND carrier that asks Caterva2 to read +another dataset. With its disk cache enabled, fetched compressed chunks are +retained inside that same carrier up to the proxy's finite `max_cache_bytes`. +Caterva2 can inspect and report its stored shape, dtype, chunk, block, and proxy +metadata without contacting the source. Outbound resolution is disabled by +default. The initial opt-in backend supports public, credential-free HTTPS sources: @@ -60,12 +62,17 @@ Source URLs containing user information, query parameters, or fragments are also rejected. These checks are applied by the server even when the carrier was created by a client that performed its own validation. -The limits bound each upstream request's time, source rank, logical -uncompressed size, chunk count, and concurrent range fetches. Set them for the -capacity of the installation; they are not inferred from untrusted carrier -metadata. +The limits validate the remote array's structure and bound connection time and +concurrent range fetches. They do not impose a network-work budget on each API +request. The proxy's own cache limit bounds its retained compressed payload, +while automatic carrier growth is also charged to the virtual server's existing +shared `quota`. If no quota remains, reads still succeed but misses are not +retained. -S3, private-source credential selection, and remote references embedded inside -persisted expressions are not enabled yet. Server credentials must eventually -be selected from an administrator-controlled destination mapping and must never -be accepted from a carrier. +Public S3 objects are supported through credential-free HTTPS object URLs. +Native `s3://` resolution, private-source credentials, and remote references +embedded inside persisted expressions are not enabled. + +Physical downloads include valid warm proxy chunks by default. Clients can pass +`include_cache=false` to download a cold carrier without mutating the hosted +proxy. Logical `api/fetch` requests continue to return array data. From a1287b9266b82a1b3d0c3a3693c299bc4730e443 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sat, 5 Sep 2026 14:37:59 +0200 Subject: [PATCH 03/20] Execute MEMORY remote proxies without caching and support unbounded DISK cache --- caterva2-server.sample.toml | 5 +- caterva2/services/remote_proxy.py | 68 ++++- caterva2/services/server.py | 5 +- caterva2/tests/test_api.py | 181 ++++++++++++- caterva2/tests/test_remote_proxy.py | 403 ++++++++++++++++++++++++++-- doc/utilities/cat2-server.md | 10 +- 6 files changed, 624 insertions(+), 48 deletions(-) diff --git a/caterva2-server.sample.toml b/caterva2-server.sample.toml index 561c3bb6..f6ad03cf 100644 --- a/caterva2-server.sample.toml +++ b/caterva2-server.sample.toml @@ -34,8 +34,9 @@ register = true # allow users to register # requests by default. The first opt-in backend is credential-free HTTPS. # Every destination must be listed exactly; redirects, URL queries, private or # otherwise non-public destination addresses, and embedded expression -# references are refused. A proxy may cache chunks inside its own carrier up to -# its persisted max_cache_bytes; that growth counts against this server's quota. +# references are refused. MEMORY carriers are accepted under the same source +# policy but execute without retained caching (same as NONE). DISK proxies cache +# chunks inside their carrier up to persisted max_cache_bytes (or unbounded if None), subject to server quota. # [server.remote_proxy] # enabled = true # allowed_hosts = ["datasets.example.org", "objects.example.org:8443"] diff --git a/caterva2/services/remote_proxy.py b/caterva2/services/remote_proxy.py index d862238f..c8752af4 100644 --- a/caterva2/services/remote_proxy.py +++ b/caterva2/services/remote_proxy.py @@ -205,10 +205,21 @@ def _validated_source(payload: dict) -> str: if max_cache_bytes is not None: raise RemoteProxyDenied("RemoteProxy cache policy 'none' cannot have max_cache_bytes") elif cache_policy == "disk": + if max_cache_bytes is not None and ( + isinstance(max_cache_bytes, bool) or not isinstance(max_cache_bytes, int) or max_cache_bytes <= 0 + ): + raise RemoteProxyDenied( + "RemoteProxy cache policy 'disk' requires positive max_cache_bytes or None" + ) + elif cache_policy == "memory": if isinstance(max_cache_bytes, bool) or not isinstance(max_cache_bytes, int) or max_cache_bytes <= 0: - raise RemoteProxyDenied("RemoteProxy cache policy 'disk' requires positive max_cache_bytes") + raise RemoteProxyDenied( + f"RemoteProxy cache policy {cache_policy!r} requires positive max_cache_bytes" + ) else: - raise RemoteProxyDenied("server RemoteProxy supports only cache policies 'none' and 'disk'") + raise RemoteProxyDenied( + "server RemoteProxy supports only cache policies 'none', 'memory', and 'disk'" + ) source = payload.get("source") if not isinstance(source, dict) or set(source) != {"kind", "version", "urlpath"}: raise RemoteProxyDenied("server RemoteProxy supports only a versioned fsspec URL source") @@ -330,16 +341,48 @@ def resolve(carrier, payload): return ServerRemoteProxy(source, expected, carrier, payload) +def _effective_cache_policy(requested: str) -> str: + """Return the runtime cache policy executed by Caterva2 ('none' or 'disk').""" + if requested in {"none", "memory"}: + return "none" + if requested == "disk": + return "disk" + raise ValueError(f"unknown cache policy: {requested!r}") + + class ServerRemoteProxy: - """Authorized remote source backed by its own carrier cache.""" + """Authorized remote source backed by its own carrier cache. + + Attributes + ---------- + requested_cache_policy : str + The cache policy requested in the carrier payload ('none', 'memory', or 'disk'). + requested_max_cache_bytes : int | None + The cache limit requested in the carrier payload. + cache_policy : str + The effective runtime cache policy executed by Caterva2 ('none' or 'disk'). + Persisted 'memory' carriers are executed using the same no-retention path + as 'none', avoiding unmanaged server RAM caching across requests. + effective_cache_policy : str + Read-only alias for `cache_policy`. + max_cache_bytes : int | None + The effective runtime cache limit (positive integer for 'disk', None for 'none'). + """ def __init__(self, source, geometry, carrier, payload): self.src = source self.shape, self.dtype, self.chunks, self.blocks = geometry self.cparams = source.cparams self.path = carrier.schunk.urlpath - self.cache_policy = payload["cache_policy"] - self.max_cache_bytes = payload["max_cache_bytes"] + self.requested_cache_policy = payload["cache_policy"] + self.requested_max_cache_bytes = payload["max_cache_bytes"] + self.cache_policy = _effective_cache_policy(self.requested_cache_policy) + self.max_cache_bytes = self.requested_max_cache_bytes if self.cache_policy == "disk" else None + + @property + def effective_cache_policy(self) -> str: + """Effective cache policy executed by the server runtime ('none' or 'disk').""" + return self.cache_policy def current_cache_bytes(self) -> int: if self.cache_policy != "disk": @@ -347,16 +390,21 @@ def current_cache_bytes(self) -> int: with carrier_thread_lock(self.path): carrier = raw_carrier(self.path, locking=True) with carrier.schunk.holding_lock(): - sizes = carrier.schunk.vlmeta.get("proxy-cache-sizes", {}) - if not isinstance(sizes, dict): - return 0 - return sum(size for size in sizes.values() if isinstance(size, int) and size >= 0) + sizes = carrier.schunk.vlmeta.get("proxy-cache-sizes") + if isinstance(sizes, dict): + return sum(size for size in sizes.values() if isinstance(size, int) and size >= 0) + return carrier.schunk.cbytes def _backend(self, cache_limit=None, *, carrier=None): if self.cache_policy != "disk" or cache_limit == 0: return blosc2.Proxy(self.src, _refresh_source=False) carrier = raw_carrier(self.path, mode="a", locking=True) if carrier is None else carrier - limit = self.max_cache_bytes if cache_limit is None else min(self.max_cache_bytes, cache_limit) + if cache_limit is None: + limit = self.max_cache_bytes + elif self.max_cache_bytes is None: + limit = cache_limit + else: + limit = min(self.max_cache_bytes, cache_limit) return blosc2.Proxy( self.src, _cache=carrier, diff --git a/caterva2/services/server.py b/caterva2/services/server.py index 7a2d4f0b..bc50caab 100644 --- a/caterva2/services/server.py +++ b/caterva2/services/server.py @@ -183,7 +183,10 @@ def remote_proxy_cache_limit(proxy: remote_proxy.ServerRemoteProxy) -> int | Non return proxy.max_cache_bytes retained = proxy.current_cache_bytes() available = max(0, settings.quota - get_disk_usage_written(0)) - return min(proxy.max_cache_bytes, retained + available) + quota_bound = retained + available + if proxy.max_cache_bytes is None: + return quota_bound + return min(proxy.max_cache_bytes, quota_bound) async def read_remote_proxy(proxy, item, abspath): diff --git a/caterva2/tests/test_api.py b/caterva2/tests/test_api.py index 3079d92e..2143ccfd 100644 --- a/caterva2/tests/test_api.py +++ b/caterva2/tests/test_api.py @@ -133,11 +133,35 @@ def test_dataset_info(client, fill_public): assert data.chunks == tuple(info["chunks"]) -def test_remote_proxy_is_discovered_but_resolution_is_disabled(client): +@pytest.mark.parametrize( + ("cache_policy", "max_cache_bytes"), + [ + (blosc2.CachePolicy.NONE, None), + (blosc2.CachePolicy.MEMORY, 1_000_000), + (blosc2.CachePolicy.DISK, 1_000_000), + (blosc2.CachePolicy.DISK, None), + ], +) +def test_remote_proxy_is_discovered_but_resolution_is_disabled( + client, cache_policy, max_cache_bytes, tmp_path +): + tag = f"{cache_policy.value}-{'unlimited' if max_cache_bytes is None else 'bounded'}" source = blosc2.arange(20, dtype=np.int32, chunks=(10,), blocks=(5,)) - fsspec.filesystem("memory").pipe_file("caterva2-disabled-reference.b2nd", source.to_cframe()) - reference = blosc2.RemoteProxy("memory://caterva2-disabled-reference.b2nd") - path = pathlib.Path(TEST_STATE_DIR) / "server/public/disabled-reference.b2nd" + name = f"caterva2-disabled-reference-{tag}.b2nd" + fsspec.filesystem("memory").pipe_file(name, source.to_cframe()) + carrier_path = tmp_path / f"carrier-{tag}.b2nd" if cache_policy == blosc2.CachePolicy.DISK else None + kwargs = {} + if cache_policy is not blosc2.CachePolicy.NONE and max_cache_bytes is not None: + kwargs["max_cache_bytes"] = max_cache_bytes + elif cache_policy is blosc2.CachePolicy.DISK and max_cache_bytes is None: + kwargs["max_cache_bytes"] = None + reference = blosc2.RemoteProxy( + f"memory://{name}", + cache_policy=cache_policy, + cache_path=carrier_path, + **kwargs, + ) + path = pathlib.Path(TEST_STATE_DIR) / f"server/public/disabled-reference-{tag}.b2nd" reference.save(path) before = path.read_bytes() @@ -149,6 +173,8 @@ def test_remote_proxy_is_discovered_but_resolution_is_disabled(client): assert info["chunks"] == [10] assert info["blocks"] == [5] assert info["accept_ranges"] == "none" + assert info["schunk"]["vlmeta"]["b2o"]["cache_policy"] == cache_policy.value + assert info["schunk"]["vlmeta"]["b2o"]["max_cache_bytes"] == max_cache_bytes response = httpx.get(f"{client.urlbase}/api/fetch/@public/{path.name}", params={"slice_": "0:2"}) assert response.status_code == 403 @@ -158,17 +184,25 @@ def test_remote_proxy_is_discovered_but_resolution_is_disabled(client): path.unlink(missing_ok=True) -def test_remote_proxy_download_can_omit_cache(client, tmp_path): +@pytest.mark.parametrize("cache_policy", [blosc2.CachePolicy.DISK, blosc2.CachePolicy.MEMORY]) +def test_remote_proxy_download_can_omit_cache(client, tmp_path, cache_policy): source = blosc2.arange(20, dtype=np.int32, chunks=(10,), blocks=(5,)) - fsspec.filesystem("memory").pipe_file("caterva2-download-reference.b2nd", source.to_cframe()) + name = f"caterva2-download-reference-{cache_policy.value}.b2nd" + fsspec.filesystem("memory").pipe_file(name, source.to_cframe()) + carrier_path = ( + tmp_path / f"warm-reference-{cache_policy.value}.b2nd" + if cache_policy == blosc2.CachePolicy.DISK + else None + ) reference = blosc2.RemoteProxy( - "memory://caterva2-download-reference.b2nd", - cache_policy=blosc2.CachePolicy.DISK, - cache_path=tmp_path / "warm-reference.b2nd", + f"memory://{name}", + cache_policy=cache_policy, + cache_path=carrier_path, max_cache_bytes=1_000_000, ) - np.testing.assert_array_equal(reference[:10], np.arange(10, dtype=np.int32)) - path = pathlib.Path(TEST_STATE_DIR) / "server/public/download-reference.b2nd" + if cache_policy == blosc2.CachePolicy.DISK: + np.testing.assert_array_equal(reference[:10], np.arange(10, dtype=np.int32)) + path = pathlib.Path(TEST_STATE_DIR) / f"server/public/download-reference-{cache_policy.value}.b2nd" reference.save(path) try: @@ -184,15 +218,134 @@ def test_remote_proxy_download_can_omit_cache(client, tmp_path): cold_response.raise_for_status() warm = blosc2.ndarray_from_cframe(warm_response.content) cold = blosc2.ndarray_from_cframe(cold_response.content) - assert warm.schunk.vlmeta.get("proxy-cache-sizes") - assert not cold.schunk.vlmeta.get("proxy-cache-sizes", {}) + if cache_policy == blosc2.CachePolicy.DISK: + assert warm.schunk.vlmeta.get("proxy-cache-sizes") + assert not cold.schunk.vlmeta.get("proxy-cache-sizes", {}) + assert len(cold_response.content) < len(warm_response.content) + else: + assert not warm.schunk.vlmeta.get("proxy-cache-sizes", {}) + assert not cold.schunk.vlmeta.get("proxy-cache-sizes", {}) + assert warm.to_cframe() == cold.to_cframe() assert cold.schunk.vlmeta["b2o"] == warm.schunk.vlmeta["b2o"] - assert len(cold_response.content) < len(warm_response.content) + assert cold.schunk.vlmeta["b2o"]["cache_policy"] == cache_policy.value + assert cold.schunk.vlmeta["b2o"]["max_cache_bytes"] == 1_000_000 assert not any(name.endswith(".b2lock") for name in client.get_list("@public")) finally: path.unlink(missing_ok=True) +@pytest.mark.asyncio +async def test_remote_proxy_memory_enabled_resolution_fetch_and_chunk(tmp_path, monkeypatch): + monkeypatch.setenv("CATERVA2_SECRET", "test-secret") + from caterva2.services import remote_proxy, server + + class _Conf: + def __init__(self, values): + self.values = values + + def get(self, key, default=None): + return self.values.get(key, default) + + server.settings.statedir = tmp_path / "statedir" + server.settings.public = server.settings.statedir / "public" + server.settings.public.mkdir(parents=True, exist_ok=True) + + data = np.arange(40, dtype=np.int32) + source = blosc2.asarray(data, chunks=(10,), blocks=(5,)) + url = "https://data.example/api-mem-source.b2nd" + mem_fs = fsspec.filesystem("memory") + mem_fs.pipe_file(url, source.to_cframe()) + mem_fs.pipe_file("dummy.b2nd", source.to_cframe()) + + dummy_proxy = blosc2.RemoteProxy( + "memory://dummy.b2nd", + cache_policy=blosc2.CachePolicy.MEMORY, + max_cache_bytes=500_000, + ) + carrier_path = server.settings.public / "api-mem.b2nd" + dummy_proxy.save(carrier_path) + + carrier = remote_proxy.raw_carrier(carrier_path, mode="a") + payload = dict(carrier.schunk.vlmeta["b2o"]) + payload["source"] = dict(payload["source"]) + payload["source"]["urlpath"] = url + carrier.schunk.vlmeta["b2o"] = payload + + remote_proxy.configure( + _Conf( + { + ".remote_proxy.enabled": True, + ".remote_proxy.allowed_hosts": ["data.example"], + } + ) + ) + + before_bytes = carrier_path.read_bytes() + before_size = carrier_path.stat().st_size + before_mtime = carrier_path.stat().st_mtime_ns + + # Instrument upstream filesystem calls + upstream_reads = [] + orig_cat_file = mem_fs.cat_file + + def traced_cat_file(path, *args, **kwargs): + upstream_reads.append(path) + return orig_cat_file(path, *args, **kwargs) + + mem_fs.cat_file = traced_cat_file + + monkeypatch.setattr(remote_proxy, "_public_addresses", lambda host, port: ("93.184.216.34",)) + monkeypatch.setattr(remote_proxy, "_https_filesystem", lambda host, addr: mem_fs) + + async with ( + server.app.router.lifespan_context(server.app), + httpx.AsyncClient( + transport=httpx.ASGITransport(app=server.app), base_url="http://test" + ) as api_client, + ): + # 1. api/info + info_resp = await api_client.get("/api/info/@public/api-mem.b2nd") + assert info_resp.status_code == 200 + info = info_resp.json() + assert info["accept_ranges"] == "none" + assert info["schunk"]["vlmeta"]["b2o"]["cache_policy"] == "memory" + assert info["schunk"]["vlmeta"]["b2o"]["max_cache_bytes"] == 500_000 + + # 2. First fetch of slice + fetch_resp1 = await api_client.get("/api/fetch/@public/api-mem.b2nd", params={"slice_": "0:10"}) + assert fetch_resp1.status_code == 200 + arr1 = blosc2.ndarray_from_cframe(fetch_resp1.content) + np.testing.assert_array_equal(arr1[:], data[:10]) + reads_after_first = len(upstream_reads) + assert reads_after_first > 0 + + # 3. Second fetch of slice: must perform repeated upstream reads + fetch_resp2 = await api_client.get("/api/fetch/@public/api-mem.b2nd", params={"slice_": "0:10"}) + assert fetch_resp2.status_code == 200 + arr2 = blosc2.ndarray_from_cframe(fetch_resp2.content) + np.testing.assert_array_equal(arr2[:], data[:10]) + assert len(upstream_reads) > reads_after_first + + # 4. Fetch chunk + reads_before_chunk = len(upstream_reads) + chunk_resp1 = await api_client.get("/api/chunk/@public/api-mem.b2nd", params={"nchunk": 0}) + assert chunk_resp1.status_code == 200 + chunk_decomp = np.frombuffer(blosc2.decompress(chunk_resp1.content), dtype=data.dtype) + np.testing.assert_array_equal(chunk_decomp, data[:10]) + assert len(upstream_reads) > reads_before_chunk + + # 5. Second fetch of chunk: must perform repeated upstream reads + reads_after_chunk1 = len(upstream_reads) + chunk_resp2 = await api_client.get("/api/chunk/@public/api-mem.b2nd", params={"nchunk": 0}) + assert chunk_resp2.status_code == 200 + assert len(upstream_reads) > reads_after_chunk1 + + # 6. Verify carrier file was not modified + assert carrier_path.read_bytes() == before_bytes + assert carrier_path.stat().st_size == before_size + assert carrier_path.stat().st_mtime_ns == before_mtime + + def test_remote_proxy_hidden_in_expression_is_denied_before_open(client): source = blosc2.arange(20, dtype=np.int32, chunks=(10,), blocks=(5,)) fsspec.filesystem("memory").pipe_file("caterva2-embedded-reference.b2nd", source.to_cframe()) diff --git a/caterva2/tests/test_remote_proxy.py b/caterva2/tests/test_remote_proxy.py index 9b3fa036..d8187dc5 100644 --- a/caterva2/tests/test_remote_proxy.py +++ b/caterva2/tests/test_remote_proxy.py @@ -44,10 +44,23 @@ def reset_policy(): remote_proxy.policy = previous -def test_resolution_is_default_deny(): +@pytest.mark.parametrize( + ("cache_policy", "max_cache_bytes"), + [ + ("none", None), + ("memory", 268435456), + ("disk", 1_000_000), + ("disk", None), + ], +) +def test_resolution_is_default_deny(cache_policy, max_cache_bytes): remote_proxy.policy = remote_proxy.Policy() with pytest.raises(remote_proxy.RemoteProxyDenied, match="disabled"): - remote_proxy._validated_source(_payload("https://data.example/array.b2nd")) + remote_proxy._validated_source( + _payload( + "https://data.example/array.b2nd", cache_policy=cache_policy, max_cache_bytes=max_cache_bytes + ) + ) @pytest.mark.parametrize( @@ -60,13 +73,33 @@ def test_resolution_is_default_deny(): "https://other.example/array.b2nd", ], ) -def test_enabled_policy_still_rejects_unsafe_destinations(url): +@pytest.mark.parametrize( + ("cache_policy", "max_cache_bytes"), + [ + ("none", None), + ("memory", 268435456), + ("disk", 1_000_000), + ], +) +def test_enabled_policy_still_rejects_unsafe_destinations(url, cache_policy, max_cache_bytes): remote_proxy.policy = remote_proxy.Policy(enabled=True, allowed_hosts=("data.example",)) with pytest.raises(remote_proxy.RemoteProxyDenied): - remote_proxy._validated_source(_payload(url)) + remote_proxy._validated_source( + _payload(url, cache_policy=cache_policy, max_cache_bytes=max_cache_bytes) + ) -def test_configured_https_destination_is_accepted(): +@pytest.mark.parametrize( + ("cache_policy", "max_cache_bytes"), + [ + ("none", None), + ("memory", 268435456), + ("memory", 500_000), + ("disk", 1_000_000), + ("disk", None), + ], +) +def test_configured_https_destination_is_accepted(cache_policy, max_cache_bytes): remote_proxy.configure( _Conf( { @@ -75,23 +108,80 @@ def test_configured_https_destination_is_accepted(): } ) ) - assert remote_proxy._validated_source(_payload("https://data.example/array.b2nd")) - assert remote_proxy._validated_source(_payload("https://data.example:8443/array.b2nd")) + assert remote_proxy._validated_source( + _payload( + "https://data.example/array.b2nd", cache_policy=cache_policy, max_cache_bytes=max_cache_bytes + ) + ) + assert remote_proxy._validated_source( + _payload( + "https://data.example:8443/array.b2nd", + cache_policy=cache_policy, + max_cache_bytes=max_cache_bytes, + ) + ) @pytest.mark.parametrize( ("payload", "match"), [ - (_payload("https://data.example/a.b2nd", cache_policy="memory"), "only cache policies"), + ( + _payload("https://data.example/a.b2nd", cache_policy="unknown", max_cache_bytes=1000), + "only cache policies", + ), ( _payload("https://data.example/a.b2nd", cache_policy="none", max_cache_bytes=1), "cannot have max_cache_bytes", ), - (_payload("https://data.example/a.b2nd", cache_policy="disk"), "requires positive"), ( _payload("https://data.example/a.b2nd", cache_policy="disk", max_cache_bytes=True), "requires positive", ), + ( + _payload("https://data.example/a.b2nd", cache_policy="disk", max_cache_bytes=False), + "requires positive", + ), + ( + _payload("https://data.example/a.b2nd", cache_policy="disk", max_cache_bytes=0), + "requires positive", + ), + ( + _payload("https://data.example/a.b2nd", cache_policy="disk", max_cache_bytes=-10), + "requires positive", + ), + ( + _payload("https://data.example/a.b2nd", cache_policy="disk", max_cache_bytes=10.5), + "requires positive", + ), + ( + _payload("https://data.example/a.b2nd", cache_policy="disk", max_cache_bytes="1000"), + "requires positive", + ), + (_payload("https://data.example/a.b2nd", cache_policy="memory"), "requires positive"), + ( + _payload("https://data.example/a.b2nd", cache_policy="memory", max_cache_bytes=True), + "requires positive", + ), + ( + _payload("https://data.example/a.b2nd", cache_policy="memory", max_cache_bytes=False), + "requires positive", + ), + ( + _payload("https://data.example/a.b2nd", cache_policy="memory", max_cache_bytes=0), + "requires positive", + ), + ( + _payload("https://data.example/a.b2nd", cache_policy="memory", max_cache_bytes=-10), + "requires positive", + ), + ( + _payload("https://data.example/a.b2nd", cache_policy="memory", max_cache_bytes=10.5), + "requires positive", + ), + ( + _payload("https://data.example/a.b2nd", cache_policy="memory", max_cache_bytes="1000"), + "requires positive", + ), ], ) def test_cache_specification_is_strict(payload, match): @@ -124,7 +214,19 @@ def test_embedded_fsspec_reference_is_recognized(): assert remote_proxy._contains_remote_reference(payload) -def test_allowed_source_is_resolved_with_the_secure_filesystem(monkeypatch): +@pytest.mark.parametrize( + ("cache_policy", "max_cache_bytes", "expected_eff_policy", "expected_eff_limit"), + [ + ("none", None, "none", None), + ("memory", 268435456, "none", None), + ("memory", 500_000, "none", None), + ("disk", 1_000_000, "disk", 1_000_000), + ("disk", None, "disk", None), + ], +) +def test_allowed_source_is_resolved_with_the_secure_filesystem( + monkeypatch, cache_policy, max_cache_bytes, expected_eff_policy, expected_eff_limit +): class Carrier: shape = (10,) dtype = np.dtype(np.int32) @@ -157,8 +259,18 @@ def fake_source(url, max_concurrency, *, _filesystem): monkeypatch.setattr(remote_proxy, "_https_filesystem", lambda host, addresses: filesystem) monkeypatch.setattr(blosc2, "FsspecNDSource", fake_source) - resolved = remote_proxy.resolve(Carrier(), _payload("https://data.example/array.b2nd")) + payload = _payload( + "https://data.example/array.b2nd", + cache_policy=cache_policy, + max_cache_bytes=max_cache_bytes, + ) + resolved = remote_proxy.resolve(Carrier(), payload) assert isinstance(resolved, remote_proxy.ServerRemoteProxy) + assert resolved.requested_cache_policy == cache_policy + assert resolved.requested_max_cache_bytes == max_cache_bytes + assert resolved.cache_policy == expected_eff_policy + assert resolved.effective_cache_policy == expected_eff_policy + assert resolved.max_cache_bytes == expected_eff_limit assert seen == { "url": "https://data.example/array.b2nd", "max_concurrency": 3, @@ -188,18 +300,35 @@ def test_cold_cframe_preserves_specification_but_not_cached_chunks(tmp_path): assert cold.to_cframe() != carrier.to_cframe() -def _server_proxy(tmp_path, name="server-cache"): +def _server_proxy(tmp_path, name="server-cache", cache_policy="disk", max_cache_bytes=1_000_000): data = np.arange(40, dtype=np.int32) source = blosc2.asarray(data, chunks=(10,), blocks=(5,)) source_url = f"memory://{name}-source.b2nd" fsspec.filesystem("memory").pipe_file(f"{name}-source.b2nd", source.to_cframe()) carrier_path = tmp_path / f"{name}.b2nd" - creator = blosc2.RemoteProxy( - source_url, - cache_policy=blosc2.CachePolicy.DISK, - cache_path=carrier_path, - max_cache_bytes=1_000_000, - ) + if cache_policy == "disk": + creator = blosc2.RemoteProxy( + source_url, + cache_policy=blosc2.CachePolicy.DISK, + cache_path=carrier_path, + max_cache_bytes=max_cache_bytes, + ) + elif cache_policy == "memory": + creator = blosc2.RemoteProxy( + source_url, + cache_policy=blosc2.CachePolicy.MEMORY, + max_cache_bytes=max_cache_bytes, + ) + creator.save(carrier_path) + elif cache_policy == "none": + creator = blosc2.RemoteProxy( + source_url, + cache_policy=blosc2.CachePolicy.NONE, + ) + creator.save(carrier_path) + else: + raise ValueError(f"unknown cache_policy: {cache_policy}") + carrier, payload = remote_proxy.inspect(carrier_path) geometry = (creator.shape, creator.dtype, creator.chunks, creator.blocks) return remote_proxy.ServerRemoteProxy(creator.src, geometry, carrier, payload), data, carrier_path @@ -224,6 +353,199 @@ def test_server_proxy_reuses_its_carrier_cache(tmp_path): assert fresh_source.traffic.requests == 0 +def test_server_proxy_memory_retains_no_cache_and_repeats_upstream_fetches(tmp_path): + proxy, data, carrier_path = _server_proxy( + tmp_path, "memory-proxy", cache_policy="memory", max_cache_bytes=500_000 + ) + assert proxy.requested_cache_policy == "memory" + assert proxy.requested_max_cache_bytes == 500_000 + assert proxy.cache_policy == "none" + assert proxy.effective_cache_policy == "none" + assert proxy.max_cache_bytes is None + assert proxy.current_cache_bytes() == 0 + + before_bytes = carrier_path.read_bytes() + before_size = carrier_path.stat().st_size + before_mtime = carrier_path.stat().st_mtime_ns + + # Instrument data calls on proxy.src + chunk_calls = [] + range_calls = [] + orig_get_chunk = proxy.src.get_chunk + orig_read_range = proxy.src.read_range + + def traced_get_chunk(n): + chunk_calls.append(n) + return orig_get_chunk(n) + + def traced_read_range(*args, **kwargs): + range_calls.append(args) + return orig_read_range(*args, **kwargs) + + proxy.src.get_chunk = traced_get_chunk + proxy.src.read_range = traced_read_range + + # First slice read + slice_item = slice(0, 10) + np.testing.assert_array_equal(proxy.read(slice_item), data[:10]) + first_chunk_count = len(chunk_calls) + first_range_count = len(range_calls) + assert first_chunk_count > 0 or first_range_count > 0 + + # Second slice read on the same runtime: no retained data, must fetch again + np.testing.assert_array_equal(proxy.read(slice_item), data[:10]) + assert len(chunk_calls) > first_chunk_count or len(range_calls) > first_range_count + + # Reconstruct the runtime and read slice again + carrier, payload = remote_proxy.inspect(carrier_path) + fresh_source = blosc2.FsspecNDSource(payload["source"]["urlpath"]) + fresh_chunk_calls = [] + fresh_orig_get_chunk = fresh_source.get_chunk + + def fresh_traced_get_chunk(n): + fresh_chunk_calls.append(n) + return fresh_orig_get_chunk(n) + + fresh_source.get_chunk = fresh_traced_get_chunk + reopened = remote_proxy.ServerRemoteProxy( + fresh_source, + (proxy.shape, proxy.dtype, proxy.chunks, proxy.blocks), + carrier, + payload, + ) + np.testing.assert_array_equal(reopened.read(slice_item), data[:10]) + assert len(fresh_chunk_calls) > 0 or fresh_source.traffic.requests > 0 + + # Repeat for get_chunk + chunk_calls.clear() + chunk0_first = proxy.get_chunk(0) + assert len(chunk_calls) == 1 + assert chunk_calls[0] == 0 + chunk0_second = proxy.get_chunk(0) + assert len(chunk_calls) == 2 + assert chunk_calls[1] == 0 + assert chunk0_first == chunk0_second + + # Invariants on carrier file + assert carrier_path.read_bytes() == before_bytes + assert carrier_path.stat().st_size == before_size + assert carrier_path.stat().st_mtime_ns == before_mtime + assert proxy.current_cache_bytes() == 0 + + +def test_memory_carrier_ignores_synthetic_cached_chunks(tmp_path): + # Create DISK proxy and warm chunks 0 and 1 + disk_proxy, data, carrier_path = _server_proxy(tmp_path, "synthetic", cache_policy="disk") + disk_proxy.read(slice(0, 20)) + raw = remote_proxy.raw_carrier(carrier_path, mode="a", locking=True) + with raw.schunk.holding_lock(): + assert raw.schunk.vlmeta.get("proxy-cache-sizes") + # Mutate carrier's payload to MEMORY + payload = dict(raw.schunk.vlmeta["b2o"]) + payload["cache_policy"] = "memory" + payload["max_cache_bytes"] = 500_000 + raw.schunk.vlmeta["b2o"] = payload + + # Update upstream source with new data + new_data = data + 1000 + fsspec.filesystem("memory").pipe_file( + "synthetic-source.b2nd", + blosc2.asarray(new_data, chunks=(10,), blocks=(5,)).to_cframe(), + ) + + carrier, payload = remote_proxy.inspect(carrier_path) + assert payload["cache_policy"] == "memory" + fresh_src = blosc2.FsspecNDSource("memory://synthetic-source.b2nd") + mem_proxy = remote_proxy.ServerRemoteProxy( + fresh_src, + (disk_proxy.shape, disk_proxy.dtype, disk_proxy.chunks, disk_proxy.blocks), + carrier, + payload, + ) + assert mem_proxy.cache_policy == "none" + assert mem_proxy.effective_cache_policy == "none" + assert mem_proxy.current_cache_bytes() == 0 + + # Logical reads must return source data, ignoring the carrier's cached chunks + np.testing.assert_array_equal(mem_proxy.read(slice(0, 20)), new_data[:20]) + chunk = mem_proxy.get_chunk(0) + assert chunk == fresh_src.get_chunk(0) + + +def test_memory_carrier_exports_preserve_policy_and_reopen_with_client_cache(tmp_path): + _proxy, data, carrier_path = _server_proxy( + tmp_path, "export-mem", cache_policy="memory", max_cache_bytes=500_000 + ) + carrier, payload = remote_proxy.inspect(carrier_path) + + warm_bytes = remote_proxy.export_cframe(carrier, payload, include_cache=True) + cold_bytes = remote_proxy.export_cframe(carrier, payload, include_cache=False) + + for kind, b in [("warm", warm_bytes), ("cold", cold_bytes)]: + out_path = tmp_path / f"export_{kind}.b2nd" + out_path.write_bytes(b) + reopened = blosc2.open(str(out_path)) + assert isinstance(reopened, blosc2.RemoteProxy) + assert reopened.cache_policy == blosc2.CachePolicy.MEMORY + assert reopened.max_cache_bytes == 500_000 + np.testing.assert_array_equal(reopened[:10], data[:10]) + + # Client memory cache reuse: uncached slice fetches then hits cache + reopened.src.traffic.reset() + res1 = reopened[10:20] + assert reopened.src.traffic.requests > 0 + np.testing.assert_array_equal(res1, data[10:20]) + reopened.src.traffic.reset() + res2 = reopened[10:20] + assert reopened.src.traffic.requests == 0 + np.testing.assert_array_equal(res2, data[10:20]) + + +def test_memory_carrier_detects_geometry_replacement_and_observes_data_replacement(tmp_path): + _proxy, data, carrier_path = _server_proxy( + tmp_path, "replacement", cache_policy="memory", max_cache_bytes=500_000 + ) + carrier, payload = remote_proxy.inspect(carrier_path) + + remote_proxy.configure( + _Conf( + { + ".remote_proxy.enabled": True, + ".remote_proxy.allowed_hosts": ["data.example"], + } + ) + ) + + url = "https://data.example/replacement-source.b2nd" + payload = dict(payload) + payload["source"] = dict(payload["source"]) + payload["source"]["urlpath"] = url + + mem_fs = fsspec.filesystem("memory") + mem_fs.pipe_file(url, blosc2.asarray(data, chunks=(10,), blocks=(5,)).to_cframe()) + + orig_public_addr = remote_proxy._public_addresses + orig_fs = remote_proxy._https_filesystem + remote_proxy._public_addresses = lambda host, port: ("93.184.216.34",) + remote_proxy._https_filesystem = lambda host, addr: mem_fs + + try: + # Case A: Source geometry changed + mismatched_source = blosc2.asarray(np.arange(60, dtype=np.int32), chunks=(10,), blocks=(5,)) + mem_fs.pipe_file(url, mismatched_source.to_cframe()) + with pytest.raises(remote_proxy.RemoteProxyDenied, match="geometry does not match"): + remote_proxy.resolve(carrier, payload) + + # Case B: Source data changed (geometry identical) + new_data = np.arange(100, 140, dtype=np.int32) + mem_fs.pipe_file(url, blosc2.asarray(new_data, chunks=(10,), blocks=(5,)).to_cframe()) + resolved = remote_proxy.resolve(carrier, payload) + np.testing.assert_array_equal(resolved.read(slice(0, 10)), new_data[:10]) + finally: + remote_proxy._public_addresses = orig_public_addr + remote_proxy._https_filesystem = orig_fs + + def test_zero_effective_quota_reads_without_retaining(tmp_path): proxy, data, carrier_path = _server_proxy(tmp_path, "zero-quota") before = carrier_path.read_bytes() @@ -250,6 +572,51 @@ def current_cache_bytes(): monkeypatch.setattr(server, "get_disk_usage_written", lambda pending: 1_000) assert server.remote_proxy_cache_limit(Proxy()) == 200 + class UnlimitedProxy: + cache_policy = "disk" + max_cache_bytes = None + + @staticmethod + def current_cache_bytes(): + return 200 + + # Under quota, unlimited disk cache is bounded by available quota (200 + 100 = 300) + monkeypatch.setattr(server, "get_disk_usage_written", lambda pending: 900) + assert server.remote_proxy_cache_limit(UnlimitedProxy()) == 300 + + # When quota is exhausted, bounded by retained bytes (200 + 0 = 200) + monkeypatch.setattr(server, "get_disk_usage_written", lambda pending: 1_000) + assert server.remote_proxy_cache_limit(UnlimitedProxy()) == 200 + + # Without quota, unlimited disk cache returns None + monkeypatch.setattr(server.settings, "quota", 0) + assert server.remote_proxy_cache_limit(UnlimitedProxy()) is None + + class MemoryProxy: + cache_policy = "none" + max_cache_bytes = None + + assert server.remote_proxy_cache_limit(MemoryProxy()) == 0 + + +def test_unlimited_disk_server_proxy_caches_without_eviction(tmp_path): + proxy, data, carrier_path = _server_proxy( + tmp_path, "unlimited-server", cache_policy="disk", max_cache_bytes=None + ) + assert proxy.requested_max_cache_bytes is None + assert proxy.max_cache_bytes is None + assert proxy.current_cache_bytes() == 0 + + # Read slices to trigger cache population + np.testing.assert_array_equal(proxy.read(slice(0, 20)), data[:20]) + np.testing.assert_array_equal(proxy.read(slice(20, 40)), data[20:]) + cached_bytes = proxy.current_cache_bytes() + assert cached_bytes > 0 + + # Reopening carrier shows data is valid + reopened = blosc2.open(carrier_path, mode="r") + np.testing.assert_array_equal(reopened[:], data) + def test_concurrent_server_proxy_fills_do_not_corrupt_carrier(tmp_path): first, data, carrier_path = _server_proxy(tmp_path, "concurrent") diff --git a/doc/utilities/cat2-server.md b/doc/utilities/cat2-server.md index 27ada0c7..3bc998e9 100644 --- a/doc/utilities/cat2-server.md +++ b/doc/utilities/cat2-server.md @@ -35,9 +35,13 @@ And then simply run `cat2-server` to start it on all network interfaces on port ## Remote reference policy A persisted `blosc2.RemoteProxy` is a B2ND carrier that asks Caterva2 to read -another dataset. With its disk cache enabled, fetched compressed chunks are -retained inside that same carrier up to the proxy's finite `max_cache_bytes`. -Caterva2 can inspect and report its stored shape, dtype, chunk, block, and proxy +another dataset. Persisted `MEMORY` carriers are accepted under the same source +policy but execute without retained caching (using the same no-retention execution +path as `NONE`), avoiding unmanaged memory use on the server while preserving +the requested client limit for download. With a `DISK` cache, fetched compressed +chunks are retained inside the carrier up to the proxy's `max_cache_bytes` (or +unbounded when `max_cache_bytes` is `None`, still subject to customer quota if +configured). Caterva2 can inspect and report its stored shape, dtype, chunk, block, and proxy metadata without contacting the source. Outbound resolution is disabled by default. From bf37df2a12be2c06fdbd923acb2c815e8529c83e Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sat, 5 Sep 2026 14:58:45 +0200 Subject: [PATCH 04/20] Prevent remote cache quota growth and fix test isolation --- caterva2-server.sample.toml | 3 +- caterva2/services/remote_proxy.py | 28 +++++++++----- caterva2/services/server.py | 15 ++++---- caterva2/tests/test_api.py | 7 ++-- caterva2/tests/test_remote_proxy.py | 58 +++++++++++++++++++++++++---- doc/utilities/cat2-server.md | 15 +++++--- 6 files changed, 93 insertions(+), 33 deletions(-) diff --git a/caterva2-server.sample.toml b/caterva2-server.sample.toml index f6ad03cf..1512e2df 100644 --- a/caterva2-server.sample.toml +++ b/caterva2-server.sample.toml @@ -36,7 +36,8 @@ register = true # allow users to register # otherwise non-public destination addresses, and embedded expression # references are refused. MEMORY carriers are accepted under the same source # policy but execute without retained caching (same as NONE). DISK proxies cache -# chunks inside their carrier up to persisted max_cache_bytes (or unbounded if None), subject to server quota. +# chunks inside their carrier up to persisted max_cache_bytes (or unbounded if None). +# With a customer quota configured, warm DISK chunks are reused but misses are not retained. # [server.remote_proxy] # enabled = true # allowed_hosts = ["datasets.example.org", "objects.example.org:8443"] diff --git a/caterva2/services/remote_proxy.py b/caterva2/services/remote_proxy.py index c8752af4..a364d60d 100644 --- a/caterva2/services/remote_proxy.py +++ b/caterva2/services/remote_proxy.py @@ -390,15 +390,18 @@ def current_cache_bytes(self) -> int: with carrier_thread_lock(self.path): carrier = raw_carrier(self.path, locking=True) with carrier.schunk.holding_lock(): - sizes = carrier.schunk.vlmeta.get("proxy-cache-sizes") - if isinstance(sizes, dict): - return sum(size for size in sizes.values() if isinstance(size, int) and size >= 0) + # The physical payload is authoritative. Uploaded size tables + # can be stale after older unbounded writers (or user supplied). return carrier.schunk.cbytes def _backend(self, cache_limit=None, *, carrier=None): - if self.cache_policy != "disk" or cache_limit == 0: + if self.cache_policy != "disk": return blosc2.Proxy(self.src, _refresh_source=False) - carrier = raw_carrier(self.path, mode="a", locking=True) if carrier is None else carrier + mode = "r" if cache_limit == 0 else "a" + carrier = raw_carrier(self.path, mode=mode, locking=True) if carrier is None else carrier + if cache_limit == 0: + # Consume warm data without writing metadata or retaining misses. + return blosc2.Proxy(self.src, _cache=carrier, _refresh_source=False) if cache_limit is None: limit = self.max_cache_bytes elif self.max_cache_bytes is None: @@ -413,10 +416,10 @@ def _backend(self, cache_limit=None, *, carrier=None): ) def read(self, item, *, cache_limit=None): - if self.cache_policy != "disk" or cache_limit == 0: + if self.cache_policy != "disk": return self._backend(cache_limit)[item] with carrier_thread_lock(self.path): - carrier = raw_carrier(self.path, mode="a", locking=True) + carrier = raw_carrier(self.path, mode="r" if cache_limit == 0 else "a", locking=True) with carrier.schunk.holding_lock(): backend = self._backend(cache_limit, carrier=carrier) return backend[item] @@ -425,7 +428,7 @@ def __getitem__(self, item): return self.read(item) def get_chunk(self, nchunk, *, cache_limit=None): - if self.cache_policy != "disk" or cache_limit == 0: + if self.cache_policy != "disk": return self.src.get_chunk(nchunk) item = tuple( slice(coord * chunk, min((coord + 1) * chunk, size)) @@ -442,10 +445,15 @@ def get_chunk(self, nchunk, *, cache_limit=None): ) ) with carrier_thread_lock(self.path): - carrier = raw_carrier(self.path, mode="a", locking=True) + carrier = raw_carrier(self.path, mode="r" if cache_limit == 0 else "a", locking=True) with carrier.schunk.holding_lock(): backend = self._backend(cache_limit, carrier=carrier) - backend.fetch(item) + try: + backend.fetch(item) + except ValueError as exc: + if cache_limit != 0 or "reading mode" not in str(exc): + raise + return self.src.get_chunk(nchunk) chunk = backend.schunk.get_chunk(nchunk) backend._enforce_cache_limit(item) return chunk diff --git a/caterva2/services/server.py b/caterva2/services/server.py index bc50caab..433bcf52 100644 --- a/caterva2/services/server.py +++ b/caterva2/services/server.py @@ -176,17 +176,18 @@ def account_chunk_written(nbytes: int) -> None: def remote_proxy_cache_limit(proxy: remote_proxy.ServerRemoteProxy) -> int | None: - """Return the per-carrier cache bound after applying remaining customer quota.""" + """Return the cache allowance; quota-enabled servers consume caches read-only. + + Payload limits cannot reserve physical metadata growth, and per-dataset locks + do not coordinate other carriers, uploads, or workers. Until all writers share + a physical-storage reservation mechanism, automatic fills must not grow disk + usage under a customer quota. Zero still permits reuse of warm carrier data. + """ if proxy.cache_policy != "disk": return 0 if not settings.quota: return proxy.max_cache_bytes - retained = proxy.current_cache_bytes() - available = max(0, settings.quota - get_disk_usage_written(0)) - quota_bound = retained + available - if proxy.max_cache_bytes is None: - return quota_bound - return min(proxy.max_cache_bytes, quota_bound) + return 0 async def read_remote_proxy(proxy, item, abspath): diff --git a/caterva2/tests/test_api.py b/caterva2/tests/test_api.py index 2143ccfd..c321f6fe 100644 --- a/caterva2/tests/test_api.py +++ b/caterva2/tests/test_api.py @@ -246,8 +246,8 @@ def __init__(self, values): def get(self, key, default=None): return self.values.get(key, default) - server.settings.statedir = tmp_path / "statedir" - server.settings.public = server.settings.statedir / "public" + monkeypatch.setattr(server.settings, "statedir", tmp_path / "statedir") + monkeypatch.setattr(server.settings, "public", server.settings.statedir / "public") server.settings.public.mkdir(parents=True, exist_ok=True) data = np.arange(40, dtype=np.int32) @@ -271,6 +271,7 @@ def get(self, key, default=None): payload["source"]["urlpath"] = url carrier.schunk.vlmeta["b2o"] = payload + monkeypatch.setattr(remote_proxy, "policy", remote_proxy.policy) remote_proxy.configure( _Conf( { @@ -292,7 +293,7 @@ def traced_cat_file(path, *args, **kwargs): upstream_reads.append(path) return orig_cat_file(path, *args, **kwargs) - mem_fs.cat_file = traced_cat_file + monkeypatch.setattr(mem_fs, "cat_file", traced_cat_file) monkeypatch.setattr(remote_proxy, "_public_addresses", lambda host, port: ("93.184.216.34",)) monkeypatch.setattr(remote_proxy, "_https_filesystem", lambda host, addr: mem_fs) diff --git a/caterva2/tests/test_remote_proxy.py b/caterva2/tests/test_remote_proxy.py index d8187dc5..1a4853d8 100644 --- a/caterva2/tests/test_remote_proxy.py +++ b/caterva2/tests/test_remote_proxy.py @@ -553,7 +553,51 @@ def test_zero_effective_quota_reads_without_retaining(tmp_path): assert carrier_path.read_bytes() == before -def test_customer_quota_reduces_the_proxy_cache_limit(monkeypatch): +@pytest.mark.parametrize("limit", [None, 1_000_000]) +def test_read_only_cache_reuses_warm_chunks_and_does_not_retain_misses(tmp_path, limit): + proxy, data, path = _server_proxy(tmp_path, "readonly-cache", max_cache_bytes=limit) + proxy.get_chunk(0) + before = path.read_bytes() + mtime = path.stat().st_mtime_ns + proxy.src.traffic.reset() + np.testing.assert_array_equal(proxy.read(slice(0, 10), cache_limit=0), data[:10]) + proxy.get_chunk(0, cache_limit=0) + assert proxy.src.traffic.requests == 0 + np.testing.assert_array_equal(proxy.read(slice(10, 20), cache_limit=0), data[10:20]) + assert proxy.src.traffic.requests > 0 + proxy.src.traffic.reset() + proxy.get_chunk(1, cache_limit=0) + assert proxy.src.traffic.requests > 0 + assert path.read_bytes() == before + assert path.stat().st_mtime_ns == mtime + + +def test_concurrent_quota_reads_do_not_grow_distinct_carriers(tmp_path, monkeypatch): + monkeypatch.setenv("CATERVA2_SECRET", "remote-proxy-test-secret") + from caterva2.services import server + + monkeypatch.setattr(server.settings, "quota", 1) + entries = [_server_proxy(tmp_path, name, max_cache_bytes=None) for name in ("left", "right")] + before = [path.read_bytes() for _, _, path in entries] + with concurrent.futures.ThreadPoolExecutor(max_workers=2) as pool: + futures = [ + pool.submit(proxy.read, (), cache_limit=server.remote_proxy_cache_limit(proxy)) + for proxy, _, _ in entries + ] + for future, (_, data, _) in zip(futures, entries, strict=True): + np.testing.assert_array_equal(future.result(timeout=10), data) + assert [path.read_bytes() for _, _, path in entries] == before + + +def test_cache_accounting_ignores_stale_size_table(tmp_path): + proxy, _, path = _server_proxy(tmp_path, "stale-table", max_cache_bytes=None) + proxy.read(()) + carrier = remote_proxy.raw_carrier(path, mode="a") + carrier.schunk.vlmeta["proxy-cache-sizes"] = {"0": 1} + assert proxy.current_cache_bytes() == carrier.schunk.cbytes + + +def test_customer_quota_disables_cache_growth(monkeypatch): monkeypatch.setenv("CATERVA2_SECRET", "remote-proxy-test-secret") from caterva2.services import server @@ -567,10 +611,10 @@ def current_cache_bytes(): monkeypatch.setattr(server.settings, "quota", 1_000) monkeypatch.setattr(server, "get_disk_usage_written", lambda pending: 900) - assert server.remote_proxy_cache_limit(Proxy()) == 300 + assert server.remote_proxy_cache_limit(Proxy()) == 0 monkeypatch.setattr(server, "get_disk_usage_written", lambda pending: 1_000) - assert server.remote_proxy_cache_limit(Proxy()) == 200 + assert server.remote_proxy_cache_limit(Proxy()) == 0 class UnlimitedProxy: cache_policy = "disk" @@ -580,13 +624,13 @@ class UnlimitedProxy: def current_cache_bytes(): return 200 - # Under quota, unlimited disk cache is bounded by available quota (200 + 100 = 300) + # Any configured quota disables automatic growth, regardless of remaining space. monkeypatch.setattr(server, "get_disk_usage_written", lambda pending: 900) - assert server.remote_proxy_cache_limit(UnlimitedProxy()) == 300 + assert server.remote_proxy_cache_limit(UnlimitedProxy()) == 0 - # When quota is exhausted, bounded by retained bytes (200 + 0 = 200) + # Exhausted quota also leaves the carrier read-only. monkeypatch.setattr(server, "get_disk_usage_written", lambda pending: 1_000) - assert server.remote_proxy_cache_limit(UnlimitedProxy()) == 200 + assert server.remote_proxy_cache_limit(UnlimitedProxy()) == 0 # Without quota, unlimited disk cache returns None monkeypatch.setattr(server.settings, "quota", 0) diff --git a/doc/utilities/cat2-server.md b/doc/utilities/cat2-server.md index 3bc998e9..4df5a1e7 100644 --- a/doc/utilities/cat2-server.md +++ b/doc/utilities/cat2-server.md @@ -40,8 +40,9 @@ policy but execute without retained caching (using the same no-retention executi path as `NONE`), avoiding unmanaged memory use on the server while preserving the requested client limit for download. With a `DISK` cache, fetched compressed chunks are retained inside the carrier up to the proxy's `max_cache_bytes` (or -unbounded when `max_cache_bytes` is `None`, still subject to customer quota if -configured). Caterva2 can inspect and report its stored shape, dtype, chunk, block, and proxy +unbounded when `max_cache_bytes` is `None`) when no customer quota is configured. +With a quota, existing warm disk chunks are reused but misses are not retained. +Caterva2 can inspect and report its stored shape, dtype, chunk, block, and proxy metadata without contacting the source. Outbound resolution is disabled by default. @@ -69,9 +70,13 @@ created by a client that performed its own validation. The limits validate the remote array's structure and bound connection time and concurrent range fetches. They do not impose a network-work budget on each API request. The proxy's own cache limit bounds its retained compressed payload, -while automatic carrier growth is also charged to the virtual server's existing -shared `quota`. If no quota remains, reads still succeed but misses are not -retained. +while a configured customer `quota` makes all DISK proxy caches read-only. +Valid warm chunks are reused, but misses are served without retention, even when +quota space remains. This conservative restriction prevents cache fills from +overspending physical storage through metadata growth or concurrent workers. +Automatic fills under quota require a future storage reservation mechanism shared +with uploads and other writers. Without a quota, normal bounded or unbounded +DISK caching applies. Public S3 objects are supported through credential-free HTTPS object URLs. Native `s3://` resolution, private-source credentials, and remote references From a1de98c9cb2094ee04d73f6b2a99173b61ce0b04 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 6 Sep 2026 00:17:40 +0200 Subject: [PATCH 05/20] Add shared quota admission for remote proxy caches --- caterva2-server.sample.toml | 8 +- caterva2/hdf5.py | 23 +- caterva2/services/remote_proxy.py | 58 ++++ caterva2/services/server.py | 409 ++++++++++++++++------ caterva2/services/storage_quota.py | 418 +++++++++++++++++++++++ caterva2/tests/test_chunk_writes.py | 7 +- caterva2/tests/test_storage_quota.py | 303 ++++++++++++++++ caterva2/tests/test_storage_quota_api.py | 225 ++++++++++++ doc/utilities/cat2-server.md | 63 +++- examples/benchmark_storage_quota.py | 69 ++++ 10 files changed, 1459 insertions(+), 124 deletions(-) create mode 100644 caterva2/services/storage_quota.py create mode 100644 caterva2/tests/test_storage_quota.py create mode 100644 caterva2/tests/test_storage_quota_api.py create mode 100644 examples/benchmark_storage_quota.py diff --git a/caterva2-server.sample.toml b/caterva2-server.sample.toml index 1512e2df..dc2f14cc 100644 --- a/caterva2-server.sample.toml +++ b/caterva2-server.sample.toml @@ -7,7 +7,8 @@ # # - listen: where the server listens to (a unix socket or a host/port) (default: localhost:8000) # - urlbase: the base url users will use to reach the server (default: http://localhost:8000) -# - quota: if defined, it will limit the disk usage (default: 0, no limit) +# - quota: limits apparent dataset bytes in public/shared/personal (default: 0, no limit) +# - quota_work_bytes: separate quota-coordinated disk staging budget (default: "1G") # - maxusers: if defined, it will limit the number of users (default: 0, no limit) # - login: if true, users will need to authenticate (default: true) # - register: if true, users will be able to register (default: false) @@ -25,6 +26,7 @@ listen = "localhost:8000" urlbase = "http://localhost:8000" quota = "10G" +# quota_work_bytes = "1G" maxusers = 5 register = true # allow users to register # publish_root = "s3://a-bucket/published" @@ -37,7 +39,9 @@ register = true # allow users to register # references are refused. MEMORY carriers are accepted under the same source # policy but execute without retained caching (same as NONE). DISK proxies cache # chunks inside their carrier up to persisted max_cache_bytes (or unbounded if None). -# With a customer quota configured, warm DISK chunks are reused but misses are not retained. +# With a customer quota, shared SQLite admission permits DISK growth when capacity +# is available; denied fills still return data without retention. See server docs +# for accounting exclusions, operational headroom, and staged replacement costs. # [server.remote_proxy] # enabled = true # allowed_hosts = ["datasets.example.org", "objects.example.org:8443"] diff --git a/caterva2/hdf5.py b/caterva2/hdf5.py index 7138ac5c..749cd600 100644 --- a/caterva2/hdf5.py +++ b/caterva2/hdf5.py @@ -373,7 +373,7 @@ def open_leaf(cls, h5file, dsetname): self.b2arr = blosc2.empty(self.dset.shape or (), dtype=self.dset.dtype, **b2args) return self - def __init__(self, b2arr, h5file=None, dsetname=None): + def __init__(self, b2arr, h5file=None, dsetname=None, *, writer=None): if b2arr is not None: # The file has been opened already, so we just need to set the filename and dataset name self.dsetname = b2arr.vlmeta["_dsetname"] @@ -426,7 +426,7 @@ def __init__(self, b2arr, h5file=None, dsetname=None): self.b2arr = blosc2.empty( shape=shape, dtype=dtype, - urlpath=urlpath, + urlpath=urlpath if writer is None else None, mode="w", **b2args, ) @@ -436,7 +436,7 @@ def __init__(self, b2arr, h5file=None, dsetname=None): del self.dset del self.fname del self.dsetname - if os.path.exists(urlpath): + if writer is None and os.path.exists(urlpath): os.remove(urlpath) return @@ -717,7 +717,7 @@ def serialize_h5_attrs_to_json(h5_attrs, indent=2): return json_str -def create_hdf5_proxies(path: str | os.PathLike) -> Iterator[HDF5Proxy]: +def create_hdf5_proxies(path: str | os.PathLike, *, writer=None) -> Iterator[HDF5Proxy]: """Create a generator of HDF5 proxies from the given HDF5 file.""" attrs_dsetname = "!_attrs_.json.b2" # the Blosc2 dataset name for the Group attributes in HDF5 h5file = h5py.File(path, "r") @@ -727,7 +727,10 @@ def create_hdf5_proxies(path: str | os.PathLike) -> Iterator[HDF5Proxy]: os.makedirs(dirname, exist_ok=True) jsonpath = os.path.join(dirname, attrs_dsetname) data = serialize_h5_attrs_to_json(h5file.attrs) - blosc2.SChunk(data=data.encode("utf-8"), urlpath=jsonpath, mode="w") + if writer is None: + blosc2.SChunk(data=data.encode("utf-8"), urlpath=jsonpath, mode="w") + else: + writer(jsonpath, blosc2.SChunk(data=data.encode("utf-8")).to_cframe()) # Recursive function to visit all groups and datasets def visit_group(group): @@ -735,14 +738,20 @@ def visit_group(group): full_path = f"{group.name}/{name}".lstrip("/") if isinstance(obj, h5py.Dataset): - yield HDF5Proxy(None, h5file, full_path) + proxy = HDF5Proxy(None, h5file, full_path, writer=writer) + if writer is not None and hasattr(proxy, "b2arr"): + writer(os.path.join(dirname, full_path + ".b2nd"), proxy.b2arr.to_cframe()) + yield proxy if isinstance(obj, h5py.Group): # Store HDF5 group attributes as JSON groupname = dirname + "/" + full_path os.makedirs(groupname, exist_ok=True) jsonpath = os.path.join(groupname, attrs_dsetname) data = serialize_h5_attrs_to_json(obj.attrs) - blosc2.SChunk(data=data.encode("utf-8"), urlpath=jsonpath, mode="w") + if writer is None: + blosc2.SChunk(data=data.encode("utf-8"), urlpath=jsonpath, mode="w") + else: + writer(jsonpath, blosc2.SChunk(data=data.encode("utf-8")).to_cframe()) # Recursively visit subgroups yield from visit_group(obj) diff --git a/caterva2/services/remote_proxy.py b/caterva2/services/remote_proxy.py index a364d60d..21449c25 100644 --- a/caterva2/services/remote_proxy.py +++ b/caterva2/services/remote_proxy.py @@ -19,6 +19,7 @@ import ipaddress import math import socket +import sqlite3 import threading import weakref from dataclasses import dataclass @@ -30,6 +31,8 @@ from blosc2.b2objects import make_b2object_carrier, write_b2object_payload from fsspec.implementations.http import HTTPFileSystem +from caterva2.services import storage_quota + class RemoteProxyDenied(ValueError): """The server policy refuses a remote reference.""" @@ -375,6 +378,7 @@ def __init__(self, source, geometry, carrier, payload): self.cparams = source.cparams self.path = carrier.schunk.urlpath self.requested_cache_policy = payload["cache_policy"] + self.requested_payload = dict(payload) self.requested_max_cache_bytes = payload["max_cache_bytes"] self.cache_policy = _effective_cache_policy(self.requested_cache_policy) self.max_cache_bytes = self.requested_max_cache_bytes if self.cache_policy == "disk" else None @@ -394,6 +398,60 @@ def current_cache_bytes(self) -> int: # can be stale after older unbounded writers (or user supplied). return carrier.schunk.cbytes + def quota_read(self, quota, item=(), *, nchunk=None): + """Assemble on an immutable candidate and admit its exact physical size.""" + if self.cache_policy != "disk" or getattr(self.src, "stamp", None) is None: + return ( + self.get_chunk(nchunk, cache_limit=0) + if nchunk is not None + else self.read(item, cache_limit=0) + ) + try: + frame, generation = quota.snapshot(self.path) + except (OSError, RuntimeError, ValueError, sqlite3.Error): + return ( + self.get_chunk(nchunk, cache_limit=0) + if nchunk is not None + else self.read(item, cache_limit=0) + ) + if frame is None or len(frame) > quota.work_bytes: + return ( + self.get_chunk(nchunk, cache_limit=0) + if nchunk is not None + else self.read(item, cache_limit=0) + ) + carrier = blosc2.ndarray_from_cframe(frame, copy=True) + if carrier.schunk.vlmeta.get("b2o") != self.requested_payload: + return ( + self.src.get_chunk(nchunk) + if nchunk is not None + else blosc2.Proxy(self.src, _refresh_source=False)[item] + ) + backend = blosc2.Proxy( + self.src, _cache=carrier, _refresh_source=False, _max_cache_bytes=self.max_cache_bytes + ) + if nchunk is None: + result = backend[item] + else: + grid = tuple(math.ceil(s / c) for s, c in zip(self.shape, self.chunks, strict=True)) + item = tuple( + slice(int(i) * c, min((int(i) + 1) * c, s)) + for i, c, s in zip(np.unravel_index(nchunk, grid), self.chunks, self.shape, strict=True) + ) + backend.fetch(item) + result = backend.schunk.get_chunk(nchunk) + backend._enforce_cache_limit(item) + candidate = carrier.to_cframe() + try: + if candidate != frame: + quota.publish(self.path, candidate, expected=generation, cache=True) + else: + quota.touch(self.path) + except (storage_quota.QuotaExceeded, storage_quota.StorageBusy, OSError, sqlite3.Error): + # The logical result already exists. Retention is strictly optional. + pass + return result + def _backend(self, cache_limit=None, *, carrier=None): if self.cache_policy != "disk": return blosc2.Proxy(self.src, _refresh_source=False) diff --git a/caterva2/services/server.py b/caterva2/services/server.py index 433bcf52..85fcb082 100644 --- a/caterva2/services/server.py +++ b/caterva2/services/server.py @@ -21,6 +21,7 @@ import os import pathlib import shutil +import sqlite3 import string import tarfile import threading @@ -50,6 +51,7 @@ import pygments import uvicorn from blosc2 import linalg_funcs_list as linalg_funcs +from blosc2.lazyexpr import LazyArrayEnum # FastAPI from fastapi import Depends, FastAPI, Form, Request, Response, UploadFile, concurrency, responses @@ -61,7 +63,7 @@ # Project from caterva2 import hdf5, models, utils -from caterva2.services import db, providers, remote_proxy, schemas, settings, srv_utils, users +from caterva2.services import db, providers, remote_proxy, schemas, settings, srv_utils, storage_quota, users from caterva2.services.notebook import inject_pyodide_bootstrap_cell BASE_DIR = pathlib.Path(__file__).resolve().parent @@ -136,6 +138,8 @@ def guess_type(path): def get_disk_usage(): + if settings.quota: + return quota_coordinator().usage()["used"] exclude = {"db.json", "db.sqlite"} return sum( path.stat().st_size @@ -175,13 +179,114 @@ def account_chunk_written(nbytes: int) -> None: _disk_usage["written"] += nbytes +_quota_instances = {} + + +def quota_coordinator(): + """Independent of authentication; one connection is opened per DB operation.""" + if not settings.quota: + return None + work_bytes = settings.parse_size(settings.conf.get(".quota_work_bytes", "1G")) + key = (str(settings.statedir), settings.quota, work_bytes) + coordinator = _quota_instances.get(key) + if coordinator is None: + coordinator = storage_quota.StorageQuota(settings.statedir, settings.quota, work_bytes=work_bytes) + _quota_instances[key] = coordinator + return coordinator + + +def write_dataset(path, data, *, expected=None, compare=False): + """Store final encoded bytes through shared admission when quota is enabled.""" + path = pathlib.Path(path) + quota = quota_coordinator() + if quota is not None: + if not compare: + _, expected = quota.snapshot(path) + quota.publish(path, data, expected=expected) + else: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_bytes(data) + + +def remove_dataset(path): + """Account deletions file-by-file; directory removal is not an atomic batch.""" + path = pathlib.Path(path) + quota = quota_coordinator() + if quota is None: + if path.is_dir(): + shutil.rmtree(path) + else: + srv_utils.unlink_with_b2lock(path) + return + if path.is_dir(): + for entry in list(path.iterdir()): + remove_dataset(entry) + with contextlib.suppress(OSError): + path.rmdir() + elif path.name.endswith(".b2lock"): + return # Stable locks are operational storage, not deletable dataset bytes. + else: + _, generation = quota.snapshot(path) + if generation is None: + raise FileNotFoundError(path) + quota.publish(path, None, expected=generation, prune=False) + + +def copy_dataset(source, destination): + source, destination = pathlib.Path(source), pathlib.Path(destination) + if source.resolve() == destination.resolve(): + return + if source.is_dir() and source.resolve() in destination.resolve().parents: + raise ValueError("cannot copy a directory into itself") + if source.is_dir(): + destination.mkdir(parents=True, exist_ok=True) + for entry in source.iterdir(): + if not entry.name.endswith(".b2lock"): + copy_dataset(entry, destination / entry.name) + else: + write_dataset(destination, source.read_bytes()) + + +def move_dataset(source, destination): + """Copy then delete only the source generation that was actually copied.""" + source, destination = pathlib.Path(source), pathlib.Path(destination) + if source.resolve() == destination.resolve(): + return + if source.is_dir(): + if source.resolve() in destination.resolve().parents: + raise ValueError("cannot move a directory into itself") + destination.mkdir(parents=True, exist_ok=True) + for entry in list(source.iterdir()): + if not entry.name.endswith(".b2lock"): + move_dataset(entry, destination / entry.name) + with contextlib.suppress(OSError): + source.rmdir() + else: + quota = quota_coordinator() + data, generation = quota.snapshot(source) + if generation is None: + raise storage_quota.StorageBusy("move source was removed") + write_dataset(destination, data) + quota.publish(source, None, expected=generation, prune=False) + + +def quota_proxy_operation(proxy, item=(), *, nchunk=None): + try: + quota = quota_coordinator() + except (OSError, sqlite3.Error): + quota = None + if quota is None: + return proxy.read(item, cache_limit=0) if nchunk is None else proxy.get_chunk(nchunk, cache_limit=0) + return proxy.quota_read(quota, item, nchunk=nchunk) + + def remote_proxy_cache_limit(proxy: remote_proxy.ServerRemoteProxy) -> int | None: - """Return the cache allowance; quota-enabled servers consume caches read-only. + """Return the allowance for the legacy, in-place cache path. Payload limits cannot reserve physical metadata growth, and per-dataset locks - do not coordinate other carriers, uploads, or workers. Until all writers share - a physical-storage reservation mechanism, automatic fills must not grow disk - usage under a customer quota. Zero still permits reuse of warm carrier data. + do not coordinate other carriers, uploads, or workers. Quota-enabled requests + use quota_proxy_operation instead; the in-place fallback must remain read-only. + Zero still permits reuse of warm carrier data. """ if proxy.cache_policy != "disk": return 0 @@ -194,16 +299,14 @@ async def read_remote_proxy(proxy, item, abspath): """Read one remote selection while serializing and accounting cache mutation.""" lock = dataset_lock(abspath) async with lock: - before = abspath.stat().st_size + if settings.quota: + return await concurrency.run_in_threadpool( + lambda: blosc2.asarray(quota_proxy_operation(proxy, item)).to_cframe() + ) cache_limit = remote_proxy_cache_limit(proxy) - data = await concurrency.run_in_threadpool( + return await concurrency.run_in_threadpool( lambda: blosc2.asarray(proxy.read(item, cache_limit=cache_limit)).to_cframe() ) - if settings.quota: - growth = max(0, abspath.stat().st_size - before) - if growth: - account_chunk_written(growth) - return data def truncate_path(path, size=35): @@ -368,6 +471,8 @@ def _setup_plugin_globals(): @contextlib.asynccontextmanager async def lifespan(app: FastAPI): + if settings.quota: + await concurrency.run_in_threadpool(quota_coordinator) # Initialize the (users) database if user_login_enabled(): await db.create_db_and_tables(settings.statedir) @@ -403,6 +508,23 @@ def custom_filesizeformat(value): app = FastAPI(lifespan=lifespan) + + +@app.exception_handler(storage_quota.QuotaExceeded) +async def quota_exceeded(request, exc): + return responses.JSONResponse(status_code=400, content={"detail": str(exc)}) + + +@app.exception_handler(storage_quota.StorageBusy) +async def storage_busy(request, exc): + return responses.JSONResponse(status_code=409, content={"detail": str(exc)}) + + +@app.exception_handler(sqlite3.Error) +async def storage_unavailable(request, exc): + return responses.JSONResponse(status_code=503, content={"detail": "storage admission is unavailable"}) + + app.add_middleware(CORSMiddleware, allow_origins=["*"], allow_methods=["*"], allow_headers=["*"]) # TODO: Support user verification @@ -1425,15 +1547,12 @@ async def get_chunk( # In case we do, this would have to be changed. chunk = container.get_chunk(nchunk) elif isinstance(container, remote_proxy.ServerRemoteProxy): - before = abspath.stat().st_size - cache_limit = remote_proxy_cache_limit(container) - chunk = await concurrency.run_in_threadpool( - lambda: container.get_chunk(nchunk, cache_limit=cache_limit) - ) if settings.quota: - growth = max(0, abspath.stat().st_size - before) - if growth: - account_chunk_written(growth) + chunk = await concurrency.run_in_threadpool( + lambda: quota_proxy_operation(container, nchunk=nchunk) + ) + else: + chunk = await concurrency.run_in_threadpool(container.get_chunk, nchunk) else: schunk = getattr(container, "schunk", container) chunk = schunk.get_chunk(nchunk) @@ -1632,6 +1751,16 @@ def publish_dataset(abspath: pathlib.Path, path: pathlib.Path) -> str: srv_utils.raise_bad_request("publishing needs fsspec, which is not installed here") destination = publish_destination(path) fs, target = fsspec.url_to_fs(destination) + if settings.quota and isinstance(fs, fsspec.implementations.local.LocalFileSystem): + target_path = pathlib.Path(target).resolve() + state = pathlib.Path(settings.statedir).resolve() + if target_path == state or state in target_path.parents: + srv_utils.raise_bad_request("local publish_root must be outside the server state directory") + if settings.quota: + quota = quota_coordinator() + frame, generation = quota.snapshot(abspath) + if generation is None: + raise storage_quota.StorageBusy("publish source was removed") # Published under a name of its own and moved into place, so that what # appears at the destination is a whole frame or nothing. A reader that # polls for the array would otherwise open it mid-copy: the file exists from @@ -1651,7 +1780,10 @@ def publish_dataset(abspath: pathlib.Path, path: pathlib.Path) -> str: if parent != target: fs.makedirs(parent, exist_ok=True) try: - with open(abspath, "rb") as source, fs.open(staging, "wb") as target_file: + with ( + io.BytesIO(frame) if settings.quota else open(abspath, "rb") as source, + fs.open(staging, "wb") as target_file, + ): shutil.copyfileobj(source, target_file) fs.mv(staging, target) except BaseException: @@ -1660,6 +1792,20 @@ def publish_dataset(abspath: pathlib.Path, path: pathlib.Path) -> str: with contextlib.suppress(Exception): fs.rm(staging) raise + if settings.quota: + array = blosc2.ndarray_from_cframe(frame, copy=True) + array.schunk.vlmeta[PUBLISHED_URL] = destination + array.schunk.vlmeta[FILL_STATE] = PUBLISHED + published_frame = array.to_cframe() + try: + quota.publish(abspath, published_frame, expected=generation) + except storage_quota.StorageBusy: + # Concurrent publishers of the identical snapshot are idempotent. + # Do not bless another upload/fill merely because its URL matches. + current, _ = quota.snapshot(abspath) + if current != published_frame: + raise + return destination with dataset_thread_lock(abspath): array = blosc2.open(abspath, mode="a", locking=True) with array.schunk.holding_lock(): @@ -1680,6 +1826,49 @@ def store_chunk(abspath: pathlib.Path, nchunk: int, chunk: bytes) -> dict: both find the slot free would otherwise both write it, and the second would move every chunk that came after the first. """ + if settings.quota: + quota = quota_coordinator() + frame, generation = quota.snapshot(abspath) + try: + array = blosc2.ndarray_from_cframe(frame, copy=True) + except (TypeError, ValueError, RuntimeError) as exc: + srv_utils.raise_bad_request(f"{abspath.name} is not a stored NDArray: {exc}") + schunk = array.schunk + if "b2o" in schunk.meta or "proxy-source" in schunk.meta: + srv_utils.raise_bad_request("chunk writes require an ordinary stored NDArray") + if not 0 <= nchunk < schunk.nchunks: + srv_utils.raise_not_found(f"{abspath.name} has no chunk {nchunk}") + try: + nbytes, _, blocksize = blosc2.get_cbuffer_sizes(chunk) + typesize = chunk_typesize(chunk) + except Exception: + srv_utils.raise_bad_request("the body is not a Blosc2 chunk") + if (nbytes, blocksize, typesize) != ( + schunk.chunksize, + schunk.blocksize, + filter_typesize(schunk.typesize), + ): + srv_utils.raise_bad_request("the chunk geometry does not match the array") + if not chunk_is_unwritten(schunk, nchunk): + raise fastapi.HTTPException(status_code=409, detail=f"chunk {nchunk} was already written") + schunk.update_chunk(nchunk, chunk) + if FILL_NONCE not in schunk.vlmeta: + schunk.vlmeta[FILL_NONCE] = uuid.uuid4().hex + schunk.vlmeta[FILL_STATE] = FILLING + written = sum(not chunk_is_unwritten(schunk, i) for i in range(schunk.nchunks)) + state = schunk.vlmeta.get(FILL_STATE, FILLING) + publish = written == schunk.nchunks and state == FILLING and bool(settings.publish_root) + if written == schunk.nchunks and state == FILLING: + state = PUBLISHING if publish else COMPLETE + schunk.vlmeta[FILL_STATE] = state + quota.publish(abspath, array.to_cframe(), expected=generation) + return { + "nchunk": nchunk, + "written": written, + "nchunks": schunk.nchunks, + "state": state, + "publish": publish, + } with dataset_thread_lock(abspath): try: array = blosc2.open(abspath, mode="a", locking=True) @@ -1812,23 +2001,12 @@ async def write_chunk( chunk = await request.body() if not chunk: srv_utils.raise_bad_request("no chunk was sent") - if settings.quota: - # The array was laid out empty, so its slots were never charged for: what - # a fill costs arrives a chunk at a time, and is checked the same way -- - # off a kept walk of the state directory rather than a fresh one, since - # this runs once per chunk (see `get_disk_usage_written`) - total_size = get_disk_usage_written(len(chunk)) - if total_size > settings.quota: - srv_utils.raise_bad_request("Write failed because quota limit has been exceeded.") # One lock per dataset in this process, and the frame's own lock across # processes: the write below blocks, so it cannot hold the event loop lock = dataset_lock(abspath) async with lock: answer = await concurrency.run_in_threadpool(store_chunk, abspath, nchunk, chunk) - if settings.quota: - # Counted only where it is checked, so the two stay paired - account_chunk_written(len(chunk)) if answer.pop("publish"): # After the response, and outside the lock: the writer that finished the # fill should not wait for the upload, and no other writer should either @@ -1972,7 +2150,34 @@ def make_expr( abspath.mkdir(exist_ok=True, parents=True) - if compute: + if settings.quota: + # Serialize before admission: metadata and compression determine the charge. + result = arr.compute() if compute else arr + try: + frame = result.to_cframe() + except (TypeError, ValueError): + if compute or func is None: + raise + # LazyUDF.save supports legacy Python UDFs that to_cframe cannot + # encode as b2objects. Build that same metadata carrier in memory. + carrier = blosc2.empty( + result.shape, + dtype=result.dtype, + chunks=result.chunks, + blocks=result.blocks, + meta={"LazyArray": LazyArrayEnum.UDF.value}, + ) + carrier.schunk.vlmeta["_LazyArray"] = { + "UDF": func, + "operands": { + f"o{i}": str(get_writable_path(pathlib.Path(vars[f"o{i}"]), user)) + for i in range(len(var_dict)) + }, + "name": result.func.__name__, + } + frame = carrier.to_cframe() + write_dataset(urlpath, frame) + elif compute: arr.compute(urlpath=urlpath, mode="w") else: arr.save(urlpath=urlpath, mode="w") @@ -2089,7 +2294,12 @@ async def move( # Make sure the destination directory exists dest_abspath.parent.mkdir(exist_ok=True, parents=True) - abspath.rename(dest_abspath) + if settings.quota: + # Reserve the copy before removing the source. No fictitious free-space + # credit; directory operations retain their existing non-atomic semantics. + move_dataset(abspath, dest_abspath) + else: + abspath.rename(dest_abspath) return str(destpath) @@ -2143,7 +2353,9 @@ async def copy( # raise fastapi.HTTPException(status_code=409, detail="The new path already exists") dest_abspath.parent.mkdir(exist_ok=True, parents=True) - if abspath.is_dir(): + if settings.quota: + copy_dataset(abspath, dest_abspath) + elif abspath.is_dir(): shutil.copytree(abspath, dest_abspath) else: shutil.copy(abspath, dest_abspath) @@ -2221,20 +2433,6 @@ async def upload_file( data = await file.read() if abspath.suffix not in srv_utils.BLOSC2_NATIVE_SUFFIXES: schunk = blosc2.SChunk(data=data) - newsize = schunk.nbytes - else: - newsize = len(data) - - if settings.quota: - try: - oldsize = abspath.stat().st_size - except FileNotFoundError: - oldsize = 0 - - total_size = get_disk_usage() - oldsize + newsize - if total_size > settings.quota: - detail = "Upload failed because quota limit has been exceeded." - raise fastapi.HTTPException(detail=detail, status_code=400) # If regular file, compress it abspath.parent.mkdir(exist_ok=True, parents=True) @@ -2243,8 +2441,7 @@ async def upload_file( abspath = abspath.with_suffix(abspath.suffix + ".b2") # Write the file - with open(abspath, "wb") as f: - f.write(data) + await concurrency.run_in_threadpool(write_dataset, abspath, data) # Return the urlpath return str(path) @@ -2290,20 +2487,6 @@ async def load_from_url( if abspath.suffix not in srv_utils.BLOSC2_NATIVE_SUFFIXES: schunk = blosc2.SChunk(data=data) - newsize = schunk.nbytes - else: - newsize = len(data) - - if settings.quota: - try: - oldsize = abspath.stat().st_size - except FileNotFoundError: - oldsize = 0 - - total_size = get_disk_usage() - oldsize + newsize - if total_size > settings.quota: - detail = "Upload failed because quota limit has been exceeded." - raise fastapi.HTTPException(detail=detail, status_code=400) # If regular file, compress it abspath.parent.mkdir(exist_ok=True, parents=True) @@ -2312,8 +2495,7 @@ async def load_from_url( abspath = abspath.with_suffix(abspath.suffix + ".b2") # Write the file - with open(abspath, "wb") as f: - f.write(data) + await concurrency.run_in_threadpool(write_dataset, abspath, data) # Return the urlpath return str(path) @@ -2358,19 +2540,16 @@ async def append_file( # Check quota # TODO To be fair we should check quota later (after compression, zip unpacking etc.) data = await file.read() - newsize = len(data) - - if settings.quota: - oldsize = abspath.stat().st_size - - total_size = get_disk_usage() + oldsize + newsize - if total_size > settings.quota: - detail = "Upload failed because quota limit has been exceeded." - raise fastapi.HTTPException(detail=detail, status_code=400) # Append the data # The original dataset (open in append mode so it can be resized/written) - orig = blosc2.open(abspath, mode="a") + if settings.quota: + frame, generation = quota_coordinator().snapshot(abspath) + orig = blosc2.ndarray_from_cframe(frame, copy=True) + if "b2o" in orig.schunk.meta or "proxy-source" in orig.schunk.meta: + srv_utils.raise_bad_request("append requires an ordinary stored NDArray") + else: + orig = blosc2.open(abspath, mode="a") # The data to append is a cframe new = blosc2.ndarray_from_cframe(data) # Check that the shapes are compatible @@ -2386,6 +2565,8 @@ async def append_file( orig.resize(result_shape) # Append the new data to orig along the first axis orig[orig.shape[0] - new_len :] = new_data + if settings.quota: + quota_coordinator().publish(abspath, orig.to_cframe(), expected=generation) # Return the new shape return result_shape @@ -2423,32 +2604,17 @@ async def unfold_file( raise fastapi.HTTPException(detail=detail, status_code=400) # Unfold the container - dirname = None if abspath.suffix in {".h5", ".hdf5"}: # Create proxies for each dataset in HDF5 file - all_dsets = list(hdf5.create_hdf5_proxies(abspath)) + all_dsets = list(hdf5.create_hdf5_proxies(abspath, writer=write_dataset if settings.quota else None)) if len(all_dsets) == 0: detail = "No arrays found in HDF5 file" raise fastapi.HTTPException(detail=detail, status_code=400) - dirname = abspath.with_suffix("") else: detail = "Target file must be a zip, tar or hdf5 container" raise fastapi.HTTPException(detail=detail, status_code=400) # Check quota - if settings.quota: - # Get the size of the datasets (proxies) in new directory - newsize = 0 - if os.path.exists(dirname): - # Traverse the directory and get the size for all files - for abspath, _ in srv_utils.walk_files(dirname): - newsize += os.path.getsize(abspath) - total_size = get_disk_usage() + newsize - if total_size > settings.quota: - # Remove the directory if it exists - shutil.rmtree(dirname) - detail = "Unfold failed because quota limit has been exceeded." - raise fastapi.HTTPException(detail=detail, status_code=400) # Return the new directory name return path.stem @@ -2481,17 +2647,17 @@ async def remove( # If abspath is a directory, remove the contents of the directory if abspath.is_dir(): - shutil.rmtree(abspath) + remove_dataset(abspath) else: # Try to unlink the file. NotADirectoryError: a path descending into a # container file (e.g. foo.h5/g) names no real file of its own. try: - srv_utils.unlink_with_b2lock(abspath) + remove_dataset(abspath) except (FileNotFoundError, NotADirectoryError): # Try adding a .b2 extension abspath = abspath.with_suffix(abspath.suffix + ".b2") try: - srv_utils.unlink_with_b2lock(abspath) + remove_dataset(abspath) except (FileNotFoundError, NotADirectoryError) as exc: raise fastapi.HTTPException( status_code=404, # not found @@ -2542,7 +2708,7 @@ async def add_notebook( file = io.StringIO() nbformat.write(nb, file) data = file.getvalue().encode() - srv_utils.compress(data, dst=abspath) + write_dataset(abspath, srv_utils.compress(data).to_cframe()) return path @@ -2665,7 +2831,8 @@ async def del_user( # Remove the personal directory of the user userid = str(users[0]["id"]) print(f"User {username} with id {userid} has been deleted") - shutil.rmtree(settings.personal / userid, ignore_errors=True) + if (settings.personal / userid).exists(): + remove_dataset(settings.personal / userid) return f"User deleted: {username}" @app.get("/api/listusers/") @@ -3825,11 +3992,6 @@ async def htmx_upload( # Read the file and check quota data = await file.read() - if settings.quota: - total_size = get_disk_usage() + len(data) - if total_size > settings.quota: - error = "Upload failed because quota limit has been exceeded." - return htmx_error(request, error) path.mkdir(exist_ok=True, parents=True) filename = pathlib.Path(file.filename) @@ -3839,6 +4001,42 @@ async def htmx_upload( suffix = filename.suffix suffixes = filename.suffixes[-2:] if suffix in [".tar", ".tgz", ".zip"] or suffixes == [".tar", ".gz"]: + if settings.quota: + # Admit encoded members independently. Never extract an archive into + # managed storage before measuring its final serialized files. + first = None + + def store_member(name, body): + nonlocal first + member = pathlib.Path(name) + if member.is_absolute() or ".." in member.parts: + raise ValueError("archive member escapes its destination") + if any(p.startswith((".", "__MACOSX")) for p in member.parts): + return + if member.suffix not in srv_utils.BLOSC2_NATIVE_SUFFIXES: + body = blosc2.SChunk(data=body).to_cframe() + member = member.with_suffix(member.suffix + ".b2") + write_dataset(path / member, body) + first = member if first is None else first + + if suffix == ".zip": + with zipfile.ZipFile(io.BytesIO(data)) as archive: + for member in archive.infolist(): + if not member.is_dir(): + if member.file_size > quota_coordinator().work_bytes: + raise storage_quota.QuotaExceeded("archive member exceeds staging budget") + store_member(member.filename, archive.read(member)) + else: + with tarfile.open(fileobj=io.BytesIO(data), mode="r:*") as archive: + for member in archive: + if member.isfile(): + if member.size > quota_coordinator().work_bytes: + raise storage_quota.QuotaExceeded("archive member exceeds staging budget") + store_member(member.name, archive.extractfile(member).read()) + elif not member.isdir(): + raise ValueError("archive links are not supported") + target = name if first is None else f"{name}/{first}" + return htmx_redirect(hx_current_url, make_url(request, "html_home", path=target), root=name) file.file.seek(0) # Reset file pointer if suffix == ".zip": with zipfile.ZipFile(file.file, "r") as archive: @@ -3890,8 +4088,7 @@ async def htmx_upload( filename = f"{filename}.b2" # Save file - with open(path / filename, "wb") as dst: - dst.write(data) + await concurrency.run_in_threadpool(write_dataset, path / filename, data) # Redirect to display new dataset path = f"{name}/{filename}" @@ -3936,7 +4133,7 @@ async def htmx_delete( if not abspath.exists(): return fastapi.HTTPException(status_code=404) - srv_utils.unlink_with_b2lock(abspath) + remove_dataset(abspath) # Redirect to home url = make_url(request, "html_home") diff --git a/caterva2/services/storage_quota.py b/caterva2/services/storage_quota.py new file mode 100644 index 00000000..7cbfb0e3 --- /dev/null +++ b/caterva2/services/storage_quota.py @@ -0,0 +1,418 @@ +"""Local, cross-process admission for immutable file replacements. + +The ledger charges regular dataset files in public/shared/personal by st_size. +SQLite, locks, and staging are operational storage; staging has its own budget. +No database transaction waits for an OS lock or performs filesystem/network I/O. +Writers publish complete files with os.replace, so existing readers keep a valid +snapshot. External filesystem writers are not part of this protocol. +""" + +from __future__ import annotations + +import contextlib +import hashlib +import json +import os +import pathlib +import sqlite3 +import time +import uuid + +ROOTS = {"public", "shared", "personal"} +WORK_BYTES = 1 << 30 + + +class QuotaExceeded(ValueError): + """The account or staging budget cannot admit this operation.""" + + +class StorageBusy(RuntimeError): + """A target changed or is already being mutated; retry from a new snapshot.""" + + +def signature(path): + try: + st = pathlib.Path(path).stat() + except FileNotFoundError: + return None + return (st.st_dev, st.st_ino, st.st_size, st.st_mtime_ns) + + +@contextlib.contextmanager +def file_lock(path, *, blocking=True, shared=False): + """An owner-death-released local lock; never remove its stable lock file.""" + with open(path, "a+b") as lock: + if os.name == "nt": + import msvcrt + + if lock.seek(0, 2) == 0: + lock.write(b"\0") + lock.flush() + lock.seek(0) + mode = msvcrt.LK_LOCK if blocking else msvcrt.LK_NBLCK + try: + msvcrt.locking(lock.fileno(), mode, 1) + except OSError as exc: + raise StorageBusy("storage object is busy") from exc + try: + yield + finally: + lock.seek(0) + msvcrt.locking(lock.fileno(), msvcrt.LK_UNLCK, 1) + else: + import fcntl + + try: + mode = fcntl.LOCK_SH if shared else fcntl.LOCK_EX + fcntl.flock(lock, mode | (0 if blocking else fcntl.LOCK_NB)) + except BlockingIOError as exc: + raise StorageBusy("storage object is busy") from exc + try: + yield + finally: + fcntl.flock(lock, fcntl.LOCK_UN) + + +def sync_directory(path): + if os.name != "nt": + fd = os.open(path, os.O_RDONLY) + try: + os.fsync(fd) + finally: + os.close(fd) + + +class StorageQuota: + def __init__(self, statedir, quota, *, work_bytes=WORK_BYTES): + self.input_root = pathlib.Path(statedir).absolute() + self.root = pathlib.Path(statedir).resolve() + if not isinstance(quota, int) or quota <= 0: + raise ValueError("quota must be a positive integer") + if not isinstance(work_bytes, int) or work_bytes <= 0: + raise ValueError("work_bytes must be a positive integer") + self.quota, self.work_bytes = quota, work_bytes + self.control = self.root / ".storage" + self.control.mkdir(parents=True, exist_ok=True) + self.dbpath = self.root / "storage.sqlite" + with self.startup_guard() as reconcile: + if not reconcile: + return # Active writers already protect a fully initialized ledger. + with self.connect() as db: + db.execute("PRAGMA journal_mode=WAL") + version = db.execute("PRAGMA user_version").fetchone()[0] + if version not in (0, 1): + raise RuntimeError("unsupported storage quota schema version") + db.executescript(""" + CREATE TABLE IF NOT EXISTS objects ( + path TEXT PRIMARY KEY, size INTEGER NOT NULL CHECK(size >= 0), + generation TEXT, cache INTEGER NOT NULL DEFAULT 0, + touched REAL NOT NULL DEFAULT 0); + CREATE TABLE IF NOT EXISTS operations ( + id TEXT PRIMARY KEY, path TEXT UNIQUE NOT NULL, + reserved INTEGER NOT NULL CHECK(reserved >= 0), + working INTEGER NOT NULL CHECK(working >= 0)); + CREATE TABLE IF NOT EXISTS account ( + id INTEGER PRIMARY KEY CHECK(id=1), quota INTEGER NOT NULL, + work_bytes INTEGER NOT NULL); + """) + initialized = db.execute("SELECT quota, work_bytes FROM account").fetchone() + if initialized is None: + # No coordinated writer can start until initialization is complete. + inventory = list(self.inventory()) + with self.transaction() as db: + db.executemany( + "INSERT OR REPLACE INTO objects(path,size,generation) VALUES(?,?,?)", inventory + ) + db.execute("INSERT INTO account VALUES(1,?,?)", (quota, work_bytes)) + db.execute("PRAGMA user_version=1") + elif initialized != (quota, work_bytes): + # A configuration change is shared by all workers. Admission + # reads the ledger value, never a stale worker-local quota. + with self.transaction() as db: + db.execute("UPDATE account SET quota=?, work_bytes=?", (quota, work_bytes)) + # Startup reconciliation covers offline edits and quota re-enablement. + # All publishes take this barrier shared; no scan runs in a DB txn. + self.recover() + inventory = list(self.inventory()) + with self.transaction() as db: + known = {row[0] for row in db.execute("SELECT path FROM objects")} + present = set() + for rel, size, generation in inventory: + present.add(rel) + db.execute( + "INSERT INTO objects(path,size,generation) VALUES(?,?,?) " + "ON CONFLICT(path) DO UPDATE SET size=excluded.size," + "cache=CASE WHEN objects.generation=excluded.generation THEN objects.cache ELSE 0 END," + "generation=excluded.generation", + (rel, size, generation), + ) + db.executemany("DELETE FROM objects WHERE path=?", ((rel,) for rel in known - present)) + + @contextlib.contextmanager + def startup_guard(self): + guard = file_lock(self.control / "initialize.lock", blocking=False) + try: + guard.__enter__() + except StorageBusy: + try: + with self.connect() as db: + ready = db.execute("SELECT quota,work_bytes FROM account").fetchone() + except sqlite3.Error: + ready = None + if ready is not None: + if ready != (self.quota, self.work_bytes): + raise StorageBusy( + "quota configuration change requires quiescent storage writers" + ) from None + yield False + else: + with file_lock(self.control / "initialize.lock"): + yield True + else: + try: + yield True + finally: + guard.__exit__(None, None, None) + + @contextlib.contextmanager + def connect(self): + db = sqlite3.connect(self.dbpath, timeout=2, isolation_level=None) + try: + db.execute("PRAGMA synchronous=FULL") + yield db + finally: + db.close() + + @contextlib.contextmanager + def transaction(self): + with self.connect() as db: + db.execute("BEGIN IMMEDIATE") + try: + yield db + db.execute("COMMIT") + except BaseException: + db.execute("ROLLBACK") + raise + + def relative(self, path): + path = pathlib.Path(path).absolute() + try: + rel = path.relative_to(self.root) + except ValueError: + try: + # Permit the configured state directory's own spelling (e.g. + # macOS /tmp -> /private/tmp), not symlinks inside dataset roots. + rel = path.relative_to(self.input_root) + except ValueError: + raise ValueError("storage target is outside the customer state directory") from None + if not rel.parts or rel.parts[0] not in ROOTS or ".." in rel.parts: + raise ValueError("storage target is outside the dataset roots") + if rel.name.endswith(".b2lock"): + raise ValueError("lock sidecars are reserved operational files") + # Reject symlinks, including parents, rather than account a different file. + current = self.root + for part in rel.parts: + current /= part + if current.is_symlink(): + raise ValueError("symlinks are not supported in quota-managed datasets") + return rel.as_posix() + + def inventory(self): + for name in sorted(ROOTS): + base = self.root / name + if not base.exists(): + continue + for path in base.rglob("*"): + if path.is_symlink(): + raise ValueError("remove dataset symlinks before enabling storage quota") + if path.is_file() and not path.name.endswith(".b2lock"): + rel = self.relative(path) + sig = signature(path) + yield rel, sig[2], json.dumps(sig) + + def lock(self, rel, *, blocking=True): + digest = hashlib.sha256(rel.encode()).hexdigest() + return file_lock(self.control / f"{digest}.lock", blocking=blocking) + + def snapshot(self, path): + rel = self.relative(path) + with self.lock(rel): + self._recover_path(rel) + sig = signature(path) + data = pathlib.Path(path).read_bytes() if sig is not None else None + return data, sig + + def _recover_path(self, rel): + """Caller owns the path lock: no previous owner can still publish.""" + with self.connect() as db: + op = db.execute("SELECT id FROM operations WHERE path=?", (rel,)).fetchone() + if op is None: + return + # Replacement is atomic: target is either the old or complete new file. + sig = signature(self.root / rel) + # A dead publisher may have renamed/unlinked without syncing the parent. + # Make that state durable before releasing its reservation. + parent = (self.root / rel).parent + if parent.exists(): + sync_directory(parent) + (self.control / f"{op[0]}.candidate").unlink(missing_ok=True) + sync_directory(self.control) + with self.transaction() as db: + self._record(db, rel, sig) + db.execute("DELETE FROM operations WHERE id=?", op) + + def recover(self): + with self.connect() as db: + paths = [row[0] for row in db.execute("SELECT path FROM operations")] + for rel in paths: + try: + with self.lock(rel, blocking=False): + self._recover_path(rel) + except StorageBusy: + continue # A live owner still holds its reservation; no TTL stealing. + + @staticmethod + def _record(db, rel, sig, *, cache=False): + if sig is None: + db.execute("DELETE FROM objects WHERE path=?", (rel,)) + else: + db.execute( + "INSERT INTO objects VALUES(?,?,?,?,?) ON CONFLICT(path) DO UPDATE SET " + "size=excluded.size,generation=excluded.generation,cache=excluded.cache," + "touched=excluded.touched", + (rel, sig[2], json.dumps(sig), int(cache), time.time()), + ) + + def usage(self): + with self.transaction() as db: + used = db.execute("SELECT coalesce(sum(size),0) FROM objects").fetchone()[0] + reserved, working = db.execute( + "SELECT coalesce(sum(reserved),0),coalesce(sum(working),0) FROM operations" + ).fetchone() + quota, budget = db.execute("SELECT quota,work_bytes FROM account").fetchone() + return {"used": used, "reserved": reserved, "working": working, "quota": quota, "work_bytes": budget} + + def publish(self, path, data, *, expected, cache=False, prune=True): + """Publish exact bytes (None deletes). A stale generation is never overwritten.""" + if data is not None and not isinstance(data, bytes): + raise TypeError("publish requires serialized bytes") + rel = self.relative(path) + for attempt in range(3): + try: + return self._publish(rel, data, expected, cache) + except QuotaExceeded: + if attempt == 0: + # Another worker may have died while this worker stays up. + # Never reclaim an operation whose OS lock is still owned. + self.recover() + elif not prune or attempt == 2 or not self.prune(exclude=rel): + raise + raise AssertionError("unreachable admission retry") + + def _publish(self, rel, data, expected, cache): + path = self.root / rel + with file_lock(self.control / "initialize.lock", shared=True), self.lock(rel): + self.relative(path) # Recheck parent symlinks after taking mutation guards. + self._recover_path(rel) + actual = signature(path) + if actual != expected: + raise StorageBusy("dataset changed while preparing its replacement") + oldsize = 0 if actual is None else actual[2] + size = 0 if data is None else len(data) + opid = uuid.uuid4().hex + with self.transaction() as db: + self._record(db, rel, actual, cache=cache) + used = db.execute("SELECT coalesce(sum(size),0) FROM objects").fetchone()[0] + reserved, working = db.execute( + "SELECT coalesce(sum(reserved),0),coalesce(sum(working),0) FROM operations" + ).fetchone() + quota, budget = db.execute("SELECT quota,work_bytes FROM account").fetchone() + growth = max(0, size - oldsize) + if (growth and used + reserved + growth > quota) or working + size > budget: + raise QuotaExceeded("customer quota or storage staging budget exceeded") + db.execute("INSERT INTO operations VALUES(?,?,?,?)", (opid, rel, growth, size)) + candidate = self.control / f"{opid}.candidate" + # From this point, any failure leaves durable intent for recovery. + if data is None: + path.unlink(missing_ok=True) + else: + fd = os.open(candidate, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + with os.fdopen(fd, "wb") as file: + file.write(data) + file.flush() + os.fsync(file.fileno()) + # Persist each new directory entry before publishing into it. + missing = [] + parent = path.parent + while not parent.exists(): + missing.append(parent) + parent = parent.parent + for directory in reversed(missing): + directory.mkdir(exist_ok=True) + sync_directory(directory.parent) + sync_directory(self.control) + os.replace(candidate, path) + sync_directory(path.parent) + sync_directory(self.control) + sig = signature(path) + with self.transaction() as db: + self._record(db, rel, sig, cache=cache) + db.execute("DELETE FROM operations WHERE id=?", (opid,)) + return sig + + def touch(self, path): + rel = self.relative(path) + now = time.time() + with self.transaction() as db: + db.execute( + "UPDATE objects SET touched=?,cache=1 WHERE path=? AND touched self.work_bytes: + continue + frame = path.read_bytes() + # Work on an immutable snapshot, then compare-and-swap on publish. + carrier = blosc2.ndarray_from_cframe(frame, copy=True) + payload = carrier.schunk.vlmeta.get("b2o") + marker = carrier.schunk.meta.get("b2o") + if marker != {"kind": "remote_proxy", "version": 1}: + continue + if not isinstance(payload, dict) or payload.get("kind") != "remote_proxy": + continue + if set(payload) != {"kind", "version", "source", "cache_policy", "max_cache_bytes"}: + continue + if payload.get("cache_policy") != "disk": + continue + for chunk in range(carrier.schunk.nchunks): + carrier.schunk.update_special(chunk, blosc2.SpecialValue.UNINIT) + for key in tuple(carrier.schunk.vlmeta): + if key in blosc2.proxy._RESERVED_VLMETA: + del carrier.schunk.vlmeta[key] + cold = carrier.to_cframe() + if len(cold) >= len(frame): + continue + self.publish(path, cold, expected=expected, cache=False, prune=False) + reclaimed += len(frame) - len(cold) + except (StorageBusy, QuotaExceeded, OSError, RuntimeError, ValueError): + continue + return reclaimed diff --git a/caterva2/tests/test_chunk_writes.py b/caterva2/tests/test_chunk_writes.py index 527b627e..e8033818 100644 --- a/caterva2/tests/test_chunk_writes.py +++ b/caterva2/tests/test_chunk_writes.py @@ -313,11 +313,16 @@ def test_a_filled_array_is_published_by_itself(presized): assert answer["written"] == NCHUNKS assert answer["state"] == "publishing" + nonce = presized.vlmeta["fill_nonce"] for _ in range(100): # the upload runs after the response published = _published("run.b2nd") if published is not None: - break + frame = blosc2.open(str(published)) + if frame.schunk.vlmeta.get("fill_nonce") == nonce: + break time.sleep(0.05) + else: + pytest.fail("the current fill was not published before the deadline") assert published is not None assert published.is_file() diff --git a/caterva2/tests/test_storage_quota.py b/caterva2/tests/test_storage_quota.py new file mode 100644 index 00000000..eea9ce5c --- /dev/null +++ b/caterva2/tests/test_storage_quota.py @@ -0,0 +1,303 @@ +"""Physical admission, independent processes, and filesystem/SQLite recovery.""" + +import concurrent.futures +import contextlib +import multiprocessing +import os +import pathlib +import sqlite3 + +import blosc2 +import fsspec +import numpy as np +import pytest + +from caterva2.services import remote_proxy, storage_quota + + +def writer_process(root, name, size, ready, release, result): + quota = storage_quota.StorageQuota(root, 100, work_bytes=1000) + original = os.replace + + def paused_replace(src, dst): + ready.set() + assert release.wait(10) + return original(src, dst) + + os.replace = paused_replace + try: + quota.publish(pathlib.Path(root) / "public" / name, b"x" * size, expected=None, prune=False) + result.put("ok") + except storage_quota.QuotaExceeded: + result.put("denied") + + +def crash_process(root, after_replace): + quota = storage_quota.StorageQuota(root, 1000, work_bytes=1000) + original = os.replace + + def crash(src, dst): + if after_replace: + original(src, dst) + os._exit(23) + + os.replace = crash + path = pathlib.Path(root) / "public" / "data" + _, generation = quota.snapshot(path) + quota.publish(path, b"new" * 20, expected=generation) + + +def test_exact_size_admission_and_replacement(tmp_path): + quota = storage_quota.StorageQuota(tmp_path, 100, work_bytes=1000) + path = tmp_path / "public" / "a" + generation = quota.publish(path, b"a" * 70, expected=None) + assert quota.usage()["used"] == 70 + with pytest.raises(storage_quota.QuotaExceeded): + quota.publish(tmp_path / "shared" / "b", b"b" * 50, expected=None) + quota.publish(path, b"a" * 90, expected=generation) + assert quota.usage()["used"] == 90 + quota.publish(path, None, expected=storage_quota.signature(path)) + assert quota.usage()["used"] == 0 + assert quota.usage()["reserved"] == 0 + + +def test_independent_processes_cannot_spend_same_capacity(tmp_path): + quota = storage_quota.StorageQuota(tmp_path, 100, work_bytes=1000) + ctx = multiprocessing.get_context("spawn") + ready, release, result = ctx.Event(), ctx.Event(), ctx.Queue() + first = ctx.Process(target=writer_process, args=(str(tmp_path), "a", 70, ready, release, result)) + first.start() + try: + assert ready.wait(10) + assert quota.usage()["reserved"] == 70 + # Recovery cannot steal a live owner's reservation, regardless of elapsed time. + quota.recover() + assert quota.usage()["reserved"] == 70 + other = storage_quota.StorageQuota(tmp_path, 100, work_bytes=1000) + with pytest.raises(storage_quota.QuotaExceeded): + other.publish(tmp_path / "public" / "b", b"b" * 50, expected=None, prune=False) + finally: + release.set() + first.join(10) + if first.is_alive(): + first.terminate() + first.join() + assert first.exitcode == 0 + assert result.get(timeout=2) == "ok" + assert quota.usage()["used"] == 70 + result.close() + result.join_thread() + + +@pytest.mark.parametrize("after_replace", [False, True]) +def test_recovery_after_worker_death(tmp_path, after_replace): + quota = storage_quota.StorageQuota(tmp_path, 1000, work_bytes=1000) + path = tmp_path / "public" / "data" + quota.publish(path, b"old", expected=None) + child = multiprocessing.get_context("spawn").Process( + target=crash_process, args=(str(tmp_path), after_replace) + ) + child.start() + child.join(10) + assert child.exitcode == 23 + assert quota.usage()["reserved"] > 0 + recovered = storage_quota.StorageQuota(tmp_path, 1000, work_bytes=1000) + assert path.read_bytes() == (b"new" * 20 if after_replace else b"old") + assert recovered.usage()["used"] == path.stat().st_size + assert recovered.usage()["reserved"] == recovered.usage()["working"] == 0 + assert not list(recovered.control.glob("*.candidate")) + recovered.recover() + assert recovered.usage()["used"] == path.stat().st_size + + +def test_failed_publication_reconciles_before_retry(tmp_path, monkeypatch): + quota = storage_quota.StorageQuota(tmp_path, 1000) + path = tmp_path / "public" / "data" + original = os.replace + + def fail(*args): + raise OSError("simulated disk error") + + monkeypatch.setattr(os, "replace", fail) + with pytest.raises(OSError): + quota.publish(path, b"hello", expected=None) + monkeypatch.setattr(os, "replace", original) + quota.publish(path, b"recovered", expected=None) + assert quota.usage()["used"] == len(b"recovered") + assert quota.usage()["reserved"] == 0 + + +def test_generation_conflict_preserves_newer_data(tmp_path): + quota = storage_quota.StorageQuota(tmp_path, 1000) + path = tmp_path / "public" / "data" + generation = quota.publish(path, b"first", expected=None) + quota.publish(path, b"second", expected=generation) + with pytest.raises(storage_quota.StorageBusy): + quota.publish(path, b"stale", expected=generation) + assert path.read_bytes() == b"second" + + +def test_staging_budget_is_independent_of_data_quota(tmp_path): + quota = storage_quota.StorageQuota(tmp_path, 1000, work_bytes=50) + with pytest.raises(storage_quota.QuotaExceeded): + quota.publish(tmp_path / "public" / "large", b"x" * 51, expected=None) + assert quota.usage()["reserved"] == 0 + + +def test_open_reader_keeps_old_snapshot_after_replacement(tmp_path): + quota = storage_quota.StorageQuota(tmp_path, 1000) + path = tmp_path / "public" / "data" + generation = quota.publish(path, b"old snapshot", expected=None) + with path.open("rb") as reader: + quota.publish(path, b"new snapshot", expected=generation) + assert reader.read() == b"old snapshot" + assert path.read_bytes() == b"new snapshot" + + +def test_restart_reconciles_offline_edits_and_reduced_quota(tmp_path): + quota = storage_quota.StorageQuota(tmp_path, 1000) + path = tmp_path / "public" / "data" + quota.publish(path, b"before", expected=None) + path.write_bytes(b"offline replacement") + restarted = storage_quota.StorageQuota(tmp_path, 10) + assert restarted.usage()["used"] == len(b"offline replacement") + with pytest.raises(storage_quota.QuotaExceeded): + restarted.publish(tmp_path / "public" / "new", b"x", expected=None) + restarted.publish(path, b"small", expected=storage_quota.signature(path)) + assert restarted.usage()["used"] == 5 + + +def test_inventory_excludes_operational_and_peer_storage(tmp_path): + for root in ("public", "personal", "shared", "peercache", "media"): + (tmp_path / root).mkdir() + (tmp_path / root / "data").write_bytes(b"123") + (tmp_path / "public" / "data.b2lock").write_bytes(b"lock") + quota = storage_quota.StorageQuota(tmp_path, 5) + assert quota.usage()["used"] == 9 + with pytest.raises(storage_quota.QuotaExceeded): + quota.publish(tmp_path / "public" / "new", b"x", expected=None) + quota.publish( + tmp_path / "public" / "data", None, expected=storage_quota.signature(tmp_path / "public" / "data") + ) + assert quota.usage()["used"] == 6 + + +def test_symlinks_and_path_escape_rejected(tmp_path): + quota = storage_quota.StorageQuota(tmp_path, 1000) + with pytest.raises(ValueError): + quota.publish(tmp_path / "public" / ".." / "outside", b"x", expected=None) + (tmp_path / "public").mkdir() + (tmp_path / "public" / "link").symlink_to(tmp_path) + with pytest.raises(ValueError): + quota.publish(tmp_path / "public" / "link" / "outside", b"x", expected=None) + + +def test_configured_state_directory_alias_is_supported(tmp_path): + root = tmp_path / "real" + root.mkdir() + alias = tmp_path / "alias" + alias.symlink_to(root, target_is_directory=True) + quota = storage_quota.StorageQuota(alias, 1000) + path = alias / "public" / "data" + quota.publish(path, b"hello", expected=None) + assert quota.snapshot(path)[0] == b"hello" + assert quota.usage()["used"] == 5 + + +def remote_fixture(tmp_path, name="proxy", *, limit=None, block=10000): + data = np.random.default_rng(1).integers(0, 256, 30000, dtype="u1") + array = blosc2.asarray(data, chunks=(10000,), blocks=(block,)) + url = f"memory://quota-{name}.b2nd" + fsspec.filesystem("memory").pipe_file(f"quota-{name}.b2nd", array.to_cframe()) + path = tmp_path / "public" / f"{name}.b2nd" + path.parent.mkdir(exist_ok=True) + creator = blosc2.RemoteProxy( + url, cache_policy=blosc2.CachePolicy.DISK, cache_path=path, max_cache_bytes=limit + ) + creator.schunk.vlmeta["user-note"] = "preserve me" + carrier, payload = remote_proxy.inspect(path) + proxy = remote_proxy.ServerRemoteProxy( + creator.src, (array.shape, array.dtype, array.chunks, array.blocks), carrier, payload + ) + return proxy, path, data + + +@pytest.mark.parametrize("limit", [None, 15000]) +def test_proxy_grows_with_quota_and_reuses_data(tmp_path, limit): + proxy, path, data = remote_fixture(tmp_path, limit=limit) + quota = storage_quota.StorageQuota(tmp_path, 100000) + before = path.stat().st_size + np.testing.assert_array_equal(proxy.quota_read(quota, slice(0, 10000)), data[:10000]) + assert path.stat().st_size > before + assert quota.usage()["used"] == path.stat().st_size + proxy.src.traffic.reset() + np.testing.assert_array_equal(proxy.quota_read(quota, slice(0, 10000)), data[:10000]) + assert proxy.src.traffic.requests == 0 + proxy.quota_read(quota, nchunk=1) + assert quota.usage()["used"] <= quota.quota + + +def test_proxy_denied_retention_still_returns_data(tmp_path): + proxy, path, data = remote_fixture(tmp_path) + before = path.read_bytes() + quota = storage_quota.StorageQuota(tmp_path, len(before) + 10032) + # A compressed chunk fits this allowance, its full carrier metadata does not. + np.testing.assert_array_equal(proxy.quota_read(quota, slice(0, 10000)), data[:10000]) + assert path.read_bytes() == before + assert quota.usage()["used"] == len(before) + + +def test_partial_block_fill_then_whole_chunk_is_consistent(tmp_path): + proxy, path, data = remote_fixture(tmp_path, block=2500) + quota = storage_quota.StorageQuota(tmp_path, 100000) + np.testing.assert_array_equal(proxy.quota_read(quota, slice(0, 200)), data[:200]) + proxy.src.traffic.reset() + np.testing.assert_array_equal(proxy.quota_read(quota, slice(0, 200)), data[:200]) + assert proxy.src.traffic.requests == 0 + result = proxy.quota_read(quota, nchunk=0) + np.testing.assert_array_equal(np.frombuffer(blosc2.decompress(result), dtype="u1"), data[:10000]) + assert quota.usage()["used"] == path.stat().st_size + + +def test_cross_proxy_pruning_preserves_descriptor_and_metadata(tmp_path): + left, lp, data = remote_fixture(tmp_path, "left") + right, rp, _ = remote_fixture(tmp_path, "right") + quota = storage_quota.StorageQuota(tmp_path, lp.stat().st_size + rp.stat().st_size + 12000) + left.quota_read(quota, slice(0, 10000)) + left.src.traffic.reset() + right.quota_read(quota, slice(0, 10000)) + assert quota.usage()["used"] <= quota.quota + raw = remote_proxy.raw_carrier(lp) + assert raw.schunk.vlmeta["user-note"] == "preserve me" + assert raw.schunk.vlmeta["b2o"]["cache_policy"] == "disk" + assert not raw.schunk.vlmeta.get("proxy-fetched") + np.testing.assert_array_equal(left.quota_read(quota, slice(0, 10000)), data[:10000]) + assert left.src.traffic.requests > 0 + + +def test_parallel_proxy_and_upload_share_quota(tmp_path): + proxy, path, data = remote_fixture(tmp_path) + quota = storage_quota.StorageQuota(tmp_path, path.stat().st_size + 12000) + upload = tmp_path / "shared" / "upload" + + def write(): + with contextlib.suppress(storage_quota.QuotaExceeded): + quota.publish(upload, b"x" * 11000, expected=None, prune=False) + + with concurrent.futures.ThreadPoolExecutor(2) as pool: + future = pool.submit(write) + np.testing.assert_array_equal(proxy.quota_read(quota, slice(0, 10000)), data[:10000]) + future.result() + measured = path.stat().st_size + (upload.stat().st_size if upload.exists() else 0) + assert quota.usage()["used"] == measured <= quota.quota + + +def test_sqlite_failure_does_not_turn_cache_miss_into_read_failure(tmp_path, monkeypatch): + proxy, _, data = remote_fixture(tmp_path) + quota = storage_quota.StorageQuota(tmp_path, 100000) + + def fail(*args, **kwargs): + raise sqlite3.OperationalError("busy") + + monkeypatch.setattr(quota, "publish", fail) + np.testing.assert_array_equal(proxy.quota_read(quota, slice(0, 10000)), data[:10000]) diff --git a/caterva2/tests/test_storage_quota_api.py b/caterva2/tests/test_storage_quota_api.py new file mode 100644 index 00000000..7ba3d4fb --- /dev/null +++ b/caterva2/tests/test_storage_quota_api.py @@ -0,0 +1,225 @@ +"""Quota-controlled writers and remote fetches through the actual ASGI routes.""" + +import io +import pathlib +import types +import uuid + +import blosc2 +import fsspec +import h5py +import httpx +import numpy as np +import pytest +import pytest_asyncio + +from caterva2.services import remote_proxy + + +@pytest_asyncio.fixture +async def quota_api(tmp_path, monkeypatch): + monkeypatch.setenv("CATERVA2_SECRET", "quota-test-secret") + from caterva2.services import server + + user = types.SimpleNamespace(id=uuid.uuid4(), is_superuser=True, is_active=True) + monkeypatch.setattr(server.settings, "statedir", tmp_path) + monkeypatch.setattr(server.settings, "quota", 100_000) + monkeypatch.setattr(server.settings, "publish_root", None) + monkeypatch.setattr(server, "_quota_instances", {}) + monkeypatch.setattr(remote_proxy, "policy", remote_proxy.policy) + for name in ("public", "shared", "personal"): + path = tmp_path / name + path.mkdir() + monkeypatch.setattr(server.settings, name, path) + overrides = dict(server.app.dependency_overrides) + server.app.dependency_overrides[server.current_active_user] = lambda: user + server.app.dependency_overrides[server.optional_user] = lambda: user + try: + async with httpx.AsyncClient( + transport=httpx.ASGITransport(app=server.app), base_url="http://test" + ) as client: + yield server, client, user + finally: + server.app.dependency_overrides.clear() + server.app.dependency_overrides.update(overrides) + + +def assert_usage(server): + quota = server.quota_coordinator() + measured = sum(row[1] for row in quota.inventory()) + assert quota.usage()["used"] == measured <= quota.usage()["quota"] + assert quota.usage()["reserved"] == quota.usage()["working"] == 0 + + +@pytest.mark.asyncio +async def test_upload_copy_append_remove_and_quota_denial(quota_api): + server, client, _ = quota_api + array = blosc2.arange(20, chunks=(10,), blocks=(5,)) + response = await client.post("/api/upload/@public/a.b2nd", files={"file": ("a.b2nd", array.to_cframe())}) + assert response.status_code == 200, response.text + response = await client.post("/api/copy/", json={"src": "@public/a.b2nd", "dst": "@shared/b.b2nd"}) + assert response.status_code == 200, response.text + response = await client.post("/api/append/@public/a.b2nd", files={"file": ("a.b2nd", array.to_cframe())}) + assert response.status_code == 200, response.text + np.testing.assert_array_equal( + blosc2.open(server.settings.public / "a.b2nd")[:], np.tile(np.arange(20), 2) + ) + assert_usage(server) + response = await client.post( + "/api/upload/@public/too-big.b2nd", files={"file": ("large", b"x" * 100_001)} + ) + assert response.status_code == 400 + assert not (server.settings.public / "too-big.b2nd").exists() + response = await client.post("/api/remove/@shared/b.b2nd") + assert response.status_code == 200, response.text + assert_usage(server) + + +@pytest.mark.asyncio +async def test_chunk_writes_charge_metadata_and_keep_generation(quota_api): + server, client, _ = quota_api + empty = blosc2.uninit(20, dtype="i4", chunks=(10,), blocks=(5,)) + source = blosc2.arange(20, dtype="i4", chunks=(10,), blocks=(5,)) + response = await client.post( + "/api/upload/@public/fill.b2nd", files={"file": ("fill.b2nd", empty.to_cframe())} + ) + assert response.status_code == 200 + for i in range(2): + response = await client.post( + "/api/chunk/@public/fill.b2nd", params={"nchunk": i}, content=source.schunk.get_chunk(i) + ) + assert response.status_code == 200, response.text + np.testing.assert_array_equal(blosc2.open(server.settings.public / "fill.b2nd")[:], np.arange(20)) + assert_usage(server) + response = await client.post( + "/api/chunk/@public/fill.b2nd", params={"nchunk": 0}, content=source.schunk.get_chunk(0) + ) + assert response.status_code == 409 + + +@pytest.mark.asyncio +async def test_expression_and_notebook_use_admission(quota_api): + server, client, user = quota_api + array = blosc2.arange(20, chunks=(10,), blocks=(5,)) + await client.post("/api/upload/@public/a.b2nd", files={"file": ("a.b2nd", array.to_cframe())}) + expr = types.SimpleNamespace( + name="expr", expression="a + 1", operands={"a": "@public/a.b2nd"}, func=None, compute=False + ) + server.make_expr(expr, user) + result = blosc2.open(server.settings.personal / str(user.id) / "expr.b2nd") + np.testing.assert_array_equal(result[:], np.arange(20) + 1) + expr.compute = True + server.make_expr(expr, user) + response = await client.post("/api/addnotebook/@public/new.ipynb") + assert response.status_code == 200, response.text + assert_usage(server) + + +@pytest.mark.asyncio +async def test_legacy_python_udf_uses_in_memory_serialization(quota_api): + server, client, user = quota_api + array = blosc2.arange(20, dtype="f8", chunks=(10,), blocks=(5,)) + server.write_dataset(server.settings.public / "a.b2nd", array.to_cframe()) + expr = types.SimpleNamespace( + name="legacy", + expression=None, + operands={"o0": "@public/a.b2nd"}, + compute=False, + dtype=np.dtype("f8"), + shape=(20,), + func="def legacy(inputs, output, offset):\n output[:] = np.logaddexp(inputs[0], 1)\n", + ) + server.make_expr(expr, user) + response = await client.get("/api/fetch/@personal/legacy.b2nd") + assert response.status_code == 200, response.text + np.testing.assert_allclose( + blosc2.ndarray_from_cframe(response.content)[:], np.logaddexp(np.arange(20), 1) + ) + assert_usage(server) + + +@pytest.mark.asyncio +async def test_hdf5_unfold_admits_each_proxy(quota_api): + server, client, _ = quota_api + buffer = io.BytesIO() + with h5py.File(buffer, "w") as file: + file.create_dataset("group/data", data=np.arange(20)) + response = await client.post("/api/upload/@public/a.h5", files={"file": ("a.h5", buffer.getvalue())}) + assert response.status_code == 200 + response = await client.post("/api/unfold/@public/a.h5") + assert response.status_code == 200, response.text + assert (server.settings.public / "a/group/data.b2nd").exists() + assert_usage(server) + + +@pytest.mark.asyncio +async def test_local_publish_cannot_bypass_managed_storage(quota_api, monkeypatch): + server, _, _ = quota_api + path = server.settings.public / "source.b2nd" + server.write_dataset(path, blosc2.arange(20).to_cframe()) + monkeypatch.setattr(server.settings, "publish_root", server.settings.public.as_uri()) + with pytest.raises(server.fastapi.HTTPException) as error: + server.publish_dataset(path, pathlib.Path("copy.b2nd")) + assert error.value.status_code == 400 + assert not (server.settings.public / "copy.b2nd").exists() + assert_usage(server) + + +@pytest.mark.asyncio +async def test_move_does_not_delete_a_concurrent_source_replacement(quota_api, monkeypatch): + server, _, _ = quota_api + source, destination = server.settings.public / "source", server.settings.shared / "dest" + server.write_dataset(source, b"original") + writer = server.write_dataset + + def racing_writer(path, data): + writer(path, data) + writer(source, b"newer data") + + monkeypatch.setattr(server, "write_dataset", racing_writer) + with pytest.raises(server.storage_quota.StorageBusy): + server.move_dataset(source, destination) + assert destination.read_bytes() == b"original" + assert source.read_bytes() == b"newer data" + assert_usage(server) + + +@pytest.mark.asyncio +async def test_disk_fetch_and_chunk_admit_growth_via_secure_source(quota_api, monkeypatch): + server, client, _ = quota_api + data = np.random.default_rng(1).integers(0, 256, 30000, dtype="u1") + array = blosc2.asarray(data, chunks=(10000,), blocks=(10000,)) + fs = fsspec.filesystem("memory") + fs.pipe_file("quota-api-source.b2nd", array.to_cframe()) + creator = blosc2.RemoteProxy("memory://quota-api-source.b2nd", cache_policy=blosc2.CachePolicy.MEMORY) + carrier = blosc2.ndarray_from_cframe(creator.to_cframe(cache_policy=blosc2.CachePolicy.DISK), copy=True) + payload = dict(carrier.schunk.vlmeta["b2o"]) + url = "https://data.example/quota.b2nd" + payload["source"] = {"kind": "fsspec", "version": 1, "urlpath": url} + carrier.schunk.vlmeta["b2o"] = payload + fs.pipe_file(url, array.to_cframe()) + monkeypatch.setattr( + remote_proxy, "policy", remote_proxy.Policy(enabled=True, allowed_hosts=("data.example",)) + ) + monkeypatch.setattr(remote_proxy, "_public_addresses", lambda *a: ("93.184.216.34",)) + monkeypatch.setattr(remote_proxy, "_https_filesystem", lambda *a: fs) + response = await client.post( + "/api/upload/@public/proxy.b2nd", files={"file": ("proxy.b2nd", carrier.to_cframe())} + ) + assert response.status_code == 200 + path = server.settings.public / "proxy.b2nd" + before = path.stat().st_size + response = await client.get("/api/fetch/@public/proxy.b2nd", params={"slice_": "0:10000"}) + assert response.status_code == 200, response.text + np.testing.assert_array_equal(blosc2.ndarray_from_cframe(response.content)[:], data[:10000]) + assert path.stat().st_size > before + response = await client.get("/api/chunk/@public/proxy.b2nd", params={"nchunk": 1}) + assert response.status_code == 200, response.text + np.testing.assert_array_equal( + np.frombuffer(blosc2.decompress(response.content), dtype="u1"), data[10000:20000] + ) + assert_usage(server) + response = await client.get("/api/download/@public/proxy.b2nd", params={"include_cache": "false"}) + assert response.status_code == 200 + cold = blosc2.ndarray_from_cframe(response.content) + assert cold.schunk.vlmeta["b2o"] == payload diff --git a/doc/utilities/cat2-server.md b/doc/utilities/cat2-server.md index 4df5a1e7..fb71b90e 100644 --- a/doc/utilities/cat2-server.md +++ b/doc/utilities/cat2-server.md @@ -41,7 +41,8 @@ path as `NONE`), avoiding unmanaged memory use on the server while preserving the requested client limit for download. With a `DISK` cache, fetched compressed chunks are retained inside the carrier up to the proxy's `max_cache_bytes` (or unbounded when `max_cache_bytes` is `None`) when no customer quota is configured. -With a quota, existing warm disk chunks are reused but misses are not retained. +With a quota, shared storage admission permits cache growth when capacity is +available; otherwise misses are returned without retention. Caterva2 can inspect and report its stored shape, dtype, chunk, block, and proxy metadata without contacting the source. Outbound resolution is disabled by default. @@ -70,13 +71,11 @@ created by a client that performed its own validation. The limits validate the remote array's structure and bound connection time and concurrent range fetches. They do not impose a network-work budget on each API request. The proxy's own cache limit bounds its retained compressed payload, -while a configured customer `quota` makes all DISK proxy caches read-only. -Valid warm chunks are reused, but misses are served without retention, even when -quota space remains. This conservative restriction prevents cache fills from -overspending physical storage through metadata growth or concurrent workers. -Automatic fills under quota require a future storage reservation mechanism shared -with uploads and other writers. Without a quota, normal bounded or unbounded -DISK caching applies. +while a configured customer `quota` additionally bounds stored dataset bytes, +including carrier metadata. A SQLite ledger shared with ordinary writers admits +the exact serialized replacement size before disk publication. A denied fill +does not fail a successfully fetched result. Without a quota, normal bounded or +unbounded DISK caching applies. Public S3 objects are supported through credential-free HTTPS object URLs. Native `s3://` resolution, private-source credentials, and remote references @@ -85,3 +84,51 @@ embedded inside persisted expressions are not enabled. Physical downloads include valid warm proxy chunks by default. Clients can pass `include_cache=false` to download a cold carrier without mutating the hosted proxy. Logical `api/fetch` requests continue to return array data. + +## Customer storage admission + +With `[server] quota` enabled, `storage.sqlite` coordinates workers sharing one +customer's local state directory. It uses Python's standard-library `sqlite3`, +independently of authentication; no additional dependency is needed. Multi-host +or network-filesystem sharing is not supported. All workers must use the same +configuration, and configuration changes require quiescent writers. + +The quota charges regular dataset files in `public`, `shared`, and `personal` by +their apparent length (`st_size`), including proxy headers and metadata. This is +an explicit change from the old whole-state-directory scan: peer-cache files +keep their separate `peer_cache_quota`; media, authentication state, SQLite/WAL, +directories and lock sidecars are operational storage, not customer data charges. +Dataset symlinks are rejected. Local `publish_root` must be outside the state +directory. Administrators must provision operational headroom separately. + +Uploads, imports, expressions, append/chunk writes, HDF5 unfolding, notebooks, +copies, moves, deletions, and remote DISK fills share admission. Under pressure, +admission may cold-replace up to four previously validated DISK cache carriers, +oldest first, preserving their descriptors and user metadata. Ordinary datasets +are never automatically pruned. A proxy's own payload cap still applies. + +The first implementation builds replacements in memory, reserves exact growth, +writes a complete staging file, and atomically replaces the target. It is a +correctness-oriented path with whole-file write amplification, not an in-place +optimization. `[server] quota_work_bytes` (default `"1G"`) bounds the aggregate +reserved disk staging bytes separately from customer quota. A candidate larger +than this budget cannot be persisted, even if customer quota has room. This is +not a hard RAM limit or a bound on HTTP upload spooling, SQLite/lock overhead, +filesystem allocated blocks, or old inodes retained by open readers. Provision +and monitor the underlying volume accordingly. + +Per-path OS locks and generation checks prevent stale candidates from replacing +newer data. Durable operation records survive worker death; recovery inspects the +atomic target and reclaims staging only after acquiring the dead owner's lock. +Startup reconciles offline changes. Do not edit managed files outside Caterva2 +while the server is running; external writers cannot be covered by admission. +If usage exceeds a reduced quota, reads and shrinking operations remain allowed +but positive growth is denied. Ledger failure never permits an unaccounted write. + +Directory/archive operations are admitted file by file, not as one transaction; +an error may leave earlier files completed. Moves currently copy then delete and +therefore need capacity for both copies. Ordinary quota denial returns HTTP 400, +generation conflicts return 409, and ledger errors return 503. Remote reads may +instead fall back to no retention. `StorageQuota.usage()` exposes committed, +reserved and disk working bytes for internal diagnostics; there is no new public +administration endpoint. diff --git a/examples/benchmark_storage_quota.py b/examples/benchmark_storage_quota.py new file mode 100644 index 00000000..5a71c32d --- /dev/null +++ b/examples/benchmark_storage_quota.py @@ -0,0 +1,69 @@ +"""Compare staged quota fills with the existing non-quota in-place path. + +Run from the checkout with ``python examples/benchmark_storage_quota.py``. +This local-memory upstream microbenchmark isolates storage overhead, not HTTP. +In-place timings are a baseline, not a quota-safe alternative implementation. +""" + +import pathlib +import tempfile +import time + +import blosc2 +import fsspec +import numpy as np + +from caterva2.services import remote_proxy, storage_quota + + +def benchmark(root, data, chunk, staged): + source = blosc2.asarray(data, chunks=(chunk,), blocks=(chunk // 4,)) + url = "memory://quota-benchmark.b2nd" + fsspec.filesystem("memory").pipe_file("quota-benchmark.b2nd", source.to_cframe()) + path = root / "public" / "proxy.b2nd" + path.parent.mkdir(parents=True) + creator = blosc2.RemoteProxy( + url, cache_policy=blosc2.CachePolicy.DISK, cache_path=path, max_cache_bytes=None + ) + carrier, payload = remote_proxy.inspect(path) + proxy = remote_proxy.ServerRemoteProxy( + creator.src, (source.shape, source.dtype, source.chunks, source.blocks), carrier, payload + ) + quota = storage_quota.StorageQuota(root, data.nbytes * 4) if staged else None + initial_bytes = path.stat().st_size + samples = [] + staged_bytes = 0 + for start in range(0, len(data), chunk): + selection = slice(start, start + chunk) + tick = time.perf_counter() + result = proxy.quota_read(quota, selection) if staged else proxy.read(selection) + samples.append(time.perf_counter() - tick) + np.testing.assert_array_equal(result, data[selection]) + if staged: + staged_bytes += path.stat().st_size + tick = time.perf_counter() + for start in range(0, len(data), chunk): + selection = slice(start, start + chunk) + if staged: + proxy.quota_read(quota, selection) + else: + proxy.read(selection) + warm_ms = (time.perf_counter() - tick) * 1000 / len(samples) + assert path.stat().st_size > initial_bytes, "benchmark did not retain any chunks" + return { + "chunk_bytes": chunk, + "staged": staged, + "cold_ms_per_fill": round(float(np.mean(samples)) * 1000, 3), + "warm_ms_per_read": round(warm_ms, 3), + "final_bytes": path.stat().st_size, + "candidate_bytes_written": staged_bytes if staged else None, + } + + +if __name__ == "__main__": + data = np.random.default_rng(1).integers(0, 256, 8 << 20, dtype="u1") + with tempfile.TemporaryDirectory(prefix="caterva2-quota-bench-") as directory: + for chunk in (256 << 10, 1 << 20): + for staged in (False, True): + root = pathlib.Path(directory) / f"{chunk}-{staged}" + print(benchmark(root, data, chunk, staged)) From bad629dbc57b8587e5a6970b136eaeaeaec39d1e Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 6 Sep 2026 14:24:05 +0200 Subject: [PATCH 06/20] Use sparse remote proxy caches by default --- caterva2-server.sample.toml | 12 +- caterva2/services/remote_proxy.py | 17 + caterva2/services/server.py | 126 +++- caterva2/services/sparse_cache.py | 687 ++++++++++++++++++ caterva2/services/storage_quota.py | 144 ++-- caterva2/tests/test_sparse_cache.py | 293 ++++++++ caterva2/tests/test_storage_quota_api.py | 73 +- doc/utilities/cat2-server.md | 86 ++- examples/benchmarks/remote_proxy_v7.md | 58 ++ examples/benchmarks/remote_proxy_v7.py | 155 ++++ .../large-contiguous.json | 197 +++++ .../remote_proxy_v7_results/large-sparse.json | 233 ++++++ .../small-contiguous.json | 197 +++++ .../remote_proxy_v7_results/small-sparse.json | 233 ++++++ 14 files changed, 2407 insertions(+), 104 deletions(-) create mode 100644 caterva2/services/sparse_cache.py create mode 100644 caterva2/tests/test_sparse_cache.py create mode 100644 examples/benchmarks/remote_proxy_v7.md create mode 100644 examples/benchmarks/remote_proxy_v7.py create mode 100644 examples/benchmarks/remote_proxy_v7_results/large-contiguous.json create mode 100644 examples/benchmarks/remote_proxy_v7_results/large-sparse.json create mode 100644 examples/benchmarks/remote_proxy_v7_results/small-contiguous.json create mode 100644 examples/benchmarks/remote_proxy_v7_results/small-sparse.json diff --git a/caterva2-server.sample.toml b/caterva2-server.sample.toml index dc2f14cc..4759feec 100644 --- a/caterva2-server.sample.toml +++ b/caterva2-server.sample.toml @@ -33,15 +33,17 @@ register = true # allow users to register # peer_cache_quota = "1G" # Persisted RemoteProxy objects are discoverable but cannot make outbound -# requests by default. The first opt-in backend is credential-free HTTPS. +# requests by default. The runtime cache uses private sparse generations and +# credential-free HTTPS is the supported source transport. # Every destination must be listed exactly; redirects, URL queries, private or # otherwise non-public destination addresses, and embedded expression # references are refused. MEMORY carriers are accepted under the same source # policy but execute without retained caching (same as NONE). DISK proxies cache -# chunks inside their carrier up to persisted max_cache_bytes (or unbounded if None). -# With a customer quota, shared SQLite admission permits DISK growth when capacity -# is available; denied fills still return data without retention. See server docs -# for accounting exclusions, operational headroom, and staged replacement costs. +# chunks in private runtime generations up to persisted max_cache_bytes (256 MiB +# by default, or unbounded if None). With a customer quota, shared SQLite admission +# permits DISK growth when capacity is available; denied fills still return data +# without retention. See server docs for accounting exclusions, operational +# headroom, and sparse lifecycle behavior. # [server.remote_proxy] # enabled = true # allowed_hosts = ["datasets.example.org", "objects.example.org:8443"] diff --git a/caterva2/services/remote_proxy.py b/caterva2/services/remote_proxy.py index 21449c25..fcc0fa9a 100644 --- a/caterva2/services/remote_proxy.py +++ b/caterva2/services/remote_proxy.py @@ -17,6 +17,7 @@ from __future__ import annotations import ipaddress +import logging import math import socket import sqlite3 @@ -33,6 +34,8 @@ from caterva2.services import storage_quota +log = logging.getLogger(__name__) + class RemoteProxyDenied(ValueError): """The server policy refuses a remote reference.""" @@ -47,6 +50,7 @@ class Policy: max_rank: int = 16 max_chunks: int = 10_000_000 max_concurrency: int = 8 + cache_backend: str = "sparse" policy = Policy() @@ -377,6 +381,7 @@ def __init__(self, source, geometry, carrier, payload): self.shape, self.dtype, self.chunks, self.blocks = geometry self.cparams = source.cparams self.path = carrier.schunk.urlpath + self.carrier_generation = storage_quota.signature(self.path) self.requested_cache_policy = payload["cache_policy"] self.requested_payload = dict(payload) self.requested_max_cache_bytes = payload["max_cache_bytes"] @@ -400,6 +405,8 @@ def current_cache_bytes(self) -> int: def quota_read(self, quota, item=(), *, nchunk=None): """Assemble on an immutable candidate and admit its exact physical size.""" + if quota.cache_backend == "sparse": + return quota.remote.read(self, item, nchunk=nchunk) if self.cache_policy != "disk" or getattr(self.src, "stamp", None) is None: return ( self.get_chunk(nchunk, cache_limit=0) @@ -526,7 +533,17 @@ def cold_cframe(carrier, payload) -> bytes: chunks=carrier.chunks, blocks=carrier.blocks, cparams=carrier.cparams, + meta={ + key: carrier.schunk.meta[key] + for key in carrier.schunk.meta + if key not in {"b2nd", "b2o", "proxy"} + }, ) + from blosc2.proxy import _RESERVED_VLMETA + + for key in carrier.schunk.vlmeta: + if key not in _RESERVED_VLMETA and key != "b2o": + cold.schunk.vlmeta[key] = carrier.schunk.vlmeta[key] write_b2object_payload(cold, payload) return cold.to_cframe() diff --git a/caterva2/services/server.py b/caterva2/services/server.py index 85fcb082..50e5c0cd 100644 --- a/caterva2/services/server.py +++ b/caterva2/services/server.py @@ -138,7 +138,7 @@ def guess_type(path): def get_disk_usage(): - if settings.quota: + if settings.quota or remote_proxy.policy.cache_backend == "sparse": return quota_coordinator().usage()["used"] exclude = {"db.json", "db.sqlite"} return sum( @@ -184,13 +184,18 @@ def account_chunk_written(nbytes: int) -> None: def quota_coordinator(): """Independent of authentication; one connection is opened per DB operation.""" - if not settings.quota: + if not settings.quota and remote_proxy.policy.cache_backend != "sparse": return None work_bytes = settings.parse_size(settings.conf.get(".quota_work_bytes", "1G")) - key = (str(settings.statedir), settings.quota, work_bytes) + key = (str(settings.statedir), settings.quota, work_bytes, remote_proxy.policy.cache_backend) coordinator = _quota_instances.get(key) if coordinator is None: - coordinator = storage_quota.StorageQuota(settings.statedir, settings.quota, work_bytes=work_bytes) + coordinator = storage_quota.StorageQuota( + settings.statedir, + settings.quota, + work_bytes=work_bytes, + cache_backend=remote_proxy.policy.cache_backend, + ) _quota_instances[key] = coordinator return coordinator @@ -244,7 +249,12 @@ def copy_dataset(source, destination): if not entry.name.endswith(".b2lock"): copy_dataset(entry, destination / entry.name) else: - write_dataset(destination, source.read_bytes()) + data = source.read_bytes() + if remote_proxy.policy.cache_backend == "sparse": + reference = remote_proxy.inspect(source) + if reference is not None: + data = remote_proxy.cold_cframe(*reference) + write_dataset(destination, data) def move_dataset(source, destination): @@ -266,6 +276,10 @@ def move_dataset(source, destination): data, generation = quota.snapshot(source) if generation is None: raise storage_quota.StorageBusy("move source was removed") + if remote_proxy.policy.cache_backend == "sparse" and source.suffix in {".b2nd", ".b2frame"}: + carrier = blosc2.ndarray_from_cframe(data) + if carrier.schunk.vlmeta.get("b2o", {}).get("kind") == "remote_proxy": + data = remote_proxy.cold_cframe(carrier, carrier.schunk.vlmeta["b2o"]) write_dataset(destination, data) quota.publish(source, None, expected=generation, prune=False) @@ -299,7 +313,7 @@ async def read_remote_proxy(proxy, item, abspath): """Read one remote selection while serializing and accounting cache mutation.""" lock = dataset_lock(abspath) async with lock: - if settings.quota: + if settings.quota or remote_proxy.policy.cache_backend == "sparse": return await concurrency.run_in_threadpool( lambda: blosc2.asarray(quota_proxy_operation(proxy, item)).to_cframe() ) @@ -471,7 +485,7 @@ def _setup_plugin_globals(): @contextlib.asynccontextmanager async def lifespan(app: FastAPI): - if settings.quota: + if settings.quota or remote_proxy.policy.cache_backend == "sparse": await concurrency.run_in_threadpool(quota_coordinator) # Initialize the (users) database if user_login_enabled(): @@ -486,7 +500,24 @@ async def lifespan(app: FastAPI): for p in providers.active: await p.startup() - yield + async def cache_maintenance(): + while True: + await asyncio.sleep(30) + try: + await concurrency.run_in_threadpool(quota_coordinator().remote.maintain) + except (OSError, sqlite3.Error, ValueError): + remote_proxy.log.exception("Sparse cache maintenance deferred") + + maintenance = ( + asyncio.create_task(cache_maintenance()) if remote_proxy.policy.cache_backend == "sparse" else None + ) + try: + yield + finally: + if maintenance is not None: + maintenance.cancel() + with contextlib.suppress(asyncio.CancelledError): + await maintenance for p in providers.active: await p.shutdown() @@ -1411,6 +1442,23 @@ async def post_fetch_data( ) +class RemoteCacheFileResponse(responses.FileResponse): + """Release artifact ownership even if streaming disconnects or fails.""" + + def __init__(self, path, *, cleanup, **kwargs): + super().__init__(path, **kwargs) + self.cleanup = cleanup + + async def __call__(self, scope, receive, send): + try: + return await super().__call__(scope, receive, send) + finally: + import anyio + + with anyio.CancelScope(shield=True): + await concurrency.run_in_threadpool(self.cleanup) + + @app.get("/api/download/{path:path}") async def download_data( path: pathlib.Path, @@ -1445,6 +1493,19 @@ async def download_data( headers.update(srv_utils.NO_RANGES) return responses.StreamingResponse(body, media_type=media_type, headers=headers) + if remote_proxy.policy.cache_backend == "sparse" and include_cache: + abspath = get_abspath(path, user) + reference = remote_proxy.inspect(abspath) if abspath.suffix in {".b2nd", ".b2frame"} else None + if reference is not None and reference[1]["cache_policy"] == "disk": + proxy = await concurrency.run_in_threadpool(lambda: remote_proxy.resolve(*reference)) + if proxy.src.stamp is not None: + artifact, etag, cleanup = await concurrency.run_in_threadpool( + lambda: quota_coordinator().remote.export(proxy) + ) + return RemoteCacheFileResponse( + artifact, cleanup=cleanup, filename=path.name, headers={"ETag": f'"{etag}"'} + ) + decompress = accept_encoding != "blosc2" # Read before creating the response: a bad path must 404 up front, not # abort the stream after the 200 headers already went out. @@ -1547,7 +1608,7 @@ async def get_chunk( # In case we do, this would have to be changed. chunk = container.get_chunk(nchunk) elif isinstance(container, remote_proxy.ServerRemoteProxy): - if settings.quota: + if settings.quota or remote_proxy.policy.cache_backend == "sparse": chunk = await concurrency.run_in_threadpool( lambda: quota_proxy_operation(container, nchunk=nchunk) ) @@ -1751,12 +1812,14 @@ def publish_dataset(abspath: pathlib.Path, path: pathlib.Path) -> str: srv_utils.raise_bad_request("publishing needs fsspec, which is not installed here") destination = publish_destination(path) fs, target = fsspec.url_to_fs(destination) - if settings.quota and isinstance(fs, fsspec.implementations.local.LocalFileSystem): + if (settings.quota or remote_proxy.policy.cache_backend == "sparse") and isinstance( + fs, fsspec.implementations.local.LocalFileSystem + ): target_path = pathlib.Path(target).resolve() state = pathlib.Path(settings.statedir).resolve() if target_path == state or state in target_path.parents: srv_utils.raise_bad_request("local publish_root must be outside the server state directory") - if settings.quota: + if settings.quota or remote_proxy.policy.cache_backend == "sparse": quota = quota_coordinator() frame, generation = quota.snapshot(abspath) if generation is None: @@ -1781,7 +1844,7 @@ def publish_dataset(abspath: pathlib.Path, path: pathlib.Path) -> str: fs.makedirs(parent, exist_ok=True) try: with ( - io.BytesIO(frame) if settings.quota else open(abspath, "rb") as source, + io.BytesIO(frame) if quota_coordinator() is not None else open(abspath, "rb") as source, fs.open(staging, "wb") as target_file, ): shutil.copyfileobj(source, target_file) @@ -1792,7 +1855,7 @@ def publish_dataset(abspath: pathlib.Path, path: pathlib.Path) -> str: with contextlib.suppress(Exception): fs.rm(staging) raise - if settings.quota: + if settings.quota or remote_proxy.policy.cache_backend == "sparse": array = blosc2.ndarray_from_cframe(frame, copy=True) array.schunk.vlmeta[PUBLISHED_URL] = destination array.schunk.vlmeta[FILL_STATE] = PUBLISHED @@ -1826,7 +1889,7 @@ def store_chunk(abspath: pathlib.Path, nchunk: int, chunk: bytes) -> dict: both find the slot free would otherwise both write it, and the second would move every chunk that came after the first. """ - if settings.quota: + if settings.quota or remote_proxy.policy.cache_backend == "sparse": quota = quota_coordinator() frame, generation = quota.snapshot(abspath) try: @@ -2150,7 +2213,7 @@ def make_expr( abspath.mkdir(exist_ok=True, parents=True) - if settings.quota: + if settings.quota or remote_proxy.policy.cache_backend == "sparse": # Serialize before admission: metadata and compression determine the charge. result = arr.compute() if compute else arr try: @@ -2294,7 +2357,7 @@ async def move( # Make sure the destination directory exists dest_abspath.parent.mkdir(exist_ok=True, parents=True) - if settings.quota: + if settings.quota or remote_proxy.policy.cache_backend == "sparse": # Reserve the copy before removing the source. No fictitious free-space # credit; directory operations retain their existing non-atomic semantics. move_dataset(abspath, dest_abspath) @@ -2353,7 +2416,7 @@ async def copy( # raise fastapi.HTTPException(status_code=409, detail="The new path already exists") dest_abspath.parent.mkdir(exist_ok=True, parents=True) - if settings.quota: + if settings.quota or remote_proxy.policy.cache_backend == "sparse": copy_dataset(abspath, dest_abspath) elif abspath.is_dir(): shutil.copytree(abspath, dest_abspath) @@ -2543,7 +2606,7 @@ async def append_file( # Append the data # The original dataset (open in append mode so it can be resized/written) - if settings.quota: + if settings.quota or remote_proxy.policy.cache_backend == "sparse": frame, generation = quota_coordinator().snapshot(abspath) orig = blosc2.ndarray_from_cframe(frame, copy=True) if "b2o" in orig.schunk.meta or "proxy-source" in orig.schunk.meta: @@ -2565,7 +2628,7 @@ async def append_file( orig.resize(result_shape) # Append the new data to orig along the first axis orig[orig.shape[0] - new_len :] = new_data - if settings.quota: + if settings.quota or remote_proxy.policy.cache_backend == "sparse": quota_coordinator().publish(abspath, orig.to_cframe(), expected=generation) # Return the new shape @@ -2606,7 +2669,11 @@ async def unfold_file( # Unfold the container if abspath.suffix in {".h5", ".hdf5"}: # Create proxies for each dataset in HDF5 file - all_dsets = list(hdf5.create_hdf5_proxies(abspath, writer=write_dataset if settings.quota else None)) + all_dsets = list( + hdf5.create_hdf5_proxies( + abspath, writer=write_dataset if quota_coordinator() is not None else None + ) + ) if len(all_dsets) == 0: detail = "No arrays found in HDF5 file" raise fastapi.HTTPException(detail=detail, status_code=400) @@ -4001,7 +4068,7 @@ async def htmx_upload( suffix = filename.suffix suffixes = filename.suffixes[-2:] if suffix in [".tar", ".tgz", ".zip"] or suffixes == [".tar", ".gz"]: - if settings.quota: + if settings.quota or remote_proxy.policy.cache_backend == "sparse": # Admit encoded members independently. Never extract an archive into # managed storage before measuring its final serialized files. first = None @@ -4171,6 +4238,23 @@ async def get_file_content(path, user, decompress=True, include_cache=True): lock = dataset_lock(abspath) async with lock: carrier, payload = remote_proxy.inspect(abspath) + if ( + remote_proxy.policy.cache_backend == "sparse" + and include_cache + and payload["cache_policy"] == "disk" + ): + + def snapshot_sparse(): + proxy = remote_proxy.resolve(carrier, payload) + if proxy.src.stamp is None: + return remote_proxy.cold_cframe(carrier, payload) + artifact, _, cleanup = quota_coordinator().remote.export(proxy) + try: + return artifact.read_bytes() + finally: + cleanup() + + return await concurrency.run_in_threadpool(snapshot_sparse) return await concurrency.run_in_threadpool( lambda: remote_proxy.export_cframe(carrier, payload, include_cache=include_cache) ) diff --git a/caterva2/services/sparse_cache.py b/caterva2/services/sparse_cache.py new file mode 100644 index 00000000..a56afa9a --- /dev/null +++ b/caterva2/services/sparse_cache.py @@ -0,0 +1,687 @@ +"""Experimental private RemoteProxy generations with shared soft admission. + +All request mutations hold the existing path lock and a generation lock. Startup +recovery discards interrupted disposable generations without resolving sources. +""" + +from __future__ import annotations + +import contextlib +import hashlib +import json +import logging +import math +import os +import re +import shutil +import sqlite3 +import time +import uuid + +import blosc2 + +from caterva2.services.storage_quota import QuotaExceeded, StorageBusy, file_lock, signature, sync_directory + +log = logging.getLogger(__name__) +SCHEMA = """ +CREATE TABLE IF NOT EXISTS remote_objects ( + object_id TEXT PRIMARY KEY, path TEXT UNIQUE, carrier_generation TEXT NOT NULL, + spec_hash TEXT NOT NULL, source_stamp TEXT, active_generation TEXT, + parent_charge_bytes INTEGER NOT NULL DEFAULT 0, updated REAL NOT NULL); +CREATE TABLE IF NOT EXISTS remote_generations ( + generation_id TEXT PRIMARY KEY, object_id TEXT NOT NULL, relpath TEXT UNIQUE NOT NULL, + state TEXT NOT NULL CHECK(state IN ('building','active','retired','trash')), + spec_hash TEXT NOT NULL, source_stamp TEXT NOT NULL, max_cache_bytes INTEGER, + payload_bytes INTEGER NOT NULL, charge_bytes INTEGER NOT NULL, inode_count INTEGER NOT NULL, + touched REAL NOT NULL, created REAL NOT NULL); +CREATE UNIQUE INDEX IF NOT EXISTS one_active_remote_generation +ON remote_generations(object_id) WHERE state='active'; +CREATE TABLE IF NOT EXISTS remote_operations ( + id TEXT PRIMARY KEY, generation_id TEXT NOT NULL UNIQUE, kind TEXT NOT NULL, + estimate INTEGER NOT NULL, previous_charge INTEGER NOT NULL, details BLOB NOT NULL, + started REAL NOT NULL); +CREATE TABLE IF NOT EXISTS remote_work ( + id TEXT PRIMARY KEY, kind TEXT NOT NULL, reserved INTEGER NOT NULL, + relpath TEXT NOT NULL, started REAL NOT NULL); +CREATE TABLE IF NOT EXISTS remote_orphans ( + id TEXT PRIMARY KEY, relpath TEXT UNIQUE NOT NULL, charge_bytes INTEGER NOT NULL, + inode_count INTEGER NOT NULL, updated REAL NOT NULL); +""" + + +def allocated(path): + st = path.lstat() + return getattr(st, "st_blocks", None) * 512 if hasattr(st, "st_blocks") else st.st_size + + +def measure(path): + """Never follow links, including an unexpected child link.""" + if not path.exists() and not path.is_symlink(): + return 0, 0 + total, count = allocated(path), 1 + if path.is_dir() and not path.is_symlink(): + for entry in path.iterdir(): + size, n = measure(entry) + total += size + count += n + return total, count + + +def sync_tree(path): + for entry in path.iterdir(): + if entry.is_symlink() or entry.is_dir(): + raise ValueError("unexpected entry in sparse frame") + with entry.open("rb") as stream: + os.fsync(stream.fileno()) + sync_directory(path) + sync_directory(path.parent) + + +class SparseCache: + def __init__(self, quota, *, initialize=True): + self.q = quota + if quota.cache_backend == "sparse": + import inspect + + required = ("with_sparse_cache", "read_cached", "trim_sparse_cache") + if ( + any(not hasattr(blosc2.RemoteProxy, name) for name in required) + or "source_descriptor" + not in inspect.signature(blosc2.RemoteProxy.with_sparse_cache).parameters + ): + raise RuntimeError("sparse backend requires the Python-Blosc2 v7 cache APIs") + self.root = quota.root / ".remote-cache" + if self.root.is_symlink(): + raise ValueError("private cache root cannot be a symlink") + self.root.mkdir(mode=0o700, exist_ok=True) + self.trash = self.root / ".trash" + if self.trash.is_symlink(): + raise ValueError("private trash cannot be a symlink") + self.trash.mkdir(mode=0o700, exist_ok=True) + if initialize: + with quota.connect() as db: + db.executescript(SCHEMA) + columns = {row[1] for row in db.execute("PRAGMA table_info(account)")} + if "cache_fill_suspended" not in columns: + db.execute( + "ALTER TABLE account ADD COLUMN cache_fill_suspended INTEGER NOT NULL DEFAULT 0" + ) + if "cache_backend" not in columns: + db.execute("ALTER TABLE account ADD COLUMN cache_backend TEXT NOT NULL DEFAULT 'sparse'") + previous = db.execute("SELECT cache_backend FROM account").fetchone()[0] + if db.execute("PRAGMA user_version").fetchone()[0] == 2 and previous != quota.cache_backend: + raise StorageBusy("backend switch requires explicit offline configuration migration") + db.execute("UPDATE account SET cache_backend=?", (quota.cache_backend,)) + db.execute("PRAGMA user_version=2") + with quota.connect() as db: + if db.execute("SELECT cache_backend FROM account").fetchone()[0] != quota.cache_backend: + raise StorageBusy("backend switch requires draining storage workers") + + @staticmethod + def totals(db): + used = db.execute("SELECT coalesce(sum(charge_bytes),0) FROM remote_generations").fetchone()[0] + used += db.execute("SELECT coalesce(sum(parent_charge_bytes),0) FROM remote_objects").fetchone()[0] + used += db.execute("SELECT coalesce(sum(charge_bytes),0) FROM remote_orphans").fetchone()[0] + reserved = db.execute("SELECT coalesce(sum(estimate),0) FROM remote_operations").fetchone()[0] + work = db.execute("SELECT coalesce(sum(reserved),0) FROM remote_work").fetchone()[0] + return used, reserved, work + + def path(self, rel): + if not re.fullmatch(r"\.remote-cache/(?:[0-9a-f]{32}/[0-9a-f]{32}|\.trash/[0-9a-f]{32})", rel): + raise ValueError("invalid private generation path") + path = self.q.root + for part in rel.split("/"): + path /= part + if path.is_symlink(): + raise ValueError("private cache symlink") + return path + + def guard(self, gid, *, blocking=True): + if not re.fullmatch("[0-9a-f]{32}", gid): + raise ValueError("invalid generation ID") + return file_lock(self.q.control / f"remote-{gid}.lock", blocking=blocking) + + def retire_path(self, rel): + """Caller owns the dataset path guard; physical cleanup is deferred.""" + with self.q.transaction() as db: + rows = db.execute("SELECT object_id FROM remote_objects WHERE path=?", (rel,)).fetchall() + for (oid,) in rows: + db.execute("UPDATE remote_generations SET state='retired' WHERE object_id=?", (oid,)) + db.execute( + "UPDATE remote_objects SET path=NULL,active_generation=NULL WHERE object_id=?", (oid,) + ) + + def _intent(self, gid, kind, estimate=0, details=None): + with self.q.transaction() as db: + db.execute( + "INSERT INTO remote_operations VALUES(?,?,?,?,?,?,?)", + (uuid.uuid4().hex, gid, kind, estimate, 0, json.dumps(details or {}), time.time()), + ) + + def _finish(self, gid, path, payload): + charge, inodes = measure(path) + parent_charge = allocated(path.parent) + with self.q.transaction() as db: + db.execute( + "UPDATE remote_generations SET charge_bytes=?,inode_count=?,payload_bytes=?,touched=? " + "WHERE generation_id=?", + (charge, inodes, payload, time.time(), gid), + ) + db.execute( + "UPDATE remote_objects SET parent_charge_bytes=? WHERE object_id=" + "(SELECT object_id FROM remote_generations WHERE generation_id=?)", + (parent_charge, gid), + ) + db.execute("DELETE FROM remote_operations WHERE generation_id=?", (gid,)) + self._suspension() + + def _suspension(self): + usage = self.q.usage() + limit = usage["quota"] + if not limit or usage["used"] <= int(limit * 0.9): + suspended = 0 + elif usage["used"] > limit: + suspended = 1 + else: + return + with self.q.transaction() as db: + db.execute("UPDATE account SET cache_fill_suspended=?", (suspended,)) + + def _admit(self, gid, estimate): + with self.q.transaction() as db: + used = db.execute("SELECT coalesce(sum(size),0) FROM objects").fetchone()[0] + reserved = db.execute("SELECT coalesce(sum(reserved),0) FROM operations").fetchone()[0] + cache, estimates, _ = self.totals(db) + limit, suspended = db.execute("SELECT quota,cache_fill_suspended FROM account").fetchone() + if limit and (suspended or used + cache + reserved + estimates + estimate > limit): + return False + db.execute( + "INSERT INTO remote_operations VALUES(?,?,?,?,?,?,?)", + (uuid.uuid4().hex, gid, "fill", estimate if limit else 0, 0, "{}", time.time()), + ) + return True + + def _attach(self, proxy, path, carrier=None): + return blosc2.RemoteProxy.with_sparse_cache( + proxy.src, + path, + source_descriptor=proxy.requested_payload["source"], + carrier=carrier, + max_cache_bytes=proxy.max_cache_bytes, + ) + + def _bind(self, proxy, rel): + sig = signature(proxy.path) + if sig is None or sig != proxy.carrier_generation: + raise StorageBusy("public carrier changed after authorization") + spec = hashlib.sha256(json.dumps(proxy.requested_payload, sort_keys=True).encode()).hexdigest() + stamp = json.dumps(proxy.src.stamp, sort_keys=True) + with self.q.connect() as db: + row = db.execute( + "SELECT o.object_id,o.active_generation,o.carrier_generation,o.spec_hash," + "o.source_stamp,g.relpath FROM remote_objects o LEFT JOIN remote_generations g " + "ON g.generation_id=o.active_generation WHERE o.path=?", + (rel,), + ).fetchone() + if row and row[1] and row[2:5] == (json.dumps(sig), spec, stamp): + path = self.path(row[5]) + if path.is_dir(): + return row[1], path + if row: + self.retire_path(rel) + with file_lock(self.q.control / "remote-migration.lock", blocking=False): + return self._build(proxy, rel, sig, spec, stamp) + + def _build(self, proxy, rel, sig, spec, stamp): + from caterva2.services import remote_proxy + + oid, gid = uuid.uuid4().hex, uuid.uuid4().hex + path = self.root / oid / gid + relative = path.relative_to(self.q.root).as_posix() + now = time.time() + if shutil.disk_usage(self.root).free < sig[2] + (1 << 30): + raise QuotaExceeded("insufficient migration headroom") + with self.q.transaction() as db: + db.execute( + "INSERT INTO remote_objects VALUES(?,?,?,?,?,?,?,?)", + (oid, rel, json.dumps(sig), spec, stamp, None, 0, now), + ) + db.execute( + "INSERT INTO remote_generations VALUES(?,?,?,?,?,?,?,?,?,?,?,?)", + (gid, oid, relative, "building", spec, stamp, proxy.max_cache_bytes, 0, 0, 0, now, now), + ) + with self.guard(gid): + self._intent(gid, "build", sig[2]) + path.parent.mkdir(mode=0o700) + carrier = remote_proxy.raw_carrier(proxy.path) + if carrier.schunk.vlmeta.get("b2o") != proxy.requested_payload or ( + carrier.shape, + carrier.dtype, + carrier.chunks, + carrier.blocks, + ) != (proxy.shape, proxy.dtype, proxy.chunks, proxy.blocks): + raise StorageBusy("carrier specification changed during source authorization") + runtime = self._attach(proxy, path, carrier) + payload_bytes = runtime.cached_payload_bytes + del runtime + sync_tree(path) + charge, count = measure(path) + cold = remote_proxy.cold_cframe(carrier, proxy.requested_payload) + del carrier + parent_charge = allocated(path.parent) + with self.q.transaction() as db: + db.execute( + "UPDATE remote_generations SET state='active',charge_bytes=?,inode_count=?," + "payload_bytes=? WHERE generation_id=?", + (charge, count, payload_bytes, gid), + ) + db.execute( + "UPDATE remote_objects SET active_generation=?,parent_charge_bytes=? WHERE object_id=?", + (gid, parent_charge, oid), + ) + db.execute( + "UPDATE remote_operations SET kind='coldify',estimate=0 WHERE generation_id=?", (gid,) + ) + new_sig = self.q.publish_locked(rel, cold, expected=sig, preserve_remote=True) + with self.q.transaction() as db: + db.execute( + "UPDATE remote_objects SET carrier_generation=? WHERE object_id=?", + (json.dumps(new_sig), oid), + ) + db.execute("DELETE FROM remote_operations WHERE generation_id=?", (gid,)) + proxy.carrier_generation = new_sig + return gid, path + + def read(self, proxy, item=(), *, nchunk=None): + """Fall back only for local retention errors, preserving upstream failures.""" + + def uncached(): + return ( + proxy.src.get_chunk(nchunk) + if nchunk is not None + else blosc2.Proxy(proxy.src, _refresh_source=False)[item] + ) + + if proxy.cache_policy != "disk" or proxy.src.stamp is None: + return uncached() + result = None + assembled = False + try: + rel = self.q.relative(proxy.path) + with file_lock(self.q.control / "initialize.lock", shared=True), self.q.lock(rel): + gid, path = self._bind(proxy, rel) + with self.guard(gid): + with self.q.connect() as db: + pending = db.execute( + "SELECT 1 FROM remote_operations WHERE generation_id=?", (gid,) + ).fetchone() + if pending: + raise StorageBusy("generation needs recovery") + runtime = self._attach(proxy, path) + try: + hit, result = runtime.read_cached(item, nchunk=nchunk) + if hit: + assembled = True + with self.q.transaction() as db: + db.execute( + "UPDATE remote_generations SET touched=? WHERE generation_id=? AND touched budget or free < budget + (1 << 30): + raise QuotaExceeded("insufficient export headroom") + db.execute( + "INSERT INTO remote_work VALUES(?,?,?,?,?)", + ( + opid, + "export", + budget, + destination.relative_to(self.q.root).as_posix(), + time.time(), + ), + ) + runtime = self._attach(proxy, path) + try: + with destination.open("xb"): + pass + runtime.save(destination, mode="w") + finally: + del runtime + exported = remote_proxy.raw_carrier(destination, mode="a") + public = remote_proxy.raw_carrier(proxy.path) + from blosc2.proxy import _RESERVED_VLMETA + + for key in public.schunk.vlmeta: + if key not in _RESERVED_VLMETA and key != "b2o": + exported.schunk.vlmeta[key] = public.schunk.vlmeta[key] + del public, exported + if destination.stat().st_size > budget: + raise QuotaExceeded("export exceeded its staging budget") + digest = hashlib.sha256() + with destination.open("rb") as stream: + os.fsync(stream.fileno()) + for block in iter(lambda: stream.read(1 << 20), b""): + digest.update(block) + sync_directory(folder) + return destination, digest.hexdigest(), cleanup + except BaseException: + cleanup() + raise + + def prune(self, *, force=False): + """Bounded whole-generation cleanup plus authorized-free chunk eviction.""" + try: + with ( + file_lock(self.q.control / "initialize.lock", shared=True), + file_lock(self.q.control / "remote-prune.lock", blocking=False), + ): + self.cleanup(max_generations=4) + usage = self.q.usage() + if not usage["quota"] or ( + usage["used"] <= usage["quota"] and not force and not usage["cache_fill_suspended"] + ): + return + with self.q.connect() as db: + rows = db.execute( + "SELECT g.generation_id,g.relpath,g.payload_bytes,o.path " + "FROM remote_generations g JOIN remote_objects o ON o.object_id=g.object_id " + "WHERE g.state='active' AND NOT EXISTS (SELECT 1 FROM remote_operations p " + "WHERE p.generation_id=g.generation_id) ORDER BY touched LIMIT 4" + ).fetchall() + remaining = 64 + for gid, private, payload, rel in rows: + try: + with self.q.lock(rel, blocking=False), self.guard(gid, blocking=False): + with self.q.connect() as db: + active = db.execute( + "SELECT 1 FROM remote_generations WHERE generation_id=? " + "AND state='active'", + (gid,), + ).fetchone() + if not active: + continue + usage = self.q.usage() + needed = max(0, usage["used"] - int(usage["quota"] * 0.9)) + if not needed or not remaining: + break + path = self.path(private) + self._intent(gid, "prune") + evicted, payload = blosc2.RemoteProxy.trim_sparse_cache( + path, max(0, payload - needed), max_chunks=remaining + ) + remaining -= len(evicted) + sync_tree(path) + self._finish(gid, path, payload) + except (OSError, sqlite3.Error, StorageBusy, ValueError): + log.debug("deferred sparse pruning", exc_info=True) + except StorageBusy: + pass + + def cleanup(self, *, max_generations=4): + with self.q.connect() as db: + rows = db.execute( + "SELECT generation_id,object_id,relpath FROM remote_generations " + "WHERE state IN ('retired','trash') LIMIT ?", + (max_generations,), + ).fetchall() + for gid, oid, rel in rows: + try: + with self.guard(gid, blocking=False): + path = self.path(rel) + trash = self.trash / gid + with self.q.transaction() as db: + db.execute( + "UPDATE remote_generations SET state='trash' WHERE generation_id=?", (gid,) + ) + if path.exists() and path != trash: + os.replace(path, trash) + sync_directory(path.parent) + sync_directory(self.trash) + with self.q.transaction() as db: + db.execute( + "UPDATE remote_generations SET relpath=? WHERE generation_id=?", + (trash.relative_to(self.q.root).as_posix(), gid), + ) + if trash.exists(): + shutil.rmtree(trash) + sync_directory(self.trash) + parent = self.root / oid + with contextlib.suppress(FileNotFoundError, OSError): + parent.rmdir() + parent_exists = parent.exists() + parent_charge = allocated(parent) if parent_exists else 0 + with self.q.transaction() as db: + db.execute("DELETE FROM remote_operations WHERE generation_id=?", (gid,)) + db.execute("DELETE FROM remote_generations WHERE generation_id=?", (gid,)) + db.execute( + "UPDATE remote_objects SET parent_charge_bytes=? WHERE object_id=?", + (parent_charge, oid), + ) + if not parent_exists: + db.execute( + "DELETE FROM remote_objects WHERE object_id=? AND path IS NULL AND parent_charge_bytes=0 AND NOT EXISTS " + "(SELECT 1 FROM remote_generations WHERE object_id=?)", + (oid, oid), + ) + except (OSError, StorageBusy): + log.debug("deferred sparse cleanup", exc_info=True) + self._suspension() + + def cleanup_empty_objects(self): + with self.q.connect() as db: + rows = db.execute( + "SELECT object_id FROM remote_objects WHERE path IS NULL AND NOT EXISTS " + "(SELECT 1 FROM remote_generations g WHERE g.object_id=remote_objects.object_id) LIMIT 64" + ).fetchall() + for (oid,) in rows: + if not re.fullmatch("[0-9a-f]{32}", oid): + raise ValueError("invalid object ID") + parent = self.root / oid + if parent.is_symlink(): + raise ValueError("invalid object directory") + with contextlib.suppress(OSError): + parent.rmdir() + parent_exists = parent.exists() + charge = allocated(parent) if parent_exists else 0 + with self.q.transaction() as db: + db.execute( + "UPDATE remote_objects SET parent_charge_bytes=? WHERE object_id=?", (charge, oid) + ) + if not parent_exists: + db.execute("DELETE FROM remote_objects WHERE object_id=?", (oid,)) + + def recover(self): + """Conservatively retire interrupted operations; never resolve a URL.""" + with self.q.connect() as db: + rows = db.execute( + "SELECT o.path,o.object_id,o.carrier_generation,g.generation_id,g.relpath " + "FROM remote_objects o JOIN remote_generations g ON g.object_id=o.object_id" + ).fetchall() + for rel, oid, expected, gid, private in rows: + try: + guard = self.q.lock(rel, blocking=False) if rel else contextlib.nullcontext() + with guard, self.guard(gid, blocking=False): + with self.q.connect() as db: + pending = db.execute( + "SELECT 1 FROM remote_operations WHERE generation_id=?", (gid,) + ).fetchone() + path = self.path(private) + if ( + pending + or (rel and json.dumps(signature(self.q.root / rel)) != expected) + or not path.exists() + ): + with self.q.transaction() as db: + db.execute( + "UPDATE remote_generations SET state='retired' WHERE generation_id=?", (gid,) + ) + db.execute( + "UPDATE remote_objects SET path=NULL,active_generation=NULL WHERE object_id=?", + (oid,), + ) + if not path.exists() and (self.trash / gid).exists(): + path = self.trash / gid + charge, count = measure(path) + parent = self.root / oid + parent_charge = allocated(parent) if parent.exists() else 0 + with self.q.transaction() as db: + db.execute( + "UPDATE remote_objects SET parent_charge_bytes=? WHERE object_id=?", + (parent_charge, oid), + ) + db.execute( + "UPDATE remote_generations SET charge_bytes=?,inode_count=? WHERE generation_id=?", + (charge, count, gid), + ) + except StorageBusy: + continue + self.cleanup(max_generations=64) + self.cleanup_empty_objects() + self.recover_exports() + self.reconcile_orphans() + + def recover_exports(self): + with self.q.connect() as db: + rows = db.execute("SELECT id,relpath FROM remote_work").fetchall() + for opid, rel in rows: + if not re.fullmatch("[0-9a-f]{32}", opid) or rel != f".storage/exports/{opid}.b2nd": + raise ValueError("invalid export registry path") + try: + with file_lock(self.q.control / f"export-{opid}.lock", blocking=False): + path = self.q.root / rel + if path.parent.is_symlink(): + raise ValueError("invalid export parent") + path.unlink(missing_ok=True) + if path.parent.exists(): + sync_directory(path.parent) + with self.q.transaction() as db: + db.execute("DELETE FROM remote_work WHERE id=?", (opid,)) + except (StorageBusy, OSError): + continue + + def reconcile_orphans(self): + # Run under the initialization barrier exclusively: no builder can be + # between row registration and directory creation during this inventory. + with self.q.connect() as db: + known = {row[0] for row in db.execute("SELECT relpath FROM remote_generations")} + objects = {row[0] for row in db.execute("SELECT object_id FROM remote_objects")} + for parent in self.root.iterdir(): + if parent.is_symlink() or not parent.is_dir(): + continue + if parent.name != ".trash" and not re.fullmatch("[0-9a-f]{32}", parent.name): + continue + for path in parent.iterdir(): + rel = path.relative_to(self.q.root).as_posix() + if rel in known or not re.fullmatch("[0-9a-f]{32}", path.name): + continue + size, count = measure(path) + with self.q.transaction() as db: + db.execute( + "INSERT INTO remote_orphans VALUES(?,?,?,?,?) ON CONFLICT(relpath) DO UPDATE SET " + "charge_bytes=excluded.charge_bytes,inode_count=excluded.inode_count,updated=excluded.updated", + (uuid.uuid4().hex, rel, size, count, time.time()), + ) + try: + if path.is_symlink() or path.is_file(): + path.unlink() + else: + shutil.rmtree(path) + sync_directory(parent) + except OSError: + continue + with self.q.transaction() as db: + db.execute("DELETE FROM remote_orphans WHERE relpath=?", (rel,)) + if parent.name != ".trash" and parent.name not in objects: + # An orphan generation's parent also consumes space/inodes. + # Keep its own charge until all children and the parent are gone. + with contextlib.suppress(OSError): + parent.rmdir() + exists = parent.exists() + charge = allocated(parent) if exists else 0 + rel = parent.relative_to(self.q.root).as_posix() + with self.q.transaction() as db: + if exists: + db.execute( + "INSERT INTO remote_orphans VALUES(?,?,?,?,?) ON CONFLICT(relpath) DO UPDATE SET " + "charge_bytes=excluded.charge_bytes,updated=excluded.updated", + (uuid.uuid4().hex, rel, charge, 1, time.time()), + ) + else: + db.execute("DELETE FROM remote_orphans WHERE relpath=?", (rel,)) + self._suspension() + + def maintain(self): + try: + with file_lock(self.q.control / "initialize.lock", blocking=False): + self.recover() + except StorageBusy: + pass + self.prune() diff --git a/caterva2/services/storage_quota.py b/caterva2/services/storage_quota.py index 7cbfb0e3..70736089 100644 --- a/caterva2/services/storage_quota.py +++ b/caterva2/services/storage_quota.py @@ -83,11 +83,16 @@ def sync_directory(path): class StorageQuota: - def __init__(self, statedir, quota, *, work_bytes=WORK_BYTES): + def __init__(self, statedir, quota, *, work_bytes=WORK_BYTES, cache_backend="contiguous"): + self.cache_backend = cache_backend + if cache_backend not in {"contiguous", "sparse"}: + raise ValueError("unknown remote cache backend") + if quota is None: + quota = 0 self.input_root = pathlib.Path(statedir).absolute() self.root = pathlib.Path(statedir).resolve() - if not isinstance(quota, int) or quota <= 0: - raise ValueError("quota must be a positive integer") + if not isinstance(quota, int) or quota < 0: + raise ValueError("quota must be a non-negative integer or None") if not isinstance(work_bytes, int) or work_bytes <= 0: raise ValueError("work_bytes must be a positive integer") self.quota, self.work_bytes = quota, work_bytes @@ -96,11 +101,14 @@ def __init__(self, statedir, quota, *, work_bytes=WORK_BYTES): self.dbpath = self.root / "storage.sqlite" with self.startup_guard() as reconcile: if not reconcile: + from caterva2.services.sparse_cache import SparseCache + + self.remote = SparseCache(self, initialize=False) return # Active writers already protect a fully initialized ledger. with self.connect() as db: db.execute("PRAGMA journal_mode=WAL") version = db.execute("PRAGMA user_version").fetchone()[0] - if version not in (0, 1): + if version not in (0, 1, 2): raise RuntimeError("unsupported storage quota schema version") db.executescript(""" CREATE TABLE IF NOT EXISTS objects ( @@ -123,13 +131,16 @@ def __init__(self, statedir, quota, *, work_bytes=WORK_BYTES): db.executemany( "INSERT OR REPLACE INTO objects(path,size,generation) VALUES(?,?,?)", inventory ) - db.execute("INSERT INTO account VALUES(1,?,?)", (quota, work_bytes)) + db.execute("INSERT INTO account(id,quota,work_bytes) VALUES(1,?,?)", (quota, work_bytes)) db.execute("PRAGMA user_version=1") elif initialized != (quota, work_bytes): # A configuration change is shared by all workers. Admission # reads the ledger value, never a stale worker-local quota. with self.transaction() as db: db.execute("UPDATE account SET quota=?, work_bytes=?", (quota, work_bytes)) + from caterva2.services.sparse_cache import SparseCache + + self.remote = SparseCache(self) # Startup reconciliation covers offline edits and quota re-enablement. # All publishes take this barrier shared; no scan runs in a DB txn. self.recover() @@ -148,6 +159,8 @@ def __init__(self, statedir, quota, *, work_bytes=WORK_BYTES): ) db.executemany("DELETE FROM objects WHERE path=?", ((rel,) for rel in known - present)) + self.remote.recover() + @contextlib.contextmanager def startup_guard(self): guard = file_lock(self.control / "initialize.lock", blocking=False) @@ -290,7 +303,18 @@ def usage(self): "SELECT coalesce(sum(reserved),0),coalesce(sum(working),0) FROM operations" ).fetchone() quota, budget = db.execute("SELECT quota,work_bytes FROM account").fetchone() - return {"used": used, "reserved": reserved, "working": working, "quota": quota, "work_bytes": budget} + remote_used, remote_reserved, remote_work = self.remote.totals(db) + suspended = db.execute("SELECT cache_fill_suspended FROM account").fetchone()[0] + return { + "used": used + remote_used, + "dataset_used": used, + "remote_cache_used": remote_used, + "reserved": reserved + remote_reserved, + "working": working + remote_work, + "quota": quota, + "work_bytes": budget, + "cache_fill_suspended": bool(suspended), + } def publish(self, path, data, *, expected, cache=False, prune=True): """Publish exact bytes (None deletes). A stale generation is never overwritten.""" @@ -310,55 +334,65 @@ def publish(self, path, data, *, expected, cache=False, prune=True): raise AssertionError("unreachable admission retry") def _publish(self, rel, data, expected, cache): - path = self.root / rel with file_lock(self.control / "initialize.lock", shared=True), self.lock(rel): - self.relative(path) # Recheck parent symlinks after taking mutation guards. - self._recover_path(rel) - actual = signature(path) - if actual != expected: - raise StorageBusy("dataset changed while preparing its replacement") - oldsize = 0 if actual is None else actual[2] - size = 0 if data is None else len(data) - opid = uuid.uuid4().hex - with self.transaction() as db: - self._record(db, rel, actual, cache=cache) - used = db.execute("SELECT coalesce(sum(size),0) FROM objects").fetchone()[0] - reserved, working = db.execute( - "SELECT coalesce(sum(reserved),0),coalesce(sum(working),0) FROM operations" - ).fetchone() - quota, budget = db.execute("SELECT quota,work_bytes FROM account").fetchone() - growth = max(0, size - oldsize) - if (growth and used + reserved + growth > quota) or working + size > budget: - raise QuotaExceeded("customer quota or storage staging budget exceeded") - db.execute("INSERT INTO operations VALUES(?,?,?,?)", (opid, rel, growth, size)) - candidate = self.control / f"{opid}.candidate" - # From this point, any failure leaves durable intent for recovery. - if data is None: - path.unlink(missing_ok=True) - else: - fd = os.open(candidate, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) - with os.fdopen(fd, "wb") as file: - file.write(data) - file.flush() - os.fsync(file.fileno()) - # Persist each new directory entry before publishing into it. - missing = [] - parent = path.parent - while not parent.exists(): - missing.append(parent) - parent = parent.parent - for directory in reversed(missing): - directory.mkdir(exist_ok=True) - sync_directory(directory.parent) - sync_directory(self.control) - os.replace(candidate, path) - sync_directory(path.parent) + return self.publish_locked(rel, data, expected=expected, cache=cache) + + def publish_locked(self, rel, data, *, expected, cache=False, preserve_remote=False): + """Publish with the initialization and path guards already held.""" + path = self.root / rel + self.relative(path) # Recheck parent symlinks after taking mutation guards. + self._recover_path(rel) + actual = signature(path) + if actual != expected: + raise StorageBusy("dataset changed while preparing its replacement") + oldsize = 0 if actual is None else actual[2] + size = 0 if data is None else len(data) + opid = uuid.uuid4().hex + with self.transaction() as db: + self._record(db, rel, actual, cache=cache) + used = db.execute("SELECT coalesce(sum(size),0) FROM objects").fetchone()[0] + reserved, working = db.execute( + "SELECT coalesce(sum(reserved),0),coalesce(sum(working),0) FROM operations" + ).fetchone() + quota, budget = db.execute("SELECT quota,work_bytes FROM account").fetchone() + remote_used, remote_reserved, remote_work = self.remote.totals(db) + used += remote_used + reserved += remote_reserved + working += remote_work + growth = max(0, size - oldsize) + if (quota and growth and used + reserved + growth > quota) or working + size > budget: + raise QuotaExceeded("customer quota or storage staging budget exceeded") + db.execute("INSERT INTO operations VALUES(?,?,?,?)", (opid, rel, growth, size)) + candidate = self.control / f"{opid}.candidate" + # From this point, any failure leaves durable intent for recovery. + if data is None: + path.unlink(missing_ok=True) + else: + fd = os.open(candidate, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + with os.fdopen(fd, "wb") as file: + file.write(data) + file.flush() + os.fsync(file.fileno()) + # Persist each new directory entry before publishing into it. + missing = [] + parent = path.parent + while not parent.exists(): + missing.append(parent) + parent = parent.parent + for directory in reversed(missing): + directory.mkdir(exist_ok=True) + sync_directory(directory.parent) sync_directory(self.control) - sig = signature(path) - with self.transaction() as db: - self._record(db, rel, sig, cache=cache) - db.execute("DELETE FROM operations WHERE id=?", (opid,)) - return sig + os.replace(candidate, path) + sync_directory(path.parent) + sync_directory(self.control) + sig = signature(path) + with self.transaction() as db: + self._record(db, rel, sig, cache=cache) + db.execute("DELETE FROM operations WHERE id=?", (opid,)) + if not preserve_remote: + self.remote.retire_path(rel) + return sig def touch(self, path): rel = self.relative(path) @@ -374,6 +408,10 @@ def prune(self, *, exclude, max_victims=4): Pruning is whole-proxy batching initially. Descriptors and user metadata survive; space is credited only after atomic replacement is complete. """ + if self.cache_backend == "sparse": + before = self.usage()["used"] + self.remote.prune(force=True) + return self.usage()["used"] < before import blosc2 reclaimed = 0 diff --git a/caterva2/tests/test_sparse_cache.py b/caterva2/tests/test_sparse_cache.py new file mode 100644 index 00000000..d1ff68a4 --- /dev/null +++ b/caterva2/tests/test_sparse_cache.py @@ -0,0 +1,293 @@ +"""Private cache lifecycle, recovery, admission and offline reclamation.""" + +import blosc2 +import fsspec +import numpy as np +import pytest +from blosc2.b2objects import make_b2object_carrier, write_b2object_payload + +from caterva2.services import remote_proxy +from caterva2.services.sparse_cache import measure +from caterva2.services.storage_quota import StorageQuota, signature + + +@pytest.fixture +def runtime(tmp_path): + for root in ("public", "shared", "personal"): + (tmp_path / root).mkdir() + data = np.random.default_rng(33).integers(0, 256, 60000, dtype="u1") + array = blosc2.asarray(data, chunks=(20000,), blocks=(5000,)) + url = "https://data.example/private-test.b2nd" + fs = fsspec.filesystem("memory") + fs.pipe_file(url, array.to_cframe()) + carrier = make_b2object_carrier( + "remote_proxy", + array.shape, + array.dtype, + chunks=array.chunks, + blocks=array.blocks, + meta={"user-fixed": {"test": True}}, + ) + carrier.schunk.vlmeta["user-variable"] = {"sample": 42} + payload = { + "kind": "remote_proxy", + "version": 1, + "source": {"kind": "fsspec", "version": 1, "urlpath": url}, + "cache_policy": "disk", + "max_cache_bytes": None, + } + write_b2object_payload(carrier, payload) + q = StorageQuota(tmp_path, 1 << 20, cache_backend="sparse") + path = tmp_path / "public/proxy.b2nd" + q.publish(path, carrier.to_cframe(), expected=None) + + def resolve(): + source = blosc2.FsspecNDSource(url, _filesystem=fs) + c = remote_proxy.raw_carrier(path) + return remote_proxy.ServerRemoteProxy( + source, (array.shape, array.dtype, array.chunks, array.blocks), c, payload + ) + + return q, resolve, data, path + + +def test_offline_pruning_and_hysteresis(runtime, monkeypatch): + q, resolve, data, _ = runtime + np.testing.assert_array_equal(q.remote.read(resolve()), data) + used = q.usage()["used"] + with q.transaction() as db: + db.execute("UPDATE account SET quota=?", (used - 10000,)) + + def forbidden(*args, **kwargs): + raise AssertionError("offline pruning contacted source") + + monkeypatch.setattr(blosc2, "FsspecNDSource", forbidden) + q.remote.prune() + assert q.usage()["used"] < used + assert q.usage()["used"] <= int((used - 10000) * 0.9) + assert not q.usage()["cache_fill_suspended"] + + +def test_failed_mutation_is_charged_and_recovered_without_network(runtime, monkeypatch): + q, resolve, data, _ = runtime + read = blosc2.RemoteProxy.__getitem__ + + def fail_after_write(self, item): + read(self, item) + raise OSError("injected interrupted cache publication") + + with monkeypatch.context() as patch: + patch.setattr(blosc2.RemoteProxy, "__getitem__", fail_after_write) + np.testing.assert_array_equal(q.remote.read(resolve()), data) + with q.connect() as db: + assert db.execute("SELECT count(*) FROM remote_operations").fetchone()[0] == 1 + + def forbidden(*args, **kwargs): + raise AssertionError("offline recovery contacted source") + + monkeypatch.setattr(blosc2, "FsspecNDSource", forbidden) + q.remote.maintain() + assert q.usage()["remote_cache_used"] == 0 + assert q.usage()["reserved"] == 0 + + +def test_replace_reuses_path_while_trash_is_charged(runtime): + q, resolve, data, path = runtime + q.remote.read(resolve(), slice(0, 5000)) + before = q.usage()["remote_cache_used"] + cold = path.read_bytes() + q.publish(path, cold, expected=signature(path)) + assert q.usage()["remote_cache_used"] == before + np.testing.assert_array_equal(q.remote.read(resolve(), slice(5000, 10000)), data[5000:10000]) + with q.connect() as db: + assert db.execute("SELECT count(*) FROM remote_objects WHERE path IS NOT NULL").fetchone()[0] == 1 + q.remote.maintain() + assert q.usage()["remote_cache_used"] > 0 + + +def test_warm_artifact_metadata_and_cleanup(runtime): + q, resolve, data, path = runtime + q.remote.read(resolve(), slice(0, 10000)) + public = remote_proxy.raw_carrier(path) + assert public.schunk.meta["user-fixed"] == {"test": True} + assert public.schunk.vlmeta["user-variable"] == {"sample": 42} + artifact, etag, cleanup = q.remote.export(resolve()) + try: + assert len(etag) == 64 + exported = remote_proxy.raw_carrier(artifact) + np.testing.assert_array_equal(exported[:10000], data[:10000]) + assert exported.schunk.vlmeta["user-variable"] == {"sample": 42} + assert exported.schunk.meta["user-fixed"] == {"test": True} + assert q.usage()["working"] > 0 + q.remote.recover_exports() # Live owner must not be stolen. + assert artifact.exists() + del exported + finally: + cleanup() + assert q.usage()["working"] == 0 + assert not artifact.exists() + + +def test_reconcile_accounts_allocated_files_and_orphans(runtime): + q, resolve, _, _ = runtime + q.remote.read(resolve(), slice(0, 5000)) + with q.connect() as db: + rows = db.execute("SELECT relpath,charge_bytes FROM remote_generations").fetchall() + for rel, charge in rows: + assert measure(q.root / rel)[0] == charge + orphan = q.remote.root / ("a" * 32) / ("b" * 32) + orphan.mkdir(parents=True) + (orphan / "payload").write_bytes(b"x" * 10000) + q.remote.maintain() + assert not orphan.exists() + + +def test_missing_active_directory_retires_binding(runtime): + q, resolve, _, path = runtime + q.remote.read(resolve(), slice(0, 5000)) + with q.connect() as db: + _gid, private = db.execute("SELECT generation_id,relpath FROM remote_generations").fetchone() + import shutil + + shutil.rmtree(q.root / private) + q.remote.maintain() + assert q.usage()["remote_cache_used"] == 0 + assert path.exists() + + +def _worker_read(statedir, ready, errors): + """Spawn-safe deterministic source; no shared in-memory transport state.""" + try: + q = StorageQuota(statedir, 1 << 20, cache_backend="sparse") + data = np.random.default_rng(33).integers(0, 256, 60000, dtype="u1") + array = blosc2.asarray(data, chunks=(20000,), blocks=(5000,)) + url = "https://data.example/private-test.b2nd" + fs = fsspec.filesystem("memory") + fs.pipe_file(url, array.to_cframe()) + ready.wait(timeout=20) + for offset in [0, 10000, 30000, 50000, 0]: + carrier = remote_proxy.raw_carrier(q.root / "public/proxy.b2nd") + source = blosc2.FsspecNDSource(url, _filesystem=fs) + source.stamp = "immutable-multiprocess-test-source" + proxy = remote_proxy.ServerRemoteProxy( + source, + (array.shape, array.dtype, array.chunks, array.blocks), + carrier, + carrier.schunk.vlmeta["b2o"], + ) + actual = q.remote.read(proxy, slice(offset, offset + 5000)) + np.testing.assert_array_equal(actual, data[offset : offset + 5000]) + errors.put(None) + except BaseException as exc: + errors.put(repr(exc)) + + +def test_multiple_workers_share_generation(runtime): + import multiprocessing + + q, _, _, _ = runtime + context = multiprocessing.get_context("spawn") + ready, errors = context.Barrier(4), context.Queue() + workers = [context.Process(target=_worker_read, args=(str(q.root), ready, errors)) for _ in range(4)] + for worker in workers: + worker.start() + try: + for _ in workers: + assert errors.get(timeout=30) is None + for worker in workers: + worker.join(timeout=10) + assert worker.exitcode == 0 + finally: + for worker in workers: + if worker.is_alive(): + worker.terminate() + worker.join() + errors.close() + with q.connect() as db: + assert db.execute("SELECT count(*) FROM remote_generations WHERE state='active'").fetchone()[0] == 1 + assert db.execute("SELECT count(*) FROM remote_operations").fetchone()[0] == 0 + + +def test_old_authorization_cannot_coldify_replacement(runtime): + q, resolve, data, path = runtime + old = resolve() + replacement = remote_proxy.raw_carrier(path) + payload = dict(replacement.schunk.vlmeta["b2o"], max_cache_bytes=10000) + cold = remote_proxy.cold_cframe(replacement, payload) + del replacement + q.publish(path, cold, expected=signature(path)) + # A replacement racing source authorization can be observed by a late stat. + old.carrier_generation = signature(path) + np.testing.assert_array_equal(q.remote.read(old, slice(0, 5000)), data[:5000]) + assert path.read_bytes() == cold + q.remote.maintain() + + +def _worker_die(statedir): + import os + + q = StorageQuota(statedir, 1 << 20, cache_backend="sparse") + data = np.random.default_rng(33).integers(0, 256, 60000, dtype="u1") + array = blosc2.asarray(data, chunks=(20000,), blocks=(5000,)) + url = "https://data.example/private-test.b2nd" + fs = fsspec.filesystem("memory") + fs.pipe_file(url, array.to_cframe()) + carrier = remote_proxy.raw_carrier(q.root / "public/proxy.b2nd") + source = blosc2.FsspecNDSource(url, _filesystem=fs) + proxy = remote_proxy.ServerRemoteProxy( + source, (array.shape, array.dtype, array.chunks, array.blocks), carrier, carrier.schunk.vlmeta["b2o"] + ) + original = blosc2.Proxy._store_chunk + + def interrupted(self, *args): + original(self, *args) + os._exit(17) + + blosc2.Proxy._store_chunk = interrupted + q.remote.read(proxy, nchunk=0) + os._exit(18) # The fault injection must actually hit a mutation boundary. + + +def test_process_death_releases_ownership_and_discards_dirty_generation(runtime): + import multiprocessing + + q, resolve, data, path = runtime + worker = multiprocessing.get_context("spawn").Process(target=_worker_die, args=(str(q.root),)) + worker.start() + worker.join(timeout=20) + if worker.is_alive(): + worker.terminate() + worker.join() + pytest.fail("worker did not reach the mutation boundary") + assert worker.exitcode == 17 + q.remote.maintain() + assert q.usage()["remote_cache_used"] == 0 + assert q.usage()["reserved"] == 0 + assert path.exists() + np.testing.assert_array_equal(q.remote.read(resolve()), data) + + +def test_failed_parent_cleanup_keeps_its_charge(runtime, monkeypatch): + from pathlib import Path + + q, resolve, _, path = runtime + q.remote.read(resolve(), slice(0, 5000)) + q.publish(path, None, expected=signature(path)) + original = Path.rmdir + + def denied(parent): + if parent.parent == q.remote.root: + raise PermissionError("injected directory cleanup failure") + return original(parent) + + with monkeypatch.context() as patch: + patch.setattr(Path, "rmdir", denied) + q.remote.maintain() + with q.connect() as db: + # APFS can report zero allocated directory blocks; ownership still + # must survive so cleanup retries even when its byte charge is zero. + assert db.execute("SELECT count(*) FROM remote_objects WHERE path IS NULL").fetchone()[0] == 1 + q.remote.maintain() + assert q.usage()["remote_cache_used"] == 0 + with q.connect() as db: + assert db.execute("SELECT count(*) FROM remote_objects").fetchone()[0] == 0 diff --git a/caterva2/tests/test_storage_quota_api.py b/caterva2/tests/test_storage_quota_api.py index 7ba3d4fb..158d2271 100644 --- a/caterva2/tests/test_storage_quota_api.py +++ b/caterva2/tests/test_storage_quota_api.py @@ -46,7 +46,7 @@ async def quota_api(tmp_path, monkeypatch): def assert_usage(server): quota = server.quota_coordinator() - measured = sum(row[1] for row in quota.inventory()) + measured = sum(row[1] for row in quota.inventory()) + quota.usage()["remote_cache_used"] assert quota.usage()["used"] == measured <= quota.usage()["quota"] assert quota.usage()["reserved"] == quota.usage()["working"] == 0 @@ -208,11 +208,10 @@ async def test_disk_fetch_and_chunk_admit_growth_via_secure_source(quota_api, mo ) assert response.status_code == 200 path = server.settings.public / "proxy.b2nd" - before = path.stat().st_size response = await client.get("/api/fetch/@public/proxy.b2nd", params={"slice_": "0:10000"}) assert response.status_code == 200, response.text np.testing.assert_array_equal(blosc2.ndarray_from_cframe(response.content)[:], data[:10000]) - assert path.stat().st_size > before + assert server.quota_coordinator().usage()["remote_cache_used"] > 0 response = await client.get("/api/chunk/@public/proxy.b2nd", params={"nchunk": 1}) assert response.status_code == 200, response.text np.testing.assert_array_equal( @@ -223,3 +222,71 @@ async def test_disk_fetch_and_chunk_admit_growth_via_secure_source(quota_api, mo assert response.status_code == 200 cold = blosc2.ndarray_from_cframe(response.content) assert cold.schunk.vlmeta["b2o"] == payload + + +@pytest.mark.asyncio +@pytest.mark.parametrize("quota_enabled", [True, False]) +async def test_sparse_boundary_lifecycle(quota_api, monkeypatch, quota_enabled): + server, client, _ = quota_api + monkeypatch.setattr(server.settings, "quota", 100_000 if quota_enabled else 0) + fs = fsspec.filesystem("memory") + data = np.random.default_rng(71).integers(0, 256, 30000, dtype="u1") + array = blosc2.asarray(data, chunks=(10000,), blocks=(10000,)) + url = "https://data.example/sparse-boundary.b2nd" + fs.pipe_file(url, array.to_cframe()) + monkeypatch.setattr( + remote_proxy, + "policy", + remote_proxy.Policy(enabled=True, allowed_hosts=("data.example",), cache_backend="sparse"), + ) + monkeypatch.setattr(remote_proxy, "_public_addresses", lambda *args: ("93.184.216.34",)) + monkeypatch.setattr(remote_proxy, "_https_filesystem", lambda *args: fs) + from blosc2.b2objects import make_b2object_carrier, write_b2object_payload + + carrier = make_b2object_carrier( + "remote_proxy", array.shape, array.dtype, chunks=array.chunks, blocks=array.blocks + ) + write_b2object_payload( + carrier, + { + "kind": "remote_proxy", + "version": 1, + "source": {"kind": "fsspec", "version": 1, "urlpath": url}, + "cache_policy": "disk", + "max_cache_bytes": 15000, + }, + ) + response = await client.post( + "/api/upload/@public/sparse.b2nd", files={"file": ("sparse.b2nd", carrier.to_cframe())} + ) + assert response.status_code == 200, response.text + for i in [0, 1, 2, 0]: + response = await client.get("/api/chunk/@public/sparse.b2nd", params={"nchunk": i}) + assert response.status_code == 200, response.text + np.testing.assert_array_equal( + np.frombuffer(blosc2.decompress(response.content), dtype="u1"), data[i * 10000 : (i + 1) * 10000] + ) + quota = server.quota_coordinator() + assert quota.usage()["remote_cache_used"] > 10000 + public = blosc2.blosc2_ext.open(str(server.settings.public / "sparse.b2nd"), "r", 0) + assert public.schunk.vlmeta.get("proxy-fetched") is None + with quota.connect() as db: + assert db.execute("SELECT count(*) FROM remote_generations WHERE state='active'").fetchone()[0] == 1 + response = await client.get("/api/download/@public/sparse.b2nd") + assert response.status_code == 200, response.text + exported = blosc2.ndarray_from_cframe(response.content) + np.testing.assert_array_equal(exported[:10000], data[:10000]) + assert quota.usage()["working"] == 0 + response = await client.get("/api/download/@public/sparse.b2nd", headers={"Range": "bytes=0-127"}) + assert response.status_code == 206, response.text + assert len(response.content) == 128 + assert quota.usage()["working"] == 0 + response = await client.get( + "/api/download/@public/sparse.b2nd", headers={"Range": "bytes=0-127", "If-Range": '"old"'} + ) + assert response.status_code == 200 + assert quota.usage()["working"] == 0 + response = await client.post("/api/remove/@public/sparse.b2nd") + assert response.status_code == 200, response.text + quota.remote.cleanup() + assert quota.usage()["remote_cache_used"] == 0 diff --git a/doc/utilities/cat2-server.md b/doc/utilities/cat2-server.md index fb71b90e..502216f7 100644 --- a/doc/utilities/cat2-server.md +++ b/doc/utilities/cat2-server.md @@ -38,16 +38,18 @@ A persisted `blosc2.RemoteProxy` is a B2ND carrier that asks Caterva2 to read another dataset. Persisted `MEMORY` carriers are accepted under the same source policy but execute without retained caching (using the same no-retention execution path as `NONE`), avoiding unmanaged memory use on the server while preserving -the requested client limit for download. With a `DISK` cache, fetched compressed -chunks are retained inside the carrier up to the proxy's `max_cache_bytes` (or -unbounded when `max_cache_bytes` is `None`) when no customer quota is configured. -With a quota, shared storage admission permits cache growth when capacity is -available; otherwise misses are returned without retention. +the requested client limit for download. With a `DISK` cache, Caterva2 stores +fetched compressed chunks in a private sparse runtime generation. The public B2ND +carrier remains portable and contains only the descriptor and user metadata. The +default proxy limit is 256 MiB; `max_cache_bytes = None` remains unbounded per +proxy. With or without a customer quota, sparse lifecycle accounting remains +active; quota controls admission and pruning, while a denied fill falls back to a +no-retention read. Caterva2 can inspect and report its stored shape, dtype, chunk, block, and proxy metadata without contacting the source. Outbound resolution is disabled by default. -The initial opt-in backend supports public, credential-free HTTPS sources: +The runtime supports public, credential-free HTTPS sources: ```toml [server.remote_proxy] @@ -72,10 +74,9 @@ The limits validate the remote array's structure and bound connection time and concurrent range fetches. They do not impose a network-work budget on each API request. The proxy's own cache limit bounds its retained compressed payload, while a configured customer `quota` additionally bounds stored dataset bytes, -including carrier metadata. A SQLite ledger shared with ordinary writers admits -the exact serialized replacement size before disk publication. A denied fill -does not fail a successfully fetched result. Without a quota, normal bounded or -unbounded DISK caching applies. +including private sparse generations. A SQLite ledger shared with ordinary +writers admits ordinary publication work and sparse cache growth. A denied fill +does not fail a successfully fetched result. Public S3 objects are supported through credential-free HTTPS object URLs. Native `s3://` resolution, private-source credentials, and remote references @@ -103,18 +104,13 @@ directory. Administrators must provision operational headroom separately. Uploads, imports, expressions, append/chunk writes, HDF5 unfolding, notebooks, copies, moves, deletions, and remote DISK fills share admission. Under pressure, -admission may cold-replace up to four previously validated DISK cache carriers, -oldest first, preserving their descriptors and user metadata. Ordinary datasets -are never automatically pruned. A proxy's own payload cap still applies. - -The first implementation builds replacements in memory, reserves exact growth, -writes a complete staging file, and atomically replaces the target. It is a -correctness-oriented path with whole-file write amplification, not an in-place -optimization. `[server] quota_work_bytes` (default `"1G"`) bounds the aggregate -reserved disk staging bytes separately from customer quota. A candidate larger -than this budget cannot be persisted, even if customer quota has room. This is -not a hard RAM limit or a bound on HTTP upload spooling, SQLite/lock overhead, -filesystem allocated blocks, or old inodes retained by open readers. Provision +the server prunes least-recently-used sparse chunks in bounded batches. Ordinary +datasets are never automatically pruned. A proxy's own payload cap still applies. + +`[server] quota_work_bytes` (default `"1G"`) bounds aggregate reserved staging +and export artifacts separately from customer quota. Sparse cache generations +charge allocated filesystem blocks, including frame metadata and directory +entries. This is not a hard RAM limit or a physical-volume guarantee; provision and monitor the underlying volume accordingly. Per-path OS locks and generation checks prevent stale candidates from replacing @@ -132,3 +128,49 @@ generation conflicts return 409, and ledger errors return 503. Remote reads may instead fall back to no retention. `StorageQuota.usage()` exposes committed, reserved and disk working bytes for internal diagnostics; there is no new public administration endpoint. + +## Sparse remote cache lifecycle + +Caterva2 uses private sparse runtime generations for retained DISK caches. The +implementation requires Python-Blosc2's authorized-source `with_sparse_cache`, +atomic `read_cached`, and offline `trim_sparse_cache` APIs, plus the C-Blosc2 +sparse eviction improvements bundled after Python-Blosc2 `a77ac97b`. + +Mutable DISK chunks live under `.remote-cache//` +outside dataset roots. The public file stays a portable carrier. First authorized +access copies valid warm seed chunks and then replaces the public carrier with a +cold copy preserving user metadata. Subsequent misses never reimport evicted seed +chunks. MEMORY and NONE continue without retention. Upload itself makes no remote +request. + +The schema-v2 registry exists even when customer quota is disabled. Sparse files, +frame metadata, locks within generations, and generation/object directories count +by allocated filesystem blocks, falling back to apparent length where allocation +information is unavailable. Ordinary datasets retain their existing apparent-size +accounting. Builds and fills reserve estimates in the same ledger used by uploads. +Actual growth can overshoot customer quota; further fills are suspended until +reclamation reaches a 90% low watermark. Ordinary datasets are never evicted. + +Removal/replacement retires the old generation; private bytes remain charged until +cleanup actually removes them. Copy and move give the destination a cold carrier +and independent identity. Maintenance wakes every 30 seconds, skips busy owners, +and retries retirement, interrupted-operation recovery, and pruning. Cache misses +still return source data if admission or local retention fails. Warm downloads use +private, immutable export artifacts with ETags and support ranges; completion and +cancellation release artifact ownership. `include_cache=false` stays cold and +non-mutating. UI consumers that require bytes retain their existing in-memory +response contract. + +The implementation intentionally uses conservative settings: full-generation +measurement/fsync after writes, whole-generation discard after interrupted work, +at most 64 chunks/four generations per prune batch, and 1 GiB operational free-space +headroom for migration. One warm export reserves the full configured +`quota_work_bytes` budget until its response finishes. These choices establish an +endpoint benchmark baseline; incremental accounting, tighter staging estimates, +bounded inventories, and configurable maintenance tuning remain follow-up work. +Process-death recovery is covered; this does not claim power-loss atomicity or a +hard filesystem/RAM bound. + +Schema v2 is the initial Caterva2 runtime schema. The previous contiguous-carrier +implementation is retained only in the benchmark report and is not a supported +runtime backend. diff --git a/examples/benchmarks/remote_proxy_v7.md b/examples/benchmarks/remote_proxy_v7.md new file mode 100644 index 00000000..e9b63193 --- /dev/null +++ b/examples/benchmarks/remote_proxy_v7.md @@ -0,0 +1,58 @@ +# RemoteProxy v5 versus experimental sparse v7 + +Measured through the real `/api/chunk` and `/api/fetch` ASGI routes, using the +rebuilt Python-Blosc2 extension after `a77ac97b` (C-Blosc2 `a54e259`). V5 is the +unmodified Caterva2 checkout at `a1de98c`; v7 is the opt-in implementation in this +change. Both use the same Python-Blosc2 implementation and C library. The source +transport is an injected deterministic in-memory range source, so these measure +server overhead rather than external HTTPS latency. Source policy resolution, +array/chunk serialization, locking, SQLite admission, filesystem writes, and fsync +are exercised. Every response is compared with the original random uint8 data. + +Runs were sequential, with three fresh-state samples for every backend/workload. +The table gives median total seconds; speedup is v5 time divided by v7 time. +Values below 1 mean sparse is slower. Raw samples, request latency distributions, +account usage, and private entry counts are in `remote_proxy_v7_results/`. + +- Small: 8 MiB source, 32 chunks of 256 KiB, eight blocks per chunk. Churn retains + approximately 1 MiB (four chunks). +- Large: 64 MiB source, 64 chunks of 1 MiB, eight blocks per chunk. Churn retains + approximately 32 MiB (32 chunks). +- Cold fills every chunk once. Warm measures three full passes after prefill. + Churn measures three full passes with the stated limit. Partial fetches one + block per chunk in round-robin order, for eight passes. Non-churn cases use an + unlimited per-proxy payload cap; customer quota is 1 GiB throughout. + +| Case | Workload | v5 seconds | sparse v7 seconds | Speedup | +| --- | --- | ---: | ---: | ---: | +| small | cold | 0.354 | 0.378 | 0.94x | +| small | warm | 0.916 | 0.379 | 2.42x | +| small | churn | 0.743 | 1.302 | 0.57x | +| small | partial | 2.567 | 1.296 | 1.98x | +| large | cold | 2.409 | 0.911 | 2.65x | +| large | warm | 9.020 | 1.023 | 8.82x | +| large | churn | 5.902 | 3.107 | 1.90x | +| large | partial | 18.827 | 2.799 | 6.73x | + +At the server boundary, sparse wins warm hits even for small caches because v5 +still snapshots and serializes the whole carrier on its read path. For a 1 MiB +resident cache, sparse's per-operation locking, metadata, fsync, and accounting +costs outweigh the saved copying during churn. At 32 MiB resident size, avoiding +whole-carrier work wins all four workloads. These results support keeping sparse +opt-in and profiling small-cache overhead before considering hybrid storage. + +This is an experimental baseline, not the completion of every v7 acceptance +criterion. It uses a full-generation measurement/fsync fallback after mutation; +source authorization/transport is retained but external DNS/TLS/network time is +not measured. Multiple workers and process death have correctness tests, not yet +throughput benchmarks. Millions of logical chunks, long-running quota-pressure +convergence, peak RSS, and power-loss durability remain separate acceptance work. + +Reproduce with the `blosc2` conda interpreter and `PYTHONPATH` selecting the +appropriate checkout; run each command sequentially. Use `--backend contiguous` +with the original v5 checkout, and `--backend sparse` with this implementation. + +```sh +python examples/benchmarks/remote_proxy_v7.py --backend sparse --repeats 3 --output small.json +python examples/benchmarks/remote_proxy_v7.py --backend sparse --chunk-size 1048576 --nchunks 64 --cache-chunks 32 --repeats 3 --output large.json +``` diff --git a/examples/benchmarks/remote_proxy_v7.py b/examples/benchmarks/remote_proxy_v7.py new file mode 100644 index 00000000..c350347c --- /dev/null +++ b/examples/benchmarks/remote_proxy_v7.py @@ -0,0 +1,155 @@ +"""Measure real ASGI fetch/chunk routes against a deterministic local range source. + +Run with PYTHONPATH selecting the desired Caterva2 checkout and the blosc2 conda +interpreter. The source transport is injected; endpoint resolution, serialization, +SQLite admission, locks, cache writes and fsync are real. No network latency is +simulated. Each sample uses a fresh state directory and validates all results. +""" + +import argparse +import asyncio +import json +import os +import pathlib +import statistics +import tempfile +import time +import types +import uuid + +os.environ.setdefault("CATERVA2_SECRET", "local-benchmark-only") +import blosc2 +import fsspec +import httpx +import numpy as np +from blosc2.b2objects import make_b2object_carrier, write_b2object_payload + +from caterva2.services import remote_proxy, server + + +async def sample(backend, workload, chunk_size, nchunks, cache_chunks): + data = np.random.default_rng(7).integers(0, 256, chunk_size * nchunks, dtype="u1") + array = blosc2.asarray( + data, + chunks=(chunk_size,), + blocks=(chunk_size // 8,), + cparams={"nthreads": 1}, + dparams={"nthreads": 1}, + ) + url = "https://data.example/benchmark.b2nd" + fs = fsspec.filesystem("memory") + fs.pipe_file(url, array.to_cframe()) + fields = {"enabled": True, "allowed_hosts": ("data.example",)} + if backend == "sparse": + fields["cache_backend"] = backend + remote_proxy.policy = remote_proxy.Policy(**fields) + remote_proxy._public_addresses = lambda *args: ("93.184.216.34",) + remote_proxy._https_filesystem = lambda *args: fs + user = types.SimpleNamespace(id=uuid.uuid4(), is_superuser=True, is_active=True) + server.app.dependency_overrides[server.current_active_user] = lambda: user + server.app.dependency_overrides[server.optional_user] = lambda: user + with tempfile.TemporaryDirectory(prefix="cat2-bench-") as directory: + root = pathlib.Path(directory) + server.settings.statedir = root + server.settings.quota = 1 << 30 + server.settings.publish_root = None + server._quota_instances = {} + for name in ("public", "shared", "personal"): + path = root / name + path.mkdir() + setattr(server.settings, name, path) + carrier = make_b2object_carrier( + "remote_proxy", array.shape, array.dtype, chunks=array.chunks, blocks=array.blocks + ) + limit = cache_chunks * (chunk_size + 256) if workload == "churn" else None + write_b2object_payload( + carrier, + { + "kind": "remote_proxy", + "version": 1, + "source": {"kind": "fsspec", "version": 1, "urlpath": url}, + "cache_policy": "disk", + "max_cache_bytes": limit, + }, + ) + async with httpx.AsyncClient( + transport=httpx.ASGITransport(app=server.app), base_url="http://test" + ) as client: + response = await client.post( + "/api/upload/@public/proxy.b2nd", files={"file": ("proxy.b2nd", carrier.to_cframe())} + ) + assert response.status_code == 200, response.text + + async def read(n, block=None): + lo = n * chunk_size + (block or 0) * (chunk_size // 8) + length = chunk_size if block is None else chunk_size // 8 + if block is None: + response = await client.get("/api/chunk/@public/proxy.b2nd", params={"nchunk": n}) + assert response.status_code == 200, response.text + actual = np.frombuffer(blosc2.decompress(response.content), dtype="u1") + else: + response = await client.get( + "/api/fetch/@public/proxy.b2nd", params={"slice_": f"{lo}:{lo + length}"} + ) + assert response.status_code == 200, response.text + actual = blosc2.ndarray_from_cframe(response.content)[:] + np.testing.assert_array_equal(actual, data[lo : lo + length]) + + if workload == "warm": + for n in range(nchunks): + await read(n) + operations = [(n, None) for n in range(nchunks)] + if workload in ("churn", "warm"): + operations *= 3 + elif workload == "partial": + operations = [(n, b) for b in range(8) for n in range(nchunks)] + times = [] + start = time.perf_counter() + for n, b in operations: + tick = time.perf_counter() + await read(n, b) + times.append(time.perf_counter() - tick) + elapsed = time.perf_counter() - start + usage = server.quota_coordinator().usage() + private = root / ".remote-cache" + return { + "seconds": elapsed, + "requests": len(times), + "median_ms": statistics.median(times) * 1000, + "p95_ms": float(np.quantile(times, 0.95)) * 1000, + "private_entries": sum(1 for _ in private.rglob("*")) if private.exists() else 0, + "usage": usage, + } + + +async def main(args): + result = { + "backend": args.backend, + "chunk_size": args.chunk_size, + "nchunks": args.nchunks, + "cache_chunks": args.cache_chunks, + "source": "in-memory range transport; real ASGI routes", + "results": {}, + } + for workload in ("cold", "warm", "churn", "partial"): + values = [ + await sample(args.backend, workload, args.chunk_size, args.nchunks, args.cache_chunks) + for _ in range(args.repeats) + ] + result["results"][workload] = { + "median_seconds": statistics.median(v["seconds"] for v in values), + "samples": values, + } + pathlib.Path(args.output).write_text(json.dumps(result, indent=2) + "\n") + print(json.dumps({k: v["median_seconds"] for k, v in result["results"].items()}, indent=2)) + + +if __name__ == "__main__": + parser = argparse.ArgumentParser() + parser.add_argument("--backend", choices=["contiguous", "sparse"], required=True) + parser.add_argument("--chunk-size", type=int, default=262144) + parser.add_argument("--nchunks", type=int, default=32) + parser.add_argument("--cache-chunks", type=int, default=4) + parser.add_argument("--repeats", type=int, default=3) + parser.add_argument("--output", required=True) + asyncio.run(main(parser.parse_args())) diff --git a/examples/benchmarks/remote_proxy_v7_results/large-contiguous.json b/examples/benchmarks/remote_proxy_v7_results/large-contiguous.json new file mode 100644 index 00000000..6f70927a --- /dev/null +++ b/examples/benchmarks/remote_proxy_v7_results/large-contiguous.json @@ -0,0 +1,197 @@ +{ + "backend": "contiguous", + "chunk_size": 1048576, + "nchunks": 64, + "cache_chunks": 32, + "source": "in-memory range transport; real ASGI routes", + "results": { + "cold": { + "median_seconds": 2.4094051250140183, + "samples": [ + { + "seconds": 2.7027057920058724, + "requests": 64, + "median_ms": 39.1818750067614, + "p95_ms": 67.01374368858524, + "private_entries": 0, + "usage": { + "used": 67112067, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + }, + { + "seconds": 2.4094051250140183, + "requests": 64, + "median_ms": 36.10052049043588, + "p95_ms": 61.19914729933952, + "private_entries": 0, + "usage": { + "used": 67112067, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + }, + { + "seconds": 2.3839995409944095, + "requests": 64, + "median_ms": 35.59593750105705, + "p95_ms": 62.05508759740041, + "private_entries": 0, + "usage": { + "used": 67112067, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + } + ] + }, + "warm": { + "median_seconds": 9.020453625009395, + "samples": [ + { + "seconds": 8.896607916016364, + "requests": 192, + "median_ms": 45.822853498975746, + "p95_ms": 51.26657483924646, + "private_entries": 0, + "usage": { + "used": 67112067, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + }, + { + "seconds": 9.074710207991302, + "requests": 192, + "median_ms": 46.756520998314954, + "p95_ms": 51.331266544002574, + "private_entries": 0, + "usage": { + "used": 67112067, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + }, + { + "seconds": 9.020453625009395, + "requests": 192, + "median_ms": 46.377249993383884, + "p95_ms": 52.138522831955925, + "private_entries": 0, + "usage": { + "used": 67112067, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + } + ] + }, + "churn": { + "median_seconds": 5.901664750010241, + "samples": [ + { + "seconds": 5.901664750010241, + "requests": 192, + "median_ms": 34.027374495053664, + "p95_ms": 37.652381087536924, + "private_entries": 0, + "usage": { + "used": 33556822, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + }, + { + "seconds": 5.898537458997453, + "requests": 192, + "median_ms": 34.00210449763108, + "p95_ms": 38.733470789156854, + "private_entries": 0, + "usage": { + "used": 33556822, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + }, + { + "seconds": 6.043646958016325, + "requests": 192, + "median_ms": 33.65360450698063, + "p95_ms": 38.70028110104613, + "private_entries": 0, + "usage": { + "used": 33556822, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + } + ] + }, + "partial": { + "median_seconds": 18.82724495898583, + "samples": [ + { + "seconds": 19.35881325000082, + "requests": 512, + "median_ms": 37.043603995698504, + "p95_ms": 47.79766858700895, + "private_entries": 0, + "usage": { + "used": 67112067, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + }, + { + "seconds": 18.82724495898583, + "requests": 512, + "median_ms": 36.18597949389368, + "p95_ms": 43.73342949838843, + "private_entries": 0, + "usage": { + "used": 67112067, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + }, + { + "seconds": 18.746478707995266, + "requests": 512, + "median_ms": 36.359187492053024, + "p95_ms": 41.93695449939696, + "private_entries": 0, + "usage": { + "used": 67112067, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + } + ] + } + } +} diff --git a/examples/benchmarks/remote_proxy_v7_results/large-sparse.json b/examples/benchmarks/remote_proxy_v7_results/large-sparse.json new file mode 100644 index 00000000..f87c8ca6 --- /dev/null +++ b/examples/benchmarks/remote_proxy_v7_results/large-sparse.json @@ -0,0 +1,233 @@ +{ + "backend": "sparse", + "chunk_size": 1048576, + "nchunks": 64, + "cache_chunks": 32, + "source": "in-memory range transport; real ASGI routes", + "results": { + "cold": { + "median_seconds": 0.9108010419877246, + "samples": [ + { + "seconds": 0.9474686250032391, + "requests": 64, + "median_ms": 14.094083002419211, + "p95_ms": 16.303023259388283, + "private_entries": 69, + "usage": { + "used": 67379665, + "dataset_used": 465, + "remote_cache_used": 67379200, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + }, + { + "seconds": 0.9108010419877246, + "requests": 64, + "median_ms": 14.194062998285517, + "p95_ms": 15.245898203284014, + "private_entries": 69, + "usage": { + "used": 67379665, + "dataset_used": 465, + "remote_cache_used": 67379200, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + }, + { + "seconds": 0.9003570000058971, + "requests": 64, + "median_ms": 13.841250503901392, + "p95_ms": 15.224403554748278, + "private_entries": 69, + "usage": { + "used": 67379665, + "dataset_used": 465, + "remote_cache_used": 67379200, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + } + ] + }, + "warm": { + "median_seconds": 1.023014083999442, + "samples": [ + { + "seconds": 1.023014083999442, + "requests": 192, + "median_ms": 5.282708501908928, + "p95_ms": 5.861722548434045, + "private_entries": 69, + "usage": { + "used": 67379665, + "dataset_used": 465, + "remote_cache_used": 67379200, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + }, + { + "seconds": 1.0096354999986943, + "requests": 192, + "median_ms": 5.2294159977464005, + "p95_ms": 5.614669049100484, + "private_entries": 69, + "usage": { + "used": 67379665, + "dataset_used": 465, + "remote_cache_used": 67379200, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + }, + { + "seconds": 1.0235531250073109, + "requests": 192, + "median_ms": 5.277042000670917, + "p95_ms": 5.745273114007431, + "private_entries": 69, + "usage": { + "used": 67379665, + "dataset_used": 465, + "remote_cache_used": 67379200, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + } + ] + }, + "churn": { + "median_seconds": 3.1067567499994766, + "samples": [ + { + "seconds": 3.1067567499994766, + "requests": 192, + "median_ms": 16.55749950441532, + "p95_ms": 17.359960582689382, + "private_entries": 37, + "usage": { + "used": 33694165, + "dataset_used": 469, + "remote_cache_used": 33693696, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + }, + { + "seconds": 3.119276000012178, + "requests": 192, + "median_ms": 16.524687482160516, + "p95_ms": 17.24347105337074, + "private_entries": 37, + "usage": { + "used": 33694165, + "dataset_used": 469, + "remote_cache_used": 33693696, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + }, + { + "seconds": 3.0811172499961685, + "requests": 192, + "median_ms": 16.44700049655512, + "p95_ms": 17.070210304518696, + "private_entries": 37, + "usage": { + "used": 33694165, + "dataset_used": 469, + "remote_cache_used": 33693696, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + } + ] + }, + "partial": { + "median_seconds": 2.798786833009217, + "samples": [ + { + "seconds": 2.798786833009217, + "requests": 512, + "median_ms": 4.363521002233028, + "p95_ms": 12.99517339648446, + "private_entries": 69, + "usage": { + "used": 67379665, + "dataset_used": 465, + "remote_cache_used": 67379200, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + }, + { + "seconds": 2.7694656669918913, + "requests": 512, + "median_ms": 4.326957990997471, + "p95_ms": 12.820158997783437, + "private_entries": 69, + "usage": { + "used": 67379665, + "dataset_used": 465, + "remote_cache_used": 67379200, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + }, + { + "seconds": 2.8031584580021445, + "requests": 512, + "median_ms": 4.344833010691218, + "p95_ms": 13.013622537255285, + "private_entries": 69, + "usage": { + "used": 67379665, + "dataset_used": 465, + "remote_cache_used": 67379200, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + } + ] + } + } +} diff --git a/examples/benchmarks/remote_proxy_v7_results/small-contiguous.json b/examples/benchmarks/remote_proxy_v7_results/small-contiguous.json new file mode 100644 index 00000000..b9375f83 --- /dev/null +++ b/examples/benchmarks/remote_proxy_v7_results/small-contiguous.json @@ -0,0 +1,197 @@ +{ + "backend": "contiguous", + "chunk_size": 262144, + "nchunks": 32, + "cache_chunks": 4, + "source": "in-memory range transport; real ASGI routes", + "results": { + "cold": { + "median_seconds": 0.35378579201642424, + "samples": [ + { + "seconds": 0.37498900000355206, + "requests": 32, + "median_ms": 10.977479003486224, + "p95_ms": 15.769316554360556, + "private_entries": 0, + "usage": { + "used": 8390767, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + }, + { + "seconds": 0.35378579201642424, + "requests": 32, + "median_ms": 11.347187508363277, + "p95_ms": 15.2050251403125, + "private_entries": 0, + "usage": { + "used": 8390767, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + }, + { + "seconds": 0.33900570901460014, + "requests": 32, + "median_ms": 10.016374988481402, + "p95_ms": 15.487539400055537, + "private_entries": 0, + "usage": { + "used": 8390767, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + } + ] + }, + "warm": { + "median_seconds": 0.9158176670025568, + "samples": [ + { + "seconds": 0.9222860420122743, + "requests": 96, + "median_ms": 9.433541490579955, + "p95_ms": 11.12018773710588, + "private_entries": 0, + "usage": { + "used": 8390767, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + }, + { + "seconds": 0.9154366670118179, + "requests": 96, + "median_ms": 9.241687497706152, + "p95_ms": 11.161926755448803, + "private_entries": 0, + "usage": { + "used": 8390767, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + }, + { + "seconds": 0.9158176670025568, + "requests": 96, + "median_ms": 9.394791501108557, + "p95_ms": 10.854114254470915, + "private_entries": 0, + "usage": { + "used": 8390767, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + } + ] + }, + "churn": { + "median_seconds": 0.7430978749762289, + "samples": [ + { + "seconds": 0.7430978749762289, + "requests": 96, + "median_ms": 7.8778125025564805, + "p95_ms": 8.609040996816475, + "private_entries": 0, + "usage": { + "used": 1049903, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + }, + { + "seconds": 0.7565104160166811, + "requests": 96, + "median_ms": 7.7562079968629405, + "p95_ms": 8.507520746206865, + "private_entries": 0, + "usage": { + "used": 1049903, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + }, + { + "seconds": 0.7259852919960395, + "requests": 96, + "median_ms": 7.727145493845455, + "p95_ms": 8.505781006533653, + "private_entries": 0, + "usage": { + "used": 1049903, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + } + ] + }, + "partial": { + "median_seconds": 2.5667394580086693, + "samples": [ + { + "seconds": 2.5667394580086693, + "requests": 256, + "median_ms": 9.81610399321653, + "p95_ms": 11.68914601294091, + "private_entries": 0, + "usage": { + "used": 8390767, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + }, + { + "seconds": 2.623016583005665, + "requests": 256, + "median_ms": 10.106271001859568, + "p95_ms": 11.919927012058906, + "private_entries": 0, + "usage": { + "used": 8390767, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + }, + { + "seconds": 2.562559834012063, + "requests": 256, + "median_ms": 9.794270488782786, + "p95_ms": 11.966812737227883, + "private_entries": 0, + "usage": { + "used": 8390767, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824 + } + } + ] + } + } +} diff --git a/examples/benchmarks/remote_proxy_v7_results/small-sparse.json b/examples/benchmarks/remote_proxy_v7_results/small-sparse.json new file mode 100644 index 00000000..5bf617e0 --- /dev/null +++ b/examples/benchmarks/remote_proxy_v7_results/small-sparse.json @@ -0,0 +1,233 @@ +{ + "backend": "sparse", + "chunk_size": 262144, + "nchunks": 32, + "cache_chunks": 4, + "source": "in-memory range transport; real ASGI routes", + "results": { + "cold": { + "median_seconds": 0.3775228329759557, + "samples": [ + { + "seconds": 0.38301454100292176, + "requests": 32, + "median_ms": 11.360541509930044, + "p95_ms": 12.926716300717088, + "private_entries": 37, + "usage": { + "used": 8528336, + "dataset_used": 464, + "remote_cache_used": 8527872, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + }, + { + "seconds": 0.3775228329759557, + "requests": 32, + "median_ms": 11.365937505615875, + "p95_ms": 12.147625148645602, + "private_entries": 37, + "usage": { + "used": 8528336, + "dataset_used": 464, + "remote_cache_used": 8527872, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + }, + { + "seconds": 0.3711140830127988, + "requests": 32, + "median_ms": 11.244020992307924, + "p95_ms": 12.172150299011264, + "private_entries": 37, + "usage": { + "used": 8528336, + "dataset_used": 464, + "remote_cache_used": 8527872, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + } + ] + }, + "warm": { + "median_seconds": 0.37866308400407434, + "samples": [ + { + "seconds": 0.37866308400407434, + "requests": 96, + "median_ms": 3.918457994586788, + "p95_ms": 4.255770494637545, + "private_entries": 37, + "usage": { + "used": 8528336, + "dataset_used": 464, + "remote_cache_used": 8527872, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + }, + { + "seconds": 0.37584758299635723, + "requests": 96, + "median_ms": 3.8801875052740797, + "p95_ms": 4.21906298288377, + "private_entries": 37, + "usage": { + "used": 8528336, + "dataset_used": 464, + "remote_cache_used": 8527872, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + }, + { + "seconds": 0.385695749981096, + "requests": 96, + "median_ms": 3.9368959987768903, + "p95_ms": 4.3533642528927885, + "private_entries": 37, + "usage": { + "used": 8528336, + "dataset_used": 464, + "remote_cache_used": 8527872, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + } + ] + }, + "churn": { + "median_seconds": 1.301555375015596, + "samples": [ + { + "seconds": 1.3065834580047522, + "requests": 96, + "median_ms": 13.58535401232075, + "p95_ms": 14.256333495723084, + "private_entries": 9, + "usage": { + "used": 1073620, + "dataset_used": 468, + "remote_cache_used": 1073152, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + }, + { + "seconds": 1.2991834589920472, + "requests": 96, + "median_ms": 13.525312504498288, + "p95_ms": 14.072707745071966, + "private_entries": 9, + "usage": { + "used": 1073620, + "dataset_used": 468, + "remote_cache_used": 1073152, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + }, + { + "seconds": 1.301555375015596, + "requests": 96, + "median_ms": 13.53947950701695, + "p95_ms": 14.202301994373556, + "private_entries": 9, + "usage": { + "used": 1073620, + "dataset_used": 468, + "remote_cache_used": 1073152, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + } + ] + }, + "partial": { + "median_seconds": 1.2956712499726564, + "samples": [ + { + "seconds": 1.2955502080149017, + "requests": 256, + "median_ms": 4.080937505932525, + "p95_ms": 11.657999762974214, + "private_entries": 37, + "usage": { + "used": 8528336, + "dataset_used": 464, + "remote_cache_used": 8527872, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + }, + { + "seconds": 1.2956712499726564, + "requests": 256, + "median_ms": 4.08633348706644, + "p95_ms": 11.630125496594701, + "private_entries": 37, + "usage": { + "used": 8528336, + "dataset_used": 464, + "remote_cache_used": 8527872, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + }, + { + "seconds": 1.308848874992691, + "requests": 256, + "median_ms": 4.083624997292645, + "p95_ms": 11.554166754649486, + "private_entries": 37, + "usage": { + "used": 8528336, + "dataset_used": 464, + "remote_cache_used": 8527872, + "reserved": 0, + "working": 0, + "quota": 1073741824, + "work_bytes": 1073741824, + "cache_fill_suspended": false + } + } + ] + } + } +} From 9a064dcb464112c708762f4099392fd9a61f7ae2 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 6 Sep 2026 14:33:50 +0200 Subject: [PATCH 07/20] Clarify sparse benchmark status --- caterva2/services/sparse_cache.py | 3 ++- examples/benchmarks/remote_proxy_v7.md | 16 ++++++++++------ 2 files changed, 12 insertions(+), 7 deletions(-) diff --git a/caterva2/services/sparse_cache.py b/caterva2/services/sparse_cache.py index a56afa9a..065a7b72 100644 --- a/caterva2/services/sparse_cache.py +++ b/caterva2/services/sparse_cache.py @@ -329,7 +329,8 @@ def uncached(): ) return result # Admission is deliberately coarse; the full-generation stat - # fallback is experimental until mutation reports are available. + # Full-generation measurement is the safe fallback until mutation reports + # are available. estimate = ( min( proxy.dtype.itemsize * math.prod(proxy.chunks), diff --git a/examples/benchmarks/remote_proxy_v7.md b/examples/benchmarks/remote_proxy_v7.md index e9b63193..d2148836 100644 --- a/examples/benchmarks/remote_proxy_v7.md +++ b/examples/benchmarks/remote_proxy_v7.md @@ -1,9 +1,10 @@ -# RemoteProxy v5 versus experimental sparse v7 +# RemoteProxy v5 versus sparse v7 Measured through the real `/api/chunk` and `/api/fetch` ASGI routes, using the rebuilt Python-Blosc2 extension after `a77ac97b` (C-Blosc2 `a54e259`). V5 is the -unmodified Caterva2 checkout at `a1de98c`; v7 is the opt-in implementation in this -change. Both use the same Python-Blosc2 implementation and C library. The source +unmodified Caterva2 checkout at `a1de98c`; v7 is the sparse-default implementation +in the current checkout. Both use the same Python-Blosc2 implementation and C library. +The source transport is an injected deterministic in-memory range source, so these measure server overhead rather than external HTTPS latency. Source policy resolution, array/chunk serialization, locking, SQLite admission, filesystem writes, and fsync @@ -38,10 +39,11 @@ At the server boundary, sparse wins warm hits even for small caches because v5 still snapshots and serializes the whole carrier on its read path. For a 1 MiB resident cache, sparse's per-operation locking, metadata, fsync, and accounting costs outweigh the saved copying during churn. At 32 MiB resident size, avoiding -whole-carrier work wins all four workloads. These results support keeping sparse -opt-in and profiling small-cache overhead before considering hybrid storage. +whole-carrier work wins all four workloads. These results support sparse as the +runtime default while retaining the contiguous path for explicit comparison and +rollback testing. -This is an experimental baseline, not the completion of every v7 acceptance +This is a benchmark baseline, not the completion of every v7 acceptance criterion. It uses a full-generation measurement/fsync fallback after mutation; source authorization/transport is retained but external DNS/TLS/network time is not measured. Multiple workers and process death have correctness tests, not yet @@ -51,6 +53,8 @@ convergence, peak RSS, and power-loss durability remain separate acceptance work Reproduce with the `blosc2` conda interpreter and `PYTHONPATH` selecting the appropriate checkout; run each command sequentially. Use `--backend contiguous` with the original v5 checkout, and `--backend sparse` with this implementation. +The backend switch belongs to the benchmark harness; Caterva2 runtime configuration +has no public backend selector and defaults to sparse. ```sh python examples/benchmarks/remote_proxy_v7.py --backend sparse --repeats 3 --output small.json From cb0d22df4f4d8a6652ba234f4875b00f03297af0 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Sun, 6 Sep 2026 14:52:39 +0200 Subject: [PATCH 08/20] Make cache maintenance interval configurable --- caterva2-server.sample.toml | 1 + caterva2/services/remote_proxy.py | 9 +++++++++ caterva2/services/server.py | 2 +- caterva2/tests/test_remote_proxy.py | 3 +++ doc/utilities/cat2-server.md | 5 +++++ 5 files changed, 19 insertions(+), 1 deletion(-) diff --git a/caterva2-server.sample.toml b/caterva2-server.sample.toml index 4759feec..23ad8b8d 100644 --- a/caterva2-server.sample.toml +++ b/caterva2-server.sample.toml @@ -52,6 +52,7 @@ register = true # allow users to register # max_rank = 16 # max_chunks = 10000000 # max_concurrency = 8 +# cache_maintenance_seconds = 60 # Mount a remote Caterva2 server's locally owned @public root as @labb. This # activates the bundled C2Cache provider. Repeat the table to mount more peers. diff --git a/caterva2/services/remote_proxy.py b/caterva2/services/remote_proxy.py index fcc0fa9a..6fef759a 100644 --- a/caterva2/services/remote_proxy.py +++ b/caterva2/services/remote_proxy.py @@ -50,6 +50,7 @@ class Policy: max_rank: int = 16 max_chunks: int = 10_000_000 max_concurrency: int = 8 + cache_maintenance_seconds: float = 60.0 cache_backend: str = "sparse" @@ -81,6 +82,7 @@ def configure(conf) -> None: max_rank = conf.get(".remote_proxy.max_rank", 16) max_chunks = conf.get(".remote_proxy.max_chunks", 10_000_000) max_concurrency = conf.get(".remote_proxy.max_concurrency", 8) + cache_maintenance_seconds = conf.get(".remote_proxy.cache_maintenance_seconds", 60.0) if not isinstance(enabled, bool): raise ValueError("remote_proxy.enabled must be true or false") @@ -90,6 +92,12 @@ def configure(conf) -> None: raise ValueError("remote_proxy.allowed_hosts must be a list of host names") if not isinstance(timeout, int | float) or isinstance(timeout, bool) or timeout <= 0: raise ValueError("remote_proxy.timeout must be positive") + if ( + not isinstance(cache_maintenance_seconds, int | float) + or isinstance(cache_maintenance_seconds, bool) + or cache_maintenance_seconds <= 0 + ): + raise ValueError("remote_proxy.cache_maintenance_seconds must be positive") for name, value in { "max_nbytes": max_nbytes, "max_rank": max_rank, @@ -107,6 +115,7 @@ def configure(conf) -> None: max_rank=max_rank, max_chunks=max_chunks, max_concurrency=max_concurrency, + cache_maintenance_seconds=float(cache_maintenance_seconds), ) diff --git a/caterva2/services/server.py b/caterva2/services/server.py index 50e5c0cd..a0599885 100644 --- a/caterva2/services/server.py +++ b/caterva2/services/server.py @@ -502,7 +502,7 @@ async def lifespan(app: FastAPI): async def cache_maintenance(): while True: - await asyncio.sleep(30) + await asyncio.sleep(remote_proxy.policy.cache_maintenance_seconds) try: await concurrency.run_in_threadpool(quota_coordinator().remote.maintain) except (OSError, sqlite3.Error, ValueError): diff --git a/caterva2/tests/test_remote_proxy.py b/caterva2/tests/test_remote_proxy.py index 1a4853d8..296fe3ad 100644 --- a/caterva2/tests/test_remote_proxy.py +++ b/caterva2/tests/test_remote_proxy.py @@ -105,9 +105,12 @@ def test_configured_https_destination_is_accepted(cache_policy, max_cache_bytes) { ".remote_proxy.enabled": True, ".remote_proxy.allowed_hosts": ["DATA.example", "data.example:8443"], + ".remote_proxy.cache_maintenance_seconds": 12.5, } ) ) + assert remote_proxy.policy.cache_maintenance_seconds == 12.5 + assert remote_proxy._validated_source( _payload( "https://data.example/array.b2nd", cache_policy=cache_policy, max_cache_bytes=max_cache_bytes diff --git a/doc/utilities/cat2-server.md b/doc/utilities/cat2-server.md index 502216f7..aa505cc4 100644 --- a/doc/utilities/cat2-server.md +++ b/doc/utilities/cat2-server.md @@ -60,8 +60,13 @@ max_nbytes = 1073741824 max_rank = 16 max_chunks = 10000000 max_concurrency = 8 +cache_maintenance_seconds = 60 ``` +The maintenance task runs every 60 seconds by default. Increase this interval for +large deployments to reduce background filesystem and SQLite scans; decrease it only +when faster cleanup is needed. + The allowlist is mandatory and matches normalized host names and explicit non-default ports exactly. Before connecting, Caterva2 resolves every address, rejects loopback, private, link-local, multicast, and other non-public results, From 17f970239a880aad9345905f372e0cc6e6b2e7c3 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Mon, 7 Sep 2026 14:29:02 +0200 Subject: [PATCH 09/20] Support immutable remote proxy descriptors --- caterva2/services/remote_proxy.py | 17 ++++++++++++++--- caterva2/tests/test_remote_proxy.py | 20 +++++++++++++++++++- caterva2/tests/test_sparse_cache.py | 2 +- caterva2/tests/test_storage_quota_api.py | 14 ++++++++++++-- doc/utilities/cat2-server.md | 5 +++++ examples/benchmarks/remote_proxy_v7.py | 7 ++++++- 6 files changed, 57 insertions(+), 8 deletions(-) diff --git a/caterva2/services/remote_proxy.py b/caterva2/services/remote_proxy.py index 6fef759a..32ed092c 100644 --- a/caterva2/services/remote_proxy.py +++ b/caterva2/services/remote_proxy.py @@ -24,6 +24,7 @@ import threading import weakref from dataclasses import dataclass +from inspect import signature from urllib.parse import urlsplit import aiohttp @@ -86,8 +87,11 @@ def configure(conf) -> None: if not isinstance(enabled, bool): raise ValueError("remote_proxy.enabled must be true or false") - if enabled and not hasattr(blosc2, "RemoteProxy"): - raise ValueError("remote_proxy.enabled requires a Python-Blosc2 version with RemoteProxy support") + if enabled and ( + not hasattr(blosc2, "RemoteProxy") + or "assume_immutable" not in signature(blosc2.RemoteProxy).parameters + ): + raise ValueError("remote_proxy.enabled requires a compatible Python-Blosc2 RemoteProxy") if not isinstance(hosts, list | tuple) or any(not isinstance(host, str) for host in hosts): raise ValueError("remote_proxy.allowed_hosts must be a list of host names") if not isinstance(timeout, int | float) or isinstance(timeout, bool) or timeout <= 0: @@ -237,10 +241,17 @@ def _validated_source(payload: dict) -> str: "server RemoteProxy supports only cache policies 'none', 'memory', and 'disk'" ) source = payload.get("source") - if not isinstance(source, dict) or set(source) != {"kind", "version", "urlpath"}: + if not isinstance(source, dict) or set(source) != { + "kind", + "version", + "urlpath", + "assume_immutable", + }: raise RemoteProxyDenied("server RemoteProxy supports only a versioned fsspec URL source") if source.get("kind") != "fsspec" or source.get("version") != 1: raise RemoteProxyDenied("server RemoteProxy supports only fsspec source version 1") + if not isinstance(source.get("assume_immutable"), bool): + raise RemoteProxyDenied("RemoteProxy source assume_immutable must be true or false") url = source.get("urlpath") if not isinstance(url, str): raise RemoteProxyDenied("RemoteProxy source URL must be a string") diff --git a/caterva2/tests/test_remote_proxy.py b/caterva2/tests/test_remote_proxy.py index 296fe3ad..096a133e 100644 --- a/caterva2/tests/test_remote_proxy.py +++ b/caterva2/tests/test_remote_proxy.py @@ -31,7 +31,7 @@ def _payload(url, *, cache_policy="none", max_cache_bytes=None): return { "kind": "remote_proxy", "version": 1, - "source": {"kind": "fsspec", "version": 1, "urlpath": url}, + "source": {"kind": "fsspec", "version": 1, "urlpath": url, "assume_immutable": True}, "cache_policy": cache_policy, "max_cache_bytes": max_cache_bytes, } @@ -193,6 +193,24 @@ def test_cache_specification_is_strict(payload, match): remote_proxy._validated_source(payload) +@pytest.mark.parametrize("value", [None, 1, "true"]) +def test_source_assume_immutable_must_be_boolean(value): + remote_proxy.policy = remote_proxy.Policy(enabled=True, allowed_hosts=("data.example",)) + payload = _payload("https://data.example/a.b2nd") + payload["source"]["assume_immutable"] = value + + with pytest.raises(remote_proxy.RemoteProxyDenied, match="must be true or false"): + remote_proxy._validated_source(payload) + + +def test_mutable_sources_are_accepted(): + remote_proxy.policy = remote_proxy.Policy(enabled=True, allowed_hosts=("data.example",)) + payload = _payload("https://data.example/a.b2nd") + payload["source"]["assume_immutable"] = False + + assert remote_proxy._validated_source(payload) == "https://data.example/a.b2nd" + + def test_private_resolution_is_rejected(monkeypatch): monkeypatch.setattr( socket, diff --git a/caterva2/tests/test_sparse_cache.py b/caterva2/tests/test_sparse_cache.py index d1ff68a4..3041b9f5 100644 --- a/caterva2/tests/test_sparse_cache.py +++ b/caterva2/tests/test_sparse_cache.py @@ -32,7 +32,7 @@ def runtime(tmp_path): payload = { "kind": "remote_proxy", "version": 1, - "source": {"kind": "fsspec", "version": 1, "urlpath": url}, + "source": {"kind": "fsspec", "version": 1, "urlpath": url, "assume_immutable": True}, "cache_policy": "disk", "max_cache_bytes": None, } diff --git a/caterva2/tests/test_storage_quota_api.py b/caterva2/tests/test_storage_quota_api.py index 158d2271..7f9256fd 100644 --- a/caterva2/tests/test_storage_quota_api.py +++ b/caterva2/tests/test_storage_quota_api.py @@ -195,7 +195,12 @@ async def test_disk_fetch_and_chunk_admit_growth_via_secure_source(quota_api, mo carrier = blosc2.ndarray_from_cframe(creator.to_cframe(cache_policy=blosc2.CachePolicy.DISK), copy=True) payload = dict(carrier.schunk.vlmeta["b2o"]) url = "https://data.example/quota.b2nd" - payload["source"] = {"kind": "fsspec", "version": 1, "urlpath": url} + payload["source"] = { + "kind": "fsspec", + "version": 1, + "urlpath": url, + "assume_immutable": True, + } carrier.schunk.vlmeta["b2o"] = payload fs.pipe_file(url, array.to_cframe()) monkeypatch.setattr( @@ -251,7 +256,12 @@ async def test_sparse_boundary_lifecycle(quota_api, monkeypatch, quota_enabled): { "kind": "remote_proxy", "version": 1, - "source": {"kind": "fsspec", "version": 1, "urlpath": url}, + "source": { + "kind": "fsspec", + "version": 1, + "urlpath": url, + "assume_immutable": True, + }, "cache_policy": "disk", "max_cache_bytes": 15000, }, diff --git a/doc/utilities/cat2-server.md b/doc/utilities/cat2-server.md index aa505cc4..787a52c1 100644 --- a/doc/utilities/cat2-server.md +++ b/doc/utilities/cat2-server.md @@ -51,6 +51,11 @@ default. The runtime supports public, credential-free HTTPS sources: +The source descriptor records `assume_immutable` as a boolean. Caterva2 resolves +the HTTPS source and its fsspec identity for each request, so existing mutable +`.b2nd` source behavior remains available when the descriptor contains +`assume_immutable=false`. + ```toml [server.remote_proxy] enabled = true diff --git a/examples/benchmarks/remote_proxy_v7.py b/examples/benchmarks/remote_proxy_v7.py index c350347c..6e96f5e0 100644 --- a/examples/benchmarks/remote_proxy_v7.py +++ b/examples/benchmarks/remote_proxy_v7.py @@ -67,7 +67,12 @@ async def sample(backend, workload, chunk_size, nchunks, cache_chunks): { "kind": "remote_proxy", "version": 1, - "source": {"kind": "fsspec", "version": 1, "urlpath": url}, + "source": { + "kind": "fsspec", + "version": 1, + "urlpath": url, + "assume_immutable": True, + }, "cache_policy": "disk", "max_cache_bytes": limit, }, From dc70c6de82688af41e4c2482d2c163ca1a562183 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Tue, 8 Sep 2026 13:11:34 +0200 Subject: [PATCH 10/20] Expose user attributes through metadata API and CLI --- caterva2/client.py | 9 ++++ caterva2/clients/cli.py | 8 ++++ caterva2/hdf5.py | 2 + caterva2/models.py | 3 ++ caterva2/services/srv_utils.py | 25 +++++++++-- .../templates/includes/info_metadata.html | 10 +++-- caterva2/tests/test_api.py | 23 +++++++++- caterva2/tests/test_attrs.py | 45 +++++++++++++++++++ caterva2/tests/test_cli.py | 28 ++++++++++++ caterva2/tests/test_ctable.py | 1 + caterva2/tests/test_hdf5_tree.py | 21 ++++++++- caterva2/tests/test_peers.py | 10 ++++- caterva2/tests/test_treestore.py | 7 ++- doc/reference/rest_api.rst | 19 ++++++++ doc/utilities/cat2-client.md | 5 +++ 15 files changed, 204 insertions(+), 12 deletions(-) create mode 100644 caterva2/tests/test_attrs.py diff --git a/caterva2/client.py b/caterva2/client.py index e918f3cf..ae5ec938 100644 --- a/caterva2/client.py +++ b/caterva2/client.py @@ -532,6 +532,15 @@ def vlmeta(self): schunk_meta = self.meta.get("schunk", self.meta) return schunk_meta.get("vlmeta", {}) + @property + def attrs(self): + """User attributes from cached metadata; changing them does not update the server. + + Older servers without an ``attrs`` field fall back to :attr:`vlmeta`. + """ + attrs = self.meta.get("attrs") + return self.vlmeta if attrs is None else attrs + def get_download_url(self, *, include_cache=True): """ Retrieves the download URL for the file. diff --git a/caterva2/clients/cli.py b/caterva2/clients/cli.py index 2277b59f..0a03664e 100644 --- a/caterva2/clients/cli.py +++ b/caterva2/clients/cli.py @@ -173,6 +173,12 @@ def cmd_info(client, args, url): return # Helpers + def _print_attrs(): + attrs = data.get("attrs") + if attrs is None: + attrs = (data.get("schunk") or data).get("vlmeta", {}) + print("attrs: " + json.dumps(attrs, indent=2, ensure_ascii=False)) + def _human_bytes(n): if n is None: return "N/A" @@ -219,6 +225,7 @@ def _filter_names(fl): print(f"cbytes : {_human_bytes(cbytes)}") print(f"ratio : {nbytes / cbytes:.2f}x" if nbytes and cbytes else "ratio : N/A") print(f"mtime : {mtime}") if mtime is not None else print("mtime : None") + _print_attrs() return # Extract fields @@ -256,6 +263,7 @@ def _filter_names(fl): print(f" filters: [{', '.join(fnames)}]") else: print(" filters: None") + _print_attrs() def _json_default(o): diff --git a/caterva2/hdf5.py b/caterva2/hdf5.py index 749cd600..358bcf6b 100644 --- a/caterva2/hdf5.py +++ b/caterva2/hdf5.py @@ -371,6 +371,8 @@ def open_leaf(cls, h5file, dsetname): self.dset = h5file[dsetname] if dsetname else h5file b2args = b2args_from_h5dset(self.dset) self.b2arr = blosc2.empty(self.dset.shape or (), dtype=self.dset.dtype, **b2args) + for name, value in b2attrs_from_h5dset(self.dset).items(): + self.b2arr.schunk.vlmeta.set_vlmeta(name, value, typesize=1) return self def __init__(self, b2arr, h5file=None, dsetname=None, *, writer=None): diff --git a/caterva2/models.py b/caterva2/models.py index e0ae08d2..391632cd 100644 --- a/caterva2/models.py +++ b/caterva2/models.py @@ -52,11 +52,13 @@ class SChunk(pydantic.BaseModel, extra="allow"): nbytes: int urlpath: str | None vlmeta: dict = {} + attrs: dict | None = None nchunks: int mtime: datetime.datetime | None = None class Metadata(pydantic.BaseModel): + attrs: dict | None = None shape: tuple chunks: tuple blocks: tuple @@ -83,6 +85,7 @@ class LazyArray(pydantic.BaseModel): class CTableMetadata(pydantic.BaseModel): + attrs: dict | None = None kind: str = "ctable" nrows: int ncols: int diff --git a/caterva2/services/srv_utils.py b/caterva2/services/srv_utils.py index 1f58c692..ca9bd59a 100644 --- a/caterva2/services/srv_utils.py +++ b/caterva2/services/srv_utils.py @@ -25,6 +25,7 @@ import fastapi import h5py import safer +from blosc2.b2objects import read_b2object_user_vlmeta from fastapi_users.exceptions import UserNotExists from sqlalchemy.future import select @@ -450,6 +451,19 @@ def is_hdf5_proxy_meta(meta): return vlmeta.get("_ftype") == "hdf5" +def user_attrs(obj): + """Read user attributes without resolving a saved remote source.""" + schunk = getattr(obj, "schunk", obj) + marker = getattr(schunk, "meta", {}).get("b2o", {}) + if isinstance(marker, dict) and marker.get("kind") == "remote_proxy": + return read_b2object_user_vlmeta(obj) + vlmeta = schunk.vlmeta + internal = {"fill_nonce", "fill_state", "published_url"} + if vlmeta.get("_ftype") == "hdf5": + internal.update({"_ftype", "_dsetname"}) + return {key: vlmeta[key] for key in vlmeta if key not in internal} + + def read_metadata(obj, mtime=None): # `mtime` is used when `obj` is an already-opened object (e.g. a container # leaf) with no file of its own; callers pass the container's mtime. @@ -515,7 +529,9 @@ def read_metadata(obj, mtime=None): schunk = get_model_from_obj(proxy.b2arr.schunk, models.SChunk, cparams=cparams) schunk.cratio = proxy.cratio schunk.cbytes = proxy.cbytes - return get_model_from_obj(proxy, models.Metadata, schunk=schunk, mtime=mtime) + return get_model_from_obj( + proxy, models.Metadata, schunk=schunk, mtime=mtime, attrs=user_attrs(proxy.b2arr) + ) elif isinstance(obj, blosc2.ndarray.NDArray): array = obj cparams = get_model_from_obj(array.schunk.cparams, models.CParams) @@ -525,12 +541,14 @@ def read_metadata(obj, mtime=None): array = hdf5.HDF5Proxy(array) schunk.cratio = array.cratio # overwrite cratio (which will be 0) with HDF5Proxy value schunk.cbytes = array.cbytes - return get_model_from_obj(array, models.Metadata, schunk=schunk, mtime=mtime) + return get_model_from_obj(array, models.Metadata, schunk=schunk, mtime=mtime, attrs=user_attrs(obj)) elif isinstance(obj, blosc2.schunk.SChunk): schunk = obj cparams = get_model_from_obj(schunk.cparams, models.CParams) cparams = reformat_cparams(cparams) - return get_model_from_obj(schunk, models.SChunk, cparams=cparams, mtime=mtime) + return get_model_from_obj( + schunk, models.SChunk, cparams=cparams, mtime=mtime, attrs=user_attrs(schunk) + ) elif isinstance(obj, blosc2.LazyArray): # overwrite operands and expression with _tosave versions for metadata display if isinstance(obj, blosc2.LazyExpr): @@ -561,6 +579,7 @@ def read_metadata(obj, mtime=None): cbytes=obj.cbytes, cratio=obj.cratio, vlmeta=dict(obj.vlmeta[:]) if obj.vlmeta[:] else {}, + attrs=user_attrs(obj), mtime=mtime, ) else: diff --git a/caterva2/services/templates/includes/info_metadata.html b/caterva2/services/templates/includes/info_metadata.html index 855c45b2..403298f1 100644 --- a/caterva2/services/templates/includes/info_metadata.html +++ b/caterva2/services/templates/includes/info_metadata.html @@ -57,7 +57,7 @@ {# All the next information is a bit too much for the main window #} {% set excluded_keys = [ - 'chunkshape', 'chunksize', 'vlmeta', + 'chunkshape', 'chunksize', 'vlmeta', 'attrs', 'contiguous', 'urlpath', 'blocksize' ] %} {% for key, value in meta %} @@ -73,7 +73,7 @@ {% endif %} {% endfor %} {% else %} - {% if key != 'urlpath' and value is not none %} + {% if key not in excluded_keys and value is not none %} {{ key }} {{ value }} @@ -82,7 +82,9 @@ {% endif %} {% endfor %} -{% if meta.schunk %} +{% if meta.attrs is defined and meta.attrs is not none %} + {% set vlmeta = meta.attrs %} +{% elif meta.schunk %} {% set vlmeta = meta.schunk.vlmeta %} {% else %} {% set vlmeta = meta.vlmeta %} @@ -90,7 +92,7 @@ {% if vlmeta %} -

VLmeta (user attributes)

+

Attributes

{% for key, value in vlmeta.items() %} diff --git a/caterva2/tests/test_api.py b/caterva2/tests/test_api.py index c321f6fe..9c2a7bf3 100644 --- a/caterva2/tests/test_api.py +++ b/caterva2/tests/test_api.py @@ -147,6 +147,8 @@ def test_remote_proxy_is_discovered_but_resolution_is_disabled( ): tag = f"{cache_policy.value}-{'unlimited' if max_cache_bytes is None else 'bounded'}" source = blosc2.arange(20, dtype=np.int32, chunks=(10,), blocks=(5,)) + source.vlmeta["experiment"] = {"id": 42, "tags": ["optical", "v2"]} + source.vlmeta["b2o"] = "user attribute" name = f"caterva2-disabled-reference-{tag}.b2nd" fsspec.filesystem("memory").pipe_file(name, source.to_cframe()) carrier_path = tmp_path / f"carrier-{tag}.b2nd" if cache_policy == blosc2.CachePolicy.DISK else None @@ -175,6 +177,21 @@ def test_remote_proxy_is_discovered_but_resolution_is_disabled( assert info["accept_ranges"] == "none" assert info["schunk"]["vlmeta"]["b2o"]["cache_policy"] == cache_policy.value assert info["schunk"]["vlmeta"]["b2o"]["max_cache_bytes"] == max_cache_bytes + assert set(info["schunk"]["vlmeta"]) == {"b2o"} + assert info["attrs"] == dict(source.vlmeta) + dataset = client.get("@public")[path.name] + assert dataset.attrs == info["attrs"] + remote = blosc2.RemoteProxy(blosc2.URLPath(f"@public/{path.name}", urlbase=client.urlbase)) + assert remote.attrs == info["attrs"] + assert remote.attrs is remote.vlmeta + panel = httpx.get( + f"{client.urlbase}/htmx/path-info/@public/{path.name}", + headers={"HX-Trigger": "meta", "HX-Current-URL": str(client.urlbase)}, + ) + panel.raise_for_status() + assert "optical" in panel.text + assert "_b2o_user_vlmeta" not in panel.text + assert "proxy-cache-sizes" not in panel.text response = httpx.get(f"{client.urlbase}/api/fetch/@public/{path.name}", params={"slice_": "0:2"}) assert response.status_code == 403 @@ -1208,17 +1225,19 @@ def test_upload_public_unauthorized(client, auth_client, examples_dir, tmp_path) @pytest.mark.parametrize("name", ["ds-1d.b2nd", "ds-hello.b2frame", "README.md"]) -def test_vlmeta(client, name): +def test_vlmeta(client, name, fill_public): myroot = client.get(TEST_CATERVA2_ROOT) ds = myroot[name] schunk_meta = ds.meta.get("schunk", ds.meta) assert ds.vlmeta is schunk_meta["vlmeta"] -def test_vlmeta_data(client): +def test_vlmeta_data(client, fill_public): myroot = client.get(TEST_CATERVA2_ROOT) ds = myroot["ds-sc-attr.b2nd"] assert ds.vlmeta == {"a": 1, "b": "foo", "c": 123.456} + assert ds.attrs == ds.vlmeta + assert ds.attrs is ds.meta["attrs"] ### Lazy expressions diff --git a/caterva2/tests/test_attrs.py b/caterva2/tests/test_attrs.py new file mode 100644 index 00000000..c562fadd --- /dev/null +++ b/caterva2/tests/test_attrs.py @@ -0,0 +1,45 @@ +"""Public attributes stay separate from storage and protocol metadata.""" + +from pathlib import Path +from types import SimpleNamespace + +import blosc2 +import pytest +from jinja2 import Environment, FileSystemLoader + +from caterva2 import File, models +from caterva2.services import srv_utils + + +@pytest.mark.parametrize("attrs", [None, {}, {"experiment": {"tags": ["optical"]}}]) +@pytest.mark.parametrize("present", [False, True]) +def test_attrs_client_and_panel_fallback(attrs, present): + meta = srv_utils.read_metadata(blosc2.arange(10)) + meta.schunk.vlmeta = {"legacy": "legacy value"} + info = meta.model_dump() + info.pop("attrs") + if present: + info["attrs"] = attrs + # The same conversion used for peer metadata preserves absent versus empty. + meta = srv_utils.get_model_from_obj(info, models.Metadata) + expected = attrs if present and attrs is not None else info["schunk"]["vlmeta"] + root = SimpleNamespace(urlbase="http://unused", name="@public") + file = File(root, "array.b2nd", meta=info) + assert file.attrs == expected + assert file.vlmeta == info["schunk"]["vlmeta"] + templates = Path(srv_utils.__file__).parent / "templates" + env = Environment(loader=FileSystemLoader(templates), autoescape=True) + rendered = env.get_template("includes/info_metadata.html").render(meta=meta) + assert ("legacy value" in rendered) == ("legacy" in expected) + assert ("optical" in rendered) == ("experiment" in expected) + + +@pytest.mark.parametrize("frame", [False, True]) +def test_attrs_preserve_user_keys_and_hide_fill_metadata(frame): + obj = blosc2.SChunk(data=b"0123456789") if frame else blosc2.arange(10) + obj.vlmeta["_user_key"] = {"tags": ["optical", "v2"]} + for key in ("fill_nonce", "fill_state", "published_url"): + obj.vlmeta[key] = "protocol value" + meta = srv_utils.read_metadata(obj) + assert meta.attrs == {"_user_key": {"tags": ["optical", "v2"]}} + assert getattr(meta, "schunk", meta).vlmeta["fill_nonce"] == "protocol value" diff --git a/caterva2/tests/test_cli.py b/caterva2/tests/test_cli.py index 7eaa672d..8df9eafc 100644 --- a/caterva2/tests/test_cli.py +++ b/caterva2/tests/test_cli.py @@ -12,6 +12,11 @@ import os import subprocess import sys +from types import SimpleNamespace + +import pytest + +from caterva2.clients.cli import cmd_info from .services import TEST_CATERVA2_ROOT @@ -43,3 +48,26 @@ def test_url(services, sub_user): urlbase = services.get_urlbase() out = cli(["url", f"{TEST_CATERVA2_ROOT}/ds-1d.b2nd"], sub_user=sub_user) assert out == f"{urlbase}/api/download/{TEST_CATERVA2_ROOT}/ds-1d.b2nd" + + +@pytest.mark.parametrize("kind", ["array", "frame", "ctable"]) +@pytest.mark.parametrize("public", ["absent", None, {}, {"experiment": {"tags": ["óptica", "v2"]}}]) +def test_info_attrs(kind, public, capsys): + storage = {"cparams": {}, "vlmeta": {"legacy": 42}} + data = {"schunk": storage} if kind == "array" else storage.copy() + if kind == "ctable": + data["kind"] = kind + if public != "absent": + data["attrs"] = public + client = SimpleNamespace(get_info=lambda path: data) + args = SimpleNamespace(dataset="@public/test", json=False) + cmd_info(client, args, "http://unused") + out = capsys.readouterr().out + expected = storage["vlmeta"] if public is None or public == "absent" else public + assert json.loads(out.split("attrs: ", 1)[1]) == expected + if "experiment" in expected: + assert "óptica" in out + + args.json = True + cmd_info(client, args, "http://unused") + assert json.loads(capsys.readouterr().out.splitlines()[-1]) == data diff --git a/caterva2/tests/test_ctable.py b/caterva2/tests/test_ctable.py index f839699d..708036ca 100644 --- a/caterva2/tests/test_ctable.py +++ b/caterva2/tests/test_ctable.py @@ -132,6 +132,7 @@ class Row: meta = srv_utils.read_metadata(table_path) assert meta.vlmeta == {"author": "Alice"} + assert meta.attrs == meta.vlmeta def test_read_metadata_nonexistent(): diff --git a/caterva2/tests/test_hdf5_tree.py b/caterva2/tests/test_hdf5_tree.py index ec1c934c..cf05f76f 100644 --- a/caterva2/tests/test_hdf5_tree.py +++ b/caterva2/tests/test_hdf5_tree.py @@ -9,6 +9,7 @@ import numpy as np import pytest +from caterva2 import hdf5 from caterva2.services import srv_utils from .services import TEST_CATERVA2_ROOT, TEST_STATE_DIR @@ -16,7 +17,9 @@ def _make_h5(path): with h5py.File(str(path), "w") as f: - f.create_dataset("/g/a", data=np.arange(6, dtype="i4").reshape(2, 3)) + ds = f.create_dataset("/g/a", data=np.arange(6, dtype="i4").reshape(2, 3)) + ds.attrs["author"] = "researcher" + ds.attrs["_user_key"] = 42 f.create_dataset("/g/b", data=np.arange(4, dtype="i4")) f.create_dataset("/h/c", data=np.arange(10, dtype="i4")) # Structured leaf for sort tests @@ -63,6 +66,10 @@ def test_info_leaf(fill_h5_public, client): info = client.get_info(f"{root.name}/{fname}/g/a") assert tuple(info["shape"]) == (2, 3) assert info["dtype"] == "int32" + assert info["attrs"] == {"author": "researcher", "_user_key": 42} + assert info["schunk"]["vlmeta"]["author"] == "researcher" + remote = blosc2.RemoteProxy(blosc2.URLPath(f"{root.name}/{fname}/g/a", urlbase=client.urlbase)) + assert remote.attrs == info["attrs"] # A leaf has no file of its own; it inherits the container's mtime. assert info["mtime"] is not None @@ -350,3 +357,15 @@ def test_chunk_of_an_hdf5_leaf_is_refused(fill_h5_public, client): response = httpx.get(f"{client.urlbase}/api/chunk/{TEST_CATERVA2_ROOT}/{path}?nchunk=0") assert response.status_code == 400 assert "slice_" in response.json()["detail"] + + +def test_legacy_hdf5_proxy_attrs(tmp_path): + path = tmp_path / "legacy.h5" + with h5py.File(path, "w") as f: + ds = f.create_dataset("g/a", data=np.arange(6, dtype="i4")) + ds.attrs["_user_key"] = "user value" + proxy = hdf5.HDF5Proxy(None, f, "g/a") + meta = srv_utils.read_metadata(proxy.b2arr) + assert meta.attrs == {"_user_key": "user value"} + assert meta.schunk.vlmeta["_ftype"] == "hdf5" + assert meta.schunk.vlmeta["_dsetname"] == "g/a" diff --git a/caterva2/tests/test_peers.py b/caterva2/tests/test_peers.py index 26896dfb..cd06281a 100644 --- a/caterva2/tests/test_peers.py +++ b/caterva2/tests/test_peers.py @@ -152,7 +152,8 @@ def two_servers(tmp_path_factory): pub = bdir / "public" pub.mkdir() data = np.random.default_rng(0).random((4, 100_000)) - blosc2.asarray(data, chunks=(1, 100_000), urlpath=str(pub / "mc.b2nd")) + arr = blosc2.asarray(data, chunks=(1, 100_000), urlpath=str(pub / "mc.b2nd")) + arr.vlmeta["experiment"] = {"id": 42, "tags": ["optical", "v2"]} # a nested dataset, to exercise path-relative listing (pub / "dir1").mkdir() blosc2.asarray(np.arange(10), urlpath=str(pub / "dir1" / "small.b2nd")) @@ -182,6 +183,13 @@ def test_list_and_info(two_servers): assert "mc.b2nd" in listing info = httpx.get(f"{urlbase}/api/info/@labb/mc.b2nd", timeout=5).json() assert tuple(info["shape"]) == (4, 100_000) + assert info["attrs"] == {"experiment": {"id": 42, "tags": ["optical", "v2"]}} + panel = httpx.get( + f"{urlbase}/htmx/path-info/@labb/mc.b2nd", + headers={"HX-Trigger": "meta", "HX-Current-URL": urlbase}, + ) + panel.raise_for_status() + assert "optical" in panel.text def test_list_is_path_relative(two_servers): diff --git a/caterva2/tests/test_treestore.py b/caterva2/tests/test_treestore.py index 62ab8939..684d13d8 100644 --- a/caterva2/tests/test_treestore.py +++ b/caterva2/tests/test_treestore.py @@ -14,7 +14,9 @@ def _make_tree(path): t = blosc2.TreeStore(str(path), mode="w") - t["/g/a"] = np.arange(6, dtype="i4").reshape(2, 3) + arr = blosc2.asarray(np.arange(6, dtype="i4").reshape(2, 3)) + arr.vlmeta["experiment"] = {"id": 42, "tags": ["optical", "v2"]} + t["/g/a"] = arr t["/g/b"] = np.arange(4, dtype="i4") t["/h/c"] = np.arange(10, dtype="i4") # Structured leaf for filter/sort tests @@ -77,6 +79,9 @@ def test_info_leaf(fill_tree_public, client): info = client.get_info(f"{root.name}/{fname}/g/a") assert tuple(info["shape"]) == (2, 3) assert info["dtype"] == "int32" + assert info["attrs"] == {"experiment": {"id": 42, "tags": ["optical", "v2"]}} + remote = blosc2.RemoteProxy(blosc2.URLPath(f"{root.name}/{fname}/g/a", urlbase=client.urlbase)) + assert remote.attrs == info["attrs"] # A leaf has no file of its own; it inherits the container's mtime. assert info["mtime"] is not None diff --git a/doc/reference/rest_api.rst b/doc/reference/rest_api.rst index a4b46b15..2cf763cb 100644 --- a/doc/reference/rest_api.rst +++ b/doc/reference/rest_api.rst @@ -9,3 +9,22 @@ A REST API is provided by the Caterva2 server. It is a simple HTTP API that allo Visit the most updated version of the REST API at: https://cat2.cloud/demo/docs. It is important to note that the REST API is not intended to be used as a replacement for the :doc:`Caterva2 Python client API `. The Python client API provides a more convenient and efficient way to interact with the Caterva2 server, and it is recommended for most use cases. However, the REST API can be useful in certain situations, such as when you need to access the Caterva2 server from a programming language that does not have a Caterva2 client library, or when you need to integrate Caterva2 with other systems that use HTTP. + +User attributes +--------------- + +``GET /api/info/{path}`` includes an ``attrs`` mapping for Blosc2 arrays and +frames, HDF5 dataset leaves, B2Z array leaves, tables, and saved remote proxies. +It contains user attributes without adapter, cache, or fill bookkeeping. +The existing ``schunk.vlmeta`` (or top-level ``vlmeta`` for frames and tables) +remains available for clients that use its protocol information. + +Saved remote proxy attributes are read from the carrier's stored snapshot; +serving them does not resolve the remote source or refresh its metadata. + +The Python client's ``File.attrs`` and ``Dataset.attrs`` use this mapping from +cached metadata. Python-Blosc2 exposes it through ``C2Array.attrs`` and +``RemoteProxy.attrs`` (also available as ``RemoteProxy.vlmeta``). These properties +provide read access, not server-side attribute writes. Clients fall back to +legacy variable metadata when ``attrs`` is absent or null; an empty mapping +means the dataset has no public attributes. diff --git a/doc/utilities/cat2-client.md b/doc/utilities/cat2-client.md index 3a17892b..8f140cb1 100644 --- a/doc/utilities/cat2-client.md +++ b/doc/utilities/cat2-client.md @@ -38,6 +38,11 @@ cat2-client roots --help A common option for many commands is `--json`, which forces the output to be in JSON format, making it easier to parse with other programs. +The `info` command prints an `attrs` section for arrays, frames, and tables, +using indented JSON so nested attributes remain readable. It prefers the +server's public `attrs` mapping and falls back to `vlmeta` for older servers. +An empty public mapping is shown as `attrs: {}`. + ## Configuration `cat2-client` can be configured using a TOML file, which is looked for as `caterva2.toml` in the current directory by default. The path can be overridden with the `--conf` generic option. Any command-line options provided will take precedence over settings from the configuration file. From 073da113668b8e9ef875665225dd21bd6fda78bd Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 11 Sep 2026 08:51:51 +0200 Subject: [PATCH 11/20] Update remote array support for the RemoteArray API --- caterva2-server.sample.toml | 2 +- caterva2/client.py | 6 +- caterva2/services/remote_proxy.py | 104 ++++++++++++----------- caterva2/services/server.py | 25 +++--- caterva2/services/sparse_cache.py | 10 +-- caterva2/services/srv_utils.py | 9 +- caterva2/services/storage_quota.py | 12 ++- caterva2/tests/test_api.py | 12 +-- caterva2/tests/test_hdf5_tree.py | 2 +- caterva2/tests/test_remote_proxy.py | 40 ++++----- caterva2/tests/test_sparse_cache.py | 14 +-- caterva2/tests/test_storage_quota.py | 4 +- caterva2/tests/test_storage_quota_api.py | 6 +- caterva2/tests/test_treestore.py | 2 +- doc/reference/rest_api.rst | 2 +- doc/utilities/cat2-server.md | 2 +- examples/benchmark_storage_quota.py | 4 +- examples/benchmarks/remote_proxy_v7.md | 2 +- 18 files changed, 137 insertions(+), 121 deletions(-) diff --git a/caterva2-server.sample.toml b/caterva2-server.sample.toml index 23ad8b8d..5b7c5964 100644 --- a/caterva2-server.sample.toml +++ b/caterva2-server.sample.toml @@ -32,7 +32,7 @@ register = true # allow users to register # publish_root = "s3://a-bucket/published" # peer_cache_quota = "1G" -# Persisted RemoteProxy objects are discoverable but cannot make outbound +# Persisted RemoteArray objects are discoverable but cannot make outbound # requests by default. The runtime cache uses private sparse generations and # credential-free HTTPS is the supported source transport. # Every destination must be listed exactly; redirects, URL queries, private or diff --git a/caterva2/client.py b/caterva2/client.py index ae5ec938..277eb354 100644 --- a/caterva2/client.py +++ b/caterva2/client.py @@ -310,7 +310,7 @@ def download(self, localpath=None, *, include_cache=True): The destination path for the downloaded file. If not specified, the file will be downloaded to the current working directory. include_cache : bool, optional - For a RemoteProxy carrier, include its valid warm cache data. + For a RemoteArray carrier, include its valid warm cache data. Pass false to download a cold proxy without changing the server copy. Returns @@ -545,7 +545,7 @@ def get_download_url(self, *, include_cache=True): """ Retrieves the download URL for the file. - ``include_cache=False`` requests a cold RemoteProxy carrier. It has no + ``include_cache=False`` requests a cold RemoteArray carrier. It has no effect on other file types. Returns @@ -1521,7 +1521,7 @@ def download(self, dataset, localpath=None, *, include_cache=True): Local path to save the downloaded dataset. Defaults to the current working directory if not specified. include_cache : bool, optional - For a RemoteProxy carrier, include its valid warm cache data. + For a RemoteArray carrier, include its valid warm cache data. Pass false to download a cold proxy without changing the server copy. Returns diff --git a/caterva2/services/remote_proxy.py b/caterva2/services/remote_proxy.py index 32ed092c..e0dee4ec 100644 --- a/caterva2/services/remote_proxy.py +++ b/caterva2/services/remote_proxy.py @@ -38,7 +38,7 @@ log = logging.getLogger(__name__) -class RemoteProxyDenied(ValueError): +class RemoteArrayDenied(ValueError): """The server policy refuses a remote reference.""" @@ -88,10 +88,10 @@ def configure(conf) -> None: if not isinstance(enabled, bool): raise ValueError("remote_proxy.enabled must be true or false") if enabled and ( - not hasattr(blosc2, "RemoteProxy") - or "assume_immutable" not in signature(blosc2.RemoteProxy).parameters + not hasattr(blosc2, "RemoteArray") + or "assume_immutable" not in signature(blosc2.RemoteArray).parameters ): - raise ValueError("remote_proxy.enabled requires a compatible Python-Blosc2 RemoteProxy") + raise ValueError("remote_proxy.enabled requires a compatible Python-Blosc2 RemoteArray") if not isinstance(hosts, list | tuple) or any(not isinstance(host, str) for host in hosts): raise ValueError("remote_proxy.allowed_hosts must be a list of host names") if not isinstance(timeout, int | float) or isinstance(timeout, bool) or timeout <= 0: @@ -144,8 +144,8 @@ def raw_carrier(path, mode="r", *, locking=False): def inspect(path): - """Return ``(raw carrier, payload)`` for a RemoteProxy, otherwise ``None``.""" - if not hasattr(blosc2, "RemoteProxy"): + """Return ``(raw carrier, payload)`` for a RemoteArray, otherwise ``None``.""" + if not hasattr(blosc2, "RemoteArray"): return None with carrier_thread_lock(path): try: @@ -154,19 +154,19 @@ def inspect(path): return None schunk = getattr(carrier, "schunk", carrier) marker = schunk.meta.get("b2o") - if not isinstance(marker, dict) or marker.get("kind") != "remote_proxy": + if not isinstance(marker, dict) or marker.get("kind") != "remote_array": return None - # Only RemoteProxy carriers need a sidecar lock. Reopen after + # Only RemoteArray carriers need a sidecar lock. Reopen after # discrimination so inspecting ordinary datasets has no filesystem # side effect, then re-read the marker and payload under that lock. carrier = raw_carrier(path, locking=True) schunk = getattr(carrier, "schunk", carrier) marker = schunk.meta.get("b2o") - if not isinstance(marker, dict) or marker.get("kind") != "remote_proxy": + if not isinstance(marker, dict) or marker.get("kind") != "remote_array": return None payload = schunk.vlmeta.get("b2o") if not isinstance(payload, dict): - raise RemoteProxyDenied("RemoteProxy carrier has no valid payload") + raise RemoteArrayDenied("RemoteArray carrier has no valid payload") return carrier, payload @@ -178,7 +178,7 @@ def guard_embedded(path) -> None: factory into that decoder, refusing those operands closes an otherwise easy way around the direct-carrier policy check. """ - if not hasattr(blosc2, "RemoteProxy"): + if not hasattr(blosc2, "RemoteArray"): return try: carrier = raw_carrier(path) @@ -190,14 +190,14 @@ def guard_embedded(path) -> None: return payload = schunk.vlmeta.get("b2o") if _contains_remote_reference(payload): - raise RemoteProxyDenied( + raise RemoteArrayDenied( "remote references embedded in persisted expressions are disabled by server policy" ) def _contains_remote_reference(value) -> bool: if isinstance(value, dict): - if value.get("kind") in {"fsspec", "remote_proxy"}: + if value.get("kind") in {"fsspec", "remote_array"}: return True return any(_contains_remote_reference(item) for item in value.values()) if isinstance(value, list | tuple): @@ -206,39 +206,41 @@ def _contains_remote_reference(value) -> bool: def is_metadata(meta) -> bool: - """Whether an api/info model describes a RemoteProxy carrier.""" + """Whether an api/info model describes a RemoteArray carrier.""" vlmeta = getattr(getattr(meta, "schunk", None), "vlmeta", None) or {} payload = vlmeta.get("b2o") - return isinstance(payload, dict) and payload.get("kind") == "remote_proxy" + return isinstance(payload, dict) and payload.get("kind") == "remote_array" def _validated_source(payload: dict) -> str: if not policy.enabled: - raise RemoteProxyDenied("RemoteProxy resolution is disabled by server policy") - if set(payload) != {"kind", "version", "source", "cache_policy", "max_cache_bytes"}: - raise RemoteProxyDenied("RemoteProxy payload contains unsupported fields") - if payload.get("kind") != "remote_proxy" or payload.get("version") != 1: - raise RemoteProxyDenied("unsupported RemoteProxy payload") + raise RemoteArrayDenied("RemoteArray resolution is disabled by server policy") + if set(payload) - {"mutable"} != {"kind", "version", "source", "cache_policy", "max_cache_bytes"}: + raise RemoteArrayDenied("RemoteArray payload contains unsupported fields") + if not isinstance(payload.get("mutable", False), bool): + raise RemoteArrayDenied("RemoteArray mutable must be true or false") + if payload.get("kind") != "remote_array" or payload.get("version") != 1: + raise RemoteArrayDenied("unsupported RemoteArray payload") cache_policy = payload.get("cache_policy") max_cache_bytes = payload.get("max_cache_bytes") if cache_policy == "none": if max_cache_bytes is not None: - raise RemoteProxyDenied("RemoteProxy cache policy 'none' cannot have max_cache_bytes") + raise RemoteArrayDenied("RemoteArray cache policy 'none' cannot have max_cache_bytes") elif cache_policy == "disk": if max_cache_bytes is not None and ( isinstance(max_cache_bytes, bool) or not isinstance(max_cache_bytes, int) or max_cache_bytes <= 0 ): - raise RemoteProxyDenied( - "RemoteProxy cache policy 'disk' requires positive max_cache_bytes or None" + raise RemoteArrayDenied( + "RemoteArray cache policy 'disk' requires positive max_cache_bytes or None" ) elif cache_policy == "memory": if isinstance(max_cache_bytes, bool) or not isinstance(max_cache_bytes, int) or max_cache_bytes <= 0: - raise RemoteProxyDenied( - f"RemoteProxy cache policy {cache_policy!r} requires positive max_cache_bytes" + raise RemoteArrayDenied( + f"RemoteArray cache policy {cache_policy!r} requires positive max_cache_bytes" ) else: - raise RemoteProxyDenied( - "server RemoteProxy supports only cache policies 'none', 'memory', and 'disk'" + raise RemoteArrayDenied( + "server RemoteArray supports only cache policies 'none', 'memory', and 'disk'" ) source = payload.get("source") if not isinstance(source, dict) or set(source) != { @@ -247,32 +249,32 @@ def _validated_source(payload: dict) -> str: "urlpath", "assume_immutable", }: - raise RemoteProxyDenied("server RemoteProxy supports only a versioned fsspec URL source") + raise RemoteArrayDenied("server RemoteArray supports only a versioned fsspec URL source") if source.get("kind") != "fsspec" or source.get("version") != 1: - raise RemoteProxyDenied("server RemoteProxy supports only fsspec source version 1") + raise RemoteArrayDenied("server RemoteArray supports only fsspec source version 1") if not isinstance(source.get("assume_immutable"), bool): - raise RemoteProxyDenied("RemoteProxy source assume_immutable must be true or false") + raise RemoteArrayDenied("RemoteArray source assume_immutable must be true or false") url = source.get("urlpath") if not isinstance(url, str): - raise RemoteProxyDenied("RemoteProxy source URL must be a string") + raise RemoteArrayDenied("RemoteArray source URL must be a string") parsed = urlsplit(url) if parsed.scheme.lower() != "https": - raise RemoteProxyDenied("server RemoteProxy currently permits only HTTPS sources") + raise RemoteArrayDenied("server RemoteArray currently permits only HTTPS sources") if parsed.username is not None or parsed.password is not None: - raise RemoteProxyDenied("RemoteProxy source URLs cannot contain user information") + raise RemoteArrayDenied("RemoteArray source URLs cannot contain user information") if parsed.query or parsed.fragment: - raise RemoteProxyDenied("RemoteProxy source URLs cannot contain a query or fragment") + raise RemoteArrayDenied("RemoteArray source URLs cannot contain a query or fragment") if parsed.hostname is None: - raise RemoteProxyDenied("RemoteProxy source URL has no host") + raise RemoteArrayDenied("RemoteArray source URL has no host") try: host = parsed.hostname.encode("idna").decode("ascii").lower() port = parsed.port or 443 except (UnicodeError, ValueError) as exc: - raise RemoteProxyDenied("RemoteProxy source URL has an invalid host or port") from exc + raise RemoteArrayDenied("RemoteArray source URL has an invalid host or port") from exc authority = host if port == 443 else f"{host}:{port}" if authority not in policy.allowed_hosts: - raise RemoteProxyDenied(f"RemoteProxy destination {authority!r} is not allowed") + raise RemoteArrayDenied(f"RemoteArray destination {authority!r} is not allowed") return url @@ -280,13 +282,13 @@ def _public_addresses(host: str, port: int) -> tuple[str, ...]: try: answers = socket.getaddrinfo(host, port, type=socket.SOCK_STREAM) except OSError as exc: - raise RemoteProxyDenied(f"RemoteProxy destination {host!r} cannot be resolved") from exc + raise RemoteArrayDenied(f"RemoteArray destination {host!r} cannot be resolved") from exc addresses = tuple(dict.fromkeys(answer[4][0] for answer in answers)) if not addresses: - raise RemoteProxyDenied(f"RemoteProxy destination {host!r} has no addresses") + raise RemoteArrayDenied(f"RemoteArray destination {host!r} has no addresses") denied = [address for address in addresses if not ipaddress.ip_address(address).is_global] if denied: - raise RemoteProxyDenied(f"RemoteProxy destination {host!r} resolves to a non-public address") + raise RemoteArrayDenied(f"RemoteArray destination {host!r} resolves to a non-public address") return addresses @@ -297,7 +299,7 @@ def __init__(self, host: str, addresses: tuple[str, ...]): async def resolve(self, host, port=0, family=socket.AF_UNSPEC): if host.encode("idna").decode("ascii").lower() != self.host: - raise OSError("redirected hosts are not allowed for RemoteProxy sources") + raise OSError("redirected hosts are not allowed for RemoteArray sources") records = [] for address in self.addresses: address_family = socket.AF_INET6 if ":" in address else socket.AF_INET @@ -348,24 +350,24 @@ def resolve(carrier, payload): expected = (tuple(carrier.shape), carrier.dtype, tuple(carrier.chunks), tuple(carrier.blocks)) actual = (tuple(source.shape), source.dtype, tuple(source.chunks), tuple(source.blocks)) if actual != expected: - raise RemoteProxyDenied( - f"RemoteProxy source geometry does not match its carrier: carrier={expected}, source={actual}" + raise RemoteArrayDenied( + f"RemoteArray source geometry does not match its carrier: carrier={expected}, source={actual}" ) if len(source.shape) > policy.max_rank: - raise RemoteProxyDenied(f"RemoteProxy rank exceeds the configured limit of {policy.max_rank}") + raise RemoteArrayDenied(f"RemoteArray rank exceeds the configured limit of {policy.max_rank}") nbytes = math.prod(source.shape) * source.dtype.itemsize if nbytes > policy.max_nbytes: - raise RemoteProxyDenied( - f"RemoteProxy logical size exceeds the configured limit of {policy.max_nbytes}" + raise RemoteArrayDenied( + f"RemoteArray logical size exceeds the configured limit of {policy.max_nbytes}" ) chunks = math.prod( math.ceil(size / chunk) for size, chunk in zip(source.shape, source.chunks, strict=True) ) if chunks > policy.max_chunks: - raise RemoteProxyDenied( - f"RemoteProxy chunk count exceeds the configured limit of {policy.max_chunks}" + raise RemoteArrayDenied( + f"RemoteArray chunk count exceeds the configured limit of {policy.max_chunks}" ) - return ServerRemoteProxy(source, expected, carrier, payload) + return ServerRemoteArray(source, expected, carrier, payload) def _effective_cache_policy(requested: str) -> str: @@ -377,7 +379,7 @@ def _effective_cache_policy(requested: str) -> str: raise ValueError(f"unknown cache policy: {requested!r}") -class ServerRemoteProxy: +class ServerRemoteArray: """Authorized remote source backed by its own carrier cache. Attributes @@ -547,7 +549,7 @@ def get_chunk(self, nchunk, *, cache_limit=None): def cold_cframe(carrier, payload) -> bytes: """Return a cache-free carrier without resolving or mutating its source.""" cold = make_b2object_carrier( - "remote_proxy", + "remote_array", carrier.shape, carrier.dtype, chunks=carrier.chunks, diff --git a/caterva2/services/server.py b/caterva2/services/server.py index a0599885..464b3dae 100644 --- a/caterva2/services/server.py +++ b/caterva2/services/server.py @@ -278,7 +278,7 @@ def move_dataset(source, destination): raise storage_quota.StorageBusy("move source was removed") if remote_proxy.policy.cache_backend == "sparse" and source.suffix in {".b2nd", ".b2frame"}: carrier = blosc2.ndarray_from_cframe(data) - if carrier.schunk.vlmeta.get("b2o", {}).get("kind") == "remote_proxy": + if carrier.schunk.vlmeta.get("b2o", {}).get("kind") == "remote_array": data = remote_proxy.cold_cframe(carrier, carrier.schunk.vlmeta["b2o"]) write_dataset(destination, data) quota.publish(source, None, expected=generation, prune=False) @@ -294,7 +294,7 @@ def quota_proxy_operation(proxy, item=(), *, nchunk=None): return proxy.quota_read(quota, item, nchunk=nchunk) -def remote_proxy_cache_limit(proxy: remote_proxy.ServerRemoteProxy) -> int | None: +def remote_proxy_cache_limit(proxy: remote_proxy.ServerRemoteArray) -> int | None: """Return the allowance for the legacy, in-place cache path. Payload limits cannot reserve physical metadata growth, and per-dataset locks @@ -371,12 +371,12 @@ def open_b2(abspath, path): carrier, payload = reference try: return remote_proxy.resolve(carrier, payload) - except remote_proxy.RemoteProxyDenied as exc: + except remote_proxy.RemoteArrayDenied as exc: raise fastapi.HTTPException(status_code=403, detail=str(exc)) from exc try: remote_proxy.guard_embedded(abspath) - except remote_proxy.RemoteProxyDenied as exc: + except remote_proxy.RemoteArrayDenied as exc: raise fastapi.HTTPException(status_code=403, detail=str(exc)) from exc container = blosc2.open(abspath) # CTable has its own storage and no table-level cparams/dparams; return early. @@ -785,7 +785,7 @@ async def get_info( response.headers["ETag"] = etag try: meta = srv_utils.read_metadata(abspath) - except remote_proxy.RemoteProxyDenied as exc: + except remote_proxy.RemoteArrayDenied as exc: raise fastapi.HTTPException(status_code=403, detail=str(exc)) from exc # A dataset with a file of its own is served by `FileResponse`, which honours # a range. Only said where it is certain: a directory or a lazy expression @@ -1245,7 +1245,7 @@ async def fetch_data( if isinstance( container, - (blosc2.NDArray, blosc2.LazyArray, hdf5.HDF5Proxy, blosc2.NDField, remote_proxy.ServerRemoteProxy), + (blosc2.NDArray, blosc2.LazyArray, hdf5.HDF5Proxy, blosc2.NDField, remote_proxy.ServerRemoteArray), ): array = container schunk = getattr(array, "schunk", None) # not really needed @@ -1285,7 +1285,7 @@ async def fetch_data( | hdf5.HDF5Proxy | blosc2.NDField | blosc2.CTable - | remote_proxy.ServerRemoteProxy, + | remote_proxy.ServerRemoteArray, ) ) and (not filter) @@ -1316,7 +1316,7 @@ async def fetch_data( srv_utils.refuse_range(range_header, path) if indices is not None: - if not isinstance(array, blosc2.NDArray | remote_proxy.ServerRemoteProxy): + if not isinstance(array, blosc2.NDArray | remote_proxy.ServerRemoteArray): srv_utils.raise_bad_request(f"{path} is not an array that can be indexed by coordinates") try: # `NDArray` reads scattered coordinates through its own sparse gather, @@ -1324,7 +1324,7 @@ async def fetch_data( # Off the event loop: bounded by `MAX_FETCH_COORDS` but not small, and # a gather that ran here would stall every other request for its # duration -- it reads, materializes and serializes, all blocking - if isinstance(array, remote_proxy.ServerRemoteProxy): + if isinstance(array, remote_proxy.ServerRemoteArray): data = await read_remote_proxy(array, indices, abspath) else: data = await concurrency.run_in_threadpool( @@ -1346,7 +1346,7 @@ async def fetch_data( data = array[() if slice_ is None else slice_] data = blosc2.asarray(data) data = data.to_cframe() - elif isinstance(array, remote_proxy.ServerRemoteProxy): + elif isinstance(array, remote_proxy.ServerRemoteArray): data = await read_remote_proxy(array, () if slice_ is None else slice_, abspath) elif isinstance(array, blosc2.NDArray): # Using NDArray.slice() allows a fast path when it is aligned with the chunks @@ -1493,7 +1493,7 @@ async def download_data( headers.update(srv_utils.NO_RANGES) return responses.StreamingResponse(body, media_type=media_type, headers=headers) - if remote_proxy.policy.cache_backend == "sparse" and include_cache: + if remote_proxy.policy.enabled and remote_proxy.policy.cache_backend == "sparse" and include_cache: abspath = get_abspath(path, user) reference = remote_proxy.inspect(abspath) if abspath.suffix in {".b2nd", ".b2frame"} else None if reference is not None and reference[1]["cache_policy"] == "disk": @@ -1607,7 +1607,7 @@ async def get_chunk( if isinstance(container, blosc2.LazyArray): # In case we do, this would have to be changed. chunk = container.get_chunk(nchunk) - elif isinstance(container, remote_proxy.ServerRemoteProxy): + elif isinstance(container, remote_proxy.ServerRemoteArray): if settings.quota or remote_proxy.policy.cache_backend == "sparse": chunk = await concurrency.run_in_threadpool( lambda: quota_proxy_operation(container, nchunk=nchunk) @@ -4240,6 +4240,7 @@ async def get_file_content(path, user, decompress=True, include_cache=True): carrier, payload = remote_proxy.inspect(abspath) if ( remote_proxy.policy.cache_backend == "sparse" + and remote_proxy.policy.enabled and include_cache and payload["cache_policy"] == "disk" ): diff --git a/caterva2/services/sparse_cache.py b/caterva2/services/sparse_cache.py index 065a7b72..cbfd6170 100644 --- a/caterva2/services/sparse_cache.py +++ b/caterva2/services/sparse_cache.py @@ -1,4 +1,4 @@ -"""Experimental private RemoteProxy generations with shared soft admission. +"""Experimental private RemoteArray generations with shared soft admission. All request mutations hold the existing path lock and a generation lock. Startup recovery discards interrupted disposable generations without resolving sources. @@ -85,9 +85,9 @@ def __init__(self, quota, *, initialize=True): required = ("with_sparse_cache", "read_cached", "trim_sparse_cache") if ( - any(not hasattr(blosc2.RemoteProxy, name) for name in required) + any(not hasattr(blosc2.RemoteArray, name) for name in required) or "source_descriptor" - not in inspect.signature(blosc2.RemoteProxy.with_sparse_cache).parameters + not in inspect.signature(blosc2.RemoteArray.with_sparse_cache).parameters ): raise RuntimeError("sparse backend requires the Python-Blosc2 v7 cache APIs") self.root = quota.root / ".remote-cache" @@ -202,7 +202,7 @@ def _admit(self, gid, estimate): return True def _attach(self, proxy, path, carrier=None): - return blosc2.RemoteProxy.with_sparse_cache( + return blosc2.RemoteArray.with_sparse_cache( proxy.src, path, source_descriptor=proxy.requested_payload["source"], @@ -475,7 +475,7 @@ def prune(self, *, force=False): break path = self.path(private) self._intent(gid, "prune") - evicted, payload = blosc2.RemoteProxy.trim_sparse_cache( + evicted, payload = blosc2.RemoteArray.trim_sparse_cache( path, max(0, payload - needed), max_chunks=remaining ) remaining -= len(evicted) diff --git a/caterva2/services/srv_utils.py b/caterva2/services/srv_utils.py index ca9bd59a..d5f8027c 100644 --- a/caterva2/services/srv_utils.py +++ b/caterva2/services/srv_utils.py @@ -455,7 +455,7 @@ def user_attrs(obj): """Read user attributes without resolving a saved remote source.""" schunk = getattr(obj, "schunk", obj) marker = getattr(schunk, "meta", {}).get("b2o", {}) - if isinstance(marker, dict) and marker.get("kind") == "remote_proxy": + if isinstance(marker, dict) and marker.get("kind") == "remote_array": return read_b2object_user_vlmeta(obj) vlmeta = schunk.vlmeta internal = {"fill_nonce", "fill_state", "published_url"} @@ -537,6 +537,13 @@ def read_metadata(obj, mtime=None): cparams = get_model_from_obj(array.schunk.cparams, models.CParams) cparams = reformat_cparams(cparams) schunk = get_model_from_obj(array.schunk, models.SChunk, cparams=cparams) + if array.schunk.meta.get("b2o", {}).get("kind") == "remote_array": + from blosc2.proxy import _RESERVED_VLMETA + + schunk.attrs = user_attrs(array) + schunk.vlmeta = { + key: value for key, value in schunk.vlmeta.items() if key not in _RESERVED_VLMETA + } if "_ftype" in schunk.vlmeta and schunk.vlmeta["_ftype"] == "hdf5": array = hdf5.HDF5Proxy(array) schunk.cratio = array.cratio # overwrite cratio (which will be 0) with HDF5Proxy value diff --git a/caterva2/services/storage_quota.py b/caterva2/services/storage_quota.py index 70736089..bf91b30e 100644 --- a/caterva2/services/storage_quota.py +++ b/caterva2/services/storage_quota.py @@ -433,11 +433,17 @@ def prune(self, *, exclude, max_victims=4): carrier = blosc2.ndarray_from_cframe(frame, copy=True) payload = carrier.schunk.vlmeta.get("b2o") marker = carrier.schunk.meta.get("b2o") - if marker != {"kind": "remote_proxy", "version": 1}: + if marker != {"kind": "remote_array", "version": 1}: continue - if not isinstance(payload, dict) or payload.get("kind") != "remote_proxy": + if not isinstance(payload, dict) or payload.get("kind") != "remote_array": continue - if set(payload) != {"kind", "version", "source", "cache_policy", "max_cache_bytes"}: + if set(payload) - {"mutable"} != { + "kind", + "version", + "source", + "cache_policy", + "max_cache_bytes", + }: continue if payload.get("cache_policy") != "disk": continue diff --git a/caterva2/tests/test_api.py b/caterva2/tests/test_api.py index 9c2a7bf3..870a2aaf 100644 --- a/caterva2/tests/test_api.py +++ b/caterva2/tests/test_api.py @@ -157,7 +157,7 @@ def test_remote_proxy_is_discovered_but_resolution_is_disabled( kwargs["max_cache_bytes"] = max_cache_bytes elif cache_policy is blosc2.CachePolicy.DISK and max_cache_bytes is None: kwargs["max_cache_bytes"] = None - reference = blosc2.RemoteProxy( + reference = blosc2.RemoteArray( f"memory://{name}", cache_policy=cache_policy, cache_path=carrier_path, @@ -181,7 +181,7 @@ def test_remote_proxy_is_discovered_but_resolution_is_disabled( assert info["attrs"] == dict(source.vlmeta) dataset = client.get("@public")[path.name] assert dataset.attrs == info["attrs"] - remote = blosc2.RemoteProxy(blosc2.URLPath(f"@public/{path.name}", urlbase=client.urlbase)) + remote = blosc2.RemoteArray(blosc2.URLPath(f"@public/{path.name}", urlbase=client.urlbase)) assert remote.attrs == info["attrs"] assert remote.attrs is remote.vlmeta panel = httpx.get( @@ -195,7 +195,7 @@ def test_remote_proxy_is_discovered_but_resolution_is_disabled( response = httpx.get(f"{client.urlbase}/api/fetch/@public/{path.name}", params={"slice_": "0:2"}) assert response.status_code == 403 - assert response.json()["detail"] == "RemoteProxy resolution is disabled by server policy" + assert response.json()["detail"] == "RemoteArray resolution is disabled by server policy" assert path.read_bytes() == before finally: path.unlink(missing_ok=True) @@ -211,7 +211,7 @@ def test_remote_proxy_download_can_omit_cache(client, tmp_path, cache_policy): if cache_policy == blosc2.CachePolicy.DISK else None ) - reference = blosc2.RemoteProxy( + reference = blosc2.RemoteArray( f"memory://{name}", cache_policy=cache_policy, cache_path=carrier_path, @@ -274,7 +274,7 @@ def get(self, key, default=None): mem_fs.pipe_file(url, source.to_cframe()) mem_fs.pipe_file("dummy.b2nd", source.to_cframe()) - dummy_proxy = blosc2.RemoteProxy( + dummy_proxy = blosc2.RemoteArray( "memory://dummy.b2nd", cache_policy=blosc2.CachePolicy.MEMORY, max_cache_bytes=500_000, @@ -367,7 +367,7 @@ def traced_cat_file(path, *args, **kwargs): def test_remote_proxy_hidden_in_expression_is_denied_before_open(client): source = blosc2.arange(20, dtype=np.int32, chunks=(10,), blocks=(5,)) fsspec.filesystem("memory").pipe_file("caterva2-embedded-reference.b2nd", source.to_cframe()) - reference = blosc2.RemoteProxy("memory://caterva2-embedded-reference.b2nd") + reference = blosc2.RemoteArray("memory://caterva2-embedded-reference.b2nd") expression = blosc2.lazyexpr("a + 1", operands={"a": reference}) path = pathlib.Path(TEST_STATE_DIR) / "server/public/embedded-reference.b2nd" expression.save(path) diff --git a/caterva2/tests/test_hdf5_tree.py b/caterva2/tests/test_hdf5_tree.py index cf05f76f..bae8778d 100644 --- a/caterva2/tests/test_hdf5_tree.py +++ b/caterva2/tests/test_hdf5_tree.py @@ -68,7 +68,7 @@ def test_info_leaf(fill_h5_public, client): assert info["dtype"] == "int32" assert info["attrs"] == {"author": "researcher", "_user_key": 42} assert info["schunk"]["vlmeta"]["author"] == "researcher" - remote = blosc2.RemoteProxy(blosc2.URLPath(f"{root.name}/{fname}/g/a", urlbase=client.urlbase)) + remote = blosc2.RemoteArray(blosc2.URLPath(f"{root.name}/{fname}/g/a", urlbase=client.urlbase)) assert remote.attrs == info["attrs"] # A leaf has no file of its own; it inherits the container's mtime. assert info["mtime"] is not None diff --git a/caterva2/tests/test_remote_proxy.py b/caterva2/tests/test_remote_proxy.py index 096a133e..d1989e7d 100644 --- a/caterva2/tests/test_remote_proxy.py +++ b/caterva2/tests/test_remote_proxy.py @@ -29,7 +29,7 @@ def get(self, key, default=None): def _payload(url, *, cache_policy="none", max_cache_bytes=None): return { - "kind": "remote_proxy", + "kind": "remote_array", "version": 1, "source": {"kind": "fsspec", "version": 1, "urlpath": url, "assume_immutable": True}, "cache_policy": cache_policy, @@ -55,7 +55,7 @@ def reset_policy(): ) def test_resolution_is_default_deny(cache_policy, max_cache_bytes): remote_proxy.policy = remote_proxy.Policy() - with pytest.raises(remote_proxy.RemoteProxyDenied, match="disabled"): + with pytest.raises(remote_proxy.RemoteArrayDenied, match="disabled"): remote_proxy._validated_source( _payload( "https://data.example/array.b2nd", cache_policy=cache_policy, max_cache_bytes=max_cache_bytes @@ -83,7 +83,7 @@ def test_resolution_is_default_deny(cache_policy, max_cache_bytes): ) def test_enabled_policy_still_rejects_unsafe_destinations(url, cache_policy, max_cache_bytes): remote_proxy.policy = remote_proxy.Policy(enabled=True, allowed_hosts=("data.example",)) - with pytest.raises(remote_proxy.RemoteProxyDenied): + with pytest.raises(remote_proxy.RemoteArrayDenied): remote_proxy._validated_source( _payload(url, cache_policy=cache_policy, max_cache_bytes=max_cache_bytes) ) @@ -189,7 +189,7 @@ def test_configured_https_destination_is_accepted(cache_policy, max_cache_bytes) ) def test_cache_specification_is_strict(payload, match): remote_proxy.policy = remote_proxy.Policy(enabled=True, allowed_hosts=("data.example",)) - with pytest.raises(remote_proxy.RemoteProxyDenied, match=match): + with pytest.raises(remote_proxy.RemoteArrayDenied, match=match): remote_proxy._validated_source(payload) @@ -199,7 +199,7 @@ def test_source_assume_immutable_must_be_boolean(value): payload = _payload("https://data.example/a.b2nd") payload["source"]["assume_immutable"] = value - with pytest.raises(remote_proxy.RemoteProxyDenied, match="must be true or false"): + with pytest.raises(remote_proxy.RemoteArrayDenied, match="must be true or false"): remote_proxy._validated_source(payload) @@ -217,7 +217,7 @@ def test_private_resolution_is_rejected(monkeypatch): "getaddrinfo", lambda *args, **kwargs: [(socket.AF_INET, socket.SOCK_STREAM, 6, "", ("127.0.0.1", 443))], ) - with pytest.raises(remote_proxy.RemoteProxyDenied, match="non-public"): + with pytest.raises(remote_proxy.RemoteArrayDenied, match="non-public"): remote_proxy._public_addresses("data.example", 443) @@ -286,7 +286,7 @@ def fake_source(url, max_concurrency, *, _filesystem): max_cache_bytes=max_cache_bytes, ) resolved = remote_proxy.resolve(Carrier(), payload) - assert isinstance(resolved, remote_proxy.ServerRemoteProxy) + assert isinstance(resolved, remote_proxy.ServerRemoteArray) assert resolved.requested_cache_policy == cache_policy assert resolved.requested_max_cache_bytes == max_cache_bytes assert resolved.cache_policy == expected_eff_policy @@ -304,7 +304,7 @@ def test_cold_cframe_preserves_specification_but_not_cached_chunks(tmp_path): source_url = "memory://cold-cframe-source.b2nd" fsspec.filesystem("memory").pipe_file("cold-cframe-source.b2nd", source.to_cframe()) carrier_path = tmp_path / "proxy.b2nd" - proxy = blosc2.RemoteProxy( + proxy = blosc2.RemoteArray( source_url, cache_policy=blosc2.CachePolicy.DISK, cache_path=carrier_path, @@ -328,21 +328,21 @@ def _server_proxy(tmp_path, name="server-cache", cache_policy="disk", max_cache_ fsspec.filesystem("memory").pipe_file(f"{name}-source.b2nd", source.to_cframe()) carrier_path = tmp_path / f"{name}.b2nd" if cache_policy == "disk": - creator = blosc2.RemoteProxy( + creator = blosc2.RemoteArray( source_url, cache_policy=blosc2.CachePolicy.DISK, cache_path=carrier_path, max_cache_bytes=max_cache_bytes, ) elif cache_policy == "memory": - creator = blosc2.RemoteProxy( + creator = blosc2.RemoteArray( source_url, cache_policy=blosc2.CachePolicy.MEMORY, max_cache_bytes=max_cache_bytes, ) - creator.save(carrier_path) + creator.save(carrier_path, mutable=True) elif cache_policy == "none": - creator = blosc2.RemoteProxy( + creator = blosc2.RemoteArray( source_url, cache_policy=blosc2.CachePolicy.NONE, ) @@ -352,7 +352,7 @@ def _server_proxy(tmp_path, name="server-cache", cache_policy="disk", max_cache_ carrier, payload = remote_proxy.inspect(carrier_path) geometry = (creator.shape, creator.dtype, creator.chunks, creator.blocks) - return remote_proxy.ServerRemoteProxy(creator.src, geometry, carrier, payload), data, carrier_path + return remote_proxy.ServerRemoteArray(creator.src, geometry, carrier, payload), data, carrier_path def test_server_proxy_reuses_its_carrier_cache(tmp_path): @@ -363,7 +363,7 @@ def test_server_proxy_reuses_its_carrier_cache(tmp_path): carrier, payload = remote_proxy.inspect(carrier_path) fresh_source = blosc2.FsspecNDSource(payload["source"]["urlpath"]) - reopened = remote_proxy.ServerRemoteProxy( + reopened = remote_proxy.ServerRemoteArray( fresh_source, (proxy.shape, proxy.dtype, proxy.chunks, proxy.blocks), carrier, @@ -428,7 +428,7 @@ def fresh_traced_get_chunk(n): return fresh_orig_get_chunk(n) fresh_source.get_chunk = fresh_traced_get_chunk - reopened = remote_proxy.ServerRemoteProxy( + reopened = remote_proxy.ServerRemoteArray( fresh_source, (proxy.shape, proxy.dtype, proxy.chunks, proxy.blocks), carrier, @@ -477,7 +477,7 @@ def test_memory_carrier_ignores_synthetic_cached_chunks(tmp_path): carrier, payload = remote_proxy.inspect(carrier_path) assert payload["cache_policy"] == "memory" fresh_src = blosc2.FsspecNDSource("memory://synthetic-source.b2nd") - mem_proxy = remote_proxy.ServerRemoteProxy( + mem_proxy = remote_proxy.ServerRemoteArray( fresh_src, (disk_proxy.shape, disk_proxy.dtype, disk_proxy.chunks, disk_proxy.blocks), carrier, @@ -505,8 +505,8 @@ def test_memory_carrier_exports_preserve_policy_and_reopen_with_client_cache(tmp for kind, b in [("warm", warm_bytes), ("cold", cold_bytes)]: out_path = tmp_path / f"export_{kind}.b2nd" out_path.write_bytes(b) - reopened = blosc2.open(str(out_path)) - assert isinstance(reopened, blosc2.RemoteProxy) + reopened = blosc2.open(str(out_path), mode="a") + assert isinstance(reopened, blosc2.RemoteArray) assert reopened.cache_policy == blosc2.CachePolicy.MEMORY assert reopened.max_cache_bytes == 500_000 np.testing.assert_array_equal(reopened[:10], data[:10]) @@ -554,7 +554,7 @@ def test_memory_carrier_detects_geometry_replacement_and_observes_data_replaceme # Case A: Source geometry changed mismatched_source = blosc2.asarray(np.arange(60, dtype=np.int32), chunks=(10,), blocks=(5,)) mem_fs.pipe_file(url, mismatched_source.to_cframe()) - with pytest.raises(remote_proxy.RemoteProxyDenied, match="geometry does not match"): + with pytest.raises(remote_proxy.RemoteArrayDenied, match="geometry does not match"): remote_proxy.resolve(carrier, payload) # Case B: Source data changed (geometry identical) @@ -687,7 +687,7 @@ def test_concurrent_server_proxy_fills_do_not_corrupt_carrier(tmp_path): first, data, carrier_path = _server_proxy(tmp_path, "concurrent") carrier, payload = remote_proxy.inspect(carrier_path) second_source = blosc2.FsspecNDSource(payload["source"]["urlpath"]) - second = remote_proxy.ServerRemoteProxy( + second = remote_proxy.ServerRemoteArray( second_source, (first.shape, first.dtype, first.chunks, first.blocks), carrier, diff --git a/caterva2/tests/test_sparse_cache.py b/caterva2/tests/test_sparse_cache.py index 3041b9f5..54c6bf8f 100644 --- a/caterva2/tests/test_sparse_cache.py +++ b/caterva2/tests/test_sparse_cache.py @@ -21,7 +21,7 @@ def runtime(tmp_path): fs = fsspec.filesystem("memory") fs.pipe_file(url, array.to_cframe()) carrier = make_b2object_carrier( - "remote_proxy", + "remote_array", array.shape, array.dtype, chunks=array.chunks, @@ -30,7 +30,7 @@ def runtime(tmp_path): ) carrier.schunk.vlmeta["user-variable"] = {"sample": 42} payload = { - "kind": "remote_proxy", + "kind": "remote_array", "version": 1, "source": {"kind": "fsspec", "version": 1, "urlpath": url, "assume_immutable": True}, "cache_policy": "disk", @@ -44,7 +44,7 @@ def runtime(tmp_path): def resolve(): source = blosc2.FsspecNDSource(url, _filesystem=fs) c = remote_proxy.raw_carrier(path) - return remote_proxy.ServerRemoteProxy( + return remote_proxy.ServerRemoteArray( source, (array.shape, array.dtype, array.chunks, array.blocks), c, payload ) @@ -70,14 +70,14 @@ def forbidden(*args, **kwargs): def test_failed_mutation_is_charged_and_recovered_without_network(runtime, monkeypatch): q, resolve, data, _ = runtime - read = blosc2.RemoteProxy.__getitem__ + read = blosc2.RemoteArray.__getitem__ def fail_after_write(self, item): read(self, item) raise OSError("injected interrupted cache publication") with monkeypatch.context() as patch: - patch.setattr(blosc2.RemoteProxy, "__getitem__", fail_after_write) + patch.setattr(blosc2.RemoteArray, "__getitem__", fail_after_write) np.testing.assert_array_equal(q.remote.read(resolve()), data) with q.connect() as db: assert db.execute("SELECT count(*) FROM remote_operations").fetchone()[0] == 1 @@ -169,7 +169,7 @@ def _worker_read(statedir, ready, errors): carrier = remote_proxy.raw_carrier(q.root / "public/proxy.b2nd") source = blosc2.FsspecNDSource(url, _filesystem=fs) source.stamp = "immutable-multiprocess-test-source" - proxy = remote_proxy.ServerRemoteProxy( + proxy = remote_proxy.ServerRemoteArray( source, (array.shape, array.dtype, array.chunks, array.blocks), carrier, @@ -234,7 +234,7 @@ def _worker_die(statedir): fs.pipe_file(url, array.to_cframe()) carrier = remote_proxy.raw_carrier(q.root / "public/proxy.b2nd") source = blosc2.FsspecNDSource(url, _filesystem=fs) - proxy = remote_proxy.ServerRemoteProxy( + proxy = remote_proxy.ServerRemoteArray( source, (array.shape, array.dtype, array.chunks, array.blocks), carrier, carrier.schunk.vlmeta["b2o"] ) original = blosc2.Proxy._store_chunk diff --git a/caterva2/tests/test_storage_quota.py b/caterva2/tests/test_storage_quota.py index eea9ce5c..2985ff91 100644 --- a/caterva2/tests/test_storage_quota.py +++ b/caterva2/tests/test_storage_quota.py @@ -211,12 +211,12 @@ def remote_fixture(tmp_path, name="proxy", *, limit=None, block=10000): fsspec.filesystem("memory").pipe_file(f"quota-{name}.b2nd", array.to_cframe()) path = tmp_path / "public" / f"{name}.b2nd" path.parent.mkdir(exist_ok=True) - creator = blosc2.RemoteProxy( + creator = blosc2.RemoteArray( url, cache_policy=blosc2.CachePolicy.DISK, cache_path=path, max_cache_bytes=limit ) creator.schunk.vlmeta["user-note"] = "preserve me" carrier, payload = remote_proxy.inspect(path) - proxy = remote_proxy.ServerRemoteProxy( + proxy = remote_proxy.ServerRemoteArray( creator.src, (array.shape, array.dtype, array.chunks, array.blocks), carrier, payload ) return proxy, path, data diff --git a/caterva2/tests/test_storage_quota_api.py b/caterva2/tests/test_storage_quota_api.py index 7f9256fd..7a7fc83a 100644 --- a/caterva2/tests/test_storage_quota_api.py +++ b/caterva2/tests/test_storage_quota_api.py @@ -191,7 +191,7 @@ async def test_disk_fetch_and_chunk_admit_growth_via_secure_source(quota_api, mo array = blosc2.asarray(data, chunks=(10000,), blocks=(10000,)) fs = fsspec.filesystem("memory") fs.pipe_file("quota-api-source.b2nd", array.to_cframe()) - creator = blosc2.RemoteProxy("memory://quota-api-source.b2nd", cache_policy=blosc2.CachePolicy.MEMORY) + creator = blosc2.RemoteArray("memory://quota-api-source.b2nd", cache_policy=blosc2.CachePolicy.MEMORY) carrier = blosc2.ndarray_from_cframe(creator.to_cframe(cache_policy=blosc2.CachePolicy.DISK), copy=True) payload = dict(carrier.schunk.vlmeta["b2o"]) url = "https://data.example/quota.b2nd" @@ -249,12 +249,12 @@ async def test_sparse_boundary_lifecycle(quota_api, monkeypatch, quota_enabled): from blosc2.b2objects import make_b2object_carrier, write_b2object_payload carrier = make_b2object_carrier( - "remote_proxy", array.shape, array.dtype, chunks=array.chunks, blocks=array.blocks + "remote_array", array.shape, array.dtype, chunks=array.chunks, blocks=array.blocks ) write_b2object_payload( carrier, { - "kind": "remote_proxy", + "kind": "remote_array", "version": 1, "source": { "kind": "fsspec", diff --git a/caterva2/tests/test_treestore.py b/caterva2/tests/test_treestore.py index 684d13d8..c013cb47 100644 --- a/caterva2/tests/test_treestore.py +++ b/caterva2/tests/test_treestore.py @@ -80,7 +80,7 @@ def test_info_leaf(fill_tree_public, client): assert tuple(info["shape"]) == (2, 3) assert info["dtype"] == "int32" assert info["attrs"] == {"experiment": {"id": 42, "tags": ["optical", "v2"]}} - remote = blosc2.RemoteProxy(blosc2.URLPath(f"{root.name}/{fname}/g/a", urlbase=client.urlbase)) + remote = blosc2.RemoteArray(blosc2.URLPath(f"{root.name}/{fname}/g/a", urlbase=client.urlbase)) assert remote.attrs == info["attrs"] # A leaf has no file of its own; it inherits the container's mtime. assert info["mtime"] is not None diff --git a/doc/reference/rest_api.rst b/doc/reference/rest_api.rst index 2cf763cb..331915ec 100644 --- a/doc/reference/rest_api.rst +++ b/doc/reference/rest_api.rst @@ -24,7 +24,7 @@ serving them does not resolve the remote source or refresh its metadata. The Python client's ``File.attrs`` and ``Dataset.attrs`` use this mapping from cached metadata. Python-Blosc2 exposes it through ``C2Array.attrs`` and -``RemoteProxy.attrs`` (also available as ``RemoteProxy.vlmeta``). These properties +``RemoteArray.attrs`` (also available as ``RemoteArray.vlmeta``). These properties provide read access, not server-side attribute writes. Clients fall back to legacy variable metadata when ``attrs`` is absent or null; an empty mapping means the dataset has no public attributes. diff --git a/doc/utilities/cat2-server.md b/doc/utilities/cat2-server.md index 787a52c1..f1f68893 100644 --- a/doc/utilities/cat2-server.md +++ b/doc/utilities/cat2-server.md @@ -34,7 +34,7 @@ And then simply run `cat2-server` to start it on all network interfaces on port ## Remote reference policy -A persisted `blosc2.RemoteProxy` is a B2ND carrier that asks Caterva2 to read +A persisted `blosc2.RemoteArray` is a B2ND carrier that asks Caterva2 to read another dataset. Persisted `MEMORY` carriers are accepted under the same source policy but execute without retained caching (using the same no-retention execution path as `NONE`), avoiding unmanaged memory use on the server while preserving diff --git a/examples/benchmark_storage_quota.py b/examples/benchmark_storage_quota.py index 5a71c32d..608729c3 100644 --- a/examples/benchmark_storage_quota.py +++ b/examples/benchmark_storage_quota.py @@ -22,11 +22,11 @@ def benchmark(root, data, chunk, staged): fsspec.filesystem("memory").pipe_file("quota-benchmark.b2nd", source.to_cframe()) path = root / "public" / "proxy.b2nd" path.parent.mkdir(parents=True) - creator = blosc2.RemoteProxy( + creator = blosc2.RemoteArray( url, cache_policy=blosc2.CachePolicy.DISK, cache_path=path, max_cache_bytes=None ) carrier, payload = remote_proxy.inspect(path) - proxy = remote_proxy.ServerRemoteProxy( + proxy = remote_proxy.ServerRemoteArray( creator.src, (source.shape, source.dtype, source.chunks, source.blocks), carrier, payload ) quota = storage_quota.StorageQuota(root, data.nbytes * 4) if staged else None diff --git a/examples/benchmarks/remote_proxy_v7.md b/examples/benchmarks/remote_proxy_v7.md index d2148836..57036e65 100644 --- a/examples/benchmarks/remote_proxy_v7.md +++ b/examples/benchmarks/remote_proxy_v7.md @@ -1,4 +1,4 @@ -# RemoteProxy v5 versus sparse v7 +# RemoteArray v5 versus sparse v7 Measured through the real `/api/chunk` and `/api/fetch` ASGI routes, using the rebuilt Python-Blosc2 extension after `a77ac97b` (C-Blosc2 `a54e259`). V5 is the From b570e64767d3ee0d27d93ec153f19f2d0e9533f7 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 11 Sep 2026 11:21:37 +0200 Subject: [PATCH 12/20] Add Caterva2 RemoteStore support --- caterva2-server.sample.toml | 2 + caterva2/services/remote_proxy.py | 10 +- caterva2/services/remote_store.py | 287 ++++++++++++++++++++++++++++ caterva2/services/server.py | 53 ++++- caterva2/services/sparse_cache.py | 234 ++++++++++++++++++++++- caterva2/services/srv_utils.py | 23 +++ caterva2/services/storage_quota.py | 8 +- caterva2/tests/test_remote_store.py | 216 +++++++++++++++++++++ doc/utilities/cat2-server.md | 38 ++++ examples/benchmark_remote_store.py | 181 ++++++++++++++++++ plans/remote-store.md | 242 +++++++++++++++++++++++ pyproject.toml | 2 +- 12 files changed, 1281 insertions(+), 15 deletions(-) create mode 100644 caterva2/services/remote_store.py create mode 100644 caterva2/tests/test_remote_store.py create mode 100644 examples/benchmark_remote_store.py create mode 100644 plans/remote-store.md diff --git a/caterva2-server.sample.toml b/caterva2-server.sample.toml index 5b7c5964..7a0a32d7 100644 --- a/caterva2-server.sample.toml +++ b/caterva2-server.sample.toml @@ -52,6 +52,8 @@ register = true # allow users to register # max_rank = 16 # max_chunks = 10000000 # max_concurrency = 8 +# max_metadata_bytes = 16777216 +# max_nodes = 100000 # cache_maintenance_seconds = 60 # Mount a remote Caterva2 server's locally owned @public root as @labb. This diff --git a/caterva2/services/remote_proxy.py b/caterva2/services/remote_proxy.py index e0dee4ec..43f147ef 100644 --- a/caterva2/services/remote_proxy.py +++ b/caterva2/services/remote_proxy.py @@ -53,6 +53,8 @@ class Policy: max_concurrency: int = 8 cache_maintenance_seconds: float = 60.0 cache_backend: str = "sparse" + max_metadata_bytes: int = 16 << 20 + max_nodes: int = 100_000 policy = Policy() @@ -83,6 +85,8 @@ def configure(conf) -> None: max_rank = conf.get(".remote_proxy.max_rank", 16) max_chunks = conf.get(".remote_proxy.max_chunks", 10_000_000) max_concurrency = conf.get(".remote_proxy.max_concurrency", 8) + max_metadata_bytes = conf.get(".remote_proxy.max_metadata_bytes", 16 << 20) + max_nodes = conf.get(".remote_proxy.max_nodes", 100_000) cache_maintenance_seconds = conf.get(".remote_proxy.cache_maintenance_seconds", 60.0) if not isinstance(enabled, bool): @@ -107,6 +111,8 @@ def configure(conf) -> None: "max_rank": max_rank, "max_chunks": max_chunks, "max_concurrency": max_concurrency, + "max_metadata_bytes": max_metadata_bytes, + "max_nodes": max_nodes, }.items(): if not isinstance(value, int) or isinstance(value, bool) or value <= 0: raise ValueError(f"remote_proxy.{name} must be a positive integer") @@ -119,6 +125,8 @@ def configure(conf) -> None: max_rank=max_rank, max_chunks=max_chunks, max_concurrency=max_concurrency, + max_metadata_bytes=max_metadata_bytes, + max_nodes=max_nodes, cache_maintenance_seconds=float(cache_maintenance_seconds), ) @@ -197,7 +205,7 @@ def guard_embedded(path) -> None: def _contains_remote_reference(value) -> bool: if isinstance(value, dict): - if value.get("kind") in {"fsspec", "remote_array"}: + if value.get("kind") in {"fsspec", "remote_array", "remote_store", "hdf5", "zarr", "b2z"}: return True return any(_contains_remote_reference(item) for item in value.values()) if isinstance(value, list | tuple): diff --git a/caterva2/services/remote_store.py b/caterva2/services/remote_store.py new file mode 100644 index 00000000..149a766a --- /dev/null +++ b/caterva2/services/remote_store.py @@ -0,0 +1,287 @@ +"""Policy-checked portable stores backed by private shared sparse generations.""" + +from __future__ import annotations + +import copy +import math +import zipfile +from contextlib import contextmanager +from pathlib import Path +from urllib.parse import urlsplit + +import blosc2 +import fastapi +from blosc2.msgpack_utils import msgpack_packb + +from caterva2.services import remote_proxy, storage_quota + + +def inspect(path): + """Inspect a store archive without opening any remote source.""" + path = Path(path) + if path.suffix != ".b2z" or not path.is_file(): + return None + from blosc2.remote_store import get_zip_offsets + + try: + offsets = get_zip_offsets(str(path)) + except zipfile.BadZipFile: + return None + entry = offsets.get("embed.b2e") + if entry is None: + return None + if not entry["stored"] or entry["length"] > remote_proxy.policy.max_metadata_bytes: + raise remote_proxy.RemoteArrayDenied("Store metadata exceeds the configured limit or is compressed") + embed = blosc2.blosc2_ext.open(str(path), "r", entry["offset"]) + if "b2remote_store" not in embed.meta: + return None + try: + manifest, _ = blosc2.RemoteStore._load_artifact_manifest(str(path)) + validate_manifest(manifest) + except (KeyError, TypeError, ValueError) as exc: + raise remote_proxy.RemoteArrayDenied(f"Invalid RemoteStore manifest: {exc}") from exc + return manifest + + +def validate_manifest(manifest): + blosc2.RemoteStore._validate_artifact_manifest(manifest) + if set(manifest["source"]) != {"urlpath", "dataset", "kind"}: + raise remote_proxy.RemoteArrayDenied("RemoteStore source contains unsupported fields") + if len(msgpack_packb(manifest)) > remote_proxy.policy.max_metadata_bytes: + raise remote_proxy.RemoteArrayDenied("RemoteStore metadata exceeds the configured limit") + if len(manifest["nodes"]) > remote_proxy.policy.max_nodes: + raise remote_proxy.RemoteArrayDenied("RemoteStore node count exceeds the configured limit") + # Reuse array policy validation, including the strict cache-policy schema. + policy = manifest.get("cache_policy", "none") + limit = manifest.get("max_cache_bytes") + if ( + policy not in {"none", "memory", "disk"} + or (policy != "none" and limit is not None and (type(limit) is not int or limit <= 0)) + or (policy == "none" and limit is not None) + or (policy == "memory" and limit is None) + ): + raise remote_proxy.RemoteArrayDenied("Invalid RemoteStore cache policy or allowance") + + +def source_url(manifest): + source = manifest["source"] + return remote_proxy._validated_source( + { + "kind": "remote_array", + "version": 1, + "source": { + "kind": "fsspec", + "version": 1, + "urlpath": source["urlpath"], + "assume_immutable": True, + }, + "cache_policy": "none", + "max_cache_bytes": None, + } + ) + + +def filesystem(manifest): + parsed = urlsplit(source_url(manifest)) + host = parsed.hostname.encode("idna").decode("ascii").lower() + addresses = remote_proxy._public_addresses(host, parsed.port or 443) + return remote_proxy._https_filesystem(host, addresses) + + +def cold_export(manifest, destination): + """Write a descriptor-only archive without resolving its source.""" + manifest = dict(manifest, caches=[]) + storage = blosc2.Storage(contiguous=True) + storage.meta = {"b2tree": {"version": 1}, "b2remote_store": {"version": 1}} + embed = blosc2.SChunk(chunksize=8192, data=None, storage=storage) + embed.vlmeta["b2remote_manifest"] = manifest + with zipfile.ZipFile(destination, "x", compression=zipfile.ZIP_STORED) as archive: + archive.writestr("embed.b2e", embed.to_cframe()) + + +class ServerRemoteStore: + """Container adapter; leaf handles contain descriptors, not live disk owners.""" + + def __init__(self, path, manifest): + self.path = str(path) + self.manifest = manifest + self.carrier_generation = storage_quota.signature(path) + self.max_cache_bytes = manifest.get("max_cache_bytes") + self.cache_policy = remote_proxy._effective_cache_policy(manifest.get("cache_policy", "none")) + + @contextmanager + def open(self, runtime=None): + fs = filesystem(self.manifest) + source = self.manifest["source"] + options = { + "_filesystem": fs, + "_source_validator": validate_array, + "_manifest_validator": validate_manifest, + "_max_nodes": remote_proxy.policy.max_nodes, + } + try: + if runtime is not None: + store = blosc2.RemoteStore.with_sparse_cache( + source["urlpath"], + runtime, + dataset=source.get("dataset"), + manifest=copy.deepcopy(self.manifest), + max_cache_bytes=self.max_cache_bytes, + carrier=self.path, + **options, + ) + else: + store = blosc2.RemoteStore( + source["urlpath"], + dataset=source.get("dataset"), + cache_policy=blosc2.CachePolicy.NONE, + _manifest=dict(copy.deepcopy(self.manifest), caches=[]), + **options, + ) + with store: + yield store + finally: + session = getattr(fs, "_session", None) + if session is not None: + fs.close_session(fs.loop, session) + + def operation(self, callback, *, cached=None): + from caterva2.services.server import quota_coordinator + + try: + source_url(self.manifest) + return quota_coordinator().remote.store_operation(self, callback, cached=cached) + except remote_proxy.RemoteArrayDenied as exc: + raise fastapi.HTTPException(status_code=403, detail=str(exc)) from exc + except ValueError as exc: + raise fastapi.HTTPException(status_code=400, detail=str(exc)) from exc + + def _key(self, key): + relative = key.strip("/") + blosc2.remote_store.RemoteDiscovery._validate(relative) + root = self.manifest["source"].get("dataset", "") + return "/".join(part for part in (root, relative) if part) + + def leaves(self, prefix="/"): + root = self.manifest["source"].get("dataset", "") + full = self._key(prefix).rstrip("/") + listed = self.manifest["listed"] + if self.manifest["source"]["kind"] in {"b2z", "hdf5"} or full in listed: + # Use known discovery offline only when all descendant groups are listed. + complete = self.manifest["source"]["kind"] in {"b2z", "hdf5"} or all( + key in listed + for key, (kind, _) in self.manifest["nodes"].items() + if kind == "group" and (key == full or key.startswith(full + "/") or not full) + ) + if complete: + return sorted( + "/" + (key[len(root) + 1 :] if root else key) + for key, (kind, _) in self.manifest["nodes"].items() + if kind == "ndarray" and (not full or key.startswith(full + "/")) + ) + + def discover(store): + result = [] + pending = [prefix.strip("/")] + while pending: + key = pending.pop() + node = store.get_info(key) + if node.kind == "ndarray": + result.append("/" + key) + elif node.kind == "group": + with store[key] as group: + pending.extend("/".join(p for p in (key, child) if p) for child in group) + if len(result) + len(pending) > remote_proxy.policy.max_nodes: + raise remote_proxy.RemoteArrayDenied("RemoteStore listing exceeds the node limit") + return sorted(result) + + return self.operation(discover) + + def get(self, key): + from caterva2.services.srv_utils import GROUP + + try: + known = self.manifest["nodes"].get(self._key(key)) + kind = known[0] if known else self.operation(lambda store: store.kind(key.strip("/"))) + if kind == "group": + return GROUP + if kind != "ndarray": + return None + return ServerStoreArray(self, key.strip("/")) + except KeyError: + return None + + def is_group(self, node): + from caterva2.services.srv_utils import GROUP + + return node is GROUP + + def is_leaf(self, key): + known = self.manifest["nodes"].get(self._key(key)) + if known is not None: + return known[0] == "ndarray" + node = self.get(key) + return node is not None and not self.is_group(node) + + def leaf_size(self, key): + return None + + def size(self, prefix="/"): + return Path(self.path).stat().st_size if prefix == "/" else None + + def close(self): + pass # Every operation closes its own runtime and child handles. + + +class ServerStoreArray(remote_proxy.ServerRemoteArray): + def __init__(self, store, key): + self.store, self.key, self.path = store, key, store.path + self.cache_policy, self.max_cache_bytes = store.cache_policy, store.max_cache_bytes + + def geometry(runtime): + with runtime[key] as array: + validate_array(array) + return array.shape, array.dtype, array.chunks, array.blocks, array.cparams, dict(array.attrs) + + self.shape, self.dtype, self.chunks, self.blocks, self.cparams, self.attrs = store.operation( + geometry + ) + + def read(self, item=(), *, cache_limit=None): + return self.quota_read(None, item) + + def get_chunk(self, nchunk, *, cache_limit=None): + return self.quota_read(None, nchunk=nchunk) + + def quota_read(self, quota, item=(), *, nchunk=None): + def read(runtime): + with runtime[self.key] as array: + validate_array(array) + if (array.shape, array.dtype, array.chunks, array.blocks) != ( + self.shape, + self.dtype, + self.chunks, + self.blocks, + ): + raise storage_quota.StorageBusy("RemoteStore leaf geometry changed") + return array[item] if nchunk is None else array.get_chunk(nchunk) + + return self.store.operation( + read, cached=lambda runtime: runtime.read_cached(self.key, item, nchunk=nchunk) + ) + + +def validate_array(array): + policy = remote_proxy.policy + if hasattr(array, "max_concurrency"): + array.max_concurrency = policy.max_concurrency + if ( + len(array.shape) > policy.max_rank + or math.prod(array.shape) * array.dtype.itemsize > policy.max_nbytes + ): + raise remote_proxy.RemoteArrayDenied("RemoteStore leaf exceeds array geometry limits") + if ( + math.prod(math.ceil(s / c) for s, c in zip(array.shape, array.chunks, strict=True)) + > policy.max_chunks + ): + raise remote_proxy.RemoteArrayDenied("RemoteStore leaf exceeds the chunk limit") diff --git a/caterva2/services/server.py b/caterva2/services/server.py index 464b3dae..1df20e4b 100644 --- a/caterva2/services/server.py +++ b/caterva2/services/server.py @@ -366,6 +366,11 @@ def open_b2(abspath, path): if root not in {"@personal", "@shared", "@public"}: raise ValueError(f"Unexpected root={root}") + from caterva2.services import remote_store + + manifest = remote_store.inspect(abspath) + if manifest is not None: + return remote_store.ServerRemoteStore(abspath, manifest) reference = remote_proxy.inspect(abspath) if reference is not None: carrier, payload = reference @@ -541,6 +546,11 @@ def custom_filesizeformat(value): app = FastAPI(lifespan=lifespan) +@app.exception_handler(remote_proxy.RemoteArrayDenied) +async def remote_reference_denied(request, exc): + return responses.JSONResponse(status_code=403, content={"detail": str(exc)}) + + @app.exception_handler(storage_quota.QuotaExceeded) async def quota_exceeded(request, exc): return responses.JSONResponse(status_code=400, content={"detail": str(exc)}) @@ -854,6 +864,10 @@ def member_window(abspath, inner_key, mtime): """ if abspath.suffix != ".b2z": return None + from caterva2.services import remote_store + + if remote_store.inspect(abspath) is not None: + return None try: store = blosc2.open(abspath) except Exception: @@ -1222,7 +1236,9 @@ async def fetch_data( if field: container = container[field] - if isinstance(container, blosc2.DictStore): + from caterva2.services.remote_store import ServerRemoteStore + + if isinstance(container, blosc2.DictStore | ServerRemoteStore): # A container is a file of leaves rather than an array: its stored image # is the file, which is what a client opening it as a store expects -- # and what the type ladder below used to die on, asking a TreeStore for @@ -1493,6 +1509,24 @@ async def download_data( headers.update(srv_utils.NO_RANGES) return responses.StreamingResponse(body, media_type=media_type, headers=headers) + from caterva2.services import remote_store + + abspath = get_abspath(path, user) + manifest = remote_store.inspect(abspath) + if manifest is not None and (remote_proxy.policy.enabled or not include_cache): + artifact, etag, cleanup = await concurrency.run_in_threadpool( + lambda: quota_coordinator().remote.export_store( + remote_store.ServerRemoteStore(abspath, manifest), + include_cache=include_cache, + ) + ) + return RemoteCacheFileResponse( + artifact, + cleanup=cleanup, + filename=path.name, + headers={"ETag": f'"{etag}"'}, + ) + if remote_proxy.policy.enabled and remote_proxy.policy.cache_backend == "sparse" and include_cache: abspath = get_abspath(path, user) reference = remote_proxy.inspect(abspath) if abspath.suffix in {".b2nd", ".b2frame"} else None @@ -4232,6 +4266,23 @@ async def get_file_content(path, user, decompress=True, include_cache=True): abspath = get_abspath(path, user) suffix = abspath.suffix + from caterva2.services import remote_store + + manifest = remote_store.inspect(abspath) + if manifest is not None and (remote_proxy.policy.enabled or not include_cache): + + def snapshot_store(): + artifact, _, cleanup = quota_coordinator().remote.export_store( + remote_store.ServerRemoteStore(abspath, manifest), + include_cache=include_cache, + ) + try: + return artifact.read_bytes() + finally: + cleanup() + + return await concurrency.run_in_threadpool(snapshot_store) + if suffix in {".b2frame", ".b2nd"}: reference = remote_proxy.inspect(abspath) if reference is not None: diff --git a/caterva2/services/sparse_cache.py b/caterva2/services/sparse_cache.py index cbfd6170..bea70483 100644 --- a/caterva2/services/sparse_cache.py +++ b/caterva2/services/sparse_cache.py @@ -8,6 +8,7 @@ import contextlib import hashlib +import io import json import logging import math @@ -69,8 +70,11 @@ def measure(path): def sync_tree(path): for entry in path.iterdir(): - if entry.is_symlink() or entry.is_dir(): + if entry.is_symlink(): raise ValueError("unexpected entry in sparse frame") + if entry.is_dir(): + sync_tree(entry) + continue with entry.open("rb") as stream: os.fsync(stream.fileno()) sync_directory(path) @@ -101,6 +105,11 @@ def __init__(self, quota, *, initialize=True): if initialize: with quota.connect() as db: db.executescript(SCHEMA) + generation_columns = {row[1] for row in db.execute("PRAGMA table_info(remote_generations)")} + if "kind" not in generation_columns: + db.execute( + "ALTER TABLE remote_generations ADD COLUMN kind TEXT NOT NULL DEFAULT 'array'" + ) columns = {row[1] for row in db.execute("PRAGMA table_info(account)")} if "cache_fill_suspended" not in columns: db.execute( @@ -109,10 +118,13 @@ def __init__(self, quota, *, initialize=True): if "cache_backend" not in columns: db.execute("ALTER TABLE account ADD COLUMN cache_backend TEXT NOT NULL DEFAULT 'sparse'") previous = db.execute("SELECT cache_backend FROM account").fetchone()[0] - if db.execute("PRAGMA user_version").fetchone()[0] == 2 and previous != quota.cache_backend: + if ( + db.execute("PRAGMA user_version").fetchone()[0] in (2, 3) + and previous != quota.cache_backend + ): raise StorageBusy("backend switch requires explicit offline configuration migration") db.execute("UPDATE account SET cache_backend=?", (quota.cache_backend,)) - db.execute("PRAGMA user_version=2") + db.execute("PRAGMA user_version=3") with quota.connect() as db: if db.execute("SELECT cache_backend FROM account").fetchone()[0] != quota.cache_backend: raise StorageBusy("backend switch requires draining storage workers") @@ -247,7 +259,7 @@ def _build(self, proxy, rel, sig, spec, stamp): (oid, rel, json.dumps(sig), spec, stamp, None, 0, now), ) db.execute( - "INSERT INTO remote_generations VALUES(?,?,?,?,?,?,?,?,?,?,?,?)", + "INSERT INTO remote_generations VALUES(?,?,?,?,?,?,?,?,?,?,?,?,'array')", (gid, oid, relative, "building", spec, stamp, proxy.max_cache_bytes, 0, 0, 0, now, now), ) with self.guard(gid): @@ -357,6 +369,141 @@ def uncached(): log.debug("deferred cache maintenance", exc_info=True) return result + def store_operation(self, store, callback, *, cached=None): + """Serialize discovery and leaf fills through the existing generation ledger.""" + from blosc2.msgpack_utils import msgpack_packb + + from caterva2.services import remote_store + + def execute(path=None, operation=callback): + with store.open(path) as runtime: + with runtime._owner.lock: + result = operation(runtime) + payload = runtime.cache_bytes if path is not None else 0 + if path is not None: + manifest = runtime._owner.disk.load() + remote_store.validate_manifest(manifest) + else: + nodes = { + key: (kind, value if kind == "unsupported" else None) + for key, (kind, value) in runtime._owner.nodes.items() + } + remote_store.validate_manifest(dict(store.manifest, nodes=nodes)) + payload = 0 + return result, payload + + if store.cache_policy != "disk": + return execute()[0] + rel = self.q.relative(store.path) + spec = hashlib.sha256(msgpack_packb(store.manifest)).hexdigest() + sig = store.carrier_generation + assembled = False + try: + with file_lock(self.q.control / "initialize.lock", shared=True), self.q.lock(rel): + if signature(store.path) != sig: + raise StorageBusy("RemoteStore carrier changed after inspection") + if hashlib.sha256(msgpack_packb(remote_store.inspect(store.path))).hexdigest() != spec: + raise StorageBusy("RemoteStore descriptor changed after inspection") + with self.q.connect() as db: + row = db.execute( + "SELECT o.active_generation,g.relpath,o.carrier_generation,o.spec_hash " + "FROM remote_objects o JOIN remote_generations g ON g.generation_id=o.active_generation " + "WHERE o.path=? AND g.state='active'", + (rel,), + ).fetchone() + if row and (row[2:] != (json.dumps(sig), spec) or not self.path(row[1]).exists()): + self.retire_path(rel) + row = None + if row: + gid, private = row[:2] + path = self.path(private) + else: + oid, gid = uuid.uuid4().hex, uuid.uuid4().hex + path = self.root / oid / gid + now = time.time() + with self.q.transaction() as db: + db.execute( + "INSERT INTO remote_objects VALUES(?,?,?,?,?,?,?,?)", + ( + oid, + rel, + json.dumps(sig), + spec, + json.dumps(store.manifest["source"]), + gid, + 0, + now, + ), + ) + db.execute( + "INSERT INTO remote_generations VALUES(?,?,?,?,?,?,?,?,?,?,?,?,?)", + ( + gid, + oid, + path.relative_to(self.q.root).as_posix(), + "active", + spec, + json.dumps(store.manifest["source"]), + store.max_cache_bytes, + 0, + 0, + 0, + now, + now, + "store", + ), + ) + with self.guard(gid): + with self.q.connect() as db: + if db.execute( + "SELECT 1 FROM remote_operations WHERE generation_id=?", (gid,) + ).fetchone(): + raise StorageBusy("Store generation needs recovery") + if row and cached is not None: + self._intent(gid, "probe") + (hit, value), payload = execute(path, cached) + sync_tree(path) + self._finish(gid, path, payload) + if hit: + return value + # ponytail: coarse soft admission, as for array fills; exact mutation reports can refine it. + if not self._admit(gid, min(store.max_cache_bytes or (1 << 20), 1 << 20) + 65536): + raise QuotaExceeded("store cache retention refused") + path.mkdir(parents=True, exist_ok=True) + result, payload = execute(path) + assembled = True + sync_tree(path) + self._finish(gid, path, payload) + if store.manifest["caches"]: + self._intent(gid, "coldify") + cold = io.BytesIO() + remote_store.cold_export(store.manifest, cold) + new_sig = self.q.publish_locked( + rel, cold.getvalue(), expected=sig, preserve_remote=True + ) + store.manifest = dict(store.manifest, caches=[]) + store.carrier_generation = new_sig + spec = hashlib.sha256(msgpack_packb(store.manifest)).hexdigest() + with self.q.transaction() as db: + db.execute( + "UPDATE remote_objects SET carrier_generation=?,spec_hash=? WHERE active_generation=?", + (json.dumps(new_sig), spec, gid), + ) + db.execute( + "UPDATE remote_generations SET spec_hash=? WHERE generation_id=?", + (spec, gid), + ) + db.execute("DELETE FROM remote_operations WHERE generation_id=?", (gid,)) + except (OSError, sqlite3.Error, QuotaExceeded): + log.debug("store retention unavailable", exc_info=True) + if not assembled: + result = execute()[0] + try: + self.prune(force=not assembled) + except (OSError, sqlite3.Error, ValueError): + log.debug("deferred store maintenance", exc_info=True) + return result + def export(self, proxy): """Return an immutable warm artifact and its response-lifetime cleanup.""" from caterva2.services import remote_proxy @@ -437,6 +584,60 @@ def cleanup(): cleanup() raise + def export_store(self, store, *, include_cache=True): + """Reserve response-lifetime staging for a portable store snapshot.""" + from caterva2.services import remote_store + + opid = uuid.uuid4().hex + folder = self.q.control / "exports" + if folder.is_symlink(): + raise ValueError("export directory cannot be a symlink") + folder.mkdir(mode=0o700, exist_ok=True) + destination = folder / f"{opid}.b2z" + owner = file_lock(self.q.control / f"export-{opid}.lock") + owner.__enter__() + + def cleanup(): + try: + destination.unlink(missing_ok=True) + sync_directory(folder) + with self.q.transaction() as db: + db.execute("DELETE FROM remote_work WHERE id=?", (opid,)) + finally: + owner.__exit__(None, None, None) + + try: + with self.q.transaction() as db: + busy = db.execute("SELECT coalesce(sum(working),0) FROM operations").fetchone()[0] + _, _, work = self.totals(db) + budget = db.execute("SELECT work_bytes FROM account").fetchone()[0] + if busy + work: + raise QuotaExceeded("export staging budget is in use") + if shutil.disk_usage(folder).free < budget: + raise QuotaExceeded("insufficient store export headroom") + db.execute( + "INSERT INTO remote_work VALUES(?,?,?,?,?)", + (opid, "export", budget, destination.relative_to(self.q.root).as_posix(), time.time()), + ) + if include_cache: + self.store_operation( + store, + lambda runtime: runtime.save(destination, mutable=store.manifest.get("mutable", False)), + ) + else: + remote_store.cold_export(store.manifest, destination) + if destination.stat().st_size > budget: + raise QuotaExceeded("store export exceeded its staging budget") + digest = hashlib.sha256() + with destination.open("rb") as stream: + os.fsync(stream.fileno()) + for block in iter(lambda: stream.read(1 << 20), b""): + digest.update(block) + return destination, digest.hexdigest(), cleanup + except BaseException: + cleanup() + raise + def prune(self, *, force=False): """Bounded whole-generation cleanup plus authorized-free chunk eviction.""" try: @@ -452,13 +653,13 @@ def prune(self, *, force=False): return with self.q.connect() as db: rows = db.execute( - "SELECT g.generation_id,g.relpath,g.payload_bytes,o.path " + "SELECT g.generation_id,g.relpath,g.payload_bytes,o.path,g.kind,g.source_stamp " "FROM remote_generations g JOIN remote_objects o ON o.object_id=g.object_id " "WHERE g.state='active' AND NOT EXISTS (SELECT 1 FROM remote_operations p " "WHERE p.generation_id=g.generation_id) ORDER BY touched LIMIT 4" ).fetchall() remaining = 64 - for gid, private, payload, rel in rows: + for gid, private, payload, rel, kind, source in rows: try: with self.q.lock(rel, blocking=False), self.guard(gid, blocking=False): with self.q.connect() as db: @@ -474,10 +675,20 @@ def prune(self, *, force=False): if not needed or not remaining: break path = self.path(private) + if not path.exists(): + continue self._intent(gid, "prune") - evicted, payload = blosc2.RemoteArray.trim_sparse_cache( - path, max(0, payload - needed), max_chunks=remaining - ) + if kind == "store": + evicted, payload = blosc2.RemoteStore.trim_sparse_cache( + path, + json.loads(source), + max(0, payload - needed), + max_chunks=remaining, + ) + else: + evicted, payload = blosc2.RemoteArray.trim_sparse_cache( + path, max(0, payload - needed), max_chunks=remaining + ) remaining -= len(evicted) sync_tree(path) self._finish(gid, path, payload) @@ -613,7 +824,10 @@ def recover_exports(self): with self.q.connect() as db: rows = db.execute("SELECT id,relpath FROM remote_work").fetchall() for opid, rel in rows: - if not re.fullmatch("[0-9a-f]{32}", opid) or rel != f".storage/exports/{opid}.b2nd": + if not re.fullmatch("[0-9a-f]{32}", opid) or rel not in { + f".storage/exports/{opid}.b2nd", + f".storage/exports/{opid}.b2z", + }: raise ValueError("invalid export registry path") try: with file_lock(self.q.control / f"export-{opid}.lock", blocking=False): diff --git a/caterva2/services/srv_utils.py b/caterva2/services/srv_utils.py index d5f8027c..0bf9b658 100644 --- a/caterva2/services/srv_utils.py +++ b/caterva2/services/srv_utils.py @@ -304,6 +304,11 @@ def open_container(abspath): is not one (single-array .b2z, corrupt/non-container file, wrong suffix).""" suffix = abspath.suffix if suffix == ".b2z": + from caterva2.services import remote_store + + manifest = remote_store.inspect(abspath) + if manifest is not None: + return remote_store.ServerRemoteStore(abspath, manifest) try: store = blosc2.open(abspath) except Exception: @@ -468,6 +473,24 @@ def read_metadata(obj, mtime=None): # `mtime` is used when `obj` is an already-opened object (e.g. a container # leaf) with no file of its own; callers pass the container's mtime. # Open dataset + from caterva2.services import remote_store + + if isinstance(obj, remote_store.ServerStoreArray): + empty = blosc2.empty(obj.shape, obj.dtype, chunks=obj.chunks, blocks=obj.blocks, cparams=obj.cparams) + result = read_metadata(empty, mtime=mtime) + result.attrs = result.schunk.attrs = obj.attrs + result.schunk.vlmeta = obj.attrs + result.accept_ranges = "none" + return result + if isinstance(obj, str | pathlib.Path): + manifest = remote_store.inspect(obj) + if manifest is not None: + path = pathlib.Path(obj) + return models.Directory( + mtime=path.stat().st_mtime, + size=path.stat().st_size, + nfiles=sum(kind == "ndarray" for kind, _ in manifest["nodes"].values()), + ) if isinstance(obj, pathlib.Path): path = obj if not path.is_file(): diff --git a/caterva2/services/storage_quota.py b/caterva2/services/storage_quota.py index bf91b30e..d0ff1c41 100644 --- a/caterva2/services/storage_quota.py +++ b/caterva2/services/storage_quota.py @@ -108,7 +108,7 @@ def __init__(self, statedir, quota, *, work_bytes=WORK_BYTES, cache_backend="con with self.connect() as db: db.execute("PRAGMA journal_mode=WAL") version = db.execute("PRAGMA user_version").fetchone()[0] - if version not in (0, 1, 2): + if version not in (0, 1, 2, 3): raise RuntimeError("unsupported storage quota schema version") db.executescript(""" CREATE TABLE IF NOT EXISTS objects ( @@ -169,7 +169,11 @@ def startup_guard(self): except StorageBusy: try: with self.connect() as db: - ready = db.execute("SELECT quota,work_bytes FROM account").fetchone() + ready = ( + db.execute("SELECT quota,work_bytes FROM account").fetchone() + if db.execute("PRAGMA user_version").fetchone()[0] == 3 + else None + ) except sqlite3.Error: ready = None if ready is not None: diff --git a/caterva2/tests/test_remote_store.py b/caterva2/tests/test_remote_store.py new file mode 100644 index 00000000..4824f2f3 --- /dev/null +++ b/caterva2/tests/test_remote_store.py @@ -0,0 +1,216 @@ +"""Remote-store policy, sparse retention, and existing container API integration.""" + +import blosc2 +import fsspec +import numpy as np +import pytest + +from caterva2.services import remote_proxy, remote_store, sparse_cache, srv_utils, storage_quota + + +@pytest.fixture +def store_runtime(tmp_path, monkeypatch): + monkeypatch.setenv("CATERVA2_SECRET", "test-secret") + from caterva2.services import server + + source = tmp_path / "source.b2z" + data = np.random.default_rng(42).integers(0, 10000, 20000, dtype="i4") + with blosc2.TreeStore(source, mode="w", threshold=0) as tree: + tree["/g/a"] = blosc2.asarray(data, chunks=(5000,), blocks=(1000,)) + tree["/g/b"] = blosc2.asarray(data + 1, chunks=(5000,), blocks=(1000,)) + fs = fsspec.filesystem("memory") + url = "https://data.example/source.b2z" + fs.pipe_file(url, source.read_bytes()) + root = tmp_path / "state" + (root / "public").mkdir(parents=True) + path = root / "public/store.b2z" + with blosc2.RemoteStore( + url, cache_policy=blosc2.CachePolicy.DISK, cache_dir=tmp_path / "creator", _filesystem=fs + ) as store: + store.save(path, include_cache=False) + q = storage_quota.StorageQuota(root, 0, cache_backend="sparse") + monkeypatch.setattr(server, "quota_coordinator", lambda: q) + monkeypatch.setattr( + remote_proxy, "policy", remote_proxy.Policy(enabled=True, allowed_hosts=("data.example",)) + ) + monkeypatch.setattr(remote_proxy, "_public_addresses", lambda *args: ("93.184.216.34",)) + monkeypatch.setattr(remote_proxy, "_https_filesystem", lambda *args: fs) + return q, path, data + + +def test_store_policy_inspection_and_shared_retention(store_runtime, monkeypatch): + q, path, data = store_runtime + original = path.read_bytes() + adapter = srv_utils.open_container(path) + assert adapter.leaves() == ["/g/a", "/g/b"] + first = adapter.get("/g/a") + np.testing.assert_array_equal(first[:5000], data[:5000]) + second = srv_utils.open_container(path).get("/g/a") + + def no_fetch(*args, **kwargs): + raise AssertionError("warm leaf fetched upstream payload") + + with monkeypatch.context() as patch: + patch.setattr(blosc2.B2ZNDSource, "get_chunk", no_fetch) + np.testing.assert_array_equal(second[:5000], data[:5000]) + assert path.read_bytes() == original + with q.connect() as db: + assert ( + db.execute( + "SELECT count(*) FROM remote_generations WHERE kind='store' AND state='active'" + ).fetchone()[0] + == 1 + ) + rel = db.execute("SELECT relpath FROM remote_generations WHERE state='active'").fetchone()[0] + assert db.execute("SELECT count(*) FROM remote_operations").fetchone()[0] == 0 + assert q.usage()["remote_cache_used"] >= sparse_cache.measure(q.root / rel)[0] > 0 + info = srv_utils.container_member_info(path, "/g/a") + assert info.shape == data.shape + assert info.accept_ranges == "none" + info.model_dump_json() + monkeypatch.setattr(remote_proxy, "policy", remote_proxy.Policy()) + assert remote_store.inspect(path) is not None + assert srv_utils.open_container(path).leaves() == ["/g/a", "/g/b"] + with pytest.raises(Exception, match=r"403|disabled"): + first[:1] + + +def test_store_retirement_and_offline_pruning(store_runtime, monkeypatch): + q, path, data = store_runtime + leaf = srv_utils.open_container(path).get("g/a") + np.testing.assert_array_equal(leaf[:], data) + with q.connect() as db: + private, source = db.execute("SELECT relpath,source_stamp FROM remote_generations").fetchone() + import json + + removed, remaining = blosc2.RemoteStore.trim_sparse_cache(q.root / private, json.loads(source), 0) + assert removed + assert remaining == 0 + q.publish(path, None, expected=storage_quota.signature(path)) + q.remote.maintain() + assert q.usage()["remote_cache_used"] == 0 + + +def test_store_quota_denial_does_not_retain(store_runtime): + q, path, data = store_runtime + q.quota = path.stat().st_size + with q.transaction() as db: + db.execute("UPDATE account SET quota=?", (q.quota,)) + leaf = srv_utils.open_container(path).get("g/a") + np.testing.assert_array_equal(leaf[:], data) + assert q.usage()["remote_cache_used"] == 0 + + +def test_store_warm_seed_is_migrated_once(store_runtime, tmp_path, monkeypatch): + q, path, data = store_runtime + manifest = remote_store.inspect(path) + warm = tmp_path / "uploaded.b2z" + with blosc2.RemoteStore( + manifest["source"]["urlpath"], + cache_dir=tmp_path / "warm-creator", + _filesystem=fsspec.filesystem("memory"), + ) as store: + with store["g/a"] as array: + array[:5000] + store.save(warm) + q.publish(path, warm.read_bytes(), expected=storage_quota.signature(path)) + assert remote_store.inspect(path)["caches"] + leaf = srv_utils.open_container(path).get("g/a") + assert remote_store.inspect(path)["caches"] == [] + with monkeypatch.context() as patch: + patch.setattr(blosc2.B2ZNDSource, "get_chunk", lambda *args: pytest.fail("seed was not retained")) + np.testing.assert_array_equal(leaf[:5000], data[:5000]) + + +def test_store_interrupted_generation_recovers_offline(store_runtime, monkeypatch): + q, path, data = store_runtime + leaf = srv_utils.open_container(path).get("g/a") + np.testing.assert_array_equal(leaf[:5000], data[:5000]) + with q.connect() as db: + gid = db.execute("SELECT generation_id FROM remote_generations WHERE state='active'").fetchone()[0] + q.remote._intent(gid, "fill") + monkeypatch.setattr( + remote_proxy, "_https_filesystem", lambda *args: pytest.fail("recovery contacted source") + ) + q.remote.maintain() + assert q.usage()["remote_cache_used"] == 0 + assert q.usage()["reserved"] == 0 + + +def test_store_warm_hit_survives_admission_denial(store_runtime, monkeypatch): + q, path, data = store_runtime + leaf = srv_utils.open_container(path).get("g/a") + np.testing.assert_array_equal(leaf[:5000], data[:5000]) + used = q.usage()["used"] + with q.transaction() as db: + db.execute("UPDATE account SET cache_fill_suspended=1,quota=?", (used,)) + monkeypatch.setattr( + blosc2.B2ZNDSource, "get_chunk", lambda *args: pytest.fail("warm hit fetched payload") + ) + np.testing.assert_array_equal(leaf[:5000], data[:5000]) + + +@pytest.mark.parametrize( + "changes", + [ + {"cache_policy": "none", "max_cache_bytes": 5}, + {"cache_policy": "memory", "max_cache_bytes": None}, + {"cache_policy": "disk", "max_cache_bytes": True}, + ], +) +def test_store_rejects_invalid_limits(store_runtime, changes): + _, path, _ = store_runtime + manifest = dict(remote_store.inspect(path), **changes) + with pytest.raises(remote_proxy.RemoteArrayDenied): + remote_store.validate_manifest(manifest) + + +@pytest.mark.asyncio +async def test_store_http_routes_and_exports(store_runtime, monkeypatch, tmp_path): + import httpx + + from caterva2.services import server + + q, path, data = store_runtime + monkeypatch.setattr(server.settings, "statedir", q.root) + monkeypatch.setattr(server.settings, "public", path.parent) + monkeypatch.setattr(server.settings, "shared", q.root / "shared") + monkeypatch.setattr(server.settings, "personal", q.root / "personal") + overrides = dict(server.app.dependency_overrides) + server.app.dependency_overrides[server.optional_user] = lambda: None + try: + async with httpx.AsyncClient( + transport=httpx.ASGITransport(app=server.app), base_url="http://test" + ) as client: + response = await client.get("/api/list/@public/store.b2z") + assert response.status_code == 200, response.text + assert response.json() == ["g/a", "g/b"] + response = await client.get("/api/info/@public/store.b2z/g/a") + assert response.status_code == 200, response.text + assert response.json()["accept_ranges"] == "none" + response = await client.get("/api/fetch/@public/store.b2z/g/a", params={"slice_": "0:5000"}) + assert response.status_code == 200, response.text + np.testing.assert_array_equal(blosc2.ndarray_from_cframe(response.content)[:], data[:5000]) + response = await client.get("/api/chunk/@public/store.b2z/g/a", params={"nchunk": 0}) + assert response.status_code == 200, response.text + np.testing.assert_array_equal( + np.frombuffer(blosc2.decompress(response.content), dtype="i4"), data[:5000] + ) + for include in (True, False): + response = await client.get( + "/api/download/@public/store.b2z", params={"include_cache": str(include).lower()} + ) + assert response.status_code == 200, response.text + out = tmp_path / f"export-{include}.b2z" + out.write_bytes(response.content) + manifest = remote_store.inspect(out) + assert bool(manifest["caches"]) == include + assert q.usage()["working"] == 0 + monkeypatch.setattr(remote_proxy, "policy", remote_proxy.Policy()) + response = await client.get("/api/fetch/@public/store.b2z/g/a") + assert response.status_code == 403 + response = await client.get("/api/download/@public/store.b2z", params={"include_cache": "false"}) + assert response.status_code == 200 + finally: + server.app.dependency_overrides.clear() + server.app.dependency_overrides.update(overrides) diff --git a/doc/utilities/cat2-server.md b/doc/utilities/cat2-server.md index f1f68893..012f17ed 100644 --- a/doc/utilities/cat2-server.md +++ b/doc/utilities/cat2-server.md @@ -98,6 +98,44 @@ proxy. Logical `api/fetch` requests continue to return array data. ## Customer storage admission +### RemoteStore references + +Caterva2 also accepts portable `blosc2.RemoteStore.save()` archives (`.b2z`) +for B2Z, HDF5, and Zarr sources. They are browsable containers: for example, +`@public/store.b2z/group/array` supports metadata, sliced fetches, and compressed +chunk reads. Known names and attributes can be inspected without contacting the +source; undiscovered metadata and array geometry require authorized discovery. +The same `[server.remote_proxy]` HTTPS policy applies to discovery and leaf reads. +`max_metadata_bytes` (default 16 MiB) and `max_nodes` (default 100,000) bound +discovery in addition to the existing per-array geometry limits. + +DISK stores use private sparse RemoteArray leaf caches, one shared discovery +manifest, and one aggregate compressed-payload allowance across all leaves. +Multiple processes can use the same cache simultaneously; operations within one +store serialize under OS locks. Independent stores can proceed concurrently. +The SQLite ledger charges allocated storage, including manifests and directories. +Requested MEMORY/NONE stores execute without retained payload. A denied cache +fill falls back to a read without retention; existing warm hits remain usable. + +Uploaded warm leaves are imported once and the public archive is replaced with +a cold descriptor. Downloads include private warm cache data by default; +`include_cache=false` produces a cold archive without network access. Export +staging is reserved until the response completes. Interrupted disposable +generations are retired by maintenance without resolving their sources. + +Sources are immutable until a new reference is published. To refresh a hosted +store, refresh it in python-blosc2, save a new archive, and upload that archive +as a replacement. Replacement/deletion retires the previous private generation. +Restart workers together when upgrading: storage schema version 3 adds store +generation accounting and prevents workers from using a partly initialized ledger. + +This support requires the current Python-Blosc2 4.13 development APIs (and their +forthcoming release). Use `RemoteStore.with_sparse_cache()` for standalone +shared runtime access; the ordinary upstream `cache_dir` constructor retains +its exclusive-owner semantics. + +### Admission and recovery + With `[server] quota` enabled, `storage.sqlite` coordinates workers sharing one customer's local state directory. It uses Python's standard-library `sqlite3`, independently of authentication; no additional dependency is needed. Multi-host diff --git a/examples/benchmark_remote_store.py b/examples/benchmark_remote_store.py new file mode 100644 index 00000000..70d31544 --- /dev/null +++ b/examples/benchmark_remote_store.py @@ -0,0 +1,181 @@ +"""Shared RemoteStore benchmark over local range I/O, including Caterva2 admission. + +Run in the blosc2 environment from the repository root: + python examples/benchmark_remote_store.py --output /tmp/remote-store.json + +The fixture maps one synthetic HTTPS host to a local file. No network latency, +TLS, or HTTP server is included. Source reads/bytes count actual file range I/O; +worker startup and fixture creation are excluded from measured phases. +""" + +import argparse +import json +import multiprocessing +import os +import platform +import tempfile +import time +from pathlib import Path + +import blosc2 +import numpy as np +from fsspec.implementations.local import LocalFileSystem + + +class FixtureFS(LocalFileSystem): + reads = 0 + nbytes = 0 + + @classmethod + def _strip_protocol(cls, path): + return super()._strip_protocol(str(path).removeprefix("https://data.example")) + + def cat_file(self, *args, **kwargs): + result = super().cat_file(*args, **kwargs) + type(self).reads += 1 + type(self).nbytes += len(result) + return result + + +def worker(engine, policy, source, state, barrier, output, size, rounds): + try: + blosc2.set_nthreads(1) + fs = FixtureFS(skip_instance_cache=True) + url = "https://data.example" + source + expected = np.random.default_rng(42).integers(0, 1 << 24, size, dtype="i4") + if engine == "caterva2": + os.environ["CATERVA2_SECRET"] = "benchmark-fixture" + from caterva2.services import remote_proxy, remote_store, server, storage_quota + + remote_proxy.policy = remote_proxy.Policy(enabled=True, allowed_hosts=("data.example",)) + remote_proxy._public_addresses = lambda *args: ("93.184.216.34",) + remote_proxy._https_filesystem = lambda *args: fs + quota = storage_quota.StorageQuota(state, 0, cache_backend="sparse") + server.quota_coordinator = lambda: quota + path = Path(state) / "public/store.b2z" + adapter = remote_store.ServerRemoteStore(path, remote_store.inspect(path)) + arrays = [adapter.get("a"), adapter.get("b")] + close = adapter.close + else: + if policy == "disk": + store = blosc2.RemoteStore.with_sparse_cache( + url, state, _filesystem=fs, max_cache_bytes=64 << 20 + ) + else: + store = blosc2.RemoteStore(url, cache_policy=blosc2.CachePolicy.NONE, _filesystem=fs) + arrays = [store["a"], store["b"]] + + def close(): + for array in arrays: + array.close() + store.close() + + phases = [] + for _ in range(2): + barrier.wait(timeout=60) + FixtureFS.reads = FixtureFS.nbytes = 0 + start = time.perf_counter() + for _ in range(rounds): + for i, array in enumerate(arrays): + np.testing.assert_array_equal(array[:], expected + i) + phases.append( + { + "seconds": time.perf_counter() - start, + "source_reads": FixtureFS.reads, + "source_bytes": FixtureFS.nbytes, + } + ) + close() + output.put(phases) + except BaseException as exc: + output.put({"error": repr(exc)}) + + +def benchmark(size=250_000, rounds=3): + ctx = multiprocessing.get_context("spawn") + results = [] + with tempfile.TemporaryDirectory(prefix="caterva2-store-bench-") as folder: + root = Path(folder).resolve() + source = root / "source.b2z" + data = np.random.default_rng(42).integers(0, 1 << 24, size, dtype="i4") + with blosc2.TreeStore(source, mode="w", threshold=0) as tree: + for i, key in enumerate(("a", "b")): + tree[key] = blosc2.asarray(data + i, chunks=(25000,), blocks=(5000,)) + for engine in ("upstream", "caterva2"): + for workers in (1, 4): + for policy in ("none", "disk"): + state = root / f"{engine}-{workers}-{policy}" + if engine == "caterva2": + (state / "public").mkdir(parents=True) + cache_policy = ( + blosc2.CachePolicy.DISK if policy == "disk" else blosc2.CachePolicy.NONE + ) + options = {"cache_dir": root / f"creator-{workers}"} if policy == "disk" else {} + with blosc2.RemoteStore( + "https://data.example" + str(source), + cache_policy=cache_policy, + _filesystem=FixtureFS(skip_instance_cache=True), + **options, + ) as store: + store.save(state / "public/store.b2z", include_cache=False) + barrier, output = ctx.Barrier(workers), ctx.Queue() + processes = [ + ctx.Process( + target=worker, + args=(engine, policy, str(source), str(state), barrier, output, size, rounds), + ) + for _ in range(workers) + ] + for process in processes: + process.start() + try: + measurements = [output.get(timeout=120) for _ in processes] + if any(isinstance(result, dict) for result in measurements): + raise RuntimeError(measurements) + for process in processes: + process.join(timeout=10) + if process.exitcode != 0: + raise RuntimeError(f"worker exit: {process.exitcode}") + finally: + for process in processes: + if process.is_alive(): + process.terminate() + process.join() + output.close() + for phase in range(2): + seconds = max(result[phase]["seconds"] for result in measurements) + results.append( + { + "engine": engine, + "workers": workers, + "policy": policy, + "phase": "cold" if phase == 0 else "warm", + "seconds": seconds, + "MiB_per_second": workers * rounds * 2 * data.nbytes / seconds / 2**20, + "source_reads": sum( + result[phase]["source_reads"] for result in measurements + ), + "source_bytes": sum( + result[phase]["source_bytes"] for result in measurements + ), + } + ) + return { + "platform": platform.platform(), + "blosc2": blosc2.__version__, + "elements_per_leaf": size, + "rounds": rounds, + "results": results, + } + + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--output", type=Path) + parser.add_argument("--size", type=int, default=250_000) + parser.add_argument("--rounds", type=int, default=3) + args = parser.parse_args() + result = json.dumps(benchmark(args.size, args.rounds), indent=2) + "\n" + if args.output: + args.output.write_text(result) + print(result, end="") diff --git a/plans/remote-store.md b/plans/remote-store.md new file mode 100644 index 00000000..6588e7f7 --- /dev/null +++ b/plans/remote-store.md @@ -0,0 +1,242 @@ +# RemoteArray update and RemoteStore support + +## Objective and agreed scope + +Update Caterva2 for the current python-blosc2 API and support persisted +`RemoteStore` references with disk caches shared by multiple server processes. +Implement the necessary upstream features directly in +`/Users/faltet/blosc/python-blosc2` before its upcoming release. + +- Replace `RemoteProxy` with `RemoteArray`, including persisted object markers. + `RemoteProxy` was never released upstream: no legacy carrier compatibility or + migration layer is required. +- Use sparse `RemoteArray` storage for the leaves of a server-managed + `RemoteStore` under `CachePolicy.DISK`. +- Share discovery metadata and enforce one aggregate compressed-payload limit + across all leaves of a store. +- Keep portable `.b2z` artifacts separate from writable private runtime caches. +- Support simultaneous handles in multiple processes sharing one local state + directory. Initially serialize operations within each store; separate stores + can proceed independently. Multi-host/network-filesystem sharing is outside + this implementation. +- Reuse Caterva2's existing policy boundary, OS locks, SQLite storage ledger, + admission, recovery, pruning, and export lifecycle. + +## Findings from the initial analysis + +The inspected python-blosc2 checkout and editable installation report +`4.13.0.dev0`. They expose `RemoteArray` and `RemoteStore`, but no `RemoteProxy`. +The array persistence marker is now `remote_array`. + +Caterva2's `services/remote_proxy.py` authorizes remote array sources, while +`services/sparse_cache.py` manages private sparse generations. Array requests +already serialize under dataset and generation locks, including cache fills. +`services/storage_quota.py` coordinates storage admission across workers. + +Upstream `StoreDiskCache` currently takes a nonblocking exclusive OS lock for +the lifetime of the store and its dependent handles. A second owner, including +another process, cannot open that cache. Discovery state and the aggregate +`CacheCoordinator` are process-local. Shortening the lifetime lock alone would +allow stale manifest writes and incorrect aggregate accounting. + +Store artifacts use a `.b2z` archive containing `embed.b2e`, a `b2remote_store` +marker, a discovery manifest, and optional leaf caches. Live storage uses a +`.b2d` directory. The existing Caterva2 cache manager assumes a single flat +sparse array directory in several build, sync, and export operations. + +Security must be integrated before enabling store resolution. Caterva2's +container adapter currently calls `blosc2.open()` directly; upstream recognizes +store artifacts and may initiate remote discovery during opening. RemoteStore +also lacks the authorized-filesystem injection path used for Caterva2 arrays. +Authorized standalone B2Z sparse attachment is currently rejected upstream. + +## 1. Update Caterva2 to RemoteArray + +- Update Python API references, capability checks, persisted marker checks and + writers, server type dispatch, tests, examples, and documentation. +- Keep the existing `[server.remote_proxy]` configuration usable; renaming the + upstream class does not require deployment configuration churn. +- Remove assumptions that a missing `RemoteProxy` means remote-reference + inspection can be skipped. +- Update embedded-expression guards to recognize current remote descriptors. +- Use the local editable python-blosc2 checkout during development. Set the + dependency floor to the release containing the completed required APIs once + that release version is established. +- Run the existing remote-array, sparse-cache, quota, and API tests before + extending their behavior for stores. + +Primary files: `caterva2/services/remote_proxy.py`, `sparse_cache.py`, +`srv_utils.py`, `server.py`, `caterva2/client.py`, and `pyproject.toml`, plus +their tests, examples, and documentation. + +## 2. Add shared RemoteStore runtime support upstream + +Provide a server-facing attachment interface modeled on +`RemoteArray.with_sparse_cache()`. It must accept authorized discovery/transport +and private runtime storage separately from the portable store artifact. +Finalize the exact API after tracing the existing constructors and callers. + +### Ownership and operation ordering + +Use a process-local thread guard paired with an operation-scoped OS lock per +store. Multiple processes may retain handles, but every operation that observes +or mutates shared cache state must synchronize through this protocol: + +1. Acquire the store lock and read the active generation and current manifest. +2. Reject stale child handles after a generation change; reopen or refresh + cached leaf state as needed within the current generation. +3. Reconstruct aggregate retained-payload accounting from shared leaf state. +4. Perform discovery, cached reads, fills, or eviction. +5. Persist leaf and manifest changes before releasing ownership. + +Apply the protocol to root/group operations and child RemoteArray operations, +including cleanup/finalization that writes metadata. Establish one consistent +lock order with Caterva2's initialization, dataset, generation, and leaf locks. +Avoid holding SQLite transactions during remote I/O. + +### Sparse leaf storage and aggregate limits + +- Reuse RemoteArray sparse storage and Proxy chunk/block machinery for leaves. +- Support authorized B2Z leaf attachment as well as HDF5 and Zarr sources. +- Preserve a single store-wide payload allowance; do not give every leaf an + independent copy of that allowance. +- Reload accounting and eviction state under the store lock so one process + sees another process's fills and evictions. +- Preserve discovery metadata when payload chunks are evicted. +- Keep source discovery and format-specific decoding in python-blosc2 rather + than reproducing them in Caterva2. + +Primary upstream files: `src/blosc2/remote_store.py`, `remote_store_cache.py`, +`remote_array.py`, and the existing cache coordinator in `proxy.py`. + +## 3. Complete recovery, refresh, and maintenance primitives + +- Record interrupted mutations before changing persistent cache state. +- Make manifest publication recoverable. The current atomic replacement of + `active_generation.json` does not make in-place updates of the manifest in + `embed.b2e` atomic. +- Recover or invalidate interrupted disposable cache data without resolving + remote sources. Reuse existing array dirty-cache recovery where applicable. +- Preserve immutable-until-refresh source semantics. Build replacement + discovery before publishing a new generation; failed refresh must preserve + the current generation. +- Detect refresh from other processes and make old child handles stale. +- Coordinate retirement and deletion with active operations and exports. +- Expose cached-only reads, bounded offline trimming, payload accounting, and + warm/cold export functionality needed by Caterva2, reusing array primitives. +- Ensure exported artifacts remain portable and never expose private cache + paths or runtime credentials. + +## 4. Integrate policy and container access in Caterva2 + +### Inspect before resolving + +Recognize `b2remote_store` and inspect the manifest without dispatching through +an unrestricted `blosc2.open()`. Apply this boundary to listings, metadata, +mountability probes, fetches, chunks, downloads, and embedded references. + +Validate uploaded manifest structure, node paths, leaf references, source +descriptors, and archive members before resolution or extraction. Treat +persisted metadata as untrusted. Validate remote leaf geometry against the +authorized source before using uploaded warm cache data. + +### Authorize discovery and leaf reads + +- Extend the existing HTTPS policy to store discovery and every leaf transport: + explicit allowlist, public DNS addresses, address pinning, no redirects, + credential-free URLs, timeouts, and bounded fetch concurrency. +- Add upstream authorized-filesystem injection and propagate it through B2Z, + HDF5, Zarr, restored manifests, and refresh paths. +- Bound discovery metadata and node counts alongside per-array rank, logical + size, and chunk limits. Choose documented defaults during implementation. +- Keep outbound resolution disabled by default. Known persisted discovery can + be inspected without network access; additional discovery requires policy + authorization. + +### Serve stores through existing routes + +Add a RemoteStore adapter beside the TreeStore/HDF5 adapters in `srv_utils.py`. +Support groups, listings, attributes, leaf metadata, sliced fetches, and +compressed chunk reads through existing server routes and browser mounts. +Report unsupported nodes consistently. + +Keep child handles alive through reads and streamed responses, and close them +explicitly afterward. Update physical downloads to export warm or cold store +artifacts without mutating the hosted descriptor. Preserve the distinction +between logical leaf fetches and portable store downloads. + +Preserve Caterva2's effective no-retention handling of requested MEMORY/NONE +policies; DISK uses the managed shared runtime cache. + +## 5. Extend quota and generation lifecycle + +Represent a store's manifest and sparse leaves as one managed store generation. +Extend the existing ledger only where its array assumptions require it. + +- Enforce the aggregate compressed-payload cap separately from customer quota. +- Charge allocated filesystem storage, including metadata, directories, leaf + caches, and retired generations awaiting cleanup. +- Admit discovery growth as well as payload fills; metadata is retained storage + even though it is outside the evictable payload limit. +- Reuse operation records, generation guards, admission fallback, and startup + reconciliation. A denied cache fill should still return successfully fetched + data without retaining it. +- Adapt flat-directory sync/measurement and offline pruning for store layouts. +- Retire store generations on replacement, deletion, and source refresh. +- Reserve export working storage and retain ownership until response cleanup. +- Preserve metadata and safely account for valid uploaded warm caches when + initializing private runtime storage; do not repeatedly restore evicted + chunks from the portable artifact. + +Primary files: `caterva2/services/sparse_cache.py`, `storage_quota.py`, and the +publication/download paths in `server.py`. + +## 6. Verification and acceptance criteria + +Extend existing test suites and multiprocessing patterns rather than adding a +new test framework. Use the `blosc2` conda environment for upstream Python, +builds, and tests, as required by its repository instructions. + +### Upstream checks + +- Two or more processes keep handles open against the same cache without an + ownership error. +- Same-leaf and different-leaf reads return correct data and reuse previously + retained chunks across processes. +- Concurrent discovery preserves both workers' discovered nodes. +- Aggregate eviction respects one store-wide limit across processes and leaves. +- Process death during leaf mutation or manifest publication leaves recoverable + state; subsequent reads are correct. +- Refresh invalidates child handles across processes; failed refresh preserves + the old generation. +- Offline trimming/recovery performs no network access. +- Warm and cold artifacts reopen correctly, with metadata preserved. +- Exercise B2Z, HDF5, and Zarr store sources with suitable deterministic fixtures. + +### Caterva2 checks + +- Existing RemoteArray, sparse cache, quota, and API tests pass with the renamed + API and current persistence markers. +- Store listings, metadata, mounts, sliced fetches, chunks, and downloads work. +- Disabled/denied sources, malformed manifests, unsafe references, and alternate + opening paths cannot bypass policy. +- Quota pressure, discovery growth, export cleanup, replacement/deletion, and + worker death leave consistent accounting and no active untracked generation. +- Measure upstream request counts as well as values: correct data alone does + not demonstrate cross-process cache reuse. +- Run repository formatting, lint, whitespace, and applicable pre-commit checks. + Preserve comments/docstrings and exactly one trailing newline in edited files. + +## Delivery order and performance boundary + +Complete the RemoteArray update first, then upstream shared-store primitives, +then Caterva2 policy/container integration and quota lifecycle, followed by +end-to-end verification and documentation. + +Benchmark competing workers reading the same and different leaves, recording +throughput, upstream traffic, and retained storage. Initial operations serialize +per store, matching the existing Caterva2 array-cache coordination model. +Parallel fills into different leaves require finer locks and coordinated +aggregate eviction; add them only if measurement shows store-level locking is +a bottleneck. This does not defer simultaneous open handles or shared-cache +correctness, both of which are required in the initial implementation. diff --git a/pyproject.toml b/pyproject.toml index c251457c..7a38f181 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -39,7 +39,7 @@ classifiers = [ "Operating System :: Unix", ] dependencies = [ - "blosc2>=4.8.1", + "blosc2>=4.13.0.dev0", "httpx[http2]", "numpy", ] From df57853a03bb3ef86d6ab10c29636917e8902291 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Mon, 14 Sep 2026 07:41:30 +0200 Subject: [PATCH 13/20] Fixed different issues: MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Indirect remote-reference policy bypass through saved expressions. - HTTP 500 when moving SChunk .b2frame files. - Unnecessary whole-file reads during overwrite/delete. - RemoteStore missing/malformed member handling. - Caterva2’s embedded HDF5 frame reader, a test false positive, and stale documentation. --- caterva2/hdf5.py | 3 ++- caterva2/services/remote_proxy.py | 29 ++++++++++++++++++++---- caterva2/services/remote_store.py | 14 +++++++++--- caterva2/services/server.py | 9 ++++---- caterva2/tests/test_providers.py | 2 +- caterva2/tests/test_remote_proxy.py | 22 ++++++++++++++++++ caterva2/tests/test_remote_store.py | 10 ++++++++ caterva2/tests/test_storage_quota_api.py | 12 ++++++++++ doc/utilities/cat2-server.md | 6 ++--- 9 files changed, 90 insertions(+), 17 deletions(-) diff --git a/caterva2/hdf5.py b/caterva2/hdf5.py index 358bcf6b..a7f8913a 100644 --- a/caterva2/hdf5.py +++ b/caterva2/hdf5.py @@ -29,7 +29,8 @@ # Losing the reference to the array may result in a segmentation fault. def b2_from_h5chunk(h5_dset: h5py.Dataset, chunk_index: int) -> blosc2.NDArray | blosc2.SChunk: h5chunk_info = h5_dset.id.get_chunk_info(chunk_index) - return blosc2.open(h5_dset.file.filename, mode="r", offset=h5chunk_info.byte_offset) + # Open the embedded frame directly; public open dispatches .h5 to an HDF5 source. + return blosc2.blosc2_ext.open(h5_dset.file.filename, "r", h5chunk_info.byte_offset) def h5dset_is_compatible(h5_dset: h5py.Dataset) -> bool: diff --git a/caterva2/services/remote_proxy.py b/caterva2/services/remote_proxy.py index 43f147ef..d6e65841 100644 --- a/caterva2/services/remote_proxy.py +++ b/caterva2/services/remote_proxy.py @@ -25,6 +25,7 @@ import weakref from dataclasses import dataclass from inspect import signature +from pathlib import Path from urllib.parse import urlsplit import aiohttp @@ -178,7 +179,7 @@ def inspect(path): return carrier, payload -def guard_embedded(path) -> None: +def guard_embedded(path, *, _seen=None) -> None: """Reject remote references hidden in another persisted B2 object. Structured LazyExpr/LazyUDF decoding resolves operand references while the @@ -188,28 +189,46 @@ def guard_embedded(path) -> None: """ if not hasattr(blosc2, "RemoteArray"): return + path = Path(path).resolve() + seen = set() if _seen is None else _seen + if path in seen: + return + seen.add(path) + if path.suffix == ".b2z": + from caterva2.services import remote_store + + if remote_store.inspect(path) is not None: + raise RemoteArrayDenied("remote stores embedded in persisted expressions are disabled") try: carrier = raw_carrier(path) except (RuntimeError, ValueError): return schunk = getattr(carrier, "schunk", carrier) marker = schunk.meta.get("b2o") + if isinstance(marker, dict) and marker.get("kind") == "remote_array": + raise RemoteArrayDenied("remote arrays embedded in persisted expressions are disabled") if not isinstance(marker, dict) or marker.get("kind") not in {"lazyexpr", "lazyudf"}: return payload = schunk.vlmeta.get("b2o") - if _contains_remote_reference(payload): + if _contains_remote_reference(payload, base_path=path.parent, seen=seen): raise RemoteArrayDenied( "remote references embedded in persisted expressions are disabled by server policy" ) -def _contains_remote_reference(value) -> bool: +def _contains_remote_reference(value, *, base_path=None, seen=None) -> bool: if isinstance(value, dict): if value.get("kind") in {"fsspec", "remote_array", "remote_store", "hdf5", "zarr", "b2z"}: return True - return any(_contains_remote_reference(item) for item in value.values()) + if base_path is not None and value.get("kind") in {"urlpath", "dictstore_key"}: + local = value.get("urlpath") + if isinstance(local, str): + guard_embedded(base_path / local, _seen=seen) + return any( + _contains_remote_reference(item, base_path=base_path, seen=seen) for item in value.values() + ) if isinstance(value, list | tuple): - return any(_contains_remote_reference(item) for item in value) + return any(_contains_remote_reference(item, base_path=base_path, seen=seen) for item in value) return False diff --git a/caterva2/services/remote_store.py b/caterva2/services/remote_store.py index 149a766a..2cc486b8 100644 --- a/caterva2/services/remote_store.py +++ b/caterva2/services/remote_store.py @@ -164,7 +164,10 @@ def _key(self, key): def leaves(self, prefix="/"): root = self.manifest["source"].get("dataset", "") - full = self._key(prefix).rstrip("/") + try: + full = self._key(prefix).rstrip("/") + except ValueError: + return [] listed = self.manifest["listed"] if self.manifest["source"]["kind"] in {"b2z", "hdf5"} or full in listed: # Use known discovery offline only when all descendant groups are listed. @@ -202,13 +205,15 @@ def get(self, key): try: known = self.manifest["nodes"].get(self._key(key)) + if known is None and self.manifest["source"]["kind"] in {"b2z", "hdf5"}: + return None kind = known[0] if known else self.operation(lambda store: store.kind(key.strip("/"))) if kind == "group": return GROUP if kind != "ndarray": return None return ServerStoreArray(self, key.strip("/")) - except KeyError: + except (KeyError, ValueError): return None def is_group(self, node): @@ -217,7 +222,10 @@ def is_group(self, node): return node is GROUP def is_leaf(self, key): - known = self.manifest["nodes"].get(self._key(key)) + try: + known = self.manifest["nodes"].get(self._key(key)) + except ValueError: + return False if known is not None: return known[0] == "ndarray" node = self.get(key) diff --git a/caterva2/services/server.py b/caterva2/services/server.py index 1df20e4b..0f8157a8 100644 --- a/caterva2/services/server.py +++ b/caterva2/services/server.py @@ -206,7 +206,7 @@ def write_dataset(path, data, *, expected=None, compare=False): quota = quota_coordinator() if quota is not None: if not compare: - _, expected = quota.snapshot(path) + expected = storage_quota.signature(path) quota.publish(path, data, expected=expected) else: path.parent.mkdir(parents=True, exist_ok=True) @@ -231,7 +231,7 @@ def remove_dataset(path): elif path.name.endswith(".b2lock"): return # Stable locks are operational storage, not deletable dataset bytes. else: - _, generation = quota.snapshot(path) + generation = storage_quota.signature(path) if generation is None: raise FileNotFoundError(path) quota.publish(path, None, expected=generation, prune=False) @@ -277,8 +277,9 @@ def move_dataset(source, destination): if generation is None: raise storage_quota.StorageBusy("move source was removed") if remote_proxy.policy.cache_backend == "sparse" and source.suffix in {".b2nd", ".b2frame"}: - carrier = blosc2.ndarray_from_cframe(data) - if carrier.schunk.vlmeta.get("b2o", {}).get("kind") == "remote_array": + schunk = blosc2.schunk_from_cframe(data) + if schunk.meta.get("b2o", {}).get("kind") == "remote_array": + carrier = blosc2.ndarray_from_cframe(data) data = remote_proxy.cold_cframe(carrier, carrier.schunk.vlmeta["b2o"]) write_dataset(destination, data) quota.publish(source, None, expected=generation, prune=False) diff --git a/caterva2/tests/test_providers.py b/caterva2/tests/test_providers.py index d966fce7..087a7157 100644 --- a/caterva2/tests/test_providers.py +++ b/caterva2/tests/test_providers.py @@ -139,7 +139,7 @@ def test_server_source_has_no_c2cache_coupling(): src = ( __import__("pathlib").Path(__file__).resolve().parent.parent / "services" / "server.py" ).read_text() - leak = re.compile(r"\b(c2cache|peers_mod|peercache|remote\.)\b") + leak = re.compile(r"\b(c2cache|peers_mod|peercache)\b|(? Date: Mon, 14 Sep 2026 08:05:46 +0200 Subject: [PATCH 14/20] Use public API for embedded HDF5 frames --- caterva2/hdf5.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/caterva2/hdf5.py b/caterva2/hdf5.py index a7f8913a..358bcf6b 100644 --- a/caterva2/hdf5.py +++ b/caterva2/hdf5.py @@ -29,8 +29,7 @@ # Losing the reference to the array may result in a segmentation fault. def b2_from_h5chunk(h5_dset: h5py.Dataset, chunk_index: int) -> blosc2.NDArray | blosc2.SChunk: h5chunk_info = h5_dset.id.get_chunk_info(chunk_index) - # Open the embedded frame directly; public open dispatches .h5 to an HDF5 source. - return blosc2.blosc2_ext.open(h5_dset.file.filename, "r", h5chunk_info.byte_offset) + return blosc2.open(h5_dset.file.filename, mode="r", offset=h5chunk_info.byte_offset) def h5dset_is_compatible(h5_dset: h5py.Dataset) -> bool: From 501c108e7974ea47273616f071f193db072cf123 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Mon, 21 Sep 2026 11:57:07 +0200 Subject: [PATCH 15/20] Serve remote HDF5 tables as CTables --- caterva2/services/remote_store.py | 53 ++++++++++++++-- caterva2/services/server.py | 31 ++++++++-- caterva2/services/srv_utils.py | 2 + caterva2/tests/test_remote_store.py | 96 +++++++++++++++++++++++++++++ 4 files changed, 172 insertions(+), 10 deletions(-) diff --git a/caterva2/services/remote_store.py b/caterva2/services/remote_store.py index 2cc486b8..37d2e5d3 100644 --- a/caterva2/services/remote_store.py +++ b/caterva2/services/remote_store.py @@ -180,7 +180,7 @@ def leaves(self, prefix="/"): return sorted( "/" + (key[len(root) + 1 :] if root else key) for key, (kind, _) in self.manifest["nodes"].items() - if kind == "ndarray" and (not full or key.startswith(full + "/")) + if kind in {"ndarray", "ctable"} and (not full or key.startswith(full + "/")) ) def discover(store): @@ -210,9 +210,11 @@ def get(self, key): kind = known[0] if known else self.operation(lambda store: store.kind(key.strip("/"))) if kind == "group": return GROUP - if kind != "ndarray": - return None - return ServerStoreArray(self, key.strip("/")) + if kind == "ndarray": + return ServerStoreArray(self, key.strip("/")) + if kind == "ctable": + return ServerStoreTable(self, key.strip("/")) + return None except (KeyError, ValueError): return None @@ -227,7 +229,7 @@ def is_leaf(self, key): except ValueError: return False if known is not None: - return known[0] == "ndarray" + return known[0] in {"ndarray", "ctable"} node = self.get(key) return node is not None and not self.is_group(node) @@ -279,6 +281,47 @@ def read(runtime): ) +class ServerStoreTable: + """Operation-scoped view of a CTable inside a portable RemoteStore.""" + + def __init__(self, store, key): + self.store, self.key, self.path = store, key, store.path + + def metadata(runtime): + with runtime[key] as table: + schema = table.schema_dict() + nbytes, cbytes = table.nbytes, table.cbytes + return { + "nrows": table.nrows, + "ncols": table.ncols, + "chunks": table.chunks, + "blocks": table.blocks, + "schema_dict": schema, + "columns": [column["name"] for column in schema.get("columns", [])], + "nbytes": nbytes, + "cbytes": cbytes, + "cratio": nbytes / cbytes if cbytes else 0, + "vlmeta": dict(table.vlmeta[:]) if table.vlmeta[:] else {}, + "attrs": dict(table.attrs), + } + + self.metadata = store.operation(metadata) + self.nrows = self.metadata["nrows"] + + def fetch(self, slice_=None, *, filter=None, field=None): + from caterva2.services.srv_utils import ctable_row_range + + def read(runtime): + with runtime[self.key] as table: + view = table.where(filter) if filter else table + start, stop = ctable_row_range(slice_, view.nrows) + if field is not None: + return blosc2.asarray(view[field][start:stop]).to_cframe() + return view.slice(start, stop).to_cframe() + + return self.store.operation(read) + + def validate_array(array): policy = remote_proxy.policy if hasattr(array, "max_concurrency"): diff --git a/caterva2/services/server.py b/caterva2/services/server.py index 0f8157a8..1758cd21 100644 --- a/caterva2/services/server.py +++ b/caterva2/services/server.py @@ -1209,16 +1209,29 @@ async def fetch_data( window = None # where a container leaf's frame lies, when it has one filter = filter.strip() if filter else filter + store_table_filter = None if filter: if field: srv_utils.raise_bad_request("Cannot handle both field and filter parameters at the same time") mtime = abspath.stat().st_mtime try: - container, _ = await concurrency.run_in_threadpool( - lambda: get_filtered_array( - abspath, path, filter, sortby=None, mtime=mtime, inner_key=inner_key + from caterva2.services.remote_store import ServerStoreTable + + container = ( + await concurrency.run_in_threadpool( + lambda: srv_utils.open_container_member(abspath, inner_key) ) + if inner_key is not None + else None ) + if isinstance(container, ServerStoreTable): + store_table_filter = filter + else: + container, _ = await concurrency.run_in_threadpool( + lambda: get_filtered_array( + abspath, path, filter, sortby=None, mtime=mtime, inner_key=inner_key + ) + ) except ValueError as exc: srv_utils.raise_bad_request(str(exc)) elif inner_key is not None: @@ -1234,7 +1247,10 @@ async def fetch_data( else: container = open_b2(abspath, path) - if field: + from caterva2.services.remote_store import ServerStoreTable + + store_table_field = field if isinstance(container, ServerStoreTable) else None + if field and store_table_field is None: container = container[field] from caterva2.services.remote_store import ServerRemoteStore @@ -1268,7 +1284,7 @@ async def fetch_data( schunk = getattr(array, "schunk", None) # not really needed typesize = array.dtype.itemsize shape = array.shape - elif isinstance(container, blosc2.CTable): + elif isinstance(container, blosc2.CTable | ServerStoreTable): array = container schunk = None typesize = 1 # not used for CTable @@ -1302,6 +1318,7 @@ async def fetch_data( | hdf5.HDF5Proxy | blosc2.NDField | blosc2.CTable + | ServerStoreTable | remote_proxy.ServerRemoteArray, ) ) @@ -1349,6 +1366,10 @@ async def fetch_data( ) except (IndexError, ValueError) as exc: srv_utils.raise_bad_request(str(exc)) + elif isinstance(array, ServerStoreTable): + data = await concurrency.run_in_threadpool( + lambda: array.fetch(slice_, filter=store_table_filter, field=store_table_field) + ) elif isinstance(array, blosc2.CTable): row_start, row_stop = srv_utils.ctable_row_range(slice_, array.nrows) view = array.slice(row_start, row_stop) diff --git a/caterva2/services/srv_utils.py b/caterva2/services/srv_utils.py index 0bf9b658..d172cb27 100644 --- a/caterva2/services/srv_utils.py +++ b/caterva2/services/srv_utils.py @@ -475,6 +475,8 @@ def read_metadata(obj, mtime=None): # Open dataset from caterva2.services import remote_store + if isinstance(obj, remote_store.ServerStoreTable): + return models.CTableMetadata(mtime=mtime, **obj.metadata) if isinstance(obj, remote_store.ServerStoreArray): empty = blosc2.empty(obj.shape, obj.dtype, chunks=obj.chunks, blocks=obj.blocks, cparams=obj.cparams) result = read_metadata(empty, mtime=mtime) diff --git a/caterva2/tests/test_remote_store.py b/caterva2/tests/test_remote_store.py index 0b700a4e..b0bd1a06 100644 --- a/caterva2/tests/test_remote_store.py +++ b/caterva2/tests/test_remote_store.py @@ -1,7 +1,10 @@ """Remote-store policy, sparse retention, and existing container API integration.""" +import io + import blosc2 import fsspec +import h5py import numpy as np import pytest @@ -38,6 +41,52 @@ def store_runtime(tmp_path, monkeypatch): return q, path, data +@pytest.fixture +def hdf5_table_runtime(tmp_path, monkeypatch): + monkeypatch.setenv("CATERVA2_SECRET", "test-secret") + from caterva2.services import server + + data = np.array( + [(value, f"v{value}".encode()) for value in np.random.default_rng(4).permutation(21)], + dtype=[("id", " 0 + + response = await client.get("/api/fetch/@public/table-store.b2z/table", params=params) + assert response.status_code == 200, response.text + assert requests == cold_requests + finally: + server.app.dependency_overrides.clear() + server.app.dependency_overrides.update(overrides) From dfd55282c03cd027fde027b4ff9ce52c3c36e4d7 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Thu, 24 Sep 2026 11:42:51 +0200 Subject: [PATCH 16/20] Integrate RemoteCTable references --- caterva2/services/remote_proxy.py | 2 +- caterva2/services/remote_store.py | 152 +++++++++++--- caterva2/services/server.py | 27 ++- caterva2/services/sparse_cache.py | 16 +- caterva2/services/srv_utils.py | 7 +- caterva2/tests/test_remote_store.py | 310 ++++++++++++++++++++++++++++ doc/utilities/cat2-server.md | 14 +- plans/remote-ctable.md | 301 +++++++++++++++++++++++++++ pyproject.toml | 2 +- 9 files changed, 785 insertions(+), 46 deletions(-) create mode 100644 plans/remote-ctable.md diff --git a/caterva2/services/remote_proxy.py b/caterva2/services/remote_proxy.py index d6e65841..89ba98e0 100644 --- a/caterva2/services/remote_proxy.py +++ b/caterva2/services/remote_proxy.py @@ -601,6 +601,6 @@ def export_cframe(carrier, payload, *, include_cache: bool) -> bytes: """Snapshot a warm or cold carrier while excluding concurrent mutations.""" path = carrier.schunk.urlpath with carrier_thread_lock(path), carrier.schunk.holding_lock(): - if include_cache: + if include_cache or payload["cache_policy"] != "disk": return carrier.to_cframe() return cold_cframe(carrier, payload) diff --git a/caterva2/services/remote_store.py b/caterva2/services/remote_store.py index 37d2e5d3..b4adc411 100644 --- a/caterva2/services/remote_store.py +++ b/caterva2/services/remote_store.py @@ -49,8 +49,14 @@ def validate_manifest(manifest): raise remote_proxy.RemoteArrayDenied("RemoteStore source contains unsupported fields") if len(msgpack_packb(manifest)) > remote_proxy.policy.max_metadata_bytes: raise remote_proxy.RemoteArrayDenied("RemoteStore metadata exceeds the configured limit") - if len(manifest["nodes"]) > remote_proxy.policy.max_nodes: - raise remote_proxy.RemoteArrayDenied("RemoteStore node count exceeds the configured limit") + pending = [manifest] + nodes = 0 + while pending: + current = pending.pop() + nodes += len(current["nodes"]) + if nodes > remote_proxy.policy.max_nodes: + raise remote_proxy.RemoteArrayDenied("RemoteStore node count exceeds the configured limit") + pending.extend(entry["manifest"] for entry in current.get("linked", {}).values()) # Reuse array policy validation, including the strict cache-policy schema. policy = manifest.get("cache_policy", "none") limit = manifest.get("max_cache_bytes") @@ -81,8 +87,28 @@ def source_url(manifest): ) -def filesystem(manifest): - parsed = urlsplit(source_url(manifest)) +def root_kind(manifest): + root = manifest["source"].get("dataset", "") + return manifest["nodes"][root][0] + + +def filesystem_for_url(url): + parsed = urlsplit( + remote_proxy._validated_source( + { + "kind": "remote_array", + "version": 1, + "source": { + "kind": "fsspec", + "version": 1, + "urlpath": url, + "assume_immutable": True, + }, + "cache_policy": "none", + "max_cache_bytes": None, + } + ) + ) host = parsed.hostname.encode("idna").decode("ascii").lower() addresses = remote_proxy._public_addresses(host, parsed.port or 443) return remote_proxy._https_filesystem(host, addresses) @@ -90,7 +116,7 @@ def filesystem(manifest): def cold_export(manifest, destination): """Write a descriptor-only archive without resolving its source.""" - manifest = dict(manifest, caches=[]) + manifest = dict(manifest, caches=[], batch_caches=[], linked={}) storage = blosc2.Storage(contiguous=True) storage.meta = {"b2tree": {"version": 1}, "b2remote_store": {"version": 1}} embed = blosc2.SChunk(chunksize=8192, data=None, storage=storage) @@ -111,11 +137,20 @@ def __init__(self, path, manifest): @contextmanager def open(self, runtime=None): - fs = filesystem(self.manifest) + filesystems = [] + + def authorized_filesystem(url): + fs = filesystem_for_url(url) + filesystems.append(fs) + return fs + source = self.manifest["source"] + fs = authorized_filesystem(source["urlpath"]) options = { "_filesystem": fs, + "_filesystem_resolver": authorized_filesystem, "_source_validator": validate_array, + "_batch_validator": validate_batch, "_manifest_validator": validate_manifest, "_max_nodes": remote_proxy.policy.max_nodes, } @@ -135,15 +170,17 @@ def open(self, runtime=None): source["urlpath"], dataset=source.get("dataset"), cache_policy=blosc2.CachePolicy.NONE, - _manifest=dict(copy.deepcopy(self.manifest), caches=[]), + _manifest=dict(copy.deepcopy(self.manifest), caches=[], batch_caches=[], linked={}), + allow_table_root=True, **options, ) with store: yield store finally: - session = getattr(fs, "_session", None) - if session is not None: - fs.close_session(fs.loop, session) + for source_fs in filesystems: + session = getattr(source_fs, "_session", None) + if session is not None: + source_fs.close_session(source_fs.loop, session) def operation(self, callback, *, cached=None): from caterva2.services.server import quota_coordinator @@ -169,7 +206,8 @@ def leaves(self, prefix="/"): except ValueError: return [] listed = self.manifest["listed"] - if self.manifest["source"]["kind"] in {"b2z", "hdf5"} or full in listed: + has_links = any(kind == "remote_store" for kind, _ in self.manifest["nodes"].values()) + if (self.manifest["source"]["kind"] in {"b2z", "hdf5"} or full in listed) and not has_links: # Use known discovery offline only when all descendant groups are listed. complete = self.manifest["source"]["kind"] in {"b2z", "hdf5"} or all( key in listed @@ -189,11 +227,14 @@ def discover(store): while pending: key = pending.pop() node = store.get_info(key) - if node.kind == "ndarray": + if node.kind in {"ndarray", "ctable"}: result.append("/" + key) - elif node.kind == "group": + elif node.kind in {"group", "remote_store"}: with store[key] as group: - pending.extend("/".join(p for p in (key, child) if p) for child in group) + if node.kind == "remote_store" and group.kind("") == "ctable": + result.append("/" + key) + else: + pending.extend("/".join(p for p in (key, child) if p) for child in group) if len(result) + len(pending) > remote_proxy.policy.max_nodes: raise remote_proxy.RemoteArrayDenied("RemoteStore listing exceeds the node limit") return sorted(result) @@ -204,11 +245,23 @@ def get(self, key): from caterva2.services.srv_utils import GROUP try: - known = self.manifest["nodes"].get(self._key(key)) - if known is None and self.manifest["source"]["kind"] in {"b2z", "hdf5"}: + full = self._key(key) + known = self.manifest["nodes"].get(full) + linked = any( + kind == "remote_store" and full.startswith(path + "/") + for path, (kind, _) in self.manifest["nodes"].items() + ) + if known is None and self.manifest["source"]["kind"] in {"b2z", "hdf5"} and not linked: return None kind = known[0] if known else self.operation(lambda store: store.kind(key.strip("/"))) - if kind == "group": + if kind == "remote_store": + + def linked_kind(store): + with store[key.strip("/")] as linked_store: + return "ctable" if linked_store.kind("") == "ctable" else "remote_store" + + kind = self.operation(linked_kind) + if kind in {"group", "remote_store"}: return GROUP if kind == "ndarray": return ServerStoreArray(self, key.strip("/")) @@ -284,11 +337,16 @@ def read(runtime): class ServerStoreTable: """Operation-scoped view of a CTable inside a portable RemoteStore.""" - def __init__(self, store, key): + def __init__(self, store, key, *, filter=None, sortby=None): self.store, self.key, self.path = store, key, store.path + self.filter, self.sortby = filter, sortby def metadata(runtime): - with runtime[key] as table: + with self._open_table(runtime) as table: + if filter: + table = table.where(filter) + if sortby: + table = table.sort_by(sortby, view=True) schema = table.schema_dict() nbytes, cbytes = table.nbytes, table.cbytes return { @@ -308,18 +366,56 @@ def metadata(runtime): self.metadata = store.operation(metadata) self.nrows = self.metadata["nrows"] + @contextmanager + def _open_table(self, runtime): + with runtime[self.key] as node: + if isinstance(node, blosc2.RemoteStore): + with node[""] as table: + yield table + else: + yield node + + def schema_dict(self): + return self.metadata["schema_dict"] + + def where(self, expression): + return type(self)(self.store, self.key, filter=expression, sortby=self.sortby) + + def sort_by(self, column, *, view=False): + return type(self)(self.store, self.key, filter=self.filter, sortby=column) + + def slice(self, start, stop): + def read(runtime): + with self._open_table(runtime) as table: + if self.filter: + table = table.where(self.filter) + if self.sortby: + table = table.sort_by(self.sortby, view=True) + return table.slice(start, stop) + + return self.store.operation(read, cached=lambda runtime: runtime.read_cached_table(read)) + def fetch(self, slice_=None, *, filter=None, field=None): from caterva2.services.srv_utils import ctable_row_range + if field is not None and field not in self.metadata["columns"]: + raise fastapi.HTTPException(status_code=400, detail=f"Unknown table field: {field}") + def read(runtime): - with runtime[self.key] as table: - view = table.where(filter) if filter else table + with self._open_table(runtime) as table: + expression = filter or self.filter + try: + view = table.where(expression) if expression else table + except (NameError, SyntaxError) as exc: + raise ValueError(f"Invalid table filter: {exc}") from exc + if self.sortby: + view = view.sort_by(self.sortby, view=True) start, stop = ctable_row_range(slice_, view.nrows) if field is not None: - return blosc2.asarray(view[field][start:stop]).to_cframe() + return view.select([field]).slice(start, stop).to_cframe() return view.slice(start, stop).to_cframe() - return self.store.operation(read) + return self.store.operation(read, cached=lambda runtime: runtime.read_cached_table(read)) def validate_array(array): @@ -336,3 +432,13 @@ def validate_array(array): > policy.max_chunks ): raise remote_proxy.RemoteArrayDenied("RemoteStore leaf exceeds the chunk limit") + + +def validate_batch(batch): + policy = remote_proxy.policy + if ( + len(batch.offsets) > policy.max_chunks + or batch.member_length > policy.max_nbytes + or len(msgpack_packb((batch.meta, batch.vlmeta))) > policy.max_metadata_bytes + ): + raise remote_proxy.RemoteArrayDenied("RemoteStore batch exceeds resource limits") diff --git a/caterva2/services/server.py b/caterva2/services/server.py index 1758cd21..cfb5663d 100644 --- a/caterva2/services/server.py +++ b/caterva2/services/server.py @@ -371,7 +371,8 @@ def open_b2(abspath, path): manifest = remote_store.inspect(abspath) if manifest is not None: - return remote_store.ServerRemoteStore(abspath, manifest) + store = remote_store.ServerRemoteStore(abspath, manifest) + return store.get("") if remote_store.root_kind(manifest) == "ctable" else store reference = remote_proxy.inspect(abspath) if reference is not None: carrier, payload = reference @@ -1222,7 +1223,7 @@ async def fetch_data( lambda: srv_utils.open_container_member(abspath, inner_key) ) if inner_key is not None - else None + else await concurrency.run_in_threadpool(lambda: open_b2(abspath, path)) ) if isinstance(container, ServerStoreTable): store_table_filter = filter @@ -1655,7 +1656,9 @@ async def get_chunk( container = open_b2(abspath, path) else: container = open_member(abspath, inner_key, abspath.stat().st_mtime) - if isinstance(container, blosc2.CTable): + from caterva2.services.remote_store import ServerStoreTable + + if isinstance(container, blosc2.CTable | ServerStoreTable): srv_utils.raise_bad_request( f"{path} is a CTable, which is a set of columns rather than one chunked array; " "fetch it with the slice_ parameter instead" @@ -3493,7 +3496,9 @@ def _filtered_array(abspath, path, filter, sortby, mtime, inner_key): # HDF5Proxy supports slicing only; no string-indexed LazyExpr yet. raise ValueError("Filtering is not supported for HDF5-backed datasets") - if isinstance(arr, blosc2.CTable): + from caterva2.services.remote_store import ServerStoreTable + + if isinstance(arr, blosc2.CTable | ServerStoreTable): if filter: arr = arr.where(filter) if sortby: @@ -3539,7 +3544,9 @@ def _desc_window(total, start, size): def _is_ctable_like(arr): """True for real CTables and provider-backed views that render through the CTable grid (ViewHandle.array is duck-typed by design).""" - return isinstance(arr, blosc2.CTable) or ( + from caterva2.services.remote_store import ServerStoreTable + + return isinstance(arr, blosc2.CTable | ServerStoreTable) or ( hasattr(arr, "nrows") and hasattr(arr, "schema_dict") and hasattr(arr, "slice") ) @@ -3618,7 +3625,9 @@ async def htmx_path_view( elif filter or sortby: try: mtime = abspath.stat().st_mtime - arr, idx = get_filtered_array(abspath, path, filter, sortby, mtime, inner_key) + arr, idx = await concurrency.run_in_threadpool( + lambda: get_filtered_array(abspath, path, filter, sortby, mtime, inner_key) + ) except TypeError as exc: return htmx_error(request, f"Error in filter: {exc}") except NameError as exc: @@ -3642,7 +3651,7 @@ async def htmx_path_view( ) else: try: - arr = open_b2(abspath, path) + arr = await concurrency.run_in_threadpool(lambda: open_b2(abspath, path)) except ValueError: return htmx_error(request, "Cannot open array; missing operand?, unknown data source?") idx = None @@ -3688,9 +3697,9 @@ def cell(value): if sort_desc: # arr is ascending-sorted; read its tail and reverse for descending order. lo, hi = _desc_window(nrows, start, size) - window = list(arr.slice(lo, hi))[::-1] + window = await concurrency.run_in_threadpool(lambda: list(arr.slice(lo, hi))[::-1]) else: - window = arr.slice(start, stop) + window = await concurrency.run_in_threadpool(lambda: arr.slice(start, stop)) rows = [fields] + [[cell(row[f]) for f in fields] for row in window] context = { "view_url": make_url(request, "htmx_path_view", path=path), diff --git a/caterva2/services/sparse_cache.py b/caterva2/services/sparse_cache.py index bea70483..650b4526 100644 --- a/caterva2/services/sparse_cache.py +++ b/caterva2/services/sparse_cache.py @@ -385,7 +385,7 @@ def execute(path=None, operation=callback): remote_store.validate_manifest(manifest) else: nodes = { - key: (kind, value if kind == "unsupported" else None) + key: (kind, value if kind in {"ctable", "remote_store", "unsupported"} else None) for key, (kind, value) in runtime._owner.nodes.items() } remote_store.validate_manifest(dict(store.manifest, nodes=nodes)) @@ -474,14 +474,18 @@ def execute(path=None, operation=callback): assembled = True sync_tree(path) self._finish(gid, path, payload) - if store.manifest["caches"]: + if ( + store.manifest["caches"] + or store.manifest.get("batch_caches") + or store.manifest.get("linked") + ): self._intent(gid, "coldify") cold = io.BytesIO() remote_store.cold_export(store.manifest, cold) new_sig = self.q.publish_locked( rel, cold.getvalue(), expected=sig, preserve_remote=True ) - store.manifest = dict(store.manifest, caches=[]) + store.manifest = dict(store.manifest, caches=[], batch_caches=[], linked={}) store.carrier_generation = new_sig spec = hashlib.sha256(msgpack_packb(store.manifest)).hexdigest() with self.q.transaction() as db: @@ -558,9 +562,7 @@ def cleanup(): ) runtime = self._attach(proxy, path) try: - with destination.open("xb"): - pass - runtime.save(destination, mode="w") + runtime.save(destination) finally: del runtime exported = remote_proxy.raw_carrier(destination, mode="a") @@ -717,6 +719,8 @@ def cleanup(self, *, max_generations=4): os.replace(path, trash) sync_directory(path.parent) sync_directory(self.trash) + # Upstream sparse attachment leaves a sibling initialization lock. + path.with_name(path.name + ".init.lock").unlink(missing_ok=True) with self.q.transaction() as db: db.execute( "UPDATE remote_generations SET relpath=? WHERE generation_id=?", diff --git a/caterva2/services/srv_utils.py b/caterva2/services/srv_utils.py index d172cb27..6b7f017c 100644 --- a/caterva2/services/srv_utils.py +++ b/caterva2/services/srv_utils.py @@ -308,6 +308,8 @@ def open_container(abspath): manifest = remote_store.inspect(abspath) if manifest is not None: + if remote_store.root_kind(manifest) != "group": + return None return remote_store.ServerRemoteStore(abspath, manifest) try: store = blosc2.open(abspath) @@ -488,10 +490,13 @@ def read_metadata(obj, mtime=None): manifest = remote_store.inspect(obj) if manifest is not None: path = pathlib.Path(obj) + if remote_store.root_kind(manifest) == "ctable": + store = remote_store.ServerRemoteStore(path, manifest) + return read_metadata(store.get(""), mtime=path.stat().st_mtime) return models.Directory( mtime=path.stat().st_mtime, size=path.stat().st_size, - nfiles=sum(kind == "ndarray" for kind, _ in manifest["nodes"].values()), + nfiles=sum(kind in {"ndarray", "ctable"} for kind, _ in manifest["nodes"].values()), ) if isinstance(obj, pathlib.Path): path = obj diff --git a/caterva2/tests/test_remote_store.py b/caterva2/tests/test_remote_store.py index b0bd1a06..1bbb13ad 100644 --- a/caterva2/tests/test_remote_store.py +++ b/caterva2/tests/test_remote_store.py @@ -1,5 +1,6 @@ """Remote-store policy, sparse retention, and existing container API integration.""" +import dataclasses import io import blosc2 @@ -11,6 +12,202 @@ from caterva2.services import remote_proxy, remote_store, sparse_cache, srv_utils, storage_quota +@pytest.fixture +def table_runtime(tmp_path, monkeypatch): + monkeypatch.setenv("CATERVA2_SECRET", "test-secret") + from caterva2.services import server + + @dataclasses.dataclass + class Row: + x: int = blosc2.field(blosc2.int64()) + text: str = blosc2.field(blosc2.vlstring(batch_rows=2)) + + source = tmp_path / "table.b2z" + blosc2.CTable(Row, [(1, "one"), (2, "two")], create_summary_index=False).to_b2z(source) + fs = fsspec.filesystem("memory") + url = "https://data.example/table.b2z" + fs.pipe_file(url, source.read_bytes()) + root = tmp_path / "state" + (root / "public").mkdir(parents=True) + path = root / "public/table-reference.b2z" + with blosc2.RemoteCTable(url, cache_policy=blosc2.CachePolicy.NONE, _filesystem=fs) as table: + table.save(path, include_cache=False) + q = storage_quota.StorageQuota(root, 0, cache_backend="sparse") + monkeypatch.setattr(server, "quota_coordinator", lambda: q) + monkeypatch.setattr( + remote_proxy, "policy", remote_proxy.Policy(enabled=True, allowed_hosts=("data.example",)) + ) + monkeypatch.setattr(remote_proxy, "_public_addresses", lambda *args: ("93.184.216.34",)) + monkeypatch.setattr(remote_proxy, "_https_filesystem", lambda *args: fs) + return path + + +def test_standalone_remote_ctable_dispatch(table_runtime): + from caterva2.services import server + + path = table_runtime + assert remote_store.root_kind(remote_store.inspect(path)) == "ctable" + assert not srv_utils.is_container_file(path) + assert srv_utils.read_metadata(path).kind == "ctable" + table = server.open_b2(path, "@public/table-reference.b2z") + result = blosc2.ctable_from_cframe(table.fetch()) + assert list(result.x[:]) == [1, 2] + assert list(result.text[:]) == ["one", "two"] + + +def test_standalone_remote_ctable_disk_cache(table_runtime, monkeypatch): + from caterva2.services import server + + path = table_runtime.with_name("cached-table.b2z") + manifest = remote_store.inspect(table_runtime) + manifest = dict(manifest, cache_policy="disk", max_cache_bytes=1 << 20) + remote_store.cold_export(manifest, path) + table = server.open_b2(path, "@public/cached-table.b2z") + assert list(blosc2.ctable_from_cframe(table.fetch()).text[:]) == ["one", "two"] + from blosc2.b2z_source import B2ZBatchSource + + def no_fetch(*args, **kwargs): + raise AssertionError("warm batch fetched upstream payload") + + monkeypatch.setattr(B2ZBatchSource, "get_chunk", no_fetch) + assert list(blosc2.ctable_from_cframe(table.fetch()).text[:]) == ["one", "two"] + + +def test_standalone_remote_ctable_warm_batch_seed(table_runtime, tmp_path, monkeypatch): + from blosc2.b2z_source import B2ZBatchSource + + path = table_runtime + manifest = remote_store.inspect(path) + warm = tmp_path / "warm-table.b2z" + with blosc2.RemoteCTable( + manifest["source"]["urlpath"], + cache_dir=tmp_path / "creator", + _filesystem=fsspec.filesystem("memory"), + ) as table: + assert list(table.text[:]) == ["one", "two"] + table.save(warm) + assert remote_store.inspect(warm)["batch_caches"] + from caterva2.services import server + + q = server.quota_coordinator() + q.publish(path, warm.read_bytes(), expected=storage_quota.signature(path)) + table = server.open_b2(path, "@public/table-reference.b2z") + assert remote_store.inspect(path)["batch_caches"] == [] + + def no_fetch(*args, **kwargs): + raise AssertionError("warm batch seed fetched upstream payload") + + monkeypatch.setattr(B2ZBatchSource, "get_chunk", no_fetch) + assert list(blosc2.ctable_from_cframe(table.fetch()).text[:]) == ["one", "two"] + + +def test_standalone_remote_ctable_warm_hit_under_quota(table_runtime, monkeypatch): + from blosc2.b2z_source import B2ZBatchSource + + from caterva2.services import server + + path = table_runtime.with_name("cached-table.b2z") + manifest = dict(remote_store.inspect(table_runtime), cache_policy="disk", max_cache_bytes=1 << 20) + remote_store.cold_export(manifest, path) + table = server.open_b2(path, "@public/cached-table.b2z") + assert list(blosc2.ctable_from_cframe(table.fetch()).text[:]) == ["one", "two"] + q = server.quota_coordinator() + used = q.usage()["used"] + with q.transaction() as db: + db.execute("UPDATE account SET cache_fill_suspended=1,quota=?", (used,)) + monkeypatch.setattr( + B2ZBatchSource, + "get_chunk", + lambda *args: pytest.fail("warm hit fetched upstream payload"), + ) + assert list(blosc2.ctable_from_cframe(table.fetch()).text[:]) == ["one", "two"] + + +def test_standalone_remote_ctable_nested_columns(table_runtime, tmp_path): + from caterva2.services import server + + @dataclasses.dataclass + class RichRow: + text: str = blosc2.field(blosc2.vlstring(nullable=True, batch_rows=2)) + tags: list[int] = blosc2.field( # noqa: RUF009 + blosc2.list(blosc2.int64(), nullable=True, batch_rows=2) + ) + category: str = blosc2.field(blosc2.dictionary(nullable=True)) + + source = tmp_path / "rich.b2z" + blosc2.CTable( + RichRow, + [("one", [1, 2], "a"), (None, None, None), ("three", [3], "b")], + create_summary_index=False, + ).to_b2z(source) + url = "https://data.example/rich.b2z" + fs = fsspec.filesystem("memory") + fs.pipe_file(url, source.read_bytes()) + path = table_runtime.with_name("rich.b2z") + with blosc2.RemoteCTable(url, cache_policy=blosc2.CachePolicy.NONE, _filesystem=fs) as remote: + remote.save(path, include_cache=False) + table = server.open_b2(path, "@public/rich.b2z") + result = blosc2.ctable_from_cframe(table.fetch()) + assert list(result.text[:]) == ["one", None, "three"] + assert list(result.tags[:]) == [[1, 2], None, [3]] + assert list(result.category[:]) == ["a", None, "b"] + projected = table.fetch(field="text") + assert list(blosc2.ctable_from_cframe(projected).text[:]) == ["one", None, "three"] + + +@pytest.mark.asyncio +async def test_standalone_remote_ctable_api(table_runtime, monkeypatch): + import httpx + + from caterva2.services import server + + path = table_runtime + monkeypatch.setattr(server.settings, "statedir", path.parents[1]) + monkeypatch.setattr(server.settings, "public", path.parent) + monkeypatch.setattr(server.settings, "shared", path.parents[1] / "shared") + monkeypatch.setattr(server.settings, "personal", path.parents[1] / "personal") + overrides = dict(server.app.dependency_overrides) + server.app.dependency_overrides[server.optional_user] = lambda: None + try: + async with httpx.AsyncClient( + transport=httpx.ASGITransport(app=server.app), base_url="http://test" + ) as client: + endpoint = "/api/info/@public/table-reference.b2z" + response = await client.get(endpoint) + assert response.status_code == 200, response.text + assert response.json()["kind"] == "ctable" + response = await client.get("/api/fetch/@public/table-reference.b2z", params={"filter": "x > 1"}) + assert response.status_code == 200, response.text + assert list(blosc2.ctable_from_cframe(response.content).x[:]) == [2] + response = await client.get("/api/fetch/@public/table-reference.b2z", params={"field": "text"}) + assert response.status_code == 200, response.text + assert list(blosc2.ctable_from_cframe(response.content).text[:]) == ["one", "two"] + response = await client.get( + "/api/fetch/@public/table-reference.b2z", params={"field": "missing"} + ) + assert response.status_code == 400, response.text + response = await client.get( + "/api/fetch/@public/table-reference.b2z", params={"filter": "missing > 1"} + ) + assert response.status_code == 400, response.text + response = await client.post("/htmx/path-view/@public/table-reference.b2z", data={"sortby": "x"}) + assert response.status_code == 200, response.text + assert "one" in response.text + assert "two" in response.text + response = await client.get( + "/api/download/@public/table-reference.b2z", params={"include_cache": "false"} + ) + assert response.status_code == 200, response.text + downloaded = path.with_name("downloaded-reference.b2z") + downloaded.write_bytes(response.content) + assert remote_store.root_kind(remote_store.inspect(downloaded)) == "ctable" + response = await client.get("/api/chunk/@public/table-reference.b2z", params={"nchunk": 0}) + assert response.status_code == 400 + finally: + server.app.dependency_overrides.clear() + server.app.dependency_overrides.update(overrides) + + @pytest.fixture def store_runtime(tmp_path, monkeypatch): monkeypatch.setenv("CATERVA2_SECRET", "test-secret") @@ -124,6 +321,119 @@ def no_fetch(*args, **kwargs): first[:1] +def test_linked_remote_store_tables_and_arrays(store_runtime, tmp_path): + _, path, data = store_runtime + fs = fsspec.filesystem("memory") + source = remote_store.inspect(path)["source"]["urlpath"] + host = tmp_path / "host.b2z" + with ( + blosc2.RemoteStore(source, dataset="g", _filesystem=fs) as linked, + blosc2.TreeStore(host, mode="w") as tree, + ): + tree["linked"] = linked + host_url = "https://data.example/host.b2z" + fs.pipe_file(host_url, host.read_bytes()) + artifact = path.with_name("linked-store.b2z") + with blosc2.RemoteStore(host_url, _filesystem=fs) as store: + store.save(artifact, include_cache=False) + + adapter = srv_utils.open_container(artifact) + assert adapter.leaves() == ["/linked/a", "/linked/b"] + np.testing.assert_array_equal(adapter.get("linked/a")[:5000], data[:5000]) + + +def test_linked_remote_store_warm_seed(store_runtime, tmp_path, monkeypatch): + q, path, data = store_runtime + fs = fsspec.filesystem("memory") + source = remote_store.inspect(path)["source"]["urlpath"] + host = tmp_path / "warm-linked-host.b2z" + with ( + blosc2.RemoteStore(source, dataset="g", _filesystem=fs) as linked, + blosc2.TreeStore(host, mode="w") as tree, + ): + tree["linked"] = linked + host_url = "https://data.example/warm-linked-host.b2z" + fs.pipe_file(host_url, host.read_bytes()) + warm = tmp_path / "warm-linked.b2z" + with blosc2.RemoteStore( + host_url, + cache_dir=tmp_path / "creator-linked", + _filesystem=fs, + _filesystem_resolver=lambda url: fs, + ) as store: + with store["linked/a"] as array: + np.testing.assert_array_equal(array[:5000], data[:5000]) + store.save(warm) + artifact = path.with_name("warm-linked-reference.b2z") + q.publish(artifact, warm.read_bytes(), expected=None) + leaf = srv_utils.open_container(artifact).get("linked/a") + assert remote_store.inspect(artifact)["linked"] == {} + monkeypatch.setattr( + blosc2.B2ZNDSource, + "get_chunk", + lambda *args: pytest.fail("warm linked leaf fetched upstream payload"), + ) + np.testing.assert_array_equal(leaf[:5000], data[:5000]) + assert q.usage()["remote_cache_used"] > 0 + with q.connect() as db: + private, source_stamp = db.execute( + "SELECT g.relpath,g.source_stamp FROM remote_generations g " + "JOIN remote_objects o ON g.object_id=o.object_id WHERE o.path=?", + ("public/warm-linked-reference.b2z",), + ).fetchone() + import json + + removed, remaining = blosc2.RemoteStore.trim_sparse_cache(q.root / private, json.loads(source_stamp), 0) + assert removed + assert remaining == 0 + + +def test_linked_remote_store_ctable(table_runtime, tmp_path): + fs = fsspec.filesystem("memory") + table_url = remote_store.inspect(table_runtime)["source"]["urlpath"] + host = tmp_path / "host-table.b2z" + with ( + blosc2.RemoteStore(table_url, allow_table_root=True, _filesystem=fs) as linked, + blosc2.TreeStore(host, mode="w") as tree, + ): + tree["linked"] = linked + host_url = "https://data.example/host-table.b2z" + fs.pipe_file(host_url, host.read_bytes()) + artifact = table_runtime.with_name("linked-table.b2z") + with blosc2.RemoteStore(host_url, _filesystem=fs) as store: + store.save(artifact, include_cache=False) + + adapter = srv_utils.open_container(artifact) + assert adapter.leaves() == ["/linked"] + table = adapter.get("linked") + assert list(blosc2.ctable_from_cframe(table.fetch()).text[:]) == ["one", "two"] + + +def test_linked_remote_store_denied_destination(store_runtime, tmp_path): + from fastapi import HTTPException + + _, path, _ = store_runtime + fs = fsspec.filesystem("memory") + original = remote_store.inspect(path)["source"]["urlpath"] + blocked = "https://blocked.example/source.b2z" + fs.pipe_file(blocked, fs.cat_file(original)) + host = tmp_path / "host.b2z" + with ( + blosc2.RemoteStore(blocked, dataset="g", _filesystem=fs) as linked, + blosc2.TreeStore(host, mode="w") as tree, + ): + tree["linked"] = linked + host_url = "https://data.example/host-with-blocked-link.b2z" + fs.pipe_file(host_url, host.read_bytes()) + artifact = path.with_name("blocked-link.b2z") + with blosc2.RemoteStore(host_url, _filesystem=fs) as store: + store.save(artifact, include_cache=False) + + adapter = srv_utils.open_container(artifact) + with pytest.raises(HTTPException, match=r"blocked\.example"): + adapter.leaves() + + def test_store_retirement_and_offline_pruning(store_runtime, monkeypatch): q, path, data = store_runtime leaf = srv_utils.open_container(path).get("g/a") diff --git a/doc/utilities/cat2-server.md b/doc/utilities/cat2-server.md index 6d1c48a8..0d6e78c1 100644 --- a/doc/utilities/cat2-server.md +++ b/doc/utilities/cat2-server.md @@ -101,10 +101,15 @@ proxy. Logical `api/fetch` requests continue to return array data. ### RemoteStore references Caterva2 also accepts portable `blosc2.RemoteStore.save()` archives (`.b2z`) -for B2Z, HDF5, and Zarr sources. They are browsable containers: for example, +for B2Z, HDF5, and Zarr sources, plus `blosc2.RemoteCTable.save()` archives +for B2Z and PyTables/HDF5 tables. Store archives are browsable containers: for example, `@public/store.b2z/group/array` supports metadata, sliced fetches, and compressed chunk reads. Known names and attributes can be inspected without contacting the source; undiscovered metadata and array geometry require authorized discovery. +Table roots and table leaves report `ctable` metadata and support row slices, +filters, selected fields, and the existing table browser. A table has no +table-level compressed-chunk endpoint. Its columns, masks, batches, and indexes +share the enclosing reference's cache allowance. The same `[server.remote_proxy]` HTTPS policy applies to discovery and leaf reads. `max_metadata_bytes` (default 16 MiB) and `max_nodes` (default 100,000) bound discovery in addition to the existing per-array geometry limits. @@ -117,8 +122,8 @@ The SQLite ledger charges allocated storage, including manifests and directories Requested MEMORY/NONE stores execute without retained payload. A denied cache fill falls back to a read without retention; existing warm hits remain usable. -Uploaded warm leaves are imported once and the public archive is replaced with -a cold descriptor. Downloads include private warm cache data by default; +Uploaded warm leaves, table batches, and linked references are imported once, +then the public archive is replaced with a cold descriptor. Downloads include private warm cache data by default; `include_cache=false` produces a cold archive without network access. Export staging is reserved until the response completes. Interrupted disposable generations are retired by maintenance without resolving their sources. @@ -129,8 +134,7 @@ as a replacement. Replacement/deletion retires the previous private generation. Restart workers together when upgrading: storage schema version 3 adds store generation accounting and prevents workers from using a partly initialized ledger. -This support requires the current Python-Blosc2 4.13 development APIs (and their -forthcoming release). Use `RemoteStore.with_sparse_cache()` for standalone +This support requires Python-Blosc2 4.14.0. Use `RemoteStore.with_sparse_cache()` for standalone shared runtime access; the ordinary upstream `cache_dir` constructor retains its exclusive-owner semantics. diff --git a/plans/remote-ctable.md b/plans/remote-ctable.md new file mode 100644 index 00000000..941ac2eb --- /dev/null +++ b/plans/remote-ctable.md @@ -0,0 +1,301 @@ +# Full RemoteCTable integration + +## Scope and baseline + +Extend the RemoteArray/RemoteStore integration described in +[remote-store.md](remote-store.md) to support read-only RemoteCTable datasets +through Caterva2's existing table API, browser, and Python client. Cover both +standalone references and tables inside stores, including all column formats +and remote index reads supported by the installed python-blosc2. + +Baseline inspected on 2026-09-24: + +- Caterva2 commit `501c108` already serves HDF5 table members using + `ServerStoreTable`, with a regression test for native index reuse. +- The local python-blosc2 checkout is at `286b7da9`, the merge of + [PR #725](https://github.com/Blosc/python-blosc2/pull/725). The `blosc2` + conda environment imports that checkout and reports `4.13.2.dev0`. +- Inspection used the installed source and tests; the GitHub PR page/API was + unavailable during planning. +- A local memory-filesystem probe created and saved a RemoteCTable, then + confirmed that Caterva2 currently reports it as `Directory` and opens it + through `ServerRemoteStore`. + +Full support means parity with Caterva2's existing CTable read operations and +preservation of upstream remote semantics. Source writes and remote index +creation are not supported by RemoteCTable. Query results remain materialized +CTable cframes; physical downloads remain portable remote-reference artifacts. +Arbitrary upstream Python methods do not each need a new HTTP endpoint. + +## Upstream changes and release target + +RemoteCTable has not been released. Its first release will be python-blosc2 +**4.14.0**; the inspected `4.13.2.dev0` version is a development identifier, +not the intended release version. + +Improve python-blosc2 directly in `/Users/faltet/blosc/python-blosc2` wherever +its API or implementation prevents this integration. Fix the underlying +behavior and add suitable server-facing APIs before integrating them into +Caterva2. Do not introduce Caterva2 monkey patches, copied upstream internals, +parallel cache/query implementations, or compatibility workarounds for the +unreleased RemoteCTable API. Adjust that API as needed before 4.14.0. + +During implementation, make focused commits in python-blosc2 as appropriate, +with regression tests and documentation for each upstream change. This is +authorized as part of the integration work. Follow that repository's agent +instructions and use the `blosc2` conda environment. Keep upstream and +Caterva2 commits separate and record the upstream commits required by the +integration. Develop against the editable checkout; require `blosc2>=4.14.0` +for the released Caterva2 integration. + +The following upstream gaps were identified by source inspection. Begin with +reproducing tests, then implement and verify the fixes: + +1. **RemoteArray-backed table columns:** + `RemoteArray._from_carrier_with_owner()` opens the secondary source without + forwarding the owner's authorized filesystem or invoking its source + validator. Add authorization/transport selection before source I/O and + preserve geometry/resource validation. +2. **Linked RemoteStores:** deferred linked-store opening does not propagate + the parent's filesystem and validation hooks, including shared-cache and + artifact restoration paths. Provide authorization for each destination; + cross-host references need separately authorized transports, not blind + reuse of the parent's host-pinned filesystem. Preserve these hooks through + refresh and nested references. +3. **Batch resource validation:** `open_ctable_batch()` has no equivalent of + the array source-validation hook. Add validation suitable for batch-backed + columns and related payloads, early enough to enforce limits before payload + reads or unbounded allocation. Define the hook contract upstream. +4. **Cached-only table operations:** add a supported operation covering a + table query's columns, masks, batches, dictionaries, and indexes. Return a + clear cache miss without fetching missing data; coordinate the check/read + with shared-cache locking and generation changes. Caterva2 can then serve + warm queries under quota pressure and apply its normal fallback on a miss. + +Also improve upstream inspection, attachment, cold-export, or lifecycle APIs +if implementation reveals that Caterva2 would otherwise need to manipulate +private owner state or reproduce format logic. Keep deployment policy, HTTP +responses, browser integration, and customer-quota decisions in Caterva2; +keep source resolution, storage formats, and cache correctness in python-blosc2. + +## Existing machinery and concrete gaps + +RemoteCTable inherits from CTable and uses the RemoteStore owner, manifest, +transport, and aggregate cache. Saved references carry `b2remote_store` and +`b2remote_manifest`; the selected root node has kind `ctable`. There is no +separate persisted table marker to invent. + +Reuse `services/remote_store.py`, `ServerStoreTable`, the `store_operation` +quota path, and existing CTable metadata/client/rendering conventions. Keep +one generation per hosted reference, with table members sharing their store's +generation and budget. + +Observed gaps: + +1. `open_b2`, `open_container`, and `read_metadata` classify every remote-store + artifact as a directory. Directory counts include only array nodes, and + live recursive discovery also omits table leaves. +2. The non-DISK constructor does not allow a table root. Its post-operation + validation also replaces CTable metadata with `None`, which the current + upstream manifest validator rejects. +3. `ServerStoreTable` implements metadata and basic fetch only. Standalone + filtering, browser paging/sorting, and generic table dispatch do not share + that operation-scoped path. +4. Field fetch uses `blosc2.asarray(view[field][...])`; this requires a format + audit for nullable, variable-length, nested, and dictionary columns. +5. Cold export and warm-carrier cleanup consider only `caches`. Current + manifests also contain `batch_caches` and nested `linked` artifacts. +6. Array geometry validation does not by itself cover table batch sources, + dictionary vocabularies, index payloads, or secondary remote references. +7. The upstream cached-only `RemoteStore.read_cached` primitive addresses + array leaves. Table fetch has no equivalent callback before quota admission. + +## 1. Recognize and open table references safely + +Add one shared dispatch decision based on the validated manifest root: +`manifest['nodes'][manifest['source'].get('dataset', '')][0]`. + +- A table root returns the operation-scoped table adapter at relative key `''`. + A group root returns the existing store adapter. Keep root-array handling + explicit and consistent with RemoteArray behavior. +- Apply the decision to metadata, fetch/filter, mountability, browser opening, + downloads/publication, and embedded-reference checks. A standalone table + appears as a table leaf; internal column/index files are not browsable datasets. +- Count and discover both array and table leaves in group listings. Support a + reference selecting a nested table, not just a source whose table is at root. +- Preserve table and linked-reference metadata when validating runtime state. +- Use a supported upstream attachment/operation interface for standalone and + nested tables, reusing the common RemoteStore cache machinery. Improve that + interface upstream where necessary so Caterva2 does not need private root + flags or additional private-owner manipulation to treat a table as a store. + Finalize the interface with the upstream changes before wiring dispatch. +- Verify NONE and effective no-retention MEMORY as well as managed DISK. + Keep table/column/view lifetimes inside each operation and materialize the + response before closing them. Never cache live remote views in web state. + +Primary files: `remote_store.py`, `srv_utils.py`, `server.py`, `sparse_cache.py`. + +## 2. Complete the policy boundary for all table data + +Perform this alongside dispatch, before enabling additional source paths. +Keep outbound access disabled by default and reuse `[server.remote_proxy]`. + +- Audit transport creation for B2Z tables, PyTables/HDF5 tables, column + RemoteArray carriers, linked remote references, indexes, dictionaries, + validity/deletion masks, and any HDF5 metadata sidecars. Every destination + must pass the allowlist, public-address validation, pinning, redirect, + timeout, and credential rules before I/O, including refresh and restoration. +- Use the upstream authorization/transport hooks completed above for secondary + sources. Do not allow an unrestricted fallback to `blosc2.open` or fsspec. +- Validate schemas, member paths, companions, index descriptors, archive + entries, nested manifests, and warm-cache geometry before using uploaded + state. Bound nested metadata in aggregate and guard reference cycles/depth. + Preserve upstream rejection of unsafe object serializers. +- Extend resource validation to table row/column counts and non-array payload + units. Keep existing array limits on physical column components and bound + batch/index metadata before allocation. Choose documented defaults from + representative supported fixtures, with explicit oversized-unit behavior. +- Apply server concurrency limits to the table itself. Upstream defaults are + 8 concurrent reads, an 8 MiB metadata buffer, and a 64 MiB row buffer; those + buffers are transport batching targets, not hard process-memory limits. +- Inspect persisted schema/discovery without outbound access where available; + fetching missing metadata still requires authorization. Policy failures must + produce consistent client errors through every entry point. + +Primary files: `remote_store.py`, `remote_proxy.py`, policy documentation/tests; +upstream `remote_store.py`, `remote_array.py`, `remote_ctable.py`, +`ctable_storage.py`, and `remote_batch.py` for the required policy hooks. + +## 3. Integrate complete table reads + +Use a single table operation to apply the query, select the requested row +window/columns, materialize, and serialize while its owner is alive. + +- Serve schema, row/column counts, attributes, user metadata, sizes, and + compression information using `CTableMetadata`, without downloading columns + just to identify the dataset. Define unavailable size values consistently. +- Match local CTable slicing, integer/negative/clipped/empty windows, filtering, + and supported field selection. Apply filters before selecting the result + window. Preserve the API's existing parameter validation. +- Exercise fixed-width scalars/vectors, fixed strings, nullable columns, UTF-8, + batch strings/lists, nested lists, structs/objects with safe serializers, + dictionaries, and deleted rows. Preserve schema and null semantics in cframes. +- Keep the existing NDArray field-response contract where representable. For + columns that cannot preserve their type/null semantics in an NDArray, define + an explicit one-column CTable response and update client decoding together; + do not silently coerce values through NumPy. Verify the corresponding local + CTable behavior so local and remote tables agree. +- Delegate filtering/index selection to upstream, including summary, ordered, + positional, membership, and imported PyTables indexes where supported. + Verify selective queries avoid unrelated payload reads and warm queries + reuse retained index data. Do not implement a second query engine. +- Return a clear unsupported-operation response for table-level compressed + chunk requests. A table is not one chunked array; only explicitly supported + physical array-column operations may use the array chunk contract. +- Map missing fields, invalid filters, unsupported column layouts, stale + generations, and source failures to existing API error conventions. + +Primary files: `remote_store.py`, fetch/filter/chunk routes in `server.py`, +`srv_utils.py`, and `client.py` where response decoding needs adjustment. + +## 4. Complete cache, quota, and artifact lifecycle + +Keep the existing per-store serialization, ledger, generation locks, recovery, +and export ownership. Extend concrete assumptions instead of adding another +cache manager or a separate table ledger kind. + +- Account for all retained column chunks, batch payloads, dictionary/index + data, HDF5 shared-record/source caches, and linked payloads under one owner + budget. Charge metadata and allocated filesystem storage to customer quota. +- Use the new upstream table cached-only operation so a warm table read + can succeed without reserving space for another fill. It must cover the + query's index, masks, and payload dependencies atomically. On a genuine miss + and denied retention, use the established no-retention fallback. +- Normalize cold manifests completely: clear array caches, batch caches, and + retained linked artifacts while keeping enough discovery/reference metadata + to reopen. Test a batch-only warm artifact explicitly. +- Restore validated warm state once, then remove its retained payload from the + portable carrier using the existing publication protocol. Do not repeatedly + resurrect evicted batch, index, or linked caches from that carrier. +- Verify trimming/recovery recognizes every table storage component and can + operate without network access. Exercise concurrent workers, interrupted + fills/exports, replacement/deletion, and restart reconciliation. +- Preserve source snapshots until explicit replacement/refresh. Integrate + table refresh with generation retirement; old views must become stale and + failed refresh must preserve the previous usable generation. First establish + the server refresh entry point, which is not currently exposed by these + routes, rather than accidentally refreshing on normal reads. +- Export warm/cold table references through the existing download/publication + lifecycle, including staging admission, response cleanup, metadata, and + mutability flags. Reopening must return RemoteCTable and expose neither + private runtime paths nor credentials. + +Primary files: `sparse_cache.py`, `remote_store.py`, `storage_quota.py` only if +needed, and download/publication routes in `server.py`. + +## 5. Browser, Python client, and peer compatibility + +- Route both standalone and nested remote tables through the existing CTable + grid: selected columns, row paging, ascending/descending sort, and the + existing table filter behavior. Execute remote work in the thread pool and + materialize only the displayed result window. Sorting/filtering may still + need wider upstream reads; avoid promising constant-cost queries. +- Extend the table adapter's operation interface for browser needs rather than + pretending it is a live CTable outside the owner lock. +- Ensure metadata selects `caterva2.Table`, and verify `slice`, `where`, + `rows`, `head`, and downloads for standalone and container-member paths. +- Verify peers consume the same table metadata and cframes, including formats + already handled by their pass-through path. Reuse existing peer caching; + this work does not replace the peer transport with RemoteCTable. +- Document reference creation/upload, supported B2Z and PyTables/HDF5 sources, + policy configuration, cache behavior, indexes, refresh, and warm/cold + downloads. Zarr remains a store/array source unless upstream adds tables. +- Set the release dependency floor to `blosc2>=4.14.0`; the current + `>=4.13.0.dev0` does not guarantee RemoteCTable or the required improvements. + +## Verification and delivery + +Extend existing pytest suites and fixtures; no new framework. Use the Python +executable in the `blosc2` conda environment. `conda run` currently emits an +activation error in this shell, while the environment's Python works directly. + +| Area | Required evidence | +| --- | --- | +| Dispatch | Standalone root and selected nested-table references report `ctable`; group listings include tables and hide internals. | +| Read parity | Local and remote table values, schemas, nulls, deleted rows, slices, filters, projections, and browser sorting agree. | +| Sources | B2Z root/nested tables and HDF5/PyTables tables; mixed array/table stores. | +| Policies | DISK plus effective no-retention NONE/MEMORY, including quota fallback. | +| Security | No unauthorized I/O through discovery, secondary carriers, indexes, linked references, restoration, or refresh. | +| Cache | Count upstream requests/bytes, verify index reuse and cross-process hits, and enforce one aggregate budget. | +| Lifecycle | Batch-only and mixed warm exports, cold reopen, trim, process death, replacement, refresh failure, and cleanup retain consistent accounting. | +| User surfaces | API, Python Table client, browser, and existing peer paths work for standalone and nested tables. | + +Start with `test_remote_store.py`, `test_remote_proxy.py`, `test_sparse_cache.py`, +`test_storage_quota.py`, `test_storage_quota_api.py`, and `test_ctable.py`; add +targeted peer cases in `test_peers.py`. Use upstream RemoteCTable fixtures as +references without duplicating its entire test suite. Run applicable pre-commit +hooks and the broader regression suite once integration is complete. + +Upstream acceptance must include denied secondary destinations with zero +unauthorized requests, hook propagation through restoration/refresh, batch +limit rejection, and cached-only hits/misses for indexed and batch-backed +queries across processes. Commit these regressions with their python-blosc2 +fixes and run the relevant upstream remote-array/store/table suites before +depending on the changes in Caterva2. + +Delivery order: + +1. Reproduce and fix the four upstream gaps, finalize supported integration + APIs, and commit the python-blosc2 changes with tests and documentation for + 4.14.0. Resolve further upstream gaps there as they are discovered. +2. Add Caterva2 regressions for dispatch, no-retention validation, and batch + cold export; fix those paths using the improved upstream APIs. +3. Complete table read/serialization semantics and metadata for all column types. +4. Complete quota, cached-only reads, warm/cold lifecycle, and refresh behavior. +5. Wire browser/client/peer parity, documentation, and the 4.14.0 dependency floor. +6. Run end-to-end and multiprocessing acceptance tests. Measure cold/warm + paging and indexed queries, upstream traffic, retained storage, and competing + workers. Change lock granularity only if these measurements justify it. + +Done means the matrix above passes for standalone and nested tables, with no +RemoteArray/RemoteStore regressions. Recognition alone is not completion. diff --git a/pyproject.toml b/pyproject.toml index 7a38f181..a4b1b922 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -39,7 +39,7 @@ classifiers = [ "Operating System :: Unix", ] dependencies = [ - "blosc2>=4.13.0.dev0", + "blosc2>=4.14.0", "httpx[http2]", "numpy", ] From 4be107cab774a1ad70271789ceb32237261717e6 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Thu, 24 Sep 2026 11:49:03 +0200 Subject: [PATCH 17/20] Add explicit remote reference refresh --- caterva2/services/remote_store.py | 19 +++++++++ caterva2/services/server.py | 30 ++++++++++++++ caterva2/tests/test_remote_store.py | 62 +++++++++++++++++++++++++++++ doc/utilities/cat2-server.md | 4 +- 4 files changed, 114 insertions(+), 1 deletion(-) diff --git a/caterva2/services/remote_store.py b/caterva2/services/remote_store.py index b4adc411..8a9e3505 100644 --- a/caterva2/services/remote_store.py +++ b/caterva2/services/remote_store.py @@ -3,7 +3,9 @@ from __future__ import annotations import copy +import io import math +import tempfile import zipfile from contextlib import contextmanager from pathlib import Path @@ -193,6 +195,23 @@ def operation(self, callback, *, cached=None): except ValueError as exc: raise fastapi.HTTPException(status_code=400, detail=str(exc)) from exc + def refreshed_bytes(self): + """Build a fresh cold descriptor before replacing the hosted reference.""" + with tempfile.TemporaryDirectory() as directory: + artifact = Path(directory) / "refreshed.b2z" + with self.open() as runtime: + runtime.refresh() + runtime.save(artifact, include_cache=False) + manifest = dict( + inspect(artifact), + cache_policy=self.manifest["cache_policy"], + max_cache_bytes=self.manifest["max_cache_bytes"], + ) + validate_manifest(manifest) + output = io.BytesIO() + cold_export(manifest, output) + return output.getvalue() + def _key(self, key): relative = key.strip("/") blosc2.remote_store.RemoteDiscovery._validate(relative) diff --git a/caterva2/services/server.py b/caterva2/services/server.py index cfb5663d..1fe135cd 100644 --- a/caterva2/services/server.py +++ b/caterva2/services/server.py @@ -2569,6 +2569,36 @@ async def upload_file( return str(path) +@app.post("/api/refresh/{path:path}") +async def refresh_remote_reference( + path: pathlib.Path, + user: db.User = Depends(current_active_user), +): + """Replace a RemoteStore or RemoteCTable reference with fresh source discovery.""" + from caterva2.services import remote_store + + abspath = get_writable_path(path, user) + manifest = remote_store.inspect(abspath) + if manifest is None: + srv_utils.raise_bad_request("The path is not a RemoteStore reference") + store = remote_store.ServerRemoteStore(abspath, manifest) + + def replace(): + expected = storage_quota.signature(abspath) + try: + data = store.refreshed_bytes() + except remote_proxy.RemoteArrayDenied as exc: + raise fastapi.HTTPException(status_code=403, detail=str(exc)) from exc + except ValueError as exc: + srv_utils.raise_bad_request(str(exc)) + except (OSError, zipfile.BadZipFile) as exc: + raise fastapi.HTTPException(status_code=502, detail=str(exc)) from exc + write_dataset(abspath, data, expected=expected, compare=True) + + await concurrency.run_in_threadpool(replace) + return str(path) + + @app.post("/api/load_from_url/{path:path}") async def load_from_url( path: pathlib.Path, diff --git a/caterva2/tests/test_remote_store.py b/caterva2/tests/test_remote_store.py index 1bbb13ad..f4409226 100644 --- a/caterva2/tests/test_remote_store.py +++ b/caterva2/tests/test_remote_store.py @@ -208,6 +208,68 @@ async def test_standalone_remote_ctable_api(table_runtime, monkeypatch): server.app.dependency_overrides.update(overrides) +@pytest.mark.asyncio +async def test_standalone_remote_ctable_refresh(table_runtime, monkeypatch): + import httpx + + from caterva2.services import server + + path = table_runtime.with_name("refreshable-table.b2z") + manifest = dict(remote_store.inspect(table_runtime), cache_policy="disk", max_cache_bytes=1 << 20) + remote_store.cold_export(manifest, path) + root = path.parents[1] + monkeypatch.setattr(server.settings, "statedir", root) + monkeypatch.setattr(server.settings, "public", path.parent) + monkeypatch.setattr(server.settings, "shared", root / "shared") + monkeypatch.setattr(server.settings, "personal", root / "personal") + overrides = dict(server.app.dependency_overrides) + server.app.dependency_overrides[server.current_active_user] = lambda: object() + server.app.dependency_overrides[server.optional_user] = lambda: None + endpoint = "/api/fetch/@public/refreshable-table.b2z" + try: + async with httpx.AsyncClient( + transport=httpx.ASGITransport(app=server.app), base_url="http://test" + ) as client: + response = await client.get(endpoint) + assert response.status_code == 200, response.text + assert list(blosc2.ctable_from_cframe(response.content).x[:]) == [1, 2] + with server.quota_coordinator().connect() as db: + previous = db.execute( + "SELECT generation_id FROM remote_generations WHERE state='active'" + ).fetchone()[0] + + @dataclasses.dataclass + class Row: + x: int = blosc2.field(blosc2.int64()) + text: str = blosc2.field(blosc2.vlstring(batch_rows=2)) + + source = table_runtime.parents[2] / "table.b2z" + blosc2.CTable(Row, [(3, "three")], create_summary_index=False).to_b2z(source, overwrite=True) + fsspec.filesystem("memory").pipe_file(manifest["source"]["urlpath"], source.read_bytes()) + response = await client.post("/api/refresh/@public/refreshable-table.b2z") + assert response.status_code == 200, response.text + response = await client.get(endpoint) + assert response.status_code == 200, response.text + assert list(blosc2.ctable_from_cframe(response.content).x[:]) == [3] + with server.quota_coordinator().connect() as db: + current = db.execute( + "SELECT generation_id FROM remote_generations WHERE state='active'" + ).fetchone()[0] + old_state = db.execute( + "SELECT state FROM remote_generations WHERE generation_id=?", (previous,) + ).fetchone() + assert current != previous + assert old_state is None or old_state[0] == "retired" + before = path.read_bytes() + fsspec.filesystem("memory").pipe_file(manifest["source"]["urlpath"], b"invalid B2Z") + response = await client.post("/api/refresh/@public/refreshable-table.b2z") + assert response.status_code in {400, 502}, response.text + assert path.read_bytes() == before + finally: + server.app.dependency_overrides.clear() + server.app.dependency_overrides.update(overrides) + + @pytest.fixture def store_runtime(tmp_path, monkeypatch): monkeypatch.setenv("CATERVA2_SECRET", "test-secret") diff --git a/doc/utilities/cat2-server.md b/doc/utilities/cat2-server.md index 0d6e78c1..5274e7d4 100644 --- a/doc/utilities/cat2-server.md +++ b/doc/utilities/cat2-server.md @@ -130,7 +130,9 @@ generations are retired by maintenance without resolving their sources. Sources are immutable until a new reference is published. To refresh a hosted store, refresh it in python-blosc2, save a new archive, and upload that archive -as a replacement. Replacement/deletion retires the previous private generation. +as a replacement. `POST /api/refresh/{path}` performs fresh discovery for an +authenticated writer and atomically replaces the cold reference; a failed refresh +leaves the old reference usable. Replacement/deletion retires the previous private generation. Restart workers together when upgrading: storage schema version 3 adds store generation accounting and prevents workers from using a partly initialized ledger. From c6eb0e7a343860a08fb6b3d50129eca67a1dc162 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Thu, 24 Sep 2026 12:00:02 +0200 Subject: [PATCH 18/20] Plan RemoteCTable API consistency follow-ups --- plans/remote-ctable.md | 140 +++++++++++++++++++++++++++++++++++++++++ 1 file changed, 140 insertions(+) diff --git a/plans/remote-ctable.md b/plans/remote-ctable.md index 941ac2eb..3422f0e3 100644 --- a/plans/remote-ctable.md +++ b/plans/remote-ctable.md @@ -1,5 +1,145 @@ # Full RemoteCTable integration +## Implementation status and API consistency review (2026-09-24) + +The first implementation is committed in Caterva2 as `dfd5528` and `4be107c`, +with python-blosc2 improvements in `5f92a80d` and `d6554ad5`. It covers root, +nested, and linked tables, table browsing and queries, secondary-source policy +hooks, batch/linked cache migration and trimming, and explicit reference refresh. +The dependency floor is now `blosc2>=4.14.0`. + +The last implementation runs reported 437 passed / 173 skipped in Caterva2 and +487 passed / 5 skipped in the relevant upstream suites, with pre-commit passing. +These are historical results, not a fresh run for this documentation review. +They establish a useful integration baseline, but do not establish every item +in the original acceptance matrix. The implementation needs the following +follow-up work before claiming complete API parity or release readiness. + +The findings below come from current source inspection and small local probes. +The original sections below remain the implementation specification; the +decisions here supersede conflicting field-response guidance there. + +### P0: Make refresh authorization and replacement consistent with other writers + +**Observed:** `server.current_active_user` becomes a dependency returning `None` +when login is disabled. Upload and remove explicitly reject a missing user; +`refresh_remote_reference()` does not. `get_writable_path()` permits the public +root with that value, so refresh can proceed anonymously in this configuration. + +- Add the same explicit authentication guard used by existing writers, before + inspecting the artifact or performing outbound I/O. Test both enabled and + disabled login configurations and assert denied calls leave source traffic, + carrier bytes, and quota state unchanged. +- Capture the inspected descriptor and its replacement signature from the same + snapshot. Currently inspection happens before `replace()` obtains a new + signature; a replacement in between can let stale discovery overwrite the + newer reference. Reuse `StorageQuota.snapshot()` and the existing publication + comparison rather than introducing another locking protocol. A racing write + must result in the existing 409 response and preserve the winner's bytes. +- Keep successful refresh atomic and preserve the existing reference on failed + discovery, policy denial, quota denial, or publication conflict. The current + failure test checks carrier bytes; also check generation and accounting state. + +### P1: Define one table projection and decoding contract + +**Observed:** `ServerStoreTable.fetch(field=...)` always returns a one-column +CTable. Local table field requests instead execute `container[field]` and enter +the general array/SChunk dispatch. A fixed-width local field is a `Column`, not +an NDArray, SChunk, or CTable, so that route is not equivalent to remote fetch. +The client also chooses its decoder from a `Table` instance or a `.b2z` suffix; +a string such as `store.b2z/table` selects the array decoder. + +- Make every table fetch, including a single-column projection, return a CTable + cframe for local and remote tables. This preserves nulls, dictionaries, nested + values, and schema, and avoids a dtype-dependent response type. Structured + NDArray field fetches keep their NDArray contract. Document the correction to + the previous local-table behavior rather than claiming backward compatibility. +- Resolve the response kind from dataset metadata for string paths; reuse + metadata already held by `Table` objects. Remove suffix-based table guessing. + Do not try unrelated binary decoders until one happens to succeed. +- Fix `Client.get_slice(..., key=..., field=...)`: it currently sends only + `field`, dropping `key` (also described as ignored in its docstring). Table + projection must preserve the requested row window; update documentation and + test bounded transfers through the real Python client. +- Allow table filtering, row windows, and projection together, in that order. + The HTTP route currently rejects `filter` plus `field`, although the remote + adapter can apply both. Use the same table pipeline for GET and POST fetch, + local and remote sources. Keep unrelated array rules explicit. +- Validate with standalone and nested paths, both `Table` objects and strings, + fixed-width and nullable/batch/dictionary columns, and empty projections by + row range. Assert decoded type, schema, nulls, and values, not only HTTP 200. + +### P1: Normalize table input validation and error responses + +**Observed:** `parse_segment()` accepts slice steps, but `ctable_row_range()` +ignores them and ignores extra tuple dimensions. The Python client rejects +non-unit steps, so raw HTTP and client calls disagree. Error handling also +differs: the remote adapter translates some filter errors and missing fields, +while local fields and filters take other branches; source errors are mapped +to 502 in refresh but not consistently in table fetch. + +- For this release, reject non-unit steps (including zero and reverse steps) + and extra table dimensions with 400 in the shared row-range helper. Preserve + the established negative-index and clipped-window behavior. Full stride + support can be a separate feature; do not silently widen selections. +- Validate field names, filter expressions, and sort columns through the same + table operation path. Return 400 for invalid query parameters, 404 for a + missing dataset/member, 403 for policy denial, 409 for generation/publication + conflicts, and 502 for recognized upstream transport or malformed-source + failures. Preserve existing quota/database handlers. Do not turn arbitrary + programming exceptions into client errors. +- Add parity cases for GET/POST fetch and browser rendering, including unknown + fields, malformed filters, unsupported selections, unavailable sources, and + stale references. Browser errors should remain understandable HTML responses. + +### P2: Complete refresh and query semantics in the client and adapter + +- Add `Client.refresh(path)` for the existing reference-refresh endpoint and + document that it accepts a hosted RemoteStore/RemoteCTable carrier. A path + inside a store must explicitly identify the owning carrier to refresh; do not + silently refresh siblings. RemoteArray refresh is not implemented by this + endpoint and must receive a clear unsupported-target response. A missing + path should receive 404, rather than the current generic 400. +- Define how callers reload cached `Table`/`Group` metadata after refresh; return + the refreshed dataset object from the convenience method or explicitly reload + its metadata. Avoid leaving `nrows` and schema silently stale. +- The internal adapter's `where()` replaces any previous filter, and + `sort_by(..., view=False)` ignores `view`. Either implement the semantics of + the CTable methods it imitates or narrow the internal interface to the browser + operations actually needed. Do not expose misleading method compatibility. +- Avoid repeated execution: adapter construction reads metadata, `where()` and + `sort_by()` execute queries to obtain metadata, and `slice()` executes them + again. Reuse one operation to obtain the requested page and required counts. + Move all blocking remote opens in fetch to the thread pool; currently the + ordinary unfiltered root/member branches can still open on the event loop. + +### P2: Finish the upstream integration contract and acceptance evidence + +- Before 4.14.0, finalize supported python-blosc2 APIs for source authorization, + batch/manifest validation, operation locking, and cold descriptor export. + Caterva2 still uses underscored hooks, `runtime._owner`, and manifest/ZIP + construction. Improve upstream APIs directly and then remove that dependency + on private state; do not add Caterva2 compatibility wrappers around it. +- Audit the table-level concurrency limit, including batches and index reads; + setting `max_concurrency` on physical array sources alone does not prove the + whole table obeys server policy. +- Reproduce the exploratory HDF5 indexed-query case that refetched data after + quota admission was suspended, even following warm reads. That assertion was + not retained in the final suite. Determine whether the probe misses retained + dependencies or the warm-up never retained them; count actual source bytes + and bulk reads as well as `get_chunk()` calls. Fix cache behavior upstream if + needed, and retain a regression for an established fully cached indexed query. +- Complete explicit remote-table Python-client and peer cases, process-shared + indexed/batch cached-only reads, and interrupted refresh/export recovery. + Record which skip conditions leave acceptance items untested. Measure cold + and warm paging/query traffic and retained storage with competing workers + before considering changes to lock granularity. + +Implement P0 first, then projection/decoding and validation together, followed +by client conveniences and the remaining upstream/acceptance work. Extend the +existing tests; keep fixes in the repository that owns the behavior. This +review updates the plan only and does not implement these follow-up changes. + ## Scope and baseline Extend the RemoteArray/RemoteStore integration described in From b0797dc05dcbda5e3bc58f1beb4bf0646b22dd29 Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Thu, 24 Sep 2026 12:15:01 +0200 Subject: [PATCH 19/20] Align RemoteCTable fetch and refresh APIs --- caterva2/client.py | 79 ++++++++++----------- caterva2/services/remote_store.py | 7 +- caterva2/services/server.py | 86 ++++++++++++++++++----- caterva2/services/srv_utils.py | 10 ++- caterva2/tests/test_ctable.py | 57 +++++++++++++++ caterva2/tests/test_remote_store.py | 104 ++++++++++++++++++++++++++++ plans/remote-ctable.md | 45 ++++++++++-- 7 files changed, 319 insertions(+), 69 deletions(-) diff --git a/caterva2/client.py b/caterva2/client.py index 277eb354..29d6c14e 100644 --- a/caterva2/client.py +++ b/caterva2/client.py @@ -593,14 +593,13 @@ def slice( provided, each slice will be applied to the corresponding dimension. as_blosc2 : bool - If True (default), the result will be returned as a Blosc2 object - (either a `SChunk` or `NDArray`). If False, it will be returned - as a NumPy array (equivalent to `self[key]`). + If True (default), return a Blosc2 object, including a CTable for + table requests. If False, return NumPy data or table row tuples. Returns ------- - NDArray or SChunk or numpy.ndarray - A new Blosc2 object containing the requested slice. + NDArray or SChunk or CTable or numpy.ndarray or list + The requested slice; table requests return a CTable or row tuples. Examples -------- @@ -1353,10 +1352,10 @@ def get_slice(self, path, key=None, as_blosc2=True, field=None, ndim=None): dimension. If str, is interpreted as filter. as_blosc2 : bool If True (default), the result will be returned as a Blosc2 object - (either a `SChunk` or `NDArray`). If False, it will be returned - as a NumPy array (equivalent to `self[key]`). + (including a CTable for table requests). If False, table rows are + returned as tuples, and other datasets as NumPy data. field: str - Shortcut to access a field in a structured array. If provided, `key` is ignored. + Select one field or table column after applying `key`. ndim: int How many dimensions the dataset has, where the caller knows. Only an `Ellipsis` in the key needs it, and only one that is not its last @@ -1364,8 +1363,8 @@ def get_slice(self, path, key=None, as_blosc2=True, field=None, ndim=None): Returns ------- - NDArray or SChunk or numpy.ndarray - A new Blosc2 object containing the requested slice. + NDArray or SChunk or CTable or numpy.ndarray or list + The requested slice or table rows. Examples -------- @@ -1385,35 +1384,16 @@ def get_slice(self, path, key=None, as_blosc2=True, field=None, ndim=None): if isinstance(path, Table): kind = "ctable" elif isinstance(path, File): - kind = None + kind = path.meta.get("kind") else: path_str = path.as_posix() if hasattr(path, "as_posix") else str(path) - kind = "ctable" if path_str.endswith(".b2z") else None + kind = self.get_info(path_str).get("kind") if isinstance(path, File): path = path.path urlbase, path = _format_paths(self.urlbase, path) - if field: # blosc2 doesn't support indexing of multiple fields - return self._fetch_data( - path, - urlbase, - {"field": field}, - auth_cookie=self.cookie, - as_blosc2=as_blosc2, - timeout=self.timeout, - kind=kind, - ) if isinstance(key, str): # The key can still be a slice expression in string format (like for CLI utils) params = {"slice_": key} if _looks_like_slice(key) else {"filter": key} - return self._fetch_data( - path, - urlbase, - params=params, - auth_cookie=self.cookie, - as_blosc2=as_blosc2, - timeout=self.timeout, - kind=kind, - ) else: # Coordinates go over as `indices` and are gathered by the server; # a plain box is a `slice_`, which says the same thing more cheaply @@ -1421,16 +1401,19 @@ def get_slice(self, path, key=None, as_blosc2=True, field=None, ndim=None): params = ( {"slice_": api_utils.slice_to_string(key, ndim)} if indices is None else {"indices": indices} ) - # Fetch and return the data as a Blosc2 object / NumPy array - return self._fetch_data( - path, - urlbase, - params, - auth_cookie=self.cookie, - as_blosc2=as_blosc2, - timeout=self.timeout, - kind=kind, - ) + if field is not None: + if "indices" in params: + raise IndexError("field cannot be combined with coordinate indices") + params["field"] = field + return self._fetch_data( + path, + urlbase, + params, + auth_cookie=self.cookie, + as_blosc2=as_blosc2, + timeout=self.timeout, + kind=kind, + ) def get_chunk(self, path, nchunk): """ @@ -1932,6 +1915,20 @@ def unfold(self, remotepath): ) return PurePosixPath(result) # return path to top directory of dset + def refresh(self, path): + """Refresh a hosted RemoteStore or RemoteCTable carrier and return a fresh object. + + Pass the carrier's ``.b2z`` path, including for a table inside a store. + This endpoint does not refresh RemoteArray references. + """ + if isinstance(path, File): + path = path.path + _, formatted = _format_paths(self.urlbase, path) + if pathlib.PurePosixPath(formatted).suffix != ".b2z": + raise ValueError("Refresh requires a hosted RemoteStore or RemoteCTable .b2z carrier") + self._post(f"{self.urlbase}/api/refresh/{formatted}", auth_cookie=self.cookie, timeout=self.timeout) + return self.get(formatted) + def remove(self, path): """ Removes a dataset or the contents of a directory from a remote repository. diff --git a/caterva2/services/remote_store.py b/caterva2/services/remote_store.py index 8a9e3505..47fe691b 100644 --- a/caterva2/services/remote_store.py +++ b/caterva2/services/remote_store.py @@ -194,6 +194,8 @@ def operation(self, callback, *, cached=None): raise fastapi.HTTPException(status_code=403, detail=str(exc)) from exc except ValueError as exc: raise fastapi.HTTPException(status_code=400, detail=str(exc)) from exc + except (OSError, zipfile.BadZipFile) as exc: + raise fastapi.HTTPException(status_code=502, detail=str(exc)) from exc def refreshed_bytes(self): """Build a fresh cold descriptor before replacing the hosted reference.""" @@ -398,9 +400,12 @@ def schema_dict(self): return self.metadata["schema_dict"] def where(self, expression): - return type(self)(self.store, self.key, filter=expression, sortby=self.sortby) + filter = f"({self.filter}) & ({expression})" if self.filter else expression + return type(self)(self.store, self.key, filter=filter, sortby=self.sortby) def sort_by(self, column, *, view=False): + if not view: + raise NotImplementedError("ServerStoreTable supports only sort_by(..., view=True)") return type(self)(self.store, self.key, filter=self.filter, sortby=column) def slice(self, start, stop): diff --git a/caterva2/services/server.py b/caterva2/services/server.py index 1fe135cd..5950c162 100644 --- a/caterva2/services/server.py +++ b/caterva2/services/server.py @@ -24,6 +24,7 @@ import sqlite3 import string import tarfile +import tempfile import threading import time import traceback @@ -201,7 +202,7 @@ def quota_coordinator(): def write_dataset(path, data, *, expected=None, compare=False): - """Store final encoded bytes through shared admission when quota is enabled.""" + """Store final encoded bytes, comparing the expected generation when requested.""" path = pathlib.Path(path) quota = quota_coordinator() if quota is not None: @@ -210,7 +211,26 @@ def write_dataset(path, data, *, expected=None, compare=False): quota.publish(path, data, expected=expected) else: path.parent.mkdir(parents=True, exist_ok=True) - path.write_bytes(data) + if not compare: + path.write_bytes(data) + else: + with dataset_thread_lock(path): + if storage_quota.signature(path) != expected: + raise storage_quota.StorageBusy("dataset changed while preparing its replacement") + candidate = None + try: + with tempfile.NamedTemporaryFile( + dir=path.parent, prefix=f".{path.name}.", delete=False + ) as file: + candidate = pathlib.Path(file.name) + file.write(data) + file.flush() + os.fsync(file.fileno()) + os.replace(candidate, path) + storage_quota.sync_directory(path.parent) + finally: + if candidate is not None: + candidate.unlink(missing_ok=True) def remove_dataset(path): @@ -1120,8 +1140,8 @@ async def fetch_data( field : str The desired field of dataset. - The field and filter parameters are incompatible, if both are giving the API will - return a "400 Bad Request" error response. + For CTables, filtering is applied before a row slice and field projection. + Other dataset types may reject a combination of field and filter. Returns ------- @@ -1212,8 +1232,6 @@ async def fetch_data( filter = filter.strip() if filter else filter store_table_filter = None if filter: - if field: - srv_utils.raise_bad_request("Cannot handle both field and filter parameters at the same time") mtime = abspath.stat().st_mtime try: from caterva2.services.remote_store import ServerStoreTable @@ -1235,23 +1253,29 @@ async def fetch_data( ) except ValueError as exc: srv_utils.raise_bad_request(str(exc)) + if field and not isinstance(container, blosc2.CTable | ServerStoreTable): + srv_utils.raise_bad_request("Cannot handle both field and filter parameters at the same time") elif inner_key is not None: # A member inside a container (e.g. a TreeStore .b2z or .h5 leaf). # A leaf that is a whole frame inside the container can be served in # ranges, by seeking to it -- what a stored dataset gets from # FileResponse, and what lets a client read its blocks. Not when a # field is projected out of it: that is computed, not stored. - window = member_window(abspath, inner_key, abspath.stat().st_mtime) - container = srv_utils.open_container_member(abspath, inner_key) + window, container = await concurrency.run_in_threadpool( + lambda: ( + member_window(abspath, inner_key, abspath.stat().st_mtime), + srv_utils.open_container_member(abspath, inner_key), + ) + ) if container is None: srv_utils.raise_not_found() else: - container = open_b2(abspath, path) + container = await concurrency.run_in_threadpool(lambda: open_b2(abspath, path)) from caterva2.services.remote_store import ServerStoreTable store_table_field = field if isinstance(container, ServerStoreTable) else None - if field and store_table_field is None: + if field and not isinstance(container, blosc2.CTable | ServerStoreTable): container = container[field] from caterva2.services.remote_store import ServerRemoteStore @@ -1372,9 +1396,20 @@ async def fetch_data( lambda: array.fetch(slice_, filter=store_table_filter, field=store_table_field) ) elif isinstance(array, blosc2.CTable): - row_start, row_stop = srv_utils.ctable_row_range(slice_, array.nrows) - view = array.slice(row_start, row_stop) - data = await concurrency.run_in_threadpool(view.to_cframe) + + def fetch_table(): + if field is not None and field not in { + column["name"] for column in array.schema_dict()["columns"] + }: + raise ValueError(f"Unknown table field: {field}") + row_start, row_stop = srv_utils.ctable_row_range(slice_, array.nrows) + view = array.select([field]) if field is not None else array + return view.slice(row_start, row_stop).to_cframe() + + try: + data = await concurrency.run_in_threadpool(fetch_table) + except (ValueError, NameError, SyntaxError) as exc: + srv_utils.raise_bad_request(str(exc)) elif isinstance(array, hdf5.HDF5Proxy): data = array.to_cframe(() if slice_ is None else slice_) elif isinstance(array, blosc2.LazyArray): @@ -2577,14 +2612,29 @@ async def refresh_remote_reference( """Replace a RemoteStore or RemoteCTable reference with fresh source discovery.""" from caterva2.services import remote_store + if not user: + raise srv_utils.raise_unauthorized("Refreshing files requires authentication") abspath = get_writable_path(path, user) - manifest = remote_store.inspect(abspath) - if manifest is None: - srv_utils.raise_bad_request("The path is not a RemoteStore reference") - store = remote_store.ServerRemoteStore(abspath, manifest) def replace(): - expected = storage_quota.signature(abspath) + quota = quota_coordinator() + if quota is not None: + snapshot, expected = quota.snapshot(abspath) + else: + with dataset_thread_lock(abspath): + expected = storage_quota.signature(abspath) + snapshot = abspath.read_bytes() if expected is not None else None + if snapshot is None: + srv_utils.raise_not_found() + with tempfile.TemporaryDirectory() as directory: + carrier = pathlib.Path(directory) / "reference.b2z" + carrier.write_bytes(snapshot) + manifest = remote_store.inspect(carrier) + if manifest is None: + if remote_proxy.inspect(abspath) is not None: + srv_utils.raise_bad_request("RemoteArray refresh is not supported") + srv_utils.raise_bad_request("The path is not a RemoteStore reference") + store = remote_store.ServerRemoteStore(abspath, manifest) try: data = store.refreshed_bytes() except remote_proxy.RemoteArrayDenied as exc: diff --git a/caterva2/services/srv_utils.py b/caterva2/services/srv_utils.py index 6b7f017c..6cd434b1 100644 --- a/caterva2/services/srv_utils.py +++ b/caterva2/services/srv_utils.py @@ -63,13 +63,19 @@ def split_container_path(path): def ctable_row_range(slice_, nrows): """Normalize an ``api/fetch`` ``slice_`` into a CTable row range - ``(start, stop)``: take the first (row) component, apply None defaults, + ``(start, stop)``: validate the row component, apply None defaults, negative wrap, and clamp to ``[0, nrows]``. Used by the local fetch branch and by peer providers, so both clamp identically.""" # slice_ is a single slice/int/tuple; extract row start/stop. # Use `is None` (not truthiness) so that stop == 0 stays 0. - sl0 = slice_[0] if isinstance(slice_, tuple) and len(slice_) > 0 else slice_ + if isinstance(slice_, tuple): + if len(slice_) > 1: + raise ValueError("CTable selections must have one row dimension") + slice_ = slice_[0] if slice_ else None + sl0 = slice_ if isinstance(sl0, slice): + if sl0.step not in (None, 1): + raise ValueError("CTable row slices support only step=1") row_start = 0 if sl0.start is None else sl0.start row_stop = nrows if sl0.stop is None else sl0.stop if row_start < 0: diff --git a/caterva2/tests/test_ctable.py b/caterva2/tests/test_ctable.py index 708036ca..6226fdf3 100644 --- a/caterva2/tests/test_ctable.py +++ b/caterva2/tests/test_ctable.py @@ -345,6 +345,63 @@ def test_client_table_class(fill_ctable_public): assert len(part) == 2 +def test_client_table_projection(fill_ctable_public, client): + fname, root = fill_ctable_public + path = f"{root.name}/{fname}" + for target in (path, root[fname]): + part = client.get_slice(target, slice(1, 3), field="y") + assert isinstance(part, blosc2.CTable) + assert [column["name"] for column in part.schema_dict()["columns"]] == ["y"] + assert list(part.y[:]) == ["v1", "v2"] + + filtered = client.get_slice(target, "x > 0", field="y") + assert isinstance(filtered, blosc2.CTable) + assert list(filtered.y[:]) == ["v1", "v2"] + assert client.get_slice(target, slice(1, 3), field="y", as_blosc2=False) == [("v1",), ("v2",)] + + empty = client.get_slice(path, slice(0, 0), field="y") + assert isinstance(empty, blosc2.CTable) + assert [column["name"] for column in empty.schema_dict()["columns"]] == ["y"] + assert len(empty) == 0 + + +def test_client_refresh_uses_carrier(monkeypatch): + client = cat2.Client("http://localhost:8000") + calls = [] + monkeypatch.setattr(client, "_post", lambda url, **kwargs: calls.append(url)) + monkeypatch.setattr(client, "get", lambda path: ("fresh", path)) + + assert client.refresh("@public/store.b2z") == ("fresh", "@public/store.b2z") + assert calls == ["http://localhost:8000/api/refresh/@public/store.b2z"] + with pytest.raises(ValueError, match="carrier"): + client.refresh("@public/store.b2z/table") + + +def test_client_nested_table_string_uses_metadata(tmp_path, monkeypatch): + path = tmp_path / "table.b2z" + _make_table(path, n=3) + with blosc2.open(path) as table: + cframe = table.slice(1, 3).to_cframe() + client = cat2.Client("http://localhost:8000") + nested = "@public/store.b2z/table" + looked_up = [] + requested = [] + monkeypatch.setattr(client, "get_info", lambda path: looked_up.append(path) or {"kind": "ctable"}) + monkeypatch.setattr( + client, + "_xget", + lambda url, **kwargs: ( + requested.append((url, kwargs["params"])) or httpx.Response(200, content=cframe) + ), + ) + + result = client.get_slice(nested, slice(1, 3), field="y") + assert isinstance(result, blosc2.CTable) + assert list(result.y[:]) == ["v1", "v2"] + assert looked_up == [nested] + assert requested == [(f"http://localhost:8000/api/fetch/{nested}", {"slice_": "1:3", "field": "y"})] + + # --------------------------------------------------------------------------- # CLI: info / show for .b2z # --------------------------------------------------------------------------- diff --git a/caterva2/tests/test_remote_store.py b/caterva2/tests/test_remote_store.py index f4409226..55074671 100644 --- a/caterva2/tests/test_remote_store.py +++ b/caterva2/tests/test_remote_store.py @@ -55,6 +55,53 @@ def test_standalone_remote_ctable_dispatch(table_runtime): assert list(result.text[:]) == ["one", "two"] +def test_standalone_remote_ctable_view_methods(table_runtime): + from caterva2.services import server + + table = server.open_b2(table_runtime, "@public/table-reference.b2z") + narrowed = table.where("x > 0").where("x < 2") + assert list(blosc2.ctable_from_cframe(narrowed.fetch()).x[:]) == [1] + with pytest.raises(NotImplementedError, match="view=True"): + table.sort_by("x") + + +def test_standalone_remote_ctable_unavailable_source(table_runtime): + import fastapi + + from caterva2.services import server + + table = server.open_b2(table_runtime, "@public/table-reference.b2z") + source = remote_store.inspect(table_runtime)["source"]["urlpath"] + fsspec.filesystem("memory").rm(source) + with pytest.raises(fastapi.HTTPException) as exc: + table.fetch() + assert exc.value.status_code == 502 + + +def test_refresh_no_quota_replacement_is_atomic(tmp_path, monkeypatch): + from caterva2.services import server + + path = tmp_path / "reference.b2z" + path.write_bytes(b"old") + expected = storage_quota.signature(path) + monkeypatch.setattr(server, "quota_coordinator", lambda: None) + + def fail_replace(*args): + raise OSError("interrupted replacement") + + with monkeypatch.context() as patch: + patch.setattr(server.os, "replace", fail_replace) + with pytest.raises(OSError, match="interrupted"): + server.write_dataset(path, b"new", expected=expected, compare=True) + assert path.read_bytes() == b"old" + assert sorted(tmp_path.iterdir()) == [path] + + path.write_bytes(b"winner") + with pytest.raises(storage_quota.StorageBusy): + server.write_dataset(path, b"new", expected=expected, compare=True) + assert path.read_bytes() == b"winner" + + def test_standalone_remote_ctable_disk_cache(table_runtime, monkeypatch): from caterva2.services import server @@ -182,6 +229,37 @@ async def test_standalone_remote_ctable_api(table_runtime, monkeypatch): response = await client.get("/api/fetch/@public/table-reference.b2z", params={"field": "text"}) assert response.status_code == 200, response.text assert list(blosc2.ctable_from_cframe(response.content).text[:]) == ["one", "two"] + response = await client.get( + "/api/fetch/@public/table-reference.b2z", params={"field": "text", "slice_": "1:2"} + ) + assert response.status_code == 200, response.text + assert list(blosc2.ctable_from_cframe(response.content).text[:]) == ["two"] + response = await client.get( + "/api/fetch/@public/table-reference.b2z", params={"field": "text", "filter": "x > 1"} + ) + assert response.status_code == 200, response.text + assert list(blosc2.ctable_from_cframe(response.content).text[:]) == ["two"] + response = await client.post( + "/api/fetch/@public/table-reference.b2z", + json={"field": "text", "filter": "x > 1", "slice_": "0:1"}, + ) + assert response.status_code == 200, response.text + assert list(blosc2.ctable_from_cframe(response.content).text[:]) == ["two"] + for selection in ("0:2:2", "0:2:0", "0:1,0:1"): + response = await client.get( + "/api/fetch/@public/table-reference.b2z", params={"slice_": selection} + ) + assert response.status_code == 400, response.text + local = path.with_name("local-table.b2z") + local.write_bytes((path.parents[2] / "table.b2z").read_bytes()) + response = await client.get( + "/api/fetch/@public/local-table.b2z", + params={"filter": "x > 1", "field": "text", "slice_": "0:1"}, + ) + assert response.status_code == 200, response.text + assert list(blosc2.ctable_from_cframe(response.content).text[:]) == ["two"] + response = await client.get("/api/fetch/@public/local-table.b2z", params={"field": "missing"}) + assert response.status_code == 400, response.text response = await client.get( "/api/fetch/@public/table-reference.b2z", params={"field": "missing"} ) @@ -230,6 +308,8 @@ async def test_standalone_remote_ctable_refresh(table_runtime, monkeypatch): async with httpx.AsyncClient( transport=httpx.ASGITransport(app=server.app), base_url="http://test" ) as client: + response = await client.post("/api/refresh/@public/missing.b2z") + assert response.status_code == 404, response.text response = await client.get(endpoint) assert response.status_code == 200, response.text assert list(blosc2.ctable_from_cframe(response.content).x[:]) == [1, 2] @@ -265,6 +345,30 @@ class Row: response = await client.post("/api/refresh/@public/refreshable-table.b2z") assert response.status_code in {400, 502}, response.text assert path.read_bytes() == before + server.app.dependency_overrides[server.current_active_user] = lambda: None + response = await client.post("/api/refresh/@public/refreshable-table.b2z") + assert response.status_code == 401, response.text + assert path.read_bytes() == before + with server.quota_coordinator().connect() as db: + assert ( + db.execute( + "SELECT generation_id FROM remote_generations WHERE state='active'" + ).fetchone()[0] + == current + ) + server.app.dependency_overrides[server.current_active_user] = lambda: object() + + def racing_refresh(store): + server.quota_coordinator().publish( + path, before + b"replacement", expected=storage_quota.signature(path) + ) + return before + + with monkeypatch.context() as patch: + patch.setattr(remote_store.ServerRemoteStore, "refreshed_bytes", racing_refresh) + response = await client.post("/api/refresh/@public/refreshable-table.b2z") + assert response.status_code == 409, response.text + assert path.read_bytes() == before + b"replacement" finally: server.app.dependency_overrides.clear() server.app.dependency_overrides.update(overrides) diff --git a/plans/remote-ctable.md b/plans/remote-ctable.md index 3422f0e3..31d71903 100644 --- a/plans/remote-ctable.md +++ b/plans/remote-ctable.md @@ -1,5 +1,38 @@ # Full RemoteCTable integration +## API consistency follow-up (2026-09-24) + +The follow-up implementation now rejects anonymous refresh, snapshots the +reference and its expected generation together, and returns 409 if a competing +replacement wins. Quota-free refresh also stages an atomic replacement and +compares the expected generation. Missing references return 404; RemoteArray targets receive +an explicit unsupported-target error. `Client.refresh()` requires the owning +`.b2z` carrier and returns a newly loaded object, so callers can replace stale +metadata after a successful refresh. + +Local and remote table field requests now return one-column CTable cframes. +The Python client chooses the decoder from dataset metadata, including for +strings naming tables inside stores, and retains slices or filters when a field +is requested. Both GET and POST fetch support filter plus field for tables. +The shared row-range helper rejects strides and extra dimensions; upstream +source I/O failures become 502. The internal remote-table adapter retains +chained filters and explicitly supports only view sorting. Ordinary remote +container opens in fetch now run in the thread pool. + +Regression coverage includes client projection and nested-path decoding, +GET/POST projection, local/remote table errors, unavailable sources, refresh +authorization, and a competing replacement. The client returns a new object; +existing `Table`/`Group` instances are not updated in place. +The full Caterva2 suite passed with 443 passed and 173 skipped, and pre-commit +passed for the changed files. + +Release validation still needs the broader P2 work below: public upstream +inspection and lifecycle APIs in python-blosc2, the table-level concurrency +audit, the indexed HDF5 cached-only reproduction, and the remaining process, +peer, interrupted-operation, and traffic measurements. Fix any upstream gaps +in python-blosc2 before 4.14.0 instead of adding Caterva2 workarounds. The +API consistency regressions do not establish these remaining acceptance items. + ## Implementation status and API consistency review (2026-09-24) The first implementation is committed in Caterva2 as `dfd5528` and `4be107c`, @@ -15,9 +48,9 @@ They establish a useful integration baseline, but do not establish every item in the original acceptance matrix. The implementation needs the following follow-up work before claiming complete API parity or release readiness. -The findings below come from current source inspection and small local probes. -The original sections below remain the implementation specification; the -decisions here supersede conflicting field-response guidance there. +The findings below record the review that led to the follow-up above. The +original sections below remain the implementation specification; the decisions +here supersede conflicting field-response guidance there. ### P0: Make refresh authorization and replacement consistent with other writers @@ -135,10 +168,8 @@ to 502 in refresh but not consistently in table fetch. and warm paging/query traffic and retained storage with competing workers before considering changes to lock granularity. -Implement P0 first, then projection/decoding and validation together, followed -by client conveniences and the remaining upstream/acceptance work. Extend the -existing tests; keep fixes in the repository that owns the behavior. This -review updates the plan only and does not implement these follow-up changes. +Implement the remaining upstream and acceptance work in the repository that +owns each behavior. Keep the client and server regressions above in Caterva2. ## Scope and baseline From 69bb51dfbc71f3327e46b35d86ac59965cd728ff Mon Sep 17 00:00:00 2001 From: Francesc Alted Date: Fri, 25 Sep 2026 20:14:00 +0200 Subject: [PATCH 20/20] Support hosted Parquet stores and raw Parquet uploads --- caterva2/services/remote_store.py | 30 ++++- caterva2/services/server.py | 56 +++++----- caterva2/services/sparse_cache.py | 5 +- caterva2/services/srv_utils.py | 5 + caterva2/tests/test_remote_store.py | 135 +++++++++++++++++++++++ caterva2/tests/test_storage_quota_api.py | 64 +++++++++++ doc/utilities/cat2-server.md | 21 +++- pyproject.toml | 6 +- 8 files changed, 286 insertions(+), 36 deletions(-) diff --git a/caterva2/services/remote_store.py b/caterva2/services/remote_store.py index 47fe691b..0ecaf644 100644 --- a/caterva2/services/remote_store.py +++ b/caterva2/services/remote_store.py @@ -47,7 +47,12 @@ def inspect(path): def validate_manifest(manifest): blosc2.RemoteStore._validate_artifact_manifest(manifest) - if set(manifest["source"]) != {"urlpath", "dataset", "kind"}: + allowed_source = ( + {"urlpath", "dataset", "kind", "options"} + if manifest["source"]["kind"] == "parquet" + else {"urlpath", "dataset", "kind"} + ) + if set(manifest["source"]) != allowed_source: raise remote_proxy.RemoteArrayDenied("RemoteStore source contains unsupported fields") if len(msgpack_packb(manifest)) > remote_proxy.policy.max_metadata_bytes: raise remote_proxy.RemoteArrayDenied("RemoteStore metadata exceeds the configured limit") @@ -118,7 +123,7 @@ def filesystem_for_url(url): def cold_export(manifest, destination): """Write a descriptor-only archive without resolving its source.""" - manifest = dict(manifest, caches=[], batch_caches=[], linked={}) + manifest = dict(manifest, caches=[], batch_caches=[], parquet_caches=[], linked={}) storage = blosc2.Storage(contiguous=True) storage.meta = {"b2tree": {"version": 1}, "b2remote_store": {"version": 1}} embed = blosc2.SChunk(chunksize=8192, data=None, storage=storage) @@ -149,6 +154,7 @@ def authorized_filesystem(url): source = self.manifest["source"] fs = authorized_filesystem(source["urlpath"]) options = { + "_source_format": source["kind"], "_filesystem": fs, "_filesystem_resolver": authorized_filesystem, "_source_validator": validate_array, @@ -156,6 +162,16 @@ def authorized_filesystem(url): "_manifest_validator": validate_manifest, "_max_nodes": remote_proxy.policy.max_nodes, } + if source["kind"] == "parquet": + from blosc2.ctable import NullPolicy + from blosc2.remote_parquet import _restore_compression_options + + conversion = dict(self.manifest["metadata"]["parquet"]["conversion"]) + conversion["_effective_null_policy"] = NullPolicy( + **self.manifest["metadata"]["parquet"]["discovery"]["options"]["null_policy"] + ) + _restore_compression_options(conversion) + options["_parquet_conversion"] = conversion try: if runtime is not None: store = blosc2.RemoteStore.with_sparse_cache( @@ -172,7 +188,13 @@ def authorized_filesystem(url): source["urlpath"], dataset=source.get("dataset"), cache_policy=blosc2.CachePolicy.NONE, - _manifest=dict(copy.deepcopy(self.manifest), caches=[], batch_caches=[], linked={}), + _manifest=dict( + copy.deepcopy(self.manifest), + caches=[], + batch_caches=[], + parquet_caches=[], + linked={}, + ), allow_table_root=True, **options, ) @@ -192,7 +214,7 @@ def operation(self, callback, *, cached=None): return quota_coordinator().remote.store_operation(self, callback, cached=cached) except remote_proxy.RemoteArrayDenied as exc: raise fastapi.HTTPException(status_code=403, detail=str(exc)) from exc - except ValueError as exc: + except (KeyError, TypeError, ValueError) as exc: raise fastapi.HTTPException(status_code=400, detail=str(exc)) from exc except (OSError, zipfile.BadZipFile) as exc: raise fastapi.HTTPException(status_code=502, detail=str(exc)) from exc diff --git a/caterva2/services/server.py b/caterva2/services/server.py index 5950c162..d480e065 100644 --- a/caterva2/services/server.py +++ b/caterva2/services/server.py @@ -957,11 +957,15 @@ def get_abspath( elif (cachedir / filepath).is_dir(): return cachedir / filepath - # HDF5 files cannot be compressed, as they are supported natively - if ( - filepath.suffix not in srv_utils.BLOSC2_NATIVE_SUFFIXES | srv_utils.HDF5_SUFFIXES - and not may_not_exist - ): + # Preserve access to Parquet files uploaded before the compression exemption. + if filepath.suffix == ".parquet" and not filepath.exists() and not may_not_exist: + compressed = filepath.with_suffix(".parquet.b2") + if compressed.is_file(): + filepath = compressed + + # HDF5 files cannot be compressed, as they are supported natively. + # Parquet also needs its original bytes for HTTP range reads. + if filepath.suffix not in srv_utils.NO_COMPRESSION_SUFFIXES and not may_not_exist: if filepath.is_file(): srv_utils.compress_file(filepath) filepath = f"{filepath}.b2" @@ -1533,7 +1537,7 @@ async def __call__(self, scope, receive, send): await concurrency.run_in_threadpool(self.cleanup) -@app.get("/api/download/{path:path}") +@app.api_route("/api/download/{path:path}", methods=["GET", "HEAD"]) async def download_data( path: pathlib.Path, user: db.User = Depends(optional_user), @@ -1541,8 +1545,8 @@ async def download_data( accept_encoding: str | None = fastapi.Header(None), range_header: str | None = fastapi.Header(None, alias="Range"), ): - # This one always streams, decompressing on the way out more often than not, - # so it never serves ranges; api/fetch on a stored dataset is what does. The + # Regular files stream, often decompressing on the way out, and refuse ranges. + # Stored Parquet files use FileResponse below to support range reads. The # refusal comes after the path is resolved, so a path that does not exist is # still a 404 rather than a 416 about a file nobody has. provider = providers.provider_for(path.parts[0]) @@ -1570,6 +1574,13 @@ async def download_data( from caterva2.services import remote_store abspath = get_abspath(path, user) + if abspath.suffix == ".parquet": + return FileResponse( + abspath, + filename=path.name, + media_type="application/vnd.apache.parquet", + headers=with_etag(abspath), + ) manifest = remote_store.inspect(abspath) if manifest is not None and (remote_proxy.policy.enabled or not include_cache): artifact, etag, cleanup = await concurrency.run_in_threadpool( @@ -2588,13 +2599,11 @@ async def upload_file( # Check quota # TODO To be fair we should check quota later (after compression, zip unpacking etc.) data = await file.read() - if abspath.suffix not in srv_utils.BLOSC2_NATIVE_SUFFIXES: - schunk = blosc2.SChunk(data=data) # If regular file, compress it abspath.parent.mkdir(exist_ok=True, parents=True) - if abspath.suffix not in srv_utils.BLOSC2_NATIVE_SUFFIXES | {".h5", ".hdf5"}: - data = schunk.to_cframe() + if abspath.suffix not in srv_utils.NO_COMPRESSION_SUFFIXES: + data = blosc2.SChunk(data=data).to_cframe() abspath = abspath.with_suffix(abspath.suffix + ".b2") # Write the file @@ -2687,13 +2696,10 @@ async def load_from_url( response.raise_for_status() data = response.content - if abspath.suffix not in srv_utils.BLOSC2_NATIVE_SUFFIXES: - schunk = blosc2.SChunk(data=data) - # If regular file, compress it abspath.parent.mkdir(exist_ok=True, parents=True) - if abspath.suffix not in srv_utils.BLOSC2_NATIVE_SUFFIXES | {".h5", ".hdf5"}: - data = schunk.to_cframe() + if abspath.suffix not in srv_utils.NO_COMPRESSION_SUFFIXES: + data = blosc2.SChunk(data=data).to_cframe() abspath = abspath.with_suffix(abspath.suffix + ".b2") # Write the file @@ -3362,7 +3368,7 @@ def add_dataset(path, abspath, mountable=False, size=None): else: relpath = pathlib.Path(*segments[2:]) abspath = rootdir / relpath - if abspath.suffix not in srv_utils.BLOSC2_NATIVE_SUFFIXES: + if abspath.suffix not in srv_utils.NO_COMPRESSION_SUFFIXES: abspath = pathlib.Path(f"{abspath}.b2") with contextlib.suppress(FileNotFoundError, NotADirectoryError): @@ -4225,7 +4231,7 @@ def store_member(name, body): raise ValueError("archive member escapes its destination") if any(p.startswith((".", "__MACOSX")) for p in member.parts): return - if member.suffix not in srv_utils.BLOSC2_NATIVE_SUFFIXES: + if member.suffix not in srv_utils.NO_COMPRESSION_SUFFIXES: body = blosc2.SChunk(data=body).to_cframe() member = member.with_suffix(member.suffix + ".b2") write_dataset(path / member, body) @@ -4282,19 +4288,17 @@ def store_member(name, body): new_members = [ member for member in members - if not (path / member).is_dir() and member.suffix not in srv_utils.BLOSC2_NATIVE_SUFFIXES + if not (path / member).is_dir() and member.suffix not in srv_utils.NO_COMPRESSION_SUFFIXES ] for member in new_members: srv_utils.compress_file(path / member) # We are done, redirect to home, and show the new files, starting with the first one - first_member = next((m for m in new_members), None) + first_member = next((m for m in members if (path / m).is_file() or m in new_members), None) path = f"{name}/{first_member}" return htmx_redirect(hx_current_url, make_url(request, "html_home", path=path), root=name) - if suffix in [".h5", ".hdf5"]: - pass - elif filename.suffix not in srv_utils.BLOSC2_NATIVE_SUFFIXES: + if filename.suffix not in srv_utils.NO_COMPRESSION_SUFFIXES: schunk = blosc2.SChunk(data=data) data = schunk.to_cframe() filename = f"{filename}.b2" @@ -4338,9 +4342,7 @@ async def htmx_delete( abspath = settings.public / path # Remove - if abspath.suffix in [".h5", ".hdf5"]: - pass - elif abspath.suffix not in {".b2frame", ".b2nd"}: + if abspath.suffix not in srv_utils.NO_COMPRESSION_SUFFIXES: abspath = abspath.with_suffix(abspath.suffix + ".b2") if not abspath.exists(): return fastapi.HTTPException(status_code=404) diff --git a/caterva2/services/sparse_cache.py b/caterva2/services/sparse_cache.py index 650b4526..70cb7ddb 100644 --- a/caterva2/services/sparse_cache.py +++ b/caterva2/services/sparse_cache.py @@ -478,6 +478,7 @@ def execute(path=None, operation=callback): store.manifest["caches"] or store.manifest.get("batch_caches") or store.manifest.get("linked") + or store.manifest.get("parquet_caches") ): self._intent(gid, "coldify") cold = io.BytesIO() @@ -485,7 +486,9 @@ def execute(path=None, operation=callback): new_sig = self.q.publish_locked( rel, cold.getvalue(), expected=sig, preserve_remote=True ) - store.manifest = dict(store.manifest, caches=[], batch_caches=[], linked={}) + store.manifest = dict( + store.manifest, caches=[], batch_caches=[], parquet_caches=[], linked={} + ) store.carrier_generation = new_sig spec = hashlib.sha256(msgpack_packb(store.manifest)).hexdigest() with self.q.transaction() as db: diff --git a/caterva2/services/srv_utils.py b/caterva2/services/srv_utils.py index 6cd434b1..c6b41db6 100644 --- a/caterva2/services/srv_utils.py +++ b/caterva2/services/srv_utils.py @@ -40,6 +40,8 @@ BLOSC2_NATIVE_SUFFIXES = BLOSC2_ARRAY_SUFFIXES | BLOSC2_TABLE_SUFFIXES | BLOSC2_FRAME_SUFFIXES HDF5_SUFFIXES = {".h5", ".hdf5"} +# Preserve these formats byte-for-byte when storing uploaded files. +NO_COMPRESSION_SUFFIXES = BLOSC2_NATIVE_SUFFIXES | HDF5_SUFFIXES | {".parquet"} # Container suffixes whose paths may descend into internal (virtual) members. BLOSC2_CONTAINER_SUFFIXES = {".b2z"} | HDF5_SUFFIXES @@ -524,6 +526,9 @@ def read_metadata(obj, mtime=None): finally: container.close() + if path.suffix == ".parquet": + return models.File(mtime=mtime, size=size) + assert path.suffix in BLOSC2_NATIVE_SUFFIXES try: reference = remote_proxy.inspect(path) diff --git a/caterva2/tests/test_remote_store.py b/caterva2/tests/test_remote_store.py index 55074671..b1c987ea 100644 --- a/caterva2/tests/test_remote_store.py +++ b/caterva2/tests/test_remote_store.py @@ -202,6 +202,141 @@ class RichRow: assert list(blosc2.ctable_from_cframe(projected).text[:]) == ["one", None, "three"] +@pytest.fixture(params=["taxi.parquet", "taxi-data"]) +def parquet_runtime(table_runtime, tmp_path, request): + import pyarrow as pa + import pyarrow.parquet as pq + + fs = fsspec.filesystem("memory") + url = f"https://data.example/{request.param}" + source = tmp_path / "taxi.parquet" + pq.write_table(pa.table({"fare": [10, 20, 30], "cab": ["a", "b", "c"]}), source, row_group_size=2) + fs.pipe_file(url, source.read_bytes()) + path = table_runtime.with_name("parquet-reference.b2z") + with blosc2.RemoteCTable( + url, + source_format="parquet", + _filesystem=fs, + cache_dir=tmp_path / "parquet-cache", + columns=["fare", "cab"], + blosc2_batch_size=2, + ) as remote: + assert list(remote.slice(0, 2).fare[:]) == [10, 20] + remote.save(path) + return path + + +def test_remote_parquet_reference(parquet_runtime, tmp_path, monkeypatch): + import blosc2.remote_parquet as parquet + import pyarrow as pa + import pyarrow.parquet as pq + + from caterva2.services import server + + path = parquet_runtime + manifest = remote_store.inspect(path) + assert manifest["parquet_caches"] + assert remote_store.root_kind(manifest) == "ctable" + q = server.quota_coordinator() + with monkeypatch.context() as patch: + patch.setattr( + parquet, "_open_source_handle", lambda *args, **kwargs: pytest.fail("warm Parquet row fetched") + ) + assert srv_utils.read_metadata(path).kind == "ctable" + table = server.open_b2(path, "@public/parquet-reference.b2z") + assert table.nrows == 3 + assert list(blosc2.ctable_from_cframe(table.fetch(slice_=slice(0, 2))).fare[:]) == [10, 20] + assert q.usage()["remote_cache_used"] > 0 + assert remote_store.inspect(path)["parquet_caches"] == [] + used = q.usage()["used"] + with q.transaction() as db: + db.execute("UPDATE account SET cache_fill_suspended=1,quota=?", (used,)) + assert list(blosc2.ctable_from_cframe(table.fetch(slice_=slice(0, 2))).fare[:]) == [10, 20] + with q.transaction() as db: + db.execute("UPDATE account SET cache_fill_suspended=0,quota=0") + for include_cache in (True, False): + export, _, cleanup = q.remote.export_store(table.store, include_cache=include_cache) + try: + exported = remote_store.inspect(export) + assert exported["source"] == manifest["source"] + assert bool(exported["parquet_caches"]) == include_cache + finally: + cleanup() + + source = tmp_path / "taxi.parquet" + pq.write_table(pa.table({"fare": [40, 50, 60, 70], "cab": ["d", "e", "f", "g"]}), source) + fsspec.filesystem("memory").pipe_file(manifest["source"]["urlpath"], source.read_bytes()) + assert list(blosc2.ctable_from_cframe(table.fetch(slice_=slice(0, 2))).fare[:]) == [10, 20] + path.write_bytes(table.store.refreshed_bytes()) + updated = server.open_b2(path, "@public/parquet-reference.b2z") + assert list(blosc2.ctable_from_cframe(updated.fetch(slice_=slice(0, 2))).fare[:]) == [40, 50] + + +@pytest.mark.parametrize("policy", ["none", "memory"]) +def test_remote_parquet_without_retention(parquet_runtime, policy): + from caterva2.services import server + + manifest = dict( + remote_store.inspect(parquet_runtime), + cache_policy=policy, + max_cache_bytes=None if policy == "none" else 1 << 20, + ) + path = parquet_runtime.with_name("uncached.b2z") + remote_store.cold_export(manifest, path) + table = server.open_b2(path, "@public/uncached.b2z") + result = blosc2.ctable_from_cframe(table.fetch(slice_=slice(1, 3), field="cab")) + assert list(result.cab[:]) == ["b", "c"] + assert server.quota_coordinator().usage()["remote_cache_used"] == 0 + path.write_bytes(table.store.refreshed_bytes()) + assert remote_store.inspect(path)["cache_policy"] == policy + + +@pytest.mark.asyncio +async def test_remote_parquet_api(parquet_runtime, monkeypatch): + import httpx + + from caterva2.services import server + + path = parquet_runtime + monkeypatch.setattr(server.settings, "statedir", path.parents[1]) + monkeypatch.setattr(server.settings, "public", path.parent) + monkeypatch.setitem(server.app.dependency_overrides, server.optional_user, lambda: None) + async with httpx.AsyncClient( + transport=httpx.ASGITransport(app=server.app), base_url="http://test" + ) as client: + response = await client.get("/api/info/@public/parquet-reference.b2z") + assert response.status_code == 200, response.text + assert response.json()["kind"] == "ctable" + response = await client.get( + "/api/fetch/@public/parquet-reference.b2z", + params={"filter": "fare >= 20", "field": "cab", "slice_": "0:1"}, + ) + assert response.status_code == 200, response.text + assert list(blosc2.ctable_from_cframe(response.content).cab[:]) == ["b"] + response = await client.post( + "/htmx/path-view/@public/parquet-reference.b2z", data={"sortby": "fare"} + ) + assert response.status_code == 200, response.text + response = await client.get( + "/api/download/@public/parquet-reference.b2z", params={"include_cache": "false"} + ) + assert response.status_code == 200, response.text + exported = path.with_name("downloaded.b2z") + exported.write_bytes(response.content) + assert remote_store.inspect(exported)["parquet_caches"] == [] + + +def test_remote_parquet_resource_limit(parquet_runtime, monkeypatch): + import fastapi + + from caterva2.services import server + + monkeypatch.setattr(remote_proxy, "policy", dataclasses.replace(remote_proxy.policy, max_nbytes=1)) + with pytest.raises(fastapi.HTTPException) as exc: + server.open_b2(parquet_runtime, "@public/parquet-reference.b2z") + assert exc.value.status_code == 403 + + @pytest.mark.asyncio async def test_standalone_remote_ctable_api(table_runtime, monkeypatch): import httpx diff --git a/caterva2/tests/test_storage_quota_api.py b/caterva2/tests/test_storage_quota_api.py index 399d0389..4f3b98f9 100644 --- a/caterva2/tests/test_storage_quota_api.py +++ b/caterva2/tests/test_storage_quota_api.py @@ -4,6 +4,7 @@ import pathlib import types import uuid +import zipfile import blosc2 import fsspec @@ -51,6 +52,69 @@ def assert_usage(server): assert quota.usage()["reserved"] == quota.usage()["working"] == 0 +@pytest.mark.asyncio +@pytest.mark.parametrize("upload", ["api", "web", "zip", "url"]) +async def test_parquet_upload_preserves_bytes_and_serves_ranges(quota_api, monkeypatch, upload): + pa = pytest.importorskip("pyarrow") + pq = pytest.importorskip("pyarrow.parquet") + server, client, _ = quota_api + buffer = io.BytesIO() + pq.write_table(pa.table({"value": [1, 2, 3]}), buffer) + data = buffer.getvalue() + if upload == "url": + async_client = httpx.AsyncClient + with monkeypatch.context() as patch: + patch.setattr( + server.httpx, + "AsyncClient", + lambda **kwargs: async_client( + **kwargs, + transport=httpx.MockTransport(lambda request: httpx.Response(200, content=data)), + ), + ) + response = await client.post( + "/api/load_from_url/@public/data.parquet", + data={"remote_url": "https://example.org/data.parquet"}, + ) + else: + filename, body = "data.parquet", data + if upload == "zip": + archive = io.BytesIO() + with zipfile.ZipFile(archive, "w") as file: + file.writestr(filename, data) + filename, body = "data.zip", archive.getvalue() + route = "/api/upload/@public/data.parquet" if upload == "api" else "/htmx/upload/@public" + response = await client.post(route, files={"file": (filename, body)}) + assert response.status_code == 200, response.text + path = server.settings.public / "data.parquet" + assert path.read_bytes() == data + assert not path.with_suffix(".parquet.b2").exists() + response = await client.get("/api/info/@public/data.parquet") + assert response.status_code == 200, response.text + assert response.json()["size"] == len(data) + url = "/api/download/@public/data.parquet" + response = await client.get(url) + assert response.status_code == 200 + assert response.content == data + response = await client.head(url) + assert response.status_code == 200 + assert int(response.headers["content-length"]) == len(data) + assert response.headers["accept-ranges"] == "bytes" + assert not response.content + response = await client.get(url, headers={"Range": "bytes=-8"}) + assert response.status_code == 206 + assert response.content == data[-8:] + assert response.headers["content-range"] == f"bytes {len(data) - 8}-{len(data) - 1}/{len(data)}" + response = await client.get(url, headers={"Range": f"bytes={len(data)}-"}) + assert response.status_code == 416 + assert path.read_bytes() == data + assert_usage(server) + response = await client.delete("/htmx/delete/@public/data.parquet") + assert response.status_code == 200, response.text + assert not path.exists() + assert_usage(server) + + @pytest.mark.asyncio async def test_upload_copy_append_remove_and_quota_denial(quota_api): server, client, _ = quota_api diff --git a/doc/utilities/cat2-server.md b/doc/utilities/cat2-server.md index 5274e7d4..4258dd03 100644 --- a/doc/utilities/cat2-server.md +++ b/doc/utilities/cat2-server.md @@ -102,7 +102,7 @@ proxy. Logical `api/fetch` requests continue to return array data. Caterva2 also accepts portable `blosc2.RemoteStore.save()` archives (`.b2z`) for B2Z, HDF5, and Zarr sources, plus `blosc2.RemoteCTable.save()` archives -for B2Z and PyTables/HDF5 tables. Store archives are browsable containers: for example, +for B2Z, PyTables/HDF5, and Parquet tables. Store archives are browsable containers: for example, `@public/store.b2z/group/array` supports metadata, sliced fetches, and compressed chunk reads. Known names and attributes can be inspected without contacting the source; undiscovered metadata and array geometry require authorized discovery. @@ -110,6 +110,13 @@ Table roots and table leaves report `ctable` metadata and support row slices, filters, selected fields, and the existing table browser. A table has no table-level compressed-chunk endpoint. Its columns, masks, batches, and indexes share the enclosing reference's cache allowance. +Parquet references use the same RemoteStore runtime path and the cache +policy saved in the archive. Save a `RemoteCTable` for a Parquet URL as `.b2z` +and upload it like any other table reference. To retain converted row groups +across requests, save with a DISK cache policy. Uploaded warm groups seed the +private cache and count against its allowance; subsequent cached-only reads can +serve them without fetching payload. A changed source requires refreshing the +hosted reference. The server does not check its version on every warm read. The same `[server.remote_proxy]` HTTPS policy applies to discovery and leaf reads. `max_metadata_bytes` (default 16 MiB) and `max_nodes` (default 100,000) bound discovery in addition to the existing per-array geometry limits. @@ -122,7 +129,7 @@ The SQLite ledger charges allocated storage, including manifests and directories Requested MEMORY/NONE stores execute without retained payload. A denied cache fill falls back to a read without retention; existing warm hits remain usable. -Uploaded warm leaves, table batches, and linked references are imported once, +Uploaded warm leaves, table batches, Parquet row groups, and linked references are imported once, then the public archive is replaced with a cold descriptor. Downloads include private warm cache data by default; `include_cache=false` produces a cold archive without network access. Export staging is reserved until the response completes. Interrupted disposable @@ -139,6 +146,16 @@ generation accounting and prevents workers from using a partly initialized ledge This support requires Python-Blosc2 4.14.0. Use `RemoteStore.with_sparse_cache()` for standalone shared runtime access; the ordinary upstream `cache_dir` constructor retains its exclusive-owner semantics. +Parquet hosting requires a python-blosc2 build with native Parquet +RemoteStore support. Caterva2 supplies its authorized filesystem to discovery, +deferred row-group reads, and refresh. + +Raw `.parquet` files uploaded through the API or web interface (including archive +uploads and URL imports) are stored unchanged, without a `.b2` wrapper. Download +them through `/api/download/@public/example.parquet` (or the corresponding shared +or personal path). This endpoint supports GET, HEAD, and HTTP byte ranges, so +remote Parquet readers can fetch only the bytes they need. Raw uploads appear as +files; use a saved `RemoteCTable` reference for Caterva2's table views. ### Admission and recovery diff --git a/pyproject.toml b/pyproject.toml index a4b1b922..37348a9d 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -62,7 +62,7 @@ server = [ "aiohttp", "aiosqlite", "caterva2[base-services]", - "caterva2[hdf5]", + "caterva2[hdf5,pyarrow]", "fastapi-mail", # Not optional: `count_written` reads a frame's offsets through # `blosc2.FsspecNDSource`, which is the whole of how a chunk fill reports @@ -77,7 +77,6 @@ server = [ "nbconvert", "nbformat", "pillow", - "pyarrow", "pygments", "python-dotenv", "python-multipart", @@ -111,6 +110,9 @@ hdf5 = [ "hdf5plugin", "msgpack", ] +pyarrow = [ + "pyarrow", +] clients = [ "rich", "textual",