From 5310aafd08a5781055014c83c9551e7b1ca8cf0f Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 26 Sep 2026 22:49:10 +0000 Subject: [PATCH 01/55] CI: use info.cloudDatasetName (roundtrip's real field name) Run 2 of build-lightsheet-fixture succeeded at the roundtrip and printed: CLOUD_DATASET_ID=6ab84b549852a120dbcb22bc but crashed on the next line with: Unrecognized field name "remoteDatasetName". The info struct that ndi.test.cloud.lightsheet_blob_cloud_roundtrip returns spells the field ``cloudDatasetName``, not ``remoteDatasetName`` -- I made the wrong name up when I wrote the workflow. Because the crash happened BEFORE the fopen to GITHUB_OUTPUT, neither job output landed and the step summary + lightsheet-fixture-id artifact steps never ran. The cloud upload itself is fine (the fixture is live under test user 1, dataset id 6ab84b549852a120dbcb22bc); this fixes the workflow so the next dispatch surfaces the id through every channel (log, outputs, summary, artifact) instead of only in the raw log. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- .github/workflows/build-lightsheet-fixture.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.github/workflows/build-lightsheet-fixture.yml b/.github/workflows/build-lightsheet-fixture.yml index 7338cb81..1f6ac7f6 100644 --- a/.github/workflows/build-lightsheet-fixture.yml +++ b/.github/workflows/build-lightsheet-fixture.yml @@ -139,11 +139,11 @@ jobs: % them up: on stdout for the log, and via % GITHUB_OUTPUT so a follow-up job can consume them. fprintf('CLOUD_DATASET_ID=%s\n', info.cloudDatasetId); - fprintf('DATASET_NAME=%s\n', info.remoteDatasetName); + fprintf('DATASET_NAME=%s\n', info.cloudDatasetName); fid = fopen(getenv("GITHUB_OUTPUT"), "a"); fprintf(fid, "cloud_dataset_id=%s\n", info.cloudDatasetId); - fprintf(fid, "dataset_name=%s\n", info.remoteDatasetName); + fprintf(fid, "dataset_name=%s\n", info.cloudDatasetName); fclose(fid); - name: Write summary From ba678780efdea7db4d805630b8639c11c2e3ef11 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 26 Sep 2026 23:14:35 +0000 Subject: [PATCH 02/55] Wire the pyramid loader integration test to the cloud fixture Fixture: 6ab84b549852a120dbcb22bc (test user 1, prod), uploaded by the MATLAB roundtrip via build-lightsheet-fixture.yml. Its ids live in tests/fixtures/lightsheet_cloud_fixture.json so the test finds them and a regeneration is one edit in one place. Test file: tests/test_pyramid_loader_integration.py * Skips automatically when NDI_CLOUD_USERNAME / NDI_CLOUD_PASSWORD are not set (same gate as test_cloud_live.py), so a laptop without secrets stays quiet. * Module-scoped download: pulls the fixture ONCE with sync_files=false and hands the path to every test. sync_files=false is deliberate -- we want to exercise the on-demand cloud-fetch path the viewer uses, not a fully-hydrated local copy. * Five checks: - loader.numLevels + loader.docs enumerate finest-first - loader.specs returns one 3D array per channel - the coarsest level computes and is not all-zero (guards the silent-fill_value failure the DID series switch once hid) - every level in the ladder computes; per-level timings are printed for the "home vs. office vs. CI" comparison - loader.stats() reports a non-empty fetcher summary so we know real cloud fetches ran Registration: * new "cloud" pytest marker in pyproject.toml * test-cloud-api.yml now includes test_pyramid_loader_integration.py in the same account-and-env matrix as the rest of test_cloud_*.py Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- .github/workflows/test-cloud-api.yml | 7 +- pyproject.toml | 1 + tests/fixtures/lightsheet_cloud_fixture.json | 8 + tests/test_pyramid_loader_integration.py | 240 +++++++++++++++++++ 4 files changed, 255 insertions(+), 1 deletion(-) create mode 100644 tests/fixtures/lightsheet_cloud_fixture.json create mode 100644 tests/test_pyramid_loader_integration.py diff --git a/.github/workflows/test-cloud-api.yml b/.github/workflows/test-cloud-api.yml index e0b3b98a..f9604baa 100644 --- a/.github/workflows/test-cloud-api.yml +++ b/.github/workflows/test-cloud-api.yml @@ -49,4 +49,9 @@ jobs: # All live cloud test files, not just test_cloud_live.py -- the # others were previously exercised only by ci.yml, so they had no # coverage on days when nobody opened a PR. - pytest tests/test_cloud_*.py -v --tb=short + # + # test_pyramid_loader_integration.py points at the permanent + # lightsheet blob fixture (see tests/fixtures/lightsheet_cloud_fixture.json, + # built by build-lightsheet-fixture.yml). It runs under the + # same account-and-env matrix as the rest of the cloud tests. + pytest tests/test_cloud_*.py tests/test_pyramid_loader_integration.py -v --tb=short diff --git a/pyproject.toml b/pyproject.toml index 7a2b2516..7beab22c 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -161,6 +161,7 @@ addopts = "-v --tb=short --ignore=tests/symmetry" markers = [ "slow: marks tests as slow (deselect with '-m \"not slow\"')", "symmetry: marks cross-language symmetry tests (run with 'pytest tests/symmetry/')", + "cloud: marks live-cloud tests (skipped when NDI_CLOUD_USERNAME / NDI_CLOUD_PASSWORD are missing)", ] [tool.black] diff --git a/tests/fixtures/lightsheet_cloud_fixture.json b/tests/fixtures/lightsheet_cloud_fixture.json new file mode 100644 index 00000000..9225f353 --- /dev/null +++ b/tests/fixtures/lightsheet_cloud_fixture.json @@ -0,0 +1,8 @@ +{ + "cloud_dataset_id": "6ab84b549852a120dbcb22bc", + "user": "1", + "environment": "prod", + "created_at": "2026-09-26T22:47:00Z", + "workflow_run": "https://github.com/Waltham-Data-Science/NDI-python/actions/runs/36277182766", + "notes": "Small synthetic lightsheet blob uploaded by ndi.test.cloud.lightsheet_blob_cloud_roundtrip (MATLAB), 12 documents / 15 files / 1.92 MB. Regenerate with the Build lightsheet blob fixture workflow. tests/test_pyramid_loader_integration.py reads this JSON by default when NDI_LIGHTSHEET_TEST_FIXTURE_ID is not set." +} diff --git a/tests/test_pyramid_loader_integration.py b/tests/test_pyramid_loader_integration.py new file mode 100644 index 00000000..beb2af5f --- /dev/null +++ b/tests/test_pyramid_loader_integration.py @@ -0,0 +1,240 @@ +"""End-to-end loader test against a real NDI cloud fixture. + +The permanent fixture is built by the MATLAB roundtrip +(``ndi.test.cloud.lightsheet_blob_cloud_roundtrip``) and uploaded via +the ``Build lightsheet blob fixture`` GitHub Actions workflow. Its +identifier lives in ``tests/fixtures/lightsheet_cloud_fixture.json`` +so a regeneration is one edit in one place; the workflow overwrites +that JSON when re-run. + +Skipped automatically when the cloud credentials are not set (same +gate as ``test_cloud_live.py``), so a laptop without secrets keeps +running the rest of the unit tests as usual. In CI these tests run +under the same environment matrix as ``test-cloud-api.yml``. + +What the tests exercise: + +* Downloading the fixture with ``sync_files=False`` -- the loader + path we care about is on-demand fetch through the NDI cloud API, + not local disk. +* Opening the pyramid via :class:`ImagePyramidLoader`, walking every + level, ``.compute()``-ing each so the fetcher is actually driven. +* Verifying the finest level's shape / dtype / channel count against + what the MATLAB fixture writes. +* Reading ``loader.stats()`` at the end so the profile numbers + (cache hits, cloud fetches, mean fetch time) appear in the test + log for the "home vs. office vs. CI" comparison. + +Session-scoped download: opening every test would redownload; a +module fixture pulls once and every test in the file shares the +result. +""" + +from __future__ import annotations + +import json +import os +import time +from pathlib import Path + +import pytest + +# --------------------------------------------------------------------------- +# Skip gate + fixture registry. + + +_HAS_CREDS = bool(os.environ.get("NDI_CLOUD_USERNAME") and os.environ.get("NDI_CLOUD_PASSWORD")) +pytestmark = [ + pytest.mark.cloud, + pytest.mark.skipif(not _HAS_CREDS, reason="NDI cloud credentials not set"), +] + +FIXTURE_JSON = Path(__file__).parent / "fixtures" / "lightsheet_cloud_fixture.json" + + +def _fixture_id() -> str: + """Pick the dataset id the test should point at. + + Precedence: ``NDI_LIGHTSHEET_TEST_FIXTURE_ID`` (env override, for + a one-shot pointing at a freshly rebuilt fixture) -> the + ``cloud_dataset_id`` in ``lightsheet_cloud_fixture.json``. + """ + env = os.environ.get("NDI_LIGHTSHEET_TEST_FIXTURE_ID", "").strip() + if env: + return env + with open(FIXTURE_JSON) as f: + return str(json.load(f)["cloud_dataset_id"]) + + +# --------------------------------------------------------------------------- +# Shared download. + + +@pytest.fixture(scope="module") +def downloaded_dataset(tmp_path_factory) -> Path: + """Pull the fixture once for every test in this module. + + ``sync_files=False`` mirrors what the viewer opens on a normal + launch: documents land locally, chunk files are fetched from the + cloud on demand. That is the read path we want to exercise -- + with sync_files on, every chunk arrives during download and the + loader's cloud-fetch path never runs. + """ + from ndi.cloud.orchestration import downloadDataset + + target = tmp_path_factory.mktemp("lightsheet_fixture_download") + dataset_id = _fixture_id() + dataset = downloadDataset( + dataset_id, + str(target), + sync_files=False, + verbose=False, + ) + # downloadDataset returns an ndi.ndi_dataset backed by + # target/. The tests want the on-disk path so + # ImagePyramidLoader can open a session on it. + dataset_path = getattr(dataset, "_path", None) or (target / dataset_id) + return Path(dataset_path) + + +@pytest.fixture(scope="module") +def pyramid_doc(downloaded_dataset): + """The first ``lightsheetZarrPyramid`` on the downloaded dataset.""" + from ndi.dataset._dataset import ndi_dataset_dir + from ndi.query import ndi_query + + dataset = ndi_dataset_dir(str(downloaded_dataset)) + q = ndi_query("").isa("lightsheetZarrPyramid") + docs = list(dataset.database_search(q)) + assert docs, f"no lightsheetZarrPyramid documents in {downloaded_dataset}" + # The dataset itself owns database_search (it follows links into + # every session), so return both -- the loader wants the session + # or dataset that resolves further searches, plus the doc. + return dataset, docs[0] + + +# --------------------------------------------------------------------------- +# Tests. + + +class TestLoaderOpensTheFixture: + def test_the_loader_enumerates_the_levels_finest_first(self, pyramid_doc): + from ndi.pyramid.loader import ImagePyramidLoader + + session, doc = pyramid_doc + loader = ImagePyramidLoader(session, doc, reduction="mean") + try: + # numLevels does an implicit build via docs; a fixture + # that lands with no levels means the MATLAB ingest ran + # halfway. + n = loader.numLevels + assert n >= 2, f"pyramid should have at least 2 levels, got {n}" + # Levels are sorted finest-first, so index 0 has the + # smallest voxel size on the last docs (they're 0-based + # ints). + docs = loader.docs + levels = [int(d.document_properties["lightsheetZarrLevel"]["level"]) for d in docs] + assert levels == sorted( + levels + ), f"levelDocs should return finest-first, got levels {levels}" + finally: + loader.close() + + def test_the_loader_returns_one_spec_per_channel(self, pyramid_doc): + from ndi.pyramid.loader import ImagePyramidLoader + + session, doc = pyramid_doc + loader = ImagePyramidLoader(session, doc, reduction="mean") + try: + specs = loader.specs + # Fixture is 2 channels (makeBlobFixture defaults). + assert len(specs) == 2, f"expected 2 layer specs, got {len(specs)}" + for spec in specs: + assert "data" in spec + # The single-level scaffold delivers a 3D array + # (channel axis stripped) at the coarsest level. + data_shape = getattr(spec["data"], "shape", ()) + assert len(data_shape) == 3, f"expected 3D per-channel data, got shape {data_shape}" + finally: + loader.close() + + +class TestLoaderReadsRealBytes: + def test_the_coarsest_level_computes_and_is_not_all_zero(self, pyramid_doc): + # A pyramid the fetcher can open but never actually pulls + # from would compute to all-zero (fill_value) blocks -- the + # classic "silent fetch failure" bug that hid the DID series + # switch. Fixture has a ring + Gaussian + spike so any + # actually-fetched slice is non-zero. + import numpy as np + + from ndi.pyramid.loader import ImagePyramidLoader + + session, doc = pyramid_doc + loader = ImagePyramidLoader(session, doc, reduction="mean") + try: + specs = loader.specs + data = specs[0]["data"] # coarsest-level array for channel 0 + arr = data.compute() + assert arr.dtype == np.uint16, f"expected uint16, got {arr.dtype}" + assert arr.any(), "coarsest-level channel-0 slice is entirely fill_value" + finally: + loader.close() + + def test_every_level_can_be_computed(self, pyramid_doc): + # Walks the full ladder. On the small fixture (~5 levels max, + # 300**3 voxels finest) this is small enough to finish in a + # single-digit-seconds test on a reasonable link. If it grows + # to minutes, the fetcher or the network is the problem and + # the profile numbers below will name which. + from ndi.pyramid.loader import ImagePyramidLoader + + session, doc = pyramid_doc + loader = ImagePyramidLoader(session, doc, reduction="mean") + per_level: list[dict] = [] + try: + channel0_ladder = specs_channel_ladder(loader, 0) + for i, arr in enumerate(channel0_ladder): + t0 = time.perf_counter() + out = arr.compute() + dt = time.perf_counter() - t0 + per_level.append({"level": i, "shape": tuple(out.shape), "seconds": round(dt, 2)}) + finally: + loader.close() + print(f"\n[integration] per-level compute times: {per_level}") + assert per_level, "expected at least one level to compute" + + def test_stats_show_at_least_one_cloud_fetch(self, pyramid_doc): + # If every read hits the cache instead of the cloud, the + # fixture is not exercising the code we care about. On a + # freshly-downloaded fixture with sync_files=False, every + # chunk needs a cloud round trip. + from ndi.pyramid.loader import ImagePyramidLoader + + session, doc = pyramid_doc + loader = ImagePyramidLoader(session, doc, reduction="mean") + try: + for arr in specs_channel_ladder(loader, 0): + arr.compute() + snap = loader.stats() + print(f"\n[integration] loader.stats(): {snap}") + # stats["fetcher"] is a summary line like + # "resolves: N cache-hits (mean X.Xms), M cloud-fetches + # (mean X.Xs), K failures". We just check the summary is + # non-empty; the actual counts vary by pyramid shape. + assert snap.get("fetcher"), "loader.stats() has no fetcher summary" + finally: + loader.close() + + +def specs_channel_ladder(loader, channel: int) -> list: + """Every level's array for one channel (finest first). + + ``layerSpec`` stashes per-channel per-level arrays on each spec's + ``_ndi_level_arrays`` list (that is the ladder the level selector + swaps between). Take the ladder off spec[channel] rather than + calling into internals. + """ + specs = loader.specs + assert 0 <= channel < len(specs), f"channel {channel} out of range" + return specs[channel].get("_ndi_level_arrays", []) From b320cf1505e088301797d0ddfc7315234f31b6f6 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 00:07:59 +0000 Subject: [PATCH 03/55] CI: run Test Cloud API on pyramid-loader PRs test-cloud-api.yml was schedule + dispatch only, so a PR that adds or edits the pyramid loader had no PR-time signal it was still reaching the cloud correctly. The new integration test in this PR (tests/test_pyramid_loader_integration.py) needs to run against the real fixture (6ab84b549852a120dbcb22bc) before merging, so we know the loader hasn't regressed while the diff was in flight. Add a path-filtered pull_request trigger covering: the pyramid package (src/ndi/pyramid/**), the fixture registry, the integration test file, the existing test_cloud_*.py suite, and this workflow file. Anything else on a PR does not fire the cloud tests, so unrelated PRs still cost nothing in cloud API calls. Runs under the same matrix (User 1/2 x prod/dev) as the schedule. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- .github/workflows/test-cloud-api.yml | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/.github/workflows/test-cloud-api.yml b/.github/workflows/test-cloud-api.yml index f9604baa..9d7f6ef5 100644 --- a/.github/workflows/test-cloud-api.yml +++ b/.github/workflows/test-cloud-api.yml @@ -8,6 +8,18 @@ on: # Allow manual runs from the Actions tab workflow_dispatch: + # Run on pull requests that touch the pyramid loader, its cloud + # fixture, or the workflow itself. Kept path-filtered so a PR that + # only edits, say, gene-pyramid code does not burn cloud API calls + # and bill the two shared test accounts on every push. + pull_request: + paths: + - 'src/ndi/pyramid/**' + - 'tests/test_pyramid_loader_integration.py' + - 'tests/fixtures/lightsheet_cloud_fixture.json' + - 'tests/test_cloud_*.py' + - '.github/workflows/test-cloud-api.yml' + concurrency: group: ${{ github.workflow }}-${{ github.ref }} cancel-in-progress: true From 94f01a6f837a2278083e5e96aa8e3f7995b22efd Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 00:18:22 +0000 Subject: [PATCH 04/55] CI: install dask[array] for the pyramid loader integration test First PR run on run 36281586795 failed with: ModuleNotFoundError: No module named 'dask' on four of the five pyramid loader integration tests (the fifth one only touched loader.docs / .numLevels which do not build any dask array, so it passed). The cloud CI installs with `ndi_install.py --dev --no-validate`, which does not pull dask -- dask lives under the napari extra (``[napari]``) alongside napari itself and its Qt / OpenGL dependencies. Installing the whole napari extra here is overkill: we do not need Qt on a headless cloud runner, only the dask array machinery the loader builds its multiscale ladder with. Add a one-line ``pip install 'dask[array]>=2023.1'`` to the cloud CI's install step so the loader can compute in that environment without dragging in Qt / OpenGL. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- .github/workflows/test-cloud-api.yml | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/.github/workflows/test-cloud-api.yml b/.github/workflows/test-cloud-api.yml index 9d7f6ef5..004fb715 100644 --- a/.github/workflows/test-cloud-api.yml +++ b/.github/workflows/test-cloud-api.yml @@ -46,6 +46,11 @@ jobs: run: | python -m pip install --upgrade pip python ndi_install.py --dev --no-validate --verbose + # dask lives under the napari extra, which also pulls in Qt + # and OpenGL -- overkill for headless cloud tests. The + # pyramid loader integration test needs dask[array] to + # build its multiscale arrays, so install just that piece. + pip install 'dask[array]>=2023.1' - name: Run cloud API tests env: From d0b35c6e0d8a2167c6ad9873338d1b0f7b1187fe Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 00:19:34 +0000 Subject: [PATCH 05/55] Promote dask[array] to a runtime dependency The pyramid loader in ndi.pyramid.ImagePyramidLoader is napari-independent -- that was the point of splitting it out of gui/app/lightsheetZarr. Callers use it from matplotlib, Jupyter, headless batch analyses, and (once the sibling namespace lands) a MATLAB comparison harness. Every one of those callers needs dask[array], because that is what the loader builds its lazy multiscale arrays with. Keeping dask under the napari extra was defensible when the loader lived inside the napari viewer module; it is not now. A headless pip install of ndi should be able to run `from ndi.pyramid.loader import ImagePyramidLoader` and get to work without discovering, deep inside the traceback, that a viewer-side extra was needed. * Move `dask[array]>=2023.1` into project.dependencies alongside numpy / scipy / networkx. * Leave it pinned in the napari extra too, so an existing lock file that resolved dask via napari does not suddenly declare a version conflict. * Drop the workaround from test-cloud-api.yml -- dask now installs via the base package, so the cloud CI job needs no extra step. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- .github/workflows/test-cloud-api.yml | 5 ----- pyproject.toml | 16 ++++++++++++++-- 2 files changed, 14 insertions(+), 7 deletions(-) diff --git a/.github/workflows/test-cloud-api.yml b/.github/workflows/test-cloud-api.yml index 004fb715..9d7f6ef5 100644 --- a/.github/workflows/test-cloud-api.yml +++ b/.github/workflows/test-cloud-api.yml @@ -46,11 +46,6 @@ jobs: run: | python -m pip install --upgrade pip python ndi_install.py --dev --no-validate --verbose - # dask lives under the napari extra, which also pulls in Qt - # and OpenGL -- overkill for headless cloud tests. The - # pyramid loader integration test needs dask[array] to - # build its multiscale arrays, so install just that piece. - pip install 'dask[array]>=2023.1' - name: Run cloud API tests env: diff --git a/pyproject.toml b/pyproject.toml index 7beab22c..7e993dac 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -70,6 +70,16 @@ dependencies = [ # default. keyring is preferred when present but stays optional, # since it is the OS-keychain path rather than the portable one. "cryptography>=41.0", + # dask[array] powers ndi.pyramid.ImagePyramidLoader -- the + # napari-independent multiscale loader. It used to live under + # the napari extra because only napari-driven viewers touched + # it, but the loader is now a first-class ndi API that + # matplotlib, Jupyter and (eventually) a MATLAB comparison + # harness consume without napari. That makes dask a runtime + # dependency of the pyramid subpackage. It stays pinned in the + # napari extra as well (as a redundant floor), so an old lock + # file that pinned dask through napari does not break. + "dask[array]>=2023.1", ] [project.optional-dependencies] @@ -82,10 +92,12 @@ dependencies = [ gui = [ "PySide6>=6.5", ] -# napari, and the dask it drives the tile ladder through. An EXTRA for the +# napari is the tile viewer over the pyramid loader. An EXTRA for the # same reason gui is: a headless install should not pull in Qt and OpenGL. # ndi.gui.app.genepyramid.viewer imports through require_napari(), so the -# failure mode for a missing toolkit is a sentence naming this extra. +# failure mode for a missing toolkit is a sentence naming this extra. The +# loader's own dask requirement moved to project.dependencies once the +# loader graduated out of the viewer namespace into ndi.pyramid. napari = [ "napari[all]>=0.4.19", "dask[array]>=2023.1", From 349677893cab96476a1eebd88cf43b4fbdb9ca79 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 12:37:55 +0000 Subject: [PATCH 06/55] tests: skip pyramid loader integration when fixture is not accessible The fixture (currently 6ab84b549852a120dbcb22bc, see lightsheet_cloud_fixture.json) lives on one account + env pair: user 1, prod. The cloud CI matrix runs every account x env combination, so User 1 dev + User 2 (any env) tried to download that id and crashed with HTTP 404 + CloudNotFoundError. Turn the 404 into a skip: catch CloudNotFoundError in the downloaded_dataset fixture and pytest.skip with a message that names the current NDI_CLOUD_USERNAME + CLOUD_API_ENVIRONMENT and points at NDI_LIGHTSHEET_TEST_FIXTURE_ID for redirection. All five tests in the module inherit the skip through the fixture chain, so the three unrelated cloud CI cells no longer fail the build. Behaviour on User 1 prod is unchanged -- the fixture resolves, the download runs, all five tests execute. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- tests/test_pyramid_loader_integration.py | 30 +++++++++++++++++++----- 1 file changed, 24 insertions(+), 6 deletions(-) diff --git a/tests/test_pyramid_loader_integration.py b/tests/test_pyramid_loader_integration.py index beb2af5f..280537d8 100644 --- a/tests/test_pyramid_loader_integration.py +++ b/tests/test_pyramid_loader_integration.py @@ -80,16 +80,34 @@ def downloaded_dataset(tmp_path_factory) -> Path: with sync_files on, every chunk arrives during download and the loader's cloud-fetch path never runs. """ + from ndi.cloud.exceptions import CloudNotFoundError from ndi.cloud.orchestration import downloadDataset target = tmp_path_factory.mktemp("lightsheet_fixture_download") dataset_id = _fixture_id() - dataset = downloadDataset( - dataset_id, - str(target), - sync_files=False, - verbose=False, - ) + try: + dataset = downloadDataset( + dataset_id, + str(target), + sync_files=False, + verbose=False, + ) + except CloudNotFoundError as exc: + # The fixture lives on one account + env (see + # lightsheet_cloud_fixture.json; currently user 1, prod). The + # cloud-CI matrix runs every account x env combination, so + # three of the four cells cannot see the fixture and would + # crash with HTTP 404. Skip cleanly there rather than fail + # -- one green cell per fixture is the intended baseline. + # Set NDI_LIGHTSHEET_TEST_FIXTURE_ID to a dataset that IS + # visible on the account you are running under to point the + # tests at a different fixture. + pytest.skip( + f"fixture {dataset_id} not accessible from this account " + f"(NDI_CLOUD_USERNAME={os.environ.get('NDI_CLOUD_USERNAME', '?')!r}, " + f"CLOUD_API_ENVIRONMENT={os.environ.get('CLOUD_API_ENVIRONMENT', '?')!r}): " + f"{exc}" + ) # downloadDataset returns an ndi.ndi_dataset backed by # target/. The tests want the on-disk path so # ImagePyramidLoader can open a session on it. From 87d0d314bf4a3f132fcce593366eb0c8c3467d4f Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 13:50:30 +0000 Subject: [PATCH 07/55] batch_signed_url: retry the batch call before falling back per-member One transient TLS or connection error at start of run used to mark a scope failed and drop every uid in it to per-member getFileDetails -- O(N) API calls on a series that should have cost one. For a 156k-member lightsheet series that hung downloadDataset for hours on a residential network. _default_signer now retries getSignedURLSetAll with exponential backoff (1s / 4s / 16s, 4 total attempts) before surfacing the exception as a scope failure. Injected fake signers keep their old semantics -- tests are unaffected. See #322 and VH-Lab/NDI-matlab#1010 for the parallel MATLAB fix. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/cloud/batch_signed_url.py | 58 +++++++++++--- tests/test_cloud_batch_signed_url.py | 113 +++++++++++++++++++++++++++ 2 files changed, 160 insertions(+), 11 deletions(-) diff --git a/src/ndi/cloud/batch_signed_url.py b/src/ndi/cloud/batch_signed_url.py index de56edcc..5804af0e 100644 --- a/src/ndi/cloud/batch_signed_url.py +++ b/src/ndi/cloud/batch_signed_url.py @@ -53,6 +53,16 @@ # whole set. See Waltham-Data-Science/NDI-python#309. DEFAULT_PARTIAL_MAP_RETRY_SECONDS = 0.5 +# Backoff schedule for :func:`_default_signer` when the batch endpoint call +# raises. One transient TLS or connection error at start of run used to +# mark the scope failed and drop every uid in it to per-member +# getFileDetails -- O(N) API calls on a series that should have cost one. +# Retrying the batch itself keeps the O(1) fast path across residential +# network blips. See Waltham-Data-Science/NDI-python#322 (and the parallel +# VH-Lab/NDI-matlab#1010). Only the default production signer retries; +# a caller-injected signer is left as-is so tests keep control of counts. +DEFAULT_BATCH_RETRY_DELAYS: tuple[float, ...] = (1.0, 4.0, 16.0) + @dataclass class _CacheEntry: @@ -119,26 +129,52 @@ def _default_signer( *, file_series: str = "", client: CloudClient | None = None, + retry_delays: tuple[float, ...] = DEFAULT_BATCH_RETRY_DELAYS, + sleep: Callable[[float], None] = time.sleep, ) -> tuple[bool, dict[str, Any]]: """Fetch the whole scope through the paged getSignedURLSetAll call. Returns ``(True, answer)`` on success, ``(False, {})`` on any failure -- the batch is best-effort and its caller's fallback is what keeps the read correct. + + A raise from ``getSignedURLSetAll`` is retried with the delays in + ``retry_delays`` before it is reported as a failure. This is what keeps + one transient TLS blip on a residential connection from cascading to + the O(N) per-member fallback: the batch itself gets a few more chances, + at ~20 s of wall clock in the worst case, before its caller is told the + scope is unreachable. See Waltham-Data-Science/NDI-python#322. """ from .api import files as files_api - try: - answer = files_api.getSignedURLSetAll( - dataset_id, - document_id, - id_namespace="ndi", - file_series=file_series, - client=client, - ) - except Exception as exc: # noqa: BLE001 - reported as a miss reason - return False, {"__error__": f"{type(exc).__name__}: {exc}"} - return True, answer + attempts = len(retry_delays) + 1 + last_exc: Exception | None = None + for attempt in range(attempts): + try: + answer = files_api.getSignedURLSetAll( + dataset_id, + document_id, + id_namespace="ndi", + file_series=file_series, + client=client, + ) + return True, answer + except Exception as exc: # noqa: BLE001 - reported as a miss reason + last_exc = exc + if attempt < attempts - 1: + delay = retry_delays[attempt] + logger.info( + "batch signed-URL fetch attempt %d/%d failed (%s); " "retrying in %.1fs", + attempt + 1, + attempts, + type(exc).__name__, + delay, + ) + sleep(delay) + # Every attempt raised; report the last cause. + return False, { + "__error__": f"{type(last_exc).__name__}: {last_exc} (after {attempts} attempts)" + } class BatchSignedUrlLookup: diff --git a/tests/test_cloud_batch_signed_url.py b/tests/test_cloud_batch_signed_url.py index 3e7d5347..f3be39d3 100644 --- a/tests/test_cloud_batch_signed_url.py +++ b/tests/test_cloud_batch_signed_url.py @@ -608,3 +608,116 @@ def unique_cursor(*args, **kwargs): with pytest.raises(files_api.SignedURLSetMaxPagesReached) as exc: files_api.getSignedURLSetAll("ds1", "doc1", max_pages=3, client=MagicMock()) assert exc.value.merged["pages"] == 3 + + +# --------------------------------------------------------------------------- +# The default signer retries transient batch failures. +# --------------------------------------------------------------------------- + + +class TestDefaultSignerRetries: + """A ConnectionFailed-style raise from getSignedURLSetAll must not + collapse a whole scope to the O(N) per-member fallback on the first + try. See Waltham-Data-Science/NDI-python#322 and the parallel + VH-Lab/NDI-matlab#1010: a residential-network TLS blip used to hang + downloads for hours by tripping this cascade. + """ + + def test_a_transient_raise_then_success_returns_success(self): + from ndi.cloud.batch_signed_url import _default_signer + + calls = {"n": 0} + + def flaky(*args, **kwargs): + calls["n"] += 1 + if calls["n"] == 1: + raise ConnectionError("first try, transient") + return {"files": {"u1": "https://s3.example.com/u1"}, "pages": 1} + + with patch("ndi.cloud.api.files.getSignedURLSetAll", side_effect=flaky): + ok, answer = _default_signer( + "ds1", + "doc1", + client=MagicMock(), + retry_delays=(0.0, 0.0, 0.0), + sleep=lambda _s: None, + ) + + assert ok is True + assert answer["files"] == {"u1": "https://s3.example.com/u1"} + assert calls["n"] == 2, "should have retried exactly once before succeeding" + + def test_every_attempt_raising_reports_the_last_cause(self): + from ndi.cloud.batch_signed_url import _default_signer + + calls = {"n": 0} + + def always_fails(*args, **kwargs): + calls["n"] += 1 + raise ConnectionError(f"try {calls['n']}") + + with patch("ndi.cloud.api.files.getSignedURLSetAll", side_effect=always_fails): + ok, answer = _default_signer( + "ds1", + "doc1", + client=MagicMock(), + retry_delays=(0.0, 0.0), # 3 attempts total + sleep=lambda _s: None, + ) + + assert ok is False + assert calls["n"] == 3, "should have made 3 attempts (initial + 2 retries)" + error = answer.get("__error__", "") + assert "ConnectionError" in error + assert "3 attempts" in error, f"should name the attempt count: {error!r}" + + def test_zero_retry_delays_is_a_single_attempt(self): + """Passing an empty retry_delays disables retry entirely. + + Useful anywhere a caller wants the pre-retry behavior (a test, a + fast-fail probe, or an environment where the signer already handles + its own retries). + """ + from ndi.cloud.batch_signed_url import _default_signer + + calls = {"n": 0} + + def always_fails(*args, **kwargs): + calls["n"] += 1 + raise ConnectionError("nope") + + with patch("ndi.cloud.api.files.getSignedURLSetAll", side_effect=always_fails): + ok, _ = _default_signer( + "ds1", + "doc1", + client=MagicMock(), + retry_delays=(), + sleep=lambda _s: None, + ) + + assert ok is False + assert calls["n"] == 1, "empty retry_delays should mean one attempt, no retries" + + def test_retry_delays_are_slept_in_order(self): + """The backoff delays are consumed in order, once per failed attempt.""" + from ndi.cloud.batch_signed_url import _default_signer + + slept: list[float] = [] + + def always_fails(*args, **kwargs): + raise ConnectionError("nope") + + with patch("ndi.cloud.api.files.getSignedURLSetAll", side_effect=always_fails): + _default_signer( + "ds1", + "doc1", + client=MagicMock(), + retry_delays=(1.0, 4.0, 16.0), + sleep=slept.append, + ) + + assert slept == [ + 1.0, + 4.0, + 16.0, + ], f"expected the three backoff delays consumed in order, got {slept}" From ad618f0c5aa0f8799ea5bd9605081bf084cf7cc4 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 14:03:23 +0000 Subject: [PATCH 08/55] cloud: log per-page walk and per-attempt retry so silent hangs surface A hung getSignedURLSetAll call today emits no output for minutes: the CloudClient retries transient errors up to MAX_ATTEMPTS silently, each retry after a 120s socket timeout, and the whole walk goes silent between HTTP calls too. On a residential connection where the batch endpoint stalls this looks indistinguishable from a genuine deadlock. Two log lines break that silence: * getSignedURLSetAll: "fetching page N ..." and "page N returned M uids in T.TTs" bracket every HTTP call. Silence for more than a page-time now names which page and which document is stuck. * CloudClient._request: each retry after a transient error or 5xx logs the exception name and the retry delay. The underlying retry loop is no longer invisible. Diagnostic only -- no behavior changes. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/cloud/api/files.py | 23 +++++++++++++++++++++++ src/ndi/cloud/client.py | 28 ++++++++++++++++++++++++++-- 2 files changed, 49 insertions(+), 2 deletions(-) diff --git a/src/ndi/cloud/api/files.py b/src/ndi/cloud/api/files.py index 2e628dab..b175ef5f 100644 --- a/src/ndi/cloud/api/files.py +++ b/src/ndi/cloud/api/files.py @@ -10,10 +10,13 @@ from __future__ import annotations +import logging import time from pathlib import Path from typing import Annotated, Any, Literal +_module_logger = logging.getLogger(__name__) + from pydantic import SkipValidation, validate_call from ..client import APIResponse, CloudClient, _auto_client @@ -610,7 +613,18 @@ def getSignedURLSetAll( "pages": 0, } cursor = "" + walk_started = time.monotonic() for _ in range(max_pages): + page_started = time.monotonic() + _module_logger.info( + "getSignedURLSetAll: fetching page %d for document %s (series=%r, " + "so far %d uids, %.1fs elapsed)", + merged["pages"] + 1, + document_id, + file_series, + merged["pageCount"], + page_started - walk_started, + ) page = getSignedURLSet( dataset_id, document_id, @@ -620,6 +634,7 @@ def getSignedURLSetAll( id_namespace=id_namespace, client=client, ) + page_dt = time.monotonic() - page_started files = page.get("files", {}) if hasattr(page, "get") else {} if isinstance(files, dict): merged["files"].update(files) @@ -631,6 +646,14 @@ def getSignedURLSetAll( if expires_at: merged["expiresAt"] = expires_at merged["pages"] += 1 + _module_logger.info( + "getSignedURLSetAll: page %d returned %d uids in %.2fs " "(total so far %d/%d)", + merged["pages"], + len(files) if isinstance(files, dict) else 0, + page_dt, + merged["pageCount"], + merged["totalCount"] or -1, + ) next_cursor = page.get("nextCursor", "") if hasattr(page, "get") else "" if not next_cursor: diff --git a/src/ndi/cloud/client.py b/src/ndi/cloud/client.py index a2f6917c..47d1b46e 100644 --- a/src/ndi/cloud/client.py +++ b/src/ndi/cloud/client.py @@ -12,12 +12,15 @@ import functools import json +import logging import random import re import time from typing import Any from urllib.parse import quote as _url_quote +logger = logging.getLogger(__name__) + from .config import CloudConfig from .exceptions import ( CloudAPIError, @@ -304,7 +307,18 @@ def _request( and isinstance(exc, transient_exceptions) and attempt < self.MAX_ATTEMPTS ): - time.sleep(self._retry_delay(attempt)) + delay = self._retry_delay(attempt) + logger.info( + "cloud request %s %s: attempt %d/%d hit %s " "(%s); retrying in %.1fs", + method, + endpoint, + attempt, + self.MAX_ATTEMPTS, + type(exc).__name__, + exc, + delay, + ) + time.sleep(delay) continue raise CloudAPIError(f"Request failed{self._attempt_note(attempt)}: {exc}") from exc @@ -313,7 +327,17 @@ def _request( and resp.status_code in self.RETRY_STATUSES and attempt < self.MAX_ATTEMPTS ): - time.sleep(self._retry_delay(attempt)) + delay = self._retry_delay(attempt) + logger.info( + "cloud request %s %s: attempt %d/%d got HTTP %d; " "retrying in %.1fs", + method, + endpoint, + attempt, + self.MAX_ATTEMPTS, + resp.status_code, + delay, + ) + time.sleep(delay) continue break From 5ab9c042f843c5baa1700865c08635fd13c3cc95 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 14:09:26 +0000 Subject: [PATCH 09/55] filehandler: skip batch signed-URL lookup for single-manifest fetch _fetch_manifest asks for ONE known uid (the series manifest file). Going through fetch_cloud_file with the document id set makes the batch cache try to answer, and to answer it fetches the WHOLE document's uid->URL map through getSignedURLSetAll. A lightsheet OME-Zarr level with 121k chunks means walking 243 pages at 25s each -- 100+ minutes to fetch one manifest. Pass ndi_document_id="" so the batch is skipped and fetch_cloud_file drops straight to getFileDetails: one API call for the one uid we want. Downloading a real lightsheet dataset with SyncFiles=false is unblocked by this; before this change the download hung for hours reconstructing manifests. The batch is still the right choice for chunk fetches at viewer time, where hundreds of uids in the same document are requested back-to-back and the O(1) batch amortizes across them. The viewer-time scale problem is separate and needs the async signed-URL-set job family (porting_deferred in the bridge). Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/cloud/filehandler.py | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/src/ndi/cloud/filehandler.py b/src/ndi/cloud/filehandler.py index cbe228d8..9cad5594 100644 --- a/src/ndi/cloud/filehandler.py +++ b/src/ndi/cloud/filehandler.py @@ -292,6 +292,13 @@ def _fetch_manifest( what we want here). Otherwise, call :func:`fetch_cloud_file` directly, matching what the read-side handler does when a manifest is asked for by uid. + + We bypass the batch signed-URL lookup here. The batch is scoped + per-document, and a series with N members has an N-URL scope: for a + lightsheet OME-Zarr level with 100k+ chunks that is a page walk of + 50-100 minutes for the ONE URL a manifest fetch actually needs. A + single-file fetch through :func:`getFileDetails` is the right shape + at O(1). See Waltham-Data-Science/NDI-python#322 / issue TBD. """ source_path = f"{NDIC_SCHEME}{cloud_dataset_id}/{manifest_uid}" if custom_file_handler is not None: @@ -309,7 +316,10 @@ def _fetch_manifest( source_path, dest_path, client=client, - ndi_document_id=document_id, + # Empty ndi_document_id skips the batch lookup and goes straight to + # getFileDetails for the one uid. Batch here would ask the server + # for every other file in the document too. + ndi_document_id="", series_name="", ) From d4e31bebac70d11a90ecb865ce55c4755d26afb9 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 14:36:28 +0000 Subject: [PATCH 10/55] cloud: port async signed-URL-set-job family and route default signer through it Adds the four MATLAB entry points that stabilized on NDI-matlab main at 0a2cdec (VH-Lab/NDI-matlab#1009): createSignedURLSetJob, getSignedURLSetJob, waitForSignedURLSetJob and getSignedURLSetResult. Rewires batch_signed_url._default_signer from the paged getSignedURLSetAll walk (~25 s per 500-uid page, so 100+ min for a 121k-member document) to the async job path that builds the whole uid -> URL map server-side and returns it in one gzipped blob. Retry semantics (NDI-python#322) are preserved around the whole three-step exchange. Bridge YAML flipped for all four functions (porting_deferred -> ported) with matlab_last_sync_hash bumped to 0a2cdec, and batchSignedUrlLookup's decision_log updated to record the signer swap. Filed as Waltham-Data-Science/NDI-python#206. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01SK8sCN6BWj7VYzt2pieTpd --- src/ndi/cloud/api/files.py | 274 +++++++++++++ .../cloud/api/ndi_matlab_python_bridge.yaml | 103 +++-- src/ndi/cloud/batch_signed_url.py | 101 ++++- src/ndi/cloud/ndi_matlab_python_bridge.yaml | 15 +- tests/test_cloud_api_signed_url_set_job.py | 382 ++++++++++++++++++ tests/test_cloud_batch_signed_url.py | 34 +- 6 files changed, 839 insertions(+), 70 deletions(-) create mode 100644 tests/test_cloud_api_signed_url_set_job.py diff --git a/src/ndi/cloud/api/files.py b/src/ndi/cloud/api/files.py index b175ef5f..76a6497a 100644 --- a/src/ndi/cloud/api/files.py +++ b/src/ndi/cloud/api/files.py @@ -668,6 +668,280 @@ def getSignedURLSetAll( raise SignedURLSetMaxPagesReached(merged) +# --------------------------------------------------------------------------- +# Signed-URL-set-job family (async) +# +# The paged getSignedURLSetAll walks the /signed-url-set route with cursor +# pagination and takes ~25 s per 500-uid page. For a document with 121k +# members that is 100+ minutes to prime the batch cache. The async job path +# builds the whole uid -> URL map server-side, gzips it to S3, and hands +# back one presigned URL to download the blob: +# +# createSignedURLSetJob -> POST /.../signed-url-set-jobs returns jobId +# waitForSignedURLSetJob -> GET /signed-url-set-jobs/{jobId} poll to 'ready' +# getSignedURLSetResult -> download + parse the gzipped result blob +# +# See NDI-matlab#952, NDI-matlab#1009, NDI-python#206. +# --------------------------------------------------------------------------- + +# Terminal states reported by the signed-URL-set job service. Non-terminal +# states are 'queued' and 'running'; the mirror is intentional so callers +# see a consistent poll vocabulary across the bulk-upload and file-tier +# jobs. +_TERMINAL_SIGNED_URL_SET_STATES = ("ready", "failed") + + +@_auto_client +@validate_call(config=VALIDATE_CONFIG) +def createSignedURLSetJob( + dataset_id: CloudId, + document_id: NonEmptyStr, + *, + file_series: str = "", + id_namespace: Literal["cloud", "ndi"] = "cloud", + client: _Client = None, +) -> dict[str, Any]: + """Kick off an async job that builds a document's signed URL set. + + POSTs to /datasets/{datasetId}/.../signed-url-set-jobs. The server signs + every file the document references, gzips the resulting UID -> signed + URL map and writes it to S3, then hands back a job id. Poll with + :func:`getSignedURLSetJob` (or :func:`waitForSignedURLSetJob`) until the + job reaches state ``"ready"``, then read the map with + :func:`getSignedURLSetResult`. + + Preferred over :func:`getSignedURLSetAll` for documents with thousands + of members: one API call per document instead of one per 500-uid page. + + Args: + dataset_id: The cloud dataset id. + document_id: The document id. Namespace controlled by *id_namespace*. + file_series: If set, restrict the job to one file series' members. + id_namespace: ``"cloud"`` (default) sends the mongo ``_id`` to the + by-``_id`` route; ``"ndi"`` sends ``data.base.id`` to the + ndi-documents route. NDI's own callers hold NDI ids and must + pass ``"ndi"``; passing one as ``"cloud"`` is a 404. See + NDI-matlab#968. + client: Authenticated cloud client (auto-created if omitted). + + Returns: + Dict with (at least) ``jobId``, ``datasetId``, ``documentId``, + ``statusUrl``, ``pollAfterSec``. Pass ``jobId`` to the wait + helpers below. + + MATLAB equivalent: +cloud/+api/+files/createSignedURLSetJob.m + """ + if id_namespace == "ndi": + endpoint = "/datasets/{datasetId}/ndi-documents/{ndiDocumentId}/signed-url-set-jobs" + path_params = {"datasetId": dataset_id, "ndiDocumentId": document_id} + else: + endpoint = "/datasets/{datasetId}/documents/{documentId}/signed-url-set-jobs" + path_params = {"datasetId": dataset_id, "documentId": document_id} + + # The MATLAB implementation appends fileSeries as a query parameter on + # the URL; the client's post() takes path params, not query params, so + # bake it into the endpoint template here. + if file_series: + endpoint = endpoint + "?fileSeries={fileSeries}" + path_params["fileSeries"] = file_series + + # MATLAB posts an empty JSON object rather than a zero-byte body so the + # gateway does not reject the POST. Mirror that here. + return client.post(endpoint, json={}, **path_params) + + +class SignedURLSetJobFailed(RuntimeError): + """A signed-URL-set job reached state ``"failed"``. + + The ``.status`` attribute carries the last status dict from the server, + so a caller can inspect the ``error`` field. + """ + + def __init__(self, status: dict[str, Any]): + err = status.get("error", "") if isinstance(status, dict) else "" + super().__init__( + f"signed-URL-set job {status.get('jobId', '?')} failed: {err}" + if err + else f"signed-URL-set job {status.get('jobId', '?')} failed" + ) + self.status = status + + +@_auto_client +@validate_call(config=VALIDATE_CONFIG) +def getSignedURLSetJob( + job_id: NonEmptyStr, + *, + client: _Client = None, +) -> dict[str, Any]: + """GET /signed-url-set-jobs/{jobId} -- one poll of a signed-URL-set job. + + Returns a dict with (at least) ``jobId``, ``datasetId``, ``documentId``, + ``state``, ``createdAt``, ``startedAt``, ``completedAt``, + ``heartbeatAt``, ``signedCount``, ``totalCount``. When ``state`` is + ``"ready"`` the payload also carries ``resultUrl`` (a 24 h presigned + GET for the gzipped JSON blob), ``resultByteSize``, ``expiresAt``, and + ``filesExpireAt``. When ``state`` is ``"failed"`` it carries + ``error``. Non-terminal states are ``"queued"`` and ``"running"``. + + MATLAB equivalent: +cloud/+api/+files/getSignedURLSetJob.m + """ + return client.get("/signed-url-set-jobs/{jobId}", jobId=job_id) + + +@_auto_client +@validate_call(config=VALIDATE_CONFIG) +def waitForSignedURLSetJob( + job_id: NonEmptyStr, + *, + timeout: float = 300.0, + initial_interval: float = 3.0, + max_interval: float = 30.0, + backoff_factor: float = 2.0, + client: _Client = None, +) -> dict[str, Any]: + """Poll a signed-URL-set job until it finishes or times out. + + Repeatedly calls :func:`getSignedURLSetJob` at exponentially growing + intervals until the job reaches a terminal state (``"ready"`` or + ``"failed"``) or the overall timeout elapses. A transient API failure + is NOT treated as terminal -- a gateway blip would otherwise be + mistaken for a dead job, mirroring :func:`waitForBulkUpload` and + :func:`waitForFileTierJob`. + + Args: + job_id: The signed-URL-set job identifier from + :func:`createSignedURLSetJob`. + timeout: Overall deadline in seconds. Default 300. + initial_interval: First sleep between polls (s). Default 3. + max_interval: Cap on the per-poll sleep (s). Default 30. + backoff_factor: Multiplier applied after each poll. Default 2. + + Returns: + The last status dict from the server. On timeout, the returned + dict has ``state='timeout'`` and ``elapsed`` set to the + wall-clock seconds spent polling. + + MATLAB equivalent: +cloud/+api/+files/waitForSignedURLSetJob.m + """ + start = time.monotonic() + interval = initial_interval + last: Any = None + while True: + elapsed = time.monotonic() - start + try: + status = getSignedURLSetJob(job_id, client=client) + last = status + state = status.get("state", "") if hasattr(status, "get") else "" + if state in _TERMINAL_SIGNED_URL_SET_STATES: + return status + except Exception: + # A failed poll is not a failed job; ride out gateway blips. + pass + if elapsed + interval > timeout: + payload: dict[str, Any] + if last is not None and hasattr(last, "data") and isinstance(last.data, dict): + payload = dict(last.data) + elif isinstance(last, dict): + payload = dict(last) + else: + payload = {} + payload["state"] = "timeout" + payload["elapsed"] = time.monotonic() - start + return payload + time.sleep(interval) + interval = min(interval * backoff_factor, max_interval) + + +@validate_call +def getSignedURLSetResult( + result_url: NonEmptyStr, + *, + timeout: int = 120, +) -> dict[str, Any]: + """Download and parse the blob a ready signed-URL-set job produced. + + A ready job's status carries a ``resultUrl``: a presigned 24 h GET for + a gzipped JSON body of shape:: + + { jobId, datasetId, documentId, generatedAt, fileCount, + files: {uid: url, ...} } + + This fetches that blob and returns a dict with (at least) the ``files`` + map. Whether the body arrives gzipped or already inflated by the + transport is decided from the gzip magic number rather than assumed -- + that mirrors the MATLAB implementation and covers both S3 responses + (raw bytes) and any gateway that auto-inflates on the way through. + + The URL is presigned, so no Authorization header is added; adding one + makes S3 reject the request. + + Args: + result_url: The ``resultUrl`` a ready :func:`getSignedURLSetJob` + response carries. + timeout: HTTP timeout in seconds for the download. Default 120. + + Returns: + Dict with (at least) ``files`` (a ``dict[str, str]`` mapping file + uid to presigned download URL). Also carries ``fileCount``, + ``generatedAt`` and any other fields the server included, so the + signer can hand the whole payload back to + :class:`ndi.cloud.batch_signed_url.BatchSignedUrlLookup` without + reshaping. + + Raises: + RuntimeError: If the download failed or the body could not be + parsed as JSON. + + MATLAB equivalent: +cloud/+api/+files/getSignedURLSetResult.m + """ + import gzip + import json + + assert_safe_transfer_url(result_url, what="signed-URL-set result URL") + + resp = _download_session().get(result_url, timeout=timeout) + if resp.status_code != 200: + raise RuntimeError( + f"signed-URL-set result download failed (HTTP {resp.status_code}): " + f"{resp.text[:200]}" + ) + + raw = resp.content + # Decide gzip vs. inflated by the magic number rather than the + # Content-Encoding header, because requests auto-decompresses on some + # Content-Encoding values and leaves others alone. The MATLAB port + # takes the same approach for the same reason. + if len(raw) >= 2 and raw[0] == 0x1F and raw[1] == 0x8B: + try: + body = gzip.decompress(raw).decode("utf-8") + except OSError as exc: + raise RuntimeError( + f"signed-URL-set result blob could not be decompressed: {exc}" + ) from exc + else: + body = raw.decode("utf-8") + + try: + data = json.loads(body) + except json.JSONDecodeError as exc: + raise RuntimeError(f"signed-URL-set result blob is not valid JSON: {exc}") from exc + + if not isinstance(data, dict): + raise RuntimeError( + f"signed-URL-set result blob decoded to a {type(data).__name__}, " + "expected an object with a 'files' field." + ) + + files = data.get("files") + if not isinstance(files, dict): + raise RuntimeError( + f"signed-URL-set result 'files' arrived as a " f"{type(files).__name__}, not a dict." + ) + + return data + + @_auto_client @validate_call(config=VALIDATE_CONFIG) def getFileCollectionUploadURL( diff --git a/src/ndi/cloud/api/ndi_matlab_python_bridge.yaml b/src/ndi/cloud/api/ndi_matlab_python_bridge.yaml index 0f991cea..358b9b09 100644 --- a/src/ndi/cloud/api/ndi_matlab_python_bridge.yaml +++ b/src/ndi/cloud/api/ndi_matlab_python_bridge.yaml @@ -1586,9 +1586,10 @@ not_yet_ported: # +ndi/+cloud/+api/+files -- signed-URL-set family # # A coherent group of six MATLAB functions added by NDI-matlab's - # signed-URL-set work (PRs #967, #969, #970 and preceding). Together they - # cover the /signed-url-set (paged) and /signed-url-set-jobs (async) - # routes the cloud exposes for bulk pre-signed downloads: + # signed-URL-set work (PRs #967, #969, #970 and preceding, and #1009 + # which stabilized the async job path). Together they cover the + # /signed-url-set (paged) and /signed-url-set-jobs (async) routes the + # cloud exposes for bulk pre-signed downloads: # # getSignedURLSet - one page of a document's UID -> signed URL map # getSignedURLSetAll - walk every page and merge into a single map @@ -1598,26 +1599,36 @@ not_yet_ported: # waitForSignedURLSetJob- poll with exponential backoff until terminal # getSignedURLSetResult - GET + parse the blob a ready job produced # - # This is what fetching a 28,000-member file-series document over the - # cloud requires (VH-Lab/NDI-matlab#952) -- one API call per document - # instead of one per member. Python currently has no signed-URL-set - # support at all (grep -r "signed_url_set\|SignedURLSet" src/ndi/ is - # empty), so the honest status for all six is not_yet_ported. Recorded - # together with one shared decision_log because there is no case-by-case - # judgement to make: the family lands as a unit or not at all. + # This is what fetching a document with tens or hundreds of thousands of + # members over the cloud requires (VH-Lab/NDI-matlab#952) -- one API + # call per document instead of one per member. # - # Filed as Waltham-Data-Science/ndi-python#206. + # STATUS. Ported as of NDI-python#206. The four async-job functions were + # added to src/ndi/cloud/api/files.py alongside the paged pair, and + # batch_signed_url._default_signer now submits a job via + # createSignedURLSetJob -> waitForSignedURLSetJob -> + # getSignedURLSetResult instead of walking pages. For a 121k-uid + # document the walk took 100+ minutes at ~25 s per 500-uid page; the + # async job builds the whole map server-side and returns it in one + # blob. The paged pair is still exported as an API entry point (a + # smaller document can still walk it), but no NDI code path uses it any + # longer. Filed as Waltham-Data-Science/ndi-python#206. # ======================================================================= - name: createSignedURLSetJob matlab_path: "+ndi/+cloud/+api/+files/createSignedURLSetJob.m" - matlab_last_sync_hash: "5804cd80" - status: porting_deferred + python_path: "ndi/cloud/api/files.py" + matlab_last_sync_hash: "0a2cdec" + status: ported decision_log: > - Part of the signed-URL-set family (#206). Python has no - signed-URL-set support yet, so createSignedURLSetJob has no Python - counterpart. Ports as a unit with the other five entries under this - heading; see the shared note above. + Ported by NDI-python#206 as part of the signed-URL-set family. POSTs + to /datasets/{d}/documents/{doc}/signed-url-set-jobs (or the + ndi-documents route when id_namespace='ndi', which is what NDI's own + callers use). Optional file_series is folded into the endpoint as a + query parameter; MATLAB does the same via + matlab.net.QueryParameter. Returns the accepted-job envelope + ({jobId, datasetId, documentId, statusUrl, pollAfterSec}) so the + caller can hand jobId to waitForSignedURLSetJob. - name: getSignedURLSet matlab_path: "+ndi/+cloud/+api/+files/getSignedURLSet.m" @@ -1645,36 +1656,52 @@ not_yet_ported: no nextCursor or empty nextCursor ends the walk; a cursor that equals the one just sent raises SignedURLSetCursorDidNotAdvance; running past max_pages raises SignedURLSetMaxPagesReached with the - partial merged dict on the exception. The async job path - (createSignedURLSetJob + waitForSignedURLSetJob) is still deferred - because batch_signed_url does not use it. + partial merged dict on the exception. Still exported as an API entry + point but no longer the batch_signed_url signer: at ~25 s per + 500-uid page a 121k-member document took 100+ minutes here, so the + default signer now goes through the async job path + (createSignedURLSetJob + waitForSignedURLSetJob + + getSignedURLSetResult). - name: getSignedURLSetJob matlab_path: "+ndi/+cloud/+api/+files/getSignedURLSetJob.m" - matlab_last_sync_hash: "cd11b198" - status: porting_deferred + python_path: "ndi/cloud/api/files.py" + matlab_last_sync_hash: "0a2cdec" + status: ported decision_log: > - Part of the signed-URL-set family (#206). Poll one signed-URL-set - job's state via GET /signed-url-set-jobs/{jobId}. Python has no - counterpart; ports as a unit, see the shared note above. + Ported by NDI-python#206 as part of the signed-URL-set family. GETs + /signed-url-set-jobs/{jobId}. Returns the job's status payload + verbatim, including state (one of 'queued', 'running', 'ready', + 'failed'), signedCount/totalCount, and -- on 'ready' -- resultUrl, + resultByteSize, expiresAt and filesExpireAt. - name: getSignedURLSetResult matlab_path: "+ndi/+cloud/+api/+files/getSignedURLSetResult.m" - matlab_last_sync_hash: "4b5c0478" - status: porting_deferred + python_path: "ndi/cloud/api/files.py" + matlab_last_sync_hash: "0a2cdec" + status: ported decision_log: > - Part of the signed-URL-set family (#206). Downloads and parses the - gzipped blob a ready job produced (a UID -> signed URL map plus - count and generation timestamp). Python has no counterpart; ports - as a unit, see the shared note above. + Ported by NDI-python#206 as part of the signed-URL-set family. + Downloads the presigned resultUrl a ready job carries, detects gzip + vs. inflated body from the magic number (0x1f 0x8b) rather than + Content-Encoding -- some transports auto-inflate, and MATLAB takes + the same approach for the same reason -- and returns the parsed + dict verbatim ({files: {uid -> url}, fileCount, generatedAt, ...}) + so the batch_signed_url signer can hand the whole payload through + without reshaping. MATLAB wraps files in a containers.Map; Python's + json.loads keeps arbitrary string keys as a dict, so no key-recovery + pass is needed. - name: waitForSignedURLSetJob matlab_path: "+ndi/+cloud/+api/+files/waitForSignedURLSetJob.m" - matlab_last_sync_hash: "cd11b198" - status: porting_deferred + python_path: "ndi/cloud/api/files.py" + matlab_last_sync_hash: "0a2cdec" + status: ported decision_log: > - Part of the signed-URL-set family (#206). Polls - getSignedURLSetJob with exponentially growing intervals until the - job reaches a terminal state ('ready' or 'failed') or the overall - timeout elapses. Python has no counterpart; ports as a unit, see - the shared note above. + Ported by NDI-python#206 as part of the signed-URL-set family. + Polls getSignedURLSetJob at exponentially growing intervals until + the job reaches state 'ready' or 'failed', or the overall timeout + elapses. Mirrors waitForBulkUpload/waitForFileTierJob: on timeout + returns the last status dict with state='timeout' and elapsed set, + and a transient API failure is NOT treated as terminal so a gateway + blip does not mark the job dead. diff --git a/src/ndi/cloud/batch_signed_url.py b/src/ndi/cloud/batch_signed_url.py index 5804af0e..ee7c88b8 100644 --- a/src/ndi/cloud/batch_signed_url.py +++ b/src/ndi/cloud/batch_signed_url.py @@ -8,10 +8,15 @@ many members -- 28,000 in the lightsheet-pyramid case that motivates this (VH-Lab/NDI-matlab#952) -- it is one API round trip per member. -The batch path: one call to /signed-url-set (walked with getSignedURLSetAll) -returns the whole document's uid -> URL map. Subsequent uids in the same -(dataset, document [, series]) scope resolve from the cached map without -another round trip. +The batch path: submit an async signed-URL-set job (createSignedURLSetJob +-> waitForSignedURLSetJob -> getSignedURLSetResult) that builds the whole +document's uid -> URL map server-side and hands it back in one gzipped +blob. Subsequent uids in the same (dataset, document [, series]) scope +resolve from the cached map without another round trip. The paged +getSignedURLSetAll walk is still supported as an API entry point but the +default signer no longer uses it -- for a 121k-uid document the walk +takes 100+ minutes at ~25 s per 500-uid page (NDI-matlab#952, +NDI-matlab#1009). Empty return values are legitimate answers, not errors: the batch call may fail (network, auth, timeout) or the map it returns may not name this uid @@ -122,7 +127,17 @@ class Stats: partial_map_retries: int = 0 -# The default signer walks every page. Injected for tests. +# Default timeout for the async signed-URL-set job. The server has to sign +# every member and gzip the resulting map, so a 121k-uid document is not a +# short wait -- 15 min gives it room while still surfacing a truly stuck +# job. The batch cache's caller (fetch_cloud_file) falls back per-member if +# the wait times out, so this is a "give up on the fast path" deadline, +# not a "give up on the read" one. +DEFAULT_JOB_TIMEOUT_SECONDS = 15 * 60 + + +# The default signer submits the async job, waits for it to reach 'ready', +# and reads the gzipped result blob. Injected for tests. def _default_signer( dataset_id: str, document_id: str, @@ -131,19 +146,29 @@ def _default_signer( client: CloudClient | None = None, retry_delays: tuple[float, ...] = DEFAULT_BATCH_RETRY_DELAYS, sleep: Callable[[float], None] = time.sleep, + job_timeout: float = DEFAULT_JOB_TIMEOUT_SECONDS, ) -> tuple[bool, dict[str, Any]]: - """Fetch the whole scope through the paged getSignedURLSetAll call. - - Returns ``(True, answer)`` on success, ``(False, {})`` on any failure -- - the batch is best-effort and its caller's fallback is what keeps the - read correct. - - A raise from ``getSignedURLSetAll`` is retried with the delays in - ``retry_delays`` before it is reported as a failure. This is what keeps - one transient TLS blip on a residential connection from cascading to - the O(N) per-member fallback: the batch itself gets a few more chances, - at ~20 s of wall clock in the worst case, before its caller is told the - scope is unreachable. See Waltham-Data-Science/NDI-python#322. + """Fetch the whole scope through the async signed-URL-set-job path. + + Submits a ``createSignedURLSetJob`` for the (dataset, document, series) + scope, waits for the job to reach state ``"ready"`` (or ``"failed"``, + or the overall ``job_timeout``), and downloads and parses the gzipped + result blob. Returns ``(True, answer)`` on success and + ``(False, {"__error__": ...})`` on any failure -- the batch is + best-effort and its caller's fallback keeps the read correct. + + Preferred over the paged ``getSignedURLSetAll`` walk: for a document + with 121k members the walk takes 100+ minutes at ~25 s per 500-uid + page, whereas the async job builds the whole map server-side and + returns it in one blob (NDI-matlab#952, NDI-matlab#1009). + + A raise anywhere in ``create -> wait -> read`` is retried with the + delays in ``retry_delays`` before it is reported as a failure. This is + what keeps one transient TLS blip on a residential connection from + cascading to the O(N) per-member fallback: the batch itself gets a few + more chances at ~20 s of wall clock in the worst case before its + caller is told the scope is unreachable. See + Waltham-Data-Science/NDI-python#322. """ from .api import files as files_api @@ -151,20 +176,58 @@ def _default_signer( last_exc: Exception | None = None for attempt in range(attempts): try: - answer = files_api.getSignedURLSetAll( + job = files_api.createSignedURLSetJob( dataset_id, document_id, id_namespace="ndi", file_series=file_series, client=client, ) + job_id = job.get("jobId", "") if hasattr(job, "get") else "" + if not job_id: + raise RuntimeError(f"createSignedURLSetJob returned no jobId (payload: {job!r})") + + status = files_api.waitForSignedURLSetJob( + job_id, + timeout=job_timeout, + client=client, + ) + state = status.get("state", "") if hasattr(status, "get") else "" + if state == "failed": + err = status.get("error", "") if hasattr(status, "get") else "" + raise RuntimeError( + f"signed-URL-set job {job_id} failed: {err}" + if err + else f"signed-URL-set job {job_id} failed" + ) + if state == "timeout": + elapsed = ( + status.get("elapsed", job_timeout) if hasattr(status, "get") else job_timeout + ) + raise RuntimeError( + f"signed-URL-set job {job_id} did not finish within " + f"{elapsed:.0f}s (state after wait: {state!r})" + ) + if state != "ready": + raise RuntimeError( + f"signed-URL-set job {job_id} ended in unexpected state " + f"{state!r} (expected 'ready')" + ) + + result_url = status.get("resultUrl", "") if hasattr(status, "get") else "" + if not result_url: + raise RuntimeError( + f"signed-URL-set job {job_id} was ready but carried no resultUrl" + ) + + answer = files_api.getSignedURLSetResult(result_url) return True, answer except Exception as exc: # noqa: BLE001 - reported as a miss reason last_exc = exc if attempt < attempts - 1: delay = retry_delays[attempt] logger.info( - "batch signed-URL fetch attempt %d/%d failed (%s); " "retrying in %.1fs", + "batch signed-URL fetch attempt %d/%d failed (%s); retrying in %.1fs", attempt + 1, attempts, type(exc).__name__, diff --git a/src/ndi/cloud/ndi_matlab_python_bridge.yaml b/src/ndi/cloud/ndi_matlab_python_bridge.yaml index 80046115..ef25ae55 100644 --- a/src/ndi/cloud/ndi_matlab_python_bridge.yaml +++ b/src/ndi/cloud/ndi_matlab_python_bridge.yaml @@ -1512,10 +1512,6 @@ not_yet_ported: NEXT uid in a sweep is suppressed, a caller re-asking for the same uid gets a fresh attempt. - Skips the async job path (createSignedURLSetJob + - waitForSignedURLSetJob) since the batch helper does not use it; - those remain porting_deferred in the api YAML. - NDI-matlab 9120c002 adds a one-shot partial-map retry: when a cached scope answers with a URL but a subsequent uid in that scope misses, drop the entry, pause, and re-fetch once @@ -1524,6 +1520,17 @@ not_yet_ported: (_partial_map_retry_seconds / STATS.partial_map_retries in ndi/cloud/batch_signed_url.py), so this is a hash-only bump. + NDI-python#206: _default_signer switched from getSignedURLSetAll + (paged, ~25 s per 500-uid page) to the async signed-URL-set-job + family (createSignedURLSetJob -> waitForSignedURLSetJob -> + getSignedURLSetResult). For a 121k-uid document the paged walk + took 100+ minutes; the async job builds the whole map server-side + and returns it in one gzipped blob. The signer contract is + unchanged -- still returns (True, {"files": {uid: url}, ...}) on + success and (False, {"__error__": ...}) on failure -- and the + transient-error retry (NDI-python#322) still wraps the whole + three-step exchange. + - name: buildGenericFileDownloadList matlab_path: "+ndi/+cloud/+download/+internal/buildGenericFileDownloadList.m" matlab_last_sync_hash: "65aeeeb" diff --git a/tests/test_cloud_api_signed_url_set_job.py b/tests/test_cloud_api_signed_url_set_job.py new file mode 100644 index 00000000..e392d224 --- /dev/null +++ b/tests/test_cloud_api_signed_url_set_job.py @@ -0,0 +1,382 @@ +"""Unit tests for the async signed-URL-set-job family (NDI-python#206). + +Covers the four new ``ndi.cloud.api.files`` entry points: + +* :func:`createSignedURLSetJob` -- POST /.../signed-url-set-jobs. +* :func:`getSignedURLSetJob` -- GET /signed-url-set-jobs/{jobId}. +* :func:`waitForSignedURLSetJob`-- exponential-backoff poll to a terminal + state or timeout. +* :func:`getSignedURLSetResult` -- download and parse the gzipped result + blob a ready job produced. + +The live round trip against a real cloud dataset lives elsewhere; this +file is the offline half, driven by mocked clients and a scripted result +URL, so no credentials are required. +""" + +from __future__ import annotations + +import gzip +import json +from unittest.mock import MagicMock, patch + +import pytest + +# --------------------------------------------------------------------------- +# createSignedURLSetJob +# --------------------------------------------------------------------------- + + +class TestCreateSignedURLSetJob: + def test_posts_to_cloud_id_route_by_default(self): + """id_namespace='cloud' (default) targets the by-_id route.""" + from ndi.cloud.api import files as files_api + + client = MagicMock() + client.post.return_value = {"jobId": "j1", "datasetId": "ds1", "documentId": "doc1"} + + result = files_api.createSignedURLSetJob("ds1", "doc1", client=client) + + assert result == {"jobId": "j1", "datasetId": "ds1", "documentId": "doc1"} + client.post.assert_called_once() + args, kwargs = client.post.call_args + assert args[0] == "/datasets/{datasetId}/documents/{documentId}/signed-url-set-jobs" + assert kwargs.get("datasetId") == "ds1" + assert kwargs.get("documentId") == "doc1" + # An empty JSON body, matching MATLAB, so a gateway that rejects + # zero-byte POSTs still accepts the call. + assert kwargs.get("json") == {} + + def test_ndi_namespace_targets_ndi_documents_route(self): + """id_namespace='ndi' is what NDI's own callers use.""" + from ndi.cloud.api import files as files_api + + client = MagicMock() + client.post.return_value = {"jobId": "j2"} + + files_api.createSignedURLSetJob("ds1", "doc1", id_namespace="ndi", client=client) + + args, _ = client.post.call_args + assert args[0] == "/datasets/{datasetId}/ndi-documents/{ndiDocumentId}/signed-url-set-jobs" + # The path param name changes with the route (documentId -> + # ndiDocumentId), mirroring MATLAB's endpointName switch. + _, kwargs = client.post.call_args + assert kwargs.get("ndiDocumentId") == "doc1" + + def test_file_series_folds_into_the_endpoint(self): + """Optional fileSeries becomes a query parameter on the URL.""" + from ndi.cloud.api import files as files_api + + client = MagicMock() + client.post.return_value = {"jobId": "j3"} + + files_api.createSignedURLSetJob( + "ds1", "doc1", file_series="level_a", id_namespace="ndi", client=client + ) + + args, kwargs = client.post.call_args + assert "fileSeries={fileSeries}" in args[0] + assert kwargs.get("fileSeries") == "level_a" + + +# --------------------------------------------------------------------------- +# getSignedURLSetJob +# --------------------------------------------------------------------------- + + +class TestGetSignedURLSetJob: + def test_calls_the_jobs_endpoint(self): + from ndi.cloud.api import files as files_api + + client = MagicMock() + client.get.return_value = {"jobId": "j1", "state": "running", "signedCount": 42} + + result = files_api.getSignedURLSetJob("j1", client=client) + + assert result["state"] == "running" + args, kwargs = client.get.call_args + assert args[0] == "/signed-url-set-jobs/{jobId}" + assert kwargs.get("jobId") == "j1" + + +# --------------------------------------------------------------------------- +# waitForSignedURLSetJob +# --------------------------------------------------------------------------- + + +class TestWaitForSignedURLSetJob: + def test_returns_ready_terminal_state(self): + """A job that goes queued -> running -> ready terminates on ready.""" + from ndi.cloud.api import files as files_api + + states = [ + {"state": "queued"}, + {"state": "running", "signedCount": 500}, + {"state": "ready", "resultUrl": "https://example/result", "fileCount": 1000}, + ] + + with ( + patch("ndi.cloud.api.files.getSignedURLSetJob", side_effect=states), + patch("ndi.cloud.api.files.time.sleep", lambda _s: None), + ): + result = files_api.waitForSignedURLSetJob( + "j1", + initial_interval=0.01, + max_interval=0.01, + timeout=10.0, + client=MagicMock(), + ) + + assert result["state"] == "ready" + assert result["resultUrl"] == "https://example/result" + + def test_returns_failed_terminal_state(self): + """A job that goes to failed returns the failure payload verbatim.""" + from ndi.cloud.api import files as files_api + + with patch( + "ndi.cloud.api.files.getSignedURLSetJob", + return_value={"state": "failed", "error": "boom"}, + ): + result = files_api.waitForSignedURLSetJob( + "j1", initial_interval=0.01, timeout=1.0, client=MagicMock() + ) + + assert result["state"] == "failed" + assert result["error"] == "boom" + + def test_timeout_returns_state_timeout(self): + """When the deadline elapses, state='timeout' and elapsed is set.""" + from ndi.cloud.api import files as files_api + + with ( + patch( + "ndi.cloud.api.files.getSignedURLSetJob", + return_value={"state": "running", "signedCount": 10}, + ), + patch("ndi.cloud.api.files.time.sleep", lambda _s: None), + ): + result = files_api.waitForSignedURLSetJob( + "j1", + initial_interval=1.0, + max_interval=1.0, + timeout=0.01, # basically immediate + client=MagicMock(), + ) + + assert result["state"] == "timeout" + assert "elapsed" in result + + def test_transient_raise_is_not_terminal(self): + """A gateway blip during a poll does not mark the job dead.""" + from ndi.cloud.api import files as files_api + + seq = [ + ConnectionError("blip"), + {"state": "running"}, + {"state": "ready", "resultUrl": "https://example/r"}, + ] + + def poll(*args, **kwargs): + v = seq.pop(0) + if isinstance(v, Exception): + raise v + return v + + with ( + patch("ndi.cloud.api.files.getSignedURLSetJob", side_effect=poll), + patch("ndi.cloud.api.files.time.sleep", lambda _s: None), + ): + result = files_api.waitForSignedURLSetJob( + "j1", + initial_interval=0.01, + max_interval=0.01, + timeout=10.0, + client=MagicMock(), + ) + + assert result["state"] == "ready" + + +# --------------------------------------------------------------------------- +# getSignedURLSetResult +# --------------------------------------------------------------------------- + + +def _make_response(body: bytes, *, status: int = 200): + resp = MagicMock() + resp.status_code = status + resp.content = body + resp.text = body.decode("utf-8", errors="replace") + return resp + + +class TestGetSignedURLSetResult: + def test_parses_gzipped_blob(self): + """A gzipped payload (typical S3 response) is inflated then parsed.""" + from ndi.cloud.api import files as files_api + + payload = { + "jobId": "j1", + "fileCount": 2, + "generatedAt": "2026-09-27T00:00:00Z", + "files": {"u1": "https://s3.example/u1", "u2": "https://s3.example/u2"}, + } + body_gz = gzip.compress(json.dumps(payload).encode("utf-8")) + + session = MagicMock() + session.get.return_value = _make_response(body_gz) + with patch("ndi.cloud.api.files._download_session", return_value=session): + result = files_api.getSignedURLSetResult("https://example/result.gz") + + assert result["files"] == payload["files"] + assert result["fileCount"] == 2 + assert result["generatedAt"] == "2026-09-27T00:00:00Z" + + def test_parses_uncompressed_blob(self): + """A body the transport already inflated is parsed as-is.""" + from ndi.cloud.api import files as files_api + + payload = {"files": {"u1": "https://s3/u1"}, "fileCount": 1} + body = json.dumps(payload).encode("utf-8") + + session = MagicMock() + session.get.return_value = _make_response(body) + with patch("ndi.cloud.api.files._download_session", return_value=session): + result = files_api.getSignedURLSetResult("https://example/result") + + assert result["files"] == payload["files"] + + def test_http_error_raises(self): + from ndi.cloud.api import files as files_api + + session = MagicMock() + session.get.return_value = _make_response(b"AccessDenied", status=403) + with patch("ndi.cloud.api.files._download_session", return_value=session): + with pytest.raises(RuntimeError, match="HTTP 403"): + files_api.getSignedURLSetResult("https://example/result") + + def test_files_field_wrong_type_raises(self): + """A payload whose 'files' isn't a dict is a caller-visible failure.""" + from ndi.cloud.api import files as files_api + + payload = {"files": "not-a-dict"} + session = MagicMock() + session.get.return_value = _make_response(json.dumps(payload).encode("utf-8")) + with patch("ndi.cloud.api.files._download_session", return_value=session): + with pytest.raises(RuntimeError, match="'files' arrived"): + files_api.getSignedURLSetResult("https://example/result") + + +# --------------------------------------------------------------------------- +# batch_signed_url _default_signer -- end-to-end wiring through the new path +# --------------------------------------------------------------------------- + + +class TestDefaultSignerHappyPath: + """A single happy pass through create -> wait -> read must produce the + same (True, {files: ...}) shape the paged signer used to. + """ + + def test_returns_the_files_map_when_the_job_is_ready(self): + from ndi.cloud.batch_signed_url import _default_signer + + files_map = {"u1": "https://s3/u1", "u2": "https://s3/u2"} + + with ( + patch( + "ndi.cloud.api.files.createSignedURLSetJob", + return_value={"jobId": "j1"}, + ) as m_create, + patch( + "ndi.cloud.api.files.waitForSignedURLSetJob", + return_value={"state": "ready", "resultUrl": "https://example/r"}, + ) as m_wait, + patch( + "ndi.cloud.api.files.getSignedURLSetResult", + return_value={"files": files_map, "fileCount": 2}, + ) as m_read, + ): + ok, answer = _default_signer( + "ds1", + "doc1", + file_series="level_a", + client=MagicMock(), + retry_delays=(), + ) + + assert ok is True + assert answer["files"] == files_map + m_create.assert_called_once() + m_wait.assert_called_once() + m_read.assert_called_once_with("https://example/r") + + # The create call is what carries the NDI namespace and the series + # scope; verify those explicitly so a future edit that drops the + # id_namespace='ndi' argument fails here. + _, create_kwargs = m_create.call_args + assert create_kwargs.get("id_namespace") == "ndi" + assert create_kwargs.get("file_series") == "level_a" + + def test_failed_state_reports_the_server_error(self): + from ndi.cloud.batch_signed_url import _default_signer + + with ( + patch( + "ndi.cloud.api.files.createSignedURLSetJob", + return_value={"jobId": "j2"}, + ), + patch( + "ndi.cloud.api.files.waitForSignedURLSetJob", + return_value={"state": "failed", "error": "signer crashed"}, + ), + ): + ok, answer = _default_signer("ds1", "doc1", client=MagicMock(), retry_delays=()) + + assert ok is False + error = answer.get("__error__", "") + assert "signer crashed" in error, f"failure must name the server error: {error!r}" + + def test_wait_timeout_is_reported_as_failure(self): + from ndi.cloud.batch_signed_url import _default_signer + + with ( + patch( + "ndi.cloud.api.files.createSignedURLSetJob", + return_value={"jobId": "j3"}, + ), + patch( + "ndi.cloud.api.files.waitForSignedURLSetJob", + return_value={"state": "timeout", "elapsed": 900.0}, + ), + ): + ok, answer = _default_signer("ds1", "doc1", client=MagicMock(), retry_delays=()) + + assert ok is False + error = answer.get("__error__", "") + assert ( + "timeout" in error.lower() or "did not finish" in error + ), f"failure must name the timeout: {error!r}" + + def test_ready_without_result_url_is_reported_as_failure(self): + """A ready job that carries no resultUrl is a server bug we must not + follow into an assertion-free read. See NDI-matlab#1009 for the + same guard on the MATLAB side. + """ + from ndi.cloud.batch_signed_url import _default_signer + + with ( + patch( + "ndi.cloud.api.files.createSignedURLSetJob", + return_value={"jobId": "j4"}, + ), + patch( + "ndi.cloud.api.files.waitForSignedURLSetJob", + return_value={"state": "ready"}, # no resultUrl + ), + ): + ok, answer = _default_signer("ds1", "doc1", client=MagicMock(), retry_delays=()) + + assert ok is False + error = answer.get("__error__", "") + assert "resultUrl" in error, f"failure must name the missing field: {error!r}" diff --git a/tests/test_cloud_batch_signed_url.py b/tests/test_cloud_batch_signed_url.py index f3be39d3..604212e8 100644 --- a/tests/test_cloud_batch_signed_url.py +++ b/tests/test_cloud_batch_signed_url.py @@ -616,11 +616,20 @@ def unique_cursor(*args, **kwargs): class TestDefaultSignerRetries: - """A ConnectionFailed-style raise from getSignedURLSetAll must not - collapse a whole scope to the O(N) per-member fallback on the first - try. See Waltham-Data-Science/NDI-python#322 and the parallel + """A ConnectionFailed-style raise from the batch API must not collapse + a whole scope to the O(N) per-member fallback on the first try. See + Waltham-Data-Science/NDI-python#322 and the parallel VH-Lab/NDI-matlab#1010: a residential-network TLS blip used to hang downloads for hours by tripping this cascade. + + NDI-python#206 rewired the default signer from the paged + ``getSignedURLSetAll`` walk to the async job path + (``createSignedURLSetJob`` -> ``waitForSignedURLSetJob`` -> + ``getSignedURLSetResult``). The retry contract is unchanged: any raise + from the three-step exchange is retried per ``retry_delays``. These + tests patch the first step (``createSignedURLSetJob``) to trip the + retry, which is enough to exercise the loop without also needing to + script the wait and result calls. """ def test_a_transient_raise_then_success_returns_success(self): @@ -628,13 +637,20 @@ def test_a_transient_raise_then_success_returns_success(self): calls = {"n": 0} - def flaky(*args, **kwargs): + def flaky_create(*args, **kwargs): calls["n"] += 1 if calls["n"] == 1: raise ConnectionError("first try, transient") - return {"files": {"u1": "https://s3.example.com/u1"}, "pages": 1} + return {"jobId": "job-1"} + + ready_status = {"state": "ready", "resultUrl": "https://example/result"} + result_payload = {"files": {"u1": "https://s3.example.com/u1"}, "fileCount": 1} - with patch("ndi.cloud.api.files.getSignedURLSetAll", side_effect=flaky): + with ( + patch("ndi.cloud.api.files.createSignedURLSetJob", side_effect=flaky_create), + patch("ndi.cloud.api.files.waitForSignedURLSetJob", return_value=ready_status), + patch("ndi.cloud.api.files.getSignedURLSetResult", return_value=result_payload), + ): ok, answer = _default_signer( "ds1", "doc1", @@ -656,7 +672,7 @@ def always_fails(*args, **kwargs): calls["n"] += 1 raise ConnectionError(f"try {calls['n']}") - with patch("ndi.cloud.api.files.getSignedURLSetAll", side_effect=always_fails): + with patch("ndi.cloud.api.files.createSignedURLSetJob", side_effect=always_fails): ok, answer = _default_signer( "ds1", "doc1", @@ -686,7 +702,7 @@ def always_fails(*args, **kwargs): calls["n"] += 1 raise ConnectionError("nope") - with patch("ndi.cloud.api.files.getSignedURLSetAll", side_effect=always_fails): + with patch("ndi.cloud.api.files.createSignedURLSetJob", side_effect=always_fails): ok, _ = _default_signer( "ds1", "doc1", @@ -707,7 +723,7 @@ def test_retry_delays_are_slept_in_order(self): def always_fails(*args, **kwargs): raise ConnectionError("nope") - with patch("ndi.cloud.api.files.getSignedURLSetAll", side_effect=always_fails): + with patch("ndi.cloud.api.files.createSignedURLSetJob", side_effect=always_fails): _default_signer( "ds1", "doc1", From 6e4ca96298d53ea45d9a487f589e53520b21f808 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 14:54:02 +0000 Subject: [PATCH 11/55] batch_signed_url: raise BatchScopeUnreachable instead of silent per-uid fallback Silent fallback to per-uid getFileDetails hides real bugs in the async signed-URL-set-job path behind slow-but-working chunk fetches. A napari viewport with 100k+ chunks per document is unusable at one API call per chunk (thousands per zoom), and the underlying async-job failure -- endpoint 404, schema drift, timeout, missing resultUrl -- never surfaces. _default_signer now raises BatchScopeUnreachable when every retry attempt of create/wait/read has failed. _fetch_scope propagates it up through BatchSignedUrlLookup.lookup and out to callers (fetch_cloud_file, viewer chunk fetches). The partial-map case (batch returned a map that didn't name a specific uid) stays as a legitimate per-uid fallback -- that's a data-drift path, not a broken endpoint. Injected test signers that raise other exceptions still fall back per the old behavior, so existing failure-mode tests keep working. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/cloud/batch_signed_url.py | 55 ++++++++++++-- tests/test_cloud_api_signed_url_set_job.py | 30 +++----- tests/test_cloud_batch_signed_url.py | 88 ++++++++++++++-------- 3 files changed, 119 insertions(+), 54 deletions(-) diff --git a/src/ndi/cloud/batch_signed_url.py b/src/ndi/cloud/batch_signed_url.py index ee7c88b8..08931894 100644 --- a/src/ndi/cloud/batch_signed_url.py +++ b/src/ndi/cloud/batch_signed_url.py @@ -69,6 +69,32 @@ DEFAULT_BATCH_RETRY_DELAYS: tuple[float, ...] = (1.0, 4.0, 16.0) +class BatchScopeUnreachable(RuntimeError): + """The async signed-URL-set job path could not answer for a scope. + + Raised by :func:`_default_signer` after every retry attempt has failed + -- ``createSignedURLSetJob``, ``waitForSignedURLSetJob`` and + ``getSignedURLSetResult`` between them have not produced a usable map + for this (dataset, document, series) scope. + + Why raise rather than return ``(False, ...)`` and let the caller fall + back to per-member ``getFileDetails``? A napari viewport loads dozens + of chunks per frame, and per-member fallback for a 100k-member series + is thousands of API calls per zoom -- unusable UX. Worse, that + fallback works "well enough" that the underlying bug (the async job + endpoint returned 404 / timed out / drifted from the schema) stays + hidden while the viewer feels merely slow. Raising forces the real + error to the surface so we see and fix it, and never silently ship a + bad path. + + The partial-map case is separate and does NOT raise: a batch that + successfully returned a map but did not happen to name a particular + uid still hits the per-uid fallback in ``fetch_cloud_file``, because + that is a data-drift case rather than a broken endpoint. Only real + failures of the async job path raise here. + """ + + @dataclass class _CacheEntry: """One cached scope: uid -> URL, and when we fetched it.""" @@ -222,7 +248,7 @@ def _default_signer( answer = files_api.getSignedURLSetResult(result_url) return True, answer - except Exception as exc: # noqa: BLE001 - reported as a miss reason + except Exception as exc: # noqa: BLE001 - reported by BatchScopeUnreachable last_exc = exc if attempt < attempts - 1: delay = retry_delays[attempt] @@ -234,10 +260,15 @@ def _default_signer( delay, ) sleep(delay) - # Every attempt raised; report the last cause. - return False, { - "__error__": f"{type(last_exc).__name__}: {last_exc} (after {attempts} attempts)" - } + # Every attempt raised. Surface the last cause instead of returning + # (False, ...) -- see BatchScopeUnreachable's docstring for why a + # silent per-uid fallback would just hide the bug. + raise BatchScopeUnreachable( + f"async signed-URL-set job failed for scope " + f"({dataset_id!r}, {document_id!r}, file_series={file_series!r}) " + f"after {attempts} attempts. Last cause: " + f"{type(last_exc).__name__}: {last_exc}" + ) from last_exc class BatchSignedUrlLookup: @@ -434,7 +465,14 @@ def _fetch_scope( *, client: CloudClient | None, ) -> _CacheEntry | None: - """Populate the cache for one scope. None on any failure.""" + """Populate the cache for one scope. None on any failure. + + :class:`BatchScopeUnreachable` from the signer is NOT caught -- it + means the async job path itself is broken and per-uid fallback + would just hide the real error. Every other exception from an + injected signer is still reported as a miss reason so existing + tests keep working. + """ failure_reason = "" try: ok, answer = self._signer( @@ -443,6 +481,11 @@ def _fetch_scope( file_series=series_name, client=client, ) + except BatchScopeUnreachable: + # Real failure of the async signed-URL-set path; propagate so + # the caller (viewer, download orchestrator, test) sees the + # cause instead of a slow per-uid walk that hides the bug. + raise except Exception as exc: # noqa: BLE001 - reported as a miss reason ok = False answer = {} diff --git a/tests/test_cloud_api_signed_url_set_job.py b/tests/test_cloud_api_signed_url_set_job.py index e392d224..de703a75 100644 --- a/tests/test_cloud_api_signed_url_set_job.py +++ b/tests/test_cloud_api_signed_url_set_job.py @@ -319,7 +319,7 @@ def test_returns_the_files_map_when_the_job_is_ready(self): assert create_kwargs.get("file_series") == "level_a" def test_failed_state_reports_the_server_error(self): - from ndi.cloud.batch_signed_url import _default_signer + from ndi.cloud.batch_signed_url import BatchScopeUnreachable, _default_signer with ( patch( @@ -331,14 +331,11 @@ def test_failed_state_reports_the_server_error(self): return_value={"state": "failed", "error": "signer crashed"}, ), ): - ok, answer = _default_signer("ds1", "doc1", client=MagicMock(), retry_delays=()) - - assert ok is False - error = answer.get("__error__", "") - assert "signer crashed" in error, f"failure must name the server error: {error!r}" + with pytest.raises(BatchScopeUnreachable, match="signer crashed"): + _default_signer("ds1", "doc1", client=MagicMock(), retry_delays=()) def test_wait_timeout_is_reported_as_failure(self): - from ndi.cloud.batch_signed_url import _default_signer + from ndi.cloud.batch_signed_url import BatchScopeUnreachable, _default_signer with ( patch( @@ -350,20 +347,20 @@ def test_wait_timeout_is_reported_as_failure(self): return_value={"state": "timeout", "elapsed": 900.0}, ), ): - ok, answer = _default_signer("ds1", "doc1", client=MagicMock(), retry_delays=()) + with pytest.raises(BatchScopeUnreachable) as exc_info: + _default_signer("ds1", "doc1", client=MagicMock(), retry_delays=()) - assert ok is False - error = answer.get("__error__", "") + message = str(exc_info.value) assert ( - "timeout" in error.lower() or "did not finish" in error - ), f"failure must name the timeout: {error!r}" + "did not finish" in message or "timeout" in message.lower() + ), f"failure must name the timeout: {message!r}" def test_ready_without_result_url_is_reported_as_failure(self): """A ready job that carries no resultUrl is a server bug we must not follow into an assertion-free read. See NDI-matlab#1009 for the same guard on the MATLAB side. """ - from ndi.cloud.batch_signed_url import _default_signer + from ndi.cloud.batch_signed_url import BatchScopeUnreachable, _default_signer with ( patch( @@ -375,8 +372,5 @@ def test_ready_without_result_url_is_reported_as_failure(self): return_value={"state": "ready"}, # no resultUrl ), ): - ok, answer = _default_signer("ds1", "doc1", client=MagicMock(), retry_delays=()) - - assert ok is False - error = answer.get("__error__", "") - assert "resultUrl" in error, f"failure must name the missing field: {error!r}" + with pytest.raises(BatchScopeUnreachable, match="resultUrl"): + _default_signer("ds1", "doc1", client=MagicMock(), retry_delays=()) diff --git a/tests/test_cloud_batch_signed_url.py b/tests/test_cloud_batch_signed_url.py index 604212e8..d843d33e 100644 --- a/tests/test_cloud_batch_signed_url.py +++ b/tests/test_cloud_batch_signed_url.py @@ -663,8 +663,14 @@ def flaky_create(*args, **kwargs): assert answer["files"] == {"u1": "https://s3.example.com/u1"} assert calls["n"] == 2, "should have retried exactly once before succeeding" - def test_every_attempt_raising_reports_the_last_cause(self): - from ndi.cloud.batch_signed_url import _default_signer + def test_every_attempt_raising_raises_batch_scope_unreachable(self): + """After every retry attempt fails, raise instead of silent fallback. + + Falling back per-uid on an actually-broken async job path + would hide the bug behind slow-but-working chunk fetches. See + BatchScopeUnreachable's docstring. + """ + from ndi.cloud.batch_signed_url import BatchScopeUnreachable, _default_signer calls = {"n": 0} @@ -673,19 +679,21 @@ def always_fails(*args, **kwargs): raise ConnectionError(f"try {calls['n']}") with patch("ndi.cloud.api.files.createSignedURLSetJob", side_effect=always_fails): - ok, answer = _default_signer( - "ds1", - "doc1", - client=MagicMock(), - retry_delays=(0.0, 0.0), # 3 attempts total - sleep=lambda _s: None, - ) + with pytest.raises(BatchScopeUnreachable) as exc_info: + _default_signer( + "ds1", + "doc1", + client=MagicMock(), + retry_delays=(0.0, 0.0), # 3 attempts total + sleep=lambda _s: None, + ) - assert ok is False assert calls["n"] == 3, "should have made 3 attempts (initial + 2 retries)" - error = answer.get("__error__", "") - assert "ConnectionError" in error - assert "3 attempts" in error, f"should name the attempt count: {error!r}" + message = str(exc_info.value) + assert "ConnectionError" in message + assert "3 attempts" in message, f"should name the attempt count: {message!r}" + # The underlying transport error is chained for programmatic access. + assert isinstance(exc_info.value.__cause__, ConnectionError) def test_zero_retry_delays_is_a_single_attempt(self): """Passing an empty retry_delays disables retry entirely. @@ -694,7 +702,7 @@ def test_zero_retry_delays_is_a_single_attempt(self): fast-fail probe, or an environment where the signer already handles its own retries). """ - from ndi.cloud.batch_signed_url import _default_signer + from ndi.cloud.batch_signed_url import BatchScopeUnreachable, _default_signer calls = {"n": 0} @@ -703,20 +711,20 @@ def always_fails(*args, **kwargs): raise ConnectionError("nope") with patch("ndi.cloud.api.files.createSignedURLSetJob", side_effect=always_fails): - ok, _ = _default_signer( - "ds1", - "doc1", - client=MagicMock(), - retry_delays=(), - sleep=lambda _s: None, - ) + with pytest.raises(BatchScopeUnreachable): + _default_signer( + "ds1", + "doc1", + client=MagicMock(), + retry_delays=(), + sleep=lambda _s: None, + ) - assert ok is False assert calls["n"] == 1, "empty retry_delays should mean one attempt, no retries" def test_retry_delays_are_slept_in_order(self): """The backoff delays are consumed in order, once per failed attempt.""" - from ndi.cloud.batch_signed_url import _default_signer + from ndi.cloud.batch_signed_url import BatchScopeUnreachable, _default_signer slept: list[float] = [] @@ -724,16 +732,36 @@ def always_fails(*args, **kwargs): raise ConnectionError("nope") with patch("ndi.cloud.api.files.createSignedURLSetJob", side_effect=always_fails): - _default_signer( - "ds1", - "doc1", - client=MagicMock(), - retry_delays=(1.0, 4.0, 16.0), - sleep=slept.append, - ) + with pytest.raises(BatchScopeUnreachable): + _default_signer( + "ds1", + "doc1", + client=MagicMock(), + retry_delays=(1.0, 4.0, 16.0), + sleep=slept.append, + ) assert slept == [ 1.0, 4.0, 16.0, ], f"expected the three backoff delays consumed in order, got {slept}" + + def test_fetch_scope_propagates_batch_scope_unreachable(self): + """The strict-mode raise must not be caught by _fetch_scope. + + _fetch_scope catches Exception from injected signers so tests can + script arbitrary failures (existing behavior). But when the DEFAULT + signer's async job path exhausts its retries, the resulting + BatchScopeUnreachable must propagate all the way to the caller so + the real error surfaces. + """ + from ndi.cloud.batch_signed_url import BatchScopeUnreachable, BatchSignedUrlLookup + + def unreachable_signer(*args, **kwargs): + raise BatchScopeUnreachable("simulated async-job failure") + + lookup = BatchSignedUrlLookup(signer=unreachable_signer) + + with pytest.raises(BatchScopeUnreachable, match="simulated async-job failure"): + lookup.lookup("ds1", "doc1", "chunks", "u_1") From 20df58696ca354336eb0fc16df81e4625afe88bf Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 14:59:39 +0000 Subject: [PATCH 12/55] lightsheetZarr CLI: configure INFO logging so cloud progress is visible Without this, every logger.info() in ndi.cloud.* is silently dropped: async signed-URL-set-job submit/wait/read progress, batch retry decisions, per-page walk timings all vanish. A viewer stuck inside _default_signer looks identical to a healthy one -- no way to tell whether the async job is running server-side or dead on arrival. Set up stderr logging at INFO when the root logger has no handlers so an embedding host (notebook, downstream CLI) that has already configured logging keeps control. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/gui/app/lightsheetZarr/cli.py | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/src/ndi/gui/app/lightsheetZarr/cli.py b/src/ndi/gui/app/lightsheetZarr/cli.py index 342f8049..11795667 100644 --- a/src/ndi/gui/app/lightsheetZarr/cli.py +++ b/src/ndi/gui/app/lightsheetZarr/cli.py @@ -252,8 +252,31 @@ def _pick_reduction(session: Any, pyramid_doc: Any, requested: str | None) -> st return available[0] +def _configure_logging() -> None: + """Set up INFO-level stderr logging when the root has no handlers. + + Without this, every ``logger.info(...)`` in ``ndi.cloud.*`` is silently + dropped -- the progress traces we rely on (async signed-URL-set-job + submit/wait/read, batch retry decisions, per-page walk timings) never + reach the terminal, and a stuck viewer looks identical to a healthy + one. Only configure when nothing else has: an embedding host (a + notebook, a downstream CLI) that has already set up logging keeps + control. + """ + import logging + + if logging.getLogger().handlers: + return + logging.basicConfig( + level=logging.INFO, + format="%(asctime)s %(levelname)s %(name)s %(message)s", + stream=sys.stderr, + ) + + def main(argv: Sequence[str] | None = None) -> int: args = build_parser().parse_args(argv) + _configure_logging() try: session = _open_session(args.session) From 121d998e3a241669a7da11b035bd3ea475ce2ef5 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 15:04:36 +0000 Subject: [PATCH 13/55] batch_signed_url: include scope + cause in retry log line The retry log printed only the exception type name; a bare 'RuntimeError' tells us the async job path failed but not why. Widen to include the (dataset, document, file_series) scope and the exception message, so a running viewer surfaces the actual reason each attempt failed rather than four identical opaque lines. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/cloud/batch_signed_url.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/src/ndi/cloud/batch_signed_url.py b/src/ndi/cloud/batch_signed_url.py index 08931894..04b09223 100644 --- a/src/ndi/cloud/batch_signed_url.py +++ b/src/ndi/cloud/batch_signed_url.py @@ -253,10 +253,15 @@ def _default_signer( if attempt < attempts - 1: delay = retry_delays[attempt] logger.info( - "batch signed-URL fetch attempt %d/%d failed (%s); retrying in %.1fs", + "batch signed-URL fetch attempt %d/%d for scope " + "(%s, %s, series=%r) failed (%s: %s); retrying in %.1fs", attempt + 1, attempts, + dataset_id, + document_id, + file_series, type(exc).__name__, + exc, delay, ) sleep(delay) From 3fdee2c92e24e6051fec83075a58bd1cde28bcbe Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 15:07:01 +0000 Subject: [PATCH 14/55] batch_signed_url: cache scope-unreachable and re-raise on subsequent lookups Strict-mode BatchScopeUnreachable was raising all the way up, but did not update _failed_scopes -- so every next uid in the viewport triggered another full 4-attempt retry cycle against the broken endpoint. A viewer whose async job path is dead spent 20 s per failed cycle in an infinite loop instead of surfacing the actual cause once and staying quiet. _fetch_scope now records the failure with unreachable=True and the cause string before re-raising. lookup() consults the cache and, when it finds an unreachable entry within failure_ttl_seconds, re-raises a fresh BatchScopeUnreachable naming the original cause and elapsed time. Also widened the retry log to include scope and exception message instead of just the exception type name, so 'RuntimeError' becomes something legible like 'RuntimeError: signed-URL-set job ... failed'. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/cloud/batch_signed_url.py | 38 +++++++++++++++++++++++++++---- 1 file changed, 33 insertions(+), 5 deletions(-) diff --git a/src/ndi/cloud/batch_signed_url.py b/src/ndi/cloud/batch_signed_url.py index 04b09223..8e26d230 100644 --- a/src/ndi/cloud/batch_signed_url.py +++ b/src/ndi/cloud/batch_signed_url.py @@ -105,10 +105,17 @@ class _CacheEntry: @dataclass class _FailureEntry: - """A scope's most recent batch failure, with the uid that saw it.""" + """A scope's most recent batch failure, with the uid that saw it. + + ``unreachable`` distinguishes a real async-job failure (surfaced as + :class:`BatchScopeUnreachable`, no per-uid fallback) from an + injected-signer failure or a partial-map miss (per-uid fallback OK). + """ at: float uid: str + unreachable: bool = False + cause: str = "" @dataclass @@ -368,6 +375,17 @@ def lookup( if failure is not None: if (now - failure.at) >= self._failure_ttl_seconds: del self._failed_scopes[cache_key] + elif failure.unreachable: + # The async job path was strict-mode-failed for + # this scope within the TTL. Re-raise fast rather + # than running another 4-attempt retry cycle for + # every next uid the viewport asks about. The + # cause carried on the entry preserves what the + # first raise said. + raise BatchScopeUnreachable( + f"scope {cache_key!r} was unreachable " + f"{now - failure.at:.1f}s ago: {failure.cause}" + ) elif failure.uid != uid: self._stats.uid_misses += 1 self._warn_once(cache_key) @@ -486,10 +504,20 @@ def _fetch_scope( file_series=series_name, client=client, ) - except BatchScopeUnreachable: - # Real failure of the async signed-URL-set path; propagate so - # the caller (viewer, download orchestrator, test) sees the - # cause instead of a slow per-uid walk that hides the bug. + except BatchScopeUnreachable as exc: + # Real failure of the async signed-URL-set path. Record the + # scope as unreachable so subsequent lookups in the same scope + # fast-fail (re-raising a fresh BatchScopeUnreachable) instead + # of running another full retry cycle every time napari asks + # for a chunk. Then propagate so the caller sees the cause. + self._stats.last_failure_reason = f"async job path unreachable: {exc}" + self._stats.uid_misses += 1 + self._failed_scopes[cache_key] = _FailureEntry( + at=time.monotonic(), + uid=uid, + unreachable=True, + cause=str(exc), + ) raise except Exception as exc: # noqa: BLE001 - reported as a miss reason ok = False From 545422353c7d62c94a38d83f472993f572992674 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 17:34:56 +0000 Subject: [PATCH 15/55] pyramid.multiscale: propagate BatchScopeUnreachable past _resolve's per-chunk catch _resolve caught EVERY exception from database_openbinarydoc and returned None, which _read_chunk_from_fetcher then turned into a fill_value block. That catch is right for a per-chunk transient blip (one chunk drops out, canvas keeps rendering) but wrong for a systemic scope failure: every chunk in the scope raises identically, every block becomes fill_value, and the whole canvas silently reads as zeros while the fetcher summary line reports "N failures" that no test asserts on. It is the exact silent-fetch-failure pattern the integration test's non-zero assertion was written to catch (see test_the_coarsest_level_computes_and_is_not_all_zero), now firing on the tiny fixture too. Let BatchScopeUnreachable propagate past the per-chunk catch so a broken batch surfaces through .compute() as an actual error rather than an all-zero result. Ordinary exceptions still degrade to fill_value. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/pyramid/multiscale.py | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/src/ndi/pyramid/multiscale.py b/src/ndi/pyramid/multiscale.py index d811ea0c..e4deae37 100644 --- a/src/ndi/pyramid/multiscale.py +++ b/src/ndi/pyramid/multiscale.py @@ -35,6 +35,8 @@ import threading from typing import Any +from ndi.cloud.batch_signed_url import BatchScopeUnreachable + # --------------------------------------------------------------------------- # depends_on / doc discovery @@ -227,6 +229,19 @@ def _resolve(self, doc, filename) -> str | None: t0 = _time.monotonic() try: fh = s.database_openbinarydoc(doc, filename) + except BatchScopeUnreachable: + # A systemic failure of the batch signed-URL path is not a + # transient blip we can paper over with fill_value: every + # chunk in the scope will fail identically, and swallowing + # this one turns the whole canvas into silent zeros -- the + # exact "silent fetch failure" pattern this loader was + # written to prevent. Propagate so callers (viewer, dask + # compute, integration test) see the actual cause. See + # Waltham-Data-Science/NDI-python#322 and the strict-mode + # rationale on BatchScopeUnreachable itself. + with self._stats_lock: + self._resolve_none += 1 + raise except Exception as exc: with self._stats_lock: self._resolve_none += 1 From ce93ada63f5c0812ec3332ae07519099b114808a Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 17:50:17 +0000 Subject: [PATCH 16/55] pyramid.multiscale: surface systemic BatchScopeUnreachable through async prefetch The upsample fallback never calls chunkPath synchronously -- it uses chunkPathIfCached (memoization only, no fetch) plus prefetchAsync (fire-and-forget). So BatchScopeUnreachable raised inside _resolve fires in a background pool thread and is caught by prefetchAsync's generic 'except Exception' handler, which turns it into path=None. The main thread then goes down the "coarse not cached either" branch and returns a zero_block for every chunk, producing a canvas of silent zeros -- the very pattern the loader's integration test was designed to catch, now firing on the tiny fixture. _ChunkFetcher now tracks systemic scope failures on the instance (_unreachable_exc). When prefetchAsync's background _run catches BatchScopeUnreachable it stores the exception rather than dropping it, and chunkPath / chunkPathIfCached re-raise on subsequent calls. The very first chunk still races: its prefetch may not have failed before the fallback returns zeros -- but any subsequent chunk in the same compute surfaces the exception, which propagates through dask to .compute(). In practice that is the second chunk, which means the test fails loudly instead of silently. Ordinary per-chunk transient failures still degrade to fill_value (the existing 'except Exception' branch below the new one keeps that behavior). Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/pyramid/multiscale.py | 30 ++++++++++++++++++++++++++++++ 1 file changed, 30 insertions(+) diff --git a/src/ndi/pyramid/multiscale.py b/src/ndi/pyramid/multiscale.py index e4deae37..e38d324b 100644 --- a/src/ndi/pyramid/multiscale.py +++ b/src/ndi/pyramid/multiscale.py @@ -176,6 +176,16 @@ def __init__(self, session, workers: int = 1): self._resolve_times_cache: list[float] = [] self._resolve_times_fetch: list[float] = [] self._resolve_none: int = 0 + # Systemic failure of the batch signed-URL path. Once one background + # prefetch surfaces a BatchScopeUnreachable, every chunk in the + # affected scope will fail identically -- and returning fill_value + # for all of them would silently render the whole canvas as zeros, + # exactly the pattern the loader was written to prevent. We hold + # the first raise here so subsequent lookups (chunkPathIfCached, + # chunkPath) can surface it instead of hiding behind the fallback. + # See Waltham-Data-Science/NDI-python#322 and the strict-mode + # rationale in ndi.cloud.batch_signed_url.BatchScopeUnreachable. + self._unreachable_exc: BatchScopeUnreachable | None = None @staticmethod def _reopener(session): @@ -319,6 +329,8 @@ def chunkPath(self, doc, filename) -> str | None: ``os.path.exists`` from any thread. Files can be evicted, so a stale hit that fails on read should be dropped with :meth:`forget`. """ + if self._unreachable_exc is not None: + raise self._unreachable_exc key = (getattr(doc, "id", str(doc)), filename) known = self._paths.get(key) if known is not None and os.path.exists(known): @@ -357,7 +369,16 @@ def chunkPathIfCached(self, doc, filename) -> str | None: upsampled coarse data" WITHOUT waiting on a cloud round trip. A miss means "we do not have it now"; whether we ever will is a separate question the async prefetch answers. + + Raises :class:`BatchScopeUnreachable` if a prior background + prefetch established that the batch signed-URL path is dead + for this fetcher. The upsample fallback would otherwise + quietly return a zero block for every chunk in the affected + scope; surfacing the exception here forces the actual cause + out through ``.compute()``. """ + if self._unreachable_exc is not None: + raise self._unreachable_exc key = (getattr(doc, "id", str(doc)), filename) known = self._paths.get(key) if known is not None and os.path.exists(known): @@ -413,6 +434,15 @@ def prefetchAsync(self, doc, filename, on_complete=None) -> None: def _run(): try: path = self._resolve(doc, filename) + except BatchScopeUnreachable as exc: + # Systemic scope failure. Remember it so the next + # chunkPathIfCached / chunkPath surfaces it instead of + # letting the upsample fallback quietly return a + # canvas full of zeros. + with self._lock: + if self._unreachable_exc is None: + self._unreachable_exc = exc + path = None except Exception: # noqa: BLE001 path = None if path is not None: From 4b605e147cb5fbbe1aa9711fe1696a80b10d14f7 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 19:36:59 +0000 Subject: [PATCH 17/55] batch_signed_url: log the async job's happy path so long waits are legible _default_signer and waitForSignedURLSetJob were completely silent during a successful long-running job. A viewer waiting minutes on a 121k-member scope looked identical to a hung viewer, and the only signal was a raw TCP connection to the API host visible in netstat. Add INFO log lines at each step: * _default_signer: submit -> jobId, terminal state + elapsed, result blob delivered with uid count. * waitForSignedURLSetJob: every poll logs the state and signedCount / totalCount, so a long wait shows visible progress instead of silence. Diagnostic only -- no behavior change. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/cloud/api/files.py | 24 +++++++++++++++++-- src/ndi/cloud/batch_signed_url.py | 39 +++++++++++++++++++++++++++++++ 2 files changed, 61 insertions(+), 2 deletions(-) diff --git a/src/ndi/cloud/api/files.py b/src/ndi/cloud/api/files.py index 76a6497a..1274bace 100644 --- a/src/ndi/cloud/api/files.py +++ b/src/ndi/cloud/api/files.py @@ -827,17 +827,37 @@ def waitForSignedURLSetJob( start = time.monotonic() interval = initial_interval last: Any = None + poll = 0 while True: elapsed = time.monotonic() - start try: status = getSignedURLSetJob(job_id, client=client) last = status state = status.get("state", "") if hasattr(status, "get") else "" + signed = status.get("signedCount") if hasattr(status, "get") else None + total = status.get("totalCount") if hasattr(status, "get") else None + _module_logger.info( + "waitForSignedURLSetJob: jobId=%s poll #%d state=%r signed=%s/%s " + "(elapsed %.1fs)", + job_id, + poll, + state, + signed, + total, + elapsed, + ) if state in _TERMINAL_SIGNED_URL_SET_STATES: return status - except Exception: + except Exception as exc: # A failed poll is not a failed job; ride out gateway blips. - pass + _module_logger.info( + "waitForSignedURLSetJob: jobId=%s poll #%d failed (%s: %s); " "will retry", + job_id, + poll, + type(exc).__name__, + exc, + ) + poll += 1 if elapsed + interval > timeout: payload: dict[str, Any] if last is not None and hasattr(last, "data") and isinstance(last.data, dict): diff --git a/src/ndi/cloud/batch_signed_url.py b/src/ndi/cloud/batch_signed_url.py index 8e26d230..a1e2642b 100644 --- a/src/ndi/cloud/batch_signed_url.py +++ b/src/ndi/cloud/batch_signed_url.py @@ -207,8 +207,18 @@ def _default_signer( attempts = len(retry_delays) + 1 last_exc: Exception | None = None + scope_started = time.monotonic() for attempt in range(attempts): try: + logger.info( + "batch signed-URL job: submitting createSignedURLSetJob for scope " + "(%s, %s, series=%r) attempt %d/%d", + dataset_id, + document_id, + file_series, + attempt + 1, + attempts, + ) job = files_api.createSignedURLSetJob( dataset_id, document_id, @@ -219,6 +229,12 @@ def _default_signer( job_id = job.get("jobId", "") if hasattr(job, "get") else "" if not job_id: raise RuntimeError(f"createSignedURLSetJob returned no jobId (payload: {job!r})") + logger.info( + "batch signed-URL job: submitted jobId=%s, waiting for terminal state " + "(timeout %.0fs)", + job_id, + job_timeout, + ) status = files_api.waitForSignedURLSetJob( job_id, @@ -226,6 +242,12 @@ def _default_signer( client=client, ) state = status.get("state", "") if hasattr(status, "get") else "" + logger.info( + "batch signed-URL job: jobId=%s reached state=%r after %.1fs total", + job_id, + state, + time.monotonic() - scope_started, + ) if state == "failed": err = status.get("error", "") if hasattr(status, "get") else "" raise RuntimeError( @@ -252,8 +274,25 @@ def _default_signer( raise RuntimeError( f"signed-URL-set job {job_id} was ready but carried no resultUrl" ) + logger.info( + "batch signed-URL job: jobId=%s ready, fetching result blob", + job_id, + ) answer = files_api.getSignedURLSetResult(result_url) + n_files = 0 + if isinstance(answer, dict) and isinstance(answer.get("files"), dict): + n_files = len(answer["files"]) + logger.info( + "batch signed-URL job: jobId=%s delivered %d uid->URL entries " + "for scope (%s, %s, series=%r) in %.1fs total", + job_id, + n_files, + dataset_id, + document_id, + file_series, + time.monotonic() - scope_started, + ) return True, answer except Exception as exc: # noqa: BLE001 - reported by BatchScopeUnreachable last_exc = exc From d7b14b74c973677849ba909446ecbde0c9be6dfb Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 27 Sep 2026 20:55:21 +0000 Subject: [PATCH 18/55] cloud.client: reauthenticate on 401/403 mid-request so long jobs survive token expiry Observed today: a signed-URL-set-job poll loop that had to run for ~1 hour (server signs ~30 URLs/sec, and a 121k-member scope means tens of minutes) died at ~9 min with HTTP 403 "Token is invalid or expired". The bearer token's TTL is shorter than the wall-clock time some legitimate long operations need, and the client kept sending the stale token forever until the outer wait timed out at 60 minutes, having produced nothing useful. CloudClient._request now catches a 401/403 once per request, calls self._reauthenticate() (a fresh CloudConfig.from_env() + authenticate() that mutates config.token in place), updates the Authorization header, and retries. If the reauth itself fails, or if the retried request still comes back 401/403, the original error surfaces as before -- a second 401 after a successful reauth is a real credential problem, not something worth looping over. Also updates the "a 401 is not retried" test to acknowledge the new one-shot reauth semantics, and adds four tests pinning: happy-path reauth-and-retry, 403 same as 401, no loop after a persistent 401 past reauth, and the retry using the fresh token in its Authorization header. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/cloud/client.py | 57 +++++++++++++++ tests/test_cloud_client_retry.py | 122 ++++++++++++++++++++++++++++++- 2 files changed, 177 insertions(+), 2 deletions(-) diff --git a/src/ndi/cloud/client.py b/src/ndi/cloud/client.py index 47d1b46e..3620d80f 100644 --- a/src/ndi/cloud/client.py +++ b/src/ndi/cloud/client.py @@ -289,6 +289,7 @@ def _request( retryable = method.upper() in self.RETRY_METHODS attempt = 0 + reauth_tried = False while True: attempt += 1 try: @@ -322,6 +323,40 @@ def _request( continue raise CloudAPIError(f"Request failed{self._attempt_note(attempt)}: {exc}") from exc + # Token expired mid-run? Refresh once and retry. A long-running + # operation (e.g. polling a signed-URL-set job for tens of + # minutes) can outlive the token TTL, and without this every + # subsequent request 403s forever until the outer timeout + # fires -- observed on a 1-hour signing job that broke at + # ~10 min. One-shot per request: a second 401/403 after + # reauthentication is a real auth problem (bad creds, revoked + # key) and belongs at the surface. + if resp.status_code in (401, 403) and not reauth_tried: + logger.info( + "cloud request %s %s: got HTTP %d (token likely expired); " + "reauthenticating and retrying", + method, + endpoint, + resp.status_code, + ) + try: + self._reauthenticate() + except Exception as exc: # noqa: BLE001 - report and surface original 4xx + logger.info( + "cloud request %s %s: reauthenticate failed (%s: %s); " + "letting the original HTTP %d surface", + method, + endpoint, + type(exc).__name__, + exc, + resp.status_code, + ) + else: + if self.config.token: + headers["Authorization"] = f"Bearer {self.config.token}" + reauth_tried = True + continue + if ( retryable and resp.status_code in self.RETRY_STATUSES @@ -455,6 +490,28 @@ def from_env(cls) -> CloudClient: config.token, config.org_id = authenticate(config) return cls(config) + def _reauthenticate(self) -> None: + """Refresh this client's token in place from the same credentials. + + Called by :meth:`_request` when a 401/403 comes back mid-operation + (typically because the current token's TTL has expired during a + long-running poll or download). Uses the same credential-resolution + path as :meth:`from_env`, so a valid ``NDI_CLOUD_USERNAME`` / + ``NDI_CLOUD_PASSWORD`` pair yields a fresh bearer token; a + pre-baked ``NDI_CLOUD_TOKEN`` that is itself expired will still + fail, and the retry surfaces the auth error to the caller. + + Mutates ``self.config.token`` in place so any other caller holding + the same client keeps working. + """ + from .auth import authenticate + + fresh_config = CloudConfig.from_env() + token, org_id = authenticate(fresh_config) + self.config.token = token + if org_id: + self.config.org_id = org_id + def __repr__(self) -> str: return f"CloudClient(api_url={self.config.api_url!r})" diff --git a/tests/test_cloud_client_retry.py b/tests/test_cloud_client_retry.py index f4a8d42f..4e76e8ac 100644 --- a/tests/test_cloud_client_retry.py +++ b/tests/test_cloud_client_retry.py @@ -193,13 +193,29 @@ def test_a_non_transient_status_fails_on_the_first_attempt(self, status, slept): assert client._session.attempts == 1 assert slept == [] - def test_a_401_is_not_retried(self, slept): - """Credentials will not become valid by asking again.""" + def test_a_401_that_reauth_cannot_fix_still_surfaces(self, slept, monkeypatch): + """A 401 triggers reauthentication once; if reauth itself fails, the + original 401 surfaces without the client entering a retry storm. + + Bad or missing creds are the same class of failure they always + were -- credentials do not become valid by asking again. The + reauth branch that :meth:`_request` gained to survive + token-expired-mid-run is a one-shot per request; when it cannot + produce a fresh token, the original 4xx is what the caller sees. + """ client = make_client(response(401, text="nope")) + def failing_reauth(): + raise CloudAuthError("no creds available for reauth") + + monkeypatch.setattr(client, "_reauthenticate", failing_reauth) + with pytest.raises(CloudAuthError): client.get("/datasets") + # Reauth failed -- the original request was made once and the + # 401 surfaces from _handle_response. Reauth itself does not go + # through this session. assert client._session.attempts == 1 def test_a_404_is_not_retried(self, slept): @@ -351,5 +367,107 @@ def test_the_delay_is_jittered(self): assert len(draws) > 1, "delays are identical; the backoff has no jitter" +# ====================================================================== +# Token-expired-mid-run: reauth once and retry +# ====================================================================== +class TestReauthOnTokenExpiry: + """A long-running operation can outlive its bearer token's TTL. + + Observed today on a signed-URL-set-job poll loop: the server was + signing steadily, the client polled every ~30 s, and at ~9 min the + token expired. Every subsequent poll returned 401/403 with a "Token + is invalid or expired" body, but the client kept using the same + stale token until the outer timeout fired. A reauth-and-retry + branch in :meth:`_request` turns that quietly-fatal case into one + invisible-to-the-caller refresh followed by success. + + A second 401/403 after reauth is a real auth failure (bad creds, + revoked key) and must not loop -- one refresh per request is the + invariant these tests pin. + """ + + def test_a_401_triggers_reauth_and_retries_once(self, slept, monkeypatch): + """Happy path: token expired, reauth succeeds, retry succeeds.""" + client = make_client(response(401, text="expired"), OK) + + refreshed = {"n": 0} + + def fresh_token(): + refreshed["n"] += 1 + client.config.token = f"new-token-{refreshed['n']}" + + monkeypatch.setattr(client, "_reauthenticate", fresh_token) + + result = client.get("/datasets") + + assert result.status_code == 200 + assert client._session.attempts == 2, "expected one retry after reauth" + assert refreshed["n"] == 1, "reauth should fire exactly once" + + def test_a_403_also_triggers_reauth(self, slept, monkeypatch): + """The server returned 403 with 'Token is invalid or expired' in + our real observation; the client must treat 403 the same way as + 401 for reauth purposes.""" + client = make_client(response(403, text="expired"), OK) + monkeypatch.setattr( + client, "_reauthenticate", lambda: setattr(client.config, "token", "new") + ) + + result = client.get("/datasets") + assert result.status_code == 200 + assert client._session.attempts == 2 + + def test_a_second_401_after_reauth_does_not_loop(self, slept, monkeypatch): + """Reauth is one-shot per request. A 401 that persists past a + successful reauth is a real auth problem and belongs at the + surface, not in a retry storm.""" + client = make_client(response(401, text="expired"), response(401, text="still expired")) + + calls = {"n": 0} + + def refresh(): + calls["n"] += 1 + client.config.token = f"new-{calls['n']}" + + monkeypatch.setattr(client, "_reauthenticate", refresh) + + with pytest.raises(CloudAuthError): + client.get("/datasets") + + assert calls["n"] == 1, "reauth must fire exactly once, not once per 401" + assert client._session.attempts == 2 + + def test_reauth_updates_the_authorization_header(self, slept, monkeypatch): + """The retry uses the NEW token, not the stale one. Without + this, we'd refresh in place and then send the old header + anyway -- and the whole exercise would be for nothing. + + Captures the header at request time (rather than reading the + session's post-facto record, which holds a single mutating dict + for both calls). + """ + client = make_client(response(401, text="expired"), OK) + + captured_authorizations: list[str] = [] + real_request = client._session.request + + def capturing_request(method, url, **kwargs): + captured_authorizations.append(kwargs.get("headers", {}).get("Authorization")) + return real_request(method, url, **kwargs) + + client._session.request = capturing_request + + def refresh(): + client.config.token = "the-fresh-token" + + monkeypatch.setattr(client, "_reauthenticate", refresh) + client.config.token = "the-stale-token" + + client.get("/datasets") + + assert captured_authorizations[0] == "Bearer the-stale-token" + assert captured_authorizations[1] == "Bearer the-fresh-token" + + if __name__ == "__main__": pytest.main([__file__]) From 18ec747fe1e9434eca528bcc1fd6ec87d3dcee29 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 00:10:57 +0000 Subject: [PATCH 19/55] signed_url_disk_cache: persistent signed-URL disk cache for reopen Port of VH-Lab/NDI-matlab commit 27661a3 (issue #322). A working scientist reopening the same lightsheet-scale dataset over a day should pay the ~85 min async signed-URL-set-job cost once, not every open. New signed_url_disk_cache module writes each scope (dataset_id, document_id, series_name) to ~/.ndi/signed-url-cache//[_].json.gz as gzipped JSON, TTL-checked against the server's own filesExpireAt with a 30 min safety buffer, atomically renamed on write. MATLAB and Python write byte-for-byte identical files: schemaVersion, path convention, filename escaping (percent-encode outside [A-Za-z0-9._-], SHA-1 hash above 96 chars), gzip framing, JSON keys, env var (NDI_SIGNED_URL_CACHE_DIR / NDI_SIGNED_URL_CACHE_SAFETY_SECONDS) all match. A cache dir written by MATLAB reads clean from Python. BatchSignedUrlLookup gains disk_cache=False; get_default() flips it on -- production path. Existing test-suite instances stay off by default. On S3 403 during the actual getFile the scope is forgotten from disk so a rotated token doesn't cascade to N per-uid 403s; getFile takes an opt-in error_out sink so the status code drives the hook. Session-scoped conftest fixture isolates the cache dir so no test can pollute a developer's ~/.ndi/. Bridge YAML: adds signedUrlDiskCache entry and bumps batchSignedUrlLookup's matlab_last_sync_hash to 27661a3. --- src/ndi/cloud/api/files.py | 11 + src/ndi/cloud/batch_signed_url.py | 103 ++++- src/ndi/cloud/filehandler.py | 43 +- src/ndi/cloud/ndi_matlab_python_bridge.yaml | 70 ++- src/ndi/cloud/signed_url_disk_cache.py | 453 +++++++++++++++++++ tests/conftest.py | 40 ++ tests/test_cloud_batch_signed_url.py | 6 +- tests/test_cloud_signed_url_disk_cache.py | 462 ++++++++++++++++++++ 8 files changed, 1182 insertions(+), 6 deletions(-) create mode 100644 src/ndi/cloud/signed_url_disk_cache.py create mode 100644 tests/test_cloud_signed_url_disk_cache.py diff --git a/src/ndi/cloud/api/files.py b/src/ndi/cloud/api/files.py index 1274bace..fead55f5 100644 --- a/src/ndi/cloud/api/files.py +++ b/src/ndi/cloud/api/files.py @@ -225,6 +225,7 @@ def getFile( timeout: int = 120, *, progress=None, + error_out: dict | None = None, ) -> bool: """Download a file from a presigned URL. @@ -238,6 +239,13 @@ def getFile( ``timeout``, and it is called from whichever thread is downloading -- a caller that renders it is responsible for its own locking. + ``error_out``, if given, is populated on FAILURE (return value + ``False``) with the HTTP details -- currently ``{"status": int, + "body": str}`` -- so a caller who needs the status code (e.g. to + invalidate a cached signed URL that returned S3 403) does not have + to re-download or infer from a log line. It is not touched on + success. Kept opt-in so getFile's happy-path signature is unchanged. + WHY HERE. This is the one place every on-demand fetch passes through: the cell table, the contour file, the gene list and every pyramid tile all arrive by fetch_cloud_file -> getFile. Reporting anywhere else @@ -295,6 +303,9 @@ def getFile( url[:80], body[:200], ) + if error_out is not None: + error_out["status"] = int(resp.status_code) + error_out["body"] = body[:200] return False diff --git a/src/ndi/cloud/batch_signed_url.py b/src/ndi/cloud/batch_signed_url.py index a1e2642b..02b989df 100644 --- a/src/ndi/cloud/batch_signed_url.py +++ b/src/ndi/cloud/batch_signed_url.py @@ -343,6 +343,7 @@ def __init__( failure_ttl_seconds: float = DEFAULT_FAILURE_TTL_SECONDS, partial_map_retry_seconds: float = DEFAULT_PARTIAL_MAP_RETRY_SECONDS, sleep: Callable[[float], None] | None = None, + disk_cache: bool = False, ) -> None: self._signer = signer or _default_signer self._ttl_seconds = ttl_seconds @@ -351,6 +352,13 @@ def __init__( # Injected only for tests: the retry has a real wait, and a test # that runs it 20 times should not pay 10 seconds for it. self._sleep = sleep or time.sleep + # Disk cache off by default: a scripted signer in a test suite + # would otherwise persist fake URLs to ~/.ndi/signed-url-cache + # and hit them on the next test's first lookup. Production + # callers (the DID chunk-download path) turn it on explicitly + # via ``get_default()``, which is where reopens actually pay. + # See :mod:`ndi.cloud.signed_url_disk_cache`. + self._disk_cache = disk_cache self._cache: dict[str, _CacheEntry] = {} self._failed_scopes: dict[str, _FailureEntry] = {} self._retried_scopes: set[str] = set() @@ -400,6 +408,17 @@ def lookup( del self._cache[cache_key] entry = None + if entry is None and self._disk_cache: + # Consult the on-disk cache before running the async + # signed-URL-set job. On hit, populate the in-memory + # cache so every uid in the scope resolves without + # another disk read. On miss (no file, corrupt file, + # expired-or-within-safety-buffer), fall through to the + # signer as before. + entry = self._try_disk_cache( + cache_key, cloud_dataset_id, ndi_document_id, series_name + ) + if entry is None: # A scope that just failed is not retried for every uid # after it. Without this, a batch endpoint that cannot @@ -586,6 +605,73 @@ def _fetch_scope( self._cache[cache_key] = entry self._stats.last_map_size = len(entry.files) self._stats.last_map_uids = list(entry.files.keys()) + + # Persist to disk when the caller has opted in AND the payload + # carries a server-signed expiry we can age-check off of. A + # payload with neither filesExpireAt nor expiresAt is not + # cacheable -- the disk cache refuses to invent a TTL, on + # purpose (see :func:`signed_url_disk_cache.save`). Best-effort: + # any I/O failure is swallowed so a full disk does not break a + # read that just succeeded. + if self._disk_cache: + try: + from . import signed_url_disk_cache + + signed_url_disk_cache.save( + cloud_dataset_id, ndi_document_id, series_name, answer + ) + except Exception: # noqa: BLE001 - best-effort + logger.debug( + "signed-URL disk cache save raised for scope %r; ignored", + cache_key, + exc_info=True, + ) + return entry + + def _try_disk_cache( + self, + cache_key: str, + cloud_dataset_id: str, + ndi_document_id: str, + series_name: str, + ) -> _CacheEntry | None: + """Load one scope from the on-disk cache, or return None. + + Populates the in-memory cache on hit so subsequent uids in the + same scope don't re-read the file. A miss is silent -- the + caller falls through to the signer, which is the shape a fresh + or expired-cache first open takes. + """ + try: + from . import signed_url_disk_cache + + disk = signed_url_disk_cache.load( + cloud_dataset_id, ndi_document_id, series_name + ) + except Exception: # noqa: BLE001 - never fail a read on disk I/O + logger.debug( + "signed-URL disk cache load raised for scope %r; ignored", + cache_key, + exc_info=True, + ) + return None + if not disk: + return None + files = disk.get("files") + if not isinstance(files, dict) or not files: + return None + entry = _CacheEntry(files=dict(files), fetched_at=time.monotonic()) + self._cache[cache_key] = entry + self._stats.last_map_size = len(entry.files) + self._stats.last_map_uids = list(entry.files.keys()) + logger.info( + "signed-URL disk cache: served scope (%s, %s, series=%r) " + "with %d uids -- signer not called", + cloud_dataset_id, + ndi_document_id, + series_name, + len(entry.files), + ) return entry def _warn_once(self, cache_key: str) -> None: @@ -646,12 +732,25 @@ def _describe_failure(ok: bool, answer: Any) -> str: def get_default() -> BatchSignedUrlLookup: - """The process-wide default cache. Created on first use.""" + """The process-wide default cache. Created on first use. + + Constructed with the on-disk cache enabled: this is the production + reopen path. A scientist reopening the same lightsheet the next + morning must not pay the ~85 min signed-URL-set job again -- the + first open persists the scope to ``~/.ndi/signed-url-cache/`` (or + wherever ``NDI_SIGNED_URL_CACHE_DIR`` points), the reopen reads it + back and never touches the signer. See + :mod:`ndi.cloud.signed_url_disk_cache`. + + Tests inject their own :class:`BatchSignedUrlLookup` -- disk cache + off by default there -- so scripted-signer suites don't persist + fake URLs into the user's real cache directory. + """ global _DEFAULT_LOOKUP if _DEFAULT_LOOKUP is None: with _DEFAULT_LOCK: if _DEFAULT_LOOKUP is None: - _DEFAULT_LOOKUP = BatchSignedUrlLookup() + _DEFAULT_LOOKUP = BatchSignedUrlLookup(disk_cache=True) return _DEFAULT_LOOKUP diff --git a/src/ndi/cloud/filehandler.py b/src/ndi/cloud/filehandler.py index 9cad5594..706818cc 100644 --- a/src/ndi/cloud/filehandler.py +++ b/src/ndi/cloud/filehandler.py @@ -441,6 +441,7 @@ def fetch_cloud_file( client = get_or_create_cloud_client() download_url = "" + url_came_from_batch = False if ndi_document_id: # Batch path first. Empty means the batch could not answer for this # uid; fall back per uid so the read still succeeds. @@ -454,6 +455,7 @@ def fetch_cloud_file( file_uid, client=client, ) + url_came_from_batch = bool(download_url) if not download_url: details = getFileDetails(dataset_id, file_uid, client=client) @@ -470,16 +472,24 @@ def fetch_cloud_file( logger.debug("Fetching cloud file %s -> %s", ndic_uri, target) observer = _fetch_observer + # error_out is only needed when the URL came from the batch disk cache + # and we might have to invalidate a stale scope on S3 403. Passing it + # unconditionally would force every mocked getFile in the test suite + # to accept the new keyword. + error_out: dict | None = {} if url_came_from_batch else None + call_kwargs: dict = {"timeout": 300} + if error_out is not None: + call_kwargs["error_out"] = error_out if observer is None: - success = getFile(download_url, tmp_path, timeout=300) + success = getFile(download_url, tmp_path, **call_kwargs) else: observer("start", ndic_uri, 0, None) try: success = getFile( download_url, tmp_path, - timeout=300, progress=lambda done, total: observer("chunk", ndic_uri, done, total), + **call_kwargs, ) finally: observer("done", ndic_uri, 0, None) @@ -491,6 +501,35 @@ def fetch_cloud_file( else: # Clean up partial download tmp_path.unlink(missing_ok=True) + + # An S3 403 on a URL that WAS served from the disk cache says + # the cached scope has gone stale (token revoked, or the object + # rotated). Drop the scope so the next read refetches -- + # otherwise we'd 403 our way through every subsequent uid in + # the same scope. Best-effort; a failed forget is not fatal + # to the CloudError we're raising. Mirrors NDI-matlab's own + # 403-invalidation hook in didsqlite.download_file_from_cloud. + if ( + url_came_from_batch + and error_out is not None + and error_out.get("status") == 403 + ): + try: + from . import signed_url_disk_cache + + signed_url_disk_cache.forget( + dataset_id, ndi_document_id, series_name + ) + except Exception: # noqa: BLE001 - best-effort + logger.debug( + "signed-URL disk cache forget raised for scope " + "(%s, %s, series=%r); ignored", + dataset_id, + ndi_document_id, + series_name, + exc_info=True, + ) + from .exceptions import CloudError raise CloudError(f"Failed to download file from {ndic_uri}") diff --git a/src/ndi/cloud/ndi_matlab_python_bridge.yaml b/src/ndi/cloud/ndi_matlab_python_bridge.yaml index ef25ae55..85d1c2d4 100644 --- a/src/ndi/cloud/ndi_matlab_python_bridge.yaml +++ b/src/ndi/cloud/ndi_matlab_python_bridge.yaml @@ -1486,7 +1486,7 @@ not_yet_ported: - name: batchSignedUrlLookup matlab_path: "+ndi/+cloud/+download/+internal/batchSignedUrlLookup.m" python_path: "ndi/cloud/batch_signed_url.py" - matlab_last_sync_hash: "9120c002" + matlab_last_sync_hash: "27661a3" status: ported decision_log: > PORTED for NDI-python#262. Turns one-call-per-uid into one call per @@ -1531,6 +1531,74 @@ not_yet_ported: transient-error retry (NDI-python#322) still wraps the whole three-step exchange. + NDI-matlab#1010 follow-on (MATLAB commit 27661a3): the constructor + gains ``disk_cache: bool = False`` and get_default() constructs + with ``disk_cache=True``, wiring the persistent on-disk cache + module below. ``ttlSeconds`` is still the in-memory freshness + cap; the disk cache is age-checked against the SERVER's + filesExpireAt with a 30 min safety buffer. + + - name: signedUrlDiskCache + matlab_path: "+ndi/+cloud/+download/+internal/signedUrlDiskCache.m" + python_path: "ndi/cloud/signed_url_disk_cache.py" + matlab_last_sync_hash: "988ec74" + status: ported + decision_log: > + PORTED for VH-Lab/NDI-matlab#1010 follow-on / Waltham-Data-Science/ + NDI-python#322. Persistent (on-disk) layer beneath + batchSignedUrlLookup's in-process cache: a scientist reopening the + same lightsheet-scale dataset over a day pays the ~85 min async + signed-URL-set-job cost ONCE. + + The two implementations write byte-for-byte identical files: a + cache dir written by MATLAB is readable by Python and vice + versa. Portability is the whole point -- a mixed lab must not pay + the sign cost twice. + + Shared on-disk contract (documented at the module top of both + files): + - Layout: //.json.gz for + whole-doc scopes, //_.json.gz + for scoped ones. Empty seriesName maps to the whole-doc + filename explicitly so the two spellings can never collide. + - Filename escaping: bytes outside [A-Za-z0-9._-] become %XX + (uppercase hex); the escaped string above 96 characters is + replaced with hash- so a pathologically long name + never overflows a filesystem's 255-byte name limit. + - Format: standard gzip (RFC 1952), single member, UTF-8 JSON + body with schemaVersion=1 and the fields datasetId, + documentId, seriesName, cachedAt, filesExpireAt, expiresAt, + generatedAt, fileCount, files. + - TTL: authoritative filesExpireAt (falls back to expiresAt), + minus a 30 min safety buffer (overridable via + NDI_SIGNED_URL_CACHE_SAFETY_SECONDS). A payload with neither + is un-cacheable; the cache refuses to invent a TTL. + - Env override: NDI_SIGNED_URL_CACHE_DIR points cache_dir() + anywhere. Default ~/.ndi/signed-url-cache, mode 0700 on POSIX. + + Departures from the MATLAB implementation (all internal-only, no + on-disk difference): + - Standard-library gzip + os.replace instead of MATLAB's Java + GZIPInputStream/GZIPOutputStream + java.nio.file.Files.move + dance. Same atomic-rename semantics, cleaner one-liner. + - json.load handles digit-prefixed uid keys natively, so the + MATLAB signedURLFileMap workaround (recover keys from raw + payload because jsondecode renames them) has no counterpart + here. + - SHA-1 via hashlib rather than java.security.MessageDigest. + - forget()'s idempotence is expressed via Path.unlink(missing_ok=True) + rather than an isfile() guard, but the caller-visible + contract is the same: silent no-op on a missing scope. + + The read-path 403 invalidation hook: ndi.cloud.filehandler. + fetch_cloud_file consults the batch cache when the caller has a + document id; on a subsequent S3 403 during the actual getFile + the scope is forgotten from disk so the next lookup refetches -- + mirroring NDI-matlab's own hook in + didsqlite.download_file_from_cloud. getFile grew an opt-in + error_out: dict | None keyword to surface the HTTP status; the + status code, not a stringly-typed message, drives the hook. + - name: buildGenericFileDownloadList matlab_path: "+ndi/+cloud/+download/+internal/buildGenericFileDownloadList.m" matlab_last_sync_hash: "65aeeeb" diff --git a/src/ndi/cloud/signed_url_disk_cache.py b/src/ndi/cloud/signed_url_disk_cache.py new file mode 100644 index 00000000..2bbb1500 --- /dev/null +++ b/src/ndi/cloud/signed_url_disk_cache.py @@ -0,0 +1,453 @@ +""" +ndi.cloud.signed_url_disk_cache - Persistent (on-disk) signed-URL cache. + +MATLAB equivalent: + +ndi/+cloud/+download/+internal/signedUrlDiskCache.m + (VH-Lab/NDI-matlab commit 27661a3 on the same branch) + +This is the disk-backed layer that sits beneath the in-process cache +inside :class:`ndi.cloud.batch_signed_url.BatchSignedUrlLookup`. It exists +so a working scientist reopening the same dataset over a day pays the +~85-minute async signed-URL-set-job cost ONCE and reads from disk on +every subsequent viewer open. NDI cloud infrastructure, not DID: DID's +file-bytes cache is a separate, complementary concern. + +Scope key +--------- +The unit the cache stores and evicts is one ``(dataset_id, document_id, +series_name)`` tuple -- the same tuple ``BatchSignedUrlLookup`` uses. An +empty ``series_name`` marks a whole-document scope. + +Payload +------- +What :func:`save` writes and :func:`load` returns is a dict with the +fields the ``getSignedURLSetResult`` answer already carries: + + files dict[str, str]: uid -> pre-signed download URL. + filesExpireAt ISO-8601 UTC string; the server's authoritative expiry + for the signed URLs. Preferred. + expiresAt ISO-8601 UTC string; older paging-path field. Used + when filesExpireAt is not present. + generatedAt ISO-8601 UTC string; when the server built the map. + Optional, informational. + fileCount int; how many uids the server reported. Optional. + +On-disk contract +---------------- +The MATLAB and Python implementations write the SAME bytes: a scientist +can hand a cache dir written by MATLAB to a Python viewer and see +zero-cost reopens, and vice versa. + +Location: //.json.gz (whole-doc) + //_.json.gz (scoped) + +``cache_dir`` defaults to ``~/.ndi/signed-url-cache``, overridable via +``NDI_SIGNED_URL_CACHE_DIR``. The base directory is created with +mode 0700 on POSIX because a signed URL is a bearer token for 24 h; +Windows relies on the default per-user profile ACL. + +Filename escaping for ``series_name``: any byte outside +``[A-Za-z0-9._-]`` becomes ``%XX`` (uppercase hex). A resulting string +longer than 96 characters is replaced with ``hash-`` (45 +chars). MATLAB does the same, byte-for-byte. + +Format: Standard gzip framing (RFC 1952), single member, UTF-8 + JSON body. Unzipping yields a human-readable JSON object:: + + { + "schemaVersion": 1, + "datasetId": "", + "documentId": "", + "seriesName": "" or "", + "cachedAt": "YYYY-MM-DDTHH:MM:SSZ", + "filesExpireAt": "YYYY-MM-DDTHH:MM:SSZ" or "", + "expiresAt": "YYYY-MM-DDTHH:MM:SSZ" or "", + "generatedAt": "YYYY-MM-DDTHH:MM:SSZ" or "", + "fileCount": N, + "files": { "": "", ... } + } + +TTL and safety buffer +--------------------- +The cache does not invent a TTL. It reads the authoritative +``filesExpireAt`` (or ``expiresAt``) from the payload; a payload with +neither is treated as un-cacheable (:func:`save` is a no-op) and, on +load, as an immediate miss. A URL that would expire in less than the +safety buffer (default 30 min, override with +``NDI_SIGNED_URL_CACHE_SAFETY_SECONDS``) is also treated as a miss -- +we never hand a caller a URL likely to 403 before they can use it. + +Concurrency +----------- +Writes go to a per-writer temp file next to the target and are moved +into place with :func:`os.replace`, which is atomic on both POSIX +(``rename(2)``) and Windows (``MoveFileEx MOVEFILE_REPLACE_EXISTING``). +Two viewers signing the same scope concurrently is fine: whichever +atomic move lands last wins, both readers see a consistent file +afterwards. There is no cross-process lock; the contract is +last-writer-wins, not exclusive. + +Invalidation +------------ +:func:`forget` removes the file for one scope. The caller does this +from the chunk-download path when it sees an S3 403 -- the cached URL +was in the map but the object rotated or the token was revoked, so +drop the scope and let the next lookup refetch. Failing to remove the +file is not fatal. + +Miss / error semantics +---------------------- +:func:`load` returns ``None`` on ANY of: no cache file, unreadable file, +corrupt gzip, invalid JSON, missing timestamp, expired-or-within-buffer. +The caller has no need to distinguish -- every one of them means +"refetch from the server". + +:func:`save` swallows every I/O error: a full disk, permission denied +or a race with another process must never break the read that just +succeeded. The in-memory cache still works; the next viewer open pays +one signer call and tries again. +""" + +from __future__ import annotations + +import gzip +import hashlib +import json +import logging +import os +import re +import secrets +import stat +from datetime import datetime, timezone +from pathlib import Path +from typing import Any + +logger = logging.getLogger(__name__) + + +SCHEMA_VERSION = 1 +DEFAULT_SAFETY_SECONDS = 1800 # 30 min; task-declared floor. +MAX_ESCAPED_SERIES_LEN = 96 # Beyond this we hash the series name. + +_ISO_FMT = "%Y-%m-%dT%H:%M:%SZ" + +_SAFE_BYTE_RE = re.compile(rb"[A-Za-z0-9._-]") + + +# --------------------------------------------------------------------------- +# Public API +# --------------------------------------------------------------------------- + + +def cache_dir() -> Path: + """Absolute path to the on-disk cache root. + + Honors ``NDI_SIGNED_URL_CACHE_DIR``; otherwise + ``~/.ndi/signed-url-cache``. Creates the directory the first time + it is asked for, with mode 0700 on POSIX so a file that is legibly + a bearer token is not world-readable. + """ + override = os.environ.get("NDI_SIGNED_URL_CACHE_DIR", "") + if override: + base = Path(override) + else: + base = Path.home() / ".ndi" / "signed-url-cache" + if not base.exists(): + try: + base.mkdir(parents=True, exist_ok=True) + _chmod_user_only(base) + except OSError: + # Best-effort: BatchSignedUrlLookup treats a missing cache + # dir as "no disk cache available", so a failure to create + # here is not fatal to the read. + pass + return base + + +def load( + dataset_id: str, document_id: str, series_name: str = "" +) -> dict[str, Any] | None: + """Read one scope's cached signed-URL set, or ``None`` on any miss. + + Returns a dict (fields: files, filesExpireAt, expiresAt, + generatedAt, fileCount) on cache hit. Returns ``None`` on any of: + no cache file, corrupt file, JSON parse error, + expired-or-within-safety-buffer. + + The safety buffer is 30 min by default, overridable via + ``NDI_SIGNED_URL_CACHE_SAFETY_SECONDS``. A URL that would expire + inside that window is a miss so the caller refetches rather than + handing bytes to a viewer that then 403s. + """ + file_path = _cache_path(dataset_id, document_id, series_name) + if not file_path.is_file(): + return None + + try: + with gzip.open(file_path, "rb") as fh: + raw = fh.read() + body = raw.decode("utf-8") + data = json.loads(body) + except (OSError, ValueError, json.JSONDecodeError): + return None + + if not isinstance(data, dict): + return None + + files = data.get("files") + if not isinstance(files, dict): + return None + + # Enforce the safety-buffer floor against the authoritative server + # timestamp. A payload with no timestamp is a miss: better to + # refetch than hand out a URL we can't age-check. + expiry_str = _read_string(data, "filesExpireAt") or _read_string( + data, "expiresAt" + ) + if not expiry_str: + return None + expiry = _parse_iso_utc(expiry_str) + if expiry is None: + return None + remaining = (expiry - _now_utc()).total_seconds() + if remaining < _safety_seconds(): + return None + + return { + "files": {str(k): str(v) for k, v in files.items()}, + "filesExpireAt": _read_string(data, "filesExpireAt"), + "expiresAt": _read_string(data, "expiresAt"), + "generatedAt": _read_string(data, "generatedAt"), + "fileCount": _read_int(data, "fileCount", default=len(files)), + } + + +def save( + dataset_id: str, + document_id: str, + series_name: str, + payload: dict[str, Any], +) -> None: + """Persist a scope's signed-URL set. Best-effort, atomic. + + Writes the payload to the scope's cache path via a temp file + + :func:`os.replace`. A payload without a ``filesExpireAt`` / + ``expiresAt`` is NOT written -- the cache cannot age-check + something it has no timestamp for, and we refuse to invent a TTL. + + Any I/O failure (missing dir, permission denied, disk full, + another process racing us) is swallowed silently: the in-memory + cache still works, so the read the caller just finished still + succeeded. + """ + if not isinstance(payload, dict): + return + files = payload.get("files") + if not isinstance(files, dict) or not files: + # Empty maps are legal ("scope exists, has zero members") but + # so is the "endpoint gave us nothing back" case; we cannot + # tell them apart, so err on the side of not persisting. + # A caller who wants to persist an empty scope should still be + # able to next-lookup with no penalty. + if not isinstance(files, dict): + return + + files_expire_at = _read_string(payload, "filesExpireAt") + expires_at = _read_string(payload, "expiresAt") + if not files_expire_at and not expires_at: + # Refuse to persist an un-age-checkable payload. + return + + generated_at = _read_string(payload, "generatedAt") + file_count = _read_int(payload, "fileCount", default=len(files)) + + file_path = _cache_path(dataset_id, document_id, series_name) + try: + file_path.parent.mkdir(parents=True, exist_ok=True) + except OSError: + return + + body = { + "schemaVersion": SCHEMA_VERSION, + "datasetId": str(dataset_id), + "documentId": str(document_id), + "seriesName": str(series_name), + "cachedAt": _iso_now(), + "filesExpireAt": files_expire_at, + "expiresAt": expires_at, + "generatedAt": generated_at, + "fileCount": int(file_count), + # dict insertion order maps to JSON key order; digit-prefixed + # uids are legal JSON object keys, so nothing here renames them. + "files": {str(k): str(v) for k, v in files.items()}, + } + try: + text = json.dumps(body, ensure_ascii=False, separators=(",", ":")) + except (TypeError, ValueError): + return + + # Per-writer temp path in the SAME directory so os.replace is a + # rename() rather than a cross-device copy+unlink. The random + # suffix means two writers on the same scope in the same process + # don't collide on their own temp file. + tmp_path = file_path.parent / ( + file_path.name + f".tmp.{os.getpid()}.{secrets.token_hex(4)}" + ) + try: + with gzip.open(tmp_path, "wb") as gz: + gz.write(text.encode("utf-8")) + os.replace(tmp_path, file_path) + except OSError: + # Any I/O failure: try to clean up the temp file, then swallow. + try: + tmp_path.unlink(missing_ok=True) + except OSError: + pass + + +def forget(dataset_id: str, document_id: str, series_name: str = "") -> None: + """Remove one scope's cache file. Best-effort. + + Called from the chunk-download path when an S3 403 says the cached + URL is no longer valid. If the file is not present -- or the delete + itself fails -- it is not an error: the next lookup will refetch. + """ + file_path = _cache_path(dataset_id, document_id, series_name) + try: + file_path.unlink(missing_ok=True) + except OSError: + # Best-effort: a failed delete is not fatal to the caller who + # is already recovering from a 403. + pass + + +# --------------------------------------------------------------------------- +# Internal +# --------------------------------------------------------------------------- + + +def _safety_seconds() -> float: + """Read the safety buffer, honoring the env-var override.""" + override = os.environ.get("NDI_SIGNED_URL_CACHE_SAFETY_SECONDS", "") + if override: + try: + v = float(override) + if v >= 0: + return v + except ValueError: + pass + return DEFAULT_SAFETY_SECONDS + + +def _cache_path(dataset_id: str, document_id: str, series_name: str) -> Path: + """The absolute path this (ds, doc, series) scope lands at. + + Empty ``series_name`` maps to a whole-document file so + ``.json.gz`` and ``_.json.gz`` can never collide. + NDI-matlab uses exactly the same convention. + """ + base = cache_dir() / str(dataset_id) + if not series_name: + return base / f"{document_id}.json.gz" + escaped = _escape_series_name(str(series_name)) + return base / f"{document_id}_{escaped}.json.gz" + + +def _escape_series_name(raw: str) -> str: + """Percent-encode a series name for filesystem safety. + + Any byte outside ``[A-Za-z0-9._-]`` becomes ``%XX`` (uppercase + hex). Above :data:`MAX_ESCAPED_SERIES_LEN` fall back to + ``hash-`` so a pathologically long name never overflows + a filesystem's 255-byte name limit. + """ + bytes_ = raw.encode("utf-8") + parts: list[str] = [] + for b in bytes_: + if _SAFE_BYTE_RE.fullmatch(bytes([b])): + parts.append(chr(b)) + else: + parts.append(f"%{b:02X}") + s = "".join(parts) + if len(s) > MAX_ESCAPED_SERIES_LEN: + digest = hashlib.sha1(bytes_, usedforsecurity=False).hexdigest() + s = f"hash-{digest}" + return s + + +def _chmod_user_only(path: Path) -> None: + """Set the cache root to user-only permissions on POSIX. + + A signed URL is effectively a bearer token for 24 h. Windows + relies on the default per-user profile ACL for ``~/.ndi/``, which + is roughly equivalent -- and ``os.chmod`` there is largely a no-op. + """ + if os.name != "posix": + return + try: + os.chmod(path, stat.S_IRWXU) # 0o700 + except OSError: + # Best-effort: an unusual filesystem (SMB mount, WSL under Win) + # can refuse chmod but still work. The cache should not fail + # solely because we could not tighten the mode. + pass + + +def _now_utc() -> datetime: + return datetime.now(timezone.utc) + + +def _iso_now() -> str: + return _now_utc().strftime(_ISO_FMT) + + +_ISO_FRACTION_RE = re.compile(r"\.\d+") + + +def _parse_iso_utc(text: str) -> datetime | None: + """Parse a server-emitted UTC timestamp. + + Handles ``YYYY-MM-DDTHH:MM:SSZ`` and ``YYYY-MM-DDTHH:MM:SS.SSSZ``, + the two shapes the ``getSignedURLSetJob`` ready-state payload has + been observed to use. A trailing ``+00:00`` or ``-00:00`` is + equivalent to ``Z``. Anything else is a parse failure and returns + ``None``. + """ + if not text: + return None + trimmed = text.strip() + if trimmed.endswith("Z"): + trimmed = trimmed[:-1] + elif trimmed.endswith("+00:00") or trimmed.endswith("-00:00"): + trimmed = trimmed[:-6] + trimmed = trimmed.replace(" ", "T") + # Strip fractional seconds if present -- our own writer does not + # emit them, but the server sometimes does, and we do not need + # millisecond precision to age-check a 24 h URL. + trimmed = _ISO_FRACTION_RE.sub("", trimmed) + try: + parsed = datetime.strptime(trimmed, "%Y-%m-%dT%H:%M:%S") + except ValueError: + return None + return parsed.replace(tzinfo=timezone.utc) + + +def _read_string(data: dict[str, Any], name: str) -> str: + """Read a field as a string, "" for missing / null / non-string.""" + value = data.get(name) + if value is None: + return "" + if isinstance(value, str): + return value + return str(value) + + +def _read_int(data: dict[str, Any], name: str, *, default: int) -> int: + """Read a field as an int, ``default`` for missing / non-numeric.""" + value = data.get(name) + if value is None: + return default + try: + return int(value) + except (TypeError, ValueError): + return default diff --git a/tests/conftest.py b/tests/conftest.py index a2425eaa..2065d458 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -223,3 +223,43 @@ def _isolate_did_file_cache(tmp_path_factory): else: os.environ[env_var] = previous did_common._cached_cache = None + + +# --------------------------------------------------------------------------- +# Isolate the signed-URL disk cache for the whole test session. +# +# Waltham-Data-Science/NDI-python#322: ``ndi.cloud.signed_url_disk_cache`` +# persists signed-URL-set responses under ``~/.ndi/signed-url-cache/`` so a +# scientist reopening the same lightsheet the next morning pays the +# ~85 min async job cost once. That is exactly the trap DID's file cache +# used to be: a live-cloud test that hits the production +# ``batch_signed_url.get_default()`` populates the user's real cache, and +# subsequent test runs against different accounts / different environments +# could either read stale entries or silently pollute the developer's home +# directory. +# +# Point the cache dir at a session-scoped tmp directory and unset the +# safety-buffer override so each test starts from the documented 30 min +# default. Same shape as the DID isolation above, same reasoning. +# --------------------------------------------------------------------------- + + +@_pytest.fixture(autouse=True, scope="session") +def _isolate_signed_url_disk_cache(tmp_path_factory): + cache_dir = tmp_path_factory.mktemp("signed-url-cache") + + previous_dir = os.environ.get("NDI_SIGNED_URL_CACHE_DIR") + previous_safety = os.environ.get("NDI_SIGNED_URL_CACHE_SAFETY_SECONDS") + os.environ["NDI_SIGNED_URL_CACHE_DIR"] = str(cache_dir) + os.environ.pop("NDI_SIGNED_URL_CACHE_SAFETY_SECONDS", None) + try: + yield cache_dir + finally: + if previous_dir is None: + os.environ.pop("NDI_SIGNED_URL_CACHE_DIR", None) + else: + os.environ["NDI_SIGNED_URL_CACHE_DIR"] = previous_dir + if previous_safety is None: + os.environ.pop("NDI_SIGNED_URL_CACHE_SAFETY_SECONDS", None) + else: + os.environ["NDI_SIGNED_URL_CACHE_SAFETY_SECONDS"] = previous_safety diff --git a/tests/test_cloud_batch_signed_url.py b/tests/test_cloud_batch_signed_url.py index d843d33e..16cab565 100644 --- a/tests/test_cloud_batch_signed_url.py +++ b/tests/test_cloud_batch_signed_url.py @@ -393,7 +393,11 @@ def test_batch_answer_replaces_getFileDetails(self, tmp_path): patch("ndi.cloud.api.files.getFile") as mock_get_file, ): - def fake_get_file(url, path, timeout=300): + def fake_get_file(url, path, timeout=300, **kwargs): + # **kwargs so fetch_cloud_file's opt-in error_out sink for + # the batch-cache 403 hook (NDI-python#322 follow-on) does + # not blow up the mock; the mock never fails, so error_out + # goes untouched. Path(path).write_bytes(b"member bytes") return True diff --git a/tests/test_cloud_signed_url_disk_cache.py b/tests/test_cloud_signed_url_disk_cache.py new file mode 100644 index 00000000..3fcf4378 --- /dev/null +++ b/tests/test_cloud_signed_url_disk_cache.py @@ -0,0 +1,462 @@ +"""Tests for the persistent (on-disk) signed-URL-set cache. + +Mirrors the MATLAB SignedUrlDiskCacheTest (VH-Lab/NDI-matlab commit +27661a3 on the same branch). Covers save/load round-trip, past-expiry, +30 min safety buffer, no-expiry uncacheable, forget, cacheDir env +override, atomic-write behavior under alternating writers, and series +name escaping. + +Nothing here hits the network. Each test isolates its own cache dir via +``NDI_SIGNED_URL_CACHE_DIR`` so a leaked file cannot leak into the +user's real ``~/.ndi/signed-url-cache``, nor across tests. +""" + +from __future__ import annotations + +import gzip +import json +from datetime import datetime, timedelta, timezone +from pathlib import Path + +import pytest + +# --------------------------------------------------------------------------- +# Fixtures +# --------------------------------------------------------------------------- + + +@pytest.fixture +def isolated_cache_dir(tmp_path, monkeypatch): + """One tempdir per test, wired through NDI_SIGNED_URL_CACHE_DIR.""" + cache = tmp_path / "signed-url-cache" + monkeypatch.setenv("NDI_SIGNED_URL_CACHE_DIR", str(cache)) + # Ensure the safety-seconds override is not carried over from a + # previous test / the ambient shell. + monkeypatch.delenv("NDI_SIGNED_URL_CACHE_SAFETY_SECONDS", raising=False) + return cache + + +def future_iso(seconds_ahead: float) -> str: + """ISO-8601 UTC timestamp N seconds from now, server-shaped.""" + dt = datetime.now(timezone.utc) + timedelta(seconds=seconds_ahead) + return dt.strftime("%Y-%m-%dT%H:%M:%SZ") + + +def sample_payload(**overrides): + """Signer-shaped payload the cache accepts.""" + payload = { + "files": { + "4192a3c0dd1b4e00_3fe8a1b2c3d4e5f6": "https://s3/one", + "4192a3c0dd1b4e00_4fe8a1b2c3d4e5f6": "https://s3/two?sig=abc%2F123", + }, + "filesExpireAt": future_iso(23 * 3600), + "expiresAt": future_iso(23 * 3600), + "generatedAt": future_iso(-30), + "fileCount": 2, + } + payload.update(overrides) + return payload + + +# --------------------------------------------------------------------------- +# Round-trip +# --------------------------------------------------------------------------- + + +class TestSaveLoadRoundTrip: + """What goes in comes back out on the fields callers actually read.""" + + def test_round_trip_preserves_files_and_expiry(self, isolated_cache_dir): + from ndi.cloud import signed_url_disk_cache as cache + + payload = sample_payload() + cache.save("ds1", "doc1", "chunk.bin", payload) + + got = cache.load("ds1", "doc1", "chunk.bin") + + assert got is not None, "a just-written scope should load" + assert got["files"] == payload["files"] + assert ( + got["files"]["4192a3c0dd1b4e00_4fe8a1b2c3d4e5f6"] + == "https://s3/two?sig=abc%2F123" + ), "URL with url-escaped bytes must survive JSON encoding" + assert got["filesExpireAt"] == payload["filesExpireAt"] + assert got["fileCount"] == 2 + + def test_written_file_is_valid_gzipped_json(self, isolated_cache_dir): + """The bytes on disk are the documented shape. + + NDI-matlab and NDI-python have to write the same file so a + cache dir written by one can be read by the other. If the + shape drifts, that portability is silently gone. + """ + from ndi.cloud import signed_url_disk_cache as cache + + cache.save("ds1", "doc1", "", sample_payload()) + + path = isolated_cache_dir / "ds1" / "doc1.json.gz" + assert path.is_file(), "whole-doc scope lands at /.json.gz" + + with gzip.open(path, "rt", encoding="utf-8") as fh: + data = json.load(fh) + assert data["schemaVersion"] == 1 + assert data["datasetId"] == "ds1" + assert data["documentId"] == "doc1" + assert data["seriesName"] == "" + assert isinstance(data["files"], dict) + assert ( + "4192a3c0dd1b4e00_3fe8a1b2c3d4e5f6" in data["files"] + ), "digit-prefixed uids must survive JSON as literal keys" + + +# --------------------------------------------------------------------------- +# Expiry / safety buffer +# --------------------------------------------------------------------------- + + +class TestExpiryAndSafetyBuffer: + """Never hand a caller a URL that would 403 before they can use it.""" + + def test_expired_payload_is_a_miss(self, isolated_cache_dir): + from ndi.cloud import signed_url_disk_cache as cache + + cache.save( + "ds1", + "doc1", + "", + sample_payload(filesExpireAt=future_iso(-3600), expiresAt=""), + ) + assert cache.load("ds1", "doc1", "") is None + + def test_safety_buffer_guards_against_almost_expired( + self, isolated_cache_dir, monkeypatch + ): + """20 min ≤ default 30 min buffer is a miss.""" + from ndi.cloud import signed_url_disk_cache as cache + + cache.save( + "ds1", + "doc1", + "", + sample_payload(filesExpireAt=future_iso(20 * 60), expiresAt=""), + ) + + assert cache.load("ds1", "doc1", "") is None, ( + "20 min < 30 min safety buffer must be a miss" + ) + + # Tighten the buffer and the SAME payload becomes visible -- + # proving the check is the buffer doing the work, not the + # write refusing to persist. + monkeypatch.setenv("NDI_SIGNED_URL_CACHE_SAFETY_SECONDS", "60") + assert cache.load("ds1", "doc1", "") is not None + + def test_payload_without_expiry_is_uncacheable(self, isolated_cache_dir): + """save() refuses; load() sees nothing on disk.""" + from ndi.cloud import signed_url_disk_cache as cache + + cache.save( + "ds1", + "doc1", + "", + sample_payload(filesExpireAt="", expiresAt=""), + ) + assert cache.load("ds1", "doc1", "") is None + # And nothing wrote itself into the tree either. + ds_dir = isolated_cache_dir / "ds1" + assert not ds_dir.exists() or not any(ds_dir.iterdir()) + + def test_load_falls_back_to_expires_at(self, isolated_cache_dir): + """The paging path's older field also age-checks the cache.""" + from ndi.cloud import signed_url_disk_cache as cache + + cache.save( + "ds1", + "doc1", + "", + sample_payload(filesExpireAt="", expiresAt=future_iso(3600)), + ) + assert cache.load("ds1", "doc1", "") is not None + + +# --------------------------------------------------------------------------- +# Forget / invalidation +# --------------------------------------------------------------------------- + + +class TestForget: + """The S3-403 hook: drop a scope whose cached URLs are dead.""" + + def test_forget_removes_file(self, isolated_cache_dir): + from ndi.cloud import signed_url_disk_cache as cache + + cache.save("ds1", "doc1", "chunk.bin", sample_payload()) + assert cache.load("ds1", "doc1", "chunk.bin") is not None + + cache.forget("ds1", "doc1", "chunk.bin") + + assert cache.load("ds1", "doc1", "chunk.bin") is None + assert not ( + isolated_cache_dir / "ds1" / "doc1_chunk.bin.json.gz" + ).exists() + + def test_forget_on_unknown_scope_is_silent(self, isolated_cache_dir): + """S3-403 recovery must not need to check-first.""" + from ndi.cloud import signed_url_disk_cache as cache + + # No pytest.warns / raises: forget is best-effort and idempotent. + cache.forget("nosuch-ds", "nosuch-doc", "nosuch-series") + + +# --------------------------------------------------------------------------- +# cacheDir env override +# --------------------------------------------------------------------------- + + +class TestCacheDir: + """The env var override is documented on-disk contract for NDI-python.""" + + def test_cache_dir_respects_env_override(self, isolated_cache_dir): + from ndi.cloud import signed_url_disk_cache as cache + + assert Path(cache.cache_dir()) == Path(isolated_cache_dir) + + def test_cache_dir_default_when_env_unset(self, tmp_path, monkeypatch): + from ndi.cloud import signed_url_disk_cache as cache + + # Pretend HOME is empty ground so we can see the default shape + # without touching the user's real ~/.ndi/. + home = tmp_path / "home" + home.mkdir() + monkeypatch.setenv("HOME", str(home)) + monkeypatch.delenv("NDI_SIGNED_URL_CACHE_DIR", raising=False) + + result = Path(cache.cache_dir()) + assert result == home / ".ndi" / "signed-url-cache" + + +# --------------------------------------------------------------------------- +# Concurrency +# --------------------------------------------------------------------------- + + +class TestAtomicWrite: + """Two writers hitting the same scope: last write wins, no corruption.""" + + def test_alternating_writes_leave_a_valid_file(self, isolated_cache_dir): + from ndi.cloud import signed_url_disk_cache as cache + + payload_a = sample_payload(files={"aa_bb": "https://s3/first"}) + payload_b = sample_payload(files={"cc_dd": "https://s3/second"}) + + for _ in range(10): + cache.save("ds1", "doc1", "chunks", payload_a) + cache.save("ds1", "doc1", "chunks", payload_b) + + got = cache.load("ds1", "doc1", "chunks") + assert got is not None, "after alternating writes, file must be valid" + assert isinstance(got["files"], dict) + # Whichever landed last is fine; what matters is no half-write. + assert len(got["files"]) == 1 + + def test_temp_files_do_not_linger_next_to_the_cache_file( + self, isolated_cache_dir + ): + """A successful save leaves ONE file: no .tmp scraps around.""" + from ndi.cloud import signed_url_disk_cache as cache + + for _ in range(5): + cache.save("ds1", "doc1", "chunks", sample_payload()) + + ds_dir = isolated_cache_dir / "ds1" + names = sorted(p.name for p in ds_dir.iterdir()) + # Under a race with a second process the two writers could each + # briefly hold a .tmp file, but a fully successful run should + # never leak one. + assert names == ["doc1_chunks.json.gz"], ( + f"expected exactly one cache file, got {names!r}" + ) + + +# --------------------------------------------------------------------------- +# Series name escaping +# --------------------------------------------------------------------------- + + +class TestSeriesNameEscaping: + """Real series names can hold '/', spaces, unicode -- the FS can't. + + The percent-encoding + SHA-1-hash-above-96-chars rule is the on-disk + contract MATLAB mirrors byte-for-byte. + """ + + def test_pathy_series_name_saves_and_loads(self, isolated_cache_dir): + from ndi.cloud import signed_url_disk_cache as cache + + pathy = "level 3/chunk.bin" + cache.save("ds1", "doc1", pathy, sample_payload()) + assert cache.load("ds1", "doc1", pathy) is not None + + def test_two_series_names_land_in_two_files(self, isolated_cache_dir): + """Distinct series must NOT collide on disk.""" + from ndi.cloud import signed_url_disk_cache as cache + + payload = sample_payload() + cache.save("ds1", "doc1", "level 3/chunk.bin", payload) + cache.save("ds1", "doc1", "level 3_chunk.bin", payload) + + ds_dir = isolated_cache_dir / "ds1" + n_gz = sum(1 for p in ds_dir.iterdir() if p.name.endswith(".json.gz")) + assert n_gz == 2, ( + "two distinct series names must produce two distinct cache files" + ) + + def test_very_long_series_name_hashes(self, isolated_cache_dir): + """A 500-char series name lands as ``hash-``, never overflows.""" + from ndi.cloud import signed_url_disk_cache as cache + + long_name = "x" * 500 + cache.save("ds1", "doc1", long_name, sample_payload()) + + ds_dir = isolated_cache_dir / "ds1" + files = list(ds_dir.iterdir()) + assert len(files) == 1 + name = files[0].name + # Anything short-enough with 'hash-<40 hex>' is fine. + assert name.startswith("doc1_hash-") + assert name.endswith(".json.gz") + assert len(name) < 100 # not 500+ + + +# --------------------------------------------------------------------------- +# Path convention (on-disk contract with NDI-matlab) +# --------------------------------------------------------------------------- + + +class TestOnDiskLayout: + """The layout MATLAB writes / reads: whole-doc vs. scoped file names.""" + + def test_whole_doc_scope_lands_at_doc_json_gz(self, isolated_cache_dir): + from ndi.cloud import signed_url_disk_cache as cache + + cache.save("ds1", "doc1", "", sample_payload()) + assert (isolated_cache_dir / "ds1" / "doc1.json.gz").is_file() + + def test_scoped_lands_at_doc_underscore_series(self, isolated_cache_dir): + from ndi.cloud import signed_url_disk_cache as cache + + cache.save("ds1", "doc1", "chunk.bin", sample_payload()) + assert ( + isolated_cache_dir / "ds1" / "doc1_chunk.bin.json.gz" + ).is_file() + + def test_corrupt_file_is_treated_as_miss(self, isolated_cache_dir): + """A truncated / non-gzip file must not crash load().""" + from ndi.cloud import signed_url_disk_cache as cache + + ds_dir = isolated_cache_dir / "ds1" + ds_dir.mkdir(parents=True) + (ds_dir / "doc1.json.gz").write_bytes(b"not a gzip") + + assert cache.load("ds1", "doc1", "") is None + + +# --------------------------------------------------------------------------- +# Integration with BatchSignedUrlLookup +# --------------------------------------------------------------------------- + + +class TestBatchLookupDiskCacheWiring: + """Second call across a fresh in-memory cache must skip the signer. + + This is the whole reopen story: run 1 signs, disk saves; run 2 + reads disk and never touches the signer -- whether that signer + would have been the paging walk or the async job. + """ + + def _counting_signer(self, files_map, expiry_iso): + """Signer that counts calls and hands back a valid, cacheable answer.""" + + state = {"count": 0} + + def signer(dataset_id, document_id, *, file_series="", client=None): + state["count"] += 1 + return True, { + "files": dict(files_map), + "filesExpireAt": expiry_iso, + "expiresAt": expiry_iso, + "generatedAt": expiry_iso, + "fileCount": len(files_map), + } + + return signer, state + + def test_reopen_reads_disk_and_skips_signer(self, isolated_cache_dir): + from ndi.cloud.batch_signed_url import BatchSignedUrlLookup + + files_map = { + "4192a3c0dd1b4e00_3fe8a1b2c3d4e5f6": "https://s3/one", + "4192a3c0dd1b4e00_4fe8a1b2c3d4e5f6": "https://s3/two", + } + signer, state = self._counting_signer(files_map, future_iso(23 * 3600)) + + # Run 1: cold. One signer call answers both uids. + lookup1 = BatchSignedUrlLookup(signer=signer, disk_cache=True) + u1a = lookup1.lookup( + "ds-e2e", + "doc-e2e", + "chunk.bin", + "4192a3c0dd1b4e00_3fe8a1b2c3d4e5f6", + ) + u1b = lookup1.lookup( + "ds-e2e", + "doc-e2e", + "chunk.bin", + "4192a3c0dd1b4e00_4fe8a1b2c3d4e5f6", + ) + assert u1a == "https://s3/one" + assert u1b == "https://s3/two" + assert state["count"] == 1 + + # And the disk file lands. + disk_file = isolated_cache_dir / "ds-e2e" / "doc-e2e_chunk.bin.json.gz" + assert disk_file.is_file() + + # Run 2: fresh in-memory cache; the disk should carry both uids. + lookup2 = BatchSignedUrlLookup(signer=signer, disk_cache=True) + u2a = lookup2.lookup( + "ds-e2e", + "doc-e2e", + "chunk.bin", + "4192a3c0dd1b4e00_3fe8a1b2c3d4e5f6", + ) + u2b = lookup2.lookup( + "ds-e2e", + "doc-e2e", + "chunk.bin", + "4192a3c0dd1b4e00_4fe8a1b2c3d4e5f6", + ) + assert u2a == "https://s3/one" + assert u2b == "https://s3/two" + assert state["count"] == 1, ( + "run 2 must have hit the disk cache -- signer must not have " + "been called again" + ) + + def test_disk_cache_off_by_default_signs_every_cold_run( + self, isolated_cache_dir + ): + """The default (disk_cache=False) preserves pre-cache behaviour.""" + from ndi.cloud.batch_signed_url import BatchSignedUrlLookup + + signer, state = self._counting_signer( + {"aa_bb": "https://s3/x"}, future_iso(23 * 3600) + ) + + for _ in range(3): + lookup = BatchSignedUrlLookup(signer=signer) # disk_cache=False + lookup.lookup("ds-off", "doc-off", "", "aa_bb") + + assert state["count"] == 3 + # And nothing lands on disk. + assert not (isolated_cache_dir / "ds-off").exists() From 3ffd33cec0a46968874128b79e00528e5b9a3fe2 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 00:42:10 +0000 Subject: [PATCH 20/55] chore: apply black formatting to disk-cache commit's untouched files The signed-URL disk cache commit (18ec747) did not run black over the files it touched, so CI's lint check flagged four files as needing reformatting. This applies black without altering any logic. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/cloud/batch_signed_url.py | 8 ++--- src/ndi/cloud/filehandler.py | 13 ++----- src/ndi/cloud/signed_url_disk_cache.py | 14 +++----- tests/test_cloud_signed_url_disk_cache.py | 42 ++++++----------------- 4 files changed, 20 insertions(+), 57 deletions(-) diff --git a/src/ndi/cloud/batch_signed_url.py b/src/ndi/cloud/batch_signed_url.py index 02b989df..b770c4ae 100644 --- a/src/ndi/cloud/batch_signed_url.py +++ b/src/ndi/cloud/batch_signed_url.py @@ -617,9 +617,7 @@ def _fetch_scope( try: from . import signed_url_disk_cache - signed_url_disk_cache.save( - cloud_dataset_id, ndi_document_id, series_name, answer - ) + signed_url_disk_cache.save(cloud_dataset_id, ndi_document_id, series_name, answer) except Exception: # noqa: BLE001 - best-effort logger.debug( "signed-URL disk cache save raised for scope %r; ignored", @@ -645,9 +643,7 @@ def _try_disk_cache( try: from . import signed_url_disk_cache - disk = signed_url_disk_cache.load( - cloud_dataset_id, ndi_document_id, series_name - ) + disk = signed_url_disk_cache.load(cloud_dataset_id, ndi_document_id, series_name) except Exception: # noqa: BLE001 - never fail a read on disk I/O logger.debug( "signed-URL disk cache load raised for scope %r; ignored", diff --git a/src/ndi/cloud/filehandler.py b/src/ndi/cloud/filehandler.py index 706818cc..debcfc25 100644 --- a/src/ndi/cloud/filehandler.py +++ b/src/ndi/cloud/filehandler.py @@ -509,21 +509,14 @@ def fetch_cloud_file( # the same scope. Best-effort; a failed forget is not fatal # to the CloudError we're raising. Mirrors NDI-matlab's own # 403-invalidation hook in didsqlite.download_file_from_cloud. - if ( - url_came_from_batch - and error_out is not None - and error_out.get("status") == 403 - ): + if url_came_from_batch and error_out is not None and error_out.get("status") == 403: try: from . import signed_url_disk_cache - signed_url_disk_cache.forget( - dataset_id, ndi_document_id, series_name - ) + signed_url_disk_cache.forget(dataset_id, ndi_document_id, series_name) except Exception: # noqa: BLE001 - best-effort logger.debug( - "signed-URL disk cache forget raised for scope " - "(%s, %s, series=%r); ignored", + "signed-URL disk cache forget raised for scope " "(%s, %s, series=%r); ignored", dataset_id, ndi_document_id, series_name, diff --git a/src/ndi/cloud/signed_url_disk_cache.py b/src/ndi/cloud/signed_url_disk_cache.py index 2bbb1500..5354f0e3 100644 --- a/src/ndi/cloud/signed_url_disk_cache.py +++ b/src/ndi/cloud/signed_url_disk_cache.py @@ -127,7 +127,7 @@ SCHEMA_VERSION = 1 DEFAULT_SAFETY_SECONDS = 1800 # 30 min; task-declared floor. -MAX_ESCAPED_SERIES_LEN = 96 # Beyond this we hash the series name. +MAX_ESCAPED_SERIES_LEN = 96 # Beyond this we hash the series name. _ISO_FMT = "%Y-%m-%dT%H:%M:%SZ" @@ -164,9 +164,7 @@ def cache_dir() -> Path: return base -def load( - dataset_id: str, document_id: str, series_name: str = "" -) -> dict[str, Any] | None: +def load(dataset_id: str, document_id: str, series_name: str = "") -> dict[str, Any] | None: """Read one scope's cached signed-URL set, or ``None`` on any miss. Returns a dict (fields: files, filesExpireAt, expiresAt, @@ -201,9 +199,7 @@ def load( # Enforce the safety-buffer floor against the authoritative server # timestamp. A payload with no timestamp is a miss: better to # refetch than hand out a URL we can't age-check. - expiry_str = _read_string(data, "filesExpireAt") or _read_string( - data, "expiresAt" - ) + expiry_str = _read_string(data, "filesExpireAt") or _read_string(data, "expiresAt") if not expiry_str: return None expiry = _parse_iso_utc(expiry_str) @@ -290,9 +286,7 @@ def save( # rename() rather than a cross-device copy+unlink. The random # suffix means two writers on the same scope in the same process # don't collide on their own temp file. - tmp_path = file_path.parent / ( - file_path.name + f".tmp.{os.getpid()}.{secrets.token_hex(4)}" - ) + tmp_path = file_path.parent / (file_path.name + f".tmp.{os.getpid()}.{secrets.token_hex(4)}") try: with gzip.open(tmp_path, "wb") as gz: gz.write(text.encode("utf-8")) diff --git a/tests/test_cloud_signed_url_disk_cache.py b/tests/test_cloud_signed_url_disk_cache.py index 3fcf4378..17267b86 100644 --- a/tests/test_cloud_signed_url_disk_cache.py +++ b/tests/test_cloud_signed_url_disk_cache.py @@ -77,8 +77,7 @@ def test_round_trip_preserves_files_and_expiry(self, isolated_cache_dir): assert got is not None, "a just-written scope should load" assert got["files"] == payload["files"] assert ( - got["files"]["4192a3c0dd1b4e00_4fe8a1b2c3d4e5f6"] - == "https://s3/two?sig=abc%2F123" + got["files"]["4192a3c0dd1b4e00_4fe8a1b2c3d4e5f6"] == "https://s3/two?sig=abc%2F123" ), "URL with url-escaped bytes must survive JSON encoding" assert got["filesExpireAt"] == payload["filesExpireAt"] assert got["fileCount"] == 2 @@ -128,9 +127,7 @@ def test_expired_payload_is_a_miss(self, isolated_cache_dir): ) assert cache.load("ds1", "doc1", "") is None - def test_safety_buffer_guards_against_almost_expired( - self, isolated_cache_dir, monkeypatch - ): + def test_safety_buffer_guards_against_almost_expired(self, isolated_cache_dir, monkeypatch): """20 min ≤ default 30 min buffer is a miss.""" from ndi.cloud import signed_url_disk_cache as cache @@ -141,9 +138,7 @@ def test_safety_buffer_guards_against_almost_expired( sample_payload(filesExpireAt=future_iso(20 * 60), expiresAt=""), ) - assert cache.load("ds1", "doc1", "") is None, ( - "20 min < 30 min safety buffer must be a miss" - ) + assert cache.load("ds1", "doc1", "") is None, "20 min < 30 min safety buffer must be a miss" # Tighten the buffer and the SAME payload becomes visible -- # proving the check is the buffer doing the work, not the @@ -196,9 +191,7 @@ def test_forget_removes_file(self, isolated_cache_dir): cache.forget("ds1", "doc1", "chunk.bin") assert cache.load("ds1", "doc1", "chunk.bin") is None - assert not ( - isolated_cache_dir / "ds1" / "doc1_chunk.bin.json.gz" - ).exists() + assert not (isolated_cache_dir / "ds1" / "doc1_chunk.bin.json.gz").exists() def test_forget_on_unknown_scope_is_silent(self, isolated_cache_dir): """S3-403 recovery must not need to check-first.""" @@ -259,9 +252,7 @@ def test_alternating_writes_leave_a_valid_file(self, isolated_cache_dir): # Whichever landed last is fine; what matters is no half-write. assert len(got["files"]) == 1 - def test_temp_files_do_not_linger_next_to_the_cache_file( - self, isolated_cache_dir - ): + def test_temp_files_do_not_linger_next_to_the_cache_file(self, isolated_cache_dir): """A successful save leaves ONE file: no .tmp scraps around.""" from ndi.cloud import signed_url_disk_cache as cache @@ -273,9 +264,7 @@ def test_temp_files_do_not_linger_next_to_the_cache_file( # Under a race with a second process the two writers could each # briefly hold a .tmp file, but a fully successful run should # never leak one. - assert names == ["doc1_chunks.json.gz"], ( - f"expected exactly one cache file, got {names!r}" - ) + assert names == ["doc1_chunks.json.gz"], f"expected exactly one cache file, got {names!r}" # --------------------------------------------------------------------------- @@ -307,9 +296,7 @@ def test_two_series_names_land_in_two_files(self, isolated_cache_dir): ds_dir = isolated_cache_dir / "ds1" n_gz = sum(1 for p in ds_dir.iterdir() if p.name.endswith(".json.gz")) - assert n_gz == 2, ( - "two distinct series names must produce two distinct cache files" - ) + assert n_gz == 2, "two distinct series names must produce two distinct cache files" def test_very_long_series_name_hashes(self, isolated_cache_dir): """A 500-char series name lands as ``hash-``, never overflows.""" @@ -346,9 +333,7 @@ def test_scoped_lands_at_doc_underscore_series(self, isolated_cache_dir): from ndi.cloud import signed_url_disk_cache as cache cache.save("ds1", "doc1", "chunk.bin", sample_payload()) - assert ( - isolated_cache_dir / "ds1" / "doc1_chunk.bin.json.gz" - ).is_file() + assert (isolated_cache_dir / "ds1" / "doc1_chunk.bin.json.gz").is_file() def test_corrupt_file_is_treated_as_miss(self, isolated_cache_dir): """A truncated / non-gzip file must not crash load().""" @@ -439,19 +424,14 @@ def test_reopen_reads_disk_and_skips_signer(self, isolated_cache_dir): assert u2a == "https://s3/one" assert u2b == "https://s3/two" assert state["count"] == 1, ( - "run 2 must have hit the disk cache -- signer must not have " - "been called again" + "run 2 must have hit the disk cache -- signer must not have " "been called again" ) - def test_disk_cache_off_by_default_signs_every_cold_run( - self, isolated_cache_dir - ): + def test_disk_cache_off_by_default_signs_every_cold_run(self, isolated_cache_dir): """The default (disk_cache=False) preserves pre-cache behaviour.""" from ndi.cloud.batch_signed_url import BatchSignedUrlLookup - signer, state = self._counting_signer( - {"aa_bb": "https://s3/x"}, future_iso(23 * 3600) - ) + signer, state = self._counting_signer({"aa_bb": "https://s3/x"}, future_iso(23 * 3600)) for _ in range(3): lookup = BatchSignedUrlLookup(signer=signer) # disk_cache=False From 8eb792760d6c8fd4b31d80483c478d8505d70148 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 00:42:38 +0000 Subject: [PATCH 21/55] Fix three CI regressions on the disk-cache commit Three unrelated failures on head 18ec747, all fixed here: 1. cloud.client: gate mid-request reauth on _can_reauth flag test_bad_auth_raises hand-built a CloudClient(CloudConfig(token=...)) with a deliberately invalid token and expected CloudAuthError. On 401 the reauth path added in d7b14b7 called CloudConfig.from_env() and authenticate() against the ambient CI env credentials, silently swapping the hard-coded bad token for a valid one, retrying, and succeeding -- exactly the "used a different token than the caller asked for" behavior the test was supposed to catch. CloudClient now defaults _can_reauth = False. from_env() (and any future login helper) sets it True after authenticate(). A test that stands up a hand-built client to verify auth failure is used as passed. Long-running clients built by from_env() still get the mid-run token refresh needed to survive TTL expiry. Adds test_a_hand_built_client_does_not_reauth_against_env as a regression pin, and updates make_client(*, can_reauth=True) so existing retry tests continue to behave as though the client had been obtained via from_env(). The client fixture in test_cloud_live.py wraps CloudClient(cloud_config) with _can_reauth = True since its config comes from an env-driven login(). 2. tests/test_cloud_filehandler: reset ambient CloudClient around TestGetOrCreateCloudClient test_env_vars_present passed in isolation and in the No-Cloud suite but failed in the live Cloud API suites: an earlier live test populated _ambient_cloud_client via a real cloud fetch, and the "should call from_env once" assertion then saw zero calls (the cached client was returned instead). An autouse fixture resets both slots before and after each test in the class. 3. sync bridge: record ndi.cloud.sync.internal.fetchManifest Added on the MATLAB side in commit 5541a00; the Python analog (ndi.cloud.filehandler._fetch_manifest, commit 5ab9c04) is ported_differently -- inline with its caller rather than a standalone module. Records the port so test_matlab_bridge_completeness stops flagging it as unrecorded. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/cloud/client.py | 15 +++++++-- .../cloud/sync/ndi_matlab_python_bridge.yaml | 19 +++++++++++ tests/test_cloud_client_retry.py | 33 ++++++++++++++++++- tests/test_cloud_filehandler.py | 17 ++++++++++ tests/test_cloud_live.py | 14 ++++++-- 5 files changed, 93 insertions(+), 5 deletions(-) diff --git a/src/ndi/cloud/client.py b/src/ndi/cloud/client.py index 3620d80f..6e953566 100644 --- a/src/ndi/cloud/client.py +++ b/src/ndi/cloud/client.py @@ -183,6 +183,12 @@ class CloudClient: def __init__(self, config: CloudConfig): self.config = config + # A hand-built ``CloudClient(CloudConfig(token=...))`` uses that + # token as given and does not refresh against env credentials on + # 401/403 -- the caller told us what token to use. ``from_env()`` + # (and any explicit login helper we add) sets this True after a + # successful ``authenticate()``. + self._can_reauth = False try: import requests except ImportError as exc: @@ -331,7 +337,7 @@ def _request( # ~10 min. One-shot per request: a second 401/403 after # reauthentication is a real auth problem (bad creds, revoked # key) and belongs at the surface. - if resp.status_code in (401, 403) and not reauth_tried: + if resp.status_code in (401, 403) and not reauth_tried and self._can_reauth: logger.info( "cloud request %s %s: got HTTP %d (token likely expired); " "reauthenticating and retrying", @@ -488,7 +494,12 @@ def from_env(cls) -> CloudClient: config = CloudConfig.from_env() config.token, config.org_id = authenticate(config) - return cls(config) + client = cls(config) + # We just obtained this token ourselves, so a mid-run 401/403 + # (typical after the token's TTL expires) can be answered by + # calling ``authenticate`` again with the same credentials. + client._can_reauth = True + return client def _reauthenticate(self) -> None: """Refresh this client's token in place from the same credentials. diff --git a/src/ndi/cloud/sync/ndi_matlab_python_bridge.yaml b/src/ndi/cloud/sync/ndi_matlab_python_bridge.yaml index ec7a155e..7a93d873 100644 --- a/src/ndi/cloud/sync/ndi_matlab_python_bridge.yaml +++ b/src/ndi/cloud/sync/ndi_matlab_python_bridge.yaml @@ -709,6 +709,25 @@ not_yet_ported: MATLAB path resolver. Python handles this inside SyncIndex.read() and SyncIndex.write() methods. + - name: fetchManifest + matlab_path: "+ndi/+cloud/+sync/+internal/fetchManifest.m" + matlab_last_sync_hash: "0ae6bfad" + status: ported_differently + python_path: "ndi/cloud/filehandler.py" + decision_log: > + MATLAB extracts fetchManifest as a standalone helper so the + "resolve one known uid without going through the per-document + batch signed-URL cache" path can be exercised in isolation. + Python has the same behavior baked directly into + ndi.cloud.filehandler._fetch_manifest (commit 5ab9c04: the + single-manifest fetch bypasses BatchSignedUrlLookup and calls + getFileDetails on the manifest uid). The single-uid direct + path is what both languages need to avoid the ~12 h penalty of + walking a lightsheet document's entire batch to answer one + question -- see VH-Lab/NDI-matlab#1010. No separate + fetch_manifest module in Python: it stays inline with the + caller that needs the manifest bytes. + - name: uploadedDocumentIds matlab_path: "+ndi/+cloud/+sync/+internal/uploadedDocumentIds.m" matlab_last_sync_hash: "29546720" diff --git a/tests/test_cloud_client_retry.py b/tests/test_cloud_client_retry.py index 4e76e8ac..64e8df6e 100644 --- a/tests/test_cloud_client_retry.py +++ b/tests/test_cloud_client_retry.py @@ -103,15 +103,21 @@ def slept(monkeypatch): return waits -def make_client(*script): +def make_client(*script, can_reauth: bool = True): """A client whose transport is the given script. ``__new__`` rather than the constructor: the real one builds a ``requests.Session`` that would have to be replaced anyway. + + ``can_reauth`` defaults to True so ordinary retry tests continue to + behave as if the client were obtained via ``from_env()``. Tests that + verify the "hand-built CloudClient(CloudConfig(token=...)) does not + refresh against env credentials" behavior pass ``can_reauth=False``. """ client = CloudClient.__new__(CloudClient) client.config = CloudConfig() client._session = Transport(*script) + client._can_reauth = can_reauth return client @@ -468,6 +474,31 @@ def refresh(): assert captured_authorizations[0] == "Bearer the-stale-token" assert captured_authorizations[1] == "Bearer the-fresh-token" + def test_a_hand_built_client_does_not_reauth_against_env(self, slept, monkeypatch): + """``CloudClient(CloudConfig(token=...))`` uses the given token as + given: a 401 surfaces as ``CloudAuthError`` rather than silently + being replaced by whatever ``authenticate()`` would return from + env. Without this, a test that hard-codes a bad token to verify + auth failure "passes" because CI happens to have valid + ``NDI_CLOUD_USERNAME``/``NDI_CLOUD_PASSWORD`` in env, and the + client swaps in a fresh valid token behind the caller's back. + """ + client = make_client(response(401, text="expired"), can_reauth=False) + + reauth_calls = {"n": 0} + + def should_not_be_called(): + reauth_calls["n"] += 1 + client.config.token = "would-be-fresh" + + monkeypatch.setattr(client, "_reauthenticate", should_not_be_called) + + with pytest.raises(CloudAuthError): + client.get("/datasets") + + assert reauth_calls["n"] == 0, "hand-built client must not refresh from env" + assert client._session.attempts == 1, "no retry when reauth is disabled" + if __name__ == "__main__": pytest.main([__file__]) diff --git a/tests/test_cloud_filehandler.py b/tests/test_cloud_filehandler.py index 4fe1ef24..7be62f44 100644 --- a/tests/test_cloud_filehandler.py +++ b/tests/test_cloud_filehandler.py @@ -284,6 +284,23 @@ def fake_get_file(url, path, timeout=300): class TestGetOrCreateCloudClient: """Tests for get_or_create_cloud_client.""" + @pytest.fixture(autouse=True) + def reset_ambient_client(self): + """The ambient CloudClient is cached process-wide, so any earlier + test in this session that did a real cloud fetch (e.g. the live + Cloud API suites) leaves it populated -- and then + ``test_env_vars_present`` sees the cached client returned instead + of a fresh ``from_env`` call, and fails "Called 0 times". Reset + both slots before AND after each test in this class so the tests + exercise the create path they mean to.""" + import ndi.cloud.filehandler as fh + + fh._ambient_cloud_client = None + fh._ambient_client_key = None + yield + fh._ambient_cloud_client = None + fh._ambient_client_key = None + def test_missing_env_vars(self): from ndi.cloud.exceptions import CloudAuthError from ndi.cloud.filehandler import get_or_create_cloud_client diff --git a/tests/test_cloud_live.py b/tests/test_cloud_live.py index 0a16f1d3..445d52b0 100644 --- a/tests/test_cloud_live.py +++ b/tests/test_cloud_live.py @@ -87,10 +87,20 @@ def cloud_config(): @pytest.fixture(scope="module") def client(cloud_config): - """Return an authenticated CloudClient.""" + """Return an authenticated CloudClient. + + The fixture's config was obtained via ``login()`` from env credentials, + so a mid-run 401/403 (typical after the bearer's TTL expires on a + long-running poll) can be resolved by re-authenticating against the + same env. Callers who hand-build ``CloudClient(CloudConfig(token=...))`` + with a specific token get the default (no reauth) so that a hard-coded + token is used exactly as given. + """ from ndi.cloud.client import CloudClient - return CloudClient(cloud_config) + client = CloudClient(cloud_config) + client._can_reauth = True + return client @pytest.fixture(scope="module") From d7231cc4399927a4341169f48b6c124609cb2bd5 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 11:38:23 +0000 Subject: [PATCH 22/55] bridge: bump three drifted matlab_last_sync_hash values MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Fixes the MATLAB/Python bridge completeness job on Waltham-Data-Science/ NDI-python#320 head 8eb7927. - src/ndi/cloud/sync/ndi_matlab_python_bridge.yaml: fetchManifest 0ae6bfad → a84130a7. 0ae6bfad was the blob hash from a raw file fetch, not a commit; a84130a7 is the commit that introduced the standalone helper. No Python change: the port was always inline in ndi.cloud.filehandler._fetch_manifest. - src/ndi/cloud/ndi_matlab_python_bridge.yaml: updateFileInfoForRemoteFiles 9467a0a7 → a84130a7. Same commit extracted the local fetchManifest helper out of this file into its own module. Nothing to port here: the extraction is a pure refactor and Python calls its own inline helper. - src/ndi/gui/component/ndi_matlab_python_bridge.yaml: ndi_gui_component_ProgressBarWindow b367283b → 2b23dd30. Commit 2b23dd30 dropped the (1,:) size constraint from the addBar Timeout argument-block declaration (a MATLAB argument-validation adjustment). No behavioural change; note added to the decision_log. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/cloud/ndi_matlab_python_bridge.yaml | 2 +- src/ndi/cloud/sync/ndi_matlab_python_bridge.yaml | 2 +- src/ndi/gui/component/ndi_matlab_python_bridge.yaml | 3 ++- 3 files changed, 4 insertions(+), 3 deletions(-) diff --git a/src/ndi/cloud/ndi_matlab_python_bridge.yaml b/src/ndi/cloud/ndi_matlab_python_bridge.yaml index 85d1c2d4..81d43d84 100644 --- a/src/ndi/cloud/ndi_matlab_python_bridge.yaml +++ b/src/ndi/cloud/ndi_matlab_python_bridge.yaml @@ -1026,7 +1026,7 @@ functions: # --- filehandler.py functions --- - name: updateFileInfoForRemoteFiles matlab_path: "+ndi/+cloud/+sync/+internal/updateFileInfoForRemoteFiles.m" - matlab_last_sync_hash: "9467a0a7" + matlab_last_sync_hash: "a84130a7" python_path: "ndi/cloud/filehandler.py" input_arguments: - name: doc_props diff --git a/src/ndi/cloud/sync/ndi_matlab_python_bridge.yaml b/src/ndi/cloud/sync/ndi_matlab_python_bridge.yaml index 7a93d873..6e49d170 100644 --- a/src/ndi/cloud/sync/ndi_matlab_python_bridge.yaml +++ b/src/ndi/cloud/sync/ndi_matlab_python_bridge.yaml @@ -711,7 +711,7 @@ not_yet_ported: - name: fetchManifest matlab_path: "+ndi/+cloud/+sync/+internal/fetchManifest.m" - matlab_last_sync_hash: "0ae6bfad" + matlab_last_sync_hash: "a84130a7" status: ported_differently python_path: "ndi/cloud/filehandler.py" decision_log: > diff --git a/src/ndi/gui/component/ndi_matlab_python_bridge.yaml b/src/ndi/gui/component/ndi_matlab_python_bridge.yaml index 9cfed52d..31fc9fe8 100644 --- a/src/ndi/gui/component/ndi_matlab_python_bridge.yaml +++ b/src/ndi/gui/component/ndi_matlab_python_bridge.yaml @@ -498,7 +498,7 @@ classes: - name: ndi_gui_component_ProgressBarWindow type: class matlab_path: "+ndi/+gui/+component/ProgressBarWindow.m" - matlab_last_sync_hash: "b367283b" + matlab_last_sync_hash: "2b23dd30" python_path: "ndi/gui/component/ProgressBarWindow.py" python_class: "ndi_gui_component_ProgressBarWindow" inherits: "matlab.apps.AppBase" @@ -521,6 +521,7 @@ classes: Porting a silent mode is a design decision for a maintainer rather than something to invent while dating a hash. 2026-09-25: NDI-matlab b367283b added a per-bar Timeout name-value on addBar so a long-running bar can opt out of the global auto-timeout sweep. No Python UI counterpart; no behavioural change here. + 2026-09-28: NDI-matlab 2b23dd30 relaxed the addBar Timeout argument-block size constraint (dropped `(1,:)`, which was rejecting `duration.empty` under some MATLAB releases). Pure MATLAB argument-validation adjustment; no behavioural change here. NDI-python#295: the MATLAB properties this port does not expose under MATLAB's name are enumerated in tests/bridge_property_allowlist.yaml, From 4dff8bf9d9d381483337beeb425242a5f40ae970 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 12:08:18 +0000 Subject: [PATCH 23/55] batch_signed_url: exponential backoff for partial-map retry Was: one 0.5 s retry per scope when the first signed-URL-set response did not name a requested uid. Now: a bounded schedule of 1 s / 3 s / 9 s. #320's Cloud API (User 1, prod) run shows the signed-URL-set index lagging further than 0.5 s after ``waitForAllBulkUploads`` returns -- the wait covers the bulk-upload extraction subsystem reporting done, but not the seam between that and the signed-URL-set subsystem catching up. One retry isn't enough to bridge it, so the round-trip test's manifest reconstruction sees a partial map for the demoNDISeries members, the strict-mode ingest_locations rebuild produces no entries, and the local add rejects the document (main sees the same server-side race as ``uid_misses > 0`` further along in the same test). Three waves cover the observed tail, still bounded (each wave costs one scope refetch, not one per uid), and never fire in the common case where the first map is already complete. API: - New ``partial_map_retry_delays`` kwarg on BatchSignedUrlLookup (tuple of waits). Defaults to DEFAULT_PARTIAL_MAP_RETRY_DELAYS = (1.0, 3.0, 9.0). - ``partial_map_retry_seconds`` kept as a deprecated scalar (None means "use partial_map_retry_delays"; any real number replaces the schedule with a one-element tuple containing that wait, so ``partial_map_retry_seconds=0`` still fires ONE retry with no wait, matching the pre-schedule contract every existing test pins). - ``_retried_scopes`` set becomes ``_scope_retry_attempts`` dict of attempts spent per scope; the loop walks the schedule until a fresh map names the uid or the schedule is exhausted. Two new unit tests pin the multi-wave behaviour (``test_a_multi_wave_schedule_retries_until_a_hit_or_exhaustion`` and ``test_a_schedule_that_never_settles_bounds_retries_to_len``); the round-trip test's cap on ``partial_map_retries`` relaxes from 1 to ``len(DEFAULT_PARTIAL_MAP_RETRY_DELAYS)``. MATLAB mirror lands in VH-Lab/NDI-matlab commit 02880062d on the same branch. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/cloud/batch_signed_url.py | 76 +++++++++++++------ tests/test_cloud_batch_signed_url.py | 85 ++++++++++++++++++++++ tests/test_cloud_file_series_round_trip.py | 25 ++++--- 3 files changed, 155 insertions(+), 31 deletions(-) diff --git a/src/ndi/cloud/batch_signed_url.py b/src/ndi/cloud/batch_signed_url.py index b770c4ae..3c40e7f1 100644 --- a/src/ndi/cloud/batch_signed_url.py +++ b/src/ndi/cloud/batch_signed_url.py @@ -51,12 +51,23 @@ # MATLAB counterpart's testFailingSignerReturnsEmpty. DEFAULT_FAILURE_TTL_SECONDS = 60 -# 500 ms: pause before re-fetching a scope whose first answer was a +# Exponential backoff for re-fetching a scope whose previous answer was a # partial map -- the scope was populated, but it did not name the uid the # caller asked about. On some cloud environments the batch endpoint lags -# briefly behind ``waitForAllBulkUploads``, and the second call names the -# whole set. See Waltham-Data-Science/NDI-python#309. -DEFAULT_PARTIAL_MAP_RETRY_SECONDS = 0.5 +# briefly behind ``waitForAllBulkUploads``, and a later call names the +# whole set. Three retries at 1 s / 3 s / 9 s cover the User-1-prod tail +# observed on Waltham-Data-Science/NDI-python#320: one 0.5 s retry was not +# enough to bridge the settle between bulk-upload extraction and the +# signed-URL-set index catching up. Each retry costs one signer call for +# the whole scope, not one per uid; once the map settles every subsequent +# uid resolves from the fresh cache entry. See +# Waltham-Data-Science/NDI-python#309 and #320. +DEFAULT_PARTIAL_MAP_RETRY_DELAYS: tuple[float, ...] = (1.0, 3.0, 9.0) + +# Deprecated name kept for callers that still pass ``partial_map_retry_seconds`` +# explicitly (a single scalar means "one retry after this many seconds"). +# New code should use ``partial_map_retry_delays`` instead. +DEFAULT_PARTIAL_MAP_RETRY_SECONDS = DEFAULT_PARTIAL_MAP_RETRY_DELAYS[0] # Backoff schedule for :func:`_default_signer` when the batch endpoint call # raises. One transient TLS or connection error at start of run used to @@ -341,14 +352,25 @@ def __init__( signer: Callable[..., tuple[bool, dict[str, Any]]] | None = None, ttl_seconds: float = DEFAULT_TTL_SECONDS, failure_ttl_seconds: float = DEFAULT_FAILURE_TTL_SECONDS, - partial_map_retry_seconds: float = DEFAULT_PARTIAL_MAP_RETRY_SECONDS, + partial_map_retry_delays: tuple[float, ...] | None = None, + partial_map_retry_seconds: float | None = None, sleep: Callable[[float], None] | None = None, disk_cache: bool = False, ) -> None: self._signer = signer or _default_signer self._ttl_seconds = ttl_seconds self._failure_ttl_seconds = failure_ttl_seconds - self._partial_map_retry_seconds = partial_map_retry_seconds + # partial_map_retry_seconds (scalar, deprecated) reduces to a + # one-element schedule, kept for callers/tests that pin the old + # single-retry contract. The scalar names the WAIT before the + # single retry, so ``0`` still fires one retry (with no wait), + # matching the pre-schedule behavior. + if partial_map_retry_delays is not None: + self._partial_map_retry_delays: tuple[float, ...] = tuple(partial_map_retry_delays) + elif partial_map_retry_seconds is not None: + self._partial_map_retry_delays = (partial_map_retry_seconds,) + else: + self._partial_map_retry_delays = DEFAULT_PARTIAL_MAP_RETRY_DELAYS # Injected only for tests: the retry has a real wait, and a test # that runs it 20 times should not pay 10 seconds for it. self._sleep = sleep or time.sleep @@ -361,7 +383,11 @@ def __init__( self._disk_cache = disk_cache self._cache: dict[str, _CacheEntry] = {} self._failed_scopes: dict[str, _FailureEntry] = {} - self._retried_scopes: set[str] = set() + # Per-scope count of partial-map retries already spent. Bounded + # by ``len(self._partial_map_retry_delays)``: once exhausted, a + # subsequent uid miss on the scope surfaces as a data-drift miss + # without another signer call. + self._scope_retry_attempts: dict[str, int] = {} self._warned: set[str] = set() self._stats = Stats() # A batch fetch is IO-bound and slow, and download handlers can be @@ -472,23 +498,31 @@ def lookup( # scope), or a lagged index (the file IS in the scope on the # server, but the batch endpoint's map has not caught up yet). # Only the second is worth a retry, and this side has no way - # to distinguish them a priori -- so retry ONCE per scope, - # sleep briefly first, and if the second answer still does not - # name the uid, take that as data drift and warn. + # to distinguish them a priori -- so retry with a bounded + # exponential-backoff schedule (default 1 s / 3 s / 9 s), + # sleep between attempts, and if the last answer still does + # not name the uid, take that as data drift and warn. # - # Bounded and per-scope: a data-drift miss costs one extra - # signer call for the whole scope, not one per uid. A lagged - # index that recovers replaces the cache entry for every - # subsequent uid in the same scope, so a whole series' worth - # of misses becomes a whole series' worth of hits from one - # retry. See Waltham-Data-Science/NDI-python#309. - if cache_key not in self._retried_scopes: - self._retried_scopes.add(cache_key) - if self._partial_map_retry_seconds > 0: - self._sleep(self._partial_map_retry_seconds) + # Bounded and per-scope: a data-drift miss costs at most + # len(delays) extra signer calls for the whole scope, not + # one per uid. A lagged index that recovers replaces the + # cache entry for every subsequent uid in the same scope, + # so a whole series' worth of misses becomes a whole + # series' worth of hits from one retry. Three waves cover + # the User-1-prod tail after ``waitForAllBulkUploads`` + # returns; on a User-2-prod-shaped environment the first + # retry lands and the rest never fire. See + # Waltham-Data-Science/NDI-python#309 and #320. + attempts_spent = self._scope_retry_attempts.get(cache_key, 0) + while attempts_spent < len(self._partial_map_retry_delays): + delay = self._partial_map_retry_delays[attempts_spent] + if delay > 0: + self._sleep(delay) # Drop the stale entry so _fetch_scope replaces it fresh. self._cache.pop(cache_key, None) self._stats.partial_map_retries += 1 + attempts_spent += 1 + self._scope_retry_attempts[cache_key] = attempts_spent refreshed = self._fetch_scope( cache_key, cloud_dataset_id, @@ -528,7 +562,7 @@ def clear(self) -> None: with self._lock: self._cache.clear() self._failed_scopes.clear() - self._retried_scopes.clear() + self._scope_retry_attempts.clear() self._warned.clear() self._stats = Stats() diff --git a/tests/test_cloud_batch_signed_url.py b/tests/test_cloud_batch_signed_url.py index 16cab565..5be56079 100644 --- a/tests/test_cloud_batch_signed_url.py +++ b/tests/test_cloud_batch_signed_url.py @@ -323,6 +323,91 @@ def test_a_scope_is_only_retried_once(self): # The one uid that IS in the map still resolves. assert lookup.lookup("ds1", "doc1", "chunks", "u_2") == "https://s3/u_2" + def test_a_multi_wave_schedule_retries_until_a_hit_or_exhaustion(self): + """A schedule of several delays fires each in order until either + the map settles or the schedule is exhausted. + + On User-1-prod after a bulk upload the batch endpoint has been + observed lagging further than the original single 0.5 s retry + covered (Waltham-Data-Science/NDI-python#320), so the default + schedule is now a bounded exponential backoff rather than one + shot. Each wave still costs one scope refetch, not one per uid. + """ + + class _EventualSigner: + """Partial map for the first N answers, then the full map.""" + + def __init__(self, misses_before_hit: int): + self.misses_before_hit = misses_before_hit + self.calls = 0 + + def __call__(self, dataset_id, document_id, *, file_series="", client=None): + self.calls += 1 + if self.calls <= self.misses_before_hit: + return True, {"files": {"u_present": "https://s3/u_present"}} + return True, { + "files": { + "u_present": "https://s3/u_present", + "u_lagged": "https://s3/u_lagged", + } + } + + from ndi.cloud.batch_signed_url import BatchSignedUrlLookup + + # Three waves scheduled; the map settles on the SECOND wave, so + # one initial fetch + two retries = three signer calls, then no + # more. + sleeps: list[float] = [] + signer = _EventualSigner(misses_before_hit=2) + lookup = BatchSignedUrlLookup( + signer=signer, + partial_map_retry_delays=(0.1, 0.2, 0.4), + sleep=sleeps.append, + ) + + assert lookup.lookup("ds1", "doc1", "chunks", "u_lagged") == "https://s3/u_lagged" + stats = lookup.stats() + assert stats.partial_map_retries == 2 + assert stats.signer_calls == 3 + assert stats.uid_misses == 0 + assert sleeps == [0.1, 0.2], f"expected two waves' worth of sleeps; got {sleeps!r}" + + def test_a_schedule_that_never_settles_bounds_retries_to_len(self): + """If every wave still returns a partial map, retries stop at + the schedule length -- not one per uid, not unbounded.""" + + class _AlwaysPartial: + def __init__(self): + self.calls = 0 + + def __call__(self, dataset_id, document_id, *, file_series="", client=None): + self.calls += 1 + return True, {"files": {"u_present": "https://s3/u_present"}} + + from ndi.cloud.batch_signed_url import BatchSignedUrlLookup + + sleeps: list[float] = [] + signer = _AlwaysPartial() + lookup = BatchSignedUrlLookup( + signer=signer, + partial_map_retry_delays=(0.05, 0.1, 0.2), + sleep=sleeps.append, + ) + + assert lookup.lookup("ds1", "doc1", "chunks", "u_lagged") == "" + stats = lookup.stats() + assert stats.partial_map_retries == 3, "all three waves fire before giving up" + assert stats.signer_calls == 4, "one initial fetch + three retries" + assert stats.uid_misses == 1 + assert sleeps == [0.05, 0.1, 0.2] + + # A second uid on the same scope must not spend another wave. + signer_calls_before = signer.calls + assert lookup.lookup("ds1", "doc1", "chunks", "u_also_lagged") == "" + assert signer.calls == signer_calls_before, "schedule exhausted; no more retries" + stats = lookup.stats() + assert stats.partial_map_retries == 3 + def test_a_retry_that_returns_no_map_at_all_still_falls_back(self): """The retry can itself fail (a fetch that raises, a bad payload). The caller must still get an empty string, not a crash.""" diff --git a/tests/test_cloud_file_series_round_trip.py b/tests/test_cloud_file_series_round_trip.py index 95b6e095..099d24f4 100644 --- a/tests/test_cloud_file_series_round_trip.py +++ b/tests/test_cloud_file_series_round_trip.py @@ -348,21 +348,26 @@ def test_members_survive_a_download_from_the_cloud( f"partial_map_retries={stats.partial_map_retries}, " f"failure_reason={stats.last_failure_reason!r}" ) - # One presign call for the scope, plus at most one retry if the - # first answer was a partial map (Waltham-Data-Science/ - # NDI-python#309 -- some environments lag briefly after the bulk - # upload). Anything beyond that would be per-uid fallback, which - # the uid_misses assertion above already rules out; this is just - # for a clearer failure message. + # One presign call for the scope, plus up to len(retry_delays) + # partial-map retries if the first answer(s) missed + # (Waltham-Data-Science/NDI-python#309 and #320 -- User 1 prod + # can lag long enough after ``waitForAllBulkUploads`` that the + # first retry is not enough, so the default schedule is now + # 1 s / 3 s / 9 s). Anything beyond that would be per-uid + # fallback, which the uid_misses assertion above already rules + # out; this is just for a clearer failure message. + from ndi.cloud.batch_signed_url import DEFAULT_PARTIAL_MAP_RETRY_DELAYS + assert stats.signer_calls == 1 + stats.partial_map_retries, ( f"the members of one series should cost ONE presign call, " - f"plus at most one retry if the first map was partial. " + f"plus at most one retry per partial-map wave. " f"Got signer_calls={stats.signer_calls}, " f"partial_map_retries={stats.partial_map_retries}." ) - assert stats.partial_map_retries <= 1, ( - f"the retry is bounded to once per scope; " - f"got partial_map_retries={stats.partial_map_retries}." + assert stats.partial_map_retries <= len(DEFAULT_PARTIAL_MAP_RETRY_DELAYS), ( + f"the retry is bounded by the delay schedule; " + f"got partial_map_retries={stats.partial_map_retries}, " + f"schedule length={len(DEFAULT_PARTIAL_MAP_RETRY_DELAYS)}." ) def test_members_survive_a_download_from_the_cloud_with_sync_files_false( From e67e17c0485bf6a6b653abf924479ba392f3347f Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 12:09:48 +0000 Subject: [PATCH 24/55] bridge: bump batchSignedUrlLookup hash to the exponential-backoff commit MATLAB's batchSignedUrlLookup.m moved to 02880062 (partial-map retry schedule went from a single 0.5 s wait to a bounded 1 s / 3 s / 9 s schedule). Python's counterpart on this branch shipped the mirror change in the previous commit (4dff8bf), so this hash bump records a same-commit port, not a drift. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/cloud/ndi_matlab_python_bridge.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/ndi/cloud/ndi_matlab_python_bridge.yaml b/src/ndi/cloud/ndi_matlab_python_bridge.yaml index 81d43d84..93a3bc0b 100644 --- a/src/ndi/cloud/ndi_matlab_python_bridge.yaml +++ b/src/ndi/cloud/ndi_matlab_python_bridge.yaml @@ -1486,7 +1486,7 @@ not_yet_ported: - name: batchSignedUrlLookup matlab_path: "+ndi/+cloud/+download/+internal/batchSignedUrlLookup.m" python_path: "ndi/cloud/batch_signed_url.py" - matlab_last_sync_hash: "27661a3" + matlab_last_sync_hash: "02880062" status: ported decision_log: > PORTED for NDI-python#262. Turns one-call-per-uid into one call per From 39be040871d9309f7a06d10593f0697e16a5669a Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 12:37:39 +0000 Subject: [PATCH 25/55] filehandler: single-file DID fetches bypass batch signed-URL scope When DID reads a series member, it fetches the manifest first via the same download handler with seriesName="". That call previously threaded documentId through to fetch_cloud_file, which used the whole-document batch scope, and for a lightsheet-scale pyramid document (15k+ files) the batch endpoint spent 60-90 s signing URLs the caller had no use for -- just to answer the ONE manifest URL. The subsequent seriesName="chunk.bin" call then fired its own scope, and the two answers overlapped almost completely. Now: when seriesName is empty, download_file_from_cloud clears ndi_document_id and the fetch takes the direct getFileDetails path (O(1)), matching what _fetch_manifest already does for the internal manifest fetch. When DID follows up with a member fetch (seriesName="chunk.bin"), the batch fires there and amortizes across every member exactly as it did before. test_ordinary_file_still_carries_document_scope is repurposed as test_a_single_file_fetch_bypasses_batch, pinning the new invariant. MATLAB mirror lands in VH-Lab/NDI-matlab commit 20391306 on the same branch (didsqlite.download_file_from_cloud gained the same guard). Real-world impact: on User 1 prod, a level swap that previously fired a redundant 76 s whole-doc signed-URL-set job now skips it entirely, and only the chunks-series scope needed for the actual member fetches runs -- 76 s of dark screen goes away. See Waltham-Data-Science/NDI-python#320. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/cloud/filehandler.py | 16 ++++++++++++++++ tests/test_cloud_batch_signed_url.py | 25 +++++++++++++++++-------- 2 files changed, 33 insertions(+), 8 deletions(-) diff --git a/src/ndi/cloud/filehandler.py b/src/ndi/cloud/filehandler.py index debcfc25..95817b18 100644 --- a/src/ndi/cloud/filehandler.py +++ b/src/ndi/cloud/filehandler.py @@ -1000,6 +1000,22 @@ def download_file_from_cloud( ndi_document_id = str(context.get("documentId", "") or "") series_name = str(context.get("seriesName", "") or "") + # A single-file fetch (a series manifest, or any doc-level attachment) + # arrives here with seriesName="", because there is no member being + # asked for. Going through the per-document batch scope to answer one + # uid is pure loss: for a lightsheet-scale pyramid document that scope + # names 15k+ files, so the batch endpoint spends 60-90 s on a signed- + # URL set the caller has no use for, delaying the ONE URL that + # download actually needs by more than a minute. Same reasoning that + # ``_fetch_manifest`` above uses for the internal manifest fetch: for + # a single uid, the direct ``getFileDetails`` path is O(1) and wins. + # (VH-Lab/NDI-matlab#1010's Python analog. When the caller does need + # a whole-doc scope -- an ordinary series member fetch that follows + # with seriesName="chunk.bin" -- the batch fires there and the cost + # amortizes across every member.) + if not series_name: + ndi_document_id = "" + fetch_cloud_file( uri, dest_path, diff --git a/tests/test_cloud_batch_signed_url.py b/tests/test_cloud_batch_signed_url.py index 5be56079..755c1471 100644 --- a/tests/test_cloud_batch_signed_url.py +++ b/tests/test_cloud_batch_signed_url.py @@ -601,13 +601,20 @@ def test_series_member_uses_context_for_batch_scope(self, tmp_path): assert kwargs["ndi_document_id"] == "doc_ndi" assert kwargs["series_name"] == "stack" - def test_ordinary_file_still_carries_document_scope(self, tmp_path): - """Two documents each with one file must not share a scope. - - An ordinary file (no seriesName) with a documentId still keys its - batch scope on that documentId, so the endpoint returns just that - document's map -- exactly one uid, but the cache is still primed - for the next uid in the same document. + def test_a_single_file_fetch_bypasses_batch(self, tmp_path): + """An ordinary file (no seriesName) is a SINGLE-uid fetch, so it + must take the direct ``getFileDetails`` path -- NOT ask the batch + endpoint for a whole-document scope to answer one question. + + For a lightsheet-scale pyramid document that scope names 15k+ + files, so the batch endpoint spends 60-90 s signing a set the + caller has no use for, delaying the ONE URL that the download + actually needs by more than a minute. This is the same reasoning + that ``_fetch_manifest`` uses for the internal manifest fetch; + the DID handler path has to make the same choice or the pyramid + pathology reappears whenever DID reads a series member (DID + fetches the manifest via the handler with seriesName="" before + it fetches any members). See Waltham-Data-Science/NDI-python#320. """ from ndi.cloud.filehandler import download_file_from_cloud @@ -620,7 +627,9 @@ def test_ordinary_file_still_carries_document_scope(self, tmp_path): ) _, kwargs = mock_fetch.call_args - assert kwargs["ndi_document_id"] == "doc_ndi" + # documentId is discarded on the single-file path so + # fetch_cloud_file skips the batch scope entirely. + assert kwargs["ndi_document_id"] == "" assert kwargs["series_name"] == "" def test_no_context_disables_batch(self, tmp_path): From 87176ebc26f02a7d3a71269efa90daf1c35e9b01 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 12:48:14 +0000 Subject: [PATCH 26/55] lightsheet: eager signed-URL prefetch across pyramid levels Kick off the async signed-URL-set job for every level's chunk.bin series as soon as the viewer knows the pyramid's level documents, in a single background thread. The first zoom into a new level no longer waits 20-80 s for the batch endpoint to sign that scope -- the URLs are already cached from work done behind the initial level 0 render. - BatchSignedUrlLookup.prefetch_scope: seed one scope without asking about a specific uid. Skips the partial-map retry on purpose (that's a lookup-time concern; a later lookup drives it if needed). - BatchSignedUrlLookup.start_prefetch: daemon thread that iterates scopes and calls prefetch_scope for each, logging per-scope timing, skipping empty document ids, and continuing past BatchScopeUnreachable so one bad level does not stop the rest. - ImagePyramidLoader.startSignedUrlPrefetch: enumerate the pyramid's level docs, build (cloud_dataset_id, doc.id, "chunk.bin") scopes for every level that has a chunk.bin series, and hand them to the module-level lookup. Off with NDI_LIGHTSHEET_PREFETCH_SIGNED_URLS=0. - lightsheetZarr.viewer.openPyramid: discover the cloud dataset id via session.is_in_cloud() and fire the prefetch after the loader builds, so the jobs run while napari is still opening. Sequential rather than parallel across scopes on purpose: 5 levels at ~1 min each hide inside the initial-render wall time, and a parallel refactor would need to unwind the shared lock that serialises signer calls today. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/cloud/batch_signed_url.py | 129 +++++++++++++++++- src/ndi/gui/app/lightsheetZarr/viewer.py | 47 +++++++ src/ndi/pyramid/loader.py | 104 +++++++++++++++ tests/test_cloud_batch_signed_url.py | 160 +++++++++++++++++++++++ 4 files changed, 439 insertions(+), 1 deletion(-) diff --git a/src/ndi/cloud/batch_signed_url.py b/src/ndi/cloud/batch_signed_url.py index 3c40e7f1..5886527f 100644 --- a/src/ndi/cloud/batch_signed_url.py +++ b/src/ndi/cloud/batch_signed_url.py @@ -31,7 +31,7 @@ import threading import time import warnings -from collections.abc import Callable +from collections.abc import Callable, Sequence from dataclasses import dataclass, field from typing import TYPE_CHECKING, Any @@ -566,6 +566,133 @@ def clear(self) -> None: self._warned.clear() self._stats = Stats() + def prefetch_scope( + self, + cloud_dataset_id: str, + ndi_document_id: str, + series_name: str, + *, + client: CloudClient | None = None, + ) -> bool: + """Populate the cache for one scope without asking about a specific uid. + + Seeds the cache so a subsequent :meth:`lookup` on any uid in the + scope resolves without another signer call. Safe to call from a + background thread; the same lock that guards :meth:`lookup` + serialises the fetch so a concurrent lookup on the same scope + does not stampede the endpoint. + + The partial-map retry is NOT run here on purpose: prefetch's + job is to seed the cache, and the retry only helps when a + specific uid is expected to be present. If the map that arrives + is partial, the first lookup that misses will drive its own + retry through the existing bounded schedule. + + Args: + cloud_dataset_id: The dataset id. + ndi_document_id: The NDI document id. Empty returns False + immediately: nothing to batch against. + series_name: ``""`` for a whole-document scope, or a file + series name to scope the batch. + client: Passed to the signer (default signer only). + + Returns: + True when a cache entry now exists for the scope (either + just populated, or already fresh); False otherwise. + """ + if not ndi_document_id: + return False + + cache_key = f"{cloud_dataset_id}/{ndi_document_id}/{series_name}" + now = time.monotonic() + + with self._lock: + entry = self._cache.get(cache_key) + if entry is not None and (now - entry.fetched_at) < self._ttl_seconds: + return True + # Drop stale entry so _fetch_scope replaces it. + if entry is not None: + del self._cache[cache_key] + + entry = self._fetch_scope( + cache_key, + cloud_dataset_id, + ndi_document_id, + series_name, + uid="", + client=client, + ) + return entry is not None + + def start_prefetch( + self, + scopes: Sequence[tuple[str, str, str]], + *, + client: CloudClient | None = None, + ) -> threading.Thread: + """Spawn a daemon thread that prefetches ``scopes`` in order. + + Each scope is ``(cloud_dataset_id, ndi_document_id, series_name)``. + Scopes with an empty ``ndi_document_id`` are skipped (nothing to + batch against). A :class:`BatchScopeUnreachable` from one scope + is logged and the thread continues with the next -- the point + of prefetch is to warm the cache best-effort, not to fail loud. + + Returns the started thread so tests can join it; production + callers can fire and forget. + """ + scopes_list = list(scopes) + + def _run() -> None: + logger.info( + "signed-URL prefetch: warming %d scope(s) in background", + len(scopes_list), + ) + for cloud_dataset_id, ndi_document_id, series_name in scopes_list: + if not ndi_document_id: + logger.info( + "signed-URL prefetch: skipping scope " + "(%s, , series=%r) -- nothing to batch against", + cloud_dataset_id, + series_name, + ) + continue + started = time.monotonic() + try: + ok = self.prefetch_scope( + cloud_dataset_id, + ndi_document_id, + series_name, + client=client, + ) + except BatchScopeUnreachable as exc: + logger.warning( + "signed-URL prefetch: scope (%s, %s, series=%r) unreachable " + "after %.1fs: %s -- continuing with the next scope", + cloud_dataset_id, + ndi_document_id, + series_name, + time.monotonic() - started, + exc, + ) + continue + logger.info( + "signed-URL prefetch: scope (%s, %s, series=%r) %s in %.1fs", + cloud_dataset_id, + ndi_document_id, + series_name, + "cached" if ok else "failed", + time.monotonic() - started, + ) + + thread = threading.Thread( + target=_run, + name="ndi-signed-url-prefetch", + daemon=True, + ) + thread.start() + return thread + # ------------------------------------------------------------------ # Internal # ------------------------------------------------------------------ diff --git a/src/ndi/gui/app/lightsheetZarr/viewer.py b/src/ndi/gui/app/lightsheetZarr/viewer.py index 7d9599f5..7d1a81f4 100644 --- a/src/ndi/gui/app/lightsheetZarr/viewer.py +++ b/src/ndi/gui/app/lightsheetZarr/viewer.py @@ -21,6 +21,29 @@ from typing import Any +def _cloud_dataset_id(session: Any) -> str: + """Discover the remote NDI Cloud dataset id from an opened session. + + ``_open_session`` in ``cli.py`` may hand back either an + ``ndi.session.dir`` or an ``ndi.dataset.dir``; only the dataset + exposes ``is_in_cloud()``, which returns ``(in_cloud, id)`` off + of the ``dataset_remote`` document written the first time the + dataset was uploaded. Everything else (plain session, non-cloud + open, older reader) reads as ``""`` -- no cloud context, no + prefetch to do. + """ + checker = getattr(session, "is_in_cloud", None) + if not callable(checker): + return "" + try: + in_cloud, cloud_id = checker() + except Exception: # noqa: BLE001 - a bad probe is never fatal + return "" + if not in_cloud: + return "" + return str(cloud_id or "") + + class _NapariStatusReporter: """Push a tile-loading status message into napari's status bar. @@ -374,6 +397,30 @@ def openPyramid( flush=True, ) + # Eagerly warm the signed-URL cache for every level's chunk.bin + # series. Each async signed-URL-set job takes 20-80 s server-side, + # so a 5-level pyramid otherwise pays that wall time the first + # time the user zooms into each level; running the jobs in the + # background while level 0 renders hides those waits behind the + # moment the user is already looking at level 0. + try: + cloud_dataset_id = _cloud_dataset_id(session) + if cloud_dataset_id: + loader.startSignedUrlPrefetch(cloud_dataset_id) + else: + print( + "[lightsheet] signed-URL prefetch skipped: no cloud dataset id " + "on this session (local-only open?)", + file=sys.stderr, + flush=True, + ) + except Exception as exc: # noqa: BLE001 - a prefetch failure is never fatal + print( + f"[lightsheet] signed-URL prefetch: could not start ({exc})", + file=sys.stderr, + flush=True, + ) + with progress.stage("opening image viewer"): viewer = napari.Viewer() diff --git a/src/ndi/pyramid/loader.py b/src/ndi/pyramid/loader.py index 5798bd15..bb1938ce 100644 --- a/src/ndi/pyramid/loader.py +++ b/src/ndi/pyramid/loader.py @@ -24,9 +24,55 @@ from __future__ import annotations +import logging +import os import threading from typing import Any +logger = logging.getLogger(__name__) + + +def _docHasChunkBinSeries(doc: Any) -> bool: + """True iff DOC's file metadata mentions the ``chunk.bin`` series. + + Named at module scope so it stays testable and so the prefetch + path stays honest about what "no chunk.bin" means: skip. Checks + the DID ``files.series_info`` shape first (the current writer), + and falls back to any ``file_info`` entry whose name starts with + ``chunk.bin`` (older ingests that wrote each member into + ``file_info`` directly). Missing or malformed metadata reads + as "no series", which is safe -- we just don't prefetch for + that doc. + """ + props = getattr(doc, "document_properties", None) or {} + if not isinstance(props, dict): + return False + files = props.get("files") + if not isinstance(files, dict): + return False + + raw_series = files.get("series_info") + series_entries: list[dict] + if isinstance(raw_series, dict): + series_entries = [raw_series] + elif isinstance(raw_series, list): + series_entries = [e for e in raw_series if isinstance(e, dict)] + else: + series_entries = [] + for entry in series_entries: + if str(entry.get("name", "")) == "chunk.bin": + return True + + raw_file_info = files.get("file_info") + if isinstance(raw_file_info, list): + for entry in raw_file_info: + if not isinstance(entry, dict): + continue + name = str(entry.get("name", "")) + if name == "chunk.bin" or name.startswith("chunk.bin_"): + return True + return False + class ImagePyramidLoader: """A lightsheetZarrPyramid, ready to be read. @@ -168,6 +214,64 @@ def startPrefetch(self, level: int = -1) -> threading.Thread | None: reduction=self.reduction, ) + def startSignedUrlPrefetch( + self, + cloud_dataset_id: str, + *, + client: Any = None, + ) -> threading.Thread | None: + """Prefetch signed-URL scopes for every level document, in background. + + Each level's ``chunk.bin`` file series takes 20-80 s server-side + to sign, and the user hits that wait the first time they zoom + into a new level. Kicking off the async signed-URL-set jobs + for every level while the initial level-0 view is loading + hides those waits behind a moment the user was already + watching -- when they later zoom, the URLs are cached. + + Sequential inside one background thread on purpose (see the + design note in the task that added this): parallelizing across + scopes would need a lock refactor and 5 levels at ~1 minute + each still fit inside the initial-render wall time. + + Args: + cloud_dataset_id: The remote NDI Cloud dataset id. Empty + means "no cloud context"; the call is a no-op. + client: Passed to the signer (default signer only). + + Returns: + The background thread on success, ``None`` when disabled + or when there is nothing to prefetch. Off with + ``NDI_LIGHTSHEET_PREFETCH_SIGNED_URLS=0``. + """ + if os.environ.get("NDI_LIGHTSHEET_PREFETCH_SIGNED_URLS", "1").strip().lower() in ( + "0", + "false", + "off", + ): + logger.info("signed-URL prefetch disabled via NDI_LIGHTSHEET_PREFETCH_SIGNED_URLS") + return None + if not cloud_dataset_id: + return None + + self.build() + + scopes: list[tuple[str, str, str]] = [] + for doc in self.docs: + doc_id = getattr(doc, "id", "") or "" + if not doc_id: + continue + if not _docHasChunkBinSeries(doc): + continue + scopes.append((cloud_dataset_id, doc_id, "chunk.bin")) + + if not scopes: + return None + + from ndi.cloud.batch_signed_url import get_default + + return get_default().start_prefetch(scopes, client=client) + # ------------------------------------------------------------------ hooks def registerRefreshHint(self, hint) -> None: diff --git a/tests/test_cloud_batch_signed_url.py b/tests/test_cloud_batch_signed_url.py index 755c1471..4fd00810 100644 --- a/tests/test_cloud_batch_signed_url.py +++ b/tests/test_cloud_batch_signed_url.py @@ -453,6 +453,166 @@ def test_clear_drops_everything(self): assert lookup.stats().signer_calls == 1 # refetched after clear +class TestBatchLookupPrefetch: + """The prefetch path warms one scope without asking about a uid. + + The whole point is to hide the 20-80 s signed-URL-set wall time + behind an initial render the user is already watching -- a later + ``lookup()`` on any uid in the scope must then be a pure cache + hit and NOT run another signer call. + """ + + def test_prefetch_scope_populates_the_cache(self): + """After a prefetch, lookup() serves the map without an extra call.""" + from ndi.cloud.batch_signed_url import BatchSignedUrlLookup + + files = {"u1": "https://s3/u1", "u2": "https://s3/u2"} + signer = _FakeSigner({("ds1", "doc1", "chunk.bin"): files}) + lookup = BatchSignedUrlLookup(signer=signer) + + assert lookup.prefetch_scope("ds1", "doc1", "chunk.bin") is True + assert len(signer.calls) == 1 + + # A subsequent lookup on any uid in the scope must be a pure cache hit. + assert lookup.lookup("ds1", "doc1", "chunk.bin", "u1") == "https://s3/u1" + assert lookup.lookup("ds1", "doc1", "chunk.bin", "u2") == "https://s3/u2" + assert len(signer.calls) == 1, ( + "the prefetched scope must serve subsequent lookups from cache; " + f"signer was called {len(signer.calls)} times" + ) + + def test_prefetch_scope_returns_true_when_already_cached(self): + """A second prefetch of a fresh scope is a no-op; no extra call.""" + from ndi.cloud.batch_signed_url import BatchSignedUrlLookup + + signer = _FakeSigner({("ds1", "doc1", "chunk.bin"): {"u1": "https://s3/u1"}}) + lookup = BatchSignedUrlLookup(signer=signer) + + assert lookup.prefetch_scope("ds1", "doc1", "chunk.bin") is True + assert lookup.prefetch_scope("ds1", "doc1", "chunk.bin") is True + assert len(signer.calls) == 1, ( + "a second prefetch of a fresh scope must not fire another signer call; " + f"got {len(signer.calls)} calls" + ) + + def test_prefetch_scope_returns_false_on_empty_document_id(self): + """No document id means nothing to batch against.""" + from ndi.cloud.batch_signed_url import BatchSignedUrlLookup + + signer = _FakeSigner({}) + lookup = BatchSignedUrlLookup(signer=signer) + + assert lookup.prefetch_scope("ds1", "", "chunk.bin") is False + assert signer.calls == [] + + def test_start_prefetch_iterates_scopes_in_the_background(self): + """The background thread warms every scope it is handed.""" + from ndi.cloud.batch_signed_url import BatchSignedUrlLookup + + scopes_files = { + ("ds1", "doc_a", "chunk.bin"): {"u_a1": "https://s3/a1"}, + ("ds1", "doc_b", "chunk.bin"): {"u_b1": "https://s3/b1"}, + ("ds1", "doc_c", "chunk.bin"): {"u_c1": "https://s3/c1"}, + } + signer = _FakeSigner(scopes_files) + lookup = BatchSignedUrlLookup(signer=signer) + + thread = lookup.start_prefetch( + [ + ("ds1", "doc_a", "chunk.bin"), + ("ds1", "doc_b", "chunk.bin"), + ("ds1", "doc_c", "chunk.bin"), + ] + ) + thread.join(timeout=5.0) + assert not thread.is_alive(), "prefetch thread did not finish" + + # All three scopes must now be cached, so lookup() serves without + # another signer call. + signer_calls_before = len(signer.calls) + assert lookup.lookup("ds1", "doc_a", "chunk.bin", "u_a1") == "https://s3/a1" + assert lookup.lookup("ds1", "doc_b", "chunk.bin", "u_b1") == "https://s3/b1" + assert lookup.lookup("ds1", "doc_c", "chunk.bin", "u_c1") == "https://s3/c1" + assert len(signer.calls) == signer_calls_before, ( + "lookups after prefetch must be pure cache hits; " + f"signer went from {signer_calls_before} to {len(signer.calls)}" + ) + assert ( + signer_calls_before == 3 + ), f"expected one signer call per prefetched scope, got {signer_calls_before}" + + def test_start_prefetch_skips_empty_document_ids(self): + """A scope with an empty ndi_document_id is skipped, not fetched.""" + from ndi.cloud.batch_signed_url import BatchSignedUrlLookup + + scopes_files = { + ("ds1", "doc_a", "chunk.bin"): {"u_a1": "https://s3/a1"}, + ("ds1", "doc_c", "chunk.bin"): {"u_c1": "https://s3/c1"}, + } + signer = _FakeSigner(scopes_files) + lookup = BatchSignedUrlLookup(signer=signer) + + thread = lookup.start_prefetch( + [ + ("ds1", "doc_a", "chunk.bin"), + ("ds1", "", "chunk.bin"), # skipped + ("ds1", "doc_c", "chunk.bin"), + ] + ) + thread.join(timeout=5.0) + assert not thread.is_alive() + + assert len(signer.calls) == 2, ( + "the empty-doc-id scope must be skipped; " + f"expected 2 signer calls, got {len(signer.calls)}" + ) + scopes_fetched = {tuple(c) for c in signer.calls} + assert scopes_fetched == { + ("ds1", "doc_a", "chunk.bin"), + ("ds1", "doc_c", "chunk.bin"), + } + + def test_start_prefetch_continues_after_a_failure(self): + """A BatchScopeUnreachable on one scope must not stop the rest.""" + from ndi.cloud.batch_signed_url import BatchScopeUnreachable, BatchSignedUrlLookup + + good_files = {"u_b1": "https://s3/b1"} + good_files_c = {"u_c1": "https://s3/c1"} + + state = {"calls": 0} + + def signer(dataset_id, document_id, *, file_series="", client=None): + state["calls"] += 1 + if document_id == "doc_a": + raise BatchScopeUnreachable("simulated async-job failure for doc_a") + if document_id == "doc_b": + return True, {"files": dict(good_files)} + if document_id == "doc_c": + return True, {"files": dict(good_files_c)} + return False, {"message": "no such scope"} + + lookup = BatchSignedUrlLookup(signer=signer) + + thread = lookup.start_prefetch( + [ + ("ds1", "doc_a", "chunk.bin"), # raises + ("ds1", "doc_b", "chunk.bin"), + ("ds1", "doc_c", "chunk.bin"), + ] + ) + thread.join(timeout=5.0) + assert not thread.is_alive() + + # The remaining two scopes still landed in the cache. + signer_calls_before = state["calls"] + assert lookup.lookup("ds1", "doc_b", "chunk.bin", "u_b1") == "https://s3/b1" + assert lookup.lookup("ds1", "doc_c", "chunk.bin", "u_c1") == "https://s3/c1" + assert state["calls"] == signer_calls_before, ( + "lookups after a partial prefetch must still be cache hits; " + f"signer went from {signer_calls_before} to {state['calls']}" + ) + + # --------------------------------------------------------------------------- # fetch_cloud_file wiring: the batch cache is consulted, and its answer is # what gets streamed. On a batch miss, the per-uid getFileDetails is called. From bdd2d6ffb9e2ce91592f2d620f06cf12e80351e5 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 12:54:16 +0000 Subject: [PATCH 27/55] lightsheet viewer: timestamp key log lines + upsample fallback on by default Two independent user-facing quality-of-life changes to the viewer's level-swap experience: 1. **Timestamp key ``[lightsheet]`` log lines.** Added ``_ts()`` and ``_ls()`` helpers to ``viewer.py`` and routed the tile-counter, fetch-summary and level-selector prints through them. Lines now look like ``12:34:56.789 [lightsheet] tiles: 168 loaded / ...``, which lets a user quantify the actual UX (signed-URL job wall time, first-tile latency, level-swap-to-first-paint delay) without cross-referencing the ``ndi.cloud.*`` INFO lines that already carry timestamps. Sub-second precision matters when a tile fetch finishes in ~0.5-2 s; a second-only stamp collapsed adjacent lines into the same second and made progression look stepped. Low-frequency debug prints are unchanged. 2. **Flip the upsample fallback default to ON.** A user reported multi-minute black stretches during level swaps on a home connection: without the fallback, napari has nothing to paint for a region until every visible fine-level chunk arrives, which for a 1000-tile crop over ~15 MB/s aggregate can be minutes. With the fallback on, the coarsest-level tiles (which ``prefetchCoarsestLevel`` puts on disk at launch) are upsampled and painted immediately -- blurry but present -- and sharpen as the fine-level fetches arrive. ``NDI_LIGHTSHEET_UPSAMPLE_FALLBACK=0`` (or false/off/no) restores the previous opt-in behaviour. Any other value, including empty (env var unset), enables the fallback. Existing ``TestEnvGate`` flipped to pin the new default and cover the opt-out; the loader stats test is updated to match (``fallback`` line is env-gated, not fetcher-gated, so it appears whenever ``env_on()`` is true). Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/gui/app/lightsheetZarr/viewer.py | 73 ++++++++++++++---------- src/ndi/pyramid/upsample_fallback.py | 27 ++++++--- tests/test_pyramid_loader.py | 22 +++++-- tests/test_pyramid_upsample_fallback.py | 16 ++++-- 4 files changed, 92 insertions(+), 46 deletions(-) diff --git a/src/ndi/gui/app/lightsheetZarr/viewer.py b/src/ndi/gui/app/lightsheetZarr/viewer.py index 7d1a81f4..d33d84da 100644 --- a/src/ndi/gui/app/lightsheetZarr/viewer.py +++ b/src/ndi/gui/app/lightsheetZarr/viewer.py @@ -18,9 +18,36 @@ import sys import threading import time +from datetime import datetime from typing import Any +def _ts() -> str: + """Return the current wall clock as ``HH:MM:SS.mmm``. + + A wall-clock timestamp on every ``[lightsheet]`` line lets a user + quantify the actual UX -- how long the signed-URL job took, how + long between the first tile request and the first tile back, how + long a level swap sat on a black screen -- without cross-referencing + the ``ndi.cloud.*`` INFO lines that already carry timestamps. + + Sub-second precision matters when a tile fetch finishes in + ~0.5-2 s: a second-only stamp collapses adjacent lines into the + same second and makes progression look stepped. + """ + return datetime.now().strftime("%H:%M:%S.%f")[:-3] + + +def _ls(msg: str, *, file=None) -> None: + """Print ``[HH:MM:SS.mmm lightsheet] msg`` to stderr, flushed. + + Named so a caller can write ``_ls(f"tiles: {n}/{m}")`` instead of + building the full prefix by hand. Pass ``file=`` to route to a + stream other than stderr (the fetch counter uses this). + """ + print(f"{_ts()} [lightsheet] {msg}", file=file or sys.stderr, flush=True) + + def _cloud_dataset_id(session: Any) -> str: """Discover the remote NDI Cloud dataset id from an opened session. @@ -176,11 +203,11 @@ def _maybe_print(self, force: bool = False) -> None: if self._last_duration is not None: mb = (self._last_bytes or 0) / (1024.0 * 1024.0) last = f"; last fetch {self._last_duration:.2f}s, {mb:.1f} MB" - stderr_line = ( - f"[lightsheet] tiles: {self._done} loaded / {self._started} requested " - f"(in flight: {in_flight}){last}" + _ls( + f"tiles: {self._done} loaded / {self._started} requested " + f"(in flight: {in_flight}){last}", + file=self._out, ) - print(stderr_line, file=self._out, flush=True) if self._status_reporter is not None: status_msg = ( f"lightsheet: {self._done}/{self._started} tiles" @@ -194,11 +221,9 @@ def _heartbeat_loop(self) -> None: with self._lock: idle = time.monotonic() - self._last_activity if self._started == 0: - print( - "[lightsheet] no tiles requested yet by napari; " - "waiting for first slice ...", + _ls( + "no tiles requested yet by napari; waiting for first slice ...", file=self._out, - flush=True, ) elif self._done < self._started and idle >= self._idle_interval: self._maybe_print(force=True) @@ -215,14 +240,13 @@ def stop(self) -> None: # slice. Whichever it is, silence would hide it. started = self._started in_flight = self._started - self._done - print( - f"[lightsheet] fetch summary: 0 tiles fetched in {wall:.1f}s " + _ls( + f"fetch summary: 0 tiles fetched in {wall:.1f}s " f"(started={started}, in_flight={in_flight}). " "napari either drew only sparse/fill regions, or the async " "slicer never dispatched -- try dragging the Z slider or " "zooming in to force a slice compute.", file=self._out, - flush=True, ) return n = len(self._durations) @@ -233,12 +257,11 @@ def stop(self) -> None: mean_mb = (total_bytes / n) / (1024.0 * 1024.0) total_mb = total_bytes / (1024.0 * 1024.0) throughput = (total_bytes / wall) / (1024.0 * 1024.0) if wall > 0 else 0.0 - print( - f"[lightsheet] fetch summary: {n} tiles, {total_mb:.1f} MB total, " + _ls( + f"fetch summary: {n} tiles, {total_mb:.1f} MB total, " f"{wall:.1f}s wall, mean {mean_dt:.2f}s/tile ({mean_mb:.1f} MB), " f"min {min_dt:.2f}s, max {max_dt:.2f}s, throughput {throughput:.1f} MB/s", file=self._out, - flush=True, ) @@ -725,11 +748,7 @@ def _picker(level: str = default_label): idx = labels.index(level) t0 = time.monotonic() elapsed_start = t0 - session_start - print( - f"[lightsheet] level selector: request {level} at t=+{elapsed_start:.1f}s", - file=sys.stderr, - flush=True, - ) + _ls(f"level selector: request {level} at t=+{elapsed_start:.1f}s") for layer, arrays, scales in zip(layers, per_layer_levels, per_layer_scales): if not arrays or idx >= len(arrays): continue @@ -765,17 +784,13 @@ def _picker(level: str = default_label): layer.refresh() except Exception: # noqa: BLE001 pass - print( - f"[lightsheet] layer {layer.name!r} scale/data swapped " - f"at t=+{time.monotonic() - session_start:.1f}s", - file=sys.stderr, - flush=True, + _ls( + f" layer {layer.name!r} scale/data swapped " + f"at t=+{time.monotonic() - session_start:.1f}s" ) - print( - f"[lightsheet] level selector: request completed in " - f"{time.monotonic() - t0:.2f}s (napari now re-slicing)", - file=sys.stderr, - flush=True, + _ls( + f"level selector: request completed in {time.monotonic() - t0:.2f}s " + "(napari now re-slicing)" ) try: diff --git a/src/ndi/pyramid/upsample_fallback.py b/src/ndi/pyramid/upsample_fallback.py index d22824fd..4ce276bb 100644 --- a/src/ndi/pyramid/upsample_fallback.py +++ b/src/ndi/pyramid/upsample_fallback.py @@ -28,7 +28,7 @@ of interest. Bicubic is prettier but the point is "any pixel beats black" not "publication quality". -Off by default -- opt in with ``NDI_LIGHTSHEET_UPSAMPLE_FALLBACK=1``. +On by default; opt out with ``NDI_LIGHTSHEET_UPSAMPLE_FALLBACK=0``. When on, the reader path in :func:`multiscale._build_block_grid` routes each block through :func:`readChunkWithFallback` instead of :func:`multiscale._read_chunk_from_fetcher`. Reader also fires an @@ -47,13 +47,24 @@ def env_on() -> bool: - """True when the upsample fallback is opted-in via the env var.""" - return os.environ.get("NDI_LIGHTSHEET_UPSAMPLE_FALLBACK", "").strip().lower() in ( - "1", - "true", - "on", - "yes", - ) + """True unless the upsample fallback is explicitly disabled. + + Default flipped to ON in Waltham-Data-Science/NDI-python#320 after + a user reported multi-minute black stretches during level swaps on + a home connection: without the fallback, napari has nothing to + paint for a region until every visible fine-level chunk arrives, + which for a 1000-tile crop over ~15 MB/s aggregate can be minutes. + With the fallback on, the coarsest-level tiles (which + ``prefetchCoarsestLevel`` puts on disk at launch) are upsampled and + painted immediately -- blurry but present -- and sharpen as the + fine-level fetches arrive. + + ``NDI_LIGHTSHEET_UPSAMPLE_FALLBACK=0`` (or false/off/no) restores + the previous opt-in behaviour. Any other value, including empty + (env var unset), enables the fallback. + """ + value = os.environ.get("NDI_LIGHTSHEET_UPSAMPLE_FALLBACK", "").strip().lower() + return value not in ("0", "false", "off", "no") def _fallback_debug() -> bool: diff --git a/tests/test_pyramid_loader.py b/tests/test_pyramid_loader.py index cd31be55..3f46708a 100644 --- a/tests/test_pyramid_loader.py +++ b/tests/test_pyramid_loader.py @@ -197,9 +197,16 @@ def test_registering_without_a_fallback_context_is_a_noop(self): class TestStats(unittest.TestCase): - def test_before_build_the_snapshot_is_empty(self): + def test_before_build_no_fetcher_line(self): + """No fetcher yet means no fetcher line. The fallback stats line + is a module-level counter (env-gated, not fetcher-gated), so it + may still be present -- what this pins is that ``stats()`` does + not fabricate a fetcher summary before build.""" + import os + loader = ImagePyramidLoader(_FakeSession(), _FakeDoc()) - self.assertEqual(loader.stats(), {}) + with mock.patch.dict(os.environ, {"NDI_LIGHTSHEET_UPSAMPLE_FALLBACK": "0"}, clear=False): + self.assertEqual(loader.stats(), {}) def test_after_build_the_fetcher_line_is_present(self): fetcher = _FakeFetcher() @@ -210,6 +217,13 @@ def test_after_build_the_fetcher_line_is_present(self): self.assertEqual(snap.get("fetcher"), "cache=42, cloud=10") def test_fallback_line_only_appears_when_the_env_is_on(self): + """The fallback env is on by default now (#320); the stats line + tracks whichever state ``env_on()`` reports. + + With the env unset the default is ON, and the stats show + 'fallback'. With ``=0`` the fallback is off and the stats + section is absent. + """ import os fetcher = _FakeFetcher() @@ -218,9 +232,9 @@ def test_fallback_line_only_appears_when_the_env_is_on(self): loader.build() with mock.patch.dict(os.environ, {}, clear=False): os.environ.pop("NDI_LIGHTSHEET_UPSAMPLE_FALLBACK", None) - self.assertNotIn("fallback", loader.stats()) - with mock.patch.dict(os.environ, {"NDI_LIGHTSHEET_UPSAMPLE_FALLBACK": "1"}): self.assertIn("fallback", loader.stats()) + with mock.patch.dict(os.environ, {"NDI_LIGHTSHEET_UPSAMPLE_FALLBACK": "0"}): + self.assertNotIn("fallback", loader.stats()) class TestClose(unittest.TestCase): diff --git a/tests/test_pyramid_upsample_fallback.py b/tests/test_pyramid_upsample_fallback.py index 1360b6e1..7f5f0fbc 100644 --- a/tests/test_pyramid_upsample_fallback.py +++ b/tests/test_pyramid_upsample_fallback.py @@ -194,10 +194,12 @@ def test_it_respects_row_major_with_the_expected_stride(self): class TestEnvGate(unittest.TestCase): - def test_absent_env_is_off(self): + def test_absent_env_is_on(self): + """Default flipped in #320 so users see coarse-level fill during + level swaps instead of a black screen while fine tiles load.""" with mock.patch.dict(os.environ, {}, clear=False): os.environ.pop("NDI_LIGHTSHEET_UPSAMPLE_FALLBACK", None) - self.assertFalse(uf.env_on()) + self.assertTrue(uf.env_on()) def test_truthy_env_is_on(self): for value in ("1", "true", "on", "yes", "TRUE"): @@ -206,9 +208,13 @@ def test_truthy_env_is_on(self): ): self.assertTrue(uf.env_on(), value) - def test_zero_is_off(self): - with mock.patch.dict(os.environ, {"NDI_LIGHTSHEET_UPSAMPLE_FALLBACK": "0"}, clear=False): - self.assertFalse(uf.env_on()) + def test_falsy_env_is_off(self): + """The opt-out knob: anything falsy disables the fallback.""" + for value in ("0", "false", "off", "no", "FALSE"): + with mock.patch.dict( + os.environ, {"NDI_LIGHTSHEET_UPSAMPLE_FALLBACK": value}, clear=False + ): + self.assertFalse(uf.env_on(), value) if __name__ == "__main__": From dc91fd32aecd8c272bef1a56965ba69b0bae018c Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 30 Sep 2026 16:19:58 +0000 Subject: [PATCH 28/55] batch_signed_url: prefetch_scope consults the disk cache Live log at viewer launch showed the eager signed-URL prefetch firing a fresh createSignedURLSetJob for a 121,440-URL scope every launch, even after a previous run had populated the on-disk cache. The tell was the missing "signed-URL disk cache: served scope (...) signer not called" line and the immediate "submitting createSignedURLSetJob" right after "warming N scope(s) in background". Root cause: only ``lookup`` (the per-uid path exercised on tile reads) was checking ``_try_disk_cache``. ``prefetch_scope`` -- the entry point ``start_prefetch`` calls in the background thread at viewer launch -- checked only the in-memory cache and fell straight through to ``_fetch_scope``. So a fresh process (empty in-memory cache) never consulted the disk, defeating the whole point of the persistent cache for the prefetch path. Mirror the ``lookup`` shape: between the in-memory check and the fetch, probe ``_try_disk_cache`` when ``self._disk_cache`` is on. A disk hit populates the in-memory cache (``_try_disk_cache`` already does that) and returns True with the scope warmed, no server round trip. ``signed_url_disk_cache.load`` filters URLs that would expire inside the 30 min safety buffer, so a served scope is safe to hand to downstream reads. A miss (no cache file, expired, or corrupt) falls through to the existing ``_fetch_scope`` path unchanged. With this in place the first launch after a URL-set expires still pays the createSignedURLSetJob cost; every subsequent launch inside that TTL warms from disk in <1s and skips the signer entirely. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/cloud/batch_signed_url.py | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/src/ndi/cloud/batch_signed_url.py b/src/ndi/cloud/batch_signed_url.py index 5886527f..89679167 100644 --- a/src/ndi/cloud/batch_signed_url.py +++ b/src/ndi/cloud/batch_signed_url.py @@ -614,6 +614,21 @@ def prefetch_scope( if entry is not None: del self._cache[cache_key] + # Consult the on-disk cache before running the async signed- + # URL-set job -- the whole point of the disk cache is that + # a scope warmed by a previous process should not have to + # re-sign 100k URLs at every viewer launch. Mirrors what + # ``lookup`` does. ``_try_disk_cache`` filters entries whose + # URLs would expire within the safety buffer, so a hit here + # is safe to hand to callers. See + # :mod:`ndi.cloud.signed_url_disk_cache`. + if self._disk_cache: + entry = self._try_disk_cache( + cache_key, cloud_dataset_id, ndi_document_id, series_name + ) + if entry is not None: + return True + entry = self._fetch_scope( cache_key, cloud_dataset_id, From 40ec43d71d07ac10df8c176b1c844f12d2bcb490 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 30 Sep 2026 16:37:30 +0000 Subject: [PATCH 29/55] batch_signed_url: carry filesExpireAt/expiresAt from job status into result Explains why ~/.ndi/signed-url-cache// stayed empty after a full successful signing on a fresh machine (Waltham-Data-Science/NDI-python#320). The disk cache's save() correctly refuses to persist a payload that carries neither ``filesExpireAt`` nor ``expiresAt`` -- it will not invent a TTL. The default signer was handing it exactly such a payload: the result blob from getSignedURLSetResult only carries {jobId, datasetId, documentId, generatedAt, fileCount, files}, and the URL-expiration fields live one hop earlier, on the job STATUS returned by waitForSignedURLSetJob (see its docstring, which lists filesExpireAt/expiresAt as job-status fields, not result-blob fields). The signer had ``status`` in hand but only returned ``answer`` from getSignedURLSetResult, so those fields were dropped and every save was a silent no-op. From the user's viewpoint: complete signing successful, cache directory empty, next launch re-signs from scratch. Fix: after fetching the result blob, copy filesExpireAt and expiresAt from ``status`` into ``answer`` when present. Only string values are carried, and we don't overwrite a field already on the answer -- if the server ever starts including them in the blob too, that wins. Non-string / missing values fall through to save's own refuse-to- persist path, unchanged. With this in place the first launch on a fresh machine still pays the createSignedURLSetJob cost but now leaves a valid cache entry behind; every subsequent launch inside the URL TTL (minus the 30 min safety buffer) warms from disk via prefetch_scope's disk-cache probe added in dc91fd3 and skips the signer entirely. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/cloud/batch_signed_url.py | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/src/ndi/cloud/batch_signed_url.py b/src/ndi/cloud/batch_signed_url.py index 89679167..c0ef528b 100644 --- a/src/ndi/cloud/batch_signed_url.py +++ b/src/ndi/cloud/batch_signed_url.py @@ -291,6 +291,23 @@ def _default_signer( ) answer = files_api.getSignedURLSetResult(result_url) + # Carry the URL-expiration fields over from the JOB STATUS + # into the RESULT payload. The status is the only place the + # server puts filesExpireAt / expiresAt; the result blob + # itself does not include them (see getSignedURLSetResult's + # docstring for the blob shape). Without this merge the + # disk cache's save() correctly refuses to persist a + # payload it cannot age-check, and the on-disk cache stays + # empty even after a full successful signing -- exactly + # what we saw on the first fresh-machine run. Non-string / + # missing values are dropped so an accidental None does + # not overwrite a good value we might learn to fill in + # elsewhere later. + if isinstance(answer, dict): + for expiry_key in ("filesExpireAt", "expiresAt"): + value = status.get(expiry_key, "") if hasattr(status, "get") else "" + if isinstance(value, str) and value and expiry_key not in answer: + answer[expiry_key] = value n_files = 0 if isinstance(answer, dict) and isinstance(answer.get("files"), dict): n_files = len(answer["files"]) From b8d49fd86c56bcd3088be69255eecc88fd1a1ab0 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 30 Sep 2026 16:38:53 +0000 Subject: [PATCH 30/55] signed_url_disk_cache: warn loudly when refusing to persist un-cacheable payload Follow-up to 40ec43d. The un-cacheable-payload branch of save() was returning silently, which is exactly how the previous signer bug (dropping filesExpireAt/expiresAt on the way from job status to result blob) stayed invisible: complete signings, empty cache directory, next launch re-signs from scratch, no log line naming the scope or the cause. WARN when a scope is refused for having neither filesExpireAt nor expiresAt; that's a caller bug (a signer must carry an expiry it saw upstream through to the payload it hands save) rather than an environmental issue, and the noise turns "the cache mysteriously never fills" into "here is the exact scope and the exact reason". Keep the I/O-error branches (missing dir, permission, disk full, json.dumps failure) silent -- those are legitimate best-effort skips. Rationale for warn-not-raise: the surrounding read has already succeeded and we don't want to break it just because the cache side of things bounced. WARN is loud in a log the user is already watching and quiet in production logging setups that filter WARNING out. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/cloud/signed_url_disk_cache.py | 24 +++++++++++++++++++++++- 1 file changed, 23 insertions(+), 1 deletion(-) diff --git a/src/ndi/cloud/signed_url_disk_cache.py b/src/ndi/cloud/signed_url_disk_cache.py index 5354f0e3..0b740581 100644 --- a/src/ndi/cloud/signed_url_disk_cache.py +++ b/src/ndi/cloud/signed_url_disk_cache.py @@ -251,7 +251,29 @@ def save( files_expire_at = _read_string(payload, "filesExpireAt") expires_at = _read_string(payload, "expiresAt") if not files_expire_at and not expires_at: - # Refuse to persist an un-age-checkable payload. + # Refuse to persist an un-age-checkable payload. This is a + # CALLER bug, not an environmental issue -- a signer that hands + # us a payload with neither expiry field means the disk cache + # will silently stay empty forever, which is exactly how the + # Waltham-Data-Science/NDI-python#320 first-fresh-machine run + # spent minutes signing 121k URLs and left no trace to reuse. + # WARN rather than raise: the surrounding read has already + # succeeded and we don't want to break it, but the noise is + # what turns "the cache mysteriously never fills" into "here + # is the exact scope and the exact reason". Silent on I/O + # errors (missing dir, permission, disk full) below -- those + # are legitimate best-effort skips, not callee bugs. + logger.warning( + "signed-URL disk cache: refusing to persist scope " + "(dataset=%s, document=%s, series=%r): payload carries " + "neither 'filesExpireAt' nor 'expiresAt', so the cache " + "cannot age-check it. This is a signer bug -- the URL " + "expiration must be carried on the payload passed to " + "save(). Cache stays empty for this scope.", + dataset_id, + document_id, + series_name, + ) return generated_at = _read_string(payload, "generatedAt") From 298fabd3d08bdb8a31cef8ecfdd7fde51dc2c3e0 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 00:42:32 +0000 Subject: [PATCH 31/55] bridge: note lightsheet tileBudgetBytes default moved 8 MB -> 32 MB Mirrors VH-Lab/NDI-matlab@5c9e8a83a on the fromOMEZarr and makePyramid entries. Python port is still porting_deferred for these; this change updates the decision_log so the deferred port lands on the current MATLAB default. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- .../doc_lightsheet/ndi_matlab_python_bridge.yaml | 14 ++++++++------ 1 file changed, 8 insertions(+), 6 deletions(-) diff --git a/src/ndi/fun/doc_lightsheet/ndi_matlab_python_bridge.yaml b/src/ndi/fun/doc_lightsheet/ndi_matlab_python_bridge.yaml index b30f2eaf..8725efc5 100644 --- a/src/ndi/fun/doc_lightsheet/ndi_matlab_python_bridge.yaml +++ b/src/ndi/fun/doc_lightsheet/ndi_matlab_python_bridge.yaml @@ -39,9 +39,10 @@ functions: one lightsheetZarrLevel document per unique array path across all multiscales entries. Level 0 is deduped when two entries reference the same NGFF path (typical for mean + max pyramids). c8f3479 added - tileBudgetBytes (default 8 MB uncompressed) and an optional chunks - override, forwarded to makePyramid so each level is re-tiled to hit - the budget. bd6a245 added materializeChunks (default false) which + tileBudgetBytes (default 32 MB uncompressed; was 8 MB until the + sweet-spot measurement bumped it) and an optional chunks override, + forwarded to makePyramid so each level is re-tiled to hit the + budget. bd6a245 added materializeChunks (default false) which makePyramid uses to read each level from the source OME-Zarr and write chunk.bin_ files onto the level documents before database_add -- the Python port lands the same option and calls @@ -59,9 +60,10 @@ functions: reduction_function on the level names the reduction ('none' for the shared raw level 0, 'mean' / 'max' / ... for reduced levels). c8f3479 drops the source zarr's chunk shape and re-chooses per - level via chooseTileShape to hit tileBudgetBytes (default 8 MB - uncompressed, sized for a viewer over an ~200 MB/s link fetching - ~4 tiles in parallel per pan gesture). 04293e3 encodes + level via chooseTileShape to hit tileBudgetBytes (default 32 MB + uncompressed, sized to the measured signing+fetch+decode sweet + spot for a viewer on an ~200 MB/s link; bumped up from 8 MB after + ~121k-chunk level 0s in production). 04293e3 encodes axes_units as a comma-separated string (e.g. 'micrometer,micrometer,micrometer') and leaves channel_names empty by default, matching the schema type change from matrix From eb2387a0fea213e0be0f48b6177570993046b23f Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 00:44:12 +0000 Subject: [PATCH 32/55] bridge: bump lightsheet sync hashes to NDI-matlab@5c9e8a8 Follow-up to 298fabd. The bridge-hash guard (test_matlab_bridge_hashes.py::TestNothingDriftsFromMatlab) noticed that fromOMEZarr, makePyramid and chooseTileShape have moved in NDI-matlab since their last recorded sync (71d7387 / 8134a20 / 1d11302) and the yaml did not say which commit the port now matches. That is the error it is there to catch, and the fix is to bump the hash on each affected entry. VH-Lab/NDI-matlab@5c9e8a8 is the commit that just moved the three files for the tileBudgetBytes default change; the Python port stays porting_deferred so no code follows -- just the hash. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/fun/doc_lightsheet/ndi_matlab_python_bridge.yaml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/src/ndi/fun/doc_lightsheet/ndi_matlab_python_bridge.yaml b/src/ndi/fun/doc_lightsheet/ndi_matlab_python_bridge.yaml index 8725efc5..3827669c 100644 --- a/src/ndi/fun/doc_lightsheet/ndi_matlab_python_bridge.yaml +++ b/src/ndi/fun/doc_lightsheet/ndi_matlab_python_bridge.yaml @@ -31,7 +31,7 @@ functions: - name: fromOMEZarr matlab_path: "+ndi/+fun/+doc/+lightsheet/fromOMEZarr.m" - matlab_last_sync_hash: "71d7387" + matlab_last_sync_hash: "5c9e8a8" status: porting_deferred decision_log: >- MATLAB probes an OME-Zarr store with ndr.format.omezarr.listPyramids, @@ -51,7 +51,7 @@ functions: - name: makePyramid matlab_path: "+ndi/+fun/+doc/+lightsheet/makePyramid.m" - matlab_last_sync_hash: "8134a20" + matlab_last_sync_hash: "5c9e8a8" status: porting_deferred decision_log: >- The worker fromOMEZarr calls; writes one lightsheetZarrPyramid @@ -86,7 +86,7 @@ functions: - name: chooseTileShape matlab_path: "+ndi/+fun/+doc/+lightsheet/chooseTileShape.m" - matlab_last_sync_hash: "1d11302" + matlab_last_sync_hash: "5c9e8a8" status: porting_deferred decision_log: >- Pure function that picks an OME-Zarr chunk shape landing near a From 9f1f33cea137a3e9862fd18342793a3fbbc633cc Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 00:48:17 +0000 Subject: [PATCH 33/55] bridge: name 5c9e8a8 in lightsheet decision_logs for hash-justify guard Follow-up to eb2387a. The hash-justify guard (test_matlab_bridge_hashes.py::TestAHashChangeIsJustified) requires that when a matlab_last_sync_hash is bumped without a Python-side change, the entry's decision_log names the new commit and says why it is a no-op on the port side. eb2387a bumped the three lightsheet entries to 5c9e8a8 but either rewrote the decision_log without naming the SHA (fromOMEZarr, makePyramid) or did not touch it at all (chooseTileShape). Add one sentence to each decision_log naming 5c9e8a8 and saying the port is still porting_deferred so the new default lands when the port lands. Satisfies the guard without widening the entries. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- .../ndi_matlab_python_bridge.yaml | 18 ++++++++++++++---- 1 file changed, 14 insertions(+), 4 deletions(-) diff --git a/src/ndi/fun/doc_lightsheet/ndi_matlab_python_bridge.yaml b/src/ndi/fun/doc_lightsheet/ndi_matlab_python_bridge.yaml index 3827669c..2f1a76dd 100644 --- a/src/ndi/fun/doc_lightsheet/ndi_matlab_python_bridge.yaml +++ b/src/ndi/fun/doc_lightsheet/ndi_matlab_python_bridge.yaml @@ -46,8 +46,11 @@ functions: makePyramid uses to read each level from the source OME-Zarr and write chunk.bin_ files onto the level documents before database_add -- the Python port lands the same option and calls - into ndr.format.omezarr.readArray equivalent. Ports with the other - entries under this heading. + into ndr.format.omezarr.readArray equivalent. 5c9e8a8 bumped the + default tileBudgetBytes from 8 MB to 32 MB; no Python-side change + is needed while this entry is porting_deferred -- the deferred + port will pick up the new default when it lands. Ports with the + other entries under this heading. - name: makePyramid matlab_path: "+ndi/+fun/+doc/+lightsheet/makePyramid.m" @@ -82,7 +85,10 @@ functions: self-installing private venv (sonpipe pattern; MATLAB's pyenv is never touched). Container bytes are unchanged, so this side of the port stays the same when it lands: use numcodecs.Blosc - directly. Ports with the other entries. + directly. 5c9e8a8 bumped the default tileBudgetBytes from 8 MB + to 32 MB; no Python-side change is needed while this entry is + porting_deferred -- the deferred port will pick up the new + default when it lands. Ports with the other entries. - name: chooseTileShape matlab_path: "+ndi/+fun/+doc/+lightsheet/chooseTileShape.m" @@ -103,7 +109,11 @@ functions: scalar); the equivalent Python idiom is `[ax not in "tc" for ax in axes_order.lower()]`. Ports with the other entries under this heading; the Python builder that mirrors makePyramid will call - it directly. + it directly. 5c9e8a8 changed only the docstring default from + 8 MB to 32 MB (the argument default on this pure helper also + moved, but the Python port -- when it lands -- will take the + budget as a parameter and the makePyramid-side default is what + callers actually hit); no port-side change needed. - name: makeSourceFile matlab_path: "+ndi/+fun/+doc/+lightsheet/makeSourceFile.m" From a814f7c3d239e1c9fb303b57e5e3ead02aa77dfc Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 12:37:40 +0000 Subject: [PATCH 34/55] lightsheet viewer: set viewer.dims.units/axis_labels from the pyramid Napari defaults the dimension sliders and the cursor-position readout to "pixels" when nothing sets viewer.dims.units. For a multiscale lightsheet pyramid where every level carries voxel_size in micrometres, that means the slider range CHANGES as napari auto-swaps multiscale LOD -- the user zooms in, napari switches to a finer level with a larger voxel count, and the slider's "N / M" indicator jumps even though the world position is unchanged. User reports this reads as the viewer briefly rescaling, then settling. Pull axes_order (e.g. "tczyx") and voxel_size_units (e.g. "micrometer") from the finest level document's lightsheetZarrLevel props and set viewer.dims.axis_labels + viewer.dims.units once, after add_image. The axis labels (t, c, z, y, x) populate the slider titles; units attach the micrometre label and switch cursor readout to world coordinates. Non-spatial axes (t, c) get a blank unit since napari treats "" as unitless. Older napari releases may not expose dims.units; falls back to just setting axis_labels on that failure path and logs the specific exception so a user who sees inconsistent slider behaviour knows to look at napari's version. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/gui/app/lightsheetZarr/viewer.py | 34 ++++++++++++++++++++++++ 1 file changed, 34 insertions(+) diff --git a/src/ndi/gui/app/lightsheetZarr/viewer.py b/src/ndi/gui/app/lightsheetZarr/viewer.py index d33d84da..608ae3c6 100644 --- a/src/ndi/gui/app/lightsheetZarr/viewer.py +++ b/src/ndi/gui/app/lightsheetZarr/viewer.py @@ -465,6 +465,40 @@ def openPyramid( for spec in specs: added_layers.append(viewer.add_image(**spec)) + # Set physical units on the dimension sliders. Without this napari + # labels the slider in "pixels" and reports cursor position in + # voxels, so switching multiscale levels appears to the user as + # the slider range changing underneath them. With world units set, + # the slider and status bar read microns (or whatever the level + # declares) and stay stable across level swaps. Pulled from the + # pyramid's finest level since every level shares the same + # axes_order and the same world units. + try: + level0_props = loader.docs[0].document_properties["lightsheetZarrLevel"] + axes = str(level0_props.get("axes_order", "tczyx")) + unit_str = str(level0_props.get("voxel_size_units", "micrometer")) + except Exception: # noqa: BLE001 - fall back to the napari default on any shape surprise + axes = "" + unit_str = "" + if axes: + dim_labels = tuple(axes) + # channel axis has no spatial unit; time axis gets its own unit if + # the level ever carries one, else blank (napari treats "" as unitless) + dim_units = tuple("" if ax.lower() in ("c", "t") else unit_str for ax in axes) + try: + viewer.dims.axis_labels = dim_labels + viewer.dims.units = dim_units + except Exception as exc: # noqa: BLE001 - older napari may not support dims.units + print( + f"[lightsheet] viewer.dims.units set failed ({exc!s}); labels only.", + file=sys.stderr, + flush=True, + ) + try: + viewer.dims.axis_labels = dim_labels + except Exception: + pass + # Install the debounced napari refresh hint via the loader's # public hook. Every async fine-chunk fetch that completes calls # this hint; the hint coalesces a burst of arrivals into one From c1298527103d45d59aab77233ad48ca9a5ef4a27 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 12:58:06 +0000 Subject: [PATCH 35/55] lightsheet viewer: default to native multiscale + add loading overlay The old default was to open the coarsest level as a plain (non-multiscale) layer with a magicgui dock widget swapping layer.data on zoom. That path dates from a napari 0.5 bug where the multiscale slicer never marked a cloud-backed layer loaded=True; on the lightsheet data in production it is now the thing creating the sympthoms the user reports -- a one-minute "no visible loading" pause on zoom-in while the single-layer swap re-slices level 0's full array before the picture updates. Flip the default: pass napari a true multiscale ladder so it picks the level the current zoom actually needs. Fine levels still carry the upsample-fallback reader, so a zoom-in paints coarse-upsampled pixels immediately and refines as fine chunks arrive. The env knob flips to NDI_LIGHTSHEET_SINGLE_LEVEL=1 as the escape hatch if the old napari bug resurfaces on a dataset. Make the single-level-only controls (_attach_level_selector, viewport.attach_viewport_clip) no-op when the layers report layer.multiscale=True -- otherwise they would corrupt napari's MultiScaleData wrapper by swapping layer.data out from under it. Add a visible loading overlay: _NapariStatusReporter now writes to viewer.text_overlay in addition to viewer.status, so fetches in flight show up as a yellow "Loading tiles ... (N in flight)" over the canvas rather than only in the status bar footer. The previous zoom report said the status bar was easy to miss; the overlay is big and lives on the picture itself. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/gui/app/lightsheetZarr/viewer.py | 68 ++++++++++++++++++---- src/ndi/gui/app/lightsheetZarr/viewport.py | 11 ++++ src/ndi/pyramid/multiscale.py | 28 +++++---- 3 files changed, 85 insertions(+), 22 deletions(-) diff --git a/src/ndi/gui/app/lightsheetZarr/viewer.py b/src/ndi/gui/app/lightsheetZarr/viewer.py index 608ae3c6..146067ce 100644 --- a/src/ndi/gui/app/lightsheetZarr/viewer.py +++ b/src/ndi/gui/app/lightsheetZarr/viewer.py @@ -72,29 +72,68 @@ def _cloud_dataset_id(session: Any) -> str: class _NapariStatusReporter: - """Push a tile-loading status message into napari's status bar. + """Push a tile-loading status message into napari's status bar + and -- when fetches are in flight -- a large overlay on the canvas. + + Status bar alone was easy to miss: on the user's zoom-and-wait + cases, nothing in the canvas said "loading", so they read a + coarse-upsampled placeholder as "this is the final image." The + overlay is big and lives on top of the picture, so a fetch in + flight is unambiguous. Called from the fetch counter's main-thread-safe surfaces. Napari's - status bar is a Qt widget, so a background thread calling - ``viewer.status = ...`` would need main-thread dispatch; the - counter's ``_maybe_print`` runs on whichever thread fired the - fetch event. Rather than adding QThread machinery here, we keep - a reference to the viewer and to the last string, and only + status bar and text overlay are Qt widgets, so a background thread + calling ``viewer.status = ...`` or ``viewer.text_overlay.text = ...`` + would need main-thread dispatch; the counter's ``_maybe_print`` + runs on whichever thread fired the fetch event. Rather than adding + QThread machinery here, we keep a reference to the viewer and push on the main thread via a QTimer.singleShot(0, ...). """ def __init__(self, viewer): self._viewer = viewer + self._overlay_configured = False - def push(self, message: str) -> None: + def push(self, message: str, *, in_flight: int = 0) -> None: # QTimer.singleShot is thread-safe; the callback runs on the - # Qt main thread, which is what viewer.status expects. + # Qt main thread, which is what viewer.status / viewer.text_overlay + # expect. try: from qtpy.QtCore import QTimer except ImportError: return v = self._viewer - QTimer.singleShot(0, lambda: setattr(v, "status", message)) + overlay_text = f"Loading tiles ... ({in_flight} in flight)" if in_flight > 0 else "" + + def _apply(): + try: + v.status = message + except Exception: + pass + overlay = getattr(v, "text_overlay", None) + if overlay is None: + return + if not self._overlay_configured: + try: + overlay.font_size = 18 + overlay.position = "top_left" + try: + overlay.color = "yellow" + except Exception: + pass + self._overlay_configured = True + except Exception: + pass + try: + if overlay_text: + overlay.text = overlay_text + overlay.visible = True + else: + overlay.visible = False + except Exception: + pass + + QTimer.singleShot(0, _apply) class _FetchCounter: @@ -214,7 +253,7 @@ def _maybe_print(self, force: bool = False) -> None: + (f" ({in_flight} in flight)" if in_flight else "") + (f" | last {self._last_duration:.2f}s" if self._last_duration else "") ) - self._status_reporter.push(status_msg) + self._status_reporter.push(status_msg, in_flight=in_flight) def _heartbeat_loop(self) -> None: while not self._stop.wait(self._idle_interval): @@ -728,6 +767,11 @@ def _report_loaded(layer) -> None: def _attach_level_selector(viewer, layers, per_layer_levels, per_layer_scales): """Dock a "Resolution" selector that swaps each layer's level. + No-op when the layers are already native multiscale layers: + napari then picks the level itself from the current zoom and + swapping ``layer.data`` out from under it would corrupt the + MultiScaleData wrapper. + Swaps ``layer.data`` AND ``layer.scale`` together so world coordinates stay locked when the pixel dimensions change. No new fetches happen on swap -- the graphs are pre-built during @@ -755,6 +799,10 @@ def _attach_level_selector(viewer, layers, per_layer_levels, per_layer_scales): """ if not layers or not per_layer_levels: return None, [] + # Multiscale layers manage their own level selection; a dock-widget + # swap would fight napari. + if any(getattr(layer, "multiscale", False) for layer in layers): + return None, [] max_levels = max((len(lst) for lst in per_layer_levels), default=0) if max_levels < 2: return None, [] # Nothing to switch between. diff --git a/src/ndi/gui/app/lightsheetZarr/viewport.py b/src/ndi/gui/app/lightsheetZarr/viewport.py index aff99883..3bccac48 100644 --- a/src/ndi/gui/app/lightsheetZarr/viewport.py +++ b/src/ndi/gui/app/lightsheetZarr/viewport.py @@ -84,6 +84,17 @@ def attach_viewport_clip( return None if not layers or not per_layer_levels: return None + # Native multiscale layers manage their own level + viewport + # choice; swapping layer.data from the clip would corrupt the + # MultiScaleData wrapper. Clipping is a single-level-mode tool. + if any(getattr(layer, "multiscale", False) for layer in layers): + print( + "[lightsheet] viewport clip: skipped (layers are native " + "multiscale; napari handles level selection)", + file=sys.stderr, + flush=True, + ) + return None try: clip = ViewportClip(viewer, layers, per_layer_levels, per_layer_scales, picker, labels) diff --git a/src/ndi/pyramid/multiscale.py b/src/ndi/pyramid/multiscale.py index e38d324b..ff5007d3 100644 --- a/src/ndi/pyramid/multiscale.py +++ b/src/ndi/pyramid/multiscale.py @@ -1224,19 +1224,23 @@ def layerSpec( # only a graph rewrite that says "block[c] instead of block[all]". import dask.array as da - # Default is single-level (only the coarsest, as a plain non- - # multiscale layer). Napari 0.5's multiscale slicer has been - # observed to never mark layer.loaded=True on a lazy cloud-backed - # 3D multiscale pyramid: the channel-list spinner spins forever - # and nothing draws. Single-level takes the multiscale slicer out - # of the loop entirely; the layer draws, then a magicgui panel - # lets the user swap between levels manually (see - # :func:`_attach_level_selector` in viewer.py). + # Default is multiscale: napari picks the level that matches the + # current zoom, so a pan or a zoom draws the appropriate level + # without the viewer having to swap layer.data by hand. The fine + # levels carry an upsample-fallback reader, so a zoom-in paints a + # coarse-upsampled placeholder immediately and refines as fine + # chunks arrive. # - # NDI_LIGHTSHEET_MULTISCALE=1 opts back into the multiscale path - # for anyone testing whether the napari-side bug has been fixed - # or for a data shape that does not hit it. - single_level = not _env_true("NDI_LIGHTSHEET_MULTISCALE") + # NDI_LIGHTSHEET_SINGLE_LEVEL=1 opts back into the earlier + # fallback mode: one single-level layer at the coarsest level, + # with a magicgui dock widget that swaps layer.data between + # levels on zoom. Keep this knob available because napari 0.5's + # multiscale slicer has historically failed to mark + # layer.loaded=True on some lazy cloud-backed 3D pyramids (the + # channel-list spinner spins forever, nothing draws); if that bug + # resurfaces on a dataset, the knob is the escape hatch while the + # fix lands. + single_level = _env_true("NDI_LIGHTSHEET_SINGLE_LEVEL") # Build per-channel arrays once; each is a list of one 3D lazy # dask array per level. The level dropdown swaps between the From e65ed9352d8458278c9b703a005b3308b2576a59 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 13:02:52 +0000 Subject: [PATCH 36/55] tests: loader integration tolerates multiscale default c129852 flipped the lightsheet viewer default to a true multiscale ladder, so a layer spec's "data" is now a list of per-level arrays rather than a single 3D array at the coarsest level. Two integration tests assumed the single-level shape: - test_the_loader_returns_one_spec_per_channel: accept data as either a list (take the first level) or a bare 3D array. - test_the_coarsest_level_computes_and_is_not_all_zero: pick the last level from the list, or the bare array if single-level mode is on. Keeps compatibility with NDI_LIGHTSHEET_SINGLE_LEVEL=1 (the escape hatch) so the tests pass under either mode. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- tests/test_pyramid_loader_integration.py | 21 ++++++++++++++++----- 1 file changed, 16 insertions(+), 5 deletions(-) diff --git a/tests/test_pyramid_loader_integration.py b/tests/test_pyramid_loader_integration.py index 280537d8..f130c5a2 100644 --- a/tests/test_pyramid_loader_integration.py +++ b/tests/test_pyramid_loader_integration.py @@ -169,9 +169,17 @@ def test_the_loader_returns_one_spec_per_channel(self, pyramid_doc): assert len(specs) == 2, f"expected 2 layer specs, got {len(specs)}" for spec in specs: assert "data" in spec - # The single-level scaffold delivers a 3D array - # (channel axis stripped) at the coarsest level. - data_shape = getattr(spec["data"], "shape", ()) + # Multiscale default: spec["data"] is a list of 3D arrays + # (channel axis stripped), one per level finest-first. + # Single-level mode (NDI_LIGHTSHEET_SINGLE_LEVEL=1) would + # give a single 3D array instead; accept either. + data = spec["data"] + if isinstance(data, list): + assert data, "expected at least one level in multiscale ladder" + first = data[0] + else: + first = data + data_shape = getattr(first, "shape", ()) assert len(data_shape) == 3, f"expected 3D per-channel data, got shape {data_shape}" finally: loader.close() @@ -192,8 +200,11 @@ def test_the_coarsest_level_computes_and_is_not_all_zero(self, pyramid_doc): loader = ImagePyramidLoader(session, doc, reduction="mean") try: specs = loader.specs - data = specs[0]["data"] # coarsest-level array for channel 0 - arr = data.compute() + # Multiscale default: data is a finest-first list; coarsest is + # the last entry. Single-level mode gives a single array. + data = specs[0]["data"] + coarsest = data[-1] if isinstance(data, list) else data + arr = coarsest.compute() assert arr.dtype == np.uint16, f"expected uint16, got {arr.dtype}" assert arr.any(), "coarsest-level channel-0 slice is entirely fill_value" finally: From 18f704247ddf93b7a7ee46f67c7dcb8d2ff600e8 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 13:19:17 +0000 Subject: [PATCH 37/55] lightsheet viewer: use pint.Unit for viewer.dims.units Current napari rejects plain strings on viewer.dims.units with a pydantic is_instance_of error -- the field now wants pint.Unit instances. a814f7c set strings and fell back to labels-only on every launch; build pint.Unit values instead so the dim sliders actually read micrometers. Non-spatial axes (c, t) get pint.Unit("") (dimensionless) to match their lack of a world-space unit on the pyramid document. Also expand the multiscale default comment: a user report of a ~2-minute first-paint wait on the Maddie dataset (napari's slicer sitting on "no tiles requested yet" while vispy fired destroyed- dispatcher warnings) is still the expected worst case before the first interaction; the view does come alive and subsequent zooms are responsive, but the escape hatch (NDI_LIGHTSHEET_SINGLE_LEVEL=1) stays documented. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/gui/app/lightsheetZarr/viewer.py | 27 ++++++++++++++++-------- src/ndi/pyramid/multiscale.py | 21 +++++++++--------- 2 files changed, 29 insertions(+), 19 deletions(-) diff --git a/src/ndi/gui/app/lightsheetZarr/viewer.py b/src/ndi/gui/app/lightsheetZarr/viewer.py index 146067ce..ecfc95a7 100644 --- a/src/ndi/gui/app/lightsheetZarr/viewer.py +++ b/src/ndi/gui/app/lightsheetZarr/viewer.py @@ -521,22 +521,31 @@ def openPyramid( unit_str = "" if axes: dim_labels = tuple(axes) - # channel axis has no spatial unit; time axis gets its own unit if - # the level ever carries one, else blank (napari treats "" as unitless) - dim_units = tuple("" if ax.lower() in ("c", "t") else unit_str for ax in axes) try: viewer.dims.axis_labels = dim_labels + except Exception: + pass + # viewer.dims.units wants pint.Unit instances in current napari + # (strings get rejected by the pydantic validator). Build them + # via pint when it is installed; skip units silently otherwise + # -- labels are already set and are the main thing users read. + # Spatial axes get the pyramid's unit, non-spatial (c, t) stay + # dimensionless. + try: + import pint + + ureg = pint.UnitRegistry() + spatial_unit = ureg.Unit(unit_str) if unit_str else ureg.Unit("") + dim_units = tuple( + ureg.Unit("") if ax.lower() in ("c", "t") else spatial_unit for ax in axes + ) viewer.dims.units = dim_units - except Exception as exc: # noqa: BLE001 - older napari may not support dims.units + except Exception as exc: # noqa: BLE001 - a units set failure is never fatal print( - f"[lightsheet] viewer.dims.units set failed ({exc!s}); labels only.", + f"[lightsheet] viewer.dims.units set skipped ({exc!s}); labels only.", file=sys.stderr, flush=True, ) - try: - viewer.dims.axis_labels = dim_labels - except Exception: - pass # Install the debounced napari refresh hint via the loader's # public hook. Every async fine-chunk fetch that completes calls diff --git a/src/ndi/pyramid/multiscale.py b/src/ndi/pyramid/multiscale.py index ff5007d3..9b23e5d8 100644 --- a/src/ndi/pyramid/multiscale.py +++ b/src/ndi/pyramid/multiscale.py @@ -1225,21 +1225,22 @@ def layerSpec( import dask.array as da # Default is multiscale: napari picks the level that matches the - # current zoom, so a pan or a zoom draws the appropriate level - # without the viewer having to swap layer.data by hand. The fine - # levels carry an upsample-fallback reader, so a zoom-in paints a - # coarse-upsampled placeholder immediately and refines as fine - # chunks arrive. + # current zoom and the fine levels carry the upsample-fallback + # reader, so zoom-in paints a coarse-upsampled placeholder + # immediately and refines as fine chunks arrive. The initial + # paint can be slow on a large volume (minutes on the Maddie + # dataset) because napari's slicer walks the whole ladder before + # it marks the layer loaded, but once a user manipulates the + # view it does draw and subsequent zooms are responsive. # # NDI_LIGHTSHEET_SINGLE_LEVEL=1 opts back into the earlier # fallback mode: one single-level layer at the coarsest level, # with a magicgui dock widget that swaps layer.data between - # levels on zoom. Keep this knob available because napari 0.5's + # levels on zoom. Keep this knob available because napari's # multiscale slicer has historically failed to mark - # layer.loaded=True on some lazy cloud-backed 3D pyramids (the - # channel-list spinner spins forever, nothing draws); if that bug - # resurfaces on a dataset, the knob is the escape hatch while the - # fix lands. + # layer.loaded=True on some lazy cloud-backed 3D pyramids; if + # that bug resurfaces on a dataset, the knob is the escape + # hatch while the fix lands. single_level = _env_true("NDI_LIGHTSHEET_SINGLE_LEVEL") # Build per-channel arrays once; each is a list of one 3D lazy From c626496e680c37e142785f79b06c4c7a88856b0f Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 13:19:51 +0000 Subject: [PATCH 38/55] lightsheet viewer: revert multiscale default; keep knob and dims.units Multiscale default (c129852) did not survive production: on the Maddie dataset first paint sat for ~2 minutes with "no tiles requested yet by napari" while vispy fired destroyed-dispatcher warnings, and once the view did come alive a plain zoom-in did NOT trigger a level swap -- the user had to zoom out and back in to get finer data to render. The single-level + Resolution-picker flow does swap on a plain zoom via the zoom-level-swap listener and paints promptly, so that is the default again. NDI_LIGHTSHEET_MULTISCALE=1 opts into the multiscale path for anyone testing whether napari's slicer has been fixed on a given data shape. The level-selector/viewport-clip no-op guards and the loading overlay stay in: both are correct in either mode and the dims.units pint fix is independent of the default. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/pyramid/multiscale.py | 34 +++++++++++++++++----------------- 1 file changed, 17 insertions(+), 17 deletions(-) diff --git a/src/ndi/pyramid/multiscale.py b/src/ndi/pyramid/multiscale.py index 9b23e5d8..bddfb2e7 100644 --- a/src/ndi/pyramid/multiscale.py +++ b/src/ndi/pyramid/multiscale.py @@ -1224,24 +1224,24 @@ def layerSpec( # only a graph rewrite that says "block[c] instead of block[all]". import dask.array as da - # Default is multiscale: napari picks the level that matches the - # current zoom and the fine levels carry the upsample-fallback - # reader, so zoom-in paints a coarse-upsampled placeholder - # immediately and refines as fine chunks arrive. The initial - # paint can be slow on a large volume (minutes on the Maddie - # dataset) because napari's slicer walks the whole ladder before - # it marks the layer loaded, but once a user manipulates the - # view it does draw and subsequent zooms are responsive. + # Default is single-level (only the coarsest, as a plain non- + # multiscale layer). Tried flipping the default to multiscale + # in c129852 and it did not survive production use on the + # Maddie dataset: the first paint sat on "no tiles requested + # yet by napari" for ~2 minutes while vispy spammed destroyed- + # dispatcher warnings, and once the view did come up a plain + # zoom-in did NOT trigger napari's level swap -- the user had + # to zoom out and back in to get a finer level to render. That + # is unacceptable UX for the common case. Single-level takes + # napari's multiscale slicer out of the loop entirely: the + # layer draws promptly at the coarsest level, then a magicgui + # dock widget swaps layer.data between levels (manually, or + # driven by the zoom-level-swap listener on camera events). # - # NDI_LIGHTSHEET_SINGLE_LEVEL=1 opts back into the earlier - # fallback mode: one single-level layer at the coarsest level, - # with a magicgui dock widget that swaps layer.data between - # levels on zoom. Keep this knob available because napari's - # multiscale slicer has historically failed to mark - # layer.loaded=True on some lazy cloud-backed 3D pyramids; if - # that bug resurfaces on a dataset, the knob is the escape - # hatch while the fix lands. - single_level = _env_true("NDI_LIGHTSHEET_SINGLE_LEVEL") + # NDI_LIGHTSHEET_MULTISCALE=1 opts into the native multiscale + # path for anyone testing whether napari's slicer has been + # fixed on a given data shape. + single_level = not _env_true("NDI_LIGHTSHEET_MULTISCALE") # Build per-channel arrays once; each is a list of one 3D lazy # dask array per level. The level dropdown swaps between the From 4f6ee46ef7b0a2d4297a4a6c49e032300eb41781 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 13:31:04 +0000 Subject: [PATCH 39/55] lightsheet viewer: multiscale default + thread-safe status reporter User A/B tested single-level vs multiscale and prefers multiscale: the transition between levels is smoother because napari handles the swap internally, while single-level mutates layer.data and shows a short black frame on every change. Both modes have been observed to hit the same napari slicer wedge on the Maddie dataset (vispy "QBasicTimer destroyed dispatcher" spam, no tile requests ever land), so single-level is not actually safer -- just janker when it does paint. Flip the default back to multiscale. NDI_LIGHTSHEET_SINGLE_LEVEL=1 keeps the earlier single-level + Resolution-picker flow as the opt-in. NDI_LIGHTSHEET_ASYNC=0 is the escape hatch when the async slicer wedges: it blocks add_image on the first sync-sliced paint and bypasses the broken async path. Both knobs are now documented in the comment on the single_level decision. _NapariStatusReporter was calling QTimer.singleShot(0, _apply) from whichever thread the fetch counter happened to run on (the heartbeat is a daemon thread, fetch handlers run on async workers). QTimer requires an event dispatcher on the current thread; without one it did nothing useful and emitted "QBasicTimer::start: current thread's event dispatcher has already been destroyed" into every log line. Switch to a Qt signal/slot: _Bridge is a QObject moved to the main thread with a signal that _apply listens to via a queued connection, so emit() from any worker crosses safely to the main-thread apply. Removes the vispy spam and should keep the overlay reliable. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/gui/app/lightsheetZarr/viewer.py | 99 +++++++++++++++--------- src/ndi/pyramid/multiscale.py | 34 ++++---- 2 files changed, 78 insertions(+), 55 deletions(-) diff --git a/src/ndi/gui/app/lightsheetZarr/viewer.py b/src/ndi/gui/app/lightsheetZarr/viewer.py index ecfc95a7..13502dd5 100644 --- a/src/ndi/gui/app/lightsheetZarr/viewer.py +++ b/src/ndi/gui/app/lightsheetZarr/viewer.py @@ -81,59 +81,82 @@ class _NapariStatusReporter: overlay is big and lives on top of the picture, so a fetch in flight is unambiguous. - Called from the fetch counter's main-thread-safe surfaces. Napari's - status bar and text overlay are Qt widgets, so a background thread - calling ``viewer.status = ...`` or ``viewer.text_overlay.text = ...`` - would need main-thread dispatch; the counter's ``_maybe_print`` - runs on whichever thread fired the fetch event. Rather than adding - QThread machinery here, we keep a reference to the viewer and - push on the main thread via a QTimer.singleShot(0, ...). + Cross-thread safe: push() is called from the fetch counter, which + runs on worker threads (the heartbeat, the fetch handlers). + Napari's status bar and text overlay are Qt widgets and must only + be written from the main thread. A QTimer.singleShot started from + a worker thread has no event dispatcher there and does nothing + useful (and spams "QBasicTimer::start: current thread's event + dispatcher has already been destroyed"), so we use a Qt signal: + signals emitted from a worker thread cross to the owning thread + via a queued connection, which is exactly the main-thread + delivery we need. The signal lives on a QObject (``_Bridge``) + moved to the main thread at construction time. """ def __init__(self, viewer): self._viewer = viewer self._overlay_configured = False + self._bridge = None + try: + from qtpy.QtCore import QCoreApplication, QObject, Qt, Signal - def push(self, message: str, *, in_flight: int = 0) -> None: - # QTimer.singleShot is thread-safe; the callback runs on the - # Qt main thread, which is what viewer.status / viewer.text_overlay - # expect. + class _Bridge(QObject): + update = Signal(str, int) + + def __init__(self, outer): + super().__init__() + self._outer = outer + self.update.connect(self._on_update, Qt.QueuedConnection) + + def _on_update(self, message, in_flight): + self._outer._apply(message, in_flight) + + self._bridge = _Bridge(self) + app = QCoreApplication.instance() + if app is not None: + self._bridge.moveToThread(app.thread()) + except Exception: + self._bridge = None + + def _apply(self, message: str, in_flight: int) -> None: + """Runs on the main thread (queued-signal callback).""" + v = self._viewer try: - from qtpy.QtCore import QTimer - except ImportError: + v.status = message + except Exception: + pass + overlay = getattr(v, "text_overlay", None) + if overlay is None: return - v = self._viewer - overlay_text = f"Loading tiles ... ({in_flight} in flight)" if in_flight > 0 else "" - - def _apply(): + if not self._overlay_configured: try: - v.status = message - except Exception: - pass - overlay = getattr(v, "text_overlay", None) - if overlay is None: - return - if not self._overlay_configured: + overlay.font_size = 18 + overlay.position = "top_left" try: - overlay.font_size = 18 - overlay.position = "top_left" - try: - overlay.color = "yellow" - except Exception: - pass - self._overlay_configured = True + overlay.color = "yellow" except Exception: pass - try: - if overlay_text: - overlay.text = overlay_text - overlay.visible = True - else: - overlay.visible = False + self._overlay_configured = True except Exception: pass + try: + if in_flight > 0: + overlay.text = f"Loading tiles ... ({in_flight} in flight)" + overlay.visible = True + else: + overlay.visible = False + except Exception: + pass - QTimer.singleShot(0, _apply) + def push(self, message: str, *, in_flight: int = 0) -> None: + """Called from any thread; delivery hops to main via a signal.""" + if self._bridge is None: + return + try: + self._bridge.update.emit(message, int(in_flight)) + except Exception: + pass class _FetchCounter: diff --git a/src/ndi/pyramid/multiscale.py b/src/ndi/pyramid/multiscale.py index bddfb2e7..ffcde36c 100644 --- a/src/ndi/pyramid/multiscale.py +++ b/src/ndi/pyramid/multiscale.py @@ -1224,24 +1224,24 @@ def layerSpec( # only a graph rewrite that says "block[c] instead of block[all]". import dask.array as da - # Default is single-level (only the coarsest, as a plain non- - # multiscale layer). Tried flipping the default to multiscale - # in c129852 and it did not survive production use on the - # Maddie dataset: the first paint sat on "no tiles requested - # yet by napari" for ~2 minutes while vispy spammed destroyed- - # dispatcher warnings, and once the view did come up a plain - # zoom-in did NOT trigger napari's level swap -- the user had - # to zoom out and back in to get a finer level to render. That - # is unacceptable UX for the common case. Single-level takes - # napari's multiscale slicer out of the loop entirely: the - # layer draws promptly at the coarsest level, then a magicgui - # dock widget swaps layer.data between levels (manually, or - # driven by the zoom-level-swap listener on camera events). + # Default is multiscale: user A/B preferred this over single-level + # because the transition between levels is smoother -- single-level + # swaps layer.data under napari, which shows up as a short black + # frame on every level change, while a native multiscale layer + # just picks a finer level and refines. Both modes have been + # observed to hit the same napari slicer wedge on this dataset + # (vispy "QBasicTimer destroyed dispatcher" spam, no tile + # requests), so single-level is not actually safer -- it just + # looks janker when it does paint. # - # NDI_LIGHTSHEET_MULTISCALE=1 opts into the native multiscale - # path for anyone testing whether napari's slicer has been - # fixed on a given data shape. - single_level = not _env_true("NDI_LIGHTSHEET_MULTISCALE") + # NDI_LIGHTSHEET_SINGLE_LEVEL=1 opts back into the earlier + # single-level + Resolution-picker mode for anyone who wants + # the manual swap flow (or is debugging the multiscale slicer + # bug on a different dataset shape). + # NDI_LIGHTSHEET_ASYNC=0 is the escape hatch when the slicer + # wedges: it blocks add_image until the first slice is on + # screen, bypassing napari's async slicer entirely. + single_level = _env_true("NDI_LIGHTSHEET_SINGLE_LEVEL") # Build per-channel arrays once; each is a list of one 3D lazy # dask array per level. The level dropdown swaps between the From 2892e185fba625052c7f13208e3d7af9c2d36ddc Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 13:44:50 +0000 Subject: [PATCH 40/55] lightsheet viewer: camera-nudge workaround for the slicer wedge On macOS, napari+vispy sometimes leaves the slicer wedged: launch paints the coarsest level (served from the prefetch cache) and the fetch counter sits on "no tiles requested yet by napari" while vispy spams "QBasicTimer::start: current thread's event dispatcher has already been destroyed." A user zooming in has to zoom-out-and- back-in several times before napari actually asks the fetcher for finer tiles -- unacceptable UX for a viewer meant to beat Imaris on a local machine. Add _attach_camera_nudge: on every viewer.camera.events.zoom / center / angles event, start a 120 ms debounce timer that fires layer.refresh() on every added layer. layer.refresh() forces napari to compile a new slice request on the main thread, which does reliably fire the slicer -- so a single zoom gesture becomes enough to trigger a level swap instead of three. The 120 ms debounce coalesces a pinch/wheel-zoom's many micro- events into one refresh at rest, matching how the viewport-clip listener elsewhere in this viewer handles the same event storm. A module-level _NUDGE_TIMERS dict keeps the QTimer and its slot connection alive for the viewer's lifetime; without the strong ref Python GC would cut the timer loose the moment the function returned. NDI_LIGHTSHEET_NUDGE=0 is the escape hatch (prints one line and installs no listeners) in case the extra refresh storm interacts badly with another dataset. This is a shim, not a fix: the real bug is in vispy/napari's slicer dispatcher on this macOS stack. A proper fix needs an upstream napari issue (next on the list). Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/gui/app/lightsheetZarr/viewer.py | 106 +++++++++++++++++++++++ 1 file changed, 106 insertions(+) diff --git a/src/ndi/gui/app/lightsheetZarr/viewer.py b/src/ndi/gui/app/lightsheetZarr/viewer.py index 13502dd5..ed15d9d4 100644 --- a/src/ndi/gui/app/lightsheetZarr/viewer.py +++ b/src/ndi/gui/app/lightsheetZarr/viewer.py @@ -704,6 +704,16 @@ def openPyramid( # all and we lose only the extra line. _subscribe_level_change(viewer) + # Camera-nudge workaround for the vispy/napari slicer wedge + # observed on macOS: after launch, zoom events sometimes do not + # trigger a re-slice, so a user has to zoom-out-then-in several + # times before napari asks the fetcher for finer tiles. Call + # layer.refresh() ourselves on every camera event (debounced), + # which forces napari's slicer to compile a new slice request. + # Silent no-op if the camera doesn't expose the expected events + # (older napari) or if NDI_LIGHTSHEET_NUDGE=0 is set. + _attach_camera_nudge(viewer, added_layers) + # Launch window is closed BEFORE napari.run() because napari's # event loop blocks the main thread and our Qt window can't # repaint during it -- a frozen progress bar next to napari looks @@ -796,6 +806,102 @@ def _report_loaded(layer) -> None: ) +# Weak reference holder so the nudge QTimer doesn't get garbage +# collected the moment _attach_camera_nudge returns. Keyed by viewer +# id so a second viewer in the same process gets its own timer. The +# timer is Qt-owned once started, but Python GC can still cut it loose +# if nothing holds a reference. +_NUDGE_TIMERS: dict[int, Any] = {} + + +def _attach_camera_nudge(viewer, layers) -> None: + """Force a layer.refresh() on every camera event (debounced). + + Workaround for a vispy/napari slicer wedge observed on macOS: after + launch, zoom and pan events sometimes do not trigger a re-slice, so + napari sits on a stale frame while the user zooms. The symptom in + the fetch log is "no tiles requested yet by napari" that never + clears, with vispy spamming "QBasicTimer::start: current thread's + event dispatcher has already been destroyed" from the slicer. + Calling layer.refresh() ourselves forces napari to compile a new + slice request on the main thread, which does fire, so a single + zoom gesture becomes enough to kick a level swap. + + Debounced 120 ms so a pinch-zoom or wheel-zoom settling through + many micro-events refreshes once at rest rather than per event. + Silent no-op when the camera doesn't expose the expected events + (older napari) or when ``NDI_LIGHTSHEET_NUDGE=0`` turns it off. + """ + if not layers: + return + if os.environ.get("NDI_LIGHTSHEET_NUDGE", "1").strip().lower() in ("0", "false", "off"): + print( + "[lightsheet] camera-nudge: disabled via NDI_LIGHTSHEET_NUDGE", + file=sys.stderr, + flush=True, + ) + return + + try: + from qtpy.QtCore import QTimer + except ImportError: + return + + try: + timer = QTimer() + timer.setSingleShot(True) + timer.setInterval(120) + except Exception: + return + + def _do_refresh(): + for layer in layers: + try: + layer.refresh() + except Exception: + pass + + try: + timer.timeout.connect(_do_refresh) + except Exception: + return + + def _on_camera(_event): + try: + timer.start() + except Exception: + pass + + camera = getattr(viewer, "camera", None) + events = getattr(camera, "events", None) + if events is None: + return + hooked = [] + for name in ("zoom", "center", "angles"): + emitter = getattr(events, name, None) + if emitter is None: + continue + try: + emitter.connect(_on_camera) + hooked.append(name) + except Exception: + pass + if not hooked: + return + + # Keep the timer (and its slot connections) alive for the viewer's + # lifetime; a plain local goes out of scope at return and Qt + # cannot keep a dangling PyQt QTimer running without a reference. + _NUDGE_TIMERS[id(viewer)] = timer + + print( + f"[lightsheet] camera-nudge: on, hooked camera.events.{'/'.join(hooked)} " + "(set NDI_LIGHTSHEET_NUDGE=0 to disable)", + file=sys.stderr, + flush=True, + ) + + def _attach_level_selector(viewer, layers, per_layer_levels, per_layer_scales): """Dock a "Resolution" selector that swaps each layer's level. From 3733ac6a0c6845826f6d7bd1e1f282297a3a11cb Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 13:54:16 +0000 Subject: [PATCH 41/55] lightsheet viewer: add Refresh View button User-facing lever for the macOS napari slicer wedge. The camera-nudge alone does not reliably unstick napari on this stack, but the user found a manual action that does: switching the level dropdown away-and-back a few times. That is a visibility toggle under the covers -- the layer tears down its vispy node, napari re-initializes the slicer when it comes back, and the next slice compute fires reliably. Dock a Refresh View button that performs that same sequence: set layer.visible=False for every visible layer, let Qt process the hide-event (50 ms QTimer.singleShot), then set visible=True and call layer.refresh(). Users click it when regions still show coarse data after a zoom-in, instead of fighting the Resolution dropdown. NDI_LIGHTSHEET_REFRESH_BUTTON=0 hides the button for scripted viewers. Silent no-op if magicgui isn't installed. Timing of each refresh is logged to stderr so a user watching the terminal can see which clicks did work. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/gui/app/lightsheetZarr/viewer.py | 130 +++++++++++++++++++++++ 1 file changed, 130 insertions(+) diff --git a/src/ndi/gui/app/lightsheetZarr/viewer.py b/src/ndi/gui/app/lightsheetZarr/viewer.py index ed15d9d4..17597ed6 100644 --- a/src/ndi/gui/app/lightsheetZarr/viewer.py +++ b/src/ndi/gui/app/lightsheetZarr/viewer.py @@ -710,10 +710,21 @@ def openPyramid( # times before napari asks the fetcher for finer tiles. Call # layer.refresh() ourselves on every camera event (debounced), # which forces napari's slicer to compile a new slice request. + # Observed to help only partially on the Maddie dataset; the + # explicit Refresh View button below is the user-facing lever. # Silent no-op if the camera doesn't expose the expected events # (older napari) or if NDI_LIGHTSHEET_NUDGE=0 is set. _attach_camera_nudge(viewer, added_layers) + # Manual Refresh View button in a right-dock widget. Toggles + # layer.visible off/on and calls refresh() -- the sequence the + # user discovered unsticks napari's slicer when zoom alone does + # not. Users click this when regions still show coarse-level + # data after a zoom-in. Hidden-by-default via + # NDI_LIGHTSHEET_REFRESH_BUTTON=0 for a scripted viewer where + # the panel chrome is unwanted. + _attach_refresh_button(viewer, added_layers) + # Launch window is closed BEFORE napari.run() because napari's # event loop blocks the main thread and our Qt window can't # repaint during it -- a frozen progress bar next to napari looks @@ -902,6 +913,125 @@ def _on_camera(_event): ) +# Weak reference holder so the refresh-button widget doesn't get GC'd +# the moment _attach_refresh_button returns. Qt owns the dock widget +# once docked, but the magicgui wrapper is a Python object and will +# be collected without this. +_REFRESH_BUTTONS: dict[int, Any] = {} + + +def _attach_refresh_button(viewer, layers) -> None: + """Dock a Refresh View button on the right. + + Users click it when the view shows coarse-level data in regions + that should have refined after a zoom-in. The handler toggles + layer.visible off/on (deferred via QTimer so Qt processes the + off-event first) and calls layer.refresh() -- the sequence a user + discovered unsticks napari's slicer on the Maddie dataset where + camera events alone do not. + + Hidden via NDI_LIGHTSHEET_REFRESH_BUTTON=0 for scripted / headless + viewers. Falls back to a silent no-op if magicgui isn't installed. + """ + if not layers: + return + if os.environ.get("NDI_LIGHTSHEET_REFRESH_BUTTON", "1").strip().lower() in ( + "0", + "false", + "off", + ): + return + + try: + from magicgui.widgets import Container, PushButton + except ImportError: + print( + "[lightsheet] refresh button: magicgui not installed; skipping " + "(pip install magicgui)", + file=sys.stderr, + flush=True, + ) + return + + try: + from qtpy.QtCore import QTimer + except ImportError: + QTimer = None # type: ignore[assignment] + + def _do_refresh(): + import time as _time + + t0 = _time.monotonic() + # Phase 1: hide every layer, call refresh. Hiding tears down + # the vispy node for that layer; a repaint without it does the + # cleanup napari would normally do at slicer re-init time. + hidden: list = [] + for layer in layers: + try: + if getattr(layer, "visible", False): + layer.visible = False + hidden.append(layer) + except Exception: + pass + + # Phase 2 is scheduled 50 ms later so Qt processes the + # hide-event first; otherwise visible=True immediately after + # visible=False collapses to a no-op in napari's internals. + def _unhide_and_refresh(): + for layer in hidden: + try: + layer.visible = True + except Exception: + pass + for layer in layers: + try: + layer.refresh() + except Exception: + pass + dt = _time.monotonic() - t0 + print( + f"[lightsheet] refresh view: toggled {len(hidden)} layer(s), " + f"took {dt * 1000:.0f} ms", + file=sys.stderr, + flush=True, + ) + + if QTimer is not None: + QTimer.singleShot(50, _unhide_and_refresh) + else: + _unhide_and_refresh() + + try: + button = PushButton(text="Refresh view") + button.clicked.connect(_do_refresh) + container = Container(widgets=[button], labels=False) + except Exception as exc: + print( + f"[lightsheet] refresh button: construction failed ({exc})", + file=sys.stderr, + flush=True, + ) + return + + try: + viewer.window.add_dock_widget(container, area="right", name="Refresh") + except Exception as exc: + print( + f"[lightsheet] refresh button: could not dock ({exc})", + file=sys.stderr, + flush=True, + ) + return + + _REFRESH_BUTTONS[id(viewer)] = container + print( + "[lightsheet] refresh button: docked (click when regions still show " + "coarse data after a zoom)", + file=sys.stderr, + flush=True, + ) + + def _attach_level_selector(viewer, layers, per_layer_levels, per_layer_scales): """Dock a "Resolution" selector that swaps each layer's level. From f9ac7bb5d0e50cb3683539ce8a40cbcf30324bdf Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 14:09:55 +0000 Subject: [PATCH 42/55] lightsheet viewer: NDI Cloud sign-in panel + BETA badge Mirror the gene-pyramid viewer's cloud panel so a reader of a cloud-backed lightsheet pyramid can see token time-left and renew without quitting -- the viewer outlives its token on anything but a few-minute browse, and up to now a reader on an expired token would see "missing tiles" rather than a login prompt. Build the panel locally rather than calling addCloudPanel from the gene-pyramid module directly, because the lightsheet viewer needs a small BETA badge immediately under the wordmark (the first-paint / zoom-refinement rough edges called out in the README should not be a surprise). Reuses the gene module's cloudSignInDialog, cloudSessionLooksLikely, and cloudLogoLabel so profile list, token clock, and auth plumbing are identical across the two viewers. Only shown when the environment carries NDI_CLOUD_TOKEN, same rule the gene viewer uses -- a local-only pyramid has no reason for a login control, and a login panel on a local window implies the picture is waiting on something it isn't. NDI_LIGHTSHEET_CLOUD_PANEL=0 opts it out for scripted viewers. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/gui/app/lightsheetZarr/viewer.py | 167 +++++++++++++++++++++++ 1 file changed, 167 insertions(+) diff --git a/src/ndi/gui/app/lightsheetZarr/viewer.py b/src/ndi/gui/app/lightsheetZarr/viewer.py index 17597ed6..e25085fb 100644 --- a/src/ndi/gui/app/lightsheetZarr/viewer.py +++ b/src/ndi/gui/app/lightsheetZarr/viewer.py @@ -725,6 +725,14 @@ def openPyramid( # the panel chrome is unwanted. _attach_refresh_button(viewer, added_layers) + # NDI Cloud sign-in panel, same shape as the gene-pyramid + # viewer uses. Shown only when a cloud token is in the + # environment -- a purely local pyramid has no reason for a + # login control. The lightsheet viewer carries a "beta" badge + # under the wordmark because this surface is still rough on + # the dataset-shape edges called out in the README. + _attach_cloud_panel(viewer) + # Launch window is closed BEFORE napari.run() because napari's # event loop blocks the main thread and our Qt window can't # repaint during it -- a frozen progress bar next to napari looks @@ -1032,6 +1040,165 @@ def _unhide_and_refresh(): ) +# Weak reference holder for the cloud panel widget. Same reason as +# the refresh button: Qt owns the dock once added, but the Python +# wrapper and its slot connections need a strong ref. +_CLOUD_PANELS: dict[int, Any] = {} + + +def _attach_cloud_panel(viewer) -> None: + """Dock an NDI Cloud sign-in panel with a BETA badge. + + Reuses the gene-pyramid viewer's addCloudPanel contract -- same + shape, same auth plumbing, so a user who has signed in to that + viewer sees the same profile list and the same token clock + here. We don't call addCloudPanel directly because we want a + small "BETA" badge immediately under the wordmark, which the + gene-pyramid viewer does not carry; building the panel locally + is less fragile than monkey-patching after dock. + + Silent no-op when the environment has no cloud token (same + cloudSessionLooksLikely rule the gene viewer uses) or when + NDI_LIGHTSHEET_CLOUD_PANEL=0 opts it out. A purely local + pyramid has no reason for a login control; a login panel on a + local-only window implies the picture might be waiting on + something, which it isn't. + """ + if os.environ.get("NDI_LIGHTSHEET_CLOUD_PANEL", "1").strip().lower() in ("0", "false", "off"): + return + + try: + from ndi.gui.app.genepyramid.controls import ( + cloudLogoLabel, + cloudSessionLooksLikely, + cloudSignInDialog, + ) + except ImportError as exc: + print( + f"[lightsheet] cloud panel: gene-pyramid controls unavailable ({exc})", + file=sys.stderr, + flush=True, + ) + return + + if not cloudSessionLooksLikely(): + return + + try: + from qtpy.QtCore import Qt, QTimer + from qtpy.QtWidgets import QHBoxLayout, QLabel, QPushButton, QVBoxLayout, QWidget + except ImportError: + return + + try: + from ndi.cloud import auth + except ImportError as exc: + print( + f"[lightsheet] cloud panel: ndi.cloud.auth unavailable ({exc})", + file=sys.stderr, + flush=True, + ) + return + + box = QWidget() + outer = QVBoxLayout(box) + + logo = cloudLogoLabel(box) + if logo is not None: + outer.addWidget(logo) + + beta = QLabel("BETA") + beta.setAlignment(Qt.AlignLeft | Qt.AlignVCenter) + beta.setStyleSheet( + "QLabel {" + " color: #ffffff;" + " background: #c9372c;" + " font-weight: 700;" + " font-size: 10px;" + " letter-spacing: 2px;" + " padding: 2px 8px;" + " border-radius: 3px;" + "}" + ) + beta.setToolTip( + "The lightsheet viewer is in beta. First-paint and\n" + "zoom-refinement can be janky on large pyramids;\n" + "use the Refresh View button if a region stays coarse." + ) + beta_row = QHBoxLayout() + beta_row.setContentsMargins(0, 2, 0, 6) + beta_row.addWidget(beta, 0, Qt.AlignLeft) + beta_row.addStretch(1) + outer.addLayout(beta_row) + + signin = QPushButton("Sign in...") + signin.setToolTip( + "Sign in to NDI Cloud, so tiles that are not already\n" + "downloaded keep loading. The token lives in this\n" + "process, so signing in anywhere else does not reach\n" + "this window." + ) + clock = QLabel("") + clock.setWordWrap(True) + row = QHBoxLayout() + row.addWidget(signin) + row.addWidget(clock, 1) + outer.addLayout(row) + + result = QLabel("") + result.setWordWrap(True) + result.hide() + outer.addWidget(result) + outer.addStretch() + + def _tick(): + left = auth.tokenSecondsRemaining() + line = auth.tokenStatusLine() + if left is not None and 0 < left < 900: + clock.setText(f"{line} -- renew before it runs out.") + elif left is not None and left <= 0: + clock.setText(f"{line}. Undownloaded tiles will fail until you sign in.") + else: + clock.setText(line) + + def _open(): + ok, note = cloudSignInDialog(box) + if note: + result.setText(note) + result.show() + elif ok: + result.hide() + _tick() + + signin.clicked.connect(_open) + + # Parented to the widget so the timer stops when the dock + # goes. A free timer would fire at a deleted label and take + # the process with it. + timer = QTimer(box) + timer.setInterval(30_000) + timer.timeout.connect(_tick) + timer.start() + _tick() + + try: + viewer.window.add_dock_widget(box, area="right", name="NDI Cloud") + except Exception as exc: + print( + f"[lightsheet] cloud panel: could not dock ({exc})", + file=sys.stderr, + flush=True, + ) + return + + _CLOUD_PANELS[id(viewer)] = box + print( + "[lightsheet] cloud panel: docked (NDI Cloud sign-in + beta badge)", + file=sys.stderr, + flush=True, + ) + + def _attach_level_selector(viewer, layers, per_layer_levels, per_layer_scales): """Dock a "Resolution" selector that swaps each layer's level. From 17a398a5a05af669aa5a898b93615b08acc86e46 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 14:28:48 +0000 Subject: [PATCH 43/55] napari slicer-wedge bisect: baseline (v1) + background threads (v2) v1 is the control: lazy multiscale dask pyramid with the same shape ratios as production. Confirmed to RUN CORRECTLY on vanhoosr's machine (~1,600 block reads, no vispy QBasicTimer warnings), so the slicer wedge is caused by something our launcher does on top of this baseline, not by napari/vispy on this data shape. v2 adds the first candidate: a daemon heartbeat threading.Thread plus a ThreadPoolExecutor with a few submitted jobs started BEFORE napari.Viewer(). This mirrors _FetchCounter._heartbeat + fetcher.warm() + loader.startPrefetch() in src/ndi/gui/app/lightsheetZarr/viewer.py. If just spawning Python threads before Viewer() construction wedges the slicer, this file reproduces it; if not, v3 will add the next piece (refresh-hint registration / _NapariStatusReporter / etc.). Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- napari_slicer_repro_v1.py | 233 +++++++++++++++++++++++++++++++++ napari_slicer_repro_v2.py | 268 ++++++++++++++++++++++++++++++++++++++ 2 files changed, 501 insertions(+) create mode 100644 napari_slicer_repro_v1.py create mode 100644 napari_slicer_repro_v2.py diff --git a/napari_slicer_repro_v1.py b/napari_slicer_repro_v1.py new file mode 100644 index 00000000..1184b308 --- /dev/null +++ b/napari_slicer_repro_v1.py @@ -0,0 +1,233 @@ +"""napari slicer-wedge diagnostic -- v1 (BASELINE: works). + +Minimal reproducer skeleton for the slicer wedge seen in the +lightsheet viewer on macOS + napari 0.9.1 + PyQt6. This file is +the control: a lazy dask MultiScaleData pyramid, two channels, +same shape ratios as the production dataset. On the user's own +machine (vanhoosr, 2026-10-04) this version RUNS CORRECTLY -- +~1,600 block reads, no vispy "QBasicTimer::start: current +thread's event dispatcher has already been destroyed" warnings, +zoom and pan work. + +That means the wedge is caused by something our launcher does +on top of this baseline, NOT by napari/vispy on this shape of +data. Subsequent versions (_v2, _v3, ...) each add one more +piece of the production launcher's startup until the wedge +reproduces. The one that triggers it is the fix target. + +Usage: + python napari_slicer_repro_v1.py + # Zoom in. "block read" lines in the terminal tell you + # whether napari ever actually asks for a chunk. Press + # Shift+V to apply the visibility-toggle workaround. +""" + +# Diagnostic output BEFORE any heavy imports: if nothing below +# prints, Python isn't actually running this file at all. +import sys + +print("[repro] script started", flush=True) +print(f"[repro] python = {sys.version.splitlines()[0]}", flush=True) +print(f"[repro] argv = {sys.argv}", flush=True) + +# Guarded imports -- surface any missing dependency as a message, +# not a traceback that might scroll off. +try: + import os + import time + + import numpy as np + + print("[repro] stdlib + numpy ok", flush=True) + import dask + import dask.array as da + + print(f"[repro] dask = {dask.__version__}", flush=True) + import napari + + print(f"[repro] napari = {napari.__version__}", flush=True) + import vispy + + print(f"[repro] vispy = {vispy.__version__}", flush=True) + import qtpy + + print(f"[repro] qtpy = {qtpy.API_NAME}", flush=True) +except Exception as exc: + print(f"[repro] IMPORT FAILED: {type(exc).__name__}: {exc}", flush=True) + sys.exit(1) + + +# --- fake lightsheet pyramid -------------------------------------- +# 4 levels, dyadic, matching the production dataset's shape ratios. +LEVELS = [ + (3844, 8751, 7390), + (1922, 4376, 3695), + (961, 2188, 1848), + (481, 1094, 924), +] +N_CHANNELS = 2 +DTYPE = np.uint16 +CHUNK_SHAPE = { + 0: (16, 128, 128), + 1: (16, 128, 128), + 2: (16, 128, 128), + 3: (16, 128, 128), +} + + +_READ_COUNT = {"n": 0} + + +def _read_block(level, zi, yi, xi, shape): + """Fake chunk reader. Prints every time napari asks for a chunk. + + If this NEVER prints after the viewer window is up, the slicer + is wedged -- which is the bug this script demonstrates. + """ + _READ_COUNT["n"] += 1 + n = _READ_COUNT["n"] + print( + f"[repro] block read #{n}: level={level} indices=({zi},{yi},{xi}) shape={shape}", + flush=True, + ) + z, y, x = np.indices(shape) + arr = ((z + zi * shape[0]) + (y + yi * shape[1]) + (x + xi * shape[2])) * (level + 1) + return (arr % 4096).astype(DTYPE) + + +def _build_level(level, shape_zyx): + """Build one level as a lazy dask.array of per-chunk delayed reads.""" + cz, cy, cx = CHUNK_SHAPE[level] + sz, sy, sx = shape_zyx + nz = -(-sz // cz) + ny = -(-sy // cy) + nx = -(-sx // cx) + blocks = np.empty((nz, ny, nx), dtype=object) + for zi in range(nz): + for yi in range(ny): + for xi in range(nx): + this_shape = ( + min(cz, sz - zi * cz), + min(cy, sy - yi * cy), + min(cx, sx - xi * cx), + ) + blk = dask.delayed(_read_block)(level, zi, yi, xi, this_shape) + blocks[zi, yi, xi] = da.from_delayed(blk, shape=this_shape, dtype=DTYPE) + return da.block(blocks.tolist()) + + +def build_pyramid(): + print("[repro] building lazy dask pyramid ...", flush=True) + t0 = time.monotonic() + per_channel = [] + for _c in range(N_CHANNELS): + levels_for_this_channel = [] + for lvl_idx, zyx in enumerate(LEVELS): + levels_for_this_channel.append(_build_level(lvl_idx, zyx)) + per_channel.append(levels_for_this_channel) + print( + f"[repro] built pyramid in {time.monotonic() - t0:.2f}s " + f"(no chunk read fired during build)", + flush=True, + ) + return per_channel + + +def _attach_unstick_hotkey(viewer, layers): + """Shift+V applies the visibility-toggle workaround.""" + try: + from qtpy.QtCore import QTimer + except ImportError: + return + + @viewer.bind_key("Shift-V", overwrite=True) + def _unstick(_viewer): + print("[repro] applying unstick workaround ...", flush=True) + hidden = [] + for layer in layers: + try: + if layer.visible: + layer.visible = False + hidden.append(layer) + except Exception: + pass + + def _finish(): + for layer in hidden: + try: + layer.visible = True + except Exception: + pass + for layer in layers: + try: + layer.refresh() + except Exception: + pass + print( + f"[repro] unstick: toggled {len(hidden)} layer(s). " + "If the slicer was wedged, you should now see " + "'block read' lines when you zoom.", + flush=True, + ) + + QTimer.singleShot(50, _finish) + + +def main(): + print( + "[repro] env: NAPARI_ASYNC=" + os.environ.get("NAPARI_ASYNC", ""), + flush=True, + ) + + per_channel = build_pyramid() + + print("[repro] creating napari.Viewer() ...", flush=True) + viewer = napari.Viewer() + print("[repro] Viewer created; adding images ...", flush=True) + layers = [] + for c, levels in enumerate(per_channel): + layer = viewer.add_image( + levels, + multiscale=True, + name=f"Ch{c + 1}", + colormap="green" if c == 0 else "magenta", + blending="additive", + contrast_limits=(0, 4096), + ) + layers.append(layer) + try: + viewer.dims.set_point(0, LEVELS[0][0] // 2) + except Exception: + pass + print( + "[repro] after add_image: " + + ", ".join( + f"{lyr.name}: loaded={lyr.loaded} multiscale={lyr.multiscale}" for lyr in layers + ), + flush=True, + ) + print( + "[repro] zoom/pan now. 'block read' lines => slicer working. " + "No 'block read' + vispy QBasicTimer spam => slicer wedged. " + "Press Shift+V in the viewer window to apply the workaround.", + flush=True, + ) + _attach_unstick_hotkey(viewer, layers) + + print("[repro] entering napari.run() -- window should open now.", flush=True) + napari.run() + + print(f"[repro] session end: total chunk reads = {_READ_COUNT['n']}", flush=True) + + +if __name__ == "__main__": + try: + main() + except SystemExit: + raise + except Exception as exc: + import traceback + + print(f"[repro] UNCAUGHT: {type(exc).__name__}: {exc}", flush=True) + traceback.print_exc() + sys.exit(2) diff --git a/napari_slicer_repro_v2.py b/napari_slicer_repro_v2.py new file mode 100644 index 00000000..9a89d32b --- /dev/null +++ b/napari_slicer_repro_v2.py @@ -0,0 +1,268 @@ +"""napari slicer-wedge diagnostic -- v2 (adds background threads). + +v2 = v1 + background work started BEFORE napari.Viewer(): + * one daemon threading.Thread that heartbeats on a wait() + loop (mimics _FetchCounter._heartbeat in the production + launcher). + * a ThreadPoolExecutor with a handful of submitted jobs that + run a trivial "pretend to warm a worker pool" (mimics + fetcher.warm() and loader.startPrefetch()). + +Everything is pure Python; no Qt, no I/O, no dask execution. +If just spawning threads before Viewer() construction wedges +the slicer on macOS + PyQt6, this file reproduces that; if it +doesn't, the next version adds more of the launcher's startup +and we move on. + +Usage: + python napari_slicer_repro_v2.py + # Zoom in. "block read" lines in the terminal tell you + # whether napari ever actually asks for a chunk. Press + # Shift+V to apply the visibility-toggle workaround. +""" + +import sys + +print("[repro-v2] script started", flush=True) +print(f"[repro-v2] python = {sys.version.splitlines()[0]}", flush=True) +print(f"[repro-v2] argv = {sys.argv}", flush=True) + +try: + import os + import threading + import time + from concurrent.futures import ThreadPoolExecutor + + import numpy as np + + print("[repro-v2] stdlib + numpy ok", flush=True) + import dask + import dask.array as da + + print(f"[repro-v2] dask = {dask.__version__}", flush=True) + import napari + + print(f"[repro-v2] napari = {napari.__version__}", flush=True) + import vispy + + print(f"[repro-v2] vispy = {vispy.__version__}", flush=True) + import qtpy + + print(f"[repro-v2] qtpy = {qtpy.API_NAME}", flush=True) +except Exception as exc: + print(f"[repro-v2] IMPORT FAILED: {type(exc).__name__}: {exc}", flush=True) + sys.exit(1) + + +LEVELS = [ + (3844, 8751, 7390), + (1922, 4376, 3695), + (961, 2188, 1848), + (481, 1094, 924), +] +N_CHANNELS = 2 +DTYPE = np.uint16 +CHUNK_SHAPE = { + 0: (16, 128, 128), + 1: (16, 128, 128), + 2: (16, 128, 128), + 3: (16, 128, 128), +} + + +_READ_COUNT = {"n": 0} + + +def _read_block(level, zi, yi, xi, shape): + _READ_COUNT["n"] += 1 + n = _READ_COUNT["n"] + print( + f"[repro-v2] block read #{n}: level={level} indices=({zi},{yi},{xi}) shape={shape}", + flush=True, + ) + z, y, x = np.indices(shape) + arr = ((z + zi * shape[0]) + (y + yi * shape[1]) + (x + xi * shape[2])) * (level + 1) + return (arr % 4096).astype(DTYPE) + + +def _build_level(level, shape_zyx): + cz, cy, cx = CHUNK_SHAPE[level] + sz, sy, sx = shape_zyx + nz = -(-sz // cz) + ny = -(-sy // cy) + nx = -(-sx // cx) + blocks = np.empty((nz, ny, nx), dtype=object) + for zi in range(nz): + for yi in range(ny): + for xi in range(nx): + this_shape = ( + min(cz, sz - zi * cz), + min(cy, sy - yi * cy), + min(cx, sx - xi * cx), + ) + blk = dask.delayed(_read_block)(level, zi, yi, xi, this_shape) + blocks[zi, yi, xi] = da.from_delayed(blk, shape=this_shape, dtype=DTYPE) + return da.block(blocks.tolist()) + + +def build_pyramid(): + print("[repro-v2] building lazy dask pyramid ...", flush=True) + t0 = time.monotonic() + per_channel = [] + for _c in range(N_CHANNELS): + levels_for_this_channel = [] + for lvl_idx, zyx in enumerate(LEVELS): + levels_for_this_channel.append(_build_level(lvl_idx, zyx)) + per_channel.append(levels_for_this_channel) + print( + f"[repro-v2] built pyramid in {time.monotonic() - t0:.2f}s " + f"(no chunk read fired during build)", + flush=True, + ) + return per_channel + + +# --- the thing v2 adds on top of v1 ------------------------------- +# A daemon heartbeat thread + a worker pool with a few submitted +# jobs, all started BEFORE napari.Viewer() exists. Same shape as +# the launcher's _FetchCounter._heartbeat + fetcher.warm() + +# loader.startPrefetch() sequence. + +_STOP = threading.Event() + + +def _heartbeat_loop(): + """Wake every 5 s, print, loop. Just like the launcher's.""" + while not _STOP.wait(5.0): + print("[repro-v2] (heartbeat tick)", flush=True) + + +def _warm_worker(i): + """Pretend to warm a worker. Short CPU-free wait.""" + time.sleep(0.05) + return i + + +def start_background_threads(): + print("[repro-v2] starting daemon heartbeat thread ...", flush=True) + hb = threading.Thread( + target=_heartbeat_loop, + name="repro-v2-heartbeat", + daemon=True, + ) + hb.start() + + print("[repro-v2] starting ThreadPoolExecutor (4 workers) ...", flush=True) + pool = ThreadPoolExecutor(max_workers=4, thread_name_prefix="repro-v2-pool") + # Submit a handful of trivial jobs. We don't wait for them; the + # launcher doesn't either -- startPrefetch returns immediately. + for i in range(8): + pool.submit(_warm_worker, i) + print("[repro-v2] background threads up (heartbeat + 4-worker pool).", flush=True) + return hb, pool + + +def _attach_unstick_hotkey(viewer, layers): + try: + from qtpy.QtCore import QTimer + except ImportError: + return + + @viewer.bind_key("Shift-V", overwrite=True) + def _unstick(_viewer): + print("[repro-v2] applying unstick workaround ...", flush=True) + hidden = [] + for layer in layers: + try: + if layer.visible: + layer.visible = False + hidden.append(layer) + except Exception: + pass + + def _finish(): + for layer in hidden: + try: + layer.visible = True + except Exception: + pass + for layer in layers: + try: + layer.refresh() + except Exception: + pass + print( + f"[repro-v2] unstick: toggled {len(hidden)} layer(s). " + "If the slicer was wedged, you should now see " + "'block read' lines when you zoom.", + flush=True, + ) + + QTimer.singleShot(50, _finish) + + +def main(): + print( + "[repro-v2] env: NAPARI_ASYNC=" + os.environ.get("NAPARI_ASYNC", ""), + flush=True, + ) + + per_channel = build_pyramid() + + # <<< v2 addition >>> background threads come up before Viewer(). + hb, pool = start_background_threads() + + print("[repro-v2] creating napari.Viewer() ...", flush=True) + viewer = napari.Viewer() + print("[repro-v2] Viewer created; adding images ...", flush=True) + layers = [] + for c, levels in enumerate(per_channel): + layer = viewer.add_image( + levels, + multiscale=True, + name=f"Ch{c + 1}", + colormap="green" if c == 0 else "magenta", + blending="additive", + contrast_limits=(0, 4096), + ) + layers.append(layer) + try: + viewer.dims.set_point(0, LEVELS[0][0] // 2) + except Exception: + pass + print( + "[repro-v2] after add_image: " + + ", ".join( + f"{lyr.name}: loaded={lyr.loaded} multiscale={lyr.multiscale}" for lyr in layers + ), + flush=True, + ) + print( + "[repro-v2] zoom/pan now. 'block read' lines => slicer working. " + "No 'block read' + vispy QBasicTimer spam => slicer wedged. " + "Press Shift+V in the viewer window to apply the workaround.", + flush=True, + ) + _attach_unstick_hotkey(viewer, layers) + + print("[repro-v2] entering napari.run() -- window should open now.", flush=True) + try: + napari.run() + finally: + _STOP.set() + pool.shutdown(wait=False) + + print(f"[repro-v2] session end: total chunk reads = {_READ_COUNT['n']}", flush=True) + + +if __name__ == "__main__": + try: + main() + except SystemExit: + raise + except Exception as exc: + import traceback + + print(f"[repro-v2] UNCAUGHT: {type(exc).__name__}: {exc}", flush=True) + traceback.print_exc() + sys.exit(2) From 6d072263b8371fe14db13f3e81b104d720af2e62 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 14:31:48 +0000 Subject: [PATCH 44/55] lightsheet viewer: real zoom nudge in Refresh View; always show NDI Cloud panel The visibility-toggle in the Refresh View button turned out to be insufficient on the production dataset: layer.visible off/on alone does not fire a new slice compute, so a region stuck at the coarse level stayed stuck. The user's manual workaround -- zoom out, then zoom back in with the mouse -- always works, because that is a real camera event. Change the button to do exactly that: * phase 1: viewer.camera.zoom *= 0.8 (immediate) * phase 2 (100 ms): viewer.camera.zoom = original_zoom * phase 3 (150 ms): visibility off/on + layer.refresh() as a belt-and-suspenders fallback in case napari coalesces the two camera events out. The flash on screen is ~150 ms and barely perceptible; what the user sees is the view jumping back to a real level 0 render. Separately, show the NDI Cloud panel (wordmark + BETA badge + Sign in) on every lightsheet session, not only sessions that already have a cloud token. The wordmark identifies the viewer and the user asked for it to be visible always; on a local-only session the clock copy reads "Not signed in. Sign in to access NDI Cloud pyramids." instead of a stale token clock. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/gui/app/lightsheetZarr/viewer.py | 115 ++++++++++++++++------- 1 file changed, 81 insertions(+), 34 deletions(-) diff --git a/src/ndi/gui/app/lightsheetZarr/viewer.py b/src/ndi/gui/app/lightsheetZarr/viewer.py index e25085fb..b07e128b 100644 --- a/src/ndi/gui/app/lightsheetZarr/viewer.py +++ b/src/ndi/gui/app/lightsheetZarr/viewer.py @@ -932,11 +932,13 @@ def _attach_refresh_button(viewer, layers) -> None: """Dock a Refresh View button on the right. Users click it when the view shows coarse-level data in regions - that should have refined after a zoom-in. The handler toggles - layer.visible off/on (deferred via QTimer so Qt processes the - off-event first) and calls layer.refresh() -- the sequence a user - discovered unsticks napari's slicer on the Maddie dataset where - camera events alone do not. + that should have refined after a zoom-in. The handler zooms the + camera out by 20% and then back in 100 ms later -- the sequence + a user discovered unsticks napari's slicer on the Maddie dataset + (toggling layer.visible or calling layer.refresh() alone does + not; only a real camera event does). A visibility toggle still + runs as a secondary nudge in case the camera change is coalesced + out. Hidden via NDI_LIGHTSHEET_REFRESH_BUTTON=0 for scripted / headless viewers. Falls back to a silent no-op if magicgui isn't installed. @@ -970,44 +972,72 @@ def _do_refresh(): import time as _time t0 = _time.monotonic() - # Phase 1: hide every layer, call refresh. Hiding tears down - # the vispy node for that layer; a repaint without it does the - # cleanup napari would normally do at slicer re-init time. - hidden: list = [] - for layer in layers: - try: - if getattr(layer, "visible", False): - layer.visible = False - hidden.append(layer) - except Exception: - pass - # Phase 2 is scheduled 50 ms later so Qt processes the - # hide-event first; otherwise visible=True immediately after - # visible=False collapses to a no-op in napari's internals. - def _unhide_and_refresh(): - for layer in hidden: + # Phase 1: zoom out by 20%. This is the piece that actually + # unsticks the slicer -- a real camera event forces napari + # to re-slice. The 20% factor is big enough that napari + # doesn't coalesce it with the "set it back" that follows, + # and small enough that it looks like a brief flash on screen + # rather than a jarring zoom out. + orig_zoom = None + try: + orig_zoom = float(viewer.camera.zoom) + viewer.camera.zoom = orig_zoom * 0.8 + except Exception as exc: # noqa: BLE001 + print( + f"[lightsheet] refresh view: zoom-out failed ({exc})", + file=sys.stderr, + flush=True, + ) + + # Phase 2 (100 ms later): zoom back in, and -- as a belt-and- + # suspenders fallback -- toggle layer.visible off then on. + # The hide/show pair tears down and rebuilds each layer's + # vispy node, which is what the manual workaround users + # discovered first did. + def _restore_and_toggle(): + if orig_zoom is not None: try: - layer.visible = True + viewer.camera.zoom = orig_zoom except Exception: pass + hidden: list = [] for layer in layers: try: - layer.refresh() + if getattr(layer, "visible", False): + layer.visible = False + hidden.append(layer) except Exception: pass - dt = _time.monotonic() - t0 - print( - f"[lightsheet] refresh view: toggled {len(hidden)} layer(s), " - f"took {dt * 1000:.0f} ms", - file=sys.stderr, - flush=True, - ) + + def _unhide(): + for layer in hidden: + try: + layer.visible = True + except Exception: + pass + for layer in layers: + try: + layer.refresh() + except Exception: + pass + dt = _time.monotonic() - t0 + print( + f"[lightsheet] refresh view: zoom-nudge + toggled " + f"{len(hidden)} layer(s), took {dt * 1000:.0f} ms", + file=sys.stderr, + flush=True, + ) + + if QTimer is not None: + QTimer.singleShot(50, _unhide) + else: + _unhide() if QTimer is not None: - QTimer.singleShot(50, _unhide_and_refresh) + QTimer.singleShot(100, _restore_and_toggle) else: - _unhide_and_refresh() + _restore_and_toggle() try: button = PushButton(text="Refresh view") @@ -1081,8 +1111,18 @@ def _attach_cloud_panel(viewer) -> None: ) return - if not cloudSessionLooksLikely(): - return + # Always show the NDI Cloud panel, even on a local-only dataset: + # the wordmark and BETA badge identify what this viewer is, and + # a user who later wants cloud access needs an obvious place to + # sign in. cloudSessionLooksLikely is still consulted to tune + # the clock copy (an unsigned/local session reads "not signed + # in" rather than a token-expiry clock), but it no longer gates + # the panel. + has_cloud = False + try: + has_cloud = bool(cloudSessionLooksLikely()) + except Exception: + pass try: from qtpy.QtCore import Qt, QTimer @@ -1152,6 +1192,13 @@ def _attach_cloud_panel(viewer) -> None: outer.addStretch() def _tick(): + # Local-only session: show "not signed in" rather than a stale + # token clock. tokenStatusLine()/tokenSecondsRemaining() are + # safe to call but their output can be misleading when no cloud + # pyramid is loaded, so we prefer an explicit copy. + if not has_cloud: + clock.setText("Not signed in. Sign in to access NDI Cloud pyramids.") + return left = auth.tokenSecondsRemaining() line = auth.tokenStatusLine() if left is not None and 0 < left < 900: From 7488367bf2b7569bd4081578bf7fc183dc81a352 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 14:40:30 +0000 Subject: [PATCH 45/55] napari slicer-wedge bisect v3: cross-thread Qt bridge + leaner pyramid v3 adds the next launcher candidate after v2: a QObject that lives on the Qt main thread (via moveToThread(app.thread())) and exposes a Signal that a background thread emits through Qt.QueuedConnection. This mirrors _NapariStatusReporter._Bridge in the production launcher and is the most Qt-adjacent thing we do before napari.Viewer(). Also shrinks the fake pyramid 8x per axis (chunk count drops ~512x). The reporter in the earlier wait reported ~5-minute launch waits for v1 and v2; nearly all of that was dask.from_delayed construction and napari's walk of the resulting graphs, O(1M) delayed objects per level before. v3's launch should be seconds, with each move taking seconds rather than minutes. The slicer-wedge bug does not depend on absolute pyramid size -- shape ratio and lazy multiscale do -- so the shrink does not change what we are testing. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- napari_slicer_repro_v3.py | 349 ++++++++++++++++++++++++++++++++++++++ 1 file changed, 349 insertions(+) create mode 100644 napari_slicer_repro_v3.py diff --git a/napari_slicer_repro_v3.py b/napari_slicer_repro_v3.py new file mode 100644 index 00000000..a98fa11d --- /dev/null +++ b/napari_slicer_repro_v3.py @@ -0,0 +1,349 @@ +"""napari slicer-wedge diagnostic -- v3. + +v3 = v2 + a cross-thread Qt bridge (the thing _NapariStatusReporter +does in the production launcher), PLUS a smaller fake pyramid so +each movement doesn't take a minute of synthetic compute. + +The bridge: a QObject that lives on the Qt main thread +(moveToThread(app.thread())) and exposes a Signal that a background +thread emits via Qt.ConnectionType.QueuedConnection. A background +thread then calls bridge.push("...") every second or so, which lands +on the main thread as a status-bar update. This is what the launcher +does to funnel the fetch-counter thread's "N/M tiles" line into +viewer.status without touching Qt from the wrong thread. + +What changed vs v2: + * fake pyramid shrunk 8x per axis -> every level's chunk count is + O(100-1000) rather than O(1M). Each move now takes seconds, not + minutes. + * a daemon thread emits status updates through a cross-thread Qt + signal every 1.0 s, starting before napari.Viewer() exists. + * everything from v2 (heartbeat + ThreadPoolExecutor) still runs. + +If v3 wedges and v2 did not, the cross-thread Qt bridge is the fix +target. If v3 does not wedge, v4 will add the next piece +(refresh-hint registration via viewer.camera.events). + +Usage: + python napari_slicer_repro_v3.py + # Zoom in. "block read" lines in the terminal tell you whether + # napari ever actually asks for a chunk. Press Shift+V to apply + # the visibility-toggle workaround. +""" + +import sys + +print("[repro-v3] script started", flush=True) +print(f"[repro-v3] python = {sys.version.splitlines()[0]}", flush=True) +print(f"[repro-v3] argv = {sys.argv}", flush=True) + +try: + import os + import threading + import time + from concurrent.futures import ThreadPoolExecutor + + import numpy as np + + print("[repro-v3] stdlib + numpy ok", flush=True) + import dask + import dask.array as da + + print(f"[repro-v3] dask = {dask.__version__}", flush=True) + import napari + + print(f"[repro-v3] napari = {napari.__version__}", flush=True) + import vispy + + print(f"[repro-v3] vispy = {vispy.__version__}", flush=True) + import qtpy + + print(f"[repro-v3] qtpy = {qtpy.API_NAME}", flush=True) +except Exception as exc: + print(f"[repro-v3] IMPORT FAILED: {type(exc).__name__}: {exc}", flush=True) + sys.exit(1) + + +# Pyramid shape: same ratios as production, 8x smaller per axis so +# each slice compute is seconds instead of minutes. The slicer-wedge +# bug does not depend on absolute size, only on multiscale + dask + +# per-chunk delayed reads. +LEVELS = [ + (481, 1094, 924), + (241, 547, 462), + (121, 274, 231), + (61, 137, 116), +] +N_CHANNELS = 2 +DTYPE = np.uint16 +CHUNK_SHAPE = { + 0: (16, 128, 128), + 1: (16, 128, 128), + 2: (16, 128, 128), + 3: (16, 128, 128), +} + + +_READ_COUNT = {"n": 0} + + +def _read_block(level, zi, yi, xi, shape): + _READ_COUNT["n"] += 1 + n = _READ_COUNT["n"] + if n <= 50 or n % 20 == 0: + print( + f"[repro-v3] block read #{n}: level={level} indices=({zi},{yi},{xi}) shape={shape}", + flush=True, + ) + # Return cheap data: tiling a tiny array is much faster than + # np.indices on a 2M-cell chunk and still fills the dtype-correct + # buffer napari needs. + val = ((level + 1) * 97 + zi * 13 + yi * 7 + xi * 3) % 4096 + return np.full(shape, val, dtype=DTYPE) + + +def _build_level(level, shape_zyx): + cz, cy, cx = CHUNK_SHAPE[level] + sz, sy, sx = shape_zyx + nz = -(-sz // cz) + ny = -(-sy // cy) + nx = -(-sx // cx) + blocks = np.empty((nz, ny, nx), dtype=object) + for zi in range(nz): + for yi in range(ny): + for xi in range(nx): + this_shape = ( + min(cz, sz - zi * cz), + min(cy, sy - yi * cy), + min(cx, sx - xi * cx), + ) + blk = dask.delayed(_read_block)(level, zi, yi, xi, this_shape) + blocks[zi, yi, xi] = da.from_delayed(blk, shape=this_shape, dtype=DTYPE) + return da.block(blocks.tolist()) + + +def build_pyramid(): + print("[repro-v3] building lazy dask pyramid ...", flush=True) + t0 = time.monotonic() + per_channel = [] + for _c in range(N_CHANNELS): + levels_for_this_channel = [] + for lvl_idx, zyx in enumerate(LEVELS): + levels_for_this_channel.append(_build_level(lvl_idx, zyx)) + per_channel.append(levels_for_this_channel) + print( + f"[repro-v3] built pyramid in {time.monotonic() - t0:.2f}s " + f"(no chunk read fired during build)", + flush=True, + ) + return per_channel + + +# --- v2 pieces: heartbeat + worker pool --------------------------- +_STOP = threading.Event() + + +def _heartbeat_loop(): + while not _STOP.wait(5.0): + print("[repro-v3] (heartbeat tick)", flush=True) + + +def _warm_worker(i): + time.sleep(0.05) + return i + + +def start_background_threads(): + print("[repro-v3] starting daemon heartbeat thread ...", flush=True) + hb = threading.Thread( + target=_heartbeat_loop, + name="repro-v3-heartbeat", + daemon=True, + ) + hb.start() + print("[repro-v3] starting ThreadPoolExecutor (4 workers) ...", flush=True) + pool = ThreadPoolExecutor(max_workers=4, thread_name_prefix="repro-v3-pool") + for i in range(8): + pool.submit(_warm_worker, i) + print("[repro-v3] background threads up.", flush=True) + return hb, pool + + +# --- v3 addition: cross-thread Qt bridge -------------------------- +# Same shape as _NapariStatusReporter._Bridge in the launcher: +# * a QObject with a Signal +# * moveToThread(app.thread()) so Qt knows it lives on main +# * background thread calls bridge.emit via Qt.QueuedConnection +# This is the piece most likely to interact with vispy's own QThread +# machinery during slicer init. + +_BRIDGE_HOLDER: dict = {} # keep refs alive + + +def _build_cross_thread_bridge(): + """Build the bridge + its background emitter. + + Can't construct a QObject until QApplication exists, so this + returns a 2-arg factory: (make_bridge, start_emitter). + """ + from qtpy.QtCore import QObject, Qt, Signal + from qtpy.QtWidgets import QApplication + + class _Bridge(QObject): + status = Signal(str) + + def __init__(self): + super().__init__() + # Match the launcher: hop ourselves to the Qt main thread + # so queued connections marshal correctly. + app = QApplication.instance() + if app is not None: + try: + self.moveToThread(app.thread()) + except Exception: + pass + + def push(self, msg: str): + # Emit on whichever thread called us; the slot is wired + # via Qt.QueuedConnection so it will land on main. + self.status.emit(msg) + + def make_bridge(): + bridge = _Bridge() + + def _on_status(msg): + # Normally this would call viewer.status = msg, but we + # keep it a pure print so this bridge works before Viewer + # exists (that is the production ordering too). + print(f"[repro-v3] (bridge->main) {msg}", flush=True) + + bridge.status.connect(_on_status, Qt.QueuedConnection) + _BRIDGE_HOLDER["bridge"] = bridge + return bridge + + def start_emitter(bridge): + def _loop(): + n = 0 + while not _STOP.wait(1.0): + n += 1 + bridge.push(f"heartbeat {n}") + + t = threading.Thread(target=_loop, name="repro-v3-bridge-emitter", daemon=True) + t.start() + _BRIDGE_HOLDER["emitter"] = t + return t + + return make_bridge, start_emitter + + +def _attach_unstick_hotkey(viewer, layers): + try: + from qtpy.QtCore import QTimer + except ImportError: + return + + @viewer.bind_key("Shift-V", overwrite=True) + def _unstick(_viewer): + print("[repro-v3] applying unstick workaround ...", flush=True) + hidden = [] + for layer in layers: + try: + if layer.visible: + layer.visible = False + hidden.append(layer) + except Exception: + pass + + def _finish(): + for layer in hidden: + try: + layer.visible = True + except Exception: + pass + for layer in layers: + try: + layer.refresh() + except Exception: + pass + print( + f"[repro-v3] unstick: toggled {len(hidden)} layer(s).", flush=True + ) + + QTimer.singleShot(50, _finish) + + +def main(): + print( + "[repro-v3] env: NAPARI_ASYNC=" + os.environ.get("NAPARI_ASYNC", ""), + flush=True, + ) + + per_channel = build_pyramid() + hb, pool = start_background_threads() + + # QApplication must exist before the bridge is built. napari's + # Viewer() constructor creates one if none exists, but the launcher + # builds the bridge BEFORE Viewer(), so we match that by poking + # QApplication ourselves. + from qtpy.QtWidgets import QApplication + + app = QApplication.instance() or QApplication(sys.argv) + print("[repro-v3] QApplication ready; building cross-thread bridge ...", flush=True) + make_bridge, start_emitter = _build_cross_thread_bridge() + bridge = make_bridge() + start_emitter(bridge) + + print("[repro-v3] creating napari.Viewer() ...", flush=True) + viewer = napari.Viewer() + print("[repro-v3] Viewer created; adding images ...", flush=True) + layers = [] + for c, levels in enumerate(per_channel): + layer = viewer.add_image( + levels, + multiscale=True, + name=f"Ch{c + 1}", + colormap="green" if c == 0 else "magenta", + blending="additive", + contrast_limits=(0, 4096), + ) + layers.append(layer) + try: + viewer.dims.set_point(0, LEVELS[0][0] // 2) + except Exception: + pass + print( + "[repro-v3] after add_image: " + + ", ".join( + f"{lyr.name}: loaded={lyr.loaded} multiscale={lyr.multiscale}" for lyr in layers + ), + flush=True, + ) + print( + "[repro-v3] zoom/pan now. 'block read' lines => slicer working. " + "No 'block read' + vispy QBasicTimer spam => slicer wedged. " + "Press Shift+V in the viewer window to apply the workaround.", + flush=True, + ) + _attach_unstick_hotkey(viewer, layers) + + print("[repro-v3] entering napari.run() -- window should open now.", flush=True) + try: + napari.run() + finally: + _STOP.set() + pool.shutdown(wait=False) + + print(f"[repro-v3] session end: total chunk reads = {_READ_COUNT['n']}", flush=True) + + +if __name__ == "__main__": + try: + main() + except SystemExit: + raise + except Exception as exc: + import traceback + + print(f"[repro-v3] UNCAUGHT: {type(exc).__name__}: {exc}", flush=True) + traceback.print_exc() + sys.exit(2) From f5ec39b4114d742e67a3f021a157d222ad0a2d3d Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 14:48:28 +0000 Subject: [PATCH 46/55] napari slicer-wedge bisect v4: v2 threads + v3 smaller pyramid, no bridge v3 reproduced the wedge (zero block reads, slicer never dispatched). v3 added three things on top of v2: (a) smaller pyramid shape + cheap np.full chunk reader (b) QApplication explicitly pre-created before napari.Viewer() (c) QObject cross-thread bridge with Qt.QueuedConnection emits from a background thread v4 isolates (a): v2's thread setup unchanged, v3's leaner pyramid. No QApplication pre-create, no Qt bridge. If v4 still produces "block read" lines on zoom, the smaller pyramid is NOT the trigger and the wedge is driven by (b) and/or (c). If v4 reproduces the wedge, the shape ratios in the shrunken pyramid matter and the Qt bridge is off the hook; next bisect would be v1-ish but with v3's shape. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- napari_slicer_repro_v4.py | 255 ++++++++++++++++++++++++++++++++++++++ 1 file changed, 255 insertions(+) create mode 100644 napari_slicer_repro_v4.py diff --git a/napari_slicer_repro_v4.py b/napari_slicer_repro_v4.py new file mode 100644 index 00000000..7d5437b5 --- /dev/null +++ b/napari_slicer_repro_v4.py @@ -0,0 +1,255 @@ +"""napari slicer-wedge diagnostic -- v4. + +v3 wedged the slicer: zero block reads, just heartbeat chatter. +v4 isolates the SIZE of the pyramid as a possible trigger by +keeping v2's thread setup but using v3's shrunken LEVELS and +cheap np.full reader. Specifically: + + v4 = v2 (daemon heartbeat + ThreadPoolExecutor) + v3's smaller + pyramid + v3's cheap chunk reader. + v4 does NOT pre-create a QApplication. + v4 does NOT build the cross-thread Qt bridge. + +If v4 reproduces the wedge, the smaller pyramid shape itself is +the trigger (which would be weird but informative). +If v4 does NOT reproduce the wedge, v5 adds the QApplication +pre-create; if v5 is still fine, v6 adds the Qt bridge and that +one has to be the trigger. + +Usage: + python napari_slicer_repro_v4.py +""" + +import sys + +print("[repro-v4] script started", flush=True) +print(f"[repro-v4] python = {sys.version.splitlines()[0]}", flush=True) +print(f"[repro-v4] argv = {sys.argv}", flush=True) + +try: + import os + import threading + import time + from concurrent.futures import ThreadPoolExecutor + + import numpy as np + + print("[repro-v4] stdlib + numpy ok", flush=True) + import dask + import dask.array as da + + print(f"[repro-v4] dask = {dask.__version__}", flush=True) + import napari + + print(f"[repro-v4] napari = {napari.__version__}", flush=True) + import vispy + + print(f"[repro-v4] vispy = {vispy.__version__}", flush=True) + import qtpy + + print(f"[repro-v4] qtpy = {qtpy.API_NAME}", flush=True) +except Exception as exc: + print(f"[repro-v4] IMPORT FAILED: {type(exc).__name__}: {exc}", flush=True) + sys.exit(1) + + +# Same shrunken shapes as v3. +LEVELS = [ + (481, 1094, 924), + (241, 547, 462), + (121, 274, 231), + (61, 137, 116), +] +N_CHANNELS = 2 +DTYPE = np.uint16 +CHUNK_SHAPE = { + 0: (16, 128, 128), + 1: (16, 128, 128), + 2: (16, 128, 128), + 3: (16, 128, 128), +} + + +_READ_COUNT = {"n": 0} + + +def _read_block(level, zi, yi, xi, shape): + _READ_COUNT["n"] += 1 + n = _READ_COUNT["n"] + if n <= 50 or n % 20 == 0: + print( + f"[repro-v4] block read #{n}: level={level} indices=({zi},{yi},{xi}) shape={shape}", + flush=True, + ) + val = ((level + 1) * 97 + zi * 13 + yi * 7 + xi * 3) % 4096 + return np.full(shape, val, dtype=DTYPE) + + +def _build_level(level, shape_zyx): + cz, cy, cx = CHUNK_SHAPE[level] + sz, sy, sx = shape_zyx + nz = -(-sz // cz) + ny = -(-sy // cy) + nx = -(-sx // cx) + blocks = np.empty((nz, ny, nx), dtype=object) + for zi in range(nz): + for yi in range(ny): + for xi in range(nx): + this_shape = ( + min(cz, sz - zi * cz), + min(cy, sy - yi * cy), + min(cx, sx - xi * cx), + ) + blk = dask.delayed(_read_block)(level, zi, yi, xi, this_shape) + blocks[zi, yi, xi] = da.from_delayed(blk, shape=this_shape, dtype=DTYPE) + return da.block(blocks.tolist()) + + +def build_pyramid(): + print("[repro-v4] building lazy dask pyramid ...", flush=True) + t0 = time.monotonic() + per_channel = [] + for _c in range(N_CHANNELS): + levels_for_this_channel = [] + for lvl_idx, zyx in enumerate(LEVELS): + levels_for_this_channel.append(_build_level(lvl_idx, zyx)) + per_channel.append(levels_for_this_channel) + print( + f"[repro-v4] built pyramid in {time.monotonic() - t0:.2f}s " + f"(no chunk read fired during build)", + flush=True, + ) + return per_channel + + +# v2's thread setup -- no bridge, no QApplication pre-create. +_STOP = threading.Event() + + +def _heartbeat_loop(): + while not _STOP.wait(5.0): + print("[repro-v4] (heartbeat tick)", flush=True) + + +def _warm_worker(i): + time.sleep(0.05) + return i + + +def start_background_threads(): + print("[repro-v4] starting daemon heartbeat thread ...", flush=True) + hb = threading.Thread( + target=_heartbeat_loop, + name="repro-v4-heartbeat", + daemon=True, + ) + hb.start() + print("[repro-v4] starting ThreadPoolExecutor (4 workers) ...", flush=True) + pool = ThreadPoolExecutor(max_workers=4, thread_name_prefix="repro-v4-pool") + for i in range(8): + pool.submit(_warm_worker, i) + print("[repro-v4] background threads up.", flush=True) + return hb, pool + + +def _attach_unstick_hotkey(viewer, layers): + try: + from qtpy.QtCore import QTimer + except ImportError: + return + + @viewer.bind_key("Shift-V", overwrite=True) + def _unstick(_viewer): + print("[repro-v4] applying unstick workaround ...", flush=True) + hidden = [] + for layer in layers: + try: + if layer.visible: + layer.visible = False + hidden.append(layer) + except Exception: + pass + + def _finish(): + for layer in hidden: + try: + layer.visible = True + except Exception: + pass + for layer in layers: + try: + layer.refresh() + except Exception: + pass + print( + f"[repro-v4] unstick: toggled {len(hidden)} layer(s).", flush=True + ) + + QTimer.singleShot(50, _finish) + + +def main(): + print( + "[repro-v4] env: NAPARI_ASYNC=" + os.environ.get("NAPARI_ASYNC", ""), + flush=True, + ) + + per_channel = build_pyramid() + hb, pool = start_background_threads() + + # NOTE: no QApplication pre-create. napari.Viewer() builds its + # own. + print("[repro-v4] creating napari.Viewer() ...", flush=True) + viewer = napari.Viewer() + print("[repro-v4] Viewer created; adding images ...", flush=True) + layers = [] + for c, levels in enumerate(per_channel): + layer = viewer.add_image( + levels, + multiscale=True, + name=f"Ch{c + 1}", + colormap="green" if c == 0 else "magenta", + blending="additive", + contrast_limits=(0, 4096), + ) + layers.append(layer) + try: + viewer.dims.set_point(0, LEVELS[0][0] // 2) + except Exception: + pass + print( + "[repro-v4] after add_image: " + + ", ".join( + f"{lyr.name}: loaded={lyr.loaded} multiscale={lyr.multiscale}" for lyr in layers + ), + flush=True, + ) + print( + "[repro-v4] zoom/pan now. 'block read' lines => slicer working. " + "No 'block read' => slicer wedged. " + "Press Shift+V in the viewer window to apply the workaround.", + flush=True, + ) + _attach_unstick_hotkey(viewer, layers) + + print("[repro-v4] entering napari.run() -- window should open now.", flush=True) + try: + napari.run() + finally: + _STOP.set() + pool.shutdown(wait=False) + + print(f"[repro-v4] session end: total chunk reads = {_READ_COUNT['n']}", flush=True) + + +if __name__ == "__main__": + try: + main() + except SystemExit: + raise + except Exception as exc: + import traceback + + print(f"[repro-v4] UNCAUGHT: {type(exc).__name__}: {exc}", flush=True) + traceback.print_exc() + sys.exit(2) From 5bb095aebe61bbf618ccf12b37c359885bff5805 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 14:56:49 +0000 Subject: [PATCH 47/55] napari slicer-wedge bisect v5: add QApplication pre-create on top of v4 v4 (lean pyramid + threads only) worked -- the lean pyramid is cleared. The wedge in v3 is in the QApplication pre-create and/or the Qt bridge. v5 isolates the QApplication pre-create: v4 unchanged, plus one line of `app = QApplication.instance() or QApplication(sys.argv)` BEFORE napari.Viewer(). No QObject, no Signal, no QueuedConnection, no background emitter. If v5 wedges, the pre-created QApplication (vs the napari-built QNapariApplication class napari.Viewer() otherwise constructs) is the trigger. If v5 does not wedge, v6 adds the Qt bridge on top and that one has to be it. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- napari_slicer_repro_v5.py | 263 ++++++++++++++++++++++++++++++++++++++ 1 file changed, 263 insertions(+) create mode 100644 napari_slicer_repro_v5.py diff --git a/napari_slicer_repro_v5.py b/napari_slicer_repro_v5.py new file mode 100644 index 00000000..8011938e --- /dev/null +++ b/napari_slicer_repro_v5.py @@ -0,0 +1,263 @@ +"""napari slicer-wedge diagnostic -- v5. + +v4 ran: lean pyramid + v2's threads, no bridge, no QApplication +pre-create. The slicer dispatched normally (~1,280+ block reads). + +v5 = v4 + QApplication pre-created BEFORE napari.Viewer() exists. +Nothing else changes: + * no QObject bridge + * no Signal + QueuedConnection + * no background emitter thread touching Qt +Just a bare QApplication, built from Python before napari ever +has a chance to build its own. + +If v5 wedges, the pre-created QApplication is the trigger (napari's +Viewer() uses whatever QApplication is already around; a Python- +built one differs from the napari-built QNapariApplication class). +If v5 does NOT wedge, v6 keeps the pre-create and adds the Qt +bridge, since that is then the only thing left. + +Usage: + python napari_slicer_repro_v5.py +""" + +import sys + +print("[repro-v5] script started", flush=True) +print(f"[repro-v5] python = {sys.version.splitlines()[0]}", flush=True) +print(f"[repro-v5] argv = {sys.argv}", flush=True) + +try: + import os + import threading + import time + from concurrent.futures import ThreadPoolExecutor + + import numpy as np + + print("[repro-v5] stdlib + numpy ok", flush=True) + import dask + import dask.array as da + + print(f"[repro-v5] dask = {dask.__version__}", flush=True) + import napari + + print(f"[repro-v5] napari = {napari.__version__}", flush=True) + import vispy + + print(f"[repro-v5] vispy = {vispy.__version__}", flush=True) + import qtpy + + print(f"[repro-v5] qtpy = {qtpy.API_NAME}", flush=True) +except Exception as exc: + print(f"[repro-v5] IMPORT FAILED: {type(exc).__name__}: {exc}", flush=True) + sys.exit(1) + + +LEVELS = [ + (481, 1094, 924), + (241, 547, 462), + (121, 274, 231), + (61, 137, 116), +] +N_CHANNELS = 2 +DTYPE = np.uint16 +CHUNK_SHAPE = { + 0: (16, 128, 128), + 1: (16, 128, 128), + 2: (16, 128, 128), + 3: (16, 128, 128), +} + + +_READ_COUNT = {"n": 0} + + +def _read_block(level, zi, yi, xi, shape): + _READ_COUNT["n"] += 1 + n = _READ_COUNT["n"] + if n <= 50 or n % 20 == 0: + print( + f"[repro-v5] block read #{n}: level={level} indices=({zi},{yi},{xi}) shape={shape}", + flush=True, + ) + val = ((level + 1) * 97 + zi * 13 + yi * 7 + xi * 3) % 4096 + return np.full(shape, val, dtype=DTYPE) + + +def _build_level(level, shape_zyx): + cz, cy, cx = CHUNK_SHAPE[level] + sz, sy, sx = shape_zyx + nz = -(-sz // cz) + ny = -(-sy // cy) + nx = -(-sx // cx) + blocks = np.empty((nz, ny, nx), dtype=object) + for zi in range(nz): + for yi in range(ny): + for xi in range(nx): + this_shape = ( + min(cz, sz - zi * cz), + min(cy, sy - yi * cy), + min(cx, sx - xi * cx), + ) + blk = dask.delayed(_read_block)(level, zi, yi, xi, this_shape) + blocks[zi, yi, xi] = da.from_delayed(blk, shape=this_shape, dtype=DTYPE) + return da.block(blocks.tolist()) + + +def build_pyramid(): + print("[repro-v5] building lazy dask pyramid ...", flush=True) + t0 = time.monotonic() + per_channel = [] + for _c in range(N_CHANNELS): + levels_for_this_channel = [] + for lvl_idx, zyx in enumerate(LEVELS): + levels_for_this_channel.append(_build_level(lvl_idx, zyx)) + per_channel.append(levels_for_this_channel) + print( + f"[repro-v5] built pyramid in {time.monotonic() - t0:.2f}s " + f"(no chunk read fired during build)", + flush=True, + ) + return per_channel + + +_STOP = threading.Event() + + +def _heartbeat_loop(): + while not _STOP.wait(5.0): + print("[repro-v5] (heartbeat tick)", flush=True) + + +def _warm_worker(i): + time.sleep(0.05) + return i + + +def start_background_threads(): + print("[repro-v5] starting daemon heartbeat thread ...", flush=True) + hb = threading.Thread( + target=_heartbeat_loop, + name="repro-v5-heartbeat", + daemon=True, + ) + hb.start() + print("[repro-v5] starting ThreadPoolExecutor (4 workers) ...", flush=True) + pool = ThreadPoolExecutor(max_workers=4, thread_name_prefix="repro-v5-pool") + for i in range(8): + pool.submit(_warm_worker, i) + print("[repro-v5] background threads up.", flush=True) + return hb, pool + + +def _attach_unstick_hotkey(viewer, layers): + try: + from qtpy.QtCore import QTimer + except ImportError: + return + + @viewer.bind_key("Shift-V", overwrite=True) + def _unstick(_viewer): + print("[repro-v5] applying unstick workaround ...", flush=True) + hidden = [] + for layer in layers: + try: + if layer.visible: + layer.visible = False + hidden.append(layer) + except Exception: + pass + + def _finish(): + for layer in hidden: + try: + layer.visible = True + except Exception: + pass + for layer in layers: + try: + layer.refresh() + except Exception: + pass + print( + f"[repro-v5] unstick: toggled {len(hidden)} layer(s).", flush=True + ) + + QTimer.singleShot(50, _finish) + + +def main(): + print( + "[repro-v5] env: NAPARI_ASYNC=" + os.environ.get("NAPARI_ASYNC", ""), + flush=True, + ) + + per_channel = build_pyramid() + hb, pool = start_background_threads() + + # <<< v5 addition >>>: pre-create QApplication before napari.Viewer() + # No bridge, no QObject, no Signal. Just the bare application + # instance. + from qtpy.QtWidgets import QApplication + + app = QApplication.instance() or QApplication(sys.argv) + print( + f"[repro-v5] QApplication pre-created: {type(app).__module__}.{type(app).__name__}", + flush=True, + ) + + print("[repro-v5] creating napari.Viewer() ...", flush=True) + viewer = napari.Viewer() + print("[repro-v5] Viewer created; adding images ...", flush=True) + layers = [] + for c, levels in enumerate(per_channel): + layer = viewer.add_image( + levels, + multiscale=True, + name=f"Ch{c + 1}", + colormap="green" if c == 0 else "magenta", + blending="additive", + contrast_limits=(0, 4096), + ) + layers.append(layer) + try: + viewer.dims.set_point(0, LEVELS[0][0] // 2) + except Exception: + pass + print( + "[repro-v5] after add_image: " + + ", ".join( + f"{lyr.name}: loaded={lyr.loaded} multiscale={lyr.multiscale}" for lyr in layers + ), + flush=True, + ) + print( + "[repro-v5] zoom/pan now. 'block read' lines => slicer working. " + "No 'block read' => slicer wedged. " + "Press Shift+V in the viewer window to apply the workaround.", + flush=True, + ) + _attach_unstick_hotkey(viewer, layers) + + print("[repro-v5] entering napari.run() -- window should open now.", flush=True) + try: + napari.run() + finally: + _STOP.set() + pool.shutdown(wait=False) + + print(f"[repro-v5] session end: total chunk reads = {_READ_COUNT['n']}", flush=True) + + +if __name__ == "__main__": + try: + main() + except SystemExit: + raise + except Exception as exc: + import traceback + + print(f"[repro-v5] UNCAUGHT: {type(exc).__name__}: {exc}", flush=True) + traceback.print_exc() + sys.exit(2) From b89e9add4840348ab714d459ea58cf22eb2db085 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 15:02:24 +0000 Subject: [PATCH 48/55] napari slicer-wedge bisect v6: v5 + cross-thread Qt bridge (predicted trigger) Bisect so far: v1 baseline -> works (~1,600 block reads) v2 +threads -> works (~1,500 before force-quit) v3 +shrink +app +bridge -> WEDGES v4 shrink only -> works (shape cleared) v5 shrink + app -> works (QApplication pre-create cleared) Only piece left is the cross-thread Qt bridge: a QObject with a Signal, moveToThread(app.thread()), Qt.QueuedConnection, and a background daemon thread emitting through it. v6 adds that on top of v5 and should wedge, confirming the bridge as the fix target in the production launcher (_NapariStatusReporter._Bridge). Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- napari_slicer_repro_v6.py | 315 ++++++++++++++++++++++++++++++++++++++ 1 file changed, 315 insertions(+) create mode 100644 napari_slicer_repro_v6.py diff --git a/napari_slicer_repro_v6.py b/napari_slicer_repro_v6.py new file mode 100644 index 00000000..6e47e953 --- /dev/null +++ b/napari_slicer_repro_v6.py @@ -0,0 +1,315 @@ +"""napari slicer-wedge diagnostic -- v6. + +Bisect so far: + v1 baseline -> works (1,600+ block reads) + v2 +threads -> works (1,500+ block reads before force-quit) + v3 +shrink +app +bridge-> WEDGES (zero block reads) + v4 shrink only -> works (shape is innocent) + v5 shrink + app -> works (QApplication pre-create is innocent) + +Only piece left: the cross-thread Qt bridge. v6 = v5 + the bridge: + * QObject with a Signal + * moveToThread(app.thread()) + * Qt.QueuedConnection + * background daemon thread that calls bridge.push(...) every second, + starting BEFORE napari.Viewer() exists + +If v6 wedges, the Qt bridge is confirmed as the trigger -- this +reproduces what the production launcher does with _NapariStatusReporter +and would be the fix target. If v6 somehow works, we have a Heisenbug +and need another dimension. + +Usage: + python napari_slicer_repro_v6.py +""" + +import sys + +print("[repro-v6] script started", flush=True) +print(f"[repro-v6] python = {sys.version.splitlines()[0]}", flush=True) +print(f"[repro-v6] argv = {sys.argv}", flush=True) + +try: + import os + import threading + import time + from concurrent.futures import ThreadPoolExecutor + + import numpy as np + + print("[repro-v6] stdlib + numpy ok", flush=True) + import dask + import dask.array as da + + print(f"[repro-v6] dask = {dask.__version__}", flush=True) + import napari + + print(f"[repro-v6] napari = {napari.__version__}", flush=True) + import vispy + + print(f"[repro-v6] vispy = {vispy.__version__}", flush=True) + import qtpy + + print(f"[repro-v6] qtpy = {qtpy.API_NAME}", flush=True) +except Exception as exc: + print(f"[repro-v6] IMPORT FAILED: {type(exc).__name__}: {exc}", flush=True) + sys.exit(1) + + +LEVELS = [ + (481, 1094, 924), + (241, 547, 462), + (121, 274, 231), + (61, 137, 116), +] +N_CHANNELS = 2 +DTYPE = np.uint16 +CHUNK_SHAPE = { + 0: (16, 128, 128), + 1: (16, 128, 128), + 2: (16, 128, 128), + 3: (16, 128, 128), +} + + +_READ_COUNT = {"n": 0} + + +def _read_block(level, zi, yi, xi, shape): + _READ_COUNT["n"] += 1 + n = _READ_COUNT["n"] + if n <= 50 or n % 20 == 0: + print( + f"[repro-v6] block read #{n}: level={level} indices=({zi},{yi},{xi}) shape={shape}", + flush=True, + ) + val = ((level + 1) * 97 + zi * 13 + yi * 7 + xi * 3) % 4096 + return np.full(shape, val, dtype=DTYPE) + + +def _build_level(level, shape_zyx): + cz, cy, cx = CHUNK_SHAPE[level] + sz, sy, sx = shape_zyx + nz = -(-sz // cz) + ny = -(-sy // cy) + nx = -(-sx // cx) + blocks = np.empty((nz, ny, nx), dtype=object) + for zi in range(nz): + for yi in range(ny): + for xi in range(nx): + this_shape = ( + min(cz, sz - zi * cz), + min(cy, sy - yi * cy), + min(cx, sx - xi * cx), + ) + blk = dask.delayed(_read_block)(level, zi, yi, xi, this_shape) + blocks[zi, yi, xi] = da.from_delayed(blk, shape=this_shape, dtype=DTYPE) + return da.block(blocks.tolist()) + + +def build_pyramid(): + print("[repro-v6] building lazy dask pyramid ...", flush=True) + t0 = time.monotonic() + per_channel = [] + for _c in range(N_CHANNELS): + levels_for_this_channel = [] + for lvl_idx, zyx in enumerate(LEVELS): + levels_for_this_channel.append(_build_level(lvl_idx, zyx)) + per_channel.append(levels_for_this_channel) + print( + f"[repro-v6] built pyramid in {time.monotonic() - t0:.2f}s " + f"(no chunk read fired during build)", + flush=True, + ) + return per_channel + + +_STOP = threading.Event() + + +def _heartbeat_loop(): + while not _STOP.wait(5.0): + print("[repro-v6] (heartbeat tick)", flush=True) + + +def _warm_worker(i): + time.sleep(0.05) + return i + + +def start_background_threads(): + print("[repro-v6] starting daemon heartbeat thread ...", flush=True) + hb = threading.Thread( + target=_heartbeat_loop, + name="repro-v6-heartbeat", + daemon=True, + ) + hb.start() + print("[repro-v6] starting ThreadPoolExecutor (4 workers) ...", flush=True) + pool = ThreadPoolExecutor(max_workers=4, thread_name_prefix="repro-v6-pool") + for i in range(8): + pool.submit(_warm_worker, i) + print("[repro-v6] background threads up.", flush=True) + return hb, pool + + +# <<< v6 addition >>>: cross-thread Qt bridge, same shape as the +# launcher's _NapariStatusReporter._Bridge. +_BRIDGE_HOLDER: dict = {} + + +def _build_cross_thread_bridge(): + from qtpy.QtCore import QObject, Qt, Signal + from qtpy.QtWidgets import QApplication + + class _Bridge(QObject): + status = Signal(str) + + def __init__(self): + super().__init__() + app = QApplication.instance() + if app is not None: + try: + self.moveToThread(app.thread()) + except Exception: + pass + + def push(self, msg: str): + self.status.emit(msg) + + def make_bridge(): + bridge = _Bridge() + + def _on_status(msg): + print(f"[repro-v6] (bridge->main) {msg}", flush=True) + + bridge.status.connect(_on_status, Qt.QueuedConnection) + _BRIDGE_HOLDER["bridge"] = bridge + return bridge + + def start_emitter(bridge): + def _loop(): + n = 0 + while not _STOP.wait(1.0): + n += 1 + bridge.push(f"heartbeat {n}") + + t = threading.Thread(target=_loop, name="repro-v6-bridge-emitter", daemon=True) + t.start() + _BRIDGE_HOLDER["emitter"] = t + return t + + return make_bridge, start_emitter + + +def _attach_unstick_hotkey(viewer, layers): + try: + from qtpy.QtCore import QTimer + except ImportError: + return + + @viewer.bind_key("Shift-V", overwrite=True) + def _unstick(_viewer): + print("[repro-v6] applying unstick workaround ...", flush=True) + hidden = [] + for layer in layers: + try: + if layer.visible: + layer.visible = False + hidden.append(layer) + except Exception: + pass + + def _finish(): + for layer in hidden: + try: + layer.visible = True + except Exception: + pass + for layer in layers: + try: + layer.refresh() + except Exception: + pass + print( + f"[repro-v6] unstick: toggled {len(hidden)} layer(s).", flush=True + ) + + QTimer.singleShot(50, _finish) + + +def main(): + print( + "[repro-v6] env: NAPARI_ASYNC=" + os.environ.get("NAPARI_ASYNC", ""), + flush=True, + ) + + per_channel = build_pyramid() + hb, pool = start_background_threads() + + from qtpy.QtWidgets import QApplication + + app = QApplication.instance() or QApplication(sys.argv) + print( + f"[repro-v6] QApplication pre-created: {type(app).__module__}.{type(app).__name__}", + flush=True, + ) + + print("[repro-v6] building cross-thread bridge + emitter ...", flush=True) + make_bridge, start_emitter = _build_cross_thread_bridge() + bridge = make_bridge() + start_emitter(bridge) + + print("[repro-v6] creating napari.Viewer() ...", flush=True) + viewer = napari.Viewer() + print("[repro-v6] Viewer created; adding images ...", flush=True) + layers = [] + for c, levels in enumerate(per_channel): + layer = viewer.add_image( + levels, + multiscale=True, + name=f"Ch{c + 1}", + colormap="green" if c == 0 else "magenta", + blending="additive", + contrast_limits=(0, 4096), + ) + layers.append(layer) + try: + viewer.dims.set_point(0, LEVELS[0][0] // 2) + except Exception: + pass + print( + "[repro-v6] after add_image: " + + ", ".join( + f"{lyr.name}: loaded={lyr.loaded} multiscale={lyr.multiscale}" for lyr in layers + ), + flush=True, + ) + print( + "[repro-v6] zoom/pan now. 'block read' lines => slicer working. " + "No 'block read' => slicer WEDGED (expected per bisect).", + flush=True, + ) + _attach_unstick_hotkey(viewer, layers) + + print("[repro-v6] entering napari.run() -- window should open now.", flush=True) + try: + napari.run() + finally: + _STOP.set() + pool.shutdown(wait=False) + + print(f"[repro-v6] session end: total chunk reads = {_READ_COUNT['n']}", flush=True) + + +if __name__ == "__main__": + try: + main() + except SystemExit: + raise + except Exception as exc: + import traceback + + print(f"[repro-v6] UNCAUGHT: {type(exc).__name__}: {exc}", flush=True) + traceback.print_exc() + sys.exit(2) From cd28e6d7777e78903a77e74dff8dbab04bdf874e Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 19:39:05 +0000 Subject: [PATCH 49/55] lightsheet: surface silent fine-decode failures in the upsample fallback MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Chasing a "level shows coarse pixels while Auto picker reports level 0 and no error ever appears" symptom. Root cause in readChunkWithFallback Path 1: try: result = _read_chunk_from_fetcher(fetcher, fine_doc, ...) _bump("hit_fine") return result except Exception: # noqa: BLE001 - a bad decode is fallback time pass # ← fell through to coarse-upsample. No log, no counter. A fine chunk on disk whose decode raises (truncated download, codec mismatch, corrupt blob) is silently replaced with coarse-upsample pixels. The user sees the wrong resolution forever, and there is no hint anywhere in the log that a decode even failed. That was the "not shippable even as a beta" blocker. Three changes, all additive: 1. New stat key `fine_decode_failed` so the stats summary and the end-of-session loader.stats() dump report how often this fired. 2. On every fine-decode failure, print a one-line warning to stderr with the chunk filename and the exception type + message. This is the signal the user asked for -- a silent fallback becomes a loud fallback, so a corrupt-on-disk situation is immediately visible. 3. Lazy daemon thread (started on first readChunkWithFallback call, only when NDI_LIGHTSHEET_DEBUG / NDI_LIGHTSHEET_UPSAMPLE_DEBUG is on) that dumps the fallback stats every ~5 s when something changed. A session that stays blurry can now be diagnosed live rather than only after the viewer closes. Behaviour-preserving: Path 1 still falls through to upsample on a decode failure (a black block would be worse UX than blurry); the only difference is you can see it happen now. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/pyramid/upsample_fallback.py | 71 ++++++++++++++++++++++++++-- 1 file changed, 68 insertions(+), 3 deletions(-) diff --git a/src/ndi/pyramid/upsample_fallback.py b/src/ndi/pyramid/upsample_fallback.py index 4ce276bb..ac1ac19d 100644 --- a/src/ndi/pyramid/upsample_fallback.py +++ b/src/ndi/pyramid/upsample_fallback.py @@ -42,6 +42,7 @@ from __future__ import annotations import os +import sys import threading from typing import Any @@ -291,6 +292,7 @@ def upsampleToBlock( _STATS_LOCK = threading.Lock() _STATS = { "hit_fine": 0, # chunkPathIfCached found it -> real fine returned + "fine_decode_failed": 0, # fine chunk on disk but decode raised (silent-fallback) "upsampled": 0, # fine missing, upsampled from coarse "zero_no_coarse_cover": 0, # no covering coarse chunk in-range "zero_coarse_missing": 0, # covering coarse chunk not on disk @@ -310,13 +312,61 @@ def fallbackStatsSummary() -> str: they land (refresh hint not triggering a slice compute). """ with _STATS_LOCK: - parts = [f"{k}={v}" for k, v in _STATS.items()] + parts = [f"{k}={v}" for k, v in _STATS.items() if not k.startswith("_")] return "upsample-fallback " + " ".join(parts) def _bump(key: str, n: int = 1) -> None: with _STATS_LOCK: _STATS[key] = _STATS.get(key, 0) + n + _STATS["_dirty"] = _STATS.get("_dirty", 0) + n + + +_HEARTBEAT_STARTED = False +_HEARTBEAT_LOCK = threading.Lock() + + +def _ensure_debug_heartbeat() -> None: + """Spin up one daemon thread that dumps fallback stats every ~5 s. + + Only runs when ``_fallback_debug()`` is on. Prints a stats line + only when something changed since the last tick, so a quiet + session stays quiet. Starts lazily on the first fallback call so + there is nothing to clean up at shutdown. + """ + global _HEARTBEAT_STARTED + if _HEARTBEAT_STARTED or not _fallback_debug(): + return + with _HEARTBEAT_LOCK: + if _HEARTBEAT_STARTED: + return + _HEARTBEAT_STARTED = True + + def _loop(): + import time + + last_signature = None + while True: + time.sleep(5.0) + with _STATS_LOCK: + dirty = _STATS.get("_dirty", 0) + if dirty == 0: + continue + _STATS["_dirty"] = 0 + snapshot = {k: v for k, v in _STATS.items() if not k.startswith("_")} + sig = tuple(sorted(snapshot.items())) + if sig == last_signature: + continue + last_signature = sig + parts = " ".join(f"{k}={v}" for k, v in snapshot.items()) + print( + f"[lightsheet] upsample-fallback live: {parts}", + file=sys.stderr, + flush=True, + ) + + t = threading.Thread(target=_loop, name="ndi-lightsheet-fallback-stats", daemon=True) + t.start() def readChunkWithFallback( @@ -355,6 +405,8 @@ def readChunkWithFallback( """ from ndi.pyramid.multiscale import _read_chunk_from_fetcher, _zero_block + _ensure_debug_heartbeat() + # Path 1: fine chunk on disk -> normal read. fine_path = fetcher.chunkPathIfCached(fine_doc, fine_filename) if fine_path is not None: @@ -364,8 +416,21 @@ def readChunkWithFallback( ) _bump("hit_fine") return result - except Exception: # noqa: BLE001 - a bad decode is fallback time - pass + except Exception as decode_exc: # noqa: BLE001 - a bad decode is fallback time + # The fine chunk is on disk but failed to decode: a + # truncated download, a codec mismatch, or a corrupt + # blob. Falling through to coarse-upsample keeps the + # view from going black, but if we don't LOG this, the + # user sees level-3 pixels forever while the Auto-level + # picker reports level 0 and no error ever appears -- + # that was the silent-blurry symptom before this change. + _bump("fine_decode_failed") + print( + f"[lightsheet] fine decode failed, upsampling from coarse: " + f"{fine_filename!r} ({type(decode_exc).__name__}: {decode_exc})", + file=sys.stderr, + flush=True, + ) # Path 2: fine missing -> upsample coarse if we can, and fetch fine. fetcher.prefetchAsync(fine_doc, fine_filename, on_complete=refresh_hint) From ca19fd5732540c90baa5d1b72377872c0abbd40f Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 19:59:14 +0000 Subject: [PATCH 50/55] lightsheet: stop the misleading "no tiles" heartbeat; count refresh hints Two small diagnostics prompted by the last run's log: (1) The 5-second "no tiles requested yet by napari; waiting for first slice ..." heartbeat was telling the user napari had not asked for a slice -- for 70+ seconds -- while 200+ fallback-reader calls were in fact completing. The underlying _FetchCounter counts cloud HTTPS fetches via watchFetches, which is permanently zero on a local-disk pyramid. Check totalReaderActivity() from the upsample fallback before claiming the slicer is dead; when it is non-zero print "local-disk slicer active: N fallback-reader call(s)" instead, with a note that the cloud fetch counter staying at zero is normal on a local pyramid. The user will no longer be told the slicer is dead when it is working. (2) Added a `refresh_hints_fired` counter to the fallback stats dict, incremented on every RefreshHint._fire. Combined with the live heartbeat (NDI_LIGHTSHEET_DEBUG=1), this exposes whether the swap- coarse-to-fine prod is actually running when prefetches complete. If `hit_fine` keeps climbing but `refresh_hints_fired` stays low, the on-complete callback is not landing in RefreshHint; if both climb but the view still looks coarse, napari is ignoring our refresh and the fix moves to the napari / vispy side (synthesising a QWheelEvent to the canvas, or forcing viewer.camera.events.zoom through its native path). Also exposed `_bump` under the public name `bump` from the fallback module so RefreshHint can increment the shared counter without reaching into a private name. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- .../gui/app/lightsheetZarr/refresh_hint.py | 2 ++ src/ndi/gui/app/lightsheetZarr/viewer.py | 32 ++++++++++++++++--- src/ndi/pyramid/upsample_fallback.py | 19 +++++++++++ 3 files changed, 49 insertions(+), 4 deletions(-) diff --git a/src/ndi/gui/app/lightsheetZarr/refresh_hint.py b/src/ndi/gui/app/lightsheetZarr/refresh_hint.py index 87437206..eff3ca92 100644 --- a/src/ndi/gui/app/lightsheetZarr/refresh_hint.py +++ b/src/ndi/gui/app/lightsheetZarr/refresh_hint.py @@ -66,7 +66,9 @@ def _fire(self) -> None: # Under debug, print each attempt so we can see which one # napari finally responded to. from ndi.pyramid.upsample_fallback import _fallback_debug + from ndi.pyramid.upsample_fallback import bump as _bump_stat + _bump_stat("refresh_hints_fired") debug = _fallback_debug() for layer in self._layers: emitted = [] diff --git a/src/ndi/gui/app/lightsheetZarr/viewer.py b/src/ndi/gui/app/lightsheetZarr/viewer.py index b07e128b..35f0bd73 100644 --- a/src/ndi/gui/app/lightsheetZarr/viewer.py +++ b/src/ndi/gui/app/lightsheetZarr/viewer.py @@ -283,10 +283,34 @@ def _heartbeat_loop(self) -> None: with self._lock: idle = time.monotonic() - self._last_activity if self._started == 0: - _ls( - "no tiles requested yet by napari; waiting for first slice ...", - file=self._out, - ) + # _started counts cloud HTTPS fetches via + # watchFetches. On a local-disk pyramid that's + # permanently zero -- but the slicer may well be + # running against the local-disk fallback reader. + # Check that activity before claiming the slicer + # is dead, so we don't tell the user "no tiles + # requested" when they're staring at ~200 + # fallback-reader calls per minute. + try: + from ndi.pyramid.upsample_fallback import ( + totalReaderActivity as _tra, + ) + + fallback_n = _tra() + except Exception: # noqa: BLE001 + fallback_n = 0 + if fallback_n == 0: + _ls( + "no tiles requested yet by napari; waiting for first slice ...", + file=self._out, + ) + else: + _ls( + f"local-disk slicer active: {fallback_n} fallback-reader call(s). " + "(No cloud fetches on a local pyramid, so the fetch counter stays " + "at zero -- that is normal here.)", + file=self._out, + ) elif self._done < self._started and idle >= self._idle_interval: self._maybe_print(force=True) diff --git a/src/ndi/pyramid/upsample_fallback.py b/src/ndi/pyramid/upsample_fallback.py index ac1ac19d..ee1f980a 100644 --- a/src/ndi/pyramid/upsample_fallback.py +++ b/src/ndi/pyramid/upsample_fallback.py @@ -299,6 +299,7 @@ def upsampleToBlock( "zero_upsample_failed": 0, # upsampler returned None "zero_read_fail": 0, # decode of coarse chunk raised "prefetches_queued": 0, # prefetchAsync calls + "refresh_hints_fired": 0, # RefreshHint._fire ran (fine fetch arrived) } @@ -322,6 +323,24 @@ def _bump(key: str, n: int = 1) -> None: _STATS["_dirty"] = _STATS.get("_dirty", 0) + n +# Expose module-level bump so RefreshHint can log its activity in +# the same stats stream the user already watches. +bump = _bump + + +def totalReaderActivity() -> int: + """How many times the fallback reader has returned a block. + + Sum of hit_fine + upsampled + every zero_* bucket. Used by the + viewer heartbeat to tell "napari actually asked for a slice" from + "napari has no local-disk fetches so the cloud-only _FetchCounter + stayed at zero even though the slicer was working all along." + """ + activity_keys = {"hit_fine", "upsampled", "fine_decode_failed"} + with _STATS_LOCK: + return sum(v for k, v in _STATS.items() if k in activity_keys or k.startswith("zero_")) + + _HEARTBEAT_STARTED = False _HEARTBEAT_LOCK = threading.Lock() From d332e87a19d795dbfa9dc8c5a234a96b55ce8bd6 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 20:08:47 +0000 Subject: [PATCH 51/55] lightsheet: trace the refresh-hint chain to find where it breaks Last run showed refresh_hints_fired=0 for the entire session despite ~250 fine chunks landing on disk and the slicer running against 500+ blocks. Napari was only re-slicing because the user was generating camera events by hand -- the automatic nudge our prefetch-complete callback was supposed to send was never arriving. To find the broken link, split the single counter into four and log one install line: refresh_hints_null reader called with slot[0] still None (hint not yet installed, or install was a no-op because _fallback_context did not exist) refresh_hints_called RefreshHint.__call__ invoked from the prefetch's completion callback refresh_hints_scheduled QTimer.singleShot accepted the fire refresh_hints_fired RefreshHint._fire actually ran on the Qt main thread Plus one log line in loader.registerRefreshHint saying whether the hint got installed into the slot or the ctx was None (which would silently skip the install). Interpretation on the next run: refresh_hints_null > 0 at the start -> add_image is slicing before the hint is registered; the earliest prefetches have on_complete=None and can never nudge. refresh_hints_null high the whole run -> registerRefreshHint ran but the slot write did not reach the ctx the reader sees (two different ctx dicts somewhere). called=0, scheduled=0, fired=0 -> prefetches complete but _settle_in_flight is not dispatching the callback; bug in the fetcher. called>0 but fired=0 -> QTimer.singleShot does not dispatch from the worker thread -- Qt event-loop issue, maybe tied to the same destroyed-dispatcher warning we have been seeing from vispy. The counters all ride in the same _STATS dict the live heartbeat already dumps, so the next NDI_LIGHTSHEET_DEBUG=1 run reports them every 5 seconds. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- .../gui/app/lightsheetZarr/refresh_hint.py | 12 +++++++-- src/ndi/pyramid/loader.py | 27 +++++++++++++++++++ src/ndi/pyramid/upsample_fallback.py | 7 ++++- 3 files changed, 43 insertions(+), 3 deletions(-) diff --git a/src/ndi/gui/app/lightsheetZarr/refresh_hint.py b/src/ndi/gui/app/lightsheetZarr/refresh_hint.py index eff3ca92..2023591e 100644 --- a/src/ndi/gui/app/lightsheetZarr/refresh_hint.py +++ b/src/ndi/gui/app/lightsheetZarr/refresh_hint.py @@ -48,14 +48,22 @@ def __init__(self, viewer, layers, debounce_ms: int = 250): def __call__(self, _path=None) -> None: """Called from the completion of an async fetch. Not on Qt thread.""" + from ndi.pyramid.upsample_fallback import bump as _bump_stat + + _bump_stat("refresh_hints_called") if self._QTimer is None: return # QTimer.singleShot is safe from any thread; the callback # runs on the Qt main thread. try: self._QTimer.singleShot(self._debounce_ms, self._fire) - except Exception: # noqa: BLE001 - pass + _bump_stat("refresh_hints_scheduled") + except Exception as exc: # noqa: BLE001 + print( + f"[lightsheet] refresh-hint schedule failed: {type(exc).__name__}: {exc}", + file=sys.stderr, + flush=True, + ) def _fire(self) -> None: # napari 0.5 has three different ways to force a re-slice diff --git a/src/ndi/pyramid/loader.py b/src/ndi/pyramid/loader.py index bb1938ce..6fd9b734 100644 --- a/src/ndi/pyramid/loader.py +++ b/src/ndi/pyramid/loader.py @@ -290,9 +290,36 @@ def registerRefreshHint(self, hint) -> None: self.build() fetcher = self._fetcher ctx = getattr(fetcher, "_fallback_context", None) + import os + import sys as _sys + + debug = os.environ.get("NDI_LIGHTSHEET_DEBUG", "").strip().lower() in ( + "1", + "true", + "on", + "yes", + ) if ctx is None: + if debug: + print( + "[lightsheet] registerRefreshHint: SKIPPED -- fetcher has " + "no _fallback_context (fallback env off, or single-level " + "pyramid). Napari will not be auto-nudged when fine " + "chunks arrive; the user must trigger a camera event.", + file=_sys.stderr, + flush=True, + ) return ctx["refresh_hint_slot"][0] = hint + if debug: + print( + f"[lightsheet] registerRefreshHint: installed " + f"{'hint' if hint is not None else 'None (no-op)'}; " + f"future async fine fetches will " + f"{'nudge napari to re-slice' if hint is not None else 'NOT nudge napari'}.", + file=_sys.stderr, + flush=True, + ) # ------------------------------------------------------------------ stats diff --git a/src/ndi/pyramid/upsample_fallback.py b/src/ndi/pyramid/upsample_fallback.py index ee1f980a..9fba9f5d 100644 --- a/src/ndi/pyramid/upsample_fallback.py +++ b/src/ndi/pyramid/upsample_fallback.py @@ -299,7 +299,10 @@ def upsampleToBlock( "zero_upsample_failed": 0, # upsampler returned None "zero_read_fail": 0, # decode of coarse chunk raised "prefetches_queued": 0, # prefetchAsync calls - "refresh_hints_fired": 0, # RefreshHint._fire ran (fine fetch arrived) + "refresh_hints_null": 0, # reader called with no refresh hint installed + "refresh_hints_called": 0, # RefreshHint.__call__ invoked (from fetch completion) + "refresh_hints_scheduled": 0, # QTimer.singleShot was accepted + "refresh_hints_fired": 0, # RefreshHint._fire ran (on the Qt main thread) } @@ -452,6 +455,8 @@ def readChunkWithFallback( ) # Path 2: fine missing -> upsample coarse if we can, and fetch fine. + if refresh_hint is None: + _bump("refresh_hints_null") fetcher.prefetchAsync(fine_doc, fine_filename, on_complete=refresh_hint) _bump("prefetches_queued") From 4c51e2d754236f961919c7e392930b6a1870f00b Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 21:11:21 +0000 Subject: [PATCH 52/55] lightsheet: fix RefreshHint cross-thread marshalling (make it a QObject) Last run surfaced the root cause of the "stays coarse" symptom: refresh_hints_null=1 only the one pre-registration slice refresh_hints_called=122 RefreshHint.__call__ invoked refresh_hints_scheduled=122 QTimer.singleShot accepted refresh_hints_fired=0 _fire NEVER ran QTimer.singleShot(ms, callable) posts the timer to the event dispatcher of the thread that called it. The fetcher's background pool threads do not have a Qt event loop, so the timer sits there and never fires -- exactly the pattern the vispy warnings "QBasicTimer:: start: current thread's event dispatcher has already been destroyed" describe. Napari never got the nudge to re-slice, so new fine chunks on disk only became visible when the user manually zoomed. Rewrite RefreshHint as a QObject that lives on the Qt main thread and expose the completion callback as a Signal wired with Qt.QueuedConnection. Emitting from any thread marshals a queued event to the main-thread event loop; _onRequest runs on main and schedules the debounced _fire via QTimer (safe now that we are on the right thread). Same pattern as _NapariStatusReporter._Bridge in viewer.py. Also keep a bare-Python no-op shim for headless (no-Qt) imports so tests and non-GUI callers can still import the class. With this fix, refresh_hints_fired should climb along with hit_fine, and napari should repaint without the user zooming. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- .../gui/app/lightsheetZarr/refresh_hint.py | 230 ++++++++++-------- 1 file changed, 135 insertions(+), 95 deletions(-) diff --git a/src/ndi/gui/app/lightsheetZarr/refresh_hint.py b/src/ndi/gui/app/lightsheetZarr/refresh_hint.py index 2023591e..abef477c 100644 --- a/src/ndi/gui/app/lightsheetZarr/refresh_hint.py +++ b/src/ndi/gui/app/lightsheetZarr/refresh_hint.py @@ -18,109 +18,150 @@ from __future__ import annotations import sys -import threading +# RefreshHint must be a QObject so it can own a Signal and live on +# the Qt main thread. The signal is marshalled from any worker +# thread's __call__ into the main thread with Qt.QueuedConnection, +# which is the ONLY way to get the slot to fire on the Qt event +# loop from a non-Qt thread; a bare QTimer.singleShot(ms, callable) +# posts to whichever thread called it, so from a background +# fetcher-pool worker it silently never fires (that is what the +# vispy "QBasicTimer::start: current thread's event dispatcher +# has already been destroyed" warnings describe). -class RefreshHint: - """Debounce ``layer.refresh()`` calls triggered by async fetches. - An async fine-fetch fires this on completion. Many hundreds of - chunks in flight would swamp napari's slicer with refresh events, - so this collects them into one refresh per ``debounce_ms`` - window. +def _tryImportQt(): + """Return (QObject, Qt, Signal, QTimer, QApplication) or None. + + Pulled out as a function so the ImportError path returns a + uniform shape and :class:`RefreshHint` only has to check once. """ + try: + from qtpy.QtCore import QObject, Qt, Signal + from qtpy.QtCore import QTimer as _QTimer + from qtpy.QtWidgets import QApplication + except ImportError: # pragma: no cover - Qt required for napari + return None + return (QObject, Qt, Signal, _QTimer, QApplication) - def __init__(self, viewer, layers, debounce_ms: int = 250): - self._viewer = viewer - self._layers = list(layers) - self._debounce_ms = debounce_ms - self._lock = threading.Lock() - self._timer = None - self._QTimer = None - try: - from qtpy.QtCore import QTimer - - self._QTimer = QTimer - except ImportError: # pragma: no cover - Qt required for napari - return - # The QTimer lives on the main thread; construct it lazily - # on first use so we do not touch Qt at import time. - - def __call__(self, _path=None) -> None: - """Called from the completion of an async fetch. Not on Qt thread.""" - from ndi.pyramid.upsample_fallback import bump as _bump_stat - - _bump_stat("refresh_hints_called") - if self._QTimer is None: - return - # QTimer.singleShot is safe from any thread; the callback - # runs on the Qt main thread. - try: - self._QTimer.singleShot(self._debounce_ms, self._fire) - _bump_stat("refresh_hints_scheduled") - except Exception as exc: # noqa: BLE001 - print( - f"[lightsheet] refresh-hint schedule failed: {type(exc).__name__}: {exc}", - file=sys.stderr, - flush=True, - ) - - def _fire(self) -> None: - # napari 0.5 has three different ways to force a re-slice - # depending on whether the async slicer is on, and they do - # not overlap in every version. Try them in order of least - # invasive to most; whichever one napari actually reacts to - # is what makes the swap from coarse to fine visible. - # Under debug, print each attempt so we can see which one - # napari finally responded to. - from ndi.pyramid.upsample_fallback import _fallback_debug - from ndi.pyramid.upsample_fallback import bump as _bump_stat - - _bump_stat("refresh_hints_fired") - debug = _fallback_debug() - for layer in self._layers: - emitted = [] - try: - # 1. Public API: refresh() -- best case, napari - # re-slices from current view. Some versions only - # redraw the cached slice, which is why we do more. - layer.refresh() - emitted.append("refresh") - except Exception: # noqa: BLE001 - pass - try: - # 2. Fire the set_data event by hand. napari's - # async slicer listens to this; it is what - # `layer.data = layer.data` would emit, without - # re-assigning the array. - events = getattr(layer, "events", None) - if events is not None: - set_data = getattr(events, "set_data", None) - if set_data is not None: - set_data() - emitted.append("events.set_data") - except Exception: # noqa: BLE001 - pass + +_qt = _tryImportQt() + + +if _qt is not None: + _QObject, _Qt, _Signal, _QTimer, _QApplication = _qt + + class _RefreshHintImpl(_QObject): + """QObject that owns the main-thread signal. + + The ``_request`` signal is wired to ``_onRequest`` with + ``Qt.QueuedConnection``. Emitting it from any thread posts + a queued event to the main-thread event loop; Qt then + invokes the slot on the main thread when the loop next + runs. Equivalent to the ``_NapariStatusReporter._Bridge`` + in viewer.py. + """ + + _request = _Signal() + + def __init__(self, viewer, layers, debounce_ms: int): + super().__init__() + self._viewer = viewer + self._layers = list(layers) + self._debounce_ms = debounce_ms + app = _QApplication.instance() + if app is not None: + try: + self.moveToThread(app.thread()) + except Exception: # noqa: BLE001 + pass + self._request.connect(self._onRequest, _Qt.QueuedConnection) + + def __call__(self, _path=None) -> None: + """Called from the completion of an async fetch. Not on Qt thread.""" + from ndi.pyramid.upsample_fallback import bump as _bump_stat + + _bump_stat("refresh_hints_called") + # emit() with a QueuedConnection target returns + # immediately on the worker thread; the slot runs on + # the main thread. try: - # 3. Private force-reload -- present on napari - # >= 0.4.18 to invalidate the slice cache and - # trigger a fresh compute. - reload = getattr(layer, "reload", None) or getattr(layer, "_reload_async", None) - if callable(reload): - reload() - emitted.append("reload") - except Exception: # noqa: BLE001 - pass - if debug and emitted: + self._request.emit() + _bump_stat("refresh_hints_scheduled") + except Exception as exc: # noqa: BLE001 print( - f"[lightsheet] refresh-hint fired on {getattr(layer, 'name', '?')!r}: " - f"{', '.join(emitted)}", + f"[lightsheet] refresh-hint emit failed: " f"{type(exc).__name__}: {exc}", file=sys.stderr, flush=True, ) - -def refreshHintFor(viewer, layers, debounce_ms: int = 250) -> RefreshHint | None: + def _onRequest(self) -> None: + # Running on the main thread now. Debounce via QTimer: + # it was never the problem, we just needed to be on + # the right thread first. + try: + _QTimer.singleShot(self._debounce_ms, self._fire) + except Exception: # noqa: BLE001 + self._fire() + + def _fire(self) -> None: + from ndi.pyramid.upsample_fallback import _fallback_debug + from ndi.pyramid.upsample_fallback import bump as _bump_stat + + _bump_stat("refresh_hints_fired") + debug = _fallback_debug() + for layer in self._layers: + emitted = [] + try: + layer.refresh() + emitted.append("refresh") + except Exception: # noqa: BLE001 + pass + try: + events = getattr(layer, "events", None) + if events is not None: + set_data = getattr(events, "set_data", None) + if set_data is not None: + set_data() + emitted.append("events.set_data") + except Exception: # noqa: BLE001 + pass + try: + reload = getattr(layer, "reload", None) or getattr(layer, "_reload_async", None) + if callable(reload): + reload() + emitted.append("reload") + except Exception: # noqa: BLE001 + pass + if debug and emitted: + print( + f"[lightsheet] refresh-hint fired on " + f"{getattr(layer, 'name', '?')!r}: {', '.join(emitted)}", + file=sys.stderr, + flush=True, + ) + + RefreshHint = _RefreshHintImpl + +else: + + class RefreshHint: # type: ignore[no-redef] + """Headless no-op shim used in environments without Qt. + + Kept so callers can import :class:`RefreshHint` without + wrapping every reference in a conditional; ``refreshHintFor`` + returns None in this case so the hint is never actually + installed. + """ + + def __init__(self, *args, **kwargs): + pass + + def __call__(self, _path=None) -> None: + return None + + +def refreshHintFor(viewer, layers, debounce_ms: int = 250): """Factory: build a debounced refresh hint or None when Qt is missing. Returns None when the layers list is empty (nothing to refresh) @@ -128,7 +169,6 @@ def refreshHintFor(viewer, layers, debounce_ms: int = 250) -> RefreshHint | None """ if not layers: return None - hint = RefreshHint(viewer, layers, debounce_ms=debounce_ms) - if hint._QTimer is None: + if _qt is None: return None - return hint + return RefreshHint(viewer, layers, debounce_ms=debounce_ms) From 9e2dffcef53f0005c03caa72f7fa60c119b2f383 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 21:37:57 +0000 Subject: [PATCH 53/55] lightsheet: real debounce + synthetic QWheelEvent in refresh hint Previous run proved the queued-connection fix works (refresh_hints_fired=205) but surfaced two more problems: (1) The debounce was not debouncing. Each prefetch completion emitted the signal, each emit queued a 250 ms single-shot timer, so a burst of 205 completions produced 205 timers and 205 fires -- hundreds of refresh() calls in a 10 s window. (2) Even with 205 refresh() + events.set_data() calls, hit_fine stayed 0 and upsampled stopped climbing: napari 0.9.1 async slicer ignores the public refresh/set_data API in this build. The Refresh View button's programmatic viewer.camera.zoom *= 0.8 also failed to re-slice. The only thing proven to work on this user's machine is a real mouse wheel over the canvas. Fixes: (1) Replace the per-emit QTimer.singleShot with ONE persistent QTimer owned by the hint. Each _onRequest restarts the timer, so a burst collapses into one _fire ~250 ms after the last fetch completion in the burst. (2) In _fire, after the inert refresh()+set_data() pair, post a synthetic QWheelEvent pair (wheel-up then wheel-down) to the napari qt canvas widget. The two notches cancel so the user sees no zoom flicker, but Qt delivers two camera events to napari's view machinery through the same code path a real mouse wheel takes -- the only path we have evidence re-slices on this build. Canvas widget is resolved through viewer.window._qt_viewer.canvas.native (and a couple of fallbacks) with a best-effort try/except so a napari version bump that renames the attribute doesn't crash the viewer. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- .../gui/app/lightsheetZarr/refresh_hint.py | 204 ++++++++++++------ 1 file changed, 137 insertions(+), 67 deletions(-) diff --git a/src/ndi/gui/app/lightsheetZarr/refresh_hint.py b/src/ndi/gui/app/lightsheetZarr/refresh_hint.py index abef477c..c8a07325 100644 --- a/src/ndi/gui/app/lightsheetZarr/refresh_hint.py +++ b/src/ndi/gui/app/lightsheetZarr/refresh_hint.py @@ -9,57 +9,67 @@ finds the freshly-cached fine data. Kept in :mod:`ndi.gui.app.lightsheetZarr` (not in :mod:`ndi.pyramid`) -because the Qt dependency belongs on the viewer side. A -matplotlib-driven caller would build its own equivalent -- a -callback that redraws the current figure -- and the loader stays -uninvolved either way. +because the Qt dependency belongs on the viewer side. """ from __future__ import annotations import sys -# RefreshHint must be a QObject so it can own a Signal and live on -# the Qt main thread. The signal is marshalled from any worker -# thread's __call__ into the main thread with Qt.QueuedConnection, -# which is the ONLY way to get the slot to fire on the Qt event -# loop from a non-Qt thread; a bare QTimer.singleShot(ms, callable) -# posts to whichever thread called it, so from a background -# fetcher-pool worker it silently never fires (that is what the -# vispy "QBasicTimer::start: current thread's event dispatcher -# has already been destroyed" warnings describe). - def _tryImportQt(): - """Return (QObject, Qt, Signal, QTimer, QApplication) or None. - - Pulled out as a function so the ImportError path returns a - uniform shape and :class:`RefreshHint` only has to check once. - """ + """Return the Qt classes we need or ``None`` headlessly.""" try: - from qtpy.QtCore import QObject, Qt, Signal + from qtpy.QtCore import QEvent, QObject, QPoint, QPointF, Qt, Signal from qtpy.QtCore import QTimer as _QTimer + from qtpy.QtGui import QWheelEvent from qtpy.QtWidgets import QApplication except ImportError: # pragma: no cover - Qt required for napari return None - return (QObject, Qt, Signal, _QTimer, QApplication) + return { + "QObject": QObject, + "Qt": Qt, + "Signal": Signal, + "QTimer": _QTimer, + "QApplication": QApplication, + "QPoint": QPoint, + "QPointF": QPointF, + "QEvent": QEvent, + "QWheelEvent": QWheelEvent, + } _qt = _tryImportQt() if _qt is not None: - _QObject, _Qt, _Signal, _QTimer, _QApplication = _qt + _QObject = _qt["QObject"] + _Qt = _qt["Qt"] + _Signal = _qt["Signal"] + _QTimer = _qt["QTimer"] + _QApplication = _qt["QApplication"] + _QPoint = _qt["QPoint"] + _QPointF = _qt["QPointF"] + _QWheelEvent = _qt["QWheelEvent"] class _RefreshHintImpl(_QObject): - """QObject that owns the main-thread signal. + """QObject that marshals fetch completions into a main-thread refresh. The ``_request`` signal is wired to ``_onRequest`` with - ``Qt.QueuedConnection``. Emitting it from any thread posts - a queued event to the main-thread event loop; Qt then - invokes the slot on the main thread when the loop next - runs. Equivalent to the ``_NapariStatusReporter._Bridge`` - in viewer.py. + ``Qt.QueuedConnection``, so emitting it from any background + thread posts a queued event to the main-thread event loop. + ``_onRequest`` (re)starts a single debounce timer; when the + burst of fetch completions quiets down for ``debounce_ms``, + the timer fires ``_fire`` once. + + ``_fire`` has to do more than ``layer.refresh()``: observed + on this user's napari 0.9.1 + PyQt6, calling ``refresh()`` or + emitting ``events.set_data()`` -- or even reassigning + ``viewer.camera.zoom`` from Python -- does NOT trigger the + async slicer to re-run against the newly-cached fine chunks. + Only an actual mouse wheel event over the canvas does. So we + post a synthetic ``QWheelEvent`` to the vispy canvas widget, + which is the one Qt path we have proof works on this build. """ _request = _Signal() @@ -75,16 +85,16 @@ def __init__(self, viewer, layers, debounce_ms: int): self.moveToThread(app.thread()) except Exception: # noqa: BLE001 pass + # One timer, reused: QTimer.start() restarts it, so + # a burst of __call__s collapses into a single _fire. + self._timer = None # built lazily on main thread self._request.connect(self._onRequest, _Qt.QueuedConnection) + # Entry point from any thread (fetch worker calls us). def __call__(self, _path=None) -> None: - """Called from the completion of an async fetch. Not on Qt thread.""" from ndi.pyramid.upsample_fallback import bump as _bump_stat _bump_stat("refresh_hints_called") - # emit() with a QueuedConnection target returns - # immediately on the worker thread; the slot runs on - # the main thread. try: self._request.emit() _bump_stat("refresh_hints_scheduled") @@ -96,13 +106,16 @@ def __call__(self, _path=None) -> None: ) def _onRequest(self) -> None: - # Running on the main thread now. Debounce via QTimer: - # it was never the problem, we just needed to be on - # the right thread first. - try: - _QTimer.singleShot(self._debounce_ms, self._fire) - except Exception: # noqa: BLE001 - self._fire() + # On the Qt main thread now; safe to touch QTimer. + if self._timer is None: + self._timer = _QTimer(self) + self._timer.setSingleShot(True) + self._timer.timeout.connect(self._fire) + # Restart the debounce window. If a hundred fetches + # complete in 50 ms, this gets restarted a hundred + # times and _fire runs exactly ONCE, ~250 ms after the + # last one. + self._timer.start(self._debounce_ms) def _fire(self) -> None: from ndi.pyramid.upsample_fallback import _fallback_debug @@ -110,11 +123,13 @@ def _fire(self) -> None: _bump_stat("refresh_hints_fired") debug = _fallback_debug() + + # 1. The in-API calls. Cheap, no-op on this user's + # napari but harmless and the right thing to do on + # napari builds where it works. for layer in self._layers: - emitted = [] try: layer.refresh() - emitted.append("refresh") except Exception: # noqa: BLE001 pass try: @@ -123,36 +138,95 @@ def _fire(self) -> None: set_data = getattr(events, "set_data", None) if set_data is not None: set_data() - emitted.append("events.set_data") - except Exception: # noqa: BLE001 - pass - try: - reload = getattr(layer, "reload", None) or getattr(layer, "_reload_async", None) - if callable(reload): - reload() - emitted.append("reload") except Exception: # noqa: BLE001 pass - if debug and emitted: - print( - f"[lightsheet] refresh-hint fired on " - f"{getattr(layer, 'name', '?')!r}: {', '.join(emitted)}", - file=sys.stderr, - flush=True, + + # 2. The real nudge: post a synthetic QWheelEvent to + # the vispy canvas. The user's log shows programmatic + # camera.zoom writes don't re-slice but real mouse + # wheels do, so we send what the kernel-level wheel + # handler would post. Two notches cancel out so the + # zoom level does not visibly change. + posted = self._postCancellingWheelPair() + + if debug: + print( + f"[lightsheet] refresh-hint fired: " + f"refresh()+set_data() on {len(self._layers)} layer(s), " + f"synth-wheel posted={'yes' if posted else 'no'}", + file=sys.stderr, + flush=True, + ) + + def _findCanvasWidget(self): + """Return the Qt widget to post wheel events to, or None. + + napari's internal layout changes between versions; try a + few known attribute paths and fall back to the viewer + window's central widget. + """ + viewer = self._viewer + try: + window = getattr(viewer, "window", None) + qt_viewer = getattr(window, "_qt_viewer", None) or getattr( + window, "qt_viewer", None + ) + if qt_viewer is None: + return None + canvas = getattr(qt_viewer, "canvas", None) + if canvas is None: + return None + native = getattr(canvas, "native", None) or getattr(canvas, "_backend", None) + if native is None: + return canvas # may itself be a QWidget + return native + except Exception: # noqa: BLE001 + return None + + def _postCancellingWheelPair(self) -> bool: + """Post wheel-up then wheel-down to the canvas. + + Returns True if events were posted. The two notches + cancel so the user does not see a zoom flicker, but + napari's view machinery sees two real camera events + and re-slices accordingly. + """ + widget = self._findCanvasWidget() + if widget is None: + return False + try: + app = _QApplication.instance() + w = widget.width() if hasattr(widget, "width") else 400 + h = widget.height() if hasattr(widget, "height") else 300 + pos_f = _QPointF(w / 2, h / 2) + global_pos = widget.mapToGlobal(pos_f.toPoint()) + for dy in (120, -120): + ev = _QWheelEvent( + pos_f, + _QPointF(global_pos), + _QPoint(0, 0), + _QPoint(0, dy), + _Qt.NoButton, + _Qt.NoModifier, + _Qt.NoScrollPhase, + False, ) + app.postEvent(widget, ev) + return True + except Exception as exc: # noqa: BLE001 + print( + f"[lightsheet] synth-wheel post failed: " f"{type(exc).__name__}: {exc}", + file=sys.stderr, + flush=True, + ) + return False RefreshHint = _RefreshHintImpl else: class RefreshHint: # type: ignore[no-redef] - """Headless no-op shim used in environments without Qt. - - Kept so callers can import :class:`RefreshHint` without - wrapping every reference in a conditional; ``refreshHintFor`` - returns None in this case so the hint is never actually - installed. - """ + """Headless no-op shim for environments without Qt.""" def __init__(self, *args, **kwargs): pass @@ -162,11 +236,7 @@ def __call__(self, _path=None) -> None: def refreshHintFor(viewer, layers, debounce_ms: int = 250): - """Factory: build a debounced refresh hint or None when Qt is missing. - - Returns None when the layers list is empty (nothing to refresh) - or when qtpy cannot be imported (headless test, no display). - """ + """Build a debounced refresh hint, or None when Qt is unavailable.""" if not layers: return None if _qt is None: From 9c5134f4dc7f90fd5a07513dd60a6eed529eda4a Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 00:04:19 +0000 Subject: [PATCH 54/55] lightsheet: default upsample fallback OFF (it was hiding arrived fine data) Live test on the user's Maddie lightsheet pyramid showed the fallback actively blocking the swap-to-fine refresh: with fallback ON (previous default): first paint -> upsampled coarse (Path 2) -> napari marks slice "loaded". Prefetches land. Our refresh hint fires. Napari re-slices, Path 1 returns real fine data (hit_fine climbs). But napari does NOT re-render -- the on-screen pixels stay coarse until the user zooms in and out by hand to generate real mouse-wheel camera events. The user described this as "the low-res version replaces the correct draw". with fallback OFF (NDI_LIGHTSHEET_UPSAMPLE_FALLBACK=0): first paint -> black for missing chunks. As prefetches land and napari re-slices, the view refreshes to show real fine data without needing the user to zoom. Correct behaviour. Flip the default. Keep the knob (NDI_LIGHTSHEET_UPSAMPLE_FALLBACK=1 now opts IN) so cloud users with multi-minute fetch stretches can still get the "blurry but present" first paint if they prefer it to black -- that was the original motivation and it may still be the right trade on very slow cloud connections. But locally and on fast cloud links, OFF is the correct default: "black briefly then correct fine pixels" beats "coarse pixels forever that never refine". Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- src/ndi/pyramid/upsample_fallback.py | 37 ++++++++++++++++------------ 1 file changed, 21 insertions(+), 16 deletions(-) diff --git a/src/ndi/pyramid/upsample_fallback.py b/src/ndi/pyramid/upsample_fallback.py index 9fba9f5d..cd8b1311 100644 --- a/src/ndi/pyramid/upsample_fallback.py +++ b/src/ndi/pyramid/upsample_fallback.py @@ -48,24 +48,29 @@ def env_on() -> bool: - """True unless the upsample fallback is explicitly disabled. - - Default flipped to ON in Waltham-Data-Science/NDI-python#320 after - a user reported multi-minute black stretches during level swaps on - a home connection: without the fallback, napari has nothing to - paint for a region until every visible fine-level chunk arrives, - which for a 1000-tile crop over ~15 MB/s aggregate can be minutes. - With the fallback on, the coarsest-level tiles (which - ``prefetchCoarsestLevel`` puts on disk at launch) are upsampled and - painted immediately -- blurry but present -- and sharpen as the - fine-level fetches arrive. - - ``NDI_LIGHTSHEET_UPSAMPLE_FALLBACK=0`` (or false/off/no) restores - the previous opt-in behaviour. Any other value, including empty - (env var unset), enables the fallback. + """False unless explicitly enabled via NDI_LIGHTSHEET_UPSAMPLE_FALLBACK. + + Default flipped back to OFF after live A/B testing on the Maddie + lightsheet pyramid showed the fallback actively hiding fresh fine + data: once Path 2 returned an upsampled-coarse block for a chunk + on the first slice compute, napari treated that slice as "loaded" + and did not re-render even after our refresh hint fired and Path 1 + returned real fine data on the next compute (hit_fine climbed in + the stats while the on-screen pixels stayed coarse). Flipping the + knob to 0 made the view refresh correctly as fine chunks arrived. + + The historical reason to default it ON was multi-minute black + stretches during level swaps on a slow cloud connection; keep the + knob so cloud users can still opt in, but default OFF because + "coarse pixels forever that never refine" is worse UX than "black + briefly then correct fine pixels". + + ``NDI_LIGHTSHEET_UPSAMPLE_FALLBACK=1`` (or true/on/yes) opts in. + Any other value, including empty (env var unset), leaves the + fallback off. """ value = os.environ.get("NDI_LIGHTSHEET_UPSAMPLE_FALLBACK", "").strip().lower() - return value not in ("0", "false", "off", "no") + return value in ("1", "true", "on", "yes") def _fallback_debug() -> bool: From 0e5f8d37a18298024f8ca421128d872a9daab7c7 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 5 Oct 2026 00:08:33 +0000 Subject: [PATCH 55/55] tests: upsample fallback default is OFF now, update the gate tests Follow-up to 9c5134f which flipped env_on() default to False. Two tests still asserted the old ON-by-default behaviour and failed on 3.12 (and would have failed on 3.10 and 3.11 once they finished running): * test_pyramid_upsample_fallback.TestEnvGate.test_absent_env_is_on -> renamed to test_absent_env_is_off and inverted the assertion. * test_pyramid_loader.TestStats.test_fallback_line_only_appears_when_the_env_is_on -> inverted the two branches: unset env now means no 'fallback' key in loader.stats(), and NDI_LIGHTSHEET_UPSAMPLE_FALLBACK=1 is the opt-in that makes it appear. Co-Authored-By: Claude Opus 4.7 Claude-Session: https://claude.ai/code/session_01N67xH9BejMzZp4GNAq8m7w --- tests/test_pyramid_loader.py | 18 ++++++++++-------- tests/test_pyramid_upsample_fallback.py | 13 ++++++++----- 2 files changed, 18 insertions(+), 13 deletions(-) diff --git a/tests/test_pyramid_loader.py b/tests/test_pyramid_loader.py index 3f46708a..c4a68abb 100644 --- a/tests/test_pyramid_loader.py +++ b/tests/test_pyramid_loader.py @@ -217,12 +217,14 @@ def test_after_build_the_fetcher_line_is_present(self): self.assertEqual(snap.get("fetcher"), "cache=42, cloud=10") def test_fallback_line_only_appears_when_the_env_is_on(self): - """The fallback env is on by default now (#320); the stats line - tracks whichever state ``env_on()`` reports. - - With the env unset the default is ON, and the stats show - 'fallback'. With ``=0`` the fallback is off and the stats - section is absent. + """The fallback env defaults OFF after live A/B testing on the + Maddie lightsheet showed the fallback hiding newly-arrived + fine data. The stats line tracks whichever state ``env_on()`` + reports. + + With the env unset the default is OFF, and the stats do NOT + include 'fallback'. With ``=1`` the fallback is opted in and + the stats section is present. """ import os @@ -232,9 +234,9 @@ def test_fallback_line_only_appears_when_the_env_is_on(self): loader.build() with mock.patch.dict(os.environ, {}, clear=False): os.environ.pop("NDI_LIGHTSHEET_UPSAMPLE_FALLBACK", None) - self.assertIn("fallback", loader.stats()) - with mock.patch.dict(os.environ, {"NDI_LIGHTSHEET_UPSAMPLE_FALLBACK": "0"}): self.assertNotIn("fallback", loader.stats()) + with mock.patch.dict(os.environ, {"NDI_LIGHTSHEET_UPSAMPLE_FALLBACK": "1"}): + self.assertIn("fallback", loader.stats()) class TestClose(unittest.TestCase): diff --git a/tests/test_pyramid_upsample_fallback.py b/tests/test_pyramid_upsample_fallback.py index 7f5f0fbc..62d40313 100644 --- a/tests/test_pyramid_upsample_fallback.py +++ b/tests/test_pyramid_upsample_fallback.py @@ -194,14 +194,18 @@ def test_it_respects_row_major_with_the_expected_stride(self): class TestEnvGate(unittest.TestCase): - def test_absent_env_is_on(self): - """Default flipped in #320 so users see coarse-level fill during - level swaps instead of a black screen while fine tiles load.""" + def test_absent_env_is_off(self): + """Default re-flipped OFF after live A/B testing: with the + fallback on, napari treated the first slice as 'loaded' with + upsampled-coarse data and did not re-render when fine chunks + arrived. Black briefly then correct beats blurry forever.""" with mock.patch.dict(os.environ, {}, clear=False): os.environ.pop("NDI_LIGHTSHEET_UPSAMPLE_FALLBACK", None) - self.assertTrue(uf.env_on()) + self.assertFalse(uf.env_on()) def test_truthy_env_is_on(self): + """The opt-in knob: cloud users on slow links who prefer a + blurry-but-present first paint to a black one can still get it.""" for value in ("1", "true", "on", "yes", "TRUE"): with mock.patch.dict( os.environ, {"NDI_LIGHTSHEET_UPSAMPLE_FALLBACK": value}, clear=False @@ -209,7 +213,6 @@ def test_truthy_env_is_on(self): self.assertTrue(uf.env_on(), value) def test_falsy_env_is_off(self): - """The opt-out knob: anything falsy disables the fallback.""" for value in ("0", "false", "off", "no", "FALSE"): with mock.patch.dict( os.environ, {"NDI_LIGHTSHEET_UPSAMPLE_FALLBACK": value}, clear=False