From 691b03bba961efb888543376b08ed88085ae9025 Mon Sep 17 00:00:00 2001 From: arthurmccray Date: Wed, 16 Sep 2026 17:57:47 -0700 Subject: [PATCH 01/12] handling zenodo being down or other download failures --- emdatabase/downloadable_dataset.py | 53 ++++++++++++++++++++-- emdatabase/tests/test_load_data.py | 72 +++++++++++++++++++++++++++++- 2 files changed, 121 insertions(+), 4 deletions(-) diff --git a/emdatabase/downloadable_dataset.py b/emdatabase/downloadable_dataset.py index 4edcc28..af1e985 100644 --- a/emdatabase/downloadable_dataset.py +++ b/emdatabase/downloadable_dataset.py @@ -39,6 +39,15 @@ class StaleIndexWarning(UserWarning): index describes, so the shipped checksum is out of date.""" +class DownloadFailedWarning(UserWarning): + """A background download failed. + + Using the path raises the error itself, but a caller that never touches the + handle - a notebook cell that only reads ``done`` - would otherwise be told + nothing at all, because the failure happened on another thread. + """ + + def _clear_upstream_cache() -> None: """Forget the fetched index documents and which families have warned.""" _UPSTREAM_CACHE.clear() @@ -183,7 +192,28 @@ def _pending_key(path: object) -> str: return os.path.normcase(os.path.abspath(str(path))) -def _release_pending(key: str, future: "Future[Path]") -> None: +def _settle_pending(key: str, future: "Future[Path]") -> None: + """What becomes of a finished download's entry in :data:`_PENDING`. + + A failure keeps its entry, so that ``result()`` and every consumer of the + path go on re-raising it. Dropped, a failed download would be + indistinguishable from a file that was already on disk: ``done`` would say + True, the file would not be there, and the exception - the only account of + what went wrong - would be discarded without anyone retrieving it. + """ + # Importing the widget module costs nothing: it never imports anywidget at + # module load, and a cancelled download is not a failure to report. + from emdatabase.widget import DownloadCancelled + + error = None if future.cancelled() else future.exception() + if error is not None and not isinstance(error, DownloadCancelled): + warnings.warn( + f"{os.path.basename(key)} was not downloaded: " + f"{type(error).__name__}: {error} " + "Using the path raises this error; call download() again to retry.", + DownloadFailedWarning, + ) + return with _PENDING_LOCK: if _PENDING.get(key) is future: del _PENDING[key] @@ -204,6 +234,12 @@ class DatasetPath(_ConcretePath): one that is already done, so :meth:`DownloadableDataset.download` returns this type whether or not it downloaded anything. + A download that failed - an unreachable host, a checksum that did not match + - is ``done`` but not successful: ``failed`` says so without blocking, and + anything that touches the file raises the original error rather than + ``FileNotFoundError``. The failure is also warned about when it happens, so + a caller who never uses the handle still hears about it. + Any path pointing at the same file waits, however it was built. ``str()`` and ``Path()`` are the exceptions: they hand back a plain value with no download attached, so ``hs.load(str(handle))`` will not block. @@ -217,7 +253,7 @@ def _attach(self, future: "Future[Path]") -> "DatasetPath": key = _pending_key(self) with _PENDING_LOCK: _PENDING[key] = future - future.add_done_callback(lambda finished: _release_pending(key, finished)) + future.add_done_callback(lambda finished: _settle_pending(key, finished)) return self def __fspath__(self) -> str: @@ -241,13 +277,24 @@ def done(self) -> bool: future = self._future return future.done() if future is not None else True + @property + def failed(self) -> bool: + """Whether the download finished with an error. Never blocks, never raises.""" + future = self._future + if future is None or not future.done() or future.cancelled(): + return False + return future.exception() is not None + def wait(self, timeout: float | None = None) -> "DatasetPath": """Block until the download finishes and return self (for chaining).""" self.result(timeout) return self def __repr__(self) -> str: # never block just to display the object - state = "done" if self.done else "downloading" + if self.failed: + state = "failed" + else: + state = "done" if self.done else "downloading" return f"" diff --git a/emdatabase/tests/test_load_data.py b/emdatabase/tests/test_load_data.py index 8694eba..593475c 100644 --- a/emdatabase/tests/test_load_data.py +++ b/emdatabase/tests/test_load_data.py @@ -16,6 +16,7 @@ import time import urllib.error import urllib.request +import warnings import zipfile from pathlib import Path @@ -28,6 +29,7 @@ _PENDING, DatasetPath, DownloadableDataset, + DownloadFailedWarning, _get_executor, _pending_key, _shutdown_executor, @@ -229,6 +231,69 @@ def test_finished_downloads_leave_no_pending_entry(tmp_path, monkeypatch): assert key not in _PENDING +def _failing_retrieve(error): + """A stand-in for ``_retrieve`` that fails the way an unreachable host does.""" + + def retrieve(destination=None, progressbar=True, chunk_size=4096, version=None, refresh=False): + raise error + + return retrieve + + +def test_a_failed_background_download_warns(tmp_path, monkeypatch): + """The failure happens on another thread, so nothing else would report it.""" + dataset = getattr(data, TINY_DATASET)() + monkeypatch.setattr( + dataset, "_retrieve", _failing_retrieve(ConnectionError("zenodo.org is unreachable")) + ) + + with warnings.catch_warnings(record=True) as caught: + warnings.simplefilter("always") + handle = dataset.download(destination=tmp_path, progressbar=False) + for _ in range(300): # the warning comes from the worker thread + if caught: + break + time.sleep(0.01) + + failures = [w for w in caught if issubclass(w.category, DownloadFailedWarning)] + assert failures, [str(w.message) for w in caught] + assert "unreachable" in str(failures[0].message) + assert handle.failed is True + assert handle.done is True # it finished - just not successfully + assert "failed" in repr(handle) + + +def test_a_failed_download_re_raises_when_the_path_is_used(tmp_path, monkeypatch): + """Not FileNotFoundError: the handle still knows why the file is not there.""" + dataset = getattr(data, TINY_DATASET)() + monkeypatch.setattr(dataset, "_retrieve", _failing_retrieve(ConnectionError("host is down"))) + + with warnings.catch_warnings(): + warnings.simplefilter("ignore", DownloadFailedWarning) + handle = dataset.download(destination=tmp_path, progressbar=False) + with pytest.raises(ConnectionError, match="host is down"): + handle.result() + with pytest.raises(ConnectionError, match="host is down"): + os.fspath(handle) + + +def test_downloading_again_after_a_failure_retries(tmp_path, monkeypatch): + """A kept failure must not stop the next attempt from replacing it.""" + dataset = getattr(data, TINY_DATASET)() + monkeypatch.setattr(dataset, "_retrieve", _failing_retrieve(ConnectionError("down"))) + + with warnings.catch_warnings(): + warnings.simplefilter("ignore", DownloadFailedWarning) + first = dataset.download(destination=tmp_path, progressbar=False) + with pytest.raises(ConnectionError): + first.result() + + monkeypatch.setattr(dataset, "_retrieve", _slow_retrieve(dataset, tmp_path)) + second = dataset.download(destination=tmp_path, progressbar=False) + assert Path(os.fspath(second)).read_bytes() == b"payload" + assert second.failed is False + + def test_keyword_overrides_leave_the_class_spec_alone(): base = getattr(data, TINY_DATASET) overridden = base(checksum="md5:" + "0" * 32) @@ -249,8 +314,13 @@ def test_a_dataset_without_a_source_is_an_error(): DownloadableDataset() +@pytest.mark.filterwarnings("ignore::emdatabase.downloadable_dataset.DownloadFailedWarning") def test_download_handle_propagates_errors(tmp_path, monkeypatch): - """A failed background download raises when the handle is consumed.""" + """A failed background download raises when the handle is consumed. + + The warning that failure also emits is this test's own doing; that it is + emitted at all is ``test_a_failed_background_download_warns``'s business. + """ dataset = getattr(data, TINY_DATASET)() def fail(*args): From ab370c320d08a8100ecdd59e45944298d803f069 Mon Sep 17 00:00:00 2001 From: arthurmccray Date: Wed, 16 Sep 2026 23:36:15 -0700 Subject: [PATCH 02/12] handling 7z files --- docs/source/contributing.rst | 36 ++++-- emdatabase/_archive.py | 153 +++++++++++++++++++------ emdatabase/data/__init__.pyi | 19 +++- emdatabase/downloadable_dataset.py | 38 ++++++- emdatabase/index/MOSS6Fig3.yaml | 80 +++++++++++++ emdatabase/index/TEMPLATE.yaml | 17 ++- emdatabase/index/json-schema.json | 35 +++++- emdatabase/metadata.py | 56 +++++++-- emdatabase/tests/test_archive.py | 177 +++++++++++++++++++++++++++++ emdatabase/tests/test_load_data.py | 36 ++++-- emdatabase/tests/test_metadata.py | 29 +++++ pyproject.toml | 1 + 12 files changed, 608 insertions(+), 69 deletions(-) create mode 100644 emdatabase/index/MOSS6Fig3.yaml diff --git a/docs/source/contributing.rst b/docs/source/contributing.rst index b1ef0f4..2f804ba 100644 --- a/docs/source/contributing.rst +++ b/docs/source/contributing.rst @@ -85,10 +85,10 @@ That prints one line per problem and exits non-zero, or prints ``valid``. A file inside an archive ------------------------ -Some data worth shipping is one file inside a multi-gigabyte zip on a record -nobody can re-publish, where downloading all of it to get one file is not -reasonable. Such an entry adds an ``archive`` block naming the zip and the -member inside it: +Some data worth shipping is one file inside a multi-gigabyte ``.zip`` or ``.7z`` +on a record nobody can re-publish, where downloading all of it to get one file is +not reasonable. Such an entry adds an ``archive`` block naming the archive and +the member inside it: .. code-block:: yaml @@ -97,14 +97,34 @@ member inside it: member: Figure_01/Panel_a/scan_x128_y128.raw ``download()`` then fetches only that member, over HTTP range requests: a few -requests read the zip's directory, and the member costs its own compressed bytes -rather than the whole archive. What comes back is a path, as for any other entry. +requests read the archive's directory, and only the member's own bytes follow. +What comes back is a path, as for any other entry. The format is taken from the +link's file name, so no field declares it. + +What that costs differs by format. A zip compresses each member on its own, so +one member is read and streamed directly. A 7z compresses files together in +solid blocks, so reaching a member means decompressing its block from the start: +cheap for a file at the front of a small block, expensive for one at the back of +a large one. It is worth measuring before adding a 7z entry - in the archive +behind ``MOSS6Fig3`` the same 15.3 GB file holds members costing +anywhere from 2 MB to 2.6 GB to reach. Note also that 7z is extracted rather +than streamed, so its progress bar fills in one step at the end and a cancel +cannot interrupt it. The entry's own ``file``, ``checksum`` and ``size_bytes`` go on describing the member, which is the file you end up with; ``checksum`` and ``size_bytes`` inside the block describe the archive instead, and are optional. Give ``member`` -as the complete path inside the zip - one file name can appear in several of its -directories, so a basename alone is ambiguous. +as the complete path inside the archive - one file name can appear in several of +its directories, so a basename alone is ambiguous. + +Data that is more than one file adds ``companions``, each naming a further member +and the ``file`` it is saved as. ``download()`` still returns the entry's own +``file``; the rest arrive beside it. Give them all a directory of their own when +a reader expects to find them together - an EMPAD ``.xml`` names its ``.raw`` by +bare file name and opens it next to itself, so ``MyDataset/acquisition_12.xml`` +and ``MyDataset/scan_x256_y256.raw`` both keeps the pairing and keeps it clear of +identically named members elsewhere. ``delete()`` removes the companions too, and +the size shown in the catalogue counts them. These entries are written by hand. The two fields CI otherwise fills in would be taken from the archive rather than from the member, so a pull request leaving diff --git a/emdatabase/_archive.py b/emdatabase/_archive.py index 1508a62..d5b5d9e 100644 --- a/emdatabase/_archive.py +++ b/emdatabase/_archive.py @@ -1,29 +1,43 @@ -"""Fetching one file out of a zip on someone else's server. +"""Fetching one file out of an archive on someone else's server. Some data worth shipping is a single member of a multi-gigabyte archive on a record that cannot be re-published, where downloading all of it to get one file -is not reasonable. A zip's directory sits at its end and names the byte range of -every member, so with HTTP range requests the whole archive never has to move: -three requests find the directory, and the member costs its own stored bytes. +is not reasonable. Both formats handled here keep their directory at a known +place, so with HTTP range requests the whole archive never has to move: a few +requests find the directory, and only the member's own bytes follow. -:class:`_HTTPRangeFile` is the seekable file ``zipfile`` reads that through, and +:class:`_HTTPRangeFile` is the seekable file the readers work through, and :class:`ArchiveMemberDownloader` is the pooch downloader :meth:`~emdatabase.downloadable_dataset.DownloadableDataset._retrieve` hands to :func:`pooch.retrieve` in place of :class:`pooch.HTTPDownloader` when an entry names an ``archive``. + +``.zip`` and ``.7z`` differ in what that costs. A zip compresses each member on +its own, so one member is read and streamed directly. 7z compresses files +together in solid blocks, so reaching a member means decompressing its block +from the start: cheap for a file at the front of a small block, expensive for +one at the back of a large one. Nothing here can change that - it is how the +archive was written - so the cost of a 7z entry is worth measuring before it is +added. """ from __future__ import annotations import io +import os +import tempfile import urllib.parse import urllib.request import zipfile -from typing import Any +from pathlib import Path +from typing import IO, Any + +import py7zr +from py7zr.callbacks import ExtractCallback from emdatabase.downloadable_dataset import USER_AGENT, Progress -# The zip directory is read in many small seeks, so an unbuffered reader would +# A zip's directory is read in many small seeks, so an unbuffered reader would # cost hundreds of requests before a single byte of the member moved. _BUFFER_SIZE = 1 << 20 @@ -42,10 +56,15 @@ def _host(url: str) -> str: return urllib.parse.urlsplit(url).netloc +def _is_sevenzip(url: str) -> bool: + """Which reader to use, taken from the link's own file name.""" + return urllib.parse.urlsplit(url).path.lower().endswith(".7z") + + class _HTTPRangeFile(io.RawIOBase): """A seekable, read-only file over HTTP range requests. - ``zipfile`` only seeks and reads, so ranges stand in for a local copy. + The archive readers only seek and read, so ranges stand in for a local copy. """ def __init__(self, url: str, timeout: float = 120) -> None: @@ -58,7 +77,7 @@ def __init__(self, url: str, timeout: float = 120) -> None: if declared is None: raise ArchiveError( f"{_host(url)} did not say how big {url} is, so the end of the archive - " - "where a zip keeps its directory - cannot be found." + "where its directory lives - cannot be found." ) self.size = int(declared) @@ -105,18 +124,45 @@ def readinto(self, buffer) -> int: # pyright: ignore[reportMissingParameterType return len(data) +class _SevenZipProgress(ExtractCallback): + """Drives a :class:`Progress` from py7zr's extract callbacks. + + py7zr reports once per file, after it has finished writing it, so a 7z + member's bar stands at nothing and then jumps to full rather than filling as + the bytes arrive. A cancel raised from ``update`` therefore cannot interrupt + a 7z extraction the way it interrupts a zip's chunk loop; the bytes are + already written by the time it is called. Giving 7z real streaming progress + would mean driving ``extract``'s writer factory instead, which is worth + doing only once an entry exists that is big enough to need it. + """ + + def __init__(self, bar: Progress) -> None: + self._bar = bar + + def report_start_preparation(self) -> None: ... + + def report_start(self, processing_file_path: str, processing_bytes: str) -> None: ... + + def report_update(self, decompressed_bytes: str) -> None: + self._bar.update(int(decompressed_bytes)) + + def report_end(self, processing_file_path: str, wrote_bytes: str) -> None: ... + + def report_postprocess(self) -> None: ... + + def report_warning(self, message: str) -> None: ... + + class ArchiveMemberDownloader: - """A pooch downloader that pulls one member out of a remote zip. + """A pooch downloader that pulls one member out of a remote archive. pooch calls a downloader as ``(url, output_file, pooch)`` from inside ``pooch.core.stream_download``, which streams to a temporary file, checks it against the entry's ``checksum`` and only then renames it into place, deleting the temporary file on any failure. So this only has to produce the member's bytes and drive ``progressbar`` the way :class:`pooch.HTTPDownloader` - does: ``total`` once before streaming, ``update(n)`` per chunk, then - ``reset()``, ``update(total)``, ``close()``. The widgets' cancel works by - raising from ``update``, which aborts the stream and takes the temporary - file with it. + does: ``total`` once before the bytes, ``update(n)`` as they arrive, then + ``reset()``, ``update(total)``, ``close()``. """ def __init__( @@ -132,25 +178,62 @@ def __init__( self.chunk_size = chunk_size def __call__(self, url: str, output_file: str, _pooch: Any = None) -> None: - bar = self.progressbar with io.BufferedReader(_HTTPRangeFile(url), buffer_size=_BUFFER_SIZE) as stream: # pyright: ignore[reportArgumentType] - with zipfile.ZipFile(stream) as archive: - try: - info = archive.getinfo(self.member) - except KeyError: - raise KeyError( - f"{url} holds no member {self.member!r}. Name the complete path " - "inside the archive: one file name can appear in several of its " - "directories." - ) from None - if bar: - bar.total = info.file_size - with archive.open(info) as member, open(output_file, "wb") as out: - while chunk := member.read(self.chunk_size): - out.write(chunk) - if bar: - bar.update(len(chunk)) - if bar: - bar.reset() - bar.update(info.file_size) - bar.close() + if _is_sevenzip(url): + total = self._from_7z(stream, url, output_file) + else: + total = self._from_zip(stream, url, output_file) + bar = self.progressbar + if bar: + bar.reset() + bar.update(total) + bar.close() + + def _missing(self, url: str) -> str: + return ( + f"{url} holds no member {self.member!r}. Name the complete path inside " + "the archive: one file name can appear in several of its directories." + ) + + def _from_zip(self, stream: IO[bytes], url: str, output_file: str) -> int: + """Stream one zip member out; each is compressed on its own.""" + bar = self.progressbar + with zipfile.ZipFile(stream) as archive: + try: + info = archive.getinfo(self.member) + except KeyError: + raise KeyError(self._missing(url)) from None + if bar: + bar.total = info.file_size + with archive.open(info) as member, open(output_file, "wb") as out: + while chunk := member.read(self.chunk_size): + out.write(chunk) + if bar: + bar.update(len(chunk)) + return info.file_size + + def _from_7z(self, stream: IO[bytes], url: str, output_file: str) -> int: + """Extract one 7z member, which py7zr will only write to a directory.""" + bar = self.progressbar + with py7zr.SevenZipFile(stream) as archive: + wanted = next( + (f for f in archive.list() if f.filename.replace("\\", "/") == self.member), + None, + ) + # py7zr extracts nothing, and raises nothing, for a name the archive + # does not hold, so the failure has to be found here or the download + # would end as a missing file with no explanation. + if wanted is None: + raise KeyError(self._missing(url)) + if bar: + bar.total = wanted.uncompressed + # Written beside pooch's temporary file, so the move is a rename + # rather than a copy, and so a failure cleans up with the directory. + with tempfile.TemporaryDirectory(dir=os.path.dirname(output_file)) as scratch: + archive.extract( + path=scratch, + targets=[wanted.filename], + callback=_SevenZipProgress(bar) if bar else None, + ) + os.replace(Path(scratch) / wanted.filename, output_file) + return wanted.uncompressed diff --git a/emdatabase/data/__init__.pyi b/emdatabase/data/__init__.pyi index 396aae7..b2e420e 100644 --- a/emdatabase/data/__init__.pyi +++ b/emdatabase/data/__init__.pyi @@ -218,6 +218,23 @@ class LayeredCuNb4DSTEM(DownloadableDataset): https://zenodo.org/records/21632101/files + """ + ... + +class MOSS6Fig3(DownloadableDataset): + """ + MOSS6Fig3 + + The MOSS-6 metal-organic framework 4D-STEM ptychography dataset shown in Fig. 3 of "Atomically resolved imaging of radiation-sensitive metal-organic frameworks via electron ptychography" (Nature Communications 16, 2025; doi 10.1038/s41467-025-56215-z). It arrives as a pair in a directory of its own: acquisition_12.xml, the EMPAD header, which is what download() hands back and what a reader is pointed at, and scan_x256_y256.raw beside it, which the header names and which holds the data - 256 x 256 scan positions of 128 x 130 little-endian float32 frames, each frame a 128 x 128 detector followed by two rows of EMPAD metadata. Read it with quantem's read_4dstem(path, file_type="empad"), or rsciio.empad, which takes the header and opens the raw next to it. The scan was taken on a double Cs-corrected FEI Titan Cubed Themis Z at 300 kV, with a 10 mrad convergence semi-angle, a 1.05 Angstrom step, about 100 nm of defocus, and an electron dose of about 98 electrons per square Angstrom. MOSS-6 is a MOF solid solution of the NU-1000 and NU-901 phases. Both files are members of the record's 15.3 GB RawData.7z and are fetched out of it directly, without downloading the archive; the raw costs about 2.6 GB of it. + + DOI: 10.5281/zenodo.13958144 + + License: CC-BY-4.0 + + You can download this dataset here: + https://zenodo.org/records/13958144/files + + """ ... @@ -359,4 +376,4 @@ class ZrNbPrecipitate(DownloadableDataset): """ ... -__all__ = ['AlNanocrystals', 'AmorphousFilm4nm4DSTEM', 'ApoferritinApollo15eps', 'BilayerWS2', 'CuZnEELSMapping', 'CuZnHAADF', 'FeAlStripes', 'HREBSDStrainPatterns', 'InSituElectrochemGrowth', 'LSMOLineScan', 'LSMOLineScanLowLoss', 'LSMOSTOLineScan', 'LSMOSTOLineScanLowLoss', 'LayeredCuNb4DSTEM', 'MgONanoCrystals', 'NiEBSDLarge', 'PdCuSiCrystallization', 'PdNiPGlass', 'PeakDetectionPolymers', 'SPEDAg', 'TutorialUNet', 'TwistedBilayerWSe2Themis', 'ZrNbPrecipitate'] \ No newline at end of file +__all__ = ['AlNanocrystals', 'AmorphousFilm4nm4DSTEM', 'ApoferritinApollo15eps', 'BilayerWS2', 'CuZnEELSMapping', 'CuZnHAADF', 'FeAlStripes', 'HREBSDStrainPatterns', 'InSituElectrochemGrowth', 'LSMOLineScan', 'LSMOLineScanLowLoss', 'LSMOSTOLineScan', 'LSMOSTOLineScanLowLoss', 'LayeredCuNb4DSTEM', 'MOSS6Fig3', 'MgONanoCrystals', 'NiEBSDLarge', 'PdCuSiCrystallization', 'PdNiPGlass', 'PeakDetectionPolymers', 'SPEDAg', 'TutorialUNet', 'TwistedBilayerWSe2Themis', 'ZrNbPrecipitate'] \ No newline at end of file diff --git a/emdatabase/downloadable_dataset.py b/emdatabase/downloadable_dataset.py index af1e985..747fecc 100644 --- a/emdatabase/downloadable_dataset.py +++ b/emdatabase/downloadable_dataset.py @@ -14,7 +14,12 @@ import yaml from emdatabase.config import LocationName -from emdatabase.metadata import DatasetMetadata, WeightsVersion, versioned_filename +from emdatabase.metadata import ( + ArchiveCompanion, + DatasetMetadata, + WeightsVersion, + versioned_filename, +) USER_AGENT = "emdatabase (https://github.com/electronmicroscopy/emdatabase)" @@ -115,6 +120,7 @@ class _Resolved: file: str pinned: bool member: str | None = None + companions: tuple[ArchiveCompanion, ...] = () class Progress(Protocol): @@ -421,6 +427,7 @@ def _resolve(self, version: str | None = None) -> _Resolved: file=md.file, pinned=True, member=md.archive.member if md.archive else None, + companions=tuple(md.archive.companions) if md.archive else (), ) if version is None: if md.latest is None: @@ -655,6 +662,7 @@ def _retrieve( return shared destination = self._resolve_destination(destination) downloader: Any + alongside: list[tuple[ArchiveCompanion, Any]] = [] if resolved.member is None: downloader = pooch.HTTPDownloader( progressbar=progressbar, # pyright: ignore[reportArgumentType] @@ -663,10 +671,16 @@ def _retrieve( ) else: # Imported here, not at the top, because _archive imports this - # module for USER_AGENT and the progress protocol. + # module for USER_AGENT and the progress protocol - and importing it + # pulls in py7zr, which a download that touches no archive should not + # pay for. from emdatabase._archive import ArchiveMemberDownloader downloader = ArchiveMemberDownloader(resolved.member, progressbar, chunk_size) + alongside = [ + (c, ArchiveMemberDownloader(c.member, progressbar, chunk_size)) + for c in resolved.companions + ] try: if refresh: # pooch keeps a file whose hash it was not given anything to @@ -683,6 +697,18 @@ def _retrieve( path=destination, downloader=downloader, # pyright: ignore[reportArgumentType] ) + # Each companion is a download of its own: its own hash, its own + # atomic write, and skipped outright if it is already there. + for companion, companion_downloader in alongside: + if refresh: + (destination / companion.file).unlink(missing_ok=True) + pooch.retrieve( + url=resolved.url, + known_hash=companion.checksum, + fname=companion.file, + path=destination, + downloader=companion_downloader, + ) finally: # pooch only closes the bar on the happy path, so a failed or # cancelled download would leave it hanging open. @@ -825,7 +851,13 @@ def delete( bool True if a file was removed, False if there was nothing to delete. """ - path = self._resolve_destination(destination) / self.filename(version) + directory = self._resolve_destination(destination) + # Companions go too. They are usually the large ones - a header is what + # the entry points at, and the gigabytes sit beside it - so leaving them + # behind would make delete() look like it had freed the space. + for companion in self._resolve(version).companions: + (directory / companion.file).unlink(missing_ok=True) + path = directory / self.filename(version) if path.exists(): path.unlink() return True diff --git a/emdatabase/index/MOSS6Fig3.yaml b/emdatabase/index/MOSS6Fig3.yaml new file mode 100644 index 0000000..c2049da --- /dev/null +++ b/emdatabase/index/MOSS6Fig3.yaml @@ -0,0 +1,80 @@ +# $schema: ./json-schema.json +MOSS6Fig3: + description: >- + The MOSS-6 metal-organic framework 4D-STEM ptychography dataset shown in Fig. 3 of + "Atomically resolved imaging of radiation-sensitive metal-organic frameworks via electron + ptychography" (Nature Communications 16, 2025; doi 10.1038/s41467-025-56215-z). It arrives + as a pair in a directory of its own: acquisition_12.xml, the EMPAD header, which is what + download() hands back and what a reader is pointed at, and scan_x256_y256.raw beside it, + which the header names and which holds the data - 256 x 256 scan positions of 128 x 130 + little-endian float32 frames, each frame a 128 x 128 detector followed by two rows of EMPAD + metadata. Read it with quantem's read_4dstem(path, file_type="empad"), or rsciio.empad, + which takes the header and opens the raw next to it. The scan was taken on a double + Cs-corrected FEI Titan Cubed Themis Z at 300 kV, with a 10 mrad convergence semi-angle, a + 1.05 Angstrom step, about 100 nm of defocus, and an electron dose of about 98 electrons per + square Angstrom. MOSS-6 is a MOF solid solution of the NU-1000 and NU-901 phases. Both + files are members of the record's 15.3 GB RawData.7z and are fetched out of it directly, + without downloading the archive; the raw costs about 2.6 GB of it. + source: https://zenodo.org/records/13958144/files + checksum: md5:6f9a865545655a2d53bf887afa98c135 + file: MOSS6Fig3/acquisition_12.xml + size_bytes: 3953 + archive: + url: https://zenodo.org/records/13958144/files/RawData.7z + member: RawData/ExperimentalData_MOSS-6_Fig.3/acquisition_12.xml + checksum: md5:97581a943b20f2b6ba5d636b859d5f10 + size_bytes: 15266871965 + companions: + - member: RawData/ExperimentalData_MOSS-6_Fig.3/scan_x256_y256.raw + file: MOSS6Fig3/scan_x256_y256.raw + size_bytes: 4362076160 + detector_manufacturer: Thermo Fisher Scientific + detector: EMPAD + microscope_vendor: Thermo Fisher Scientific + microscope_model: Titan Cubed Themis Z + camera_length: 576.6 mm + voltage: 300 kV + license: CC-BY-4.0 + technique: + - 4D-STEM + doi: 10.5281/zenodo.13958144 + tags: + - Ptychography + - Metal-Organic Framework + - Low Dose + authors: + Guanxing Li: + affiliation: King Abdullah University of Science and Technology + orcid: 0000-0003-1573-4528 + Ming Xu: + affiliation: Nanjing Normal University + Wen-Qi Tang: + affiliation: Nanjing Normal University + Ying Liu: + affiliation: Chongqing University + Cailing Chen: + affiliation: King Abdullah University of Science and Technology + orcid: 0000-0003-2598-1354 + Daliang Zhang: + affiliation: Chongqing University + Lingmei Liu: + affiliation: Chongqing University + orcid: 0000-0002-3273-9884 + Shoucong Ning: + affiliation: University of Science and Technology of China + Hui Zhang: + affiliation: South China University of Technology + orcid: 0000-0003-2269-320X + Zhi-Yuan Gu: + affiliation: Nanjing Normal University + orcid: 0000-0002-6245-4759 + Zhiping Lai: + affiliation: King Abdullah University of Science and Technology + orcid: 0000-0001-9555-6009 + David A. Muller: + affiliation: Cornell University + orcid: 0000-0003-4129-0473 + Yu Han: + affiliation: South China University of Technology + orcid: 0000-0003-1462-1118 + kind: dataset diff --git a/emdatabase/index/TEMPLATE.yaml b/emdatabase/index/TEMPLATE.yaml index f4292d9..d35378b 100644 --- a/emdatabase/index/TEMPLATE.yaml +++ b/emdatabase/index/TEMPLATE.yaml @@ -23,7 +23,7 @@ MyDatasetName: file: MyDatasetName.zspy # The file's Content-Length in bytes; the weekly CI check compares it to the server. size_bytes: 1000000 - # Only for a file that lives inside a zip on a record you cannot re-publish, + # Only for a file that lives inside a zip or 7z on a record you cannot republish, # where downloading the whole archive to get one file is not reasonable. # `download()` fetches just this member, with HTTP range requests, and hands # back a path like any other entry. `url`, `checksum` and `size_bytes` here @@ -31,11 +31,23 @@ MyDatasetName: # `checksum` and `size_bytes` above go on describing the member, the file you # end up with. Give the complete path inside the zip: the same file name can # appear in several of its directories. Not allowed on a `kind: weights` entry. + # A 7z compresses files together in solid blocks, so reaching a member means + # decompressing its block from the start: measure the cost before adding one. # archive: # url: https://zenodo.org/records/0000000/files/Figures.zip # member: Figure_01/Panel_a/scan_x128_y128.raw # checksum: md5:0123456789abcdef0123456789abcdef # size_bytes: 2000000 + # # For data that is more than one file. Each companion comes out of the same + # # archive and is saved at its own `file`; `download()` still hands back the + # # entry's own `file` above. Give both a directory of their own when a reader + # # expects to find them side by side - an EMPAD `.xml` names its `.raw` by + # # bare file name and opens it next to itself. + # companions: + # - member: Figure_01/Panel_a/scan_x128_y128.raw + # file: MyDatasetName/scan_x128_y128.raw + # checksum: md5:0123456789abcdef0123456789abcdef + # size_bytes: 1000000 # Who made the detector. See vendors.yaml for the names already in use. detector_manufacturer: Direct Electron # The detector model. @@ -65,7 +77,8 @@ MyDatasetName: authors: Jane Doe: affiliation: University of Somewhere - # ORCID as `0000-0000-0000-0000`. + # ORCID as `0000-0000-0000-0000`. The last character may be `X`, which is + # a legal check digit rather than a typo. orcid: 0000-0002-1825-0097 # What the entry hands out: `dataset` (the default) or `weights` for a model checkpoint. kind: dataset diff --git a/emdatabase/index/json-schema.json b/emdatabase/index/json-schema.json index 76c3da5..54e8790 100644 --- a/emdatabase/index/json-schema.json +++ b/emdatabase/index/json-schema.json @@ -91,7 +91,7 @@ }, "orcid": { "type": "string", - "pattern": "^\\d{4}-\\d{4}-\\d{4}-\\d{4}$" + "pattern": "^\\d{4}-\\d{4}-\\d{4}-\\d{3}[\\dX]$" } }, "required": [ @@ -276,6 +276,13 @@ "size_bytes": { "type": "integer", "minimum": 0 + }, + "companions": { + "type": "array", + "minItems": 1, + "items": { + "$ref": "#/$defs/archiveCompanion" + } } }, "required": [ @@ -283,6 +290,32 @@ "member" ], "additionalProperties": false + }, + "archiveCompanion": { + "type": "object", + "properties": { + "member": { + "type": "string", + "minLength": 1 + }, + "file": { + "type": "string", + "minLength": 1 + }, + "checksum": { + "type": "string", + "pattern": "^md5:[A-Fa-f0-9]{32}$" + }, + "size_bytes": { + "type": "integer", + "minimum": 0 + } + }, + "required": [ + "member", + "file" + ], + "additionalProperties": false } } } \ No newline at end of file diff --git a/emdatabase/metadata.py b/emdatabase/metadata.py index eb27143..68f8395 100644 --- a/emdatabase/metadata.py +++ b/emdatabase/metadata.py @@ -320,9 +320,26 @@ class WeightsVersion: size_bytes: int | None = None +@dataclass(frozen=True) +class ArchiveCompanion: + """A second file the entry's own file cannot be read without. + + An EMPAD acquisition is the plain case: the ``.xml`` is what a reader is + pointed at, but it names its ``.raw`` by bare file name and expects to find + it in the same directory. Both come out of the same archive, and ``file`` + gives each its name on disk - including a directory, which is how the two + are kept together and away from identically named members elsewhere. + """ + + member: str + file: str + checksum: str | None = None + size_bytes: int | None = None + + @dataclass(frozen=True) class ArchiveMember: - """One file inside a zip on someone else's record. + """One file inside an archive on someone else's record. An entry names this when the data worth shipping is a single member of a multi-gigabyte archive that cannot be re-published. ``download()`` fetches @@ -330,14 +347,20 @@ class ArchiveMember: ``checksum`` and ``size_bytes`` go on describing the file the user ends up with, and the two here describe the archive they come out of. - ``member`` is the complete path inside the zip: one file name can appear in - several directories of the same archive. + ``member`` is the complete path inside the archive: one file name can appear + in several of its directories. A ``.zip`` and a ``.7z`` are both read, told + apart by the link's own file name. + + ``companions`` names further members fetched alongside, for data that is more + than one file; see :class:`ArchiveCompanion`. ``download()`` still hands back + the entry's own ``file``. """ url: str member: str checksum: str | None = None size_bytes: int | None = None + companions: tuple[ArchiveCompanion, ...] = () @dataclass(frozen=True, repr=False) @@ -409,7 +432,11 @@ def from_spec( } ) if values.get("archive") is not None: - values["archive"] = ArchiveMember(**values["archive"]) + archive = dict(values["archive"]) + archive["companions"] = tuple( + ArchiveCompanion(**c) for c in archive.get("companions") or () + ) + values["archive"] = ArchiveMember(**archive) if values.get("latest") is not None: values["latest"] = WeightsVersion(**values["latest"]) values["versions"] = { @@ -420,10 +447,23 @@ def from_spec( except TypeError as error: raise TypeError(f"{_where(origin)}: {error}") from None + @property + def total_bytes(self) -> int | None: + """Every byte the entry puts on disk, companions included, or None. + + :attr:`size_bytes` describes the entry's own file. For a multi-member + entry that is the small one - an EMPAD header beside gigabytes of raw - + so it is this, not that, which answers "how big is this dataset". + """ + if self.size_bytes is None: + return None + companions = self.archive.companions if self.archive else () + return self.size_bytes + sum(c.size_bytes or 0 for c in companions) + @property def size(self) -> str: - """:attr:`size_bytes` formatted for display, or ``""`` if unknown.""" - return format_size(self.size_bytes) + """:attr:`total_bytes` formatted for display, or ``""`` if unknown.""" + return format_size(self.total_bytes) @property def headline(self) -> str: @@ -457,7 +497,9 @@ def __str__(self) -> str: elif entry.name == "model": value = " · ".join(p for p in (value.class_, value.framework, value.quantem) if p) elif entry.name == "archive": - value = f"{value.member} in {value.url}" + value = f"{value.member} in {value.url}" + ( + f" (+{len(value.companions)} alongside)" if value.companions else "" + ) elif entry.name == "latest": value = " · ".join(p for p in (value.checksum, value.url) if p) elif entry.name == "versions": diff --git a/emdatabase/tests/test_archive.py b/emdatabase/tests/test_archive.py index a26f39c..14f163e 100644 --- a/emdatabase/tests/test_archive.py +++ b/emdatabase/tests/test_archive.py @@ -12,9 +12,11 @@ import hashlib import io +import tempfile import zipfile from pathlib import Path +import py7zr import pytest from emdatabase import config @@ -153,3 +155,178 @@ def test_filepath_and_delete_behave_as_for_any_other_dataset(archived, dest): assert ds.filepath() == dest / "scan.raw" assert ds.delete() is True assert ds.filepath() is None + + +# --- the same, out of a 7z ------------------------------------------------- +# +# A zip compresses each member on its own; a 7z compresses them together in +# solid blocks and can only be extracted to a directory. Both end up at the same +# place, so the behaviour asserted here is deliberately the zip's. + +MEMBER_7Z = "Fig_01/Panel_g-h_Themis/scan.raw" + + +def _sevenzip_bytes() -> bytes: + """A small .7z holding the member and a decoy sharing its file name.""" + with tempfile.TemporaryDirectory() as work: + work = Path(work) + (work / "good").write_bytes(CONTENT) + (work / "decoy").write_bytes(DECOY_CONTENT) + built = work / "built.7z" + with py7zr.SevenZipFile(built, "w") as archive: + archive.write(work / "decoy", DECOY) + archive.write(work / "good", MEMBER_7Z) + return built.read_bytes() + + +@pytest.fixture +def archived_7z(http_server): + """Build a dataset whose file is one member of a 7z served from localhost.""" + base, directory = http_server + (directory / "Fig_01.7z").write_bytes(_sevenzip_bytes()) + + def make(**overrides): + archive = {"url": f"{base}/Fig_01.7z", "member": MEMBER_7Z, **overrides.pop("archive", {})} + spec = { + "description": "One headerless raw file inside a 7z.", + "source": base, + "file": "scan.raw", + "checksum": f"md5:{hashlib.md5(CONTENT).hexdigest()}", + "size_bytes": len(CONTENT), + "archive": archive, + **overrides, + } + return DownloadableDataset(**spec) + + return make + + +def test_a_7z_member_comes_out_byte_for_byte(archived_7z, dest): + path = archived_7z().download(destination=dest, progressbar=False, background=False) + assert Path(path).read_bytes() == CONTENT + + +def test_a_7z_member_the_archive_does_not_hold_names_it(archived_7z, dest): + """py7zr extracts nothing and raises nothing for a name it lacks, so we check.""" + ds = archived_7z(archive={"member": "Fig_01/Panel_g-h_Themis/missing.raw"}) + with pytest.raises(KeyError, match="missing.raw"): + ds.download(destination=dest, progressbar=False, background=False) + assert list(dest.iterdir()) == [] + + +def test_a_wrong_checksum_on_a_7z_leaves_no_file(archived_7z, dest): + ds = archived_7z(checksum="md5:" + "0" * 32) + with pytest.raises(ValueError): + ds.download(destination=dest, progressbar=False, background=False) + assert list(dest.iterdir()) == [] + + +def test_the_7z_progress_object_reaches_the_total(archived_7z, dest): + """py7zr reports once at the end, so only the final state is meaningful.""" + bar = _Recorder() + archived_7z().download(destination=dest, progressbar=bar, background=False) + assert bar.total == len(CONTENT) + assert bar.done == len(CONTENT) + assert bar.closed + + +def test_a_7z_host_that_ignores_range_fails_loudly(archived_7z, dest, http_server): + base, _ = http_server + ds = archived_7z(archive={"url": f"{base}/attached/Fig_01.7z"}) + with pytest.raises(ArchiveError, match="Range"): + ds.download(destination=dest, progressbar=False, background=False) + assert list(dest.iterdir()) == [] + + +# --- more than one member -------------------------------------------------- +# +# Some data is not one file. An EMPAD acquisition is a header naming a raw it +# expects to find beside it, so both have to arrive, under the names the archive +# gave them, in a directory of their own. + +COMPANION = "Fig_01/Panel_g-h_Themis/scan_x256_y256.raw" +COMPANION_CONTENT = b"raw detector frames " * 500 + + +@pytest.fixture +def paired(http_server): + """A dataset whose file is a header naming a second member beside it.""" + base, directory = http_server + buffer = io.BytesIO() + with zipfile.ZipFile(buffer, "w", zipfile.ZIP_DEFLATED) as archive: + archive.writestr(MEMBER, CONTENT) + archive.writestr(COMPANION, COMPANION_CONTENT) + (directory / "Paired.zip").write_bytes(buffer.getvalue()) + + def make(**overrides): + spec = { + "description": "A header and the file it names.", + "source": base, + "file": "Paired/scan.raw", + "checksum": f"md5:{hashlib.md5(CONTENT).hexdigest()}", + "size_bytes": len(CONTENT), + "archive": { + "url": f"{base}/Paired.zip", + "member": MEMBER, + "companions": [ + { + "member": COMPANION, + "file": "Paired/scan_x256_y256.raw", + "checksum": f"md5:{hashlib.md5(COMPANION_CONTENT).hexdigest()}", + "size_bytes": len(COMPANION_CONTENT), + } + ], + }, + **overrides, + } + return DownloadableDataset(**spec) + + return make + + +def test_a_companion_lands_beside_the_entrys_own_file(paired, dest): + """Both members, one directory, the names the archive gave them.""" + path = paired().download(destination=dest, progressbar=False, background=False) + + assert Path(path).name == "scan.raw" # the entry's own file is what comes back + folder = dest / "Paired" + assert sorted(p.name for p in folder.iterdir()) == ["scan.raw", "scan_x256_y256.raw"] + assert (folder / "scan_x256_y256.raw").read_bytes() == COMPANION_CONTENT + + +def test_a_companion_with_a_wrong_checksum_fails(paired, dest): + """Each member is verified on its own, not just the one handed back.""" + ds = paired( + archive={ + "url": paired().metadata.archive.url, + "member": MEMBER, + "companions": [ + { + "member": COMPANION, + "file": "Paired/scan_x256_y256.raw", + "checksum": "md5:" + "0" * 32, + } + ], + } + ) + with pytest.raises(ValueError): + ds.download(destination=dest, progressbar=False, background=False) + + +def test_delete_removes_the_companion_too(paired, dest): + """The companion is usually the large one, so leaving it would free nothing.""" + config.set({"locations": {"personal": str(dest)}}) + ds = paired() + ds.download(progressbar=False, background=False) + folder = dest / "Paired" + + assert len(list(folder.iterdir())) == 2 + assert ds.delete() is True + assert list(folder.iterdir()) == [] + + +def test_the_size_shown_counts_the_companion(paired): + """`size_bytes` is the entry's own file; `size` is what the dataset occupies.""" + md = paired().metadata + assert md.size_bytes == len(CONTENT) + assert md.total_bytes == len(CONTENT) + len(COMPANION_CONTENT) diff --git a/emdatabase/tests/test_load_data.py b/emdatabase/tests/test_load_data.py index 593475c..c78c2bb 100644 --- a/emdatabase/tests/test_load_data.py +++ b/emdatabase/tests/test_load_data.py @@ -20,10 +20,11 @@ import zipfile from pathlib import Path +import py7zr import pytest import emdatabase.data as data -from emdatabase._archive import _HTTPRangeFile +from emdatabase._archive import _HTTPRangeFile, _is_sevenzip from emdatabase.data import MgONanoCrystals, NiEBSDLarge from emdatabase.downloadable_dataset import ( _PENDING, @@ -72,15 +73,21 @@ def _head(url, timeout=60): return urllib.request.urlopen(request, timeout=timeout) -def _archive_member(url, member): - """The directory entry for one member of a remote zip. +def _archive_member_size(url, member): + """The uncompressed size of one member of a remote archive. A few small range requests rather than the whole archive, which is the - reason an entry names a member in the first place. + reason an entry names a member in the first place. A zip and a 7z report + that size through different objects, so the size itself is what comes back. """ with io.BufferedReader(_HTTPRangeFile(url), buffer_size=1 << 20) as stream: + if _is_sevenzip(url): + with py7zr.SevenZipFile(stream) as archive: + found = [f for f in archive.list() if f.filename.replace("\\", "/") == member] + assert found, f"{url} holds no member {member!r}" + return found[0].uncompressed with zipfile.ZipFile(stream) as archive: - return archive.getinfo(member) + return archive.getinfo(member).file_size @pytest.mark.network @@ -122,13 +129,18 @@ def test_source_url_resolves(name, version): f"{name}: {url} is {int(length)} bytes, but the YAML declares {declared}" ) if archive is not None: - # What rots for an archive entry is the member being renamed or moved - # inside a zip whose own size never changes. - info = _archive_member(url, archive.member) - assert info.file_size == resolved.size_bytes, ( - f"{name}: {archive.member} is {info.file_size} bytes inside {url}, " - f"but the YAML declares {resolved.size_bytes}" - ) + # What rots for an archive entry is a member being renamed or moved + # inside an archive whose own size never changes. Companions are checked + # too: they are usually the large ones, and nothing else would notice. + members = [(archive.member, resolved.size_bytes)] + members += [(c.member, c.size_bytes) for c in archive.companions] + for member, declared in members: + if declared is None: + continue + size = _archive_member_size(url, member) + assert size == declared, ( + f"{name}: {member} is {size} bytes inside {url}, but the YAML declares {declared}" + ) @pytest.mark.parametrize("name", ALL_DATASETS) diff --git a/emdatabase/tests/test_metadata.py b/emdatabase/tests/test_metadata.py index f178177..982ee1e 100644 --- a/emdatabase/tests/test_metadata.py +++ b/emdatabase/tests/test_metadata.py @@ -13,6 +13,7 @@ from emdatabase.metadata import ( TEMPLATE_PATH, + ArchiveCompanion, ArchiveMember, Author, DatasetMetadata, @@ -102,6 +103,23 @@ def test_author_schema_and_dataclass_agree(): assert list(author_schema["properties"]) == [f.name for f in dataclasses.fields(Author)] +def test_an_orcid_ending_in_x_is_accepted(): + """``X`` is a legal ORCID check digit (ISO 7064 mod 11-2), not a typo, so a + digits-only pattern would reject about one valid ORCID in eleven.""" + + def entry(orcid): + return { + "description": "A dataset by someone with an ORCID.", + "source": "https://example.com", + "file": "f.zspy", + "authors": {"Jane Doe": {"affiliation": "Somewhere", "orcid": orcid}}, + } + + assert validate_document({"Fine_Name": entry("0000-0003-2269-320X")}) == [] + assert validate_document({"Fine_Name": entry("0000-0003-2269-320Y")}) # only X, not any letter + assert validate_document({"Fine_Name": entry("0000-0003-2269-32XX")}) # and only at the end + + def test_weights_file_schema_and_dataclass_agree(): """``latest`` and every dated version are the same three fields.""" weights_file = SCHEMA["$defs"]["weightsFile"] @@ -118,6 +136,17 @@ def test_archive_member_schema_and_dataclass_agree(): assert ENTRY_SCHEMA["properties"]["archive"] == {"$ref": "#/$defs/archiveMember"} +def test_archive_companion_schema_and_dataclass_agree(): + """A companion comes out of the same archive and is saved beside the entry's file.""" + companion = SCHEMA["$defs"]["archiveCompanion"] + fields = [f.name for f in dataclasses.fields(ArchiveCompanion)] + assert list(companion["properties"]) == fields + assert companion["required"] == ["member", "file"] + assert SCHEMA["$defs"]["archiveMember"]["properties"]["companions"]["items"] == { + "$ref": "#/$defs/archiveCompanion" + } + + @pytest.mark.parametrize( ("file", "expected"), [("w.pt", "w_260902.pt"), ("weights", "weights_260902"), ("a.tar.gz", "a.tar_260902.gz")], diff --git a/pyproject.toml b/pyproject.toml index 595f386..a97a633 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -11,6 +11,7 @@ authors = [ ] dependencies = [ "pooch", + "py7zr", "pyyaml", "tqdm>=4.67.1", ] From 0f66b0670686bd198a012b0d9a4e18bdd6d62473 Mon Sep 17 00:00:00 2001 From: arthurmccray Date: Thu, 17 Sep 2026 00:21:09 -0700 Subject: [PATCH 03/12] bugfixes - stale failed download, shared location w/ multiple files --- emdatabase/_archive.py | 37 +++++++++++++------- emdatabase/downloadable_dataset.py | 56 ++++++++++++++++++++++++++---- emdatabase/tests/conftest.py | 19 ++++++++++ emdatabase/tests/test_archive.py | 44 +++++++++++++++++++++++ emdatabase/tests/test_load_data.py | 27 ++++++++++++++ 5 files changed, 164 insertions(+), 19 deletions(-) diff --git a/emdatabase/_archive.py b/emdatabase/_archive.py index d5b5d9e..3af94f7 100644 --- a/emdatabase/_archive.py +++ b/emdatabase/_archive.py @@ -107,18 +107,31 @@ def readinto(self, buffer) -> int: # pyright: ignore[reportMissingParameterType self.url, headers={"User-Agent": USER_AGENT, "Range": f"bytes={self.pos}-{end}"}, ) - with urllib.request.urlopen(request, timeout=self.timeout) as response: - # A host that does not do ranges answers 200 with the whole body. - # Reading it would quietly pull the entire archive, which is the one - # thing this exists to avoid, so it is an error rather than a - # fallback. - if response.status != 206: - raise ArchiveError( - f"{_host(self.url)} ignored a Range request and answered " - f"{response.status}, so fetching one member would mean downloading " - f"all {self.size} bytes of {self.url}." - ) - data = response.read() + try: + with urllib.request.urlopen(request, timeout=self.timeout) as response: + # A host that does not do ranges answers 200 with the whole body. + # Reading it would quietly pull the entire archive, which is the one + # thing this exists to avoid, so it is an error rather than a + # fallback. + if response.status != 206: + raise ArchiveError( + f"{_host(self.url)} ignored a Range request and answered " + f"{response.status}, so fetching one member would mean downloading " + f"all {self.size} bytes of {self.url}." + ) + data = response.read() + except OSError as error: + # A host that answered the HEAD and then stopped serving - a 503 + # from one under load, a dropped connection, a range it will not + # give - fails here as an OSError, and ``zipfile`` turns any OSError + # raised while it reads the directory into + # BadZipFile("File is not a zip file"). Reporting an outage as a + # corrupt archive is the substitution ArchiveError exists to + # prevent, so the transport failure is named rather than left to it. + raise ArchiveError( + f"{_host(self.url)} did not serve bytes {self.pos}-{end} of {self.url}: " + f"{type(error).__name__}: {error}" + ) from error buffer[: len(data)] = data self.pos += len(data) return len(data) diff --git a/emdatabase/downloadable_dataset.py b/emdatabase/downloadable_dataset.py index 747fecc..d2b8a12 100644 --- a/emdatabase/downloadable_dataset.py +++ b/emdatabase/downloadable_dataset.py @@ -225,6 +225,24 @@ def _settle_pending(key: str, future: "Future[Path]") -> None: del _PENDING[key] +def _discard_settled(key: str) -> None: + """Drop a kept failure once the file has arrived by some other route. + + :func:`_settle_pending` keeps a failed download's entry so that it goes on + being reported, but the entry is keyed by the path alone and so outlives the + attempt it belongs to: without this, a blocking download that succeeds after + a background one failed would hand back a handle that still raises the old + error - and reprs as ``[failed]`` - for a file that is now on disk. + + A download still running is left alone. It is writing the same file, and + dropping it would let its handle report itself done before the bytes are. + """ + with _PENDING_LOCK: + future = _PENDING.get(key) + if future is not None and future.done(): + del _PENDING[key] + + class DatasetPath(_ConcretePath): """The local path to a dataset, which may still be downloading. @@ -614,9 +632,11 @@ def download( reports download state. With ``background`` False it is already done. """ if not background: - return DatasetPath( - self._retrieve(destination, progressbar, chunk_size, version, refresh) - ) + path = self._retrieve(destination, progressbar, chunk_size, version, refresh) + # The bytes are here, so an earlier failure kept for this path no + # longer describes it and must not be re-raised by the handle. + _discard_settled(_pending_key(path)) + return DatasetPath(path) # Where the file will end up: the copy the search order finds, unless a # destination or a refresh asks for a fresh one. existing = None if destination is not None or refresh else self.filepath(version) @@ -790,13 +810,32 @@ def _retrieve_latest(self, resolved: _Resolved, destination: Path, downloader: A ) return filepath + def _is_complete(self, directory: Path, version: str | None = None) -> bool: + """Whether ``directory`` holds the whole dataset, companions included. + + An entry with companions is not on disk until they are: its own file is + a header, the data sits beside it under the name the archive gave it, + and a header whose raw is missing fails in the reader rather than here. + A directory holding only part of the set is passed over, so the rest is + fetched instead of the dataset reporting itself already downloaded. + """ + if not (directory / self.filename(version)).exists(): + return False + return all((directory / c.file).exists() for c in self._resolve(version).companions) + def _find_in_shared_locations(self, version: str | None = None) -> Path | None: - """Path to an existing copy in a configured shared location, or None.""" + """Path to an existing, complete copy in a shared location, or None.""" from emdatabase import config name = self.filename(version) - shared = (loc.path / name for loc in config.locations() if loc.kind != "personal") - return next((path for path in shared if path.exists()), None) + return next( + ( + location.path / name + for location in config.locations() + if location.kind != "personal" and self._is_complete(location.path, version) + ), + None, + ) def filepaths(self, version: str | None = None) -> list[Path]: """Every copy of the dataset on disk, in search order. @@ -810,11 +849,14 @@ def filepaths(self, version: str | None = None) -> list[Path]: ``version`` asks about one dated version of a weights family; with no version it is the ``latest`` file, which is a different name on disk. + + A copy counts only if it is complete: an entry that names companions is + not there unless they are beside it. """ from emdatabase import config name = self.filename(version) - return [d / name for d in config.data_search_dirs() if (d / name).exists()] + return [d / name for d in config.data_search_dirs() if self._is_complete(d, version)] def filepath(self, version: str | None = None) -> Path | None: """Return the local file path of the dataset if present. diff --git a/emdatabase/tests/conftest.py b/emdatabase/tests/conftest.py index 072544a..c01243c 100644 --- a/emdatabase/tests/conftest.py +++ b/emdatabase/tests/conftest.py @@ -106,10 +106,29 @@ def send_head(self): self.send_header("Content-Disposition", f'attachment; filename="{name}"') self.end_headers() return io.BytesIO(body) + if self.path.startswith("/flaky/"): + return self._die_mid_read() if "Range" in self.headers: return self._send_range() return super().send_head() + def _die_mid_read(self): + """Says how big the file is, then refuses to serve any of it. + + A host that goes down between the HEAD that finds the archive's + directory and the ranges that read it - Zenodo under load answering + ``503`` - which is a failure arriving mid-read rather than up front. + """ + body = (Path(self.directory) / self.path[len("/flaky/") :]).read_bytes() + if self.command == "HEAD": + self.send_response(200) + self.send_header("Content-Type", "application/octet-stream") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + return None + self.send_error(503) + return None + def _send_range(self): """``206`` with the requested slice, for ``bytes=-[]``.""" body = Path(self.translate_path(self.path)).read_bytes() diff --git a/emdatabase/tests/test_archive.py b/emdatabase/tests/test_archive.py index 14f163e..1090e3f 100644 --- a/emdatabase/tests/test_archive.py +++ b/emdatabase/tests/test_archive.py @@ -238,6 +238,21 @@ def test_a_7z_host_that_ignores_range_fails_loudly(archived_7z, dest, http_serve assert list(dest.iterdir()) == [] +def test_a_host_that_stops_serving_mid_read_says_so(archived, dest, http_server): + """The archive is found, then the host goes down while its bytes are being read. + + ``zipfile`` turns any ``OSError`` raised while it reads the directory into + ``BadZipFile("File is not a zip file")``, so an outage reported as it + happens is the difference between a message about the host and one blaming + the archive. + """ + base, _ = http_server + ds = archived(archive={"url": f"{base}/flaky/Fig_01.zip"}) + with pytest.raises(ArchiveError, match="503"): + ds.download(destination=dest, progressbar=False, background=False) + assert list(dest.iterdir()) == [] + + # --- more than one member -------------------------------------------------- # # Some data is not one file. An EMPAD acquisition is a header naming a raw it @@ -325,6 +340,35 @@ def test_delete_removes_the_companion_too(paired, dest): assert list(folder.iterdir()) == [] +def test_a_shared_copy_missing_its_companion_is_not_used(paired, tmp_path, dest): + """A header without its raw is not the dataset, so the whole set is fetched.""" + group = tmp_path / "group" + (group / "Paired").mkdir(parents=True) + (group / "Paired/scan.raw").write_bytes(CONTENT) # the header arrived; the raw did not + config.set({"locations": {"group": str(group), "personal": str(dest)}}) + ds = paired() + + assert ds.filepath() is None # a partial copy is not a copy + path = ds.download(progressbar=False, background=False) + + assert Path(path) == dest / "Paired/scan.raw" # not the shared header + assert (dest / "Paired/scan_x256_y256.raw").read_bytes() == COMPANION_CONTENT + + +def test_a_complete_shared_copy_is_still_used_as_is(paired, tmp_path, dest): + """The shortcut stays a shortcut: both members there means nothing is fetched.""" + group = tmp_path / "group" + (group / "Paired").mkdir(parents=True) + (group / "Paired/scan.raw").write_bytes(CONTENT) + (group / "Paired/scan_x256_y256.raw").write_bytes(COMPANION_CONTENT) + config.set({"locations": {"group": str(group), "personal": str(dest)}}) + + path = paired().download(progressbar=False, background=False) + + assert Path(path) == group / "Paired/scan.raw" + assert list(dest.iterdir()) == [] # nothing downloaded to the personal dir + + def test_the_size_shown_counts_the_companion(paired): """`size_bytes` is the entry's own file; `size` is what the dataset occupies.""" md = paired().metadata diff --git a/emdatabase/tests/test_load_data.py b/emdatabase/tests/test_load_data.py index c78c2bb..4a95e6d 100644 --- a/emdatabase/tests/test_load_data.py +++ b/emdatabase/tests/test_load_data.py @@ -306,6 +306,33 @@ def test_downloading_again_after_a_failure_retries(tmp_path, monkeypatch): assert second.failed is False +def test_a_blocking_download_clears_an_earlier_failure(tmp_path, monkeypatch): + """A kept failure is about a file that is not there; this one is. + + The entry is keyed by path, not by attempt, so a failure left behind would + be found by every handle to that path - including the one just handed back + by the download that succeeded. + """ + dataset = getattr(data, TINY_DATASET)() + monkeypatch.setattr(dataset, "_retrieve", _failing_retrieve(ConnectionError("host is down"))) + + with warnings.catch_warnings(): + warnings.simplefilter("ignore", DownloadFailedWarning) + failed = dataset.download(destination=tmp_path, progressbar=False) + with pytest.raises(ConnectionError): + failed.result() + + monkeypatch.setattr(dataset, "_retrieve", _slow_retrieve(dataset, tmp_path)) + handle = dataset.download(destination=tmp_path, progressbar=False, background=False) + + assert handle.failed is False + assert "failed" not in repr(handle) + assert Path(os.fspath(handle)).read_bytes() == b"payload" + assert _pending_key(handle) not in _PENDING + # and not just this handle: any path to the file, however it was built + assert DatasetPath(tmp_path / dataset.file).failed is False + + def test_keyword_overrides_leave_the_class_spec_alone(): base = getattr(data, TINY_DATASET) overridden = base(checksum="md5:" + "0" * 32) From a6d636cb7cc1a2d9e5be1bf80fd006ccb9a95aac Mon Sep 17 00:00:00 2001 From: arthurmccray Date: Thu, 17 Sep 2026 20:58:27 -0700 Subject: [PATCH 04/12] making it easier for datasets to contain multiple files (and a single head/main file) --- docs/source/contributing.rst | 49 ++++--- emdatabase/_archive.py | 24 ++-- emdatabase/catalogue.py | 2 +- emdatabase/downloadable_dataset.py | 136 ++++++++++-------- emdatabase/index/MOSS6Fig3.yaml | 10 +- emdatabase/index/TEMPLATE.yaml | 21 ++- .../index/TwistedBilayerWSe2Themis.yaml | 9 +- emdatabase/index/json-schema.json | 58 ++++++-- emdatabase/metadata.py | 104 +++++++------- emdatabase/new_dataset.py | 16 ++- emdatabase/tests/test_archive.py | 77 ++++++---- emdatabase/tests/test_fill_download_fields.py | 40 ++++-- emdatabase/tests/test_load_data.py | 21 +-- emdatabase/tests/test_metadata.py | 74 +++++++--- emdatabase/tests/test_new_dataset.py | 3 +- 15 files changed, 398 insertions(+), 246 deletions(-) diff --git a/docs/source/contributing.rst b/docs/source/contributing.rst index 2f804ba..c504bfd 100644 --- a/docs/source/contributing.rst +++ b/docs/source/contributing.rst @@ -88,15 +88,19 @@ A file inside an archive Some data worth shipping is one file inside a multi-gigabyte ``.zip`` or ``.7z`` on a record nobody can re-publish, where downloading all of it to get one file is not reasonable. Such an entry adds an ``archive`` block naming the archive and -the member inside it: +every file taken out of it: .. code-block:: yaml archive: url: https://zenodo.org/records/0000000/files/Figures.zip - member: Figure_01/Panel_a/scan_x128_y128.raw + members: + - member: Figure_01/Panel_a/scan_x128_y128.raw + file: MyDatasetName/scan_x128_y128.raw + checksum: md5:0123456789abcdef0123456789abcdef + size_bytes: 1000000 -``download()`` then fetches only that member, over HTTP range requests: a few +``download()`` then fetches only those members, over HTTP range requests: a few requests read the archive's directory, and only the member's own bytes follow. What comes back is a path, as for any other entry. The format is taken from the link's file name, so no field declares it. @@ -111,24 +115,25 @@ anywhere from 2 MB to 2.6 GB to reach. Note also that 7z is extracted rather than streamed, so its progress bar fills in one step at the end and a cancel cannot interrupt it. -The entry's own ``file``, ``checksum`` and ``size_bytes`` go on describing the -member, which is the file you end up with; ``checksum`` and ``size_bytes`` -inside the block describe the archive instead, and are optional. Give ``member`` -as the complete path inside the archive - one file name can appear in several of -its directories, so a basename alone is ambiguous. - -Data that is more than one file adds ``companions``, each naming a further member -and the ``file`` it is saved as. ``download()`` still returns the entry's own -``file``; the rest arrive beside it. Give them all a directory of their own when -a reader expects to find them together - an EMPAD ``.xml`` names its ``.raw`` by -bare file name and opens it next to itself, so ``MyDataset/acquisition_12.xml`` -and ``MyDataset/scan_x256_y256.raw`` both keeps the pairing and keeps it clear of -identically named members elsewhere. ``delete()`` removes the companions too, and -the size shown in the catalogue counts them. +An archive entry leaves out the top-level ``file``, ``checksum`` and +``size_bytes``. The first two are the first member's, stated there instead so +that no fact about a file appears at two levels; the entry's ``size_bytes`` is +every member's added up, which is the one number ``size`` shows. ``checksum`` and +``size_bytes`` directly under ``archive`` describe the archive itself, and are +optional. Give ``member`` as the complete path inside the archive - one file name +can appear in several of its directories, so a basename alone is ambiguous. + +Data that is more than one file lists them all under ``members``. The first is +what ``download()`` returns; the rest arrive beside it. Give them a directory of +their own when a reader expects to find them together - an EMPAD ``.xml`` names +its ``.raw`` by bare file name and opens it next to itself, so +``MyDataset/acquisition_12.xml`` and ``MyDataset/scan_x256_y256.raw`` both keeps +the pairing and keeps it clear of identically named members elsewhere. +``delete()`` removes them all, and the size shown in the catalogue counts them. These entries are written by hand. The two fields CI otherwise fills in would be -taken from the archive rather than from the member, so a pull request leaving -them blank is refused rather than guessed at. ``download_url``, and the download +taken from the archive rather than from the member, so a pull request leaving any +member's blank is refused rather than guessed at. ``download_url``, and the download link on the docs site, point at the archive: the member has no link of its own. A host that ignores ``Range`` and answers with the whole archive fails loudly, @@ -203,9 +208,9 @@ a dataset with one, fail as well. files. It downloads the file behind each changed entry that is missing its ``checksum`` or ``size_bytes`` - a weights family's ``latest`` and each dated version on their own links - fills the fields in and pushes the result back to -the branch. An entry naming an ``archive`` is refused instead: its ``checksum`` -and ``size_bytes`` describe the member inside the zip, and the only thing there -is to download is the whole archive, so those two are filled in by hand. A +the branch. An entry naming an ``archive`` is refused instead: each member's +``checksum`` and ``size_bytes`` describe that file inside the zip, and the only +thing there is to download is the whole archive, so they are filled in by hand. A fork's branch cannot be pushed to, so a pull request from one fails instead and prints the values to paste in. An entry coming in through the issue form is filled in the same way before its pull request is opened, so it diff --git a/emdatabase/_archive.py b/emdatabase/_archive.py index 3af94f7..95de3a1 100644 --- a/emdatabase/_archive.py +++ b/emdatabase/_archive.py @@ -181,25 +181,29 @@ class ArchiveMemberDownloader: def __init__( self, member: str, - progressbar: Progress | bool = False, + progressbar: Progress | None = None, chunk_size: int = 4096, ) -> None: self.member = member - # `True` means "build your own bar", which pooch's HTTPDownloader does - # and this does not: _retrieve has already swapped it for a Progress. - self.progressbar = None if isinstance(progressbar, bool) else progressbar + # pooch's `progressbar=True` - "build your own bar", which this does not + # do - stops at _retrieve, which has already turned it into a Progress + # or into nothing. + self.progressbar = progressbar self.chunk_size = chunk_size def __call__(self, url: str, output_file: str, _pooch: Any = None) -> None: with io.BufferedReader(_HTTPRangeFile(url), buffer_size=_BUFFER_SIZE) as stream: # pyright: ignore[reportArgumentType] if _is_sevenzip(url): - total = self._from_7z(stream, url, output_file) + self._from_7z(stream, url, output_file) else: - total = self._from_zip(stream, url, output_file) + self._from_zip(stream, url, output_file) bar = self.progressbar if bar: + # The member's size is whatever the reader set the bar to; a second + # copy of it threaded back through the return value could only + # disagree with what the bar is actually showing. bar.reset() - bar.update(total) + bar.update(bar.total) bar.close() def _missing(self, url: str) -> str: @@ -208,7 +212,7 @@ def _missing(self, url: str) -> str: "the archive: one file name can appear in several of its directories." ) - def _from_zip(self, stream: IO[bytes], url: str, output_file: str) -> int: + def _from_zip(self, stream: IO[bytes], url: str, output_file: str) -> None: """Stream one zip member out; each is compressed on its own.""" bar = self.progressbar with zipfile.ZipFile(stream) as archive: @@ -223,9 +227,8 @@ def _from_zip(self, stream: IO[bytes], url: str, output_file: str) -> int: out.write(chunk) if bar: bar.update(len(chunk)) - return info.file_size - def _from_7z(self, stream: IO[bytes], url: str, output_file: str) -> int: + def _from_7z(self, stream: IO[bytes], url: str, output_file: str) -> None: """Extract one 7z member, which py7zr will only write to a directory.""" bar = self.progressbar with py7zr.SevenZipFile(stream) as archive: @@ -249,4 +252,3 @@ def _from_7z(self, stream: IO[bytes], url: str, output_file: str) -> int: callback=_SevenZipProgress(bar) if bar else None, ) os.replace(Path(scratch) / wanted.filename, output_file) - return wanted.uncompressed diff --git a/emdatabase/catalogue.py b/emdatabase/catalogue.py index a5bf394..d6187a1 100644 --- a/emdatabase/catalogue.py +++ b/emdatabase/catalogue.py @@ -115,7 +115,7 @@ def entry(name: str, ds: DownloadableDataset) -> dict: "url": ds.download_url, # For an entry fetched out of a zip, `url` is the archive rather than # the file; this is the member inside it, and "" for everything else. - "archive": md.archive.member if md.archive else "", + "archive": md.archive.members[0].member if md.archive and md.archive.members else "", "latest_checksum": ds.checksum or "", "versions": _versions(ds), "model_class": md.model.class_ if md.model else "", diff --git a/emdatabase/downloadable_dataset.py b/emdatabase/downloadable_dataset.py index d2b8a12..695c283 100644 --- a/emdatabase/downloadable_dataset.py +++ b/emdatabase/downloadable_dataset.py @@ -15,7 +15,7 @@ from emdatabase.config import LocationName from emdatabase.metadata import ( - ArchiveCompanion, + ArchiveMember, DatasetMetadata, WeightsVersion, versioned_filename, @@ -120,7 +120,7 @@ class _Resolved: file: str pinned: bool member: str | None = None - companions: tuple[ArchiveCompanion, ...] = () + companions: tuple[ArchiveMember, ...] = () class Progress(Protocol): @@ -219,30 +219,21 @@ def _settle_pending(key: str, future: "Future[Path]") -> None: "Using the path raises this error; call download() again to retry.", DownloadFailedWarning, ) + # The entry is kept for the life of the process, and a traceback holds + # every frame of the download and its locals - for an archive that is + # the downloader, its range file and its read buffers. What the handle + # owes its caller is which error and what it said, both of which the + # exception still carries; re-raising it gets a fresh traceback from + # the raise itself. + error.__traceback__ = None + error.__context__ = None + error.__cause__ = None return with _PENDING_LOCK: if _PENDING.get(key) is future: del _PENDING[key] -def _discard_settled(key: str) -> None: - """Drop a kept failure once the file has arrived by some other route. - - :func:`_settle_pending` keeps a failed download's entry so that it goes on - being reported, but the entry is keyed by the path alone and so outlives the - attempt it belongs to: without this, a blocking download that succeeds after - a background one failed would hand back a handle that still raises the old - error - and reprs as ``[failed]`` - for a file that is now on disk. - - A download still running is left alone. It is writing the same file, and - dropping it would let its handle report itself done before the bytes are. - """ - with _PENDING_LOCK: - future = _PENDING.get(key) - if future is not None and future.done(): - del _PENDING[key] - - class DatasetPath(_ConcretePath): """The local path to a dataset, which may still be downloading. @@ -438,14 +429,25 @@ def _resolve(self, version: str | None = None) -> _Resolved: if md.kind != "weights": if version is not None: raise ValueError(f"{type(self).__name__} is a dataset and has no versions") + if md.archive is not None: + # The archive is the only thing there is a link to, and every + # file the entry puts on disk is a member of it: the first is + # the one handed back, the rest arrive beside it. + return _Resolved( + url=md.archive.url, + checksum=md.checksum, + size_bytes=md.size_bytes, + file=md.file, + pinned=True, + member=md.archive.members[0].member, + companions=md.archive.members[1:], + ) return _Resolved( - url=md.archive.url if md.archive else (md.url or f"{md.source}/{md.file}"), + url=md.url or f"{md.source}/{md.file}", checksum=md.checksum, size_bytes=md.size_bytes, file=md.file, pinned=True, - member=md.archive.member if md.archive else None, - companions=tuple(md.archive.companions) if md.archive else (), ) if version is None: if md.latest is None: @@ -633,10 +635,13 @@ def download( """ if not background: path = self._retrieve(destination, progressbar, chunk_size, version, refresh) - # The bytes are here, so an earlier failure kept for this path no - # longer describes it and must not be re-raised by the handle. - _discard_settled(_pending_key(path)) - return DatasetPath(path) + # Attached like any other download, so that "the newest attempt owns + # the entry" stays one rule rather than two: _attach replaces a + # failure kept for this path, and the done-callback - which fires + # immediately for a future that is already settled - clears it. + settled: Future[Path] = Future() + settled.set_result(Path(path)) + return DatasetPath(path)._attach(settled) # Where the file will end up: the copy the search order finds, unless a # destination or a refresh asks for a fresh one. existing = None if destination is not None or refresh else self.filepath(version) @@ -681,14 +686,23 @@ def _retrieve( if shared is not None: return shared destination = self._resolve_destination(destination) + pin = newer or resolved # the newer link on main, if there is one + # pooch's `progressbar=True` - "build your own bar" - has already become + # a _TqdmProgress above; what a downloader takes is a Progress or nothing. + bar = None if isinstance(progressbar, bool) else progressbar downloader: Any - alongside: list[tuple[ArchiveCompanion, Any]] = [] + # Every file this download puts on disk, the entry's own first. A + # companion is a download of its own - its own hash, its own atomic + # write, and skipped outright if it is already there - differing only in + # which member of the archive the bytes come from. + parts: list[tuple[str, str | None, Any]] if resolved.member is None: downloader = pooch.HTTPDownloader( progressbar=progressbar, # pyright: ignore[reportArgumentType] chunk_size=chunk_size, headers={"User-Agent": USER_AGENT}, ) + parts = [(resolved.file, pin.checksum, downloader)] else: # Imported here, not at the top, because _archive imports this # module for USER_AGENT and the progress protocol - and importing it @@ -696,39 +710,34 @@ def _retrieve( # pay for. from emdatabase._archive import ArchiveMemberDownloader - downloader = ArchiveMemberDownloader(resolved.member, progressbar, chunk_size) - alongside = [ - (c, ArchiveMemberDownloader(c.member, progressbar, chunk_size)) - for c in resolved.companions + downloader = ArchiveMemberDownloader(resolved.member, bar, chunk_size) + parts = [ + (resolved.file, pin.checksum, downloader), + *( + (c.file, c.checksum, ArchiveMemberDownloader(c.member, bar, chunk_size)) + for c in resolved.companions + ), ] try: if refresh: # pooch keeps a file whose hash it was not given anything to # check against, so the copy has to go before it will re-fetch. - (destination / resolved.file).unlink(missing_ok=True) + for name, _, _ in parts: + (destination / name).unlink(missing_ok=True) if newer is None and not resolved.pinned: filepath = self._retrieve_latest(resolved, destination, downloader) else: - pin = newer or resolved # the newer link on main, if there is one - filepath = pooch.retrieve( - url=pin.url, - known_hash=pin.checksum, - fname=resolved.file, - path=destination, - downloader=downloader, # pyright: ignore[reportArgumentType] - ) - # Each companion is a download of its own: its own hash, its own - # atomic write, and skipped outright if it is already there. - for companion, companion_downloader in alongside: - if refresh: - (destination / companion.file).unlink(missing_ok=True) + written = [ pooch.retrieve( - url=resolved.url, - known_hash=companion.checksum, - fname=companion.file, + url=pin.url, + known_hash=checksum, + fname=name, path=destination, - downloader=companion_downloader, + downloader=part, # pyright: ignore[reportArgumentType] ) + for name, checksum, part in parts + ] + filepath = written[0] # parts[0] is the entry's own file finally: # pooch only closes the bar on the happy path, so a failed or # cancelled download would leave it hanging open. @@ -810,7 +819,7 @@ def _retrieve_latest(self, resolved: _Resolved, destination: Path, downloader: A ) return filepath - def _is_complete(self, directory: Path, version: str | None = None) -> bool: + def _is_complete(self, directory: Path, resolved: _Resolved) -> bool: """Whether ``directory`` holds the whole dataset, companions included. An entry with companions is not on disk until they are: its own file is @@ -818,21 +827,25 @@ def _is_complete(self, directory: Path, version: str | None = None) -> bool: and a header whose raw is missing fails in the reader rather than here. A directory holding only part of the set is passed over, so the rest is fetched instead of the dataset reporting itself already downloaded. + + Takes the resolved entry rather than a version: the answer differs per + directory, but what is being looked for does not, and the search order + asks about several directories in a row. """ - if not (directory / self.filename(version)).exists(): + if not (directory / resolved.file).exists(): return False - return all((directory / c.file).exists() for c in self._resolve(version).companions) + return all((directory / c.file).exists() for c in resolved.companions) def _find_in_shared_locations(self, version: str | None = None) -> Path | None: """Path to an existing, complete copy in a shared location, or None.""" from emdatabase import config - name = self.filename(version) + resolved = self._resolve(version) return next( ( - location.path / name + location.path / resolved.file for location in config.locations() - if location.kind != "personal" and self._is_complete(location.path, version) + if location.kind != "personal" and self._is_complete(location.path, resolved) ), None, ) @@ -855,8 +868,10 @@ def filepaths(self, version: str | None = None) -> list[Path]: """ from emdatabase import config - name = self.filename(version) - return [d / name for d in config.data_search_dirs() if self._is_complete(d, version)] + resolved = self._resolve(version) + return [ + d / resolved.file for d in config.data_search_dirs() if self._is_complete(d, resolved) + ] def filepath(self, version: str | None = None) -> Path | None: """Return the local file path of the dataset if present. @@ -897,9 +912,10 @@ def delete( # Companions go too. They are usually the large ones - a header is what # the entry points at, and the gigabytes sit beside it - so leaving them # behind would make delete() look like it had freed the space. - for companion in self._resolve(version).companions: + resolved = self._resolve(version) + for companion in resolved.companions: (directory / companion.file).unlink(missing_ok=True) - path = directory / self.filename(version) + path = directory / resolved.file if path.exists(): path.unlink() return True diff --git a/emdatabase/index/MOSS6Fig3.yaml b/emdatabase/index/MOSS6Fig3.yaml index c2049da..4a0fcd6 100644 --- a/emdatabase/index/MOSS6Fig3.yaml +++ b/emdatabase/index/MOSS6Fig3.yaml @@ -16,15 +16,15 @@ MOSS6Fig3: files are members of the record's 15.3 GB RawData.7z and are fetched out of it directly, without downloading the archive; the raw costs about 2.6 GB of it. source: https://zenodo.org/records/13958144/files - checksum: md5:6f9a865545655a2d53bf887afa98c135 - file: MOSS6Fig3/acquisition_12.xml - size_bytes: 3953 archive: url: https://zenodo.org/records/13958144/files/RawData.7z - member: RawData/ExperimentalData_MOSS-6_Fig.3/acquisition_12.xml checksum: md5:97581a943b20f2b6ba5d636b859d5f10 size_bytes: 15266871965 - companions: + members: + - member: RawData/ExperimentalData_MOSS-6_Fig.3/acquisition_12.xml + file: MOSS6Fig3/acquisition_12.xml + checksum: md5:6f9a865545655a2d53bf887afa98c135 + size_bytes: 3953 - member: RawData/ExperimentalData_MOSS-6_Fig.3/scan_x256_y256.raw file: MOSS6Fig3/scan_x256_y256.raw size_bytes: 4362076160 diff --git a/emdatabase/index/TEMPLATE.yaml b/emdatabase/index/TEMPLATE.yaml index d35378b..fd97455 100644 --- a/emdatabase/index/TEMPLATE.yaml +++ b/emdatabase/index/TEMPLATE.yaml @@ -33,17 +33,24 @@ MyDatasetName: # appear in several of its directories. Not allowed on a `kind: weights` entry. # A 7z compresses files together in solid blocks, so reaching a member means # decompressing its block from the start: measure the cost before adding one. + # For a file that lives inside a .zip or .7z on someone else's record. The + # entry's own `file`, `checksum` and `size_bytes` above are left out: they are + # the first member's, stated once, here. `checksum` and `size_bytes` directly + # under `archive` describe the archive itself and are optional. # archive: # url: https://zenodo.org/records/0000000/files/Figures.zip - # member: Figure_01/Panel_a/scan_x128_y128.raw # checksum: md5:0123456789abcdef0123456789abcdef # size_bytes: 2000000 - # # For data that is more than one file. Each companion comes out of the same - # # archive and is saved at its own `file`; `download()` still hands back the - # # entry's own `file` above. Give both a directory of their own when a reader - # # expects to find them side by side - an EMPAD `.xml` names its `.raw` by - # # bare file name and opens it next to itself. - # companions: + # # Every file this entry puts on disk, described the same way. The first is + # # the one `download()` hands back; any others arrive beside it, which is + # # what data that is more than one file needs. Give them a directory of + # # their own when a reader expects to find them side by side - an EMPAD + # # `.xml` names its `.raw` by bare file name and opens it next to itself. + # members: + # - member: Figure_01/Panel_a/acquisition_12.xml + # file: MyDatasetName/acquisition_12.xml + # checksum: md5:0123456789abcdef0123456789abcdef + # size_bytes: 4000 # - member: Figure_01/Panel_a/scan_x128_y128.raw # file: MyDatasetName/scan_x128_y128.raw # checksum: md5:0123456789abcdef0123456789abcdef diff --git a/emdatabase/index/TwistedBilayerWSe2Themis.yaml b/emdatabase/index/TwistedBilayerWSe2Themis.yaml index 7f62bc9..7981746 100644 --- a/emdatabase/index/TwistedBilayerWSe2Themis.yaml +++ b/emdatabase/index/TwistedBilayerWSe2Themis.yaml @@ -12,14 +12,15 @@ TwistedBilayerWSe2Themis: archive holds a Talos acquisition of the same scan under Fig_01/Panel_c-d_Talos/, so the member path matters. source: https://zenodo.org/records/10431683/files - checksum: md5:6de543cc5026f3e6b8f34a2189cb2fb4 - file: TwistedBilayerWSe2_Themis_x128_y128.raw - size_bytes: 1090519040 archive: url: https://zenodo.org/records/10431683/files/Fig_01.zip - member: Fig_01/Panel_g-h_Themis/scan_x128_y128.raw checksum: md5:6b340d06ddf33415e402f96bb0ff17ff size_bytes: 1908605317 + members: + - member: Fig_01/Panel_g-h_Themis/scan_x128_y128.raw + file: TwistedBilayerWSe2_Themis_x128_y128.raw + checksum: md5:6de543cc5026f3e6b8f34a2189cb2fb4 + size_bytes: 1090519040 detector_manufacturer: Thermo Fisher Scientific detector: EMPAD microscope_vendor: Thermo Fisher Scientific diff --git a/emdatabase/index/json-schema.json b/emdatabase/index/json-schema.json index 54e8790..9751b82 100644 --- a/emdatabase/index/json-schema.json +++ b/emdatabase/index/json-schema.json @@ -31,7 +31,7 @@ "minimum": 0 }, "archive": { - "$ref": "#/$defs/archiveMember" + "$ref": "#/$defs/archive" }, "detector_manufacturer": { "type": "string" @@ -156,10 +156,48 @@ }, "required": [ "description", - "source", - "file" + "source" ], "additionalProperties": false, + "allOf": [ + { + "if": { + "required": [ + "archive" + ] + }, + "then": { + "allOf": [ + { + "not": { + "required": [ + "file" + ] + } + }, + { + "not": { + "required": [ + "checksum" + ] + } + }, + { + "not": { + "required": [ + "size_bytes" + ] + } + } + ] + }, + "else": { + "required": [ + "file" + ] + } + } + ], "if": { "properties": { "kind": { @@ -258,17 +296,13 @@ ], "additionalProperties": false }, - "archiveMember": { + "archive": { "type": "object", "properties": { "url": { "type": "string", "format": "uri" }, - "member": { - "type": "string", - "minLength": 1 - }, "checksum": { "type": "string", "pattern": "^md5:[A-Fa-f0-9]{32}$" @@ -277,21 +311,21 @@ "type": "integer", "minimum": 0 }, - "companions": { + "members": { "type": "array", "minItems": 1, "items": { - "$ref": "#/$defs/archiveCompanion" + "$ref": "#/$defs/archiveMember" } } }, "required": [ "url", - "member" + "members" ], "additionalProperties": false }, - "archiveCompanion": { + "archiveMember": { "type": "object", "properties": { "member": { diff --git a/emdatabase/metadata.py b/emdatabase/metadata.py index 68f8395..edb8e43 100644 --- a/emdatabase/metadata.py +++ b/emdatabase/metadata.py @@ -12,10 +12,10 @@ :class:`WeightsVersion`, and :func:`versioned_filename` is what a dated copy is called on disk. -An entry whose data is one file inside a zip on a record that cannot be -re-published names that archive as an :class:`ArchiveMember`. The entry's own -``file``, ``checksum`` and ``size_bytes`` go on describing the member, which is -all a caller ever receives. +An entry whose data lives inside a zip on a record that cannot be re-published +names that archive as an :class:`Archive`, and every file taken out of it as an +:class:`ArchiveMember`. The entry's own ``file`` and ``checksum`` are the first +member's, which is what a caller receives; its ``size_bytes`` counts them all. This module also owns the small amount of shared knowledge about where the dataset files live - :func:`dataset_files`, :func:`index_entries`, @@ -321,14 +321,15 @@ class WeightsVersion: @dataclass(frozen=True) -class ArchiveCompanion: - """A second file the entry's own file cannot be read without. - - An EMPAD acquisition is the plain case: the ``.xml`` is what a reader is - pointed at, but it names its ``.raw`` by bare file name and expects to find - it in the same directory. Both come out of the same archive, and ``file`` - gives each its name on disk - including a directory, which is how the two - are kept together and away from identically named members elsewhere. +class ArchiveMember: + """One file inside an archive, and what it is called once it is out. + + ``member`` is the complete path inside the archive: one file name can appear + in several of its directories, so a basename alone is ambiguous. ``file`` is + the name on disk - including a directory, which is how the files of one + dataset are kept together and away from identically named members + elsewhere. ``checksum`` and ``size_bytes`` describe this file, never the + archive it is taken out of. """ member: str @@ -338,29 +339,28 @@ class ArchiveCompanion: @dataclass(frozen=True) -class ArchiveMember: - """One file inside an archive on someone else's record. - - An entry names this when the data worth shipping is a single member of a - multi-gigabyte archive that cannot be re-published. ``download()`` fetches - only that member, with HTTP range requests; the entry's ``file``, - ``checksum`` and ``size_bytes`` go on describing the file the user ends up - with, and the two here describe the archive they come out of. - - ``member`` is the complete path inside the archive: one file name can appear - in several of its directories. A ``.zip`` and a ``.7z`` are both read, told - apart by the link's own file name. - - ``companions`` names further members fetched alongside, for data that is more - than one file; see :class:`ArchiveCompanion`. ``download()`` still hands back - the entry's own ``file``. +class Archive: + """An archive on someone else's record, and the files taken out of it. + + An entry names this when the data worth shipping sits inside a + multi-gigabyte ``.zip`` or ``.7z`` that cannot be re-published. + ``download()`` fetches only the members named here, with HTTP range + requests, and hands back the first of them. A ``.zip`` and a ``.7z`` are + both read, told apart by the link's own file name. + + ``checksum`` and ``size_bytes`` here describe the archive. Every file the + entry puts on disk is one of ``members``, described the same way whether it + is the one handed back or one that has to arrive beside it - an EMPAD + ``.xml`` names its ``.raw`` by bare file name and expects to open it next to + itself. ``members[0]`` is the entry's own file, and is where the entry's + ``file`` and ``checksum`` come from, so that no fact about a file is stated + at two levels; the entry's ``size_bytes`` is every member's added up. """ url: str - member: str checksum: str | None = None size_bytes: int | None = None - companions: tuple[ArchiveCompanion, ...] = () + members: tuple[ArchiveMember, ...] = () @dataclass(frozen=True, repr=False) @@ -382,7 +382,7 @@ class DatasetMetadata: url: str | None = None checksum: str | None = None size_bytes: int | None = None - archive: ArchiveMember | None = None + archive: Archive | None = None detector_manufacturer: str | None = None detector: str | None = None microscope_vendor: str | None = None @@ -433,10 +433,23 @@ def from_spec( ) if values.get("archive") is not None: archive = dict(values["archive"]) - archive["companions"] = tuple( - ArchiveCompanion(**c) for c in archive.get("companions") or () + archive["members"] = tuple( + ArchiveMember(**m) for m in archive.get("members") or () ) - values["archive"] = ArchiveMember(**archive) + values["archive"] = Archive(**archive) + # The entry's own file is the first member, and is described + # there rather than at the top level, so that its name and hash + # are not stated in two places that can disagree. `size_bytes` + # counts every member: one number for how much disk the dataset + # takes, which is what anyone asking a size is asking. + members = values["archive"].members + if members: + values.setdefault("file", members[0].file) + values.setdefault("checksum", members[0].checksum) + sizes = [m.size_bytes for m in members] + values.setdefault( + "size_bytes", None if sizes[0] is None else sum(s or 0 for s in sizes) + ) if values.get("latest") is not None: values["latest"] = WeightsVersion(**values["latest"]) values["versions"] = { @@ -447,23 +460,10 @@ def from_spec( except TypeError as error: raise TypeError(f"{_where(origin)}: {error}") from None - @property - def total_bytes(self) -> int | None: - """Every byte the entry puts on disk, companions included, or None. - - :attr:`size_bytes` describes the entry's own file. For a multi-member - entry that is the small one - an EMPAD header beside gigabytes of raw - - so it is this, not that, which answers "how big is this dataset". - """ - if self.size_bytes is None: - return None - companions = self.archive.companions if self.archive else () - return self.size_bytes + sum(c.size_bytes or 0 for c in companions) - @property def size(self) -> str: - """:attr:`total_bytes` formatted for display, or ``""`` if unknown.""" - return format_size(self.total_bytes) + """:attr:`size_bytes` formatted for display, or ``""`` if unknown.""" + return format_size(self.size_bytes) @property def headline(self) -> str: @@ -497,8 +497,10 @@ def __str__(self) -> str: elif entry.name == "model": value = " · ".join(p for p in (value.class_, value.framework, value.quantem) if p) elif entry.name == "archive": - value = f"{value.member} in {value.url}" + ( - f" (+{len(value.companions)} alongside)" if value.companions else "" + beside = len(value.members) - 1 + first = value.members[0].member if value.members else "?" + value = f"{first} in {value.url}" + ( + f" (+{beside} alongside)" if beside > 0 else "" ) elif entry.name == "latest": value = " · ".join(p for p in (value.checksum, value.url) if p) diff --git a/emdatabase/new_dataset.py b/emdatabase/new_dataset.py index d6b265e..f606c5e 100644 --- a/emdatabase/new_dataset.py +++ b/emdatabase/new_dataset.py @@ -353,14 +353,16 @@ def fill_download_fields(document: dict[str, Any]) -> list[str]: if pin: lines += _fill_pin(label, pin, pin.get("url", "")) elif entry.get("archive"): - # The entry's checksum and size_bytes describe the member inside the - # archive. Following the link here would hash the whole zip and - # write a value that is wrong in a way nothing downstream notices. - if not (entry.get("checksum") and entry.get("size_bytes")): + # Each member's checksum and size_bytes describe that file inside + # the archive. Following the link here would hash the whole archive + # and write a value that is wrong in a way nothing downstream + # notices. + members = entry["archive"].get("members") or () + if not members or not all(m.get("checksum") and m.get("size_bytes") for m in members): raise ValueError( - f"{name}: checksum and size_bytes describe the member inside the " - "archive, which is not something this can download. Fill them in " - "by hand." + f"{name}: each member's checksum and size_bytes describe that file " + "inside the archive, which is not something this can download. Fill " + "them in by hand." ) else: url = entry.get("url") or f"{entry.get('source', '')}/{entry.get('file', '')}" diff --git a/emdatabase/tests/test_archive.py b/emdatabase/tests/test_archive.py index 1090e3f..b25b9c0 100644 --- a/emdatabase/tests/test_archive.py +++ b/emdatabase/tests/test_archive.py @@ -22,6 +22,7 @@ from emdatabase import config from emdatabase._archive import ArchiveError from emdatabase.downloadable_dataset import DownloadableDataset +from emdatabase.metadata import format_size from emdatabase.widget import DownloadCancelled MEMBER = "Fig_01/Panel_g-h_Themis/scan.raw" @@ -47,13 +48,20 @@ def archived(http_server): (directory / "Fig_01.zip").write_bytes(_zip_bytes()) def make(**overrides): - archive = {"url": f"{base}/Fig_01.zip", "member": MEMBER, **overrides.pop("archive", {})} + member = { + "member": MEMBER, + "file": "scan.raw", + "checksum": overrides.pop("checksum", f"md5:{hashlib.md5(CONTENT).hexdigest()}"), + "size_bytes": len(CONTENT), + } + archive = { + "url": f"{base}/Fig_01.zip", + "members": [member], + **overrides.pop("archive", {}), + } spec = { "description": "One headerless raw file inside a zip.", "source": base, - "file": "scan.raw", - "checksum": f"md5:{hashlib.md5(CONTENT).hexdigest()}", - "size_bytes": len(CONTENT), "archive": archive, **overrides, } @@ -132,7 +140,11 @@ def update(self, n): def test_a_member_the_archive_does_not_hold_names_it(archived, dest): - ds = archived(archive={"member": "Fig_01/Panel_g-h_Themis/missing.raw"}) + ds = archived( + archive={ + "members": [{"member": "Fig_01/Panel_g-h_Themis/missing.raw", "file": "scan.raw"}] + } + ) with pytest.raises(KeyError, match="missing.raw"): ds.download(destination=dest, progressbar=False, background=False) @@ -163,8 +175,6 @@ def test_filepath_and_delete_behave_as_for_any_other_dataset(archived, dest): # solid blocks and can only be extracted to a directory. Both end up at the same # place, so the behaviour asserted here is deliberately the zip's. -MEMBER_7Z = "Fig_01/Panel_g-h_Themis/scan.raw" - def _sevenzip_bytes() -> bytes: """A small .7z holding the member and a decoy sharing its file name.""" @@ -175,7 +185,7 @@ def _sevenzip_bytes() -> bytes: built = work / "built.7z" with py7zr.SevenZipFile(built, "w") as archive: archive.write(work / "decoy", DECOY) - archive.write(work / "good", MEMBER_7Z) + archive.write(work / "good", MEMBER) return built.read_bytes() @@ -186,13 +196,20 @@ def archived_7z(http_server): (directory / "Fig_01.7z").write_bytes(_sevenzip_bytes()) def make(**overrides): - archive = {"url": f"{base}/Fig_01.7z", "member": MEMBER_7Z, **overrides.pop("archive", {})} + member = { + "member": MEMBER, + "file": "scan.raw", + "checksum": overrides.pop("checksum", f"md5:{hashlib.md5(CONTENT).hexdigest()}"), + "size_bytes": len(CONTENT), + } + archive = { + "url": f"{base}/Fig_01.7z", + "members": [member], + **overrides.pop("archive", {}), + } spec = { "description": "One headerless raw file inside a 7z.", "source": base, - "file": "scan.raw", - "checksum": f"md5:{hashlib.md5(CONTENT).hexdigest()}", - "size_bytes": len(CONTENT), "archive": archive, **overrides, } @@ -208,7 +225,11 @@ def test_a_7z_member_comes_out_byte_for_byte(archived_7z, dest): def test_a_7z_member_the_archive_does_not_hold_names_it(archived_7z, dest): """py7zr extracts nothing and raises nothing for a name it lacks, so we check.""" - ds = archived_7z(archive={"member": "Fig_01/Panel_g-h_Themis/missing.raw"}) + ds = archived_7z( + archive={ + "members": [{"member": "Fig_01/Panel_g-h_Themis/missing.raw", "file": "scan.raw"}] + } + ) with pytest.raises(KeyError, match="missing.raw"): ds.download(destination=dest, progressbar=False, background=False) assert list(dest.iterdir()) == [] @@ -277,19 +298,21 @@ def make(**overrides): spec = { "description": "A header and the file it names.", "source": base, - "file": "Paired/scan.raw", - "checksum": f"md5:{hashlib.md5(CONTENT).hexdigest()}", - "size_bytes": len(CONTENT), "archive": { "url": f"{base}/Paired.zip", - "member": MEMBER, - "companions": [ + "members": [ + { + "member": MEMBER, + "file": "Paired/scan.raw", + "checksum": f"md5:{hashlib.md5(CONTENT).hexdigest()}", + "size_bytes": len(CONTENT), + }, { "member": COMPANION, "file": "Paired/scan_x256_y256.raw", "checksum": f"md5:{hashlib.md5(COMPANION_CONTENT).hexdigest()}", "size_bytes": len(COMPANION_CONTENT), - } + }, ], }, **overrides, @@ -314,13 +337,17 @@ def test_a_companion_with_a_wrong_checksum_fails(paired, dest): ds = paired( archive={ "url": paired().metadata.archive.url, - "member": MEMBER, - "companions": [ + "members": [ + { + "member": MEMBER, + "file": "Paired/scan.raw", + "checksum": f"md5:{hashlib.md5(CONTENT).hexdigest()}", + }, { "member": COMPANION, "file": "Paired/scan_x256_y256.raw", "checksum": "md5:" + "0" * 32, - } + }, ], } ) @@ -370,7 +397,7 @@ def test_a_complete_shared_copy_is_still_used_as_is(paired, tmp_path, dest): def test_the_size_shown_counts_the_companion(paired): - """`size_bytes` is the entry's own file; `size` is what the dataset occupies.""" + """One number for how much disk the dataset takes, every member counted.""" md = paired().metadata - assert md.size_bytes == len(CONTENT) - assert md.total_bytes == len(CONTENT) + len(COMPANION_CONTENT) + assert md.size_bytes == len(CONTENT) + len(COMPANION_CONTENT) + assert md.size == format_size(md.size_bytes) # the same number, formatted diff --git a/emdatabase/tests/test_fill_download_fields.py b/emdatabase/tests/test_fill_download_fields.py index 0f43397..066aa42 100644 --- a/emdatabase/tests/test_fill_download_fields.py +++ b/emdatabase/tests/test_fill_download_fields.py @@ -25,7 +25,22 @@ CONTENT = b"a small 4D-STEM dataset, allegedly" * 100 MD5 = f"md5:{hashlib.md5(CONTENT).hexdigest()}" # Deliberately unreachable: an archive entry must never be downloaded here. -ARCHIVE = {"url": "https://example.invalid/Figures.zip", "member": "Fig/Panel/scan.raw"} +ARCHIVE = { + "url": "https://example.invalid/Figures.zip", + "members": [{"member": "Fig/Panel/scan.raw", "file": FILE}], +} +# The same archive with every member described, which is what CI must leave alone. +FILLED_ARCHIVE = { + "url": "https://example.invalid/Figures.zip", + "members": [ + { + "member": "Fig/Panel/scan.raw", + "file": FILE, + "checksum": MD5, + "size_bytes": len(CONTENT), + } + ], +} @pytest.fixture(scope="module") @@ -50,16 +65,17 @@ def index(http_server, tmp_path): directory.mkdir() def write(**extra): - document = { - "MyData": { - "description": "A 4D-STEM dataset of something.", - "source": base, - "file": FILE, - "license": "CC-BY-4.0", - "technique": ["4D-STEM"], - **extra, - } + entry = { + "description": "A 4D-STEM dataset of something.", + "source": base, + "file": FILE, + "license": "CC-BY-4.0", + "technique": ["4D-STEM"], + **extra, } + # None drops a key, which is how an archive entry leaves out the + # top-level `file`: for those it is the first member's. + document = {"MyData": {key: value for key, value in entry.items() if value is not None}} path = directory / "MyData.yaml" path.write_text(yaml.safe_dump(document, sort_keys=False), encoding="utf-8") return path @@ -90,7 +106,7 @@ def test_a_blank_entry_is_filled_in_and_written(script, index, tmp_path): def test_an_archive_entry_with_blank_fields_is_refused(script, index, tmp_path): """Following the link would hash the whole zip, not the member inside it.""" _, directory, write = index - path = write(archive=ARCHIVE) + path = write(file=None, archive=ARCHIVE) before = path.read_text(encoding="utf-8") code, summary = _run(script, directory, tmp_path) @@ -102,7 +118,7 @@ def test_an_archive_entry_with_blank_fields_is_refused(script, index, tmp_path): def test_a_complete_archive_entry_is_not_downloaded(script, index, tmp_path): """The archive URL does not resolve, so getting here at all means it was left alone.""" _, directory, write = index - path = write(checksum=MD5, size_bytes=len(CONTENT), archive=ARCHIVE) + path = write(file=None, archive=FILLED_ARCHIVE) before = path.read_text(encoding="utf-8") code, _ = _run(script, directory, tmp_path) diff --git a/emdatabase/tests/test_load_data.py b/emdatabase/tests/test_load_data.py index 4a95e6d..60790e5 100644 --- a/emdatabase/tests/test_load_data.py +++ b/emdatabase/tests/test_load_data.py @@ -24,7 +24,7 @@ import pytest import emdatabase.data as data -from emdatabase._archive import _HTTPRangeFile, _is_sevenzip +from emdatabase._archive import _BUFFER_SIZE, _HTTPRangeFile, _is_sevenzip from emdatabase.data import MgONanoCrystals, NiEBSDLarge from emdatabase.downloadable_dataset import ( _PENDING, @@ -80,7 +80,7 @@ def _archive_member_size(url, member): reason an entry names a member in the first place. A zip and a 7z report that size through different objects, so the size itself is what comes back. """ - with io.BufferedReader(_HTTPRangeFile(url), buffer_size=1 << 20) as stream: + with io.BufferedReader(_HTTPRangeFile(url), buffer_size=_BUFFER_SIZE) as stream: if _is_sevenzip(url): with py7zr.SevenZipFile(stream) as archive: found = [f for f in archive.list() if f.filename.replace("\\", "/") == member] @@ -130,16 +130,17 @@ def test_source_url_resolves(name, version): ) if archive is not None: # What rots for an archive entry is a member being renamed or moved - # inside an archive whose own size never changes. Companions are checked - # too: they are usually the large ones, and nothing else would notice. - members = [(archive.member, resolved.size_bytes)] - members += [(c.member, c.size_bytes) for c in archive.companions] - for member, declared in members: - if declared is None: + # inside an archive whose own size never changes. Every member is + # checked: the ones beside the entry's own file are usually the large + # ones, and nothing else would notice. + members = [(m.member, m.size_bytes) for m in archive.members] + for member, member_declared in members: + if member_declared is None: continue size = _archive_member_size(url, member) - assert size == declared, ( - f"{name}: {member} is {size} bytes inside {url}, but the YAML declares {declared}" + assert size == member_declared, ( + f"{name}: {member} is {size} bytes inside {url}, but the YAML declares " + f"{member_declared}" ) diff --git a/emdatabase/tests/test_metadata.py b/emdatabase/tests/test_metadata.py index 982ee1e..89a42f3 100644 --- a/emdatabase/tests/test_metadata.py +++ b/emdatabase/tests/test_metadata.py @@ -13,7 +13,7 @@ from emdatabase.metadata import ( TEMPLATE_PATH, - ArchiveCompanion, + Archive, ArchiveMember, Author, DatasetMetadata, @@ -91,11 +91,23 @@ def test_schema_and_dataclass_agree(): assert list(ENTRY_SCHEMA["properties"]) == [ f.name for f in dataclasses.fields(DatasetMetadata) ] - assert set(ENTRY_SCHEMA["required"]) == { + no_default = { f.name for f in dataclasses.fields(DatasetMetadata) if f.default is dataclasses.MISSING and f.default_factory is dataclasses.MISSING } + # Every entry needs a `file` too, but an archive entry leaves it out - it is + # the first member's - so it is required in the branch for entries without + # an archive rather than unconditionally at the top. + assert set(ENTRY_SCHEMA["required"]) == no_default - {"file"} + branch = ENTRY_SCHEMA["allOf"][0] + assert branch["if"] == {"required": ["archive"]} + assert branch["else"] == {"required": ["file"]} + assert branch["then"]["allOf"] == [ + {"not": {"required": ["file"]}}, + {"not": {"required": ["checksum"]}}, + {"not": {"required": ["size_bytes"]}}, + ] def test_author_schema_and_dataclass_agree(): @@ -128,25 +140,53 @@ def test_weights_file_schema_and_dataclass_agree(): assert ENTRY_SCHEMA["properties"]["latest"] == {"$ref": "#/$defs/weightsFile"} +def test_archive_schema_and_dataclass_agree(): + """The archive itself: a link, what it hashes to, and the files inside it.""" + archive = SCHEMA["$defs"]["archive"] + assert list(archive["properties"]) == [f.name for f in dataclasses.fields(Archive)] + assert archive["required"] == ["url", "members"] + assert ENTRY_SCHEMA["properties"]["archive"] == {"$ref": "#/$defs/archive"} + + def test_archive_member_schema_and_dataclass_agree(): - """The archive a member is fetched out of is the same four fields, in two files.""" - archive = SCHEMA["$defs"]["archiveMember"] - assert list(archive["properties"]) == [f.name for f in dataclasses.fields(ArchiveMember)] - assert archive["required"] == ["url", "member"] - assert ENTRY_SCHEMA["properties"]["archive"] == {"$ref": "#/$defs/archiveMember"} - - -def test_archive_companion_schema_and_dataclass_agree(): - """A companion comes out of the same archive and is saved beside the entry's file.""" - companion = SCHEMA["$defs"]["archiveCompanion"] - fields = [f.name for f in dataclasses.fields(ArchiveCompanion)] - assert list(companion["properties"]) == fields - assert companion["required"] == ["member", "file"] - assert SCHEMA["$defs"]["archiveMember"]["properties"]["companions"]["items"] == { - "$ref": "#/$defs/archiveCompanion" + """Every file the entry puts on disk is a member, described the same way.""" + member = SCHEMA["$defs"]["archiveMember"] + assert list(member["properties"]) == [f.name for f in dataclasses.fields(ArchiveMember)] + assert member["required"] == ["member", "file"] + assert SCHEMA["$defs"]["archive"]["properties"]["members"]["items"] == { + "$ref": "#/$defs/archiveMember" } +def test_an_archive_entry_takes_its_file_from_the_first_member(): + """Stated once, in the member, rather than at two levels that can disagree.""" + spec = { + "description": "A header and the raw it names.", + "source": "https://zenodo.org/records/1/files", + "archive": { + "url": "https://zenodo.org/records/1/files/Raw.zip", + "members": [ + { + "member": "Raw/acquisition_12.xml", + "file": "Paired/acquisition_12.xml", + "checksum": "md5:" + "a" * 32, + "size_bytes": 3953, + }, + { + "member": "Raw/scan.raw", + "file": "Paired/scan.raw", + "size_bytes": 4000, + }, + ], + }, + } + md = DatasetMetadata.from_spec(spec) + + assert md.file == "Paired/acquisition_12.xml" + assert md.checksum == "md5:" + "a" * 32 + assert md.size_bytes == 3953 + 4000 # every member, not just the first + + @pytest.mark.parametrize( ("file", "expected"), [("w.pt", "w_260902.pt"), ("weights", "weights_260902"), ("a.tar.gz", "a.tar_260902.gz")], diff --git a/emdatabase/tests/test_new_dataset.py b/emdatabase/tests/test_new_dataset.py index b80362f..5bcfae8 100644 --- a/emdatabase/tests/test_new_dataset.py +++ b/emdatabase/tests/test_new_dataset.py @@ -477,10 +477,9 @@ def test_build_document_keeps_the_archive_block(): entry = { "description": "One headerless raw file inside a zip.", "source": "https://zenodo.org/records/1/files", - "file": "scan.raw", "archive": { "url": "https://zenodo.org/records/1/files/Fig_01.zip", - "member": "Fig_01/Panel/scan.raw", + "members": [{"member": "Fig_01/Panel/scan.raw", "file": "scan.raw"}], }, } written = build_document("Archived", entry)["Archived"] From e4a8177a82d921308351b3456ae0ac6d5edef2e5 Mon Sep 17 00:00:00 2001 From: arthurmccray Date: Thu, 17 Sep 2026 21:10:44 -0700 Subject: [PATCH 05/12] fixing multiple download pbar titles --- emdatabase/downloadable_dataset.py | 41 +++++++++++++++++++++++------- emdatabase/tests/test_archive.py | 13 +++++++++- 2 files changed, 44 insertions(+), 10 deletions(-) diff --git a/emdatabase/downloadable_dataset.py b/emdatabase/downloadable_dataset.py index 695c283..adb68da 100644 --- a/emdatabase/downloadable_dataset.py +++ b/emdatabase/downloadable_dataset.py @@ -328,6 +328,10 @@ class _TqdmProgress: A background download calls :meth:`open` up front instead, on the calling thread. A pool thread does not carry the running cell, so a notebook bar built there is shown in whichever cell ran last, or not at all. + + A download that is more than one file drives one of these through a bar per + file, in turn: :attr:`desc` is set to each as it starts, so the label names + the file whose bytes are moving rather than the first of them. """ def __init__(self, desc: str = "") -> None: @@ -335,6 +339,19 @@ def __init__(self, desc: str = "") -> None: self._bar: Any = None # a tqdm.auto bar; which backend depends on the host self._total = 0 + @property + def desc(self) -> str: + """What the bar is labelled with, which is the file it is showing.""" + return self._desc + + @desc.setter + def desc(self, value: str) -> None: + self._desc = value + if self._bar is not None: + # set_description_str, not set_description: the latter appends a + # ": " of its own, which the constructor's `desc=` does not. + self._bar.set_description_str(value) + @property def total(self) -> int: return self._total @@ -727,16 +744,22 @@ def _retrieve( if newer is None and not resolved.pinned: filepath = self._retrieve_latest(resolved, destination, downloader) else: - written = [ - pooch.retrieve( - url=pin.url, - known_hash=checksum, - fname=name, - path=destination, - downloader=part, # pyright: ignore[reportArgumentType] + written = [] + for name, checksum, part in parts: + if isinstance(bar, _TqdmProgress): + # Each member closes the bar and the next one opens a + # fresh one, so the label has to follow. The directory + # is the same for every member and only costs width. + bar.desc = Path(name).name + written.append( + pooch.retrieve( + url=pin.url, + known_hash=checksum, + fname=name, + path=destination, + downloader=part, # pyright: ignore[reportArgumentType] + ) ) - for name, checksum, part in parts - ] filepath = written[0] # parts[0] is the entry's own file finally: # pooch only closes the bar on the happy path, so a failed or diff --git a/emdatabase/tests/test_archive.py b/emdatabase/tests/test_archive.py index b25b9c0..f6c69e5 100644 --- a/emdatabase/tests/test_archive.py +++ b/emdatabase/tests/test_archive.py @@ -21,7 +21,7 @@ from emdatabase import config from emdatabase._archive import ArchiveError -from emdatabase.downloadable_dataset import DownloadableDataset +from emdatabase.downloadable_dataset import DownloadableDataset, _TqdmProgress from emdatabase.metadata import format_size from emdatabase.widget import DownloadCancelled @@ -332,6 +332,17 @@ def test_a_companion_lands_beside_the_entrys_own_file(paired, dest): assert (folder / "scan_x256_y256.raw").read_bytes() == COMPANION_CONTENT +def test_each_member_labels_its_own_bar(paired, archived, dest): + """Both files get a bar, so the second must not carry the first one's name.""" + bar = _TqdmProgress("placeholder") + paired().download(destination=dest, progressbar=bar, background=False) + assert bar.desc == "scan_x256_y256.raw" # the file fetched last, not the header + + solo = _TqdmProgress("placeholder") + archived().download(destination=dest / "one", progressbar=solo, background=False) + assert solo.desc == "scan.raw" + + def test_a_companion_with_a_wrong_checksum_fails(paired, dest): """Each member is verified on its own, not just the one handed back.""" ds = paired( From 04daf070789ba71e1f04ffa873b391e0c1d04dc4 Mon Sep 17 00:00:00 2001 From: arthurmccray Date: Thu, 17 Sep 2026 21:22:23 -0700 Subject: [PATCH 06/12] fixing bug of zenodo updates not checking weights checksum changes --- .github/scripts/check_latest_weights.py | 31 +++++++++++++++++---- docs/source/contributing.rst | 14 ++++++---- emdatabase/index/PeakDetectionPolymers.yaml | 4 --- emdatabase/tests/test_check_weights.py | 28 +++++++++++++++++++ 4 files changed, 63 insertions(+), 14 deletions(-) diff --git a/.github/scripts/check_latest_weights.py b/.github/scripts/check_latest_weights.py index 29e29a8..268a11d 100644 --- a/.github/scripts/check_latest_weights.py +++ b/.github/scripts/check_latest_weights.py @@ -27,8 +27,10 @@ round. Those bytes are immutable and permanent, so nothing is downloaded and nothing is copied to GitHub; retraining publishes a new record under the same concept record instead, which the file URL cannot show. The script asks the -Zenodo API for the concept's newest record, and a new record id becomes a dated -version pointing at that record. +Zenodo API for the concept's newest record, and a new record serving a file +whose md5 differs becomes a dated version pointing at that record. One serving +the same md5 is not a new state of the weights - a record is versioned whole, so +anything published beside them mints a new id - so only the link moves. """ from __future__ import annotations @@ -262,8 +264,13 @@ def _check_zenodo( """Ask the Zenodo API whether this concept has a newer record. A record's files never change, so there is nothing to download and nothing - to archive: a new version of the model is a new record id, and the dated - version written for it points straight at that record. + to archive: the API's md5 is what the link serves, and a dated version + written for it points straight at that record. + + A newer record is not by itself a newer file. Zenodo versions the whole + record, so a change to anything in it publishes a new id while the file this + entry names may be byte for byte what it was; only a checksum that differs + is a new state of the weights worth a date of its own. """ try: record = fetch_json(f"{link.api}/{link.record_id}") @@ -299,10 +306,24 @@ def _check_zenodo( report.lines.append(f"- unchanged; Zenodo record {new_id} is still the latest version") return + url = link.file_url(new_id, str(served.get("key") or link.key)) + if latest.checksum and checksum == latest.checksum: + # A newer record is not a newer file. Zenodo versions the whole record, + # so anything else in it changing - a zip published beside the weights - + # is a new id while this file stays byte for byte what it was. Only the + # link moves: a dated version would pin a second date to bytes that + # already have one. + report.lines.append( + f"- unchanged; Zenodo record {new_id} is newer than {link.record_id} but serves " + f"the same `{checksum}`, so the link moved and no version was filed" + ) + entry["latest"]["url"] = url + report.changed = True + return + date = published.strftime("%y%m%d") if _date_taken(report, metadata, date, checksum): return - url = link.file_url(new_id, str(served.get("key") or link.key)) report.lines.append( f"- new Zenodo record {new_id}: the index has `{latest.checksum}` from record " f"{link.record_id}, and the new record serves `{checksum}` " diff --git a/docs/source/contributing.rst b/docs/source/contributing.rst index c504bfd..b44dc76 100644 --- a/docs/source/contributing.rst +++ b/docs/source/contributing.rst @@ -239,11 +239,15 @@ For a Zenodo record file, nothing is downloaded and nothing is copied to GitHub. The job asks the Zenodo API for the newest record of the concept the current record belongs to. If that is still the record the entry points at, the run reports it unchanged, and reports an error if the API's md5 is not the one -in the index. If a newer record has been published, the job adds a dated -version - dated by the new record's publication date - pointing at the file in -that record, and moves ``latest`` to it. The file it looks for in the new -record is the one whose name matches the current link; if the name has changed -and the record holds more than one file, the run fails rather than guess. +in the index. If a newer record has been published and it serves a different +md5, the job adds a dated version - dated by the new record's publication date - +pointing at the file in that record, and moves ``latest`` to it. A newer record +serving the same md5 only moves ``latest``: a record is versioned whole, so +publishing anything beside the weights - a zip of the training data, say - makes +a new id while the weights stand still, and a second date for the same bytes +would claim they had changed. The file it looks for in the new record is the one +whose name matches the current link; if the name has changed and the record +holds more than one file, the run fails rather than guess. Removing an entry ----------------- diff --git a/emdatabase/index/PeakDetectionPolymers.yaml b/emdatabase/index/PeakDetectionPolymers.yaml index fa438e1..4631ee3 100644 --- a/emdatabase/index/PeakDetectionPolymers.yaml +++ b/emdatabase/index/PeakDetectionPolymers.yaml @@ -42,7 +42,3 @@ PeakDetectionPolymers: url: https://zenodo.org/records/22311217/files/best.pth size_bytes: 57343166 checksum: md5:1d2524b53c6f8be4a85d0530312a4ade - "260914": - url: https://zenodo.org/records/22741835/files/best.pth - checksum: md5:1d2524b53c6f8be4a85d0530312a4ade - size_bytes: 57343166 diff --git a/emdatabase/tests/test_check_weights.py b/emdatabase/tests/test_check_weights.py index 12ddbe9..9f2d3f0 100644 --- a/emdatabase/tests/test_check_weights.py +++ b/emdatabase/tests/test_check_weights.py @@ -360,6 +360,34 @@ def test_a_new_zenodo_record_becomes_a_dated_version(script, gh, zenodo, tmp_pat assert _md5(LATEST_BYTES) in report and _md5(NEW_BYTES) in report +def test_a_new_zenodo_record_serving_the_same_file_files_no_version(script, gh, zenodo, tmp_path): + """A new record is not a new file. + + Zenodo versions the whole record, so publishing anything beside the weights + - a zip of the training data, say - mints a new id while the weights stand + still. Two dates for one checksum is the state this guards against. + """ + base, _, directory, path, publish = zenodo + publish( + OLD_RECORD, + "2026-01-01", + [_zenodo_file(base, OLD_RECORD, FILE, LATEST_BYTES)], + newest=False, + ) + publish(NEW_RECORD, "2026-03-04", [_zenodo_file(base, NEW_RECORD, FILE, LATEST_BYTES)]) + + code, summary = _run(script, directory, tmp_path) + + assert code == 0 + assert gh == [] + entry = _entry(path) + assert list(entry["versions"]) == [OLD_DATE] # no second date for the same bytes + assert entry["latest"]["checksum"] == _md5(LATEST_BYTES) + # The link still moves to the record Zenodo serves now. + assert entry["latest"]["url"] == f"{base}/records/{NEW_RECORD}/files/{FILE}" + assert "no version was filed" in summary.read_text() + + def test_a_zenodo_record_without_the_named_file_fails(script, gh, zenodo, tmp_path): base, _, directory, path, publish = zenodo publish( From 968764062cc79a4afd17804c485ce3793f441eca Mon Sep 17 00:00:00 2001 From: arthurmccray Date: Thu, 17 Sep 2026 21:38:35 -0700 Subject: [PATCH 07/12] updating descriptions of ptycho datasets --- emdatabase/data/__init__.pyi | 4 ++-- emdatabase/index/MOSS6Fig3.yaml | 18 ++++++------------ emdatabase/index/TwistedBilayerWSe2Themis.yaml | 14 +++++--------- 3 files changed, 13 insertions(+), 23 deletions(-) diff --git a/emdatabase/data/__init__.pyi b/emdatabase/data/__init__.pyi index b2e420e..caf1cb7 100644 --- a/emdatabase/data/__init__.pyi +++ b/emdatabase/data/__init__.pyi @@ -225,7 +225,7 @@ class MOSS6Fig3(DownloadableDataset): """ MOSS6Fig3 - The MOSS-6 metal-organic framework 4D-STEM ptychography dataset shown in Fig. 3 of "Atomically resolved imaging of radiation-sensitive metal-organic frameworks via electron ptychography" (Nature Communications 16, 2025; doi 10.1038/s41467-025-56215-z). It arrives as a pair in a directory of its own: acquisition_12.xml, the EMPAD header, which is what download() hands back and what a reader is pointed at, and scan_x256_y256.raw beside it, which the header names and which holds the data - 256 x 256 scan positions of 128 x 130 little-endian float32 frames, each frame a 128 x 128 detector followed by two rows of EMPAD metadata. Read it with quantem's read_4dstem(path, file_type="empad"), or rsciio.empad, which takes the header and opens the raw next to it. The scan was taken on a double Cs-corrected FEI Titan Cubed Themis Z at 300 kV, with a 10 mrad convergence semi-angle, a 1.05 Angstrom step, about 100 nm of defocus, and an electron dose of about 98 electrons per square Angstrom. MOSS-6 is a MOF solid solution of the NU-1000 and NU-901 phases. Both files are members of the record's 15.3 GB RawData.7z and are fetched out of it directly, without downloading the archive; the raw costs about 2.6 GB of it. + The MOSS-6 metal-organic framework 4D-STEM ptychography dataset shown in Fig. 3 of "Atomically resolved imaging of radiation-sensitive metal-organic frameworks via electron ptychography" (Nature Communications 16, 2025; doi 10.1038/s41467-025-56215-z). It downloads two files into a directory of its own: acquisition_12.xml, the EMPAD header, which is what download() hands back, and scan_x256_y256.raw beside it. Read it with quantem's read_4dstem(path, file_type="empad"), or rsciio.empad, which takes the header. The scan was taken on a Cs-corrected FEI Titan at 300 kV, with a 10 mrad convergence semi-angle, a 1.05 Angstrom step, about 100 nm of defocus, and an electron dose of about 98 e/A^2. DOI: 10.5281/zenodo.13958144 @@ -350,7 +350,7 @@ class TwistedBilayerWSe2Themis(DownloadableDataset): """ TwistedBilayerWSe2Themis - Electron ptychography of a twisted bilayer WSe2, acquired at 80 kV on an uncorrected Thermo Fisher Themis with an EMPAD. From the record for "Achieving sub-0.5-Angstrom resolution ptychography in an uncorrected electron microscope", which reaches 0.44 Angstrom without an aberration corrector (Science 384, adl2029). The file is one member of the record's 1.9 GB Fig_01.zip and is fetched out of it directly, without downloading the archive. It is a headerless raw: a 128 x 128 scan of 128 x 130 little-endian float32 frames, each frame the 128 x 128 detector followed by two rows of EMPAD metadata. Read it with numpy.fromfile(path, dtype="- The MOSS-6 metal-organic framework 4D-STEM ptychography dataset shown in Fig. 3 of "Atomically resolved imaging of radiation-sensitive metal-organic frameworks via electron - ptychography" (Nature Communications 16, 2025; doi 10.1038/s41467-025-56215-z). It arrives - as a pair in a directory of its own: acquisition_12.xml, the EMPAD header, which is what - download() hands back and what a reader is pointed at, and scan_x256_y256.raw beside it, - which the header names and which holds the data - 256 x 256 scan positions of 128 x 130 - little-endian float32 frames, each frame a 128 x 128 detector followed by two rows of EMPAD - metadata. Read it with quantem's read_4dstem(path, file_type="empad"), or rsciio.empad, - which takes the header and opens the raw next to it. The scan was taken on a double - Cs-corrected FEI Titan Cubed Themis Z at 300 kV, with a 10 mrad convergence semi-angle, a - 1.05 Angstrom step, about 100 nm of defocus, and an electron dose of about 98 electrons per - square Angstrom. MOSS-6 is a MOF solid solution of the NU-1000 and NU-901 phases. Both - files are members of the record's 15.3 GB RawData.7z and are fetched out of it directly, - without downloading the archive; the raw costs about 2.6 GB of it. + ptychography" (Nature Communications 16, 2025; doi 10.1038/s41467-025-56215-z). It downloads + two files into a directory of its own: acquisition_12.xml, the EMPAD header, which is what + download() hands back, and scan_x256_y256.raw beside it. Read it with quantem's + read_4dstem(path, file_type="empad"), or rsciio.empad, which takes the header. The scan was + taken on a Cs-corrected FEI Titan at 300 kV, with a 10 mrad convergence semi-angle, a 1.05 + Angstrom step, about 100 nm of defocus, and an electron dose of about 98 e/A^2. source: https://zenodo.org/records/13958144/files archive: url: https://zenodo.org/records/13958144/files/RawData.7z diff --git a/emdatabase/index/TwistedBilayerWSe2Themis.yaml b/emdatabase/index/TwistedBilayerWSe2Themis.yaml index 7981746..db57ecf 100644 --- a/emdatabase/index/TwistedBilayerWSe2Themis.yaml +++ b/emdatabase/index/TwistedBilayerWSe2Themis.yaml @@ -2,15 +2,11 @@ TwistedBilayerWSe2Themis: description: >- Electron ptychography of a twisted bilayer WSe2, acquired at 80 kV on an uncorrected - Thermo Fisher Themis with an EMPAD. From the record for "Achieving sub-0.5-Angstrom - resolution ptychography in an uncorrected electron microscope", which reaches 0.44 Angstrom - without an aberration corrector (Science 384, adl2029). The file is one member of the - record's 1.9 GB Fig_01.zip and is fetched out of it directly, without downloading the - archive. It is a headerless raw: a 128 x 128 scan of 128 x 130 little-endian float32 - frames, each frame the 128 x 128 detector followed by two rows of EMPAD metadata. Read it - with numpy.fromfile(path, dtype=" Date: Thu, 17 Sep 2026 21:47:28 -0700 Subject: [PATCH 08/12] updating wording in example --- examples/changing_download_location.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/examples/changing_download_location.py b/examples/changing_download_location.py index 4cea5fa..3d862c4 100644 --- a/examples/changing_download_location.py +++ b/examples/changing_download_location.py @@ -10,14 +10,14 @@ :func:`emdatabase.config.remove_location` are how you manage them. Every call here passes ``persist=False``, so this example changes nothing on -disk. Drop it and the location is written to +disk. If ``persist=True`` (the default), the location is written to ``~/.config/emdatabase/config.yaml``, which is read on every import. """ from emdatabase import config # %% -# The personal location is where datasets download to. Unset, it is pooch's +# The personal location is where datasets download to. By default, it is pooch's # cache directory for emdatabase (``~/.cache/emdatabase`` on Linux). print("configured :", config.get("locations.personal")) print("download dir :", config.data_dir()) From f409507661b9ef2d27e8bbd04315dd1ef92e64be Mon Sep 17 00:00:00 2001 From: arthurmccray Date: Thu, 17 Sep 2026 22:12:10 -0700 Subject: [PATCH 09/12] updating techniques and fixing wordings --- docs/source/_build_docs.py | 12 ++++++------ emdatabase/data/__init__.pyi | 2 +- emdatabase/index/ApoferritinApollo15eps.yaml | 1 + emdatabase/index/CuZnEELSMapping.yaml | 1 + emdatabase/index/FeAlStripes.yaml | 1 + emdatabase/index/InSituElectrochemGrowth.yaml | 1 + emdatabase/index/LSMOLineScan.yaml | 1 + emdatabase/index/LSMOLineScanLowLoss.yaml | 1 + emdatabase/index/LSMOSTOLineScan.yaml | 1 + emdatabase/index/LSMOSTOLineScanLowLoss.yaml | 1 + emdatabase/index/MOSS6Fig3.yaml | 6 +++--- emdatabase/index/PdCuSiCrystallization.yaml | 1 + emdatabase/index/PeakDetectionPolymers.yaml | 1 + emdatabase/index/SPEDAg.yaml | 1 + examples/README.rst | 2 +- 15 files changed, 22 insertions(+), 11 deletions(-) diff --git a/docs/source/_build_docs.py b/docs/source/_build_docs.py index bae0c99..4cd8c34 100644 --- a/docs/source/_build_docs.py +++ b/docs/source/_build_docs.py @@ -350,8 +350,7 @@ def generate_weights_html() -> str: "

Trained model checkpoints, one entry per model. Downloading an entry " "follows its latest link, which serves whatever the current " "weights are; every earlier state of that link is kept as a dated version, " - "pinned to its checksum. Pick one with the selector; the load snippet " - "opens it with weights_only=True.

" + "pinned to its checksum." "

download() warns when the index on the project's " "main branch has newer weights than your installed release, " "and download(refresh=True) fetches them; the " @@ -379,12 +378,13 @@ def generate_add_dataset_html() -> str: '

' '
' "

Add a Dataset

" - "

Submissions go through an issue. Fill in the new-dataset issue form and " + "

Submissions go through an issue on GitHub. Fill in the new-dataset issue form and " "an action turns it into the entry’s YAML file, fills in whatever it can " "work out for itself, and opens the pull request.

" + "The checksum and the size are filled in automatically by " + "downloading the file, so both can be left blank in most cases. If the file is very large," + " you may want to provide the checksum manually.

" "
" - '

The checksum and the size are filled in automatically by ' - "downloading the file, so both can be left blank.

" '
' 'From a terminal" "
" + escape(_CLI_COMMAND) + "
" "

It asks the same questions at the prompt and writes the YAML file, leaving " - "the pull request to you. A trained model checkpoint takes " + "the pull request to you. A model checkpoint takes " "--kind weights as well.

" "

It computes the checksum from the file on your own machine rather than on " "GitHub, which is the route to take for a very large file.

" diff --git a/emdatabase/data/__init__.pyi b/emdatabase/data/__init__.pyi index caf1cb7..2618678 100644 --- a/emdatabase/data/__init__.pyi +++ b/emdatabase/data/__init__.pyi @@ -225,7 +225,7 @@ class MOSS6Fig3(DownloadableDataset): """ MOSS6Fig3 - The MOSS-6 metal-organic framework 4D-STEM ptychography dataset shown in Fig. 3 of "Atomically resolved imaging of radiation-sensitive metal-organic frameworks via electron ptychography" (Nature Communications 16, 2025; doi 10.1038/s41467-025-56215-z). It downloads two files into a directory of its own: acquisition_12.xml, the EMPAD header, which is what download() hands back, and scan_x256_y256.raw beside it. Read it with quantem's read_4dstem(path, file_type="empad"), or rsciio.empad, which takes the header. The scan was taken on a Cs-corrected FEI Titan at 300 kV, with a 10 mrad convergence semi-angle, a 1.05 Angstrom step, about 100 nm of defocus, and an electron dose of about 98 e/A^2. + The MOSS-6 metal-organic framework 4D-STEM ptychography dataset shown in Fig. 3 of "Atomically resolved imaging of radiation-sensitive metal-organic frameworks via electron ptychography" (Nature Communications 16, 2025; doi 10.1038/s41467-025-56215-z). It downloads two files into a directory of its own: acquisition_12.xml, the EMPAD header, which is what download() hands back, and scan_x256_y256.raw beside it. Read it with quantem's read_4dstem(path, file_type="empad"), or rsciio.empad, which takes the header. The scan was taken on a Cs-corrected FEI Titan at 300 kV, with a 10 mrad convergence semi-angle, a 1.05 Angstrom step, about 100 nm of defocus, and an electron dose of about 98 e/A^2. DOI: 10.5281/zenodo.13958144 diff --git a/emdatabase/index/ApoferritinApollo15eps.yaml b/emdatabase/index/ApoferritinApollo15eps.yaml index 83b6aa8..89d933d 100644 --- a/emdatabase/index/ApoferritinApollo15eps.yaml +++ b/emdatabase/index/ApoferritinApollo15eps.yaml @@ -16,6 +16,7 @@ ApoferritinApollo15eps: license: CC0-1.0 technique: - Cryo + - TEM doi: 10.5281/zenodo.21632101 tags: - Cryo-EM diff --git a/emdatabase/index/CuZnEELSMapping.yaml b/emdatabase/index/CuZnEELSMapping.yaml index 9f9d5d3..5957833 100644 --- a/emdatabase/index/CuZnEELSMapping.yaml +++ b/emdatabase/index/CuZnEELSMapping.yaml @@ -21,6 +21,7 @@ CuZnEELSMapping: license: Unspecified technique: - EELS + - STEM tags: - Spectrum Image - Elemental Mapping diff --git a/emdatabase/index/FeAlStripes.yaml b/emdatabase/index/FeAlStripes.yaml index 95b2d8b..1b16c96 100644 --- a/emdatabase/index/FeAlStripes.yaml +++ b/emdatabase/index/FeAlStripes.yaml @@ -12,6 +12,7 @@ FeAlStripes: license: CC-BY-4.0 technique: - 4D-STEM + - Lorentz tags: - Strain Mapping authors: diff --git a/emdatabase/index/InSituElectrochemGrowth.yaml b/emdatabase/index/InSituElectrochemGrowth.yaml index 0517445..2e90643 100644 --- a/emdatabase/index/InSituElectrochemGrowth.yaml +++ b/emdatabase/index/InSituElectrochemGrowth.yaml @@ -16,6 +16,7 @@ InSituElectrochemGrowth: doi: 10.5281/zenodo.21632101 technique: - In-situ + - TEM tags: - In-situ - Electrochemistry diff --git a/emdatabase/index/LSMOLineScan.yaml b/emdatabase/index/LSMOLineScan.yaml index e126b18..0248bbe 100644 --- a/emdatabase/index/LSMOLineScan.yaml +++ b/emdatabase/index/LSMOLineScan.yaml @@ -20,6 +20,7 @@ LSMOLineScan: license: Unspecified technique: - EELS + - STEM tags: - Line Scan - Core Loss diff --git a/emdatabase/index/LSMOLineScanLowLoss.yaml b/emdatabase/index/LSMOLineScanLowLoss.yaml index c4c3ad9..36b69a5 100644 --- a/emdatabase/index/LSMOLineScanLowLoss.yaml +++ b/emdatabase/index/LSMOLineScanLowLoss.yaml @@ -19,6 +19,7 @@ LSMOLineScanLowLoss: license: Unspecified technique: - EELS + - STEM tags: - Line Scan - Low Loss diff --git a/emdatabase/index/LSMOSTOLineScan.yaml b/emdatabase/index/LSMOSTOLineScan.yaml index d637555..371c2fe 100644 --- a/emdatabase/index/LSMOSTOLineScan.yaml +++ b/emdatabase/index/LSMOSTOLineScan.yaml @@ -21,6 +21,7 @@ LSMOSTOLineScan: license: Unspecified technique: - EELS + - STEM tags: - Line Scan - Core Loss diff --git a/emdatabase/index/LSMOSTOLineScanLowLoss.yaml b/emdatabase/index/LSMOSTOLineScanLowLoss.yaml index 157ce66..163fe84 100644 --- a/emdatabase/index/LSMOSTOLineScanLowLoss.yaml +++ b/emdatabase/index/LSMOSTOLineScanLowLoss.yaml @@ -19,6 +19,7 @@ LSMOSTOLineScanLowLoss: license: Unspecified technique: - EELS + - STEM tags: - Line Scan - Low Loss diff --git a/emdatabase/index/MOSS6Fig3.yaml b/emdatabase/index/MOSS6Fig3.yaml index 1ba0792..83ccce4 100644 --- a/emdatabase/index/MOSS6Fig3.yaml +++ b/emdatabase/index/MOSS6Fig3.yaml @@ -5,10 +5,10 @@ MOSS6Fig3: "Atomically resolved imaging of radiation-sensitive metal-organic frameworks via electron ptychography" (Nature Communications 16, 2025; doi 10.1038/s41467-025-56215-z). It downloads two files into a directory of its own: acquisition_12.xml, the EMPAD header, which is what - download() hands back, and scan_x256_y256.raw beside it. Read it with quantem's + download() hands back, and scan_x256_y256.raw beside it. Read it with quantem's read_4dstem(path, file_type="empad"), or rsciio.empad, which takes the header. The scan was - taken on a Cs-corrected FEI Titan at 300 kV, with a 10 mrad convergence semi-angle, a 1.05 - Angstrom step, about 100 nm of defocus, and an electron dose of about 98 e/A^2. + taken on a Cs-corrected FEI Titan at 300 kV, with a 10 mrad convergence semi-angle, a 1.05 + Angstrom step, about 100 nm of defocus, and an electron dose of about 98 e/A^2. source: https://zenodo.org/records/13958144/files archive: url: https://zenodo.org/records/13958144/files/RawData.7z diff --git a/emdatabase/index/PdCuSiCrystallization.yaml b/emdatabase/index/PdCuSiCrystallization.yaml index bb3d64f..52aae7a 100644 --- a/emdatabase/index/PdCuSiCrystallization.yaml +++ b/emdatabase/index/PdCuSiCrystallization.yaml @@ -20,6 +20,7 @@ PdCuSiCrystallization: doi: 10.5281/zenodo.15490547 technique: - 4D-STEM + - In-situ tags: - Amorphous - Metallic Glass diff --git a/emdatabase/index/PeakDetectionPolymers.yaml b/emdatabase/index/PeakDetectionPolymers.yaml index 4631ee3..09cb960 100644 --- a/emdatabase/index/PeakDetectionPolymers.yaml +++ b/emdatabase/index/PeakDetectionPolymers.yaml @@ -9,6 +9,7 @@ PeakDetectionPolymers: file: best.pth license: CC-BY-4.0 technique: + - 4D-STEM - ML - peak finding doi: 10.5281/zenodo.22311216 tags: diff --git a/emdatabase/index/SPEDAg.yaml b/emdatabase/index/SPEDAg.yaml index e273d28..2deb829 100644 --- a/emdatabase/index/SPEDAg.yaml +++ b/emdatabase/index/SPEDAg.yaml @@ -12,6 +12,7 @@ SPEDAg: license: CC-BY-4.0 technique: - 4D-STEM + - Diffraction tags: - Orientation Mapping - Grain Boundaries diff --git a/examples/README.rst b/examples/README.rst index 35e9207..35028a1 100644 --- a/examples/README.rst +++ b/examples/README.rst @@ -1,4 +1,4 @@ Examples ======== -Below are a gallery of examples for using the EM-Database API to download +Below is a gallery of examples for using the EM-Database API to download EM Data into different applications. From 9be6c8b0fc71712013d6cc2e251986e53a918cff Mon Sep 17 00:00:00 2001 From: Arthur McCray Date: Thu, 17 Sep 2026 22:29:16 -0700 Subject: [PATCH 10/12] Revise README --- README.md | 178 +++--------------------------------------------------- 1 file changed, 9 insertions(+), 169 deletions(-) diff --git a/README.md b/README.md index a86bd9c..b612d86 100644 --- a/README.md +++ b/README.md @@ -1,168 +1,25 @@ emdatabase ---------- -This is a simple project for aggregating different Electron Microscopy files which are hosted over different sources. It uses pooch to download datasets and should be -used as a way to host simple example datasets for method validation. +This is a project for aggregating different Electron Microscopy files which are hosted over different sources. It is intended to simplify downloading example datasets and trained machine learning model weights for tutorials and method validation. -Downloads go to `~/.cache/emdatabase` by default; shared read-only locations can be added with `emdatabase.add_location`. - -List of datasets https://electronmicroscopy.github.io/emdatabase/datasets.html +A list of all datasets and model weights can be found in our [docs](https://electronmicroscopy.github.io/emdatabase/datasets.html). ## Installation ```bash pip install emdatabase ``` +You can install `emdatabase` via pip or as a local install in the usual way. ## Usage -Every dataset is a class under `emdatabase.data`. Calling `download()` fetches the -file to the data directory, verifies its checksum, and returns a path handle. Files -that are already present are not downloaded again. - -```python -import emdatabase.data as data -import hyperspy.api as hs - -path = data.LayeredCuNb4DSTEM().download() -s = hs.load(path, lazy=True) -``` - -By default the download runs on a background thread so a notebook cell returns -immediately. The handle it returns *is* the file path — a `pathlib.Path` subclass -pointing at the file's final location — so you can hand it straight to a loader as -above; it only blocks at the moment the file is actually opened. Read `path.done` -to check progress without blocking, or call `path.result()` to wait explicitly. -`download(background=False)` blocks instead, and returns the same type. - -Any path pointing at the same file waits, however it was built, so a derived path -(`handle.parent / handle.name`, `handle.with_suffix(...)`) behaves too. The -exceptions are `str(handle)` and `Path(handle)`: both hand back an ordinary value -with no download attached, so `hs.load(str(handle))` will *not* wait. Keeping -`str()` non-blocking is deliberate — `repr()` needs it — so pass the handle itself. - -## Finding a dataset - -`search()` is the browser widget's search box, callable from Python; `filter()` -matches named fields. Both return dataset objects, so a result can be downloaded -directly. - -```python -import emdatabase - -emdatabase.list_datasets() # everything -emdatabase.search("amorphous") # any field -emdatabase.search("jeol eels") # all terms, any field -emdatabase.filter(technique="4D-STEM", tags="Strain") # exact, case-insensitive -emdatabase.filter(microscope_vendor=["JEOL", "Hitachi"]) # a list means any of -emdatabase.filter(downloaded=True) # what is already here -``` - -`technique`, `tags`, `authors` and `version` are several values per dataset, so they -test membership: a dataset that is both in-situ and 4D-STEM matches either. - -An unknown field raises rather than being ignored, so a typo cannot quietly return -the whole index. - -## Configuration - -Data lives in named **locations**. `personal` is the one writable location, -where downloads go; every other one is read-only and searched first, so a copy -already on a group drive is used instead of refetched. - -```python -from emdatabase import config - -config.add_location("/group/example_data") # read-only -config.add_location("/big/disk/emdatabase", name="personal") # where downloads go -config.locations() -``` - -``` -[Location(name='example_data', path=PosixPath('/group/example_data'), kind='shared'), - Location(name='personal', path=PosixPath('/big/disk/emdatabase'), kind='personal')] -``` - -`locations()` is the search order: the shared locations in the order they were -added, then `personal` last. A location is named after the last component of its -path unless you pass `name=`, and that name is the provenance — it is what -`catalogue.entry()["location"]`, `emdatabase.filter(location="example_data")` and -the browser widget report for a copy found there. Nothing is written to a shared -location unless you name it as a download's destination, which is how one is -seeded. - -Removing one takes either the name or the path; `"personal"` is not deleted but -reset, putting downloads back in the default cache directory: - -```python -config.remove_location("example_data") -config.remove_location("/group/example_data") # the same thing, by path -config.remove_location("personal") -``` - -Both functions persist to `~/.config/emdatabase/config.yaml`, which is read on -every import. Pass `persist=False` to change this process only, or use -`config.set` as a context manager for a change that lasts for a block: - -```python -config.add_location("/scratch/em", name="personal", persist=False) # this process -with config.set({"locations.personal": "/scratch/em"}): # this block - ... -``` - -The path does not have to exist when you add it — a share may be mounted later — -but you get a warning saying so. - -### Seeding a shared location - -`destination=` takes a location's name, which is how the copy gets onto the share -in the first place — run it once, from an account with write access: - -```python -from emdatabase import data - -data.CuZnHAADF().download(destination="example_data") -``` - -The file is written with your umask, so `chmod` it group-readable afterwards if -your umask is not; emdatabase does not set permissions for you. - -### The key underneath - -Configuration is dask-style: shipped defaults, then every `*.yaml` in -`~/.config/emdatabase/` (or wherever `EMDATABASE_CONFIG` points), then -environment variables, then `config.set` — each layer overriding the one before. -There are two keys, and `add_location` is a wrapper over writing the first one -yourself: - -```yaml -# ~/.config/emdatabase/config.yaml -locations: - example_data: /group/example_data - cluster: /cluster/em_data - personal: /big/disk/emdatabase -check_updates: true -``` - -`personal: null` means pooch's cache directory (`~/.cache/emdatabase` on Linux), -and `config.data_dir()` reports whichever it resolves to. - -`check_updates` is whether downloading a model's `latest` weights asks the index -on the project's `main` branch — kept current by a weekly job — whether newer -weights have been published, and warns if they have; `download(refresh=True)` -fetches them. Set it to `false` to skip the request. - -On HPC, where a config file is often the wrong place to put a machine-specific -path, set the same key from the environment instead — prefix `EMDATABASE_`, -double underscore to nest — which needs no file and no write access: - -```bash -export EMDATABASE_LOCATIONS__PERSONAL=/scratch/emdatabase -export EMDATABASE_LOCATIONS__GROUP=/group/example_data -``` +Examples of how to download data and configure storage locations can be found in our [docs](https://electronmicroscopy.github.io/emdatabase/examples/index.html), as well as in the [quantem-tutorials](https://github.com/electronmicroscopy/quantem-tutorials/tree/main/tutorials/core) repository. ## Adding a dataset +We welcome contributions of new or existing data! + Datasets are described by a YAML file in `emdatabase/index/`, one entry per file, validated against `emdatabase/index/json-schema.json`. The class name is generated from the top-level key: @@ -185,26 +42,9 @@ vocabulary they come from: `acquisition` (how the data was taken) and `ml_task` (what a model does). A dataset declares acquisition techniques only; a `kind: weights` entry declares one of those plus the ML tasks it performs. -`size_bytes` is the file's `Content-Length` in bytes; the test suite checks it against -the server on every run. `emdatabase/index/vendors.yaml` lists the microscope -vendors and detector manufacturers already in use - a new one is fine, but a name close -to one already on the list fails CI as a misspelling. A technique close to one in -`techniques.yaml` fails the same way. - Submissions go through an issue. Fill in the [new dataset issue form](https://github.com/electronmicroscopy/emdatabase/issues/new?template=new_dataset.yaml) -and an action writes the file and opens the pull request; the -[Add Dataset page](https://electronmicroscopy.github.io/emdatabase/add_dataset.html) -says what to have ready first. Or run `python -m emdatabase.new_dataset `, which -fetches the checksum and size, prompts for the rest and writes the file for you to open -a pull request with. See [CONTRIBUTING.md](CONTRIBUTING.md). - -Both take one download link and split it into `source`, `file` and, when the file is -not served at `source/file`, `url`. A Google Drive share link - what the share button -copies - is rewritten to the `uc?export=download&id=` link that serves the file. -Authors go in as one per line, `Name; Affiliation; ORCID`, with the ORCID optional. +and an action writes the file and opens the pull request. For very large datasets it might be easier to use the CLI +as described on the [Add Dataset page](https://electronmicroscopy.github.io/emdatabase/add_dataset.html). +See [CONTRIBUTING.md](CONTRIBUTING.md). -Neither route needs the checksum or the size. An entry that is missing either one has -the file downloaded on GitHub and the fields filled in and pushed back to the branch; a -pull request from a fork, whose branch cannot be pushed to, fails with the values to -paste in instead. From 85ffded5ccaa596740cb75fbdf668b449c0bd181 Mon Sep 17 00:00:00 2001 From: arthurmccray Date: Thu, 17 Sep 2026 22:29:36 -0700 Subject: [PATCH 11/12] fixing check of hash for multiple file dsets --- docs/source/contributing.rst | 5 +-- emdatabase/new_dataset.py | 8 +++-- emdatabase/tests/test_fill_download_fields.py | 35 +++++++++++++++++++ 3 files changed, 44 insertions(+), 4 deletions(-) diff --git a/docs/source/contributing.rst b/docs/source/contributing.rst index b44dc76..4227f37 100644 --- a/docs/source/contributing.rst +++ b/docs/source/contributing.rst @@ -132,8 +132,9 @@ the pairing and keeps it clear of identically named members elsewhere. ``delete()`` removes them all, and the size shown in the catalogue counts them. These entries are written by hand. The two fields CI otherwise fills in would be -taken from the archive rather than from the member, so a pull request leaving any -member's blank is refused rather than guessed at. ``download_url``, and the download +taken from the archive rather than from the member, so a pull request leaving the +entry's own blank is refused rather than guessed at. A member beside it may leave +its ``checksum`` out, and is then fetched unverified. ``download_url``, and the download link on the docs site, point at the archive: the member has no link of its own. A host that ignores ``Range`` and answers with the whole archive fails loudly, diff --git a/emdatabase/new_dataset.py b/emdatabase/new_dataset.py index f606c5e..38dc675 100644 --- a/emdatabase/new_dataset.py +++ b/emdatabase/new_dataset.py @@ -357,10 +357,14 @@ def fill_download_fields(document: dict[str, Any]) -> list[str]: # the archive. Following the link here would hash the whole archive # and write a value that is wrong in a way nothing downstream # notices. + # A later member may leave its checksum out - the schema allows it, + # and pooch then fetches that file unverified - so only the entry's + # own file, which is the first member, is required here. members = entry["archive"].get("members") or () - if not members or not all(m.get("checksum") and m.get("size_bytes") for m in members): + first = members[0] if members else {} + if not (first.get("checksum") and first.get("size_bytes")): raise ValueError( - f"{name}: each member's checksum and size_bytes describe that file " + f"{name}: the first member's checksum and size_bytes describe that file " "inside the archive, which is not something this can download. Fill " "them in by hand." ) diff --git a/emdatabase/tests/test_fill_download_fields.py b/emdatabase/tests/test_fill_download_fields.py index 066aa42..26bc051 100644 --- a/emdatabase/tests/test_fill_download_fields.py +++ b/emdatabase/tests/test_fill_download_fields.py @@ -41,6 +41,24 @@ } ], } +# A second member with no checksum of its own, which the schema allows: the one +# shipped multi-file entry has exactly this shape, its 4.3 GB raw unhashed. +PAIRED_ARCHIVE = { + "url": "https://example.invalid/Figures.zip", + "members": [ + { + "member": "Fig/Panel/scan.raw", + "file": FILE, + "checksum": MD5, + "size_bytes": len(CONTENT), + }, + { + "member": "Fig/Panel/scan_x256.raw", + "file": "scan_x256.raw", + "size_bytes": 4362076160, + }, + ], +} @pytest.fixture(scope="module") @@ -115,6 +133,23 @@ def test_an_archive_entry_with_blank_fields_is_refused(script, index, tmp_path): assert path.read_text(encoding="utf-8") == before # nothing guessed, nothing written +def test_a_member_without_a_checksum_is_not_refused(script, index, tmp_path): + """Only the entry's own file is asked for. + + A member beside it may leave its checksum out - the schema requires only + ``member`` and ``file`` - so requiring one of every member would refuse the + entry the feature was built for. + """ + _, directory, write = index + path = write(file=None, archive=PAIRED_ARCHIVE) + before = path.read_text(encoding="utf-8") + + code, _ = _run(script, directory, tmp_path) + + assert code == 0 + assert path.read_text(encoding="utf-8") == before + + def test_a_complete_archive_entry_is_not_downloaded(script, index, tmp_path): """The archive URL does not resolve, so getting here at all means it was left alone.""" _, directory, write = index From b9c0ace2b92d7e5787470073a95af4456da10d15 Mon Sep 17 00:00:00 2001 From: arthurmccray Date: Thu, 17 Sep 2026 23:00:53 -0700 Subject: [PATCH 12/12] test fix --- emdatabase/tests/test_widget.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/emdatabase/tests/test_widget.py b/emdatabase/tests/test_widget.py index 7486682..1dad469 100644 --- a/emdatabase/tests/test_widget.py +++ b/emdatabase/tests/test_widget.py @@ -107,7 +107,11 @@ def test_browse_returns_widget_populated_from_the_catalogue(): widget = _browser() assert widget.n_total > 0 assert widget.groups - assert widget.n_total == sum(len(g["items"]) for g in widget.groups) + # As for the payload above: a dataset declaring several techniques is in + # each of their groups, so the groups overlap and n_total is the distinct + # rows rather than the group sizes added up. + names = {it["name"] for g in widget.groups for it in g["items"]} + assert widget.n_total == len(names) assert isinstance(widget.data_dir, str) and widget.data_dir