diff --git a/.github/scripts/check_latest_weights.py b/.github/scripts/check_latest_weights.py index 5448f11..c2e8264 100644 --- a/.github/scripts/check_latest_weights.py +++ b/.github/scripts/check_latest_weights.py @@ -36,7 +36,6 @@ import argparse import datetime import json -import os import re import shutil import subprocess @@ -51,7 +50,6 @@ from emdatabase.metadata import ( INDEX_DIR, - NON_DATASET_FILES, DatasetMetadata, WeightsVersion, dataset_files, @@ -77,11 +75,6 @@ def asset_url(tag: str, asset: str) -> str: return f"{RELEASE_DOWNLOADS}/{tag}/{asset}" -def is_archived(url: str, tag: str) -> bool: - """Whether a version's link already points into the archive release.""" - return url.startswith(f"{RELEASE_DOWNLOADS}/{tag}/") - - @dataclass(frozen=True) class ZenodoLink: """A Zenodo record file link, split into the pieces the API needs.""" @@ -118,23 +111,9 @@ def fetch_json(url: str) -> dict[str, Any]: return json.loads(response.read().decode("utf-8")) -def index_files(index_dir: Path | None) -> list[Path]: - """Every dataset YAML to check, sorted by name.""" - if index_dir is None: - return dataset_files() - return sorted(p for p in index_dir.rglob("*.y*ml") if p.name not in NON_DATASET_FILES) - - def run_gh(args: list[str], *, check: bool = True) -> subprocess.CompletedProcess[str]: - """Run one ``gh`` command; the only place this script shells out. - - ``gh`` reads ``GH_TOKEN``, while a workflow hands the job ``GITHUB_TOKEN``. - """ - env = os.environ.copy() - token = env.get("GH_TOKEN") or env.get("GITHUB_TOKEN") - if token: - env["GH_TOKEN"] = token - return subprocess.run(["gh", *args], check=check, capture_output=True, text=True, env=env) + """Run one ``gh`` command; the only place this script shells out.""" + return subprocess.run(["gh", *args], check=check, capture_output=True, text=True) def upload_asset(tag: str, path: Path, asset: str) -> None: @@ -212,9 +191,7 @@ def check_family( """Download one family's ``latest`` and update ``entry`` in place.""" report = Report(lines=[f"### {name}"]) latest = metadata.latest - if latest is None: - report.lines.append("- no `latest` link, so there is nothing to follow") - return report + assert latest is not None # the schema requires one of every weights entry link = zenodo_link(latest.url) if link is not None: @@ -250,19 +227,6 @@ def check_family( return report -def _newest_record(link: ZenodoLink, record: dict[str, Any]) -> dict[str, Any]: - """The newest record of this concept. - - ``links.latest`` is what Zenodo publishes for it; the concept record id is - the fallback, which the API resolves to the same place. - """ - follow = (record.get("links") or {}).get("latest") - if not follow: - concept = record.get("conceptrecid") - follow = f"{link.api}/{concept}" if concept else None - return fetch_json(str(follow)) if follow else record - - def _record_file(record: dict[str, Any], key: str) -> dict[str, Any] | None: """The file entry named ``key``, or the only one when the name has changed.""" files = [item for item in record.get("files") or [] if isinstance(item, dict)] @@ -272,13 +236,18 @@ def _record_file(record: dict[str, Any], key: str) -> dict[str, Any] | None: return files[0] if len(files) == 1 else None -def _record_date(record: dict[str, Any]) -> str: - """The record's publication date as ``YYMMDD``, or today's if it has none.""" - published = str((record.get("metadata") or {}).get("publication_date", "")) - try: - return datetime.date.fromisoformat(published[:10]).strftime("%y%m%d") - except ValueError: - return version_date() +def _date_taken(report: Report, metadata: DatasetMetadata, date: str, checksum: str) -> bool: + """Whether version ``date`` already holds a different file; a failure if it does.""" + existing = metadata.versions.get(date) + if existing is None or existing.checksum == checksum: + return False + report.ok = False + report.lines.append( + f"- **version `{date}` already exists with `{existing.checksum}`**, and the newest file " + f"is `{checksum}`. Two states of this file share one date; refile the earlier one by " + "hand under the date it was published." + ) + return True def _check_zenodo( @@ -295,13 +264,15 @@ def _check_zenodo( version written for it points straight at that record. """ try: - newest = _newest_record(link, fetch_json(f"{link.api}/{link.record_id}")) - except (OSError, ValueError) as error: + record = fetch_json(f"{link.api}/{link.record_id}") + newest = fetch_json(record["links"]["latest"]) + new_id = str(newest["id"]) + published = datetime.date.fromisoformat(newest["metadata"]["publication_date"][:10]) + except (OSError, KeyError, ValueError) as error: report.ok = False report.lines.append(f"- **could not read the Zenodo API** for {latest.url}: {error}") return - new_id = str(newest.get("id", "")) served = _record_file(newest, link.key) if served is None: report.ok = False @@ -326,15 +297,8 @@ def _check_zenodo( report.lines.append(f"- unchanged; Zenodo record {new_id} is still the latest version") return - date = _record_date(newest) - existing = metadata.versions.get(date) - if existing is not None and existing.checksum != checksum: - report.ok = False - report.lines.append( - f"- **version `{date}` already exists with `{existing.checksum}`**, and Zenodo record " - f"{new_id} serves `{checksum}`. Two states of this file share one date; file the " - "earlier one under the date it was published." - ) + date = published.strftime("%y%m%d") + if _date_taken(report, metadata, date, checksum): return url = link.file_url(new_id, str(served.get("key") or link.key)) report.lines.append( @@ -342,12 +306,9 @@ def _check_zenodo( f"{link.record_id}, and the new record serves `{checksum}` " f"({format_size(size_bytes)}), filed as version `{date}`" ) - entry["latest"] = {"url": url, "checksum": checksum, "size_bytes": size_bytes} - entry.setdefault("versions", {})[date] = { - "url": url, - "checksum": checksum, - "size_bytes": size_bytes, - } + pin = {"url": url, "checksum": checksum, "size_bytes": size_bytes} + entry["latest"] = pin + entry["versions"][date] = dict(pin) # a copy, or the YAML would use an anchor report.changed = True @@ -368,7 +329,7 @@ def _backfill( date for date, version in sorted(metadata.versions.items()) if version.checksum == served.checksum - and not is_archived(version.url, options.archive_tag) + and not version.url.startswith(asset_url(options.archive_tag, "")) and zenodo_link(version.url) is None ] if not unarchived: @@ -393,14 +354,7 @@ def _new_version( ) -> None: """Record what the link serves now as a version dated today.""" date = version_date() - existing = metadata.versions.get(date) - if existing is not None and existing.checksum != served.checksum: - report.ok = False - report.lines.append( - f"- **version `{date}` already exists with `{existing.checksum}`**, and the link now " - f"serves `{served.checksum}`. Two states of this file share one date; archive the " - "earlier one by hand and file it under the date it was published." - ) + if _date_taken(report, metadata, date, served.checksum): return report.lines.append( f"- the file changed: the index has `{latest.checksum}` and the link now serves " @@ -410,7 +364,7 @@ def _new_version( _archive(report, options, served, asset) entry["latest"]["checksum"] = served.checksum entry["latest"]["size_bytes"] = served.size_bytes - entry.setdefault("versions", {})[date] = { + entry["versions"][date] = { "url": asset_url(options.archive_tag, asset), "checksum": served.checksum, "size_bytes": served.size_bytes, @@ -464,6 +418,7 @@ def _parser() -> argparse.ArgumentParser: parser.add_argument( "--index", type=Path, + default=INDEX_DIR, help=f"directory of dataset YAML to check (default {INDEX_DIR})", ) parser.add_argument("--summary", type=Path, help="write a markdown report of the run here") @@ -490,12 +445,10 @@ def main(argv: list[str] | None = None) -> int: lines: list[str] = [] ok = True - for path in index_files(args.index): + for path in dataset_files(args.index): report = check_file(path, options) ok &= report.ok lines += report.lines - if not lines: - lines = ["No weights families in the index."] summary = "\n".join(lines) print(summary) diff --git a/.github/scripts/fill_download_fields.py b/.github/scripts/fill_download_fields.py index 954e626..aa0f6d5 100644 --- a/.github/scripts/fill_download_fields.py +++ b/.github/scripts/fill_download_fields.py @@ -30,7 +30,7 @@ from emdatabase.new_dataset import build_document, fill_download_fields, write_document -def index_files(index_dir: Path | None, paths: list[Path]) -> list[Path]: +def index_files(index_dir: Path, paths: list[Path]) -> list[Path]: """Every dataset YAML to fill in: the ones named, or a whole directory. ``vendors.yaml`` and the rest of ``index/`` are not dataset collections, so @@ -40,9 +40,7 @@ def index_files(index_dir: Path | None, paths: list[Path]) -> list[Path]: if paths: # A pull request that removes an entry names a file that is gone. return [path for path in paths if path.name not in NON_DATASET_FILES and path.exists()] - if index_dir is None: - return dataset_files() - return sorted(p for p in index_dir.rglob("*.y*ml") if p.name not in NON_DATASET_FILES) + return dataset_files(index_dir) def fill_file(path: Path) -> tuple[list[str], bool]: @@ -80,6 +78,7 @@ def _parser() -> argparse.ArgumentParser: parser.add_argument( "--index", type=Path, + default=INDEX_DIR, help=f"directory of dataset YAML to fill in (default {INDEX_DIR})", ) parser.add_argument("--summary", type=Path, help="write a markdown report of the run here") diff --git a/.github/scripts/issue_to_yaml.py b/.github/scripts/issue_to_yaml.py index 603650c..3c3fed2 100644 --- a/.github/scripts/issue_to_yaml.py +++ b/.github/scripts/issue_to_yaml.py @@ -139,7 +139,9 @@ def build_yaml(data): source, filename, link = split_url(url) if not source: sys.exit(f"{data['URL']!r} is not a link to a file") - filename = data["File Name"] or filename + # Whoever opened the issue typed this, and it is joined onto a directory + # later (check_latest_weights.py), so keep the name and nothing else. + filename = Path(data["File Name"] or filename).name if not filename: sys.exit(f"{data['URL']!r} does not end in a file name; fill in --File Name--") # The issue may already carry the size; without it the server is asked for @@ -166,14 +168,12 @@ def build_yaml(data): } entry["authors"], problems = parse_authors(data["Authors"]) if data["Kind"] == "weights": - entry["kind"] = "weights" model = { "class": data["Model Class"], "framework": data["Model Framework"], "quantem": data["Model quantem"], } - entry["model"] = {k: v for k, v in model.items() if v} - entry = as_weights_family(entry, data["Version Date"] or version_date()) + entry = as_weights_family(entry, data["Version Date"] or version_date(), model) return build_document(name, entry), name, problems @@ -181,6 +181,9 @@ def write_yaml(issue_file, out_dir): """Parse one issue body and write the entry it describes into ``out_dir``.""" document, dataset_name, problems = build_yaml(parse_issue_body(Path(issue_file).read_text())) out_path = Path(out_dir) / f"{dataset_name}.yaml" + if out_path.exists(): + # Anyone can open an issue; the pull request it opens adds, never replaces. + problems.append(f"{out_path} already exists; choose another --Dataset Name--") # Nothing is downloaded for an issue that is already known to be wrong. if not problems: for line in fill_download_fields(document): diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 2d9a95c..c9dada5 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -38,7 +38,7 @@ jobs: run: python -m emdatabase._create_stubs --check build-with-pip: - name: ${{ matrix.os }}-py${{ matrix.python-version }}${{ matrix.LABEL }} + name: ${{ matrix.os }}-py${{ matrix.python-version }} runs-on: ${{ matrix.os }} timeout-minutes: 15 strategy: @@ -47,35 +47,18 @@ jobs: os: [ubuntu-latest, windows-latest, macos-latest] python-version: ["3.12", "3.13"] steps: - - uses: actions/checkout@v3 + - uses: actions/checkout@v4 - name: Set up Python ${{ matrix.python-version }} - uses: actions/setup-python@v4 + uses: actions/setup-python@v5 with: python-version: ${{ matrix.python-version }} - - name: Get the number of CPUs - id: cpus - run: | - import os, platform - num_cpus = os.cpu_count() - print(f"Number of CPU: {num_cpus}") - print(f"Architecture: {platform.machine()}") - output_file = os.environ["GITHUB_OUTPUT"] - with open(output_file, "a", encoding="utf-8") as output_stream: - output_stream.write(f"count={num_cpus}\n") - shell: python - - name: Install dependencies and package shell: bash run: | pip install -U -e .'[tests]' - - name: Install oldest supported versions - if: contains(matrix.LABEL, 'oldest') - run: | - pip install ${{ matrix.DEPENDENCIES }} - - name: Display Python, pip and package versions run: | python -V @@ -83,7 +66,6 @@ jobs: pip list - name: Run docstring tests - continue-on-error: true run: | pytest --doctest-modules --doctest-continue-on-failure --ignore-glob=emdatabase/tests emdatabase diff --git a/.github/workflows/check_weights.yml b/.github/workflows/check_weights.yml index 3899318..c271098 100644 --- a/.github/workflows/check_weights.yml +++ b/.github/workflows/check_weights.yml @@ -29,7 +29,7 @@ jobs: - name: Follow each latest link env: - GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} run: | python .github/scripts/check_latest_weights.py \ --summary changes.md --keep-dir oversize diff --git a/.github/workflows/documentation.yaml b/.github/workflows/documentation.yaml index 375e3da..c641a87 100644 --- a/.github/workflows/documentation.yaml +++ b/.github/workflows/documentation.yaml @@ -47,6 +47,7 @@ jobs: - name: Sphinx build timeout-minutes: 15 run: | + set -o pipefail make html -C docs 2>&1 | tee sphinx.log # A failed gallery example only warns (see conf.py), so say so where a diff --git a/.github/workflows/fill_download_fields.yml b/.github/workflows/fill_download_fields.yml index 73e7d8a..dcb3bb6 100644 --- a/.github/workflows/fill_download_fields.yml +++ b/.github/workflows/fill_download_fields.yml @@ -1,8 +1,8 @@ name: fill download fields -# The docs form and the issue form both let the checksum and the size be blank, -# because a contributor cannot be asked to md5 a 100 GB file by hand, while the -# test suite requires both on every entry. This job downloads whatever a pull +# The issue form lets the checksum and the size be blank, because a contributor +# cannot be asked to md5 a 100 GB file by hand, while the test suite requires +# both on every entry. This job downloads whatever a pull # request left blank, fills it in and pushes the result back to the branch. A # fork's branch cannot be pushed to, so a pull request from one is failed with # the values it would have written instead. diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml index cd3dc2b..484e8cf 100644 --- a/.github/workflows/publish.yml +++ b/.github/workflows/publish.yml @@ -48,16 +48,17 @@ jobs: sys.exit("json-schema.json missing from the wheel") PY - - name: Check the version matches the release tag - if: github.event_name == 'release' + - name: Check the version is a release, and matches the tag run: | python - <<'PY' - import os, sys, tomllib + import os, re, sys, tomllib with open("pyproject.toml", "rb") as f: version = tomllib.load(f)["project"]["version"] tag = os.environ["TAG"].lstrip("v") print(f"pyproject version {version!r}, release tag {tag!r}") - if version != tag: + if not re.fullmatch(r"\d+\.\d+\.\d+", version): + sys.exit(f"{version} is not a release version, and PyPI never frees one up again") + if tag and version != tag: sys.exit(f"version mismatch: pyproject says {version}, tag says {tag}") PY env: diff --git a/docs/source/_build_docs.py b/docs/source/_build_docs.py index e06ace3..dcdb5c4 100644 --- a/docs/source/_build_docs.py +++ b/docs/source/_build_docs.py @@ -1,634 +1,104 @@ +"""The docs site's app pages, written over Sphinx's output by ``conf.py``. + +Every page is self-contained: the widget's CSS and shared JS are inlined, the +catalogue is baked in as JSON at build time, and nothing is loaded from +outside, so search and the list work with no backend. +""" + import json -from collections import defaultdict +from html import escape from importlib import resources from pathlib import Path -import yaml +from emdatabase.metadata import acquisition_techniques, versioned_filename -from emdatabase.metadata import NON_DATASET_FILES, acquisition_techniques - - -def parse_datasets(yaml_dir): - """Parse all YAML files and organize by technique. - - An entry may declare several techniques, in which case it is listed under - each of them; ``techniques`` on the record is all of them, so the table can - still draw it as one row. - """ - datasets_by_technique = defaultdict(list) - - for yaml_file in sorted(Path(yaml_dir).glob("*.yaml")): - if yaml_file.name in NON_DATASET_FILES: - continue - with open(yaml_file, "r") as f: - data = yaml.safe_load(f) - - for name, info in data.items(): - if info.get("kind") == "weights": - continue # the Model Weights page, not this one - techniques = info.get("technique") or ["Unknown"] - if isinstance(techniques, str): - techniques = [techniques] - record = { - "name": name, - "techniques": list(techniques), - "description": info.get("description", ""), - "tags": info.get("tags", []), - "source": info.get("source", ""), - "file": info.get("file", ""), - "license": info.get("license", ""), - "detector": info.get("detector", "Unknown"), - "detector_manufacturer": info.get("detector_manufacturer", "Unknown"), - } - for technique in techniques: - datasets_by_technique[technique].append(record) - - return dict(datasets_by_technique) - - -def generate_html_table(datasets_by_technique): - """Generate HTML with filterable table and technique tabs.""" - from emdatabase import catalogue - - all_tags = set() - all_detectors = {} # Changed to dict: {manufacturer: [detectors]} - technique_tags = {} - technique_detectors = {} - - for technique, datasets in datasets_by_technique.items(): - tags = set() - detectors = {} - for dataset in datasets: - tags.update(dataset["tags"]) - all_tags.update(dataset["tags"]) - manufacturer = dataset.get("detector_manufacturer", "Unknown") - detector = dataset.get("detector", "Unknown") - - if manufacturer not in detectors: - detectors[manufacturer] = set() - detectors[manufacturer].add(detector) - - if manufacturer not in all_detectors: - all_detectors[manufacturer] = set() - all_detectors[manufacturer].add(detector) - - technique_tags[technique] = sorted(tags) - technique_detectors[technique] = {m: sorted(d) for m, d in detectors.items()} - - all_detectors = {m: sorted(d) for m, d in all_detectors.items()} - - technique_tabs_json = json.dumps(catalogue.ordered_groups(datasets_by_technique)) - technique_tags_json = __import__("json").dumps(technique_tags) - technique_detectors_json = __import__("json").dumps(technique_detectors) - all_tags_sorted = sorted(all_tags) - all_detectors_json = __import__("json").dumps(all_detectors) - - html = """ - - -
- - - - -| Technique | -Dataset | -Description | -
- Tags
-
-
-
-
- |
-
- Detector
-
-
-
-
- |
- File | -License | -
|---|---|---|---|---|---|---|
| {techniques_str} | -{dataset["name"]} | -{dataset["description"]} | -{tags_str} | -{detector_full} | -{dataset["file"]} | -{dataset["license"]} | -