Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
107 changes: 30 additions & 77 deletions .github/scripts/check_latest_weights.py
Original file line number Diff line number Diff line change
Expand Up @@ -36,7 +36,6 @@
import argparse
import datetime
import json
import os
import re
import shutil
import subprocess
Expand All @@ -51,7 +50,6 @@

from emdatabase.metadata import (
INDEX_DIR,
NON_DATASET_FILES,
DatasetMetadata,
WeightsVersion,
dataset_files,
Expand All @@ -77,11 +75,6 @@ def asset_url(tag: str, asset: str) -> str:
return f"{RELEASE_DOWNLOADS}/{tag}/{asset}"


def is_archived(url: str, tag: str) -> bool:
"""Whether a version's link already points into the archive release."""
return url.startswith(f"{RELEASE_DOWNLOADS}/{tag}/")


@dataclass(frozen=True)
class ZenodoLink:
"""A Zenodo record file link, split into the pieces the API needs."""
Expand Down Expand Up @@ -118,23 +111,9 @@ def fetch_json(url: str) -> dict[str, Any]:
return json.loads(response.read().decode("utf-8"))


def index_files(index_dir: Path | None) -> list[Path]:
"""Every dataset YAML to check, sorted by name."""
if index_dir is None:
return dataset_files()
return sorted(p for p in index_dir.rglob("*.y*ml") if p.name not in NON_DATASET_FILES)


def run_gh(args: list[str], *, check: bool = True) -> subprocess.CompletedProcess[str]:
"""Run one ``gh`` command; the only place this script shells out.

``gh`` reads ``GH_TOKEN``, while a workflow hands the job ``GITHUB_TOKEN``.
"""
env = os.environ.copy()
token = env.get("GH_TOKEN") or env.get("GITHUB_TOKEN")
if token:
env["GH_TOKEN"] = token
return subprocess.run(["gh", *args], check=check, capture_output=True, text=True, env=env)
"""Run one ``gh`` command; the only place this script shells out."""
return subprocess.run(["gh", *args], check=check, capture_output=True, text=True)


def upload_asset(tag: str, path: Path, asset: str) -> None:
Expand Down Expand Up @@ -212,9 +191,7 @@ def check_family(
"""Download one family's ``latest`` and update ``entry`` in place."""
report = Report(lines=[f"### {name}"])
latest = metadata.latest
if latest is None:
report.lines.append("- no `latest` link, so there is nothing to follow")
return report
assert latest is not None # the schema requires one of every weights entry

link = zenodo_link(latest.url)
if link is not None:
Expand Down Expand Up @@ -250,19 +227,6 @@ def check_family(
return report


def _newest_record(link: ZenodoLink, record: dict[str, Any]) -> dict[str, Any]:
"""The newest record of this concept.

``links.latest`` is what Zenodo publishes for it; the concept record id is
the fallback, which the API resolves to the same place.
"""
follow = (record.get("links") or {}).get("latest")
if not follow:
concept = record.get("conceptrecid")
follow = f"{link.api}/{concept}" if concept else None
return fetch_json(str(follow)) if follow else record


def _record_file(record: dict[str, Any], key: str) -> dict[str, Any] | None:
"""The file entry named ``key``, or the only one when the name has changed."""
files = [item for item in record.get("files") or [] if isinstance(item, dict)]
Expand All @@ -272,13 +236,18 @@ def _record_file(record: dict[str, Any], key: str) -> dict[str, Any] | None:
return files[0] if len(files) == 1 else None


def _record_date(record: dict[str, Any]) -> str:
"""The record's publication date as ``YYMMDD``, or today's if it has none."""
published = str((record.get("metadata") or {}).get("publication_date", ""))
try:
return datetime.date.fromisoformat(published[:10]).strftime("%y%m%d")
except ValueError:
return version_date()
def _date_taken(report: Report, metadata: DatasetMetadata, date: str, checksum: str) -> bool:
"""Whether version ``date`` already holds a different file; a failure if it does."""
existing = metadata.versions.get(date)
if existing is None or existing.checksum == checksum:
return False
report.ok = False
report.lines.append(
f"- **version `{date}` already exists with `{existing.checksum}`**, and the newest file "
f"is `{checksum}`. Two states of this file share one date; refile the earlier one by "
"hand under the date it was published."
)
return True


def _check_zenodo(
Expand All @@ -295,13 +264,15 @@ def _check_zenodo(
version written for it points straight at that record.
"""
try:
newest = _newest_record(link, fetch_json(f"{link.api}/{link.record_id}"))
except (OSError, ValueError) as error:
record = fetch_json(f"{link.api}/{link.record_id}")
newest = fetch_json(record["links"]["latest"])
new_id = str(newest["id"])
published = datetime.date.fromisoformat(newest["metadata"]["publication_date"][:10])
except (OSError, KeyError, ValueError) as error:
report.ok = False
report.lines.append(f"- **could not read the Zenodo API** for {latest.url}: {error}")
return

new_id = str(newest.get("id", ""))
served = _record_file(newest, link.key)
if served is None:
report.ok = False
Expand All @@ -326,28 +297,18 @@ def _check_zenodo(
report.lines.append(f"- unchanged; Zenodo record {new_id} is still the latest version")
return

date = _record_date(newest)
existing = metadata.versions.get(date)
if existing is not None and existing.checksum != checksum:
report.ok = False
report.lines.append(
f"- **version `{date}` already exists with `{existing.checksum}`**, and Zenodo record "
f"{new_id} serves `{checksum}`. Two states of this file share one date; file the "
"earlier one under the date it was published."
)
date = published.strftime("%y%m%d")
if _date_taken(report, metadata, date, checksum):
return
url = link.file_url(new_id, str(served.get("key") or link.key))
report.lines.append(
f"- new Zenodo record {new_id}: the index has `{latest.checksum}` from record "
f"{link.record_id}, and the new record serves `{checksum}` "
f"({format_size(size_bytes)}), filed as version `{date}`"
)
entry["latest"] = {"url": url, "checksum": checksum, "size_bytes": size_bytes}
entry.setdefault("versions", {})[date] = {
"url": url,
"checksum": checksum,
"size_bytes": size_bytes,
}
pin = {"url": url, "checksum": checksum, "size_bytes": size_bytes}
entry["latest"] = pin
entry["versions"][date] = dict(pin) # a copy, or the YAML would use an anchor
report.changed = True


Expand All @@ -368,7 +329,7 @@ def _backfill(
date
for date, version in sorted(metadata.versions.items())
if version.checksum == served.checksum
and not is_archived(version.url, options.archive_tag)
and not version.url.startswith(asset_url(options.archive_tag, ""))
and zenodo_link(version.url) is None
]
if not unarchived:
Expand All @@ -393,14 +354,7 @@ def _new_version(
) -> None:
"""Record what the link serves now as a version dated today."""
date = version_date()
existing = metadata.versions.get(date)
if existing is not None and existing.checksum != served.checksum:
report.ok = False
report.lines.append(
f"- **version `{date}` already exists with `{existing.checksum}`**, and the link now "
f"serves `{served.checksum}`. Two states of this file share one date; archive the "
"earlier one by hand and file it under the date it was published."
)
if _date_taken(report, metadata, date, served.checksum):
return
report.lines.append(
f"- the file changed: the index has `{latest.checksum}` and the link now serves "
Expand All @@ -410,7 +364,7 @@ def _new_version(
_archive(report, options, served, asset)
entry["latest"]["checksum"] = served.checksum
entry["latest"]["size_bytes"] = served.size_bytes
entry.setdefault("versions", {})[date] = {
entry["versions"][date] = {
"url": asset_url(options.archive_tag, asset),
"checksum": served.checksum,
"size_bytes": served.size_bytes,
Expand Down Expand Up @@ -464,6 +418,7 @@ def _parser() -> argparse.ArgumentParser:
parser.add_argument(
"--index",
type=Path,
default=INDEX_DIR,
help=f"directory of dataset YAML to check (default {INDEX_DIR})",
)
parser.add_argument("--summary", type=Path, help="write a markdown report of the run here")
Expand All @@ -490,12 +445,10 @@ def main(argv: list[str] | None = None) -> int:

lines: list[str] = []
ok = True
for path in index_files(args.index):
for path in dataset_files(args.index):
report = check_file(path, options)
ok &= report.ok
lines += report.lines
if not lines:
lines = ["No weights families in the index."]

summary = "\n".join(lines)
print(summary)
Expand Down
7 changes: 3 additions & 4 deletions .github/scripts/fill_download_fields.py
Original file line number Diff line number Diff line change
Expand Up @@ -30,7 +30,7 @@
from emdatabase.new_dataset import build_document, fill_download_fields, write_document


def index_files(index_dir: Path | None, paths: list[Path]) -> list[Path]:
def index_files(index_dir: Path, paths: list[Path]) -> list[Path]:
"""Every dataset YAML to fill in: the ones named, or a whole directory.

``vendors.yaml`` and the rest of ``index/`` are not dataset collections, so
Expand All @@ -40,9 +40,7 @@ def index_files(index_dir: Path | None, paths: list[Path]) -> list[Path]:
if paths:
# A pull request that removes an entry names a file that is gone.
return [path for path in paths if path.name not in NON_DATASET_FILES and path.exists()]
if index_dir is None:
return dataset_files()
return sorted(p for p in index_dir.rglob("*.y*ml") if p.name not in NON_DATASET_FILES)
return dataset_files(index_dir)


def fill_file(path: Path) -> tuple[list[str], bool]:
Expand Down Expand Up @@ -80,6 +78,7 @@ def _parser() -> argparse.ArgumentParser:
parser.add_argument(
"--index",
type=Path,
default=INDEX_DIR,
help=f"directory of dataset YAML to fill in (default {INDEX_DIR})",
)
parser.add_argument("--summary", type=Path, help="write a markdown report of the run here")
Expand Down
11 changes: 7 additions & 4 deletions .github/scripts/issue_to_yaml.py
Original file line number Diff line number Diff line change
Expand Up @@ -139,7 +139,9 @@ def build_yaml(data):
source, filename, link = split_url(url)
if not source:
sys.exit(f"{data['URL']!r} is not a link to a file")
filename = data["File Name"] or filename
# Whoever opened the issue typed this, and it is joined onto a directory
# later (check_latest_weights.py), so keep the name and nothing else.
filename = Path(data["File Name"] or filename).name
if not filename:
sys.exit(f"{data['URL']!r} does not end in a file name; fill in --File Name--")
# The issue may already carry the size; without it the server is asked for
Expand All @@ -166,21 +168,22 @@ def build_yaml(data):
}
entry["authors"], problems = parse_authors(data["Authors"])
if data["Kind"] == "weights":
entry["kind"] = "weights"
model = {
"class": data["Model Class"],
"framework": data["Model Framework"],
"quantem": data["Model quantem"],
}
entry["model"] = {k: v for k, v in model.items() if v}
entry = as_weights_family(entry, data["Version Date"] or version_date())
entry = as_weights_family(entry, data["Version Date"] or version_date(), model)
return build_document(name, entry), name, problems


def write_yaml(issue_file, out_dir):
"""Parse one issue body and write the entry it describes into ``out_dir``."""
document, dataset_name, problems = build_yaml(parse_issue_body(Path(issue_file).read_text()))
out_path = Path(out_dir) / f"{dataset_name}.yaml"
if out_path.exists():
# Anyone can open an issue; the pull request it opens adds, never replaces.
problems.append(f"{out_path} already exists; choose another --Dataset Name--")
# Nothing is downloaded for an issue that is already known to be wrong.
if not problems:
for line in fill_download_fields(document):
Expand Down
24 changes: 3 additions & 21 deletions .github/workflows/build.yml
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,7 @@ jobs:
run: python -m emdatabase._create_stubs --check

build-with-pip:
name: ${{ matrix.os }}-py${{ matrix.python-version }}${{ matrix.LABEL }}
name: ${{ matrix.os }}-py${{ matrix.python-version }}
runs-on: ${{ matrix.os }}
timeout-minutes: 15
strategy:
Expand All @@ -47,43 +47,25 @@ jobs:
os: [ubuntu-latest, windows-latest, macos-latest]
python-version: ["3.12", "3.13"]
steps:
- uses: actions/checkout@v3
- uses: actions/checkout@v4

- name: Set up Python ${{ matrix.python-version }}
uses: actions/setup-python@v4
uses: actions/setup-python@v5
with:
python-version: ${{ matrix.python-version }}

- name: Get the number of CPUs
id: cpus
run: |
import os, platform
num_cpus = os.cpu_count()
print(f"Number of CPU: {num_cpus}")
print(f"Architecture: {platform.machine()}")
output_file = os.environ["GITHUB_OUTPUT"]
with open(output_file, "a", encoding="utf-8") as output_stream:
output_stream.write(f"count={num_cpus}\n")
shell: python

- name: Install dependencies and package
shell: bash
run: |
pip install -U -e .'[tests]'

- name: Install oldest supported versions
if: contains(matrix.LABEL, 'oldest')
run: |
pip install ${{ matrix.DEPENDENCIES }}

- name: Display Python, pip and package versions
run: |
python -V
pip -V
pip list

- name: Run docstring tests
continue-on-error: true
run: |
pytest --doctest-modules --doctest-continue-on-failure --ignore-glob=emdatabase/tests emdatabase

Expand Down
2 changes: 1 addition & 1 deletion .github/workflows/check_weights.yml
Original file line number Diff line number Diff line change
Expand Up @@ -29,7 +29,7 @@ jobs:

- name: Follow each latest link
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
python .github/scripts/check_latest_weights.py \
--summary changes.md --keep-dir oversize
Expand Down
1 change: 1 addition & 0 deletions .github/workflows/documentation.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -47,6 +47,7 @@ jobs:
- name: Sphinx build
timeout-minutes: 15
run: |
set -o pipefail
make html -C docs 2>&1 | tee sphinx.log

# A failed gallery example only warns (see conf.py), so say so where a
Expand Down
6 changes: 3 additions & 3 deletions .github/workflows/fill_download_fields.yml
Original file line number Diff line number Diff line change
@@ -1,8 +1,8 @@
name: fill download fields

# The docs form and the issue form both let the checksum and the size be blank,
# because a contributor cannot be asked to md5 a 100 GB file by hand, while the
# test suite requires both on every entry. This job downloads whatever a pull
# The issue form lets the checksum and the size be blank, because a contributor
# cannot be asked to md5 a 100 GB file by hand, while the test suite requires
# both on every entry. This job downloads whatever a pull
# request left blank, fills it in and pushes the result back to the branch. A
# fork's branch cannot be pushed to, so a pull request from one is failed with
# the values it would have written instead.
Expand Down
9 changes: 5 additions & 4 deletions .github/workflows/publish.yml
Original file line number Diff line number Diff line change
Expand Up @@ -48,16 +48,17 @@ jobs:
sys.exit("json-schema.json missing from the wheel")
PY

- name: Check the version matches the release tag
if: github.event_name == 'release'
- name: Check the version is a release, and matches the tag
run: |
python - <<'PY'
import os, sys, tomllib
import os, re, sys, tomllib
with open("pyproject.toml", "rb") as f:
version = tomllib.load(f)["project"]["version"]
tag = os.environ["TAG"].lstrip("v")
print(f"pyproject version {version!r}, release tag {tag!r}")
if version != tag:
if not re.fullmatch(r"\d+\.\d+\.\d+", version):
sys.exit(f"{version} is not a release version, and PyPI never frees one up again")
if tag and version != tag:
sys.exit(f"version mismatch: pyproject says {version}, tag says {tag}")
PY
env:
Expand Down
Loading
Loading