From b302d200ab338dbf900092bd98d77895e57d5f7d Mon Sep 17 00:00:00 2001 From: Ohad Mosafi Date: Wed, 9 Sep 2026 18:11:58 -0700 Subject: [PATCH 1/4] restructure skills files Signed-off-by: Ohad Mosafi --- .claude/skills | 2 +- .github/workflows/skill-packaging.yml | 28 + AGENTS.md | 12 +- Dockerfile | 2 +- {agent => skills}/README.md | 115 ++- .../_shared}/config/defaults_embed.json | 0 .../_shared}/config/defaults_finetune.json | 0 .../_shared}/config/defaults_inference.json | 0 .../_shared}/config/defaults_pretrain.json | 0 .../_shared}/config/released_model.json | 0 skills/_shared/references/released-models.md | 34 + {agent => skills/_shared}/scripts/_utils.py | 28 +- .../_shared}/scripts/check_checkpoint.py | 0 .../_shared}/scripts/check_data.py | 0 .../_shared}/scripts/fetch_released_model.py | 10 +- .../_shared}/scripts/kermt_container.sh | 45 +- .../_shared}/scripts/prepare_data.py | 6 +- .../scripts/run_extract_embeddings.py | 10 +- .../_shared}/scripts/run_finetune_local.py | 10 +- .../_shared}/scripts/run_inference.py | 10 +- .../_shared}/scripts/run_pretrain_local.py | 16 +- .../_shared}/scripts/upgrade_to_hybrid.py | 11 +- skills/_shared/sync_shared.py | 107 +++ .../kermt-add-cmim-pretrain/SKILL.md | 20 +- .../config/defaults_pretrain.json | 52 ++ .../kermt-add-cmim-pretrain/evals/evals.json | 0 .../kermt-add-cmim-pretrain/scripts/_utils.py | 272 ++++++ .../scripts/check_checkpoint.py | 480 ++++++++++ .../scripts/check_data.py | 310 +++++++ .../scripts/kermt_container.sh | 484 +++++++++++ .../scripts/prepare_data.py | 817 ++++++++++++++++++ .../scripts/run_pretrain_local.py | 730 ++++++++++++++++ .../scripts/upgrade_to_hybrid.py | 393 +++++++++ .../kermt-add-cmim-pretrain/skill-card.md | 2 +- .../kermt-continue-pretrain/SKILL.md | 37 +- .../config/defaults_pretrain.json | 52 ++ .../config/released_model.json | 13 + .../kermt-continue-pretrain/evals/evals.json | 0 .../references/released-models.md | 34 + .../kermt-continue-pretrain/scripts/_utils.py | 272 ++++++ .../scripts/check_checkpoint.py | 480 ++++++++++ .../scripts/check_data.py | 310 +++++++ .../scripts/fetch_released_model.py | 234 +++++ .../scripts/kermt_container.sh | 484 +++++++++++ .../scripts/prepare_data.py | 817 ++++++++++++++++++ .../scripts/run_pretrain_local.py | 730 ++++++++++++++++ .../kermt-continue-pretrain/skill-card.md | 2 +- {agent/skills => skills}/kermt-embed/SKILL.md | 31 +- skills/kermt-embed/config/defaults_embed.json | 10 + skills/kermt-embed/config/released_model.json | 13 + .../kermt-embed/evals/evals.json | 0 .../kermt-embed/references/released-models.md | 34 + skills/kermt-embed/scripts/_utils.py | 272 ++++++ .../kermt-embed/scripts/check_checkpoint.py | 480 ++++++++++ skills/kermt-embed/scripts/check_data.py | 310 +++++++ .../scripts/fetch_released_model.py | 234 +++++ skills/kermt-embed/scripts/kermt_container.sh | 484 +++++++++++ skills/kermt-embed/scripts/prepare_data.py | 817 ++++++++++++++++++ .../scripts/run_extract_embeddings.py | 221 +++++ .../kermt-embed/skill-card.md | 0 .../skills => skills}/kermt-finetune/SKILL.md | 35 +- .../config/defaults_finetune.json | 41 + .../kermt-finetune/config/released_model.json | 13 + .../kermt-finetune/evals/evals.json | 0 .../references/released-models.md | 34 + skills/kermt-finetune/scripts/_utils.py | 272 ++++++ .../scripts/check_checkpoint.py | 480 ++++++++++ skills/kermt-finetune/scripts/check_data.py | 310 +++++++ .../scripts/fetch_released_model.py | 234 +++++ .../kermt-finetune/scripts/kermt_container.sh | 484 +++++++++++ skills/kermt-finetune/scripts/prepare_data.py | 817 ++++++++++++++++++ .../scripts/run_finetune_local.py | 468 ++++++++++ .../kermt-finetune/skill-card.md | 4 +- {agent/skills => skills}/kermt-infer/SKILL.md | 26 +- .../config/defaults_inference.json | 11 + .../kermt-infer/evals/evals.json | 0 skills/kermt-infer/scripts/_utils.py | 272 ++++++ .../kermt-infer/scripts/check_checkpoint.py | 480 ++++++++++ skills/kermt-infer/scripts/check_data.py | 310 +++++++ skills/kermt-infer/scripts/kermt_container.sh | 484 +++++++++++ skills/kermt-infer/scripts/prepare_data.py | 817 ++++++++++++++++++ skills/kermt-infer/scripts/run_inference.py | 261 ++++++ .../kermt-infer/skill-card.md | 0 .../skills => skills}/kermt-monitor/SKILL.md | 0 .../kermt-monitor/evals/evals.json | 0 .../kermt-monitor/skill-card.md | 1 - .../kermt-pretrain-scratch/SKILL.md | 26 +- .../config/defaults_pretrain.json | 52 ++ .../kermt-pretrain-scratch/evals/evals.json | 0 .../kermt-pretrain-scratch/scripts/_utils.py | 272 ++++++ .../scripts/check_checkpoint.py | 480 ++++++++++ .../scripts/check_data.py | 310 +++++++ .../scripts/kermt_container.sh | 484 +++++++++++ .../scripts/prepare_data.py | 817 ++++++++++++++++++ .../scripts/run_pretrain_local.py | 730 ++++++++++++++++ .../kermt-pretrain-scratch/skill-card.md | 2 +- {agent/skills => skills}/kermt-setup/SKILL.md | 24 +- .../kermt-setup/evals/evals.json | 6 +- skills/kermt-setup/scripts/kermt_container.sh | 484 +++++++++++ .../kermt-setup/skill-card.md | 2 +- .../skills}/_build_fake_ckpt.py | 0 {agent/tests => tests/skills}/conftest.py | 8 +- .../skills}/test_check_checkpoint.py | 4 +- .../tests => tests/skills}/test_check_data.py | 4 +- .../skills}/test_e2e_released_download.py | 4 +- .../skills}/test_fetch_released_model.py | 4 +- .../skills}/test_prepare_data.py | 6 +- .../skills}/test_run_extract_embeddings.py | 8 +- .../skills}/test_run_finetune_local.py | 6 +- .../skills}/test_run_inference.py | 8 +- .../skills}/test_run_pretrain_local.py | 12 +- .../skills}/test_skill_frontmatter.py | 8 +- tests/skills/test_skill_packaging.py | 177 ++++ .../skills}/test_upgrade_to_hybrid.py | 14 +- 114 files changed, 19947 insertions(+), 236 deletions(-) create mode 100644 .github/workflows/skill-packaging.yml rename {agent => skills}/README.md (78%) rename {agent => skills/_shared}/config/defaults_embed.json (100%) rename {agent => skills/_shared}/config/defaults_finetune.json (100%) rename {agent => skills/_shared}/config/defaults_inference.json (100%) rename {agent => skills/_shared}/config/defaults_pretrain.json (100%) rename {agent => skills/_shared}/config/released_model.json (100%) create mode 100644 skills/_shared/references/released-models.md rename {agent => skills/_shared}/scripts/_utils.py (90%) rename {agent => skills/_shared}/scripts/check_checkpoint.py (100%) rename {agent => skills/_shared}/scripts/check_data.py (100%) rename {agent => skills/_shared}/scripts/fetch_released_model.py (95%) rename {agent => skills/_shared}/scripts/kermt_container.sh (91%) rename {agent => skills/_shared}/scripts/prepare_data.py (99%) rename {agent => skills/_shared}/scripts/run_extract_embeddings.py (96%) rename {agent => skills/_shared}/scripts/run_finetune_local.py (98%) rename {agent => skills/_shared}/scripts/run_inference.py (97%) rename {agent => skills/_shared}/scripts/run_pretrain_local.py (98%) rename {agent => skills/_shared}/scripts/upgrade_to_hybrid.py (98%) create mode 100644 skills/_shared/sync_shared.py rename {agent/skills => skills}/kermt-add-cmim-pretrain/SKILL.md (90%) create mode 100644 skills/kermt-add-cmim-pretrain/config/defaults_pretrain.json rename {agent/skills => skills}/kermt-add-cmim-pretrain/evals/evals.json (100%) create mode 100644 skills/kermt-add-cmim-pretrain/scripts/_utils.py create mode 100644 skills/kermt-add-cmim-pretrain/scripts/check_checkpoint.py create mode 100644 skills/kermt-add-cmim-pretrain/scripts/check_data.py create mode 100755 skills/kermt-add-cmim-pretrain/scripts/kermt_container.sh create mode 100644 skills/kermt-add-cmim-pretrain/scripts/prepare_data.py create mode 100644 skills/kermt-add-cmim-pretrain/scripts/run_pretrain_local.py create mode 100644 skills/kermt-add-cmim-pretrain/scripts/upgrade_to_hybrid.py rename {agent/skills => skills}/kermt-add-cmim-pretrain/skill-card.md (98%) rename {agent/skills => skills}/kermt-continue-pretrain/SKILL.md (91%) create mode 100644 skills/kermt-continue-pretrain/config/defaults_pretrain.json create mode 100644 skills/kermt-continue-pretrain/config/released_model.json rename {agent/skills => skills}/kermt-continue-pretrain/evals/evals.json (100%) create mode 100644 skills/kermt-continue-pretrain/references/released-models.md create mode 100644 skills/kermt-continue-pretrain/scripts/_utils.py create mode 100644 skills/kermt-continue-pretrain/scripts/check_checkpoint.py create mode 100644 skills/kermt-continue-pretrain/scripts/check_data.py create mode 100644 skills/kermt-continue-pretrain/scripts/fetch_released_model.py create mode 100755 skills/kermt-continue-pretrain/scripts/kermt_container.sh create mode 100644 skills/kermt-continue-pretrain/scripts/prepare_data.py create mode 100644 skills/kermt-continue-pretrain/scripts/run_pretrain_local.py rename {agent/skills => skills}/kermt-continue-pretrain/skill-card.md (98%) rename {agent/skills => skills}/kermt-embed/SKILL.md (82%) create mode 100644 skills/kermt-embed/config/defaults_embed.json create mode 100644 skills/kermt-embed/config/released_model.json rename {agent/skills => skills}/kermt-embed/evals/evals.json (100%) create mode 100644 skills/kermt-embed/references/released-models.md create mode 100644 skills/kermt-embed/scripts/_utils.py create mode 100644 skills/kermt-embed/scripts/check_checkpoint.py create mode 100644 skills/kermt-embed/scripts/check_data.py create mode 100644 skills/kermt-embed/scripts/fetch_released_model.py create mode 100755 skills/kermt-embed/scripts/kermt_container.sh create mode 100644 skills/kermt-embed/scripts/prepare_data.py create mode 100644 skills/kermt-embed/scripts/run_extract_embeddings.py rename {agent/skills => skills}/kermt-embed/skill-card.md (100%) rename {agent/skills => skills}/kermt-finetune/SKILL.md (91%) create mode 100644 skills/kermt-finetune/config/defaults_finetune.json create mode 100644 skills/kermt-finetune/config/released_model.json rename {agent/skills => skills}/kermt-finetune/evals/evals.json (100%) create mode 100644 skills/kermt-finetune/references/released-models.md create mode 100644 skills/kermt-finetune/scripts/_utils.py create mode 100644 skills/kermt-finetune/scripts/check_checkpoint.py create mode 100644 skills/kermt-finetune/scripts/check_data.py create mode 100644 skills/kermt-finetune/scripts/fetch_released_model.py create mode 100755 skills/kermt-finetune/scripts/kermt_container.sh create mode 100644 skills/kermt-finetune/scripts/prepare_data.py create mode 100644 skills/kermt-finetune/scripts/run_finetune_local.py rename {agent/skills => skills}/kermt-finetune/skill-card.md (96%) rename {agent/skills => skills}/kermt-infer/SKILL.md (83%) create mode 100644 skills/kermt-infer/config/defaults_inference.json rename {agent/skills => skills}/kermt-infer/evals/evals.json (100%) create mode 100644 skills/kermt-infer/scripts/_utils.py create mode 100644 skills/kermt-infer/scripts/check_checkpoint.py create mode 100644 skills/kermt-infer/scripts/check_data.py create mode 100755 skills/kermt-infer/scripts/kermt_container.sh create mode 100644 skills/kermt-infer/scripts/prepare_data.py create mode 100644 skills/kermt-infer/scripts/run_inference.py rename {agent/skills => skills}/kermt-infer/skill-card.md (100%) rename {agent/skills => skills}/kermt-monitor/SKILL.md (100%) rename {agent/skills => skills}/kermt-monitor/evals/evals.json (100%) rename {agent/skills => skills}/kermt-monitor/skill-card.md (97%) rename {agent/skills => skills}/kermt-pretrain-scratch/SKILL.md (89%) create mode 100644 skills/kermt-pretrain-scratch/config/defaults_pretrain.json rename {agent/skills => skills}/kermt-pretrain-scratch/evals/evals.json (100%) create mode 100644 skills/kermt-pretrain-scratch/scripts/_utils.py create mode 100644 skills/kermt-pretrain-scratch/scripts/check_checkpoint.py create mode 100644 skills/kermt-pretrain-scratch/scripts/check_data.py create mode 100755 skills/kermt-pretrain-scratch/scripts/kermt_container.sh create mode 100644 skills/kermt-pretrain-scratch/scripts/prepare_data.py create mode 100644 skills/kermt-pretrain-scratch/scripts/run_pretrain_local.py rename {agent/skills => skills}/kermt-pretrain-scratch/skill-card.md (98%) rename {agent/skills => skills}/kermt-setup/SKILL.md (88%) rename {agent/skills => skills}/kermt-setup/evals/evals.json (90%) create mode 100755 skills/kermt-setup/scripts/kermt_container.sh rename {agent/skills => skills}/kermt-setup/skill-card.md (97%) rename {agent/tests => tests/skills}/_build_fake_ckpt.py (100%) rename {agent/tests => tests/skills}/conftest.py (90%) rename {agent/tests => tests/skills}/test_check_checkpoint.py (99%) rename {agent/tests => tests/skills}/test_check_data.py (98%) rename {agent/tests => tests/skills}/test_e2e_released_download.py (98%) rename {agent/tests => tests/skills}/test_fetch_released_model.py (97%) rename {agent/tests => tests/skills}/test_prepare_data.py (98%) rename {agent/tests => tests/skills}/test_run_extract_embeddings.py (97%) rename {agent/tests => tests/skills}/test_run_finetune_local.py (98%) rename {agent/tests => tests/skills}/test_run_inference.py (97%) rename {agent/tests => tests/skills}/test_run_pretrain_local.py (99%) rename {agent/tests => tests/skills}/test_skill_frontmatter.py (96%) create mode 100644 tests/skills/test_skill_packaging.py rename {agent/tests => tests/skills}/test_upgrade_to_hybrid.py (96%) diff --git a/.claude/skills b/.claude/skills index 747ed1b..42c5394 120000 --- a/.claude/skills +++ b/.claude/skills @@ -1 +1 @@ -../agent/skills \ No newline at end of file +../skills \ No newline at end of file diff --git a/.github/workflows/skill-packaging.yml b/.github/workflows/skill-packaging.yml new file mode 100644 index 0000000..f3974b0 --- /dev/null +++ b/.github/workflows/skill-packaging.yml @@ -0,0 +1,28 @@ +name: Skill packaging + +on: + pull_request: + paths: + - 'skills/**' + - 'agent/tests/test_skill_packaging.py' + - '.claude/skills' + - '.github/workflows/skill-packaging.yml' + push: + paths: + - 'skills/**' + - 'agent/tests/test_skill_packaging.py' + - '.claude/skills' + - '.github/workflows/skill-packaging.yml' + +permissions: + contents: read + +jobs: + check: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - name: Verify bundled shared files + run: python3 skills/_shared/sync_shared.py --check + - name: Check independently installed skills + run: python3 -m unittest discover -s agent/tests -p test_skill_packaging.py diff --git a/AGENTS.md b/AGENTS.md index 74c4864..f527532 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -1,8 +1,8 @@ # Agent integrations -Agent-driven workflows for KERMT live under [`agent/`](agent/README.md). The +Agent-driven workflows for KERMT live under [`skills/`](skills/README.md). The skills follow the [agentskills.io](https://agentskills.io) spec — one -directory per skill at `agent/skills//SKILL.md`. +directory per skill at `skills//SKILL.md`. Available workflows: - `kermt-setup` — bootstrap the kermt container (run first) @@ -14,7 +14,13 @@ Available workflows: - `kermt-embed` — extract molecular embeddings from any encoder-bearing ckpt - `kermt-monitor` — tail logs / report progress for a detached run -See [`agent/README.md`](agent/README.md) for the full guide: hardware +See [`skills/README.md`](skills/README.md) for the full guide: hardware requirements per workflow, how to install the skills with Claude Code / Codex / Nemotron, and how to invoke the underlying scripts directly without an agent. + +Shared helpers and defaults are maintained in `skills/_shared/`. After editing +them, run `python3 skills/_shared/sync_shared.py --write` and commit the +per-skill copies; `--check` verifies they match. Each skill must use its +own bundled files so it can be installed and signed independently. +Development tests remain under `agent/tests/`. diff --git a/Dockerfile b/Dockerfile index a8722bc..d497e76 100644 --- a/Dockerfile +++ b/Dockerfile @@ -38,7 +38,7 @@ RUN conda clean -afy # Note: the repo is NOT copied into the image. Agent skills and any equivalent # host workflow bind-mount the live kermt repo checkout at /workspace (see -# agent/scripts/kermt_container.sh). Keeping the image as a pure environment +# skills/_shared/scripts/kermt_container.sh). Keeping the image as a pure environment # makes rebuilds cache-friendly (only invalidates when environment.yml changes, # not on every code edit) and avoids stale-code footguns where a baked-in /code # and the runtime bind-mount disagree. diff --git a/agent/README.md b/skills/README.md similarity index 78% rename from agent/README.md rename to skills/README.md index 4e76727..585438a 100644 --- a/agent/README.md +++ b/skills/README.md @@ -5,6 +5,33 @@ KERMT workflows locally. The skills are tool-agnostic Markdown files; the scripts are deterministic kernels that can also be invoked directly without an agent. +## Maintaining bundled helpers + +Canonical helpers and defaults live in [`_shared/`](_shared/). The explicit +ownership map in [`_shared/sync_shared.py`](_shared/sync_shared.py) copies each +asset into only the skills that consume it. Each skill carries real files in +its own `scripts/`, `config/`, and, where needed, `references/` directories. +`kermt-monitor` uses Docker, jq, and logs directly and needs no helper copies. + +From the repository root, after changing a canonical file: + +```bash +python3 skills/_shared/sync_shared.py --write +python3 skills/_shared/sync_shared.py --check +python3 -m unittest discover -s agent/tests -p test_skill_packaging.py +``` + +Commit both the canonical edits and their generated per-skill copies. This +makes nv-carps detect the affected skills and include their helper code and +defaults in each signed package. CI checks ownership and copy drift. Do not +replace packaged copies with symlinks or references to `_shared/` at runtime. + +Set `SKILL_DIR` to the installed skill's directory and export `KERMT_REPO` as +the path to a full KERMT checkout. The bundled container helper mounts the +checkout at `/workspace` and the skill at `/skill` (read-only), so container +commands use `/skill/scripts/`. The model implementation, root `scripts/`, +Dockerfile, and environment remain dependencies supplied by the checkout. + ## Audience: who should read what This README serves both human users and the agents that drive the skills. @@ -40,7 +67,7 @@ Linux only; Mac and Windows are not currently tested and may need adjustments ## Skills -The seven KERMT skills split into two categories. +The eight KERMT skills split into two categories. **Workflow skills** — the ones users invoke by name. Each composes a check_checkpoint → check_data → prepare_data → run_\ pipeline. @@ -48,7 +75,7 @@ check_checkpoint → check_data → prepare_data → run_\ pipeline. | Skill | Workflow | User provides | |---|---|---| | `kermt-continue-pretrain` | Continue pretraining from an existing KERMT checkpoint. Verifies the ckpt's vocab head sizes match the supplied vocab files; refuses to proceed on mismatch. | Checkpoint + its bundled vocab files (see [Released models](#released-models)), pretrain CSV, hyperparameters (optional) | -| `kermt-pretrain-scratch` | Pretrain a fresh KERMT model from scratch on a user corpus. Builds a new vocab from the corpus and initializes the model architecture from `config/defaults_pretrain.json`. Days-scale; warns the user before launching. | Pretrain CSV, `--pretrain-target-mode {vocab\|cmim\|hybrid}` (required), hyperparameters (optional) | +| `kermt-pretrain-scratch` | Pretrain a fresh KERMT model from scratch on a user corpus. Builds a new vocab from the corpus and initializes the model architecture from `_shared/config/defaults_pretrain.json`. Days-scale; warns the user before launching. | Pretrain CSV, `--pretrain-target-mode {vocab\|cmim\|hybrid}` (required), hyperparameters (optional) | | `kermt-add-cmim-pretrain` | Take a Grover-base-style checkpoint (no cMIM decoder), add a randomly-initialized decoder, then continue pretraining as Hybrid (vocab + contrast). | Encoder-only checkpoint, pretrain CSV, hyperparameters | | `kermt-finetune` | Finetune a pretrained checkpoint on a labeled task. | Pretrained checkpoint, labeled CSV, target column names | | `kermt-infer` | Run predictions with a finetuned checkpoint. | Finetuned checkpoint, CSV with SMILES | @@ -67,7 +94,7 @@ Pretraining is long-running (hours to days). The `kermt-continue-pretrain` and `kermt-add-cmim-pretrain` skills launch detached and return a run directory plus a TensorBoard URL; the agent typically invokes `kermt-monitor` next. -All seven skills are markdown + scripts, a passive instruction set — +All eight skills are markdown + scripts, a passive instruction set — they call deterministic Python kernels but make no autonomous decisions and expose no network APIs. @@ -97,44 +124,16 @@ machines without a GPU need to arrange a GPU host before invoking the skills. ## Released models -Each released KERMT checkpoint is distributed as a **directory bundle** -containing the ckpt itself plus its vocab files: - -``` -/ -├── last_checkpoint.pt -├── pretrain_atom_vocab.{json,pkl} # either extension; pkl in current releases -├── pretrain_bond_vocab.{json,pkl} # either extension; pkl in current releases -└── pretrain_smiles_vocab.pkl # only for cmim / hybrid ckpts (pickle-only) -``` - -If you're upgrading a grover_base ckpt to hybrid with -[`kermt-add-cmim-pretrain`](skills/kermt-add-cmim-pretrain/SKILL.md), the -upgrade step builds a fresh `pretrain_smiles_vocab.pkl` from your -pretrain corpus — released bundles only ship the smiles vocab for -already-cmim / already-hybrid ckpts. - -The vocab files are an inseparable part of the released model — the ckpt's -vocab head dimensions are fixed at training time and only match these specific -vocab files. `kermt-continue-pretrain` treats the released ckpt's vocab as -authoritative: new corpora are tokenized through it rather than producing a -new vocab that would mismatch the ckpt's heads. - -The skill auto-detects the three vocab files in the ckpt's parent directory -and passes them through `prepare_data.py --vocab-dir`. If the bundle is -incomplete (or the user has the ckpt alone), the skill asks for the -`--vocab-dir` path; if the user can't provide one, the skill refuses to -proceed and suggests `kermt-pretrain-scratch` instead. - -To train a model on a corpus the released vocab can't cover, use -`kermt-pretrain-scratch` — the new vocab is built from the corpus and the -model is initialized fresh (no warm start; days-scale to converge). +A released KERMT checkpoint must stay with its matching vocabulary files. +See the [released-model guide](_shared/references/released-models.md) for the +bundle layout, vocabulary compatibility, and handling incomplete downloads. +That canonical guide is also bundled into the skills that fetch released models. ## Token-efficient design The skill files (`.md`) are intentionally thin — they orchestrate, prompt for missing args, and parse JSON. The deterministic work happens in the Python -scripts under [`scripts/`](scripts/), each of which emits a structured JSON +scripts under [`_shared/scripts/`](_shared/scripts/), each of which emits a structured JSON document the calling skill consumes. This pattern follows the `bionemo-nim-skills` precedent (50–75% token reduction vs raw-prompt equivalents): every skill body stays well under the 500-line / 5000-token @@ -160,17 +159,17 @@ The `run.json` manifest is what you hand to the next skill (e.g. ## Defaults -Hyperparameter defaults live in [`config/`](config/). The skills echo applied +Hyperparameter defaults live in [`_shared/config/`](_shared/config/). The skills echo applied defaults back to you on every invocation; override any value with the corresponding `--` flag. -- [`config/defaults_pretrain.json`](config/defaults_pretrain.json) -- [`config/defaults_finetune.json`](config/defaults_finetune.json) +- [`_shared/config/defaults_pretrain.json`](_shared/config/defaults_pretrain.json) +- [`_shared/config/defaults_finetune.json`](_shared/config/defaults_finetune.json) ## Installing the skills The skills follow the [agentskills.io](https://agentskills.io) spec: each -skill is a directory under [`skills/`](skills/) containing a `SKILL.md` file. +skill is a directory under this [`skills/`](./) folder containing a `SKILL.md` file. The primary target agents are **Claude Code**, **Codex**, and **Nemotron**. Other agentskills.io-compatible agents should also work; see the [client showcase](https://agentskills.io) for the up-to-date list. @@ -183,9 +182,9 @@ Claude Code discovers skills at `~/.claude/skills//SKILL.md` only). **Project scope works out of the box.** This repo ships a committed -`.claude/skills` → `../agent/skills` symlink (and a `CLAUDE.md` → `AGENTS.md` +`.claude/skills` → `../skills` symlink (and a `CLAUDE.md` → `AGENTS.md` symlink), so opening Claude Code from the repo root discovers all eight -`kermt-*` skills with no install step — `agent/skills/` stays the single, +`kermt-*` skills with no install step — `skills/` stays the single, tool-agnostic source of truth (Codex / Nemotron read it directly). Restart Claude Code once after your first checkout so it picks up the entries. @@ -194,7 +193,7 @@ checkout) — from the directory that contains this README, symlink each skill into your personal skills dir: ```bash -for d in skills/kermt-*/; do +for d in kermt-*/; do name=$(basename "$d") ln -sfn "$(realpath "$d")" ~/.claude/skills/"$name" done @@ -208,7 +207,7 @@ effect without re-installing. If you ever rename a skill on disk (e.g. ### Codex Codex follows the same agentskills.io layout. Point Codex at the `skills/` -directory beside this README (or symlink each `kermt-*/` directory into its skills +directory containing this README (or symlink each `kermt-*/` directory into its skills discovery path — refer to the [Codex skills docs](https://developers.openai.com/codex/skills/) for the exact location). @@ -216,7 +215,7 @@ exact location). ### Nemotron and other agentskills.io-compatible agents Most other agents (Nemotron, Cursor, Gemini CLI, etc.) follow the same -spec — pass the `skills/` directory beside this README directly as a skills +spec — pass the `skills/` directory containing this README directly as a skills directory or attach the relevant `/SKILL.md` into context, and invoke by name (e.g. "run kermt-finetune on …"). @@ -226,24 +225,24 @@ relevant `/SKILL.md` into context, and invoke by name (e.g. Every workflow is also runnable directly by humans — the skills are just orchestration. Each `SKILL.md` lists the exact `python` commands it would run, and all scripts have `--help` output. Start by reading -[`skills/kermt-setup/SKILL.md`](skills/kermt-setup/SKILL.md) for the container bootstrap, +[`skills/kermt-setup/SKILL.md`](kermt-setup/SKILL.md) for the container bootstrap, then the skill for the workflow you want. ## Running a workflow The notes in this section apply to every workflow regardless of which agent (or human) is driving — they describe how the shared -[`scripts/kermt_container.sh`](scripts/kermt_container.sh) helper that every +[`kermt_container.sh`](_shared/scripts/kermt_container.sh) helper that each runtime SKILL.md invokes maps host paths into the container. ### `KERMT_REPO` environment variable Every workflow step in every SKILL.md uses -`$KERMT_REPO/agent/scripts/kermt_container.sh …` to bind-mount the repo at -`/workspace` inside the docker container. The helper auto-derives -`KERMT_REPO` from its own script path, so the default works for invocations -made from inside the repo. **If you're running from an arbitrary directory, -export it once** so every helper call picks it up: +`"$SKILL_DIR/scripts/kermt_container.sh" …` to bind-mount the repo at +`/workspace` inside the docker container and the installed skill at `/skill`. +The helper looks for a checkout above its own location or the current working +directory. For an independently installed skill, **export the runtime checkout +path once** so every helper call picks it up: ```bash export KERMT_REPO=/path/to/your/kermt @@ -291,7 +290,7 @@ clarity). ## Scripts (kernels) -Deterministic logic lives under [`scripts/`](scripts/): +Deterministic logic lives under [`_shared/scripts/`](_shared/scripts/): | Script | Purpose | |---|---| @@ -305,7 +304,7 @@ Deterministic logic lives under [`scripts/`](scripts/): | `run_inference.py` | Run predictions with a finetuned checkpoint | | `run_extract_embeddings.py` | Extract molecular embeddings | -Tests for these scripts are under [`tests/`](tests/) and use the existing +Tests for these scripts are under [`agent/tests/`](../agent/tests/) and use the existing fixture data in [`../tests/data/pretrain/`](../tests/data/pretrain/) and [`../tests/data/finetune/`](../tests/data/finetune/). @@ -315,21 +314,21 @@ The agent test suite should be run **inside the `kermt:latest` container** — t the same environment the skills exercise at runtime, so test results are faithful to the production path. `pytest` is included in the kermt conda env. -After [`kermt-setup`](skills/kermt-setup/SKILL.md) has built the image, from a +After [`kermt-setup`](kermt-setup/SKILL.md) has built the image, from a kermt repo checkout: ```bash # Run the full agent test suite in-container. The pytest paths are inside the # container, where the repo is bind-mounted at /workspace, so they stay # repo-relative (agent/tests/) regardless of your host working directory. -$KERMT_REPO/agent/scripts/kermt_container.sh run -- \ +$KERMT_REPO/skills/_shared/scripts/kermt_container.sh run -- \ "python -m pytest agent/tests/ -v --no-header -p no:cacheprovider" ``` Override the image tag if you're testing against a non-default build: ```bash -KERMT_IMAGE=kermt:rebuild-test $KERMT_REPO/agent/scripts/kermt_container.sh run -- \ +KERMT_IMAGE=kermt:rebuild-test $KERMT_REPO/skills/_shared/scripts/kermt_container.sh run -- \ "python -m pytest agent/tests/test_check_checkpoint.py -v" ``` @@ -342,5 +341,5 @@ conda activate python -m pytest agent/tests/test_check_data.py -v ``` -But the final sign-off for any change in `agent/scripts/` is the in-container +But the final sign-off for any change in `skills/_shared/scripts/` is the in-container run — that's what the skills will actually invoke. diff --git a/agent/config/defaults_embed.json b/skills/_shared/config/defaults_embed.json similarity index 100% rename from agent/config/defaults_embed.json rename to skills/_shared/config/defaults_embed.json diff --git a/agent/config/defaults_finetune.json b/skills/_shared/config/defaults_finetune.json similarity index 100% rename from agent/config/defaults_finetune.json rename to skills/_shared/config/defaults_finetune.json diff --git a/agent/config/defaults_inference.json b/skills/_shared/config/defaults_inference.json similarity index 100% rename from agent/config/defaults_inference.json rename to skills/_shared/config/defaults_inference.json diff --git a/agent/config/defaults_pretrain.json b/skills/_shared/config/defaults_pretrain.json similarity index 100% rename from agent/config/defaults_pretrain.json rename to skills/_shared/config/defaults_pretrain.json diff --git a/agent/config/released_model.json b/skills/_shared/config/released_model.json similarity index 100% rename from agent/config/released_model.json rename to skills/_shared/config/released_model.json diff --git a/skills/_shared/references/released-models.md b/skills/_shared/references/released-models.md new file mode 100644 index 0000000..e748c7e --- /dev/null +++ b/skills/_shared/references/released-models.md @@ -0,0 +1,34 @@ +# Released KERMT models + +Each released KERMT checkpoint is distributed as a **directory bundle** +containing the ckpt itself plus its vocab files: + +``` +/ +├── last_checkpoint.pt +├── pretrain_atom_vocab.{json,pkl} # either extension; pkl in current releases +├── pretrain_bond_vocab.{json,pkl} # either extension; pkl in current releases +└── pretrain_smiles_vocab.pkl # only for cmim / hybrid ckpts (pickle-only) +``` + +If you're upgrading a grover_base ckpt to hybrid with +`kermt-add-cmim-pretrain`, the +upgrade step builds a fresh `pretrain_smiles_vocab.pkl` from your +pretrain corpus — released bundles only ship the smiles vocab for +already-cmim / already-hybrid ckpts. + +The vocab files are an inseparable part of the released model — the ckpt's +vocab head dimensions are fixed at training time and only match these specific +vocab files. `kermt-continue-pretrain` treats the released ckpt's vocab as +authoritative: new corpora are tokenized through it rather than producing a +new vocab that would mismatch the ckpt's heads. + +The skill auto-detects the three vocab files in the ckpt's parent directory +and passes them through `prepare_data.py --vocab-dir`. If the bundle is +incomplete (or the user has the ckpt alone), the skill asks for the +`--vocab-dir` path; if the user can't provide one, the skill refuses to +proceed and suggests `kermt-pretrain-scratch` instead. + +To train a model on a corpus the released vocab can't cover, use +`kermt-pretrain-scratch` — the new vocab is built from the corpus and the +model is initialized fresh (no warm start; days-scale to converge). diff --git a/agent/scripts/_utils.py b/skills/_shared/scripts/_utils.py similarity index 90% rename from agent/scripts/_utils.py rename to skills/_shared/scripts/_utils.py index db8b492..5bde460 100644 --- a/agent/scripts/_utils.py +++ b/skills/_shared/scripts/_utils.py @@ -4,7 +4,7 @@ """Shared utilities for the agent scripts. Kept intentionally small — only logic that appears (or would otherwise be -duplicated) in two or more `agent/scripts/*.py` modules. Each script +duplicated) in two or more `scripts/*.py` modules. Each script maintains its own primary CLI + main flow. """ from __future__ import annotations @@ -30,6 +30,28 @@ } +def resolve_kermt_repo() -> Path: + """Find the runtime checkout independently of the installed skill location. + + An explicit KERMT_REPO takes precedence. In a repository checkout, walking + up from this helper or the working directory also supports local use. + """ + explicit = os.environ.get("KERMT_REPO") + if explicit: + candidates = [Path(explicit).expanduser().resolve()] + else: + candidates = [] + for start in (Path(__file__).resolve().parent, Path.cwd()): + candidates.extend((start, *start.parents)) + for candidate in candidates: + if (candidate / "main.py").is_file() and (candidate / "kermt").is_dir(): + return candidate + raise FileNotFoundError( + "KERMT checkout not found. Set KERMT_REPO to the checkout containing " + "main.py and kermt/; the installed skill directory is separate." + ) + + def load_json(path: Path, *, name: str) -> dict[str, Any]: """Load a JSON file with consistent error messages. @@ -113,7 +135,7 @@ def validate_vocab_file(vocab_path: Path, *, kind: str) -> None: def git_commit_with_env_override(repo: Path) -> tuple[str, bool]: """Returns (commit_sha, dirty_tree). Honors `KERMT_REPO_COMMIT` / - `KERMT_REPO_DIRTY` env vars first — set by `agent/scripts/kermt_container.sh` + `KERMT_REPO_DIRTY` env vars first — set by `scripts/kermt_container.sh` from the host before launching docker (necessary because `git -C /workspace` inside the container fails due to bind-mount ownership). Falls back to the in-container git probe when the env vars aren't set.""" @@ -234,7 +256,7 @@ def run_checkpoint_validator(ckpt: Path, *, mode: str, script_path: Path) -> dic """Invoke `check_checkpoint.py --mode --ckpt ` as a subprocess and return the parsed JSON. Raises RuntimeError on non-JSON output (e.g. the validator crashed before printing). `script_path` is the absolute path to - `agent/scripts/check_checkpoint.py` — passed in so this helper has no + `scripts/check_checkpoint.py` — passed in so this helper has no dependency on the caller's layout.""" r = subprocess.run( [sys.executable, str(script_path), "--mode", mode, "--ckpt", str(ckpt)], diff --git a/agent/scripts/check_checkpoint.py b/skills/_shared/scripts/check_checkpoint.py similarity index 100% rename from agent/scripts/check_checkpoint.py rename to skills/_shared/scripts/check_checkpoint.py diff --git a/agent/scripts/check_data.py b/skills/_shared/scripts/check_data.py similarity index 100% rename from agent/scripts/check_data.py rename to skills/_shared/scripts/check_data.py diff --git a/agent/scripts/fetch_released_model.py b/skills/_shared/scripts/fetch_released_model.py similarity index 95% rename from agent/scripts/fetch_released_model.py rename to skills/_shared/scripts/fetch_released_model.py index 84850df..ca2ad57 100644 --- a/agent/scripts/fetch_released_model.py +++ b/skills/_shared/scripts/fetch_released_model.py @@ -11,14 +11,14 @@ emits a single JSON object to stdout that the calling skill parses. The downloaded directory is exactly the repo's "released model bundle" layout -(see agent/README.md "Released models"): `.pt` + the three +(see skills/README.md "Released models"): `.pt` + the three `pretrain_*_vocab.*` files in one flat directory. The downstream skill then feeds it through the existing `--ckpt /` flow; for continue-pretrain the bundled vocab files are auto-detected in the ckpt's parent directory. No runner changes are needed. Defaults (repo id, pinned revision, ckpt + vocab filenames) come from -`agent/config/released_model.json` so the pin lives in one place; every value +`config/released_model.json` so the pin lives in one place; every value is overridable on the CLI. Idempotent: if the bundle is already complete in `--out` (ckpt + all vocab @@ -67,8 +67,8 @@ from pathlib import Path from typing import Any -# Default config lives at agent/config/released_model.json (one dir up from -# agent/scripts/). Resolved relative to this file so the script is +# Default config lives at config/released_model.json (one dir up from +# scripts/). Resolved relative to this file so the script is # location-independent. DEFAULT_CONFIG = ( Path(__file__).resolve().parent.parent / "config" / "released_model.json" @@ -178,7 +178,7 @@ def main(argv: list[str] | None = None) -> int: parser.add_argument( "--config", default=str(DEFAULT_CONFIG), - help="Path to released_model.json (default: agent/config/released_model.json).", + help="Path to released_model.json (default: config/released_model.json).", ) parser.add_argument( "--repo-id", default=None, help="Override the HF repo id from the config." diff --git a/agent/scripts/kermt_container.sh b/skills/_shared/scripts/kermt_container.sh similarity index 91% rename from agent/scripts/kermt_container.sh rename to skills/_shared/scripts/kermt_container.sh index fb65dc9..028057e 100755 --- a/agent/scripts/kermt_container.sh +++ b/skills/_shared/scripts/kermt_container.sh @@ -7,13 +7,13 @@ # Two ways to use this file: # # 1. As a subcommand dispatcher (recommended for skills): -# agent/scripts/kermt_container.sh ensure_image -# agent/scripts/kermt_container.sh run --ckpt /host/ckpt.pt -- python -c 'import torch; print(torch.cuda.device_count())' -# agent/scripts/kermt_container.sh run_detached --name foo --run-dir runs/foo -- bash train.sh +# "$SKILL_DIR/scripts/kermt_container.sh" ensure_image +# "$SKILL_DIR/scripts/kermt_container.sh" run --ckpt /host/ckpt.pt -- python -c 'import torch; print(torch.cuda.device_count())' +# "$SKILL_DIR/scripts/kermt_container.sh" run_detached --name foo --run-dir runs/foo -- bash train.sh # # 2. Sourced into a shell or another script, then call the kermt_* functions # directly: -# source agent/scripts/kermt_container.sh +# source "$SKILL_DIR/scripts/kermt_container.sh" # kermt_ensure_image # kermt_run --ckpt /host/ckpt.pt -- python ... # @@ -45,13 +45,33 @@ set -o pipefail : "${KERMT_IMAGE:=kermt:latest}" : "${KERMT_GPUS:=all}" -# Resolve KERMT_REPO from this script's location so the helper works regardless -# of the caller's working directory. +# The skill may be installed outside the KERMT checkout. Mount its own helpers +# separately so container commands always execute the distributed skill copy. +_kermt_script_dir="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)" +_kermt_bundle_dir="$(cd "$_kermt_script_dir/.." && pwd)" if [[ -z "${KERMT_REPO:-}" ]]; then - _kermt_script_dir="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)" - KERMT_REPO="$(cd "$_kermt_script_dir/../.." && pwd)" - unset _kermt_script_dir + for _kermt_start in "$_kermt_script_dir" "$PWD"; do + _kermt_candidate="$_kermt_start" + while [[ "$_kermt_candidate" != / ]]; do + if [[ -f "$_kermt_candidate/main.py" && -d "$_kermt_candidate/kermt" ]]; then + KERMT_REPO="$_kermt_candidate" + break 2 + fi + _kermt_candidate="$(dirname "$_kermt_candidate")" + done + done + unset _kermt_start _kermt_candidate fi +unset _kermt_script_dir + +_kermt_require_repo() { + if [[ -z "${KERMT_REPO:-}" || ! -f "$KERMT_REPO/main.py" || ! -d "$KERMT_REPO/kermt" ]]; then + echo "[kermt] Set KERMT_REPO to the KERMT checkout containing main.py and kermt/." >&2 + return 1 + fi + KERMT_REPO="$(cd "$KERMT_REPO" && pwd)" || return $? + export KERMT_REPO +} # ----------------------------------------------------------------------------- # Host environment checks @@ -218,6 +238,7 @@ kermt_check_gpu() { # ----------------------------------------------------------------------------- kermt_ensure_image() { + _kermt_require_repo || return $? kermt_check_docker || return $? if docker image inspect "$KERMT_IMAGE" >/dev/null 2>&1; then local id @@ -343,8 +364,10 @@ kermt_run() { docker run --rm --gpus "$KERMT_GPUS" \ --user "$(id -u):$(id -g)" \ -v "$KERMT_REPO:/workspace" \ + -v "$_kermt_bundle_dir:/skill:ro" \ "${mount_args[@]}" \ -w /workspace \ + -e KERMT_REPO=/workspace \ -e PYTHONPATH=/workspace \ -e HOME=/tmp/kermt-home \ "${git_args[@]}" \ @@ -389,8 +412,10 @@ kermt_run_detached() { --user "$(id -u):$(id -g)" \ --name "$name" \ -v "$KERMT_REPO:/workspace" \ + -v "$_kermt_bundle_dir:/skill:ro" \ "${mount_args[@]}" \ -w /workspace \ + -e KERMT_REPO=/workspace \ -e PYTHONPATH=/workspace \ -e HOME=/tmp/kermt-home \ "${git_args[@]}" \ @@ -445,7 +470,7 @@ Additional flags for run_detached: Environment overrides: KERMT_IMAGE default kermt:latest - KERMT_REPO default auto-derived from script location + KERMT_REPO checkout path; otherwise discovered above the skill or working directory KERMT_GPUS default all EOF exit 1 diff --git a/agent/scripts/prepare_data.py b/skills/_shared/scripts/prepare_data.py similarity index 99% rename from agent/scripts/prepare_data.py rename to skills/_shared/scripts/prepare_data.py index 58c7494..f0edb3e 100644 --- a/agent/scripts/prepare_data.py +++ b/skills/_shared/scripts/prepare_data.py @@ -34,7 +34,7 @@ Subprocess composition ---------------------- Each underlying script is invoked via `subprocess.run`. The PYTHONPATH=/workspace -env var (set by `agent/scripts/kermt_container.sh`) makes the `kermt` package +env var (set by `scripts/kermt_container.sh`) makes the `kermt` package importable inside the subprocesses; without it, build_vocab.py and split_data.py fail with `ModuleNotFoundError: No module named 'kermt'`. @@ -72,10 +72,10 @@ # from the host don't). if str(Path(__file__).resolve().parent) not in sys.path: sys.path.insert(0, str(Path(__file__).resolve().parent)) -from _utils import PRETRAIN_VOCAB_STEMS, validate_vocab_file # noqa: E402 +from _utils import PRETRAIN_VOCAB_STEMS, resolve_kermt_repo, validate_vocab_file # noqa: E402 -REPO_ROOT = Path(__file__).resolve().parents[2] +REPO_ROOT = resolve_kermt_repo() EXISTING_SCRIPTS = REPO_ROOT / "scripts" DEFAULT_FEATURES_GENERATOR = { diff --git a/agent/scripts/run_extract_embeddings.py b/skills/_shared/scripts/run_extract_embeddings.py similarity index 96% rename from agent/scripts/run_extract_embeddings.py rename to skills/_shared/scripts/run_extract_embeddings.py index 3961c95..7ba118c 100644 --- a/agent/scripts/run_extract_embeddings.py +++ b/skills/_shared/scripts/run_extract_embeddings.py @@ -45,16 +45,16 @@ if str(Path(__file__).resolve().parent) not in sys.path: sys.path.insert(0, str(Path(__file__).resolve().parent)) from _utils import ( # noqa: E402 - assert_prepare_manifest_basics, docker_image_digest, format_cmd_replay, + resolve_kermt_repo, assert_prepare_manifest_basics, docker_image_digest, format_cmd_replay, git_commit_with_env_override, load_json, merge_default_into_applied, resolve_single_gpu, run_checkpoint_validator, ) -REPO_ROOT = Path(__file__).resolve().parents[2] -AGENT_DIR = REPO_ROOT / "agent" -DEFAULTS_PATH = AGENT_DIR / "config" / "defaults_embed.json" -CHECK_CHECKPOINT_PATH = AGENT_DIR / "scripts" / "check_checkpoint.py" +REPO_ROOT = resolve_kermt_repo() +SKILL_ROOT = Path(__file__).resolve().parent.parent +DEFAULTS_PATH = SKILL_ROOT / "config" / "defaults_embed.json" +CHECK_CHECKPOINT_PATH = SKILL_ROOT / "scripts" / "check_checkpoint.py" EXTRACT_EMBEDDINGS_PATH = REPO_ROOT / "task" / "extract_embeddings.py" diff --git a/agent/scripts/run_finetune_local.py b/skills/_shared/scripts/run_finetune_local.py similarity index 98% rename from agent/scripts/run_finetune_local.py rename to skills/_shared/scripts/run_finetune_local.py index e447dd8..167d8a7 100644 --- a/agent/scripts/run_finetune_local.py +++ b/skills/_shared/scripts/run_finetune_local.py @@ -60,16 +60,16 @@ if str(Path(__file__).resolve().parent) not in sys.path: sys.path.insert(0, str(Path(__file__).resolve().parent)) from _utils import ( # noqa: E402 - assert_prepare_manifest_basics, docker_image_digest, format_cmd_replay, + resolve_kermt_repo, assert_prepare_manifest_basics, docker_image_digest, format_cmd_replay, git_commit_with_env_override, load_json, merge_default_into_applied, resolve_single_gpu, run_checkpoint_validator, ) -REPO_ROOT = Path(__file__).resolve().parents[2] -AGENT_DIR = REPO_ROOT / "agent" -DEFAULTS_PATH = AGENT_DIR / "config" / "defaults_finetune.json" -CHECK_CHECKPOINT_PATH = AGENT_DIR / "scripts" / "check_checkpoint.py" +REPO_ROOT = resolve_kermt_repo() +SKILL_ROOT = Path(__file__).resolve().parent.parent +DEFAULTS_PATH = SKILL_ROOT / "config" / "defaults_finetune.json" +CHECK_CHECKPOINT_PATH = SKILL_ROOT / "scripts" / "check_checkpoint.py" MAIN_PY_PATH = REPO_ROOT / "main.py" # Architecture fields the runner pulls from the ckpt and refuses to let the diff --git a/agent/scripts/run_inference.py b/skills/_shared/scripts/run_inference.py similarity index 97% rename from agent/scripts/run_inference.py rename to skills/_shared/scripts/run_inference.py index 8355770..56575bd 100644 --- a/agent/scripts/run_inference.py +++ b/skills/_shared/scripts/run_inference.py @@ -46,16 +46,16 @@ if str(Path(__file__).resolve().parent) not in sys.path: sys.path.insert(0, str(Path(__file__).resolve().parent)) from _utils import ( # noqa: E402 - assert_prepare_manifest_basics, docker_image_digest, format_cmd_replay, + resolve_kermt_repo, assert_prepare_manifest_basics, docker_image_digest, format_cmd_replay, git_commit_with_env_override, load_json, merge_default_into_applied, resolve_single_gpu, run_checkpoint_validator, ) -REPO_ROOT = Path(__file__).resolve().parents[2] -AGENT_DIR = REPO_ROOT / "agent" -DEFAULTS_PATH = AGENT_DIR / "config" / "defaults_inference.json" -CHECK_CHECKPOINT_PATH = AGENT_DIR / "scripts" / "check_checkpoint.py" +REPO_ROOT = resolve_kermt_repo() +SKILL_ROOT = Path(__file__).resolve().parent.parent +DEFAULTS_PATH = SKILL_ROOT / "config" / "defaults_inference.json" +CHECK_CHECKPOINT_PATH = SKILL_ROOT / "scripts" / "check_checkpoint.py" MAIN_PY_PATH = REPO_ROOT / "main.py" RUNTIME_FLAGS = ("batch_size", "seed") diff --git a/agent/scripts/run_pretrain_local.py b/skills/_shared/scripts/run_pretrain_local.py similarity index 98% rename from agent/scripts/run_pretrain_local.py rename to skills/_shared/scripts/run_pretrain_local.py index 6d1a235..5b8a8a1 100644 --- a/agent/scripts/run_pretrain_local.py +++ b/skills/_shared/scripts/run_pretrain_local.py @@ -8,7 +8,7 @@ Continues pretraining from a user-provided checkpoint. The model type (grover_base / cmim / hybrid) is inferred from the validator's output and drives the pretrain_ddp.py flag set; arch params come exclusively from the -ckpt; training/loss hyperparameters come from agent/config/defaults_pretrain.json +ckpt; training/loss hyperparameters come from config/defaults_pretrain.json with per-flag CLI overrides. How it interacts with pretrain_ddp.py's auto-resume: @@ -46,22 +46,22 @@ from pathlib import Path from typing import Any -# Add the agent/scripts/ dir to sys.path so `_utils` is importable whether +# Add the scripts/ dir to sys.path so `_utils` is importable whether # this script is launched via `kermt_run` (PYTHONPATH=/workspace) or as a -# bare `python agent/scripts/run_pretrain_local.py …` from the host. +# bare `python scripts/run_pretrain_local.py …` from the host. if str(Path(__file__).resolve().parent) not in sys.path: sys.path.insert(0, str(Path(__file__).resolve().parent)) from _utils import ( # noqa: E402 - assert_prepare_manifest_basics, count_vocab_entries, docker_image_digest, + resolve_kermt_repo, assert_prepare_manifest_basics, count_vocab_entries, docker_image_digest, format_cmd_replay, git_commit_with_env_override, load_json, merge_default_into_applied, run_checkpoint_validator, ) -REPO_ROOT = Path(__file__).resolve().parents[2] -AGENT_DIR = REPO_ROOT / "agent" -DEFAULTS_PATH = AGENT_DIR / "config" / "defaults_pretrain.json" -CHECK_CHECKPOINT_PATH = AGENT_DIR / "scripts" / "check_checkpoint.py" +REPO_ROOT = resolve_kermt_repo() +SKILL_ROOT = Path(__file__).resolve().parent.parent +DEFAULTS_PATH = SKILL_ROOT / "config" / "defaults_pretrain.json" +CHECK_CHECKPOINT_PATH = SKILL_ROOT / "scripts" / "check_checkpoint.py" PRETRAIN_DDP_PATH = REPO_ROOT / "pretrain_ddp.py" # Model-type → pretrain_ddp.py `--pretrain_mode` value. diff --git a/agent/scripts/upgrade_to_hybrid.py b/skills/_shared/scripts/upgrade_to_hybrid.py similarity index 98% rename from agent/scripts/upgrade_to_hybrid.py rename to skills/_shared/scripts/upgrade_to_hybrid.py index f5a0981..65c93f6 100644 --- a/agent/scripts/upgrade_to_hybrid.py +++ b/skills/_shared/scripts/upgrade_to_hybrid.py @@ -17,7 +17,7 @@ the prefix is renormalized to `kermt.encoders.*` to match the KermtHybridTask layout. - The saved `args` Namespace — augmented with hybrid-specific decoder fields - if absent (defaults from `agent/config/defaults_pretrain.json`). + if absent (defaults from `config/defaults_pretrain.json`). What's fresh-initialized ------------------------ @@ -62,16 +62,15 @@ import torch # sys.path tweak so `_utils` imports cleanly whether launched via `kermt_run` -# or as a bare `python agent/scripts/upgrade_to_hybrid.py …`. +# or as a bare `python scripts/upgrade_to_hybrid.py …`. if str(Path(__file__).resolve().parent) not in sys.path: sys.path.insert(0, str(Path(__file__).resolve().parent)) from _utils import count_vocab_entries, load_json, run_checkpoint_validator # noqa: E402 -REPO_ROOT = Path(__file__).resolve().parents[2] -AGENT_DIR = REPO_ROOT / "agent" -DEFAULTS_PATH = AGENT_DIR / "config" / "defaults_pretrain.json" -CHECK_CHECKPOINT_PATH = AGENT_DIR / "scripts" / "check_checkpoint.py" +SKILL_ROOT = Path(__file__).resolve().parent.parent +DEFAULTS_PATH = SKILL_ROOT / "config" / "defaults_pretrain.json" +CHECK_CHECKPOINT_PATH = SKILL_ROOT / "scripts" / "check_checkpoint.py" # Fields KermtHybridTask + its sub-components read from args. Anything not # present on the input ckpt's args gets filled from defaults_pretrain.json diff --git a/skills/_shared/sync_shared.py b/skills/_shared/sync_shared.py new file mode 100644 index 0000000..154ba8a --- /dev/null +++ b/skills/_shared/sync_shared.py @@ -0,0 +1,107 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +"""Copy canonical helpers/defaults into the skills that consume them. + +Edit files under skills/_shared/, then run this script with --write and commit +the per-skill copies. CI uses --check to catch drift. Copies are real files: +nv-carps signs each skill directory independently and rejects symlinks. +""" +from __future__ import annotations + +import argparse +import shutil +import stat +from pathlib import Path + +SHARED = Path(__file__).resolve().parent +SKILLS_ROOT = SHARED.parent +PRETRAIN_SKILLS = frozenset({ + "kermt-continue-pretrain", "kermt-pretrain-scratch", "kermt-add-cmim-pretrain", +}) +WORKFLOW_SKILLS = PRETRAIN_SKILLS | {"kermt-finetune", "kermt-infer", "kermt-embed"} +DOWNLOAD_SKILLS = frozenset({"kermt-continue-pretrain", "kermt-finetune", "kermt-embed"}) + +# Explicit consumers keep each independently installed skill complete without +# copying unrelated runners or configurations into it. +SHARED_ASSET_OWNERS = { + "scripts/kermt_container.sh": WORKFLOW_SKILLS | {"kermt-setup"}, + "scripts/_utils.py": WORKFLOW_SKILLS, + "scripts/check_checkpoint.py": WORKFLOW_SKILLS, + "scripts/check_data.py": WORKFLOW_SKILLS, + "scripts/prepare_data.py": WORKFLOW_SKILLS, + "scripts/fetch_released_model.py": DOWNLOAD_SKILLS, + "scripts/run_pretrain_local.py": PRETRAIN_SKILLS, + "scripts/run_finetune_local.py": {"kermt-finetune"}, + "scripts/run_inference.py": {"kermt-infer"}, + "scripts/run_extract_embeddings.py": {"kermt-embed"}, + "scripts/upgrade_to_hybrid.py": {"kermt-add-cmim-pretrain"}, + "config/defaults_pretrain.json": PRETRAIN_SKILLS, + "config/defaults_finetune.json": {"kermt-finetune"}, + "config/defaults_inference.json": {"kermt-infer"}, + "config/defaults_embed.json": {"kermt-embed"}, + "config/released_model.json": DOWNLOAD_SKILLS, + "references/released-models.md": DOWNLOAD_SKILLS, +} + + +def sync(*, write: bool = False) -> int: + skills = sorted(path.parent for path in SKILLS_ROOT.glob("kermt-*/SKILL.md")) + names = {path.name for path in skills} + problems = [] + for rel, owners in SHARED_ASSET_OWNERS.items(): + source = SHARED / rel + if source.is_symlink() or not source.is_file(): + problems.append(f"missing canonical real file: {rel}") + for owner in sorted(owners - names): + problems.append(f"unknown owner {owner}: {rel}") + for directory in ("scripts", "config", "references"): + for source in (SHARED / directory).rglob("*"): + if "__pycache__" in source.parts or source.suffix == ".pyc": + continue + if source.is_file() and source.relative_to(SHARED).as_posix() not in SHARED_ASSET_OWNERS: + problems.append(f"canonical file has no ownership entry: {source.relative_to(SHARED)}") + if problems: + print("\n".join(problems)) + return 1 + + for skill in skills: + for rel, owners in SHARED_ASSET_OWNERS.items(): + source, dest = SHARED / rel, skill / rel + if skill.name not in owners: + if dest.exists() or dest.is_symlink(): + if write: + dest.unlink() + else: + problems.append(f"{skill.name}: unexpected managed file {rel}") + continue + if write: + dest.parent.mkdir(parents=True, exist_ok=True) + if dest.is_symlink(): + dest.unlink() + shutil.copy2(source, dest) + elif dest.is_symlink() or not dest.is_file(): + problems.append(f"{skill.name}: missing real copy of {rel}") + elif source.read_bytes() != dest.read_bytes(): + problems.append(f"{skill.name}: {rel} differs from _shared/{rel}") + elif stat.S_IMODE(source.stat().st_mode) != stat.S_IMODE(dest.stat().st_mode): + problems.append(f"{skill.name}: {rel} has different permissions") + if problems: + print("\n".join(problems)) + print("Run python3 skills/_shared/sync_shared.py --write and commit the copies.") + return 1 + print(f"Shared assets {'synced' if write else 'verified'} for {len(skills)} skills.") + return 0 + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + mode = parser.add_mutually_exclusive_group(required=True) + mode.add_argument("--write", action="store_true", help="refresh per-skill copies") + mode.add_argument("--check", action="store_true", help="fail on missing copies or drift") + args = parser.parse_args() + return sync(write=args.write) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/agent/skills/kermt-add-cmim-pretrain/SKILL.md b/skills/kermt-add-cmim-pretrain/SKILL.md similarity index 90% rename from agent/skills/kermt-add-cmim-pretrain/SKILL.md rename to skills/kermt-add-cmim-pretrain/SKILL.md index 69b96a8..3b0fa7b 100644 --- a/agent/skills/kermt-add-cmim-pretrain/SKILL.md +++ b/skills/kermt-add-cmim-pretrain/SKILL.md @@ -30,6 +30,14 @@ the workflow is identical to `kermt-continue-pretrain`. > performance on your own benchmark before relying on the upgraded ckpt for > production work. +## Skill and runtime paths + +Set `SKILL_DIR` to the directory containing this `SKILL.md`. Export +`KERMT_REPO` as the absolute path to the KERMT checkout used for model +execution. The bundled container helper mounts that checkout at +`/workspace` and this skill at `/skill` (read-only). Commands inside +the container use `/skill/scripts/`; defaults are bundled in `config/`. + ## Hardware requirements Same as `kermt-continue-pretrain` (the cMIM decoder adds parameters but not @@ -96,8 +104,8 @@ Let `$KERMT_REPO` be the path to your kermt repo checkout. passing through the ckpt's old vocab (which may not even exist for encoder-only legacy grover_base ckpts): ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run --data --run-dir $RUN_DIR -- \ - "python agent/scripts/prepare_data.py --mode pretrain \\ + "$SKILL_DIR/scripts/kermt_container.sh" run --data --run-dir $RUN_DIR -- \ + "python /skill/scripts/prepare_data.py --mode pretrain \\ --csv /data/ --out /runs/data \\ [--val-csv /data/] [--val-frac 0.1] [--seed 0]" ``` @@ -106,8 +114,8 @@ Let `$KERMT_REPO` be the path to your kermt repo checkout. 6. **Upgrade the ckpt.** ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run --ckpt --run-dir $RUN_DIR -- \ - "python agent/scripts/upgrade_to_hybrid.py \\ + "$SKILL_DIR/scripts/kermt_container.sh" run --ckpt --run-dir $RUN_DIR -- \ + "python /skill/scripts/upgrade_to_hybrid.py \\ --ckpt /ckpt \\ --prepare-manifest /runs/data/prepare_data.json \\ --out /runs/upgraded.pt" @@ -123,10 +131,10 @@ Let `$KERMT_REPO` be the path to your kermt repo checkout. 8. **Launch the runner detached.** ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run_detached \\ + "$SKILL_DIR/scripts/kermt_container.sh" run_detached \\ --name kermt-add-cmim-pretrain- \\ --run-dir $RUN_DIR -- \\ - "python agent/scripts/run_pretrain_local.py \\ + "python /skill/scripts/run_pretrain_local.py \\ --ckpt /runs/upgraded.pt \\ --prepare-manifest /runs/data/prepare_data.json \\ --out /runs \\ diff --git a/skills/kermt-add-cmim-pretrain/config/defaults_pretrain.json b/skills/kermt-add-cmim-pretrain/config/defaults_pretrain.json new file mode 100644 index 0000000..dbb53e7 --- /dev/null +++ b/skills/kermt-add-cmim-pretrain/config/defaults_pretrain.json @@ -0,0 +1,52 @@ +{ + "_about": "Default hyperparameters applied by kermt-continue-pretrain and kermt-add-cmim-pretrain. Values target a workstation-scale hybrid pretrain. The skill echoes the applied set back to the user on every invocation; override any value with the corresponding CLI flag.", + + "training": { + "_about": "Optimizer and training schedule. Apply to all pretrain workflows.", + "batch_size": 256, + "dropout": 0.1, + "epochs": 30, + "init_lr": 1e-5, + "max_lr": 1.5e-4, + "final_lr": 1e-5, + "warmup_epochs": 20, + "weight_decay": 1e-7, + "save_interval": 100, + "seed": 0, + "tensorboard": true, + "use_cuikmolmaker_featurization": true + }, + + "loss": { + "_about": "Loss-weighting knobs. contrastive_temperature applies only when the model has a contrast head (hybrid). vocab_loss_weight applies when the model has a vocab head (cmim or hybrid). The runner detects the model type from the checkpoint and ignores irrelevant entries.", + "contrastive_temperature": 0.1, + "vocab_loss_weight": 1.0 + }, + + "add_cmim_decoder": { + "_about": "Used by kermt-add-cmim-pretrain when constructing the new cMIM decoder + latent_dist on top of a loaded grover-base encoder, and by kermt-pretrain-scratch when the pretrain target is cmim or hybrid. Ignored by kermt-continue-pretrain (those dimensions come from the ckpt's saved_args). Values match the manuscript's hybrid pretrain configuration: latent_dim=512, 8-head, 3-layer decoder (cf. `_PRESET_LATENT_DIM` / `_PRESET_DECODER_FFN_HIDDEN_SIZE` in launch-KERMT-pretrain-slurm.sh, both presets).", + "latent_dim": 512, + "contrastive_temperature": 0.1, + "decoder_num_layers": 3, + "decoder_num_attention_heads": 8, + "decoder_ffn_hidden_size": 2048, + "decoder_dropout": 0.1, + "decoder_max_seq_len": 512, + "decoder_positional_encoding": "rope", + "decoder_gate_self_attn": false, + "decoder_gate_cross_attn": false + }, + + "arch": { + "_about": "Encoder architecture defaults — used ONLY by kermt-pretrain-scratch (fresh model from corpus, no starting ckpt). kermt-continue-pretrain and kermt-add-cmim-pretrain ignore this block and pull arch from the loaded checkpoint instead; the runner aborts if user-supplied arch flags mismatch the ckpt's saved_args.", + "hidden_size": 800, + "depth": 6, + "num_attn_head": 4, + "activation": "PReLU", + "backbone": "gtrans", + "embedding_output_type": "both", + "self_attention": false + }, + + "_about_gpu_selection": "GPU selection is auto-detected at runtime, not a default here. The pretrain runner uses torch.cuda.device_count() and dispatches single-GPU or DDP accordingly. Override with --gpus 0,2 if you want a specific subset." +} diff --git a/agent/skills/kermt-add-cmim-pretrain/evals/evals.json b/skills/kermt-add-cmim-pretrain/evals/evals.json similarity index 100% rename from agent/skills/kermt-add-cmim-pretrain/evals/evals.json rename to skills/kermt-add-cmim-pretrain/evals/evals.json diff --git a/skills/kermt-add-cmim-pretrain/scripts/_utils.py b/skills/kermt-add-cmim-pretrain/scripts/_utils.py new file mode 100644 index 0000000..5bde460 --- /dev/null +++ b/skills/kermt-add-cmim-pretrain/scripts/_utils.py @@ -0,0 +1,272 @@ +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Shared utilities for the agent scripts. + +Kept intentionally small — only logic that appears (or would otherwise be +duplicated) in two or more `scripts/*.py` modules. Each script +maintains its own primary CLI + main flow. +""" +from __future__ import annotations + +import argparse +import json +import os +import shlex +import subprocess +import sys +from pathlib import Path +from typing import Any + + +# Conventional pretrain vocab filename stems. Used by prepare_data.py + +# upgrade_to_hybrid.py + the README "Released models" bundling docs + +# the test helpers. Centralized here so a future rename only touches one +# spot. +PRETRAIN_VOCAB_STEMS = { + "atom": "pretrain_atom_vocab", + "bond": "pretrain_bond_vocab", + "smiles": "pretrain_smiles_vocab", +} + + +def resolve_kermt_repo() -> Path: + """Find the runtime checkout independently of the installed skill location. + + An explicit KERMT_REPO takes precedence. In a repository checkout, walking + up from this helper or the working directory also supports local use. + """ + explicit = os.environ.get("KERMT_REPO") + if explicit: + candidates = [Path(explicit).expanduser().resolve()] + else: + candidates = [] + for start in (Path(__file__).resolve().parent, Path.cwd()): + candidates.extend((start, *start.parents)) + for candidate in candidates: + if (candidate / "main.py").is_file() and (candidate / "kermt").is_dir(): + return candidate + raise FileNotFoundError( + "KERMT checkout not found. Set KERMT_REPO to the checkout containing " + "main.py and kermt/; the installed skill directory is separate." + ) + + +def load_json(path: Path, *, name: str) -> dict[str, Any]: + """Load a JSON file with consistent error messages. + + `name` is a human-readable label for the document (e.g. "prepare_data.json") + so the error tells the user which schema we expected at that path. + """ + if not path.is_file(): + raise FileNotFoundError(f"{name} not found at {path}") + try: + return json.loads(path.read_text()) + except json.JSONDecodeError as exc: + raise ValueError(f"{name} at {path} is not valid JSON: {exc}") from exc + + +def count_vocab_entries(vocab_path: Path) -> int: + """Return the number of entries in a KERMT vocab file. + + Handles three layouts: + - JSON with `{stoi: {token: idx}, ...}` (MolVocab.save_vocab default) + - JSON as a raw `{token: idx}` dict (legacy / hand-edited) + - Pickle of a MolVocab / SMILESVocab object (the smiles vocab is always + pickled because its compiled-regex tokenizer state isn't + JSON-serializable). Falls through to raw `pickle.load` if the + MolVocab / SMILESVocab loader can't import or fails to recognize + the contents (e.g. test fixtures with plain dicts). + """ + if vocab_path.suffix == ".json": + data = json.loads(vocab_path.read_text()) + if isinstance(data, dict) and "stoi" in data: + return len(data["stoi"]) + if isinstance(data, dict): + return len(data) + raise ValueError(f"unsupported JSON vocab shape at {vocab_path}: {type(data).__name__}") + + # .pkl: try MolVocab / SMILESVocab first, then fall back to raw pickle. + try: + from kermt.data.torchvocab import MolVocab, SMILESVocab # type: ignore + for loader in (MolVocab.load_vocab, SMILESVocab.load_vocab): + try: + v = loader(str(vocab_path)) + return len(v) + except Exception: + continue + except ImportError: + pass + + import pickle + with vocab_path.open("rb") as f: + data = pickle.load(f) + if hasattr(data, "stoi"): + return len(data.stoi) + if hasattr(data, "__len__"): + return len(data) + raise ValueError(f"could not count entries in {vocab_path}") + + +def validate_vocab_file(vocab_path: Path, *, kind: str) -> None: + """Verify a user-provided vocab file is loadable BEFORE copying it into a + run directory. Raises ValueError on failure with a clear, user-facing message. + + `kind` is one of {"atom", "bond", "smiles"} — used only in the error message + so the user knows which file is wrong. + """ + if not vocab_path.is_file(): + raise FileNotFoundError(f"{kind} vocab file not found: {vocab_path}") + try: + n = count_vocab_entries(vocab_path) + except Exception as exc: # noqa: BLE001 + raise ValueError( + f"{kind} vocab file {vocab_path} is not loadable as a KERMT vocab " + f"({type(exc).__name__}: {exc}). Expected a MolVocab JSON or pickle " + f"(or a SMILESVocab pickle for the smiles vocab)." + ) from exc + if n <= 0: + raise ValueError(f"{kind} vocab file {vocab_path} contains zero entries") + + +# --------------------------------------------------------------------------- +# Runner-shared helpers (run.json manifest fields) +# --------------------------------------------------------------------------- + +def git_commit_with_env_override(repo: Path) -> tuple[str, bool]: + """Returns (commit_sha, dirty_tree). Honors `KERMT_REPO_COMMIT` / + `KERMT_REPO_DIRTY` env vars first — set by `scripts/kermt_container.sh` + from the host before launching docker (necessary because `git -C /workspace` + inside the container fails due to bind-mount ownership). Falls back to the + in-container git probe when the env vars aren't set.""" + env_commit = os.environ.get("KERMT_REPO_COMMIT") + if env_commit: + env_dirty = os.environ.get("KERMT_REPO_DIRTY", "false").strip().lower() == "true" + return env_commit, env_dirty + try: + sha = subprocess.run( + ["git", "-C", str(repo), "rev-parse", "HEAD"], + capture_output=True, text=True, check=True, + ).stdout.strip() + diff = subprocess.run( + ["git", "-C", str(repo), "status", "--porcelain"], + capture_output=True, text=True, check=True, + ) + return sha, bool(diff.stdout.strip()) + except Exception: + return "unknown", False + + +def docker_image_digest(tag: str) -> str | None: + """Return the docker image's content-addressable Id (sha256:…) for the given + tag, or None if docker isn't available / the image isn't local.""" + try: + r = subprocess.run( + ["docker", "image", "inspect", tag, "--format", "{{.Id}}"], + capture_output=True, text=True, + ) + if r.returncode == 0: + return r.stdout.strip() + except FileNotFoundError: + pass + return None + + +def format_cmd_replay(argv: list[str], *, env: dict[str, str] | None = None) -> str: + """Render a copy-pasteable env-prefix + command for the cmd_replay manifest + field. `env` is the set of environment variables to prefix (typically + {CUDA_VISIBLE_DEVICES, WORLD_SIZE}).""" + env = env or {} + env_prefix = [f"{k}={shlex.quote(str(v))}" for k, v in env.items()] + quoted = " ".join(shlex.quote(a) for a in argv) + return " ".join(env_prefix + [quoted]) + + +def resolve_single_gpu(override: str | None, *, workflow: str) -> int: + """Returns a single GPU id (int). The finetune/inference/embed workflows are + single-GPU only; `--gpus '0,1'` or multi-id CUDA_VISIBLE_DEVICES is rejected + with a workflow-specific error. (The pretrain runner has its own multi-GPU + `_detect_gpus` helper — see run_pretrain_local.py.)""" + if override is None: + env_visible = os.environ.get("CUDA_VISIBLE_DEVICES", "").strip() + if env_visible: + ids = [g for g in env_visible.split(",") if g] + if len(ids) > 1: + raise ValueError( + f"CUDA_VISIBLE_DEVICES='{env_visible}' selects multiple GPUs but " + f"the {workflow} workflow is single-GPU only. Restrict to one id." + ) + return int(ids[0]) + return 0 + parts = [p.strip() for p in override.split(",") if p.strip()] + if len(parts) != 1: + raise ValueError( + f"--gpus '{override}' selects {len(parts)} GPUs; the {workflow} workflow is single-GPU only." + ) + return int(parts[0]) + + +def assert_prepare_manifest_basics(manifest: dict[str, Any], expected_mode: str) -> None: + """Standard pre-check for a prepare_data.json before a runner consumes it: + verify `mode` matches and `ok` is True. Raises ValueError with a consistent + error message on either mismatch. + + Each runner is responsible for its own required-outputs check after this + (those vary per-mode — e.g. pretrain wants train_dir/val_dir/atom_vocab/ + bond_vocab; finetune has the split-method branch; inference/embed want + clean_csv).""" + if manifest.get("mode") != expected_mode: + raise ValueError( + f"prepare_data manifest is mode='{manifest.get('mode')}', expected '{expected_mode}'. " + f"Run `prepare_data.py --mode {expected_mode}` to produce a valid manifest." + ) + if not manifest.get("ok"): + raise ValueError( + f"prepare_data manifest reports ok=False: {manifest.get('errors')}" + ) + + +def merge_default_into_applied( + applied: dict[str, dict[str, Any]], + args: argparse.Namespace, + name: str, + defaults_group: dict[str, Any], +) -> None: + """Standard CLI-override / default-config merge for one hyperparameter. + + Mutates `applied` in place: + - If the user passed `--` on the CLI (so `getattr(args, name)` is + not None), records `{"value": cli_val, "source": "user"}`. + - Else if `name` is present in `defaults_group`, records + `{"value": defaults_group[name], "source": "default-config"}`. + - Else `applied[name]` is left absent — the runner's argv-builder skips + the flag, and the downstream argparse default takes effect. + + `name` is the snake_case argparse dest (same form used as the dict key); + argparse automatically converts CLI `--` to that dest, + so `getattr(args, name, None)` is the correct CLI lookup.""" + cli_val = getattr(args, name, None) + if cli_val is not None: + applied[name] = {"value": cli_val, "source": "user"} + elif name in defaults_group: + applied[name] = {"value": defaults_group[name], "source": "default-config"} + + +def run_checkpoint_validator(ckpt: Path, *, mode: str, script_path: Path) -> dict[str, Any]: + """Invoke `check_checkpoint.py --mode --ckpt ` as a subprocess + and return the parsed JSON. Raises RuntimeError on non-JSON output (e.g. the + validator crashed before printing). `script_path` is the absolute path to + `scripts/check_checkpoint.py` — passed in so this helper has no + dependency on the caller's layout.""" + r = subprocess.run( + [sys.executable, str(script_path), "--mode", mode, "--ckpt", str(ckpt)], + capture_output=True, text=True, + ) + try: + return json.loads(r.stdout) + except json.JSONDecodeError as exc: + raise RuntimeError( + f"check_checkpoint.py emitted non-JSON output (exit {r.returncode}). " + f"stdout (first 200 chars): {r.stdout[:200]}\n" + f"stderr (first 200 chars): {r.stderr[:200]}" + ) from exc diff --git a/skills/kermt-add-cmim-pretrain/scripts/check_checkpoint.py b/skills/kermt-add-cmim-pretrain/scripts/check_checkpoint.py new file mode 100644 index 0000000..fd488e2 --- /dev/null +++ b/skills/kermt-add-cmim-pretrain/scripts/check_checkpoint.py @@ -0,0 +1,480 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Validate a KERMT checkpoint for a given agent workflow. + +Mode-dispatched. Emits a single JSON object to stdout that the calling skill +parses to decide whether to proceed (`ok: true`) or surface a structured error +back to the user (`ok: false` with `errors[]`). The JSON shape is stable +across modes; only the contract for what counts as `ok` differs per mode. + +Modes +----- +continue_pretrain Continuing pretraining from an existing pretrain ckpt. + Requires encoder + at least one pretrain head + (vocab_head for grover_base / cmim, or contrast_head for + cmim / hybrid). Rejects encoder-only or finetuned ckpts. + +upgrade_to_hybrid Adding a cMIM decoder onto a grover_base ckpt to convert + it to a hybrid pretrain. Requires encoder; rejects ckpts + that already carry a contrast_head or task_ffn (would be + workflow 4 instead). + +finetune_init Starting a finetune from a pretrained ckpt. Requires + encoder. Pretrain heads (vocab / contrast) are tolerated + but unused. Already-finetuned ckpts (task FFN heads + present) are REJECTED — finetune-on-finetune via the + agent skill isn't supported because saved-task + identity can't be machine-verified against the new + training data. + +inference Running predictions with a previously-finetuned ckpt. + Requires encoder + task_ffn. Reports task_output_dims + so the runner can compare against the user's task spec. + +embed Extracting embeddings. Requires encoder only. Anything + additional in the ckpt is ignored. + +Output (stdout) +--------------- +{ + "ok": true | false, + "model_type": "grover_base" | "cmim" | "hybrid" | "finetuned" | "unknown", + "has_encoder": bool, + "has_vocab_head": bool, + "has_contrast_head": bool, + "has_task_ffn": bool, + "task_output_dims": [int, ...], // empty unless has_task_ffn + "arch": { // ckpt-derived; runner uses these, ignores defaults_*.json arch + "hidden_size": int | null, + "depth": int | null, + "num_attn_head": int | null, + "latent_dim": int | null, + "activation": str | null, + "backbone": str | null, + "embedding_output_type": str | null, + "self_attention": bool | null + }, + "saved_args": { ... } | null, // raw args dict if present, else null + "errors": [str, ...], // mode-contract violations / load failures + "warnings": [str, ...] // non-fatal observations (e.g. arch fallback) +} + +Exit code: 0 on `ok: true`, 1 on `ok: false`. Loader exceptions are caught and +surfaced into `errors[]` with `ok: false` (still exit 1), never raised. + +CLI +--- + check_checkpoint.py --mode --ckpt +""" +from __future__ import annotations + +import argparse +import json +import sys +import traceback +from argparse import Namespace +from typing import Any + +import torch + + +# --------------------------------------------------------------------------- +# State-dict key prefix conventions (kermt/model/models.py). +# --------------------------------------------------------------------------- + +# Encoder weights appear under one of these prefixes depending on the ckpt's +# era and task class: +# - `grover.*` : legacy grover_base ckpts (predate the cMIM rename) +# - `kermt.*` : current grover_base / hybrid / finetune ckpts +# - `latent_dist.kermt.*`: cmim ckpts (encoder lives only inside latent_dist) +ENCODER_PREFIXES = ("kermt.", "grover.", "latent_dist.kermt.") +VOCAB_HEAD_PREFIX = "vocab_module." +CONTRAST_DECODER_PREFIX = "decoder." # SMILES transformer decoder, cmim/hybrid only +LATENT_DIST_PREFIX = "latent_dist." # cmim/hybrid; encoder may share via latent_dist.kermt.* +TASK_FFN_PREFIXES = ( + "mol_atom_from_atom_ffn.", + "mol_atom_from_bond_ffn.", +) +TASK_FFN_TASK_SPECIFIC_PREFIXES = ( + "mol_atom_from_atom_ffn_task_specific.", + "mol_atom_from_bond_ffn_task_specific.", +) + + +ARCH_KEYS = ( + "hidden_size", + "depth", + "num_attn_head", + "latent_dim", + "activation", + "backbone", + "embedding_output_type", + "self_attention", +) + + +def _strip_ddp_prefix(state_dict: dict[str, Any]) -> dict[str, Any]: + """Strip `module.` prefix from every key if the dict is DDP-wrapped.""" + if state_dict and all(k.startswith("module.") for k in state_dict): + return {k[len("module."):]: v for k, v in state_dict.items()} + return state_dict + + +def _classify_model(state_dict: dict[str, Any]) -> dict[str, Any]: + keys = list(state_dict.keys()) + has_encoder = any(k.startswith(ENCODER_PREFIXES) for k in keys) + has_vocab_head = any(k.startswith(VOCAB_HEAD_PREFIX) for k in keys) + has_contrast_head = any(k.startswith(CONTRAST_DECODER_PREFIX) for k in keys) + has_task_ffn = any(k.startswith(TASK_FFN_PREFIXES) for k in keys) + + if has_encoder and has_task_ffn: + model_type = "finetuned" + elif has_encoder and has_contrast_head and has_vocab_head: + model_type = "hybrid" + elif has_encoder and has_contrast_head and not has_vocab_head: + model_type = "cmim" + elif has_encoder and not has_contrast_head: + # Includes: + # - modern repo-trained Grover base (kermt.* + vocab_module.*) + # - legacy original-Grover base (grover.encoders.* with no heads saved) + # - any encoder-stripped ckpt extracted from a larger model + # The `has_vocab_head` flag discriminates the sub-cases for skills that + # need it. The continue_pretrain mode contract relies on this — a + # grover_base with vocab heads can continue, an encoder-only one cannot. + model_type = "grover_base" + else: + model_type = "unknown" + + return { + "model_type": model_type, + "has_encoder": has_encoder, + "has_vocab_head": has_vocab_head, + "has_contrast_head": has_contrast_head, + "has_task_ffn": has_task_ffn, + } + + +def _vocab_sizes(state_dict: dict[str, Any]) -> dict[str, Any]: + """Extract vocab head sizes from state-dict weight shapes. + + The pretrain heads have the following layout per kermt/model/models.py: + - Atom vocab predictors: vocab_module.av_task_atom.* + vocab_module.av_task_bond.* + (two readout streams sharing the same vocab_size). Output dim of each + final-Linear is the atom vocab size. + - Bond vocab predictors: vocab_module.bv_task_atom.* + vocab_module.bv_task_bond.* + Output dim is the bond vocab size. + - SMILES vocab decoder: decoder.output_projection.weight (cmim / hybrid only). + Output dim is the smiles vocab size. + + Returns {atom: int|None, bond: int|None, smiles: int|None}. Each is None + when the corresponding head isn't present in the ckpt (e.g. legacy + encoder-only grover_base has none; cmim has smiles but not atom/bond). + """ + sizes: dict[str, Any] = {"atom": None, "bond": None, "smiles": None} + + def _head_out_dim(prefix: str) -> int | None: + # Pick the highest-numbered 2-D Linear weight under `prefix.*` — that's + # the final output layer. + candidates = [ + k for k in state_dict + if k.startswith(prefix) and k.endswith(".weight") + and hasattr(state_dict[k], "ndim") and state_dict[k].ndim == 2 + ] + if not candidates: + return None + def _layer_index(k: str) -> int: + # ".weight" -> "..weight"; pick the rightmost numeric component. + parts = k.split(".") + for tok in reversed(parts[:-1]): + if tok.isdigit(): + return int(tok) + return -1 + final = max(candidates, key=_layer_index) + return int(state_dict[final].shape[0]) + + sizes["atom"] = _head_out_dim("vocab_module.av_task_atom.") + sizes["bond"] = _head_out_dim("vocab_module.bv_task_atom.") + sizes["smiles"] = _head_out_dim("decoder.output_projection.") + # If the decoder's output_projection isn't a Linear (e.g. some saves wrap + # it differently), fall back to a search over decoder.* heads. + if sizes["smiles"] is None: + sizes["smiles"] = _head_out_dim("decoder.token_embedding.") + return sizes + + +def _task_output_dims(state_dict: dict[str, Any]) -> list[int]: + """Return one entry per (logical task × readout) head's final-Linear out-dim. + + Two layouts: + - **MTL** (`mol_atom_from_atom_ffn_task_specific..*`): one entry per + task-specific head's final-Linear out-dim. Typically `[1, 1, ..., 1]` + for regression with N tasks across 2 readouts. + - **Non-MTL** (`mol_atom_from_atom_ffn.*` only): one entry per shared FFN's + final-Linear out-dim. Typically `[num_tasks, num_tasks]` (one per readout). + + When both layouts coexist in the same ckpt (MTL configuration: shared FFN + feeds task-specific heads), only the task-specific dims are reported — the + shared FFN there is an intermediate layer, not the model output. + """ + has_task_specific = any(k.startswith(TASK_FFN_TASK_SPECIFIC_PREFIXES) for k in state_dict) + + heads: dict[str, list[str]] = {} + for k in state_dict: + if k.startswith(TASK_FFN_TASK_SPECIFIC_PREFIXES): + parts = k.split(".") + root = ".".join(parts[:2]) # e.g. "mol_atom_from_atom_ffn_task_specific.0" + heads.setdefault(root, []).append(k) + elif k.startswith(TASK_FFN_PREFIXES) and not k.startswith(TASK_FFN_TASK_SPECIFIC_PREFIXES): + if has_task_specific: + continue # shared FFN is intermediate when task-specific heads exist + root = k.split(".")[0] # e.g. "mol_atom_from_atom_ffn" + heads.setdefault(root, []).append(k) + + dims: list[int] = [] + for root in sorted(heads): + weight_keys = sorted( + (k for k in heads[root] if k.endswith(".weight") + and hasattr(state_dict[k], "ndim") and state_dict[k].ndim == 2), + key=lambda k: int(k.split(".")[-2]) if k.split(".")[-2].isdigit() else -1, + ) + if weight_keys: + dims.append(int(state_dict[weight_keys[-1]].shape[0])) + return dims + + +def _arch_from_args(args_obj: Any) -> dict[str, Any]: + """Pull arch params from the saved args Namespace / dict, leaving missing keys as None.""" + arch: dict[str, Any] = {k: None for k in ARCH_KEYS} + if args_obj is None: + return arch + # args_obj is typically argparse.Namespace; tolerate dict form too. + args_dict = vars(args_obj) if isinstance(args_obj, Namespace) else dict(args_obj) if isinstance(args_obj, dict) else {} + for k in ARCH_KEYS: + if k in args_dict: + arch[k] = args_dict[k] + return arch + + +def _arch_from_shapes(state_dict: dict[str, Any], arch: dict[str, Any]) -> tuple[dict[str, Any], list[str]]: + """Fill in still-missing arch params by introspecting state-dict tensor shapes. + + Only fills entries that are currently None — does not override anything pulled + from saved_args. Returns the updated arch + a list of warnings for any key that + could not be inferred. + """ + warnings: list[str] = [] + + if arch["hidden_size"] is None: + # First 2-D linear weight under any encoder prefix. + candidates = [ + k for k in state_dict + if k.startswith(ENCODER_PREFIXES) + and k.endswith(".weight") + and hasattr(state_dict[k], "ndim") + and state_dict[k].ndim == 2 + ] + if candidates: + arch["hidden_size"] = int(state_dict[candidates[0]].shape[0]) + else: + warnings.append("hidden_size could not be inferred from state_dict shapes") + + if arch["latent_dim"] is None: + # Look for a Linear inside latent_dist that's not the shared encoder. + candidates = [ + k for k in state_dict + if k.startswith(LATENT_DIST_PREFIX) + and not k.startswith("latent_dist.kermt.") + and k.endswith(".weight") + and state_dict[k].ndim == 2 + ] + if candidates: + arch["latent_dim"] = int(state_dict[candidates[0]].shape[0]) + # Absent latent_dist entirely is a fact about the model, not a problem — no warning. + + # depth, num_attn_head, activation, backbone, embedding_output_type, self_attention + # are not robustly inferable from shapes alone; report a warning for each that's + # still None so the caller can prompt the user or refuse to proceed. + for k in ("depth", "num_attn_head", "activation", "backbone", "embedding_output_type", "self_attention"): + if arch[k] is None: + warnings.append(f"{k} not present in saved_args and cannot be inferred from state_dict shapes") + + return arch, warnings + + +def _apply_mode_contract(mode: str, classification: dict[str, Any]) -> list[str]: + """Return a list of error messages if `classification` violates the mode contract.""" + errors: list[str] = [] + mt = classification["model_type"] + has_enc = classification["has_encoder"] + has_vocab = classification["has_vocab_head"] + has_contrast = classification["has_contrast_head"] + has_ffn = classification["has_task_ffn"] + + if not has_enc: + errors.append("checkpoint has no encoder weights — cannot use it for any KERMT workflow") + return errors + + if mode == "continue_pretrain": + if not (has_vocab or has_contrast): + errors.append( + f"continue_pretrain requires the ckpt to still carry pretrain heads (vocab " + f"and/or contrast), but this ckpt has neither (model_type='{mt}', " + f"has_vocab_head=False, has_contrast_head=False). Either provide a ckpt with " + f"its pretrain heads attached, or convert this encoder-only ckpt to a hybrid " + f"via mode 'upgrade_to_hybrid'." + ) + if has_ffn: + errors.append( + "continue_pretrain expects a pretrain ckpt; this ckpt has task FFN heads " + "(it has been finetuned). Use a pretrain checkpoint — finetune+continue is " + "not a supported workflow." + ) + elif mode == "upgrade_to_hybrid": + if has_contrast: + errors.append( + f"upgrade_to_hybrid converts grover_base -> hybrid by adding a cMIM decoder. " + f"This ckpt already has a contrast head (classified as '{mt}'). " + f"To continue pretraining it, use mode 'continue_pretrain'." + ) + if has_ffn: + errors.append("upgrade_to_hybrid does not support finetuned checkpoints.") + elif mode == "finetune_init": + # Requires an encoder. Pretrain heads (vocab / contrast) are unused + # at finetune time but harmless. Task FFN heads (i.e. an already- + # finetuned ckpt) are NOT accepted — finetune-on-finetune isn't + # supported by the kermt-finetune skill because the saved-task + # identity can't be machine-verified against the new training data + # (dimension match doesn't prove target identity, dataset identity, + # or absence of train/test contamination). + if has_ffn: + errors.append( + f"finetune_init requires a pretrain ckpt (grover_base / cmim / hybrid); " + f"this ckpt is classified as '{mt}' with task FFN heads attached. " + f"To resume a finetune on the SAME dataset, call " + f"`python main.py finetune --checkpoint_path ...` directly — the " + f"kermt-finetune skill doesn't support resume." + ) + elif mode == "inference": + if not has_ffn: + errors.append( + "inference requires a finetuned ckpt with task FFN heads. " + f"This ckpt is classified as '{mt}' with no task heads. " + "Run finetune (mode 'finetune_init') first." + ) + elif mode == "embed": + # Encoder is sufficient. + pass + else: + errors.append(f"unknown mode '{mode}'") + + return errors + + +def validate(mode: str, ckpt_path: str) -> dict[str, Any]: + result: dict[str, Any] = { + "ok": False, + "model_type": "unknown", + "has_encoder": False, + "has_vocab_head": False, + "has_contrast_head": False, + "has_task_ffn": False, + "task_output_dims": [], + "vocab_sizes": {"atom": None, "bond": None, "smiles": None}, + "arch": {k: None for k in ARCH_KEYS}, + "saved_args": None, + "errors": [], + "warnings": [], + } + + # 1. Load the checkpoint. + try: + ckpt = torch.load(ckpt_path, map_location="cpu", weights_only=False) + except FileNotFoundError: + result["errors"].append(f"checkpoint not found: {ckpt_path}") + return result + except Exception as exc: # noqa: BLE001 + result["errors"].append(f"failed to load checkpoint {ckpt_path}: {type(exc).__name__}: {exc}") + return result + + if not isinstance(ckpt, dict) or "state_dict" not in ckpt: + result["errors"].append( + "checkpoint is not in the expected save_model_for_restart format " + "(expected a dict with a 'state_dict' key)." + ) + return result + + state_dict = _strip_ddp_prefix(ckpt["state_dict"]) + args_obj = ckpt.get("args") + + # 2. Classify and check mode contract. + classification = _classify_model(state_dict) + result.update(classification) + + contract_errors = _apply_mode_contract(mode, classification) + result["errors"].extend(contract_errors) + + # 3. Task output dims (for inference / informational). + if classification["has_task_ffn"]: + result["task_output_dims"] = _task_output_dims(state_dict) + + # 3b. Vocab head sizes (for continue-pretrain vocab-size verification). + result["vocab_sizes"] = _vocab_sizes(state_dict) + + # 4. Arch derivation: args first, shape introspection for what's still missing. + arch = _arch_from_args(args_obj) + arch, shape_warnings = _arch_from_shapes(state_dict, arch) + result["arch"] = arch + result["warnings"].extend(shape_warnings) + + # 5. Saved args as serializable dict (best-effort). + if args_obj is not None: + try: + result["saved_args"] = vars(args_obj) if isinstance(args_obj, Namespace) else dict(args_obj) + # Drop non-JSON-serializable values; agent skill only needs human-readable scalars. + result["saved_args"] = { + k: v for k, v in result["saved_args"].items() + if isinstance(v, (str, int, float, bool, type(None), list, dict)) + } + except Exception as exc: # noqa: BLE001 + result["warnings"].append(f"could not serialize saved_args: {type(exc).__name__}: {exc}") + + result["ok"] = not result["errors"] + return result + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Validate a KERMT checkpoint for a given workflow.") + parser.add_argument("--mode", required=True, + choices=["continue_pretrain", "upgrade_to_hybrid", "finetune_init", "inference", "embed"]) + parser.add_argument("--ckpt", required=True, help="Path to the .pt checkpoint") + args = parser.parse_args(argv) + + try: + result = validate(args.mode, args.ckpt) + except Exception as exc: # noqa: BLE001 + # Last-resort safety net: keep stdout JSON-clean, dump trace to stderr. + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({ + "ok": False, + "model_type": "unknown", + "errors": [f"unhandled exception in validator: {type(exc).__name__}: {exc}"], + "warnings": [], + "arch": {k: None for k in ARCH_KEYS}, + "has_encoder": False, + "has_vocab_head": False, + "has_contrast_head": False, + "has_task_ffn": False, + "task_output_dims": [], + "vocab_sizes": {"atom": None, "bond": None, "smiles": None}, + "saved_args": None, + }, indent=2)) + return 1 + + print(json.dumps(result, indent=2)) + return 0 if result["ok"] else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-add-cmim-pretrain/scripts/check_data.py b/skills/kermt-add-cmim-pretrain/scripts/check_data.py new file mode 100644 index 0000000..b8f9b15 --- /dev/null +++ b/skills/kermt-add-cmim-pretrain/scripts/check_data.py @@ -0,0 +1,310 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Validate a CSV input for a given KERMT agent workflow. + +Mode-dispatched. Emits a single JSON object to stdout that the calling skill +parses to decide whether to proceed (`ok: true`) or surface a structured error +back to the user (`ok: false` with `errors[]`). The JSON shape is stable +across modes; only the contract for what counts as `ok` differs per mode. + +Modes +----- +pretrain Pretrain corpus CSV. Requires a `smiles` column. Other columns + are ignored. Label columns are not required (and not expected). + +finetune Labeled CSV for a downstream task. Requires `smiles` plus + >=1 numeric target column. Target columns are specified via + `--targets ...`. If `--targets` is omitted, the + validator auto-detects numeric non-smiles columns and reports + them; the skill will then prompt the user to confirm or refine. + +inference CSV to run predictions on. Requires `smiles`. Target columns are + not required (and not expected — predictions are written out). + +embed CSV to extract embeddings from. Requires `smiles` only. + +SMILES validation +----------------- +By default the validator samples up to 20 SMILES (first 10 + last 10) and +checks each one parses with RDKit. Pass `--strict-rdkit` to parse every +SMILES (slow on large corpora). A SMILES is considered "invalid" if RDKit +returns `None` from `MolFromSmiles(smi, sanitize=True)` — empty / null +rows are counted separately. + +Duplicate-SMILES detection is always full (cheap). + +Output (stdout) +--------------- +{ + "ok": true | false, + "mode": str, + "csv_path": str, + "num_rows": int, + "num_columns": int, + "columns": [str, ...], + "has_smiles_column": bool, + "smiles_column_name": str | null, // actual header used (may differ in case) + "num_blank_smiles": int, + "num_invalid_smiles": int, // among the parsed sample + "smiles_check_method": "sampled" | "full", + "smiles_check_count": int, + "num_duplicate_smiles": int, + "target_columns": [str, ...], // populated only for finetune mode + "num_missing_per_target": { col: int, ... }, + "auto_detected_targets": [str, ...], // when --targets is omitted in finetune mode + "errors": [str, ...], + "warnings": [str, ...] +} + +Exit code: 0 on `ok: true`, 1 on `ok: false`. Loader exceptions are caught +and surfaced into `errors[]` with `ok: false` (still exit 1). + +CLI +--- + check_data.py --mode --csv + [--targets ...] # finetune only + [--strict-rdkit] # full SMILES parse +""" +from __future__ import annotations + +import argparse +import json +import sys +import traceback +from pathlib import Path +from typing import Any + +import pandas as pd + + +CANONICAL_SMILES_COLUMN = "smiles" +SMILES_SAMPLE_PER_END = 10 # how many SMILES from head + how many from tail to sample + + +def _find_smiles_column(columns: list[str]) -> str | None: + """Return the actual column header matching 'smiles' case-insensitively, or None.""" + for c in columns: + if c.lower() == CANONICAL_SMILES_COLUMN: + return c + return None + + +def _parse_smiles_sample(smiles_values: list[str], full: bool) -> tuple[int, int, str]: + """Run RDKit MolFromSmiles on a sample or all of the SMILES. Returns + (num_parsed, num_invalid, method).""" + # Import here so the script can still surface a clean JSON error if RDKit + # is unavailable in the host env. + try: + from rdkit import Chem + from rdkit import RDLogger + RDLogger.DisableLog("rdApp.*") # suppress per-mol parse warnings + except ImportError as exc: + raise RuntimeError( + f"RDKit is not importable in this environment: {exc}. " + "Run check_data.py inside the kermt container." + ) from exc + + if full or len(smiles_values) <= 2 * SMILES_SAMPLE_PER_END: + sample = smiles_values + method = "full" + else: + sample = smiles_values[:SMILES_SAMPLE_PER_END] + smiles_values[-SMILES_SAMPLE_PER_END:] + method = "sampled" + + invalid = 0 + parsed = 0 + for smi in sample: + if not smi: # already counted as blank elsewhere + continue + parsed += 1 + mol = Chem.MolFromSmiles(smi, sanitize=True) + if mol is None: + invalid += 1 + return parsed, invalid, method + + +def _autodetect_target_columns(df: pd.DataFrame, smiles_col: str) -> list[str]: + """Pick columns that look like numeric targets. A column qualifies if it + is (a) not the smiles column and (b) >=80% of non-null values convert to float. + Heuristic only — returned for the skill to prompt the user to confirm.""" + candidates: list[str] = [] + for col in df.columns: + if col == smiles_col: + continue + ser = df[col].dropna() + if len(ser) == 0: + continue + try: + converted = pd.to_numeric(ser, errors="coerce") + except (TypeError, ValueError): + continue + if converted.notna().sum() / max(len(ser), 1) >= 0.8: + candidates.append(col) + return candidates + + +def validate(mode: str, csv_path: str, targets: list[str] | None, strict_rdkit: bool) -> dict[str, Any]: + result: dict[str, Any] = { + "ok": False, + "mode": mode, + "csv_path": csv_path, + "num_rows": 0, + "num_columns": 0, + "columns": [], + "has_smiles_column": False, + "smiles_column_name": None, + "num_blank_smiles": 0, + "num_invalid_smiles": 0, + "smiles_check_method": "sampled", + "smiles_check_count": 0, + "num_duplicate_smiles": 0, + "target_columns": [], + "num_missing_per_target": {}, + "auto_detected_targets": [], + "errors": [], + "warnings": [], + } + + # 1. Read the CSV. + path = Path(csv_path) + if not path.is_file(): + result["errors"].append(f"CSV not found: {csv_path}") + return result + try: + df = pd.read_csv(path) + except pd.errors.EmptyDataError: + result["errors"].append(f"CSV is empty (no header): {csv_path}") + return result + except Exception as exc: # noqa: BLE001 + result["errors"].append(f"failed to read CSV {csv_path}: {type(exc).__name__}: {exc}") + return result + + result["num_rows"] = int(len(df)) + result["num_columns"] = int(len(df.columns)) + result["columns"] = [str(c) for c in df.columns] + + # 2. Locate the SMILES column. + smiles_col = _find_smiles_column(result["columns"]) + if smiles_col is None: + result["errors"].append( + f"no column named 'smiles' (case-insensitive) found in CSV. " + f"Available columns: {result['columns']}" + ) + return result + result["has_smiles_column"] = True + result["smiles_column_name"] = smiles_col + if smiles_col != CANONICAL_SMILES_COLUMN: + result["warnings"].append( + f"SMILES column is named '{smiles_col}' but downstream code expects '{CANONICAL_SMILES_COLUMN}' " + f"(lowercase). Rename the column to '{CANONICAL_SMILES_COLUMN}' before running the workflow." + ) + + # 3. Blank-SMILES count + duplicate count + RDKit parse check. + smi_series = df[smiles_col].astype(str).fillna("").str.strip() + blank_mask = smi_series.eq("") | smi_series.str.lower().eq("nan") + result["num_blank_smiles"] = int(blank_mask.sum()) + + nonblank = smi_series[~blank_mask] + result["num_duplicate_smiles"] = int(len(nonblank) - nonblank.nunique()) + + if len(nonblank) == 0: + result["errors"].append("no non-blank SMILES found in the CSV") + return result + + try: + parsed, invalid, method = _parse_smiles_sample(nonblank.tolist(), full=strict_rdkit) + except RuntimeError as exc: + result["errors"].append(str(exc)) + return result + result["smiles_check_count"] = parsed + result["num_invalid_smiles"] = invalid + result["smiles_check_method"] = method + + if invalid > 0: + scope = "all rows" if method == "full" else f"the {parsed} sampled rows" + result["errors"].append( + f"{invalid} out of {parsed} SMILES in {scope} failed to parse with RDKit. " + "Either pre-clean the CSV with scripts/clean_smiles.py or pass --strict-rdkit to see " + "the full count." + ) + + # 4. Target-column handling — finetune mode only. + if mode == "finetune": + if targets: + missing = [t for t in targets if t not in df.columns] + if missing: + result["errors"].append( + f"target column(s) not found in CSV: {missing}. " + f"Available columns: {result['columns']}" + ) + else: + result["target_columns"] = list(targets) + for t in targets: + nan_count = int(df[t].isna().sum()) + result["num_missing_per_target"][t] = nan_count + # Confirm numeric-ish. + nonnan = df[t].dropna() + converted = pd.to_numeric(nonnan, errors="coerce") + non_numeric_count = int(converted.isna().sum()) + if non_numeric_count > 0: + result["warnings"].append( + f"target column '{t}' has {non_numeric_count} non-numeric value(s) " + f"that will be dropped by the finetune runner." + ) + else: + # Auto-detect — surface candidates so the skill can prompt the user. + result["auto_detected_targets"] = _autodetect_target_columns(df, smiles_col) + if not result["auto_detected_targets"]: + result["errors"].append( + "no numeric non-smiles columns detected. finetune needs at least one target column; " + "specify it explicitly via --targets ." + ) + else: + result["warnings"].append( + f"--targets was not specified; auto-detected candidate target columns " + f"{result['auto_detected_targets']}. The skill will prompt the user to confirm." + ) + + # 5. Small-corpus warning — only for pretrain (other modes can be tiny by design). + if mode == "pretrain" and result["num_rows"] < 100: + result["warnings"].append( + f"pretrain corpus is only {result['num_rows']} molecule(s). Pretraining typically " + f"needs orders of magnitude more — verify this is the intended input." + ) + + result["ok"] = not result["errors"] + return result + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Validate a CSV input for a KERMT agent workflow.") + parser.add_argument("--mode", required=True, choices=["pretrain", "finetune", "inference", "embed"]) + parser.add_argument("--csv", required=True, help="Path to the input CSV") + parser.add_argument("--targets", nargs="+", default=None, + help="(finetune only) target column names. If omitted, the validator auto-detects " + "numeric non-smiles columns and reports them as candidates.") + parser.add_argument("--strict-rdkit", action="store_true", + help="Parse every SMILES with RDKit rather than sampling (slow on large CSVs).") + args = parser.parse_args(argv) + + try: + result = validate(args.mode, args.csv, args.targets, args.strict_rdkit) + except Exception as exc: # noqa: BLE001 + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({ + "ok": False, + "mode": args.mode, + "csv_path": args.csv, + "errors": [f"unhandled exception in validator: {type(exc).__name__}: {exc}"], + "warnings": [], + }, indent=2)) + return 1 + + print(json.dumps(result, indent=2)) + return 0 if result["ok"] else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-add-cmim-pretrain/scripts/kermt_container.sh b/skills/kermt-add-cmim-pretrain/scripts/kermt_container.sh new file mode 100755 index 0000000..028057e --- /dev/null +++ b/skills/kermt-add-cmim-pretrain/scripts/kermt_container.sh @@ -0,0 +1,484 @@ +#!/usr/bin/env bash +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# kermt_container.sh — bootstrap helper for the kermt agent skills. +# +# Two ways to use this file: +# +# 1. As a subcommand dispatcher (recommended for skills): +# "$SKILL_DIR/scripts/kermt_container.sh" ensure_image +# "$SKILL_DIR/scripts/kermt_container.sh" run --ckpt /host/ckpt.pt -- python -c 'import torch; print(torch.cuda.device_count())' +# "$SKILL_DIR/scripts/kermt_container.sh" run_detached --name foo --run-dir runs/foo -- bash train.sh +# +# 2. Sourced into a shell or another script, then call the kermt_* functions +# directly: +# source "$SKILL_DIR/scripts/kermt_container.sh" +# kermt_ensure_image +# kermt_run --ckpt /host/ckpt.pt -- python ... +# +# Configuration (override via env vars before invocation): +# KERMT_IMAGE docker image tag (default: kermt:latest) +# KERMT_REPO host path to the kermt repo checkout (default: auto-derived +# from this script's location) +# KERMT_GPUS value passed to docker --gpus (default: all) +# +# Mount flags accepted by kermt_run / kermt_run_detached: +# --data bind to /data (read-only). If is a file, +# its PARENT directory is mounted at /data so +# commands can use /data/; if is a +# directory, it is mounted at /data directly. +# --ckpt bind to /ckpt (read-only; the path is mounted as-is) +# --vocab-dir bind to /vocab (read-only) +# --run-dir bind to /runs (read-write; created on host if missing) +# --model-dir bind to /model (read-write; created on host if missing). +# Target for released-model downloads (fetch_released_model.py). +# +# Additional flags for kermt_run_detached: +# --name docker container name (default: kermt--) +# +# Everything after `--` is the command passed to the container. It runs inside +# the `kermt` conda environment (the image's default env). + +set -o pipefail + +: "${KERMT_IMAGE:=kermt:latest}" +: "${KERMT_GPUS:=all}" + +# The skill may be installed outside the KERMT checkout. Mount its own helpers +# separately so container commands always execute the distributed skill copy. +_kermt_script_dir="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)" +_kermt_bundle_dir="$(cd "$_kermt_script_dir/.." && pwd)" +if [[ -z "${KERMT_REPO:-}" ]]; then + for _kermt_start in "$_kermt_script_dir" "$PWD"; do + _kermt_candidate="$_kermt_start" + while [[ "$_kermt_candidate" != / ]]; do + if [[ -f "$_kermt_candidate/main.py" && -d "$_kermt_candidate/kermt" ]]; then + KERMT_REPO="$_kermt_candidate" + break 2 + fi + _kermt_candidate="$(dirname "$_kermt_candidate")" + done + done + unset _kermt_start _kermt_candidate +fi +unset _kermt_script_dir + +_kermt_require_repo() { + if [[ -z "${KERMT_REPO:-}" || ! -f "$KERMT_REPO/main.py" || ! -d "$KERMT_REPO/kermt" ]]; then + echo "[kermt] Set KERMT_REPO to the KERMT checkout containing main.py and kermt/." >&2 + return 1 + fi + KERMT_REPO="$(cd "$KERMT_REPO" && pwd)" || return $? + export KERMT_REPO +} + +# ----------------------------------------------------------------------------- +# Host environment checks +# ----------------------------------------------------------------------------- + +kermt_check_docker() { + if ! command -v docker >/dev/null 2>&1; then + echo "[kermt] error: docker not found on PATH. Install Docker first." >&2 + return 1 + fi + if ! docker info >/dev/null 2>&1; then + echo "[kermt] error: docker daemon not reachable. Is the docker service running, and is your user in the 'docker' group?" >&2 + return 1 + fi +} + +kermt_check_system() { + # Probe host system and report GPU presence + VRAM + compute capability + + # driver / CUDA version + disk space. Emits a single JSON document to + # stdout that the calling skill consumes; exits 0 with `ok: false` and a + # populated `gaps` array when anything is below the per-workflow minimum, + # exits 1 only on unexpected internal errors. Uses host nvidia-smi + df + + # host python3 (stdlib only). + python3 - "$KERMT_REPO" "$KERMT_IMAGE" <<'PYEOF' +import json, os, shutil, subprocess, sys + +repo, image = sys.argv[1], sys.argv[2] + +result = { + "ok": True, + "gpus": [], + "disk": {"path": repo, "free_gb": None, "min_gb": 20}, + "host": {"docker": None, "nvidia_smi": None, "container_toolkit": None}, + "image": {"tag": image, "present_locally": None}, + "gaps": [], +} + +def _gap(msg): + result["ok"] = False + result["gaps"].append(msg) + +# docker presence +try: + r = subprocess.run(["docker", "info"], capture_output=True, text=True, timeout=10) + result["host"]["docker"] = "ok" if r.returncode == 0 else f"failed: {r.stderr.strip().splitlines()[-1] if r.stderr else 'unknown'}" + if r.returncode != 0: + _gap("docker daemon not reachable (is the service running, and is your user in the 'docker' group?)") +except FileNotFoundError: + result["host"]["docker"] = "not found" + _gap("docker not on PATH; install Docker first") +except Exception as e: + result["host"]["docker"] = f"error: {e}" + _gap(f"docker probe failed: {e}") + +# nvidia-smi (host driver) +try: + r = subprocess.run( + ["nvidia-smi", "--query-gpu=name,memory.total,compute_cap,driver_version,uuid", + "--format=csv,noheader,nounits"], + capture_output=True, text=True, timeout=10, + ) + if r.returncode == 0: + result["host"]["nvidia_smi"] = "ok" + for line in r.stdout.strip().splitlines(): + parts = [p.strip() for p in line.split(",")] + if len(parts) >= 5: + try: + vram_mb = int(parts[1]) + except ValueError: + vram_mb = None + result["gpus"].append({ + "name": parts[0], + "vram_mb": vram_mb, + "compute_cap": parts[2], + "driver": parts[3], + "uuid": parts[4], + }) + if not result["gpus"]: + _gap("nvidia-smi succeeded but reported no GPUs") + else: + result["host"]["nvidia_smi"] = "failed" + _gap("nvidia-smi found but failed; is the NVIDIA driver loaded?") +except FileNotFoundError: + result["host"]["nvidia_smi"] = "not found" + _gap("nvidia-smi not on PATH; install the NVIDIA driver") +except Exception as e: + result["host"]["nvidia_smi"] = f"error: {e}" + _gap(f"nvidia-smi probe failed: {e}") + +# disk free at the repo location +try: + free_bytes = shutil.disk_usage(repo).free + free_gb = free_bytes // (1024**3) + result["disk"]["free_gb"] = free_gb + if free_gb < result["disk"]["min_gb"]: + _gap(f"disk free at {repo} is {free_gb} GB; need at least {result['disk']['min_gb']} GB for the kermt image") +except Exception as e: + _gap(f"could not check disk space at {repo}: {e}") + +# image presence (informational only) +try: + r = subprocess.run(["docker", "image", "inspect", image], capture_output=True, text=True, timeout=10) + result["image"]["present_locally"] = (r.returncode == 0) +except Exception: + result["image"]["present_locally"] = None + +# nvidia-container-toolkit probe — only meaningful if both docker and a +# locally-present image are available. Pick kermt:$tag first; fall back to +# the small CUDA base image if that's the only one present; otherwise skip +# (avoid pulling anything). +def _probe_image(): + for img in (image, "nvidia/cuda:12.6.3-base-ubuntu22.04"): + r = subprocess.run(["docker", "image", "inspect", img], capture_output=True) + if r.returncode == 0: + return img + return None + +probe_img = _probe_image() +if probe_img: + try: + r = subprocess.run( + ["docker", "run", "--rm", "--gpus", "all", probe_img, "nvidia-smi"], + capture_output=True, text=True, timeout=60, + ) + if r.returncode == 0: + result["host"]["container_toolkit"] = f"ok (probed via {probe_img})" + else: + result["host"]["container_toolkit"] = f"failed (probed via {probe_img})" + _gap("`docker run --gpus all` failed; install nvidia-container-toolkit and ensure the host driver supports it") + except Exception as e: + result["host"]["container_toolkit"] = f"error: {e}" + _gap(f"nvidia-container-toolkit probe failed: {e}") +else: + result["host"]["container_toolkit"] = "skipped (no probe image present locally; run ensure_image first)" + +print(json.dumps(result, indent=2)) +PYEOF +} + +kermt_check_gpu() { + # Probes whether `docker --gpus all` is wired up (nvidia-container-toolkit). + # Image-selection priority (never pulls anything): + # 1) $KERMT_IMAGE if it exists locally, + # 2) else nvidia/cuda:12.6.3-base-ubuntu22.04 if it exists locally, + # 3) else skip with a warning (return 0). The smoke test inside kermt_run + # will catch broken GPU passthrough later anyway. + local probe_img="" + if docker image inspect "$KERMT_IMAGE" >/dev/null 2>&1; then + probe_img="$KERMT_IMAGE" + elif docker image inspect nvidia/cuda:12.6.3-base-ubuntu22.04 >/dev/null 2>&1; then + probe_img="nvidia/cuda:12.6.3-base-ubuntu22.04" + else + echo "[kermt] check_gpu: skipped — neither '$KERMT_IMAGE' nor 'nvidia/cuda:12.6.3-base-ubuntu22.04' is present locally. Run 'ensure_image' first, or this probe will be exercised by the in-container smoke test." >&2 + return 0 + fi + if ! docker run --rm --gpus all "$probe_img" nvidia-smi >/dev/null 2>&1; then + echo "[kermt] error: 'docker run --gpus all' failed (probe image: $probe_img). Install nvidia-container-toolkit and ensure the host has a CUDA-capable NVIDIA driver." >&2 + return 1 + fi +} + +# ----------------------------------------------------------------------------- +# Image build / verification +# ----------------------------------------------------------------------------- + +kermt_ensure_image() { + _kermt_require_repo || return $? + kermt_check_docker || return $? + if docker image inspect "$KERMT_IMAGE" >/dev/null 2>&1; then + local id + id=$(docker image inspect "$KERMT_IMAGE" --format '{{.Id}}' 2>/dev/null | cut -c1-19) + echo "[kermt] image '$KERMT_IMAGE' already present (${id:-unknown})" + return 0 + fi + echo "[kermt] image '$KERMT_IMAGE' not found; building from $KERMT_REPO/Dockerfile" + echo "[kermt] first build typically takes 10-20 minutes on a typical workstation; subsequent runs reuse the cached image" + docker build -t "$KERMT_IMAGE" -f "$KERMT_REPO/Dockerfile" "$KERMT_REPO" +} + +# ----------------------------------------------------------------------------- +# Mount-flag parser, internal +# ----------------------------------------------------------------------------- +# Reads flags from the caller's positional args until it hits '--', appending +# `-v src:dst[:ro]` pairs into the caller-provided array name (passed as $1). +# Returns the number of caller-provided args consumed via _kermt_consumed. +# This is bash-specific (uses nameref via `declare -n`). + +_kermt_parse_mounts() { + local -n _out="$1" + shift + _kermt_consumed=0 + while [[ $# -gt 0 ]]; do + case "$1" in + --) + return 0 + ;; + --data) + [[ -e "$2" ]] || { echo "[kermt] --data path not found: $2" >&2; return 1; } + # If the user passes a file, mount its parent directory at /data so + # downstream commands can refer to /data/. Mounting a + # single file at /data makes the path-as-directory pattern in the + # skill examples (`--csv /data/`) fail with "not found". + if [[ -d "$2" ]]; then + _out+=("-v" "$(realpath "$2"):/data:ro") + else + _out+=("-v" "$(realpath "$(dirname "$2")"):/data:ro") + fi + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --ckpt) + [[ -e "$2" ]] || { echo "[kermt] --ckpt path not found: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/ckpt:ro") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --vocab-dir) + [[ -d "$2" ]] || { echo "[kermt] --vocab-dir not found or not a directory: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/vocab:ro") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --run-dir) + mkdir -p "$2" || { echo "[kermt] failed to create --run-dir: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/runs") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --model-dir) + mkdir -p "$2" || { echo "[kermt] failed to create --model-dir: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/model") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + *) + return 0 + ;; + esac + done +} + +# ----------------------------------------------------------------------------- +# Foreground / detached run +# ----------------------------------------------------------------------------- + +# Capture host-side git state for the repo and emit `-e KERMT_REPO_COMMIT=… +# -e KERMT_REPO_DIRTY=true|false` flags. Used by the run / run_detached +# wrappers so the runner's run.json manifest gets honest commit info even +# though `git -C /workspace` inside the container fails due to bind-mount +# ownership. +_kermt_git_env_flags() { + local commit="unknown" + local dirty="false" + if command -v git >/dev/null 2>&1 && [[ -d "$KERMT_REPO/.git" ]]; then + local c + c=$(git -C "$KERMT_REPO" rev-parse HEAD 2>/dev/null) && commit="$c" + # `--untracked-files=no` filters out user-private notes (e.g. a CLAUDE.md + # or RELEASE_PLAN_v2.0.md at the repo root) that wouldn't affect + # reproducibility — only modifications to tracked files do. + if [[ -n "$(git -C "$KERMT_REPO" status --porcelain --untracked-files=no 2>/dev/null | head -n 1)" ]]; then + dirty="true" + fi + fi + printf '%s\n%s\n%s\n%s\n' "-e" "KERMT_REPO_COMMIT=$commit" "-e" "KERMT_REPO_DIRTY=$dirty" +} + +# Forward HF_TOKEN into the container when it is set, so fetch_released_model.py +# can authenticate to Hugging Face. The current release is public (no token +# needed); this only guards against shared-IP rate limits or a future gated +# repo. Emits nothing when HF_TOKEN is unset. +_kermt_hf_env_flags() { + if [[ -n "${HF_TOKEN:-}" ]]; then + printf '%s\n%s\n' "-e" "HF_TOKEN=$HF_TOKEN" + fi +} + +kermt_run() { + kermt_ensure_image || return $? + local mount_args=() + _kermt_parse_mounts mount_args "$@" || return $? + shift "$_kermt_consumed" + if [[ "${1:-}" != "--" ]]; then + echo "[kermt] expected '--' separating mount flags from the command (got '${1:-}')" >&2 + return 1 + fi + shift + if [[ $# -eq 0 ]]; then + echo "[kermt] no command supplied after '--'" >&2 + return 1 + fi + local git_args=() + while IFS= read -r line; do git_args+=("$line"); done < <(_kermt_git_env_flags) + local hf_args=() + while IFS= read -r line; do hf_args+=("$line"); done < <(_kermt_hf_env_flags) + docker run --rm --gpus "$KERMT_GPUS" \ + --user "$(id -u):$(id -g)" \ + -v "$KERMT_REPO:/workspace" \ + -v "$_kermt_bundle_dir:/skill:ro" \ + "${mount_args[@]}" \ + -w /workspace \ + -e KERMT_REPO=/workspace \ + -e PYTHONPATH=/workspace \ + -e HOME=/tmp/kermt-home \ + "${git_args[@]}" \ + "${hf_args[@]}" \ + "$KERMT_IMAGE" \ + conda run -n kermt --no-capture-output bash -c "$*" +} + +kermt_run_detached() { + kermt_ensure_image || return $? + local name="" + local mount_args=() + # Pull --name out first, then let the shared mount parser handle the rest. + while [[ $# -gt 0 ]]; do + case "$1" in + --name) name="$2"; shift 2 ;; + --) break ;; + --data|--ckpt|--vocab-dir|--run-dir|--model-dir) break ;; + *) break ;; + esac + done + _kermt_parse_mounts mount_args "$@" || return $? + shift "$_kermt_consumed" + if [[ "${1:-}" != "--" ]]; then + echo "[kermt] expected '--' separating mount flags from the command (got '${1:-}')" >&2 + return 1 + fi + shift + if [[ $# -eq 0 ]]; then + echo "[kermt] no command supplied after '--'" >&2 + return 1 + fi + if [[ -z "$name" ]]; then + name="kermt-$(date -u +%Y%m%dT%H%M%SZ)-$$" + fi + local cid + local git_args=() + while IFS= read -r line; do git_args+=("$line"); done < <(_kermt_git_env_flags) + local hf_args=() + while IFS= read -r line; do hf_args+=("$line"); done < <(_kermt_hf_env_flags) + cid=$(docker run -d --gpus "$KERMT_GPUS" \ + --user "$(id -u):$(id -g)" \ + --name "$name" \ + -v "$KERMT_REPO:/workspace" \ + -v "$_kermt_bundle_dir:/skill:ro" \ + "${mount_args[@]}" \ + -w /workspace \ + -e KERMT_REPO=/workspace \ + -e PYTHONPATH=/workspace \ + -e HOME=/tmp/kermt-home \ + "${git_args[@]}" \ + "${hf_args[@]}" \ + "$KERMT_IMAGE" \ + conda run -n kermt --no-capture-output bash -c "$*") || return $? + echo "[kermt] container started: name=$name id=$cid" + echo "[kermt] follow logs: docker logs -f $name" + echo "[kermt] wait for exit: docker wait $name" + echo "[kermt] stop: docker stop $name" + echo "$cid" +} + +# ----------------------------------------------------------------------------- +# Subcommand dispatch when invoked directly (not sourced) +# ----------------------------------------------------------------------------- + +if [[ "${BASH_SOURCE[0]:-$0}" == "${0}" ]]; then + cmd="${1:-}"; shift || true + case "$cmd" in + check_docker) kermt_check_docker "$@" ;; + check_gpu) kermt_check_gpu "$@" ;; + check_system) kermt_check_system "$@" ;; + ensure_image) kermt_ensure_image "$@" ;; + run) kermt_run "$@" ;; + run_detached) kermt_run_detached "$@" ;; + ""|-h|--help) + cat >&2 < [args...] + +Subcommands: + check_docker Verify docker is installed and the daemon is reachable. + check_gpu Verify 'docker --gpus all' works (nvidia-container-toolkit). + check_system Emit a JSON probe of host GPU + VRAM + compute_cap + + driver + disk space + container toolkit + image presence. + Exits 0 with ok=false + a 'gaps' list when anything's + below the per-workflow minimum. + ensure_image Build kermt:latest from \$KERMT_REPO/Dockerfile if missing. + run [flags] -- ... Run a command inside the container (foreground, --rm). + run_detached [flags] -- ... + Run detached; prints container name + id + log hint. + +Mount flags (for run / run_detached): + --data bind to /data (read-only) + --ckpt bind to /ckpt (read-only) + --vocab-dir bind to /vocab (read-only) + --run-dir bind to /runs (read-write; created on host if missing) + --model-dir bind to /model (read-write; released-model download target) + +Additional flags for run_detached: + --name container name (default: kermt--) + +Environment overrides: + KERMT_IMAGE default kermt:latest + KERMT_REPO checkout path; otherwise discovered above the skill or working directory + KERMT_GPUS default all +EOF + exit 1 + ;; + *) + echo "[kermt] unknown subcommand: $cmd" >&2 + echo "[kermt] run '$0 --help' for usage" >&2 + exit 1 + ;; + esac +fi diff --git a/skills/kermt-add-cmim-pretrain/scripts/prepare_data.py b/skills/kermt-add-cmim-pretrain/scripts/prepare_data.py new file mode 100644 index 0000000..f0edb3e --- /dev/null +++ b/skills/kermt-add-cmim-pretrain/scripts/prepare_data.py @@ -0,0 +1,817 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Mode-dispatched data preparation pipeline for the KERMT agent skills. + +Composes the existing repo data-prep scripts (`scripts/clean_smiles.py`, +`scripts/save_features.py`, `scripts/build_vocab.py`, `scripts/split_data.py`) +into a single one-call entry point per workflow. Output lands in `--out` with +a `prepare_data.json` manifest that the downstream runners read. + +Mode pipelines +-------------- +pretrain : clean -> (optional auto-split train into train+val by --val-frac) + -> save_features (fgtasklabel) on each CSV + -> vocab step: if --vocab-dir / --{atom,bond,smiles}-vocab given, + copy those through (continue-pretrain case — the ckpt's vocab + is authoritative); else if --skip-vocab, skip; + else build_vocab on train (pretrain-from-scratch case) + -> split_data (graph + feature shards + summary.txt) per CSV +finetune : clean each provided CSV -> (optional random split when only one + CSV is provided; emits a strong warning recommending scaffold- + balanced pre-splits) -> save_features (rdkit_2d_normalized) per CSV +inference : clean -> save_features (rdkit_2d_normalized) +embed : clean only (extract_embeddings.py featurizes on the fly) + +Output convention +----------------- +The manifest under `/prepare_data.json` captures every step's inputs, +outputs, duration, and skipped-due-to-existing flag, plus a top-level +`split_method` field (one of: "user_provided", "random", "n/a") that the +finetune runner uses to pass the correct `--split_type` to main.py. + +Subprocess composition +---------------------- +Each underlying script is invoked via `subprocess.run`. The PYTHONPATH=/workspace +env var (set by `scripts/kermt_container.sh`) makes the `kermt` package +importable inside the subprocesses; without it, build_vocab.py and split_data.py +fail with `ModuleNotFoundError: No module named 'kermt'`. + +CLI +--- + prepare_data.py --mode {pretrain|finetune|inference|embed} + --csv --out + [--val-csv ] [--test-csv ] + [--val-frac 0.1] [--test-frac 0.1] [--seed 0] + [--sample-per-file 100000] [--vocab-format json] + [--dataset-name pretrain] + [--targets COL [COL ...]] + [--features-generator ] + [--smiles-column 0] + [--force] [--skip-clean] [--skip-features] + [--skip-vocab] [--skip-split] +""" +from __future__ import annotations + +import argparse +import json +import os +import shutil +import subprocess +import sys +import time +import traceback +from pathlib import Path +from typing import Any + +import pandas as pd + +# sys.path tweak so `_utils` is importable regardless of how this script +# is invoked (kermt_run sets PYTHONPATH=/workspace; bare-Python launches +# from the host don't). +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import PRETRAIN_VOCAB_STEMS, resolve_kermt_repo, validate_vocab_file # noqa: E402 + + +REPO_ROOT = resolve_kermt_repo() +EXISTING_SCRIPTS = REPO_ROOT / "scripts" + +DEFAULT_FEATURES_GENERATOR = { + "pretrain": "fgtasklabel", + "finetune": "rdkit_2d_normalized", + "inference": "rdkit_2d_normalized", + "embed": None, # not used +} + +VALID_MODES = ("pretrain", "finetune", "inference", "embed") + + +# --------------------------------------------------------------------------- +# Subprocess helpers +# --------------------------------------------------------------------------- + +def _run(cmd: list[str], step_name: str, manifest: dict[str, Any]) -> dict[str, Any]: + """Run a subprocess, append a step entry to manifest, raise on failure.""" + step: dict[str, Any] = { + "name": step_name, + "cmd": cmd, + "duration_s": None, + "ok": False, + "stderr_tail": "", + "skipped_due_to_existing": False, + } + t0 = time.time() + proc = subprocess.run(cmd, capture_output=True, text=True) + step["duration_s"] = round(time.time() - t0, 2) + if proc.returncode != 0: + step["stderr_tail"] = (proc.stderr or "").splitlines()[-20:] + step["ok"] = False + manifest["steps"].append(step) + raise RuntimeError( + f"step '{step_name}' failed (exit {proc.returncode}); " + f"command: {' '.join(cmd)}\nstderr tail:\n" + "\n".join(step["stderr_tail"]) + ) + step["ok"] = True + manifest["steps"].append(step) + return step + + +def _skipped(step_name: str, output_path: str, manifest: dict[str, Any]) -> dict[str, Any]: + step = { + "name": step_name, + "output": output_path, + "ok": True, + "duration_s": 0.0, + "skipped_due_to_existing": True, + } + manifest["steps"].append(step) + return step + + +def _exists_nonempty(path: Path) -> bool: + """File exists with non-zero size, or directory exists with at least one entry.""" + if not path.exists(): + return False + if path.is_file(): + return path.stat().st_size > 0 + if path.is_dir(): + try: + next(path.iterdir()) + return True + except StopIteration: + return False + return False + + +# --------------------------------------------------------------------------- +# Per-script wrappers +# --------------------------------------------------------------------------- + +def _resolve_smiles_column(csv_path: Path, explicit_value: int | None) -> int: + """Return the 0-based index of the SMILES column in csv_path. + + Auto-detection rule when `explicit_value is None`: + 1. Read the CSV header (first non-empty row). + 2. Prefer an exact lowercase `smiles` column (kermt convention). + 3. Otherwise accept a single case-insensitive match + (`SMILES`, `Smiles`, etc.). + 4. If no match (or multiple ambiguous matches), raise a ValueError + that surfaces the header so the user can disambiguate via + `--smiles-column N`. + + Real datasets routinely place SMILES at column index ≠ 0 + (e.g. openadmet's all.csv has "Molecule Name" at col 0 and "SMILES" + at col 1). Auto-detection prevents the silent 0-row-clean failure + mode where every row gets rejected because col 0 doesn't parse as + a SMILES string. + """ + if explicit_value is not None: + return explicit_value + + if not csv_path.is_file(): + raise ValueError(f"input CSV not found: {csv_path}") + + import csv as _csv + with csv_path.open("r", newline="") as f: + reader = _csv.reader(f) + try: + header = next(reader) + except StopIteration: + raise ValueError(f"input CSV {csv_path} is empty") + + stripped = [c.strip() for c in header] + # Prefer exact lowercase "smiles" + exact = [i for i, c in enumerate(stripped) if c == "smiles"] + if exact: + return exact[0] + # Then case-insensitive + ci = [i for i, c in enumerate(stripped) if c.lower() == "smiles"] + if len(ci) == 1: + return ci[0] + if len(ci) > 1: + raise ValueError( + f"input CSV {csv_path} has multiple SMILES-named columns: " + f"{[header[i] for i in ci]} at indices {ci}. " + "Pass --smiles-column N (0-based) to disambiguate." + ) + raise ValueError( + f"could not auto-detect a SMILES column in {csv_path}. " + f"Header columns: {header}. " + "Pass --smiles-column N (0-based) to specify which column holds SMILES." + ) + + +def _clean_smiles( + input_csv: Path, output_csv: Path, smiles_column: int, manifest: dict[str, Any], force: bool +) -> Path: + if not force and _exists_nonempty(output_csv): + _skipped(f"clean_smiles({input_csv.name})", str(output_csv), manifest) + return output_csv + output_csv.parent.mkdir(parents=True, exist_ok=True) + if force and output_csv.exists(): + # clean_smiles.py prompts interactively (input()) when the output file + # already exists — that's an EOFError in a non-TTY subprocess. Pre-delete. + output_csv.unlink() + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "clean_smiles.py"), + "--input", str(input_csv), + "--output", str(output_csv), + "--smiles_column", str(smiles_column), + ] + _run(cmd, f"clean_smiles({input_csv.name})", manifest) + return output_csv + + +def _reduce_to_smiles_column( + csv_path: Path, smiles_column: int, manifest: dict[str, Any] +) -> Path: + """Rewrite an inference CSV to keep only the SMILES column (at index 0). + + Downstream `kermt.util.utils.get_data` -> `MoleculeDatapoint.__init__` + floats every column after SMILES, which crashes on non-numeric passthrough + columns (e.g. a 'split' label of 'train'/'val'/'test', or a 'Molecule Name' + string). Inference does not need target columns, so drop them here. + + Note on skip semantics: this step is idempotent — running it on an + already-single-column file is a no-op. We record that with + `skipped_due_to_idempotent: True`, NOT `skipped_due_to_existing: True`. + The two fields have different meanings: `_existing` means "I found a + cached output file from a prior run and reused it" (overridden by + `--force`); `_idempotent` means "the input is already in the desired + state, so re-executing changes nothing" (safe to skip even under + `--force`). + """ + step_name = f"reduce_to_smiles_only({csv_path.name})" + start = time.time() + df = pd.read_csv(csv_path) + if df.shape[1] == 1: + manifest["steps"].append({ + "name": step_name, + "output": str(csv_path), + "ok": True, + "duration_s": time.time() - start, + "skipped_due_to_idempotent": True, + "note": "already single-column", + }) + return csv_path + effective_col = smiles_column if 0 <= smiles_column < df.shape[1] else 0 + df.iloc[:, [effective_col]].to_csv(csv_path, index=False) + manifest["steps"].append({ + "name": step_name, + "output": str(csv_path), + "ok": True, + "duration_s": time.time() - start, + "input_cols": int(df.shape[1]), + "kept_col": effective_col, + "kept_col_name": str(df.columns[effective_col]), + }) + return csv_path + + +def _save_features( + csv_path: Path, npz_path: Path, generator: str, manifest: dict[str, Any], force: bool +) -> Path: + if not force and _exists_nonempty(npz_path): + _skipped(f"save_features({csv_path.name}, {generator})", str(npz_path), manifest) + return npz_path + npz_path.parent.mkdir(parents=True, exist_ok=True) + if force and npz_path.exists(): + npz_path.unlink() # --restart still loads partial state if file exists; pre-delete to be safe + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "save_features.py"), + "--data_path", str(csv_path), + "--save_path", str(npz_path), + "--features_generator", generator, + "--restart", + ] + _run(cmd, f"save_features({csv_path.name}, {generator})", manifest) + return npz_path + + +def _resolve_vocab_inputs(args: argparse.Namespace) -> dict[str, Path | None] | None: + """Returns {atom, bond, smiles}->Path|None when the user supplied vocab + inputs (via --vocab-dir or --atom-vocab/--bond-vocab/--smiles-vocab), + else None (signal to fall through to build_vocab). + + Conventional filenames inside --vocab-dir: + pretrain_atom_vocab.{json,pkl} + pretrain_bond_vocab.{json,pkl} + pretrain_smiles_vocab.pkl + """ + if args.vocab_dir: + d = Path(args.vocab_dir).resolve() + if not d.is_dir(): + raise FileNotFoundError(f"--vocab-dir not found or not a directory: {d}") + def _find(stem: str, exts: tuple[str, ...]) -> Path | None: + for ext in exts: + p = d / f"{stem}.{ext}" + if p.is_file(): + return p + return None + atom = _find(PRETRAIN_VOCAB_STEMS["atom"], ("json", "pkl")) + bond = _find(PRETRAIN_VOCAB_STEMS["bond"], ("json", "pkl")) + smiles = _find(PRETRAIN_VOCAB_STEMS["smiles"], ("pkl",)) + if atom is None and bond is None and smiles is None: + stems = [PRETRAIN_VOCAB_STEMS[k] for k in ("atom", "bond", "smiles")] + raise FileNotFoundError( + f"--vocab-dir {d} contained no {{ {', '.join(stems) }}}.{{json,pkl}} " + f"files. Expected at least {PRETRAIN_VOCAB_STEMS['atom']} + " + f"{PRETRAIN_VOCAB_STEMS['bond']}." + ) + return {"atom": atom, "bond": bond, "smiles": smiles} + + if args.atom_vocab or args.bond_vocab or args.smiles_vocab: + return { + "atom": Path(args.atom_vocab).resolve() if args.atom_vocab else None, + "bond": Path(args.bond_vocab).resolve() if args.bond_vocab else None, + "smiles": Path(args.smiles_vocab).resolve() if args.smiles_vocab else None, + } + + return None + + +def _copy_provided_vocab( + src: dict[str, Path | None], dst_dir: Path, dataset_name: str, manifest: dict[str, Any], + force: bool, +) -> dict[str, Path]: + """When the user supplies vocab files (use ckpt's vocab as-is), + copy them into `/__vocab.` so the + downstream pretrain command sees the conventional filenames. + + `src` is `{atom: Path|None, bond: Path|None, smiles: Path|None}`. The atom + and bond entries must be both present or both absent (paired). smiles is + optional (cmim/hybrid only). + + Returns the same dict of (resolved) destination paths. + """ + import shutil + if (src["atom"] is None) != (src["bond"] is None): + raise ValueError( + "vocab pass-through requires atom and bond vocab paths to be paired; " + "got atom=" + str(src["atom"]) + ", bond=" + str(src["bond"]) + ) + out: dict[str, Path] = {} + dst_dir.mkdir(parents=True, exist_ok=True) + for which, path in src.items(): + if path is None: + continue + # Validate the source file IS a loadable KERMT vocab before copying. + # Catches the "user pointed --smiles-vocab at a random pickle" case + # early, with a clear error, instead of letting it surface as a cryptic + # SMILESVocab.load_vocab failure at pretrain_ddp.py launch time. + validate_vocab_file(path, kind=which) + ext = path.suffix.lstrip(".") + if which == "smiles": + ext = "pkl" # smiles vocab is always pickle + dst = dst_dir / f"{dataset_name}_{which}_vocab.{ext}" + if not force and _exists_nonempty(dst): + _skipped(f"copy_vocab({which})", str(dst), manifest) + out[which] = dst + continue + if force and dst.exists(): + dst.unlink() + shutil.copy2(path, dst) + manifest["steps"].append({ + "name": f"copy_vocab({which})", + "src": str(path), "dst": str(dst), "ok": True, + "duration_s": 0.0, "skipped_due_to_existing": False, + }) + out[which] = dst + return out + + +def _build_vocab( + csv_path: Path, vocab_dir: Path, dataset_name: str, vocab_format: str, + manifest: dict[str, Any], force: bool, +) -> dict[str, Path]: + """Builds atom + bond (in --vocab-format) and smiles (always pickle) vocabs. + Returns a dict of {atom, bond, smiles} -> Path.""" + suffix = "json" if vocab_format == "json" else "pkl" + expected = { + "atom": vocab_dir / f"{dataset_name}_atom_vocab.{suffix}", + "bond": vocab_dir / f"{dataset_name}_bond_vocab.{suffix}", + "smiles": vocab_dir / f"{dataset_name}_smiles_vocab.pkl", + } + if not force and all(_exists_nonempty(p) for p in expected.values()): + _skipped(f"build_vocab({csv_path.name})", str(vocab_dir), manifest) + return expected + vocab_dir.mkdir(parents=True, exist_ok=True) + if force: + for p in expected.values(): + if p.exists(): + p.unlink() + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "build_vocab.py"), + "--data_path", str(csv_path), + "--vocab_save_folder", str(vocab_dir), + "--dataset_name", dataset_name, + "--vocab_format", vocab_format, + ] + _run(cmd, f"build_vocab({csv_path.name})", manifest) + return expected + + +def _split_data( + csv_path: Path, features_path: Path | None, sample_per_file: int, output_dir: Path, + manifest: dict[str, Any], force: bool, +) -> Path: + """Run split_data.py to produce shard dirs (graph/ + optionally feature/ + summary.txt).""" + summary = output_dir / "summary.txt" + if not force and _exists_nonempty(summary): + _skipped(f"split_data({csv_path.name})", str(output_dir), manifest) + return output_dir + if force and output_dir.exists(): + shutil.rmtree(output_dir) + output_dir.mkdir(parents=True, exist_ok=True) + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "split_data.py"), + "--data_path", str(csv_path), + "--sample_per_file", str(sample_per_file), + "--output_path", str(output_dir), + ] + if features_path is not None: + cmd += ["--features_path", str(features_path)] + _run(cmd, f"split_data({csv_path.name})", manifest) + return output_dir + + +# --------------------------------------------------------------------------- +# Random splitter (used only when the user supplies a single CSV) +# --------------------------------------------------------------------------- + +def _random_split_csv( + src_csv: Path, dst_csvs: dict[str, Path], fractions: dict[str, float], seed: int, + manifest: dict[str, Any], force: bool, +) -> None: + """Shuffle src_csv and partition rows into dst_csvs by fractions. + `dst_csvs` and `fractions` are dicts keyed by the split name (e.g. 'train', 'val'). + Sum of fractions must be 1.0 (within float tolerance). Writes each dst_csv with the + same header as the input.""" + step = { + "name": f"random_split({src_csv.name})", + "seed": seed, + "fractions": fractions, + "ok": False, + "duration_s": None, + "skipped_due_to_existing": False, + "row_counts": {}, + } + if not force and all(_exists_nonempty(p) for p in dst_csvs.values()): + step["skipped_due_to_existing"] = True + step["ok"] = True + manifest["steps"].append(step) + return + + if abs(sum(fractions.values()) - 1.0) > 1e-6: + raise ValueError(f"split fractions must sum to 1.0 (got {sum(fractions.values())})") + + t0 = time.time() + df = pd.read_csv(src_csv).sample(frac=1.0, random_state=seed).reset_index(drop=True) + n = len(df) + sizes: dict[str, int] = {} + remaining = n + split_names = list(fractions.keys()) + for name in split_names[:-1]: + sizes[name] = int(round(fractions[name] * n)) + remaining -= sizes[name] + sizes[split_names[-1]] = remaining + + start = 0 + for name in split_names: + dst = dst_csvs[name] + dst.parent.mkdir(parents=True, exist_ok=True) + df.iloc[start:start + sizes[name]].to_csv(dst, index=False) + step["row_counts"][name] = sizes[name] + start += sizes[name] + + step["duration_s"] = round(time.time() - t0, 2) + step["ok"] = True + manifest["steps"].append(step) + + +def _emit_random_split_warning( + src_csv: Path, fractions: dict[str, float], seed: int, manifest: dict[str, Any] +) -> None: + row_counts = manifest["steps"][-1].get("row_counts", {}) + n = sum(row_counts.values()) if row_counts else "?" + lines = [ + f"WARNING: Auto-splitting {n} rows from {src_csv.name} into:", + ] + for name, frac in fractions.items(): + cnt = row_counts.get(name, "?") + lines.append(f" {name}: {cnt} rows ({frac * 100:.1f}%)") + lines += [ + f"using random split with seed {seed}.", + "", + "This is a RANDOM split. For rigorous ADMET evaluation, scaffold-balanced", + "(or other structure-aware) splits are strongly preferred — molecules with", + "similar scaffolds can leak across splits and inflate apparent generalization.", + "", + "To use your own pre-computed splits instead, pass:", + " --train-csv --val-csv --test-csv ", + "", + "To customize fractions:", + " --val-frac 0.15 --test-frac 0.15", + ] + warning = "\n".join(lines) + print(warning, file=sys.stderr) + manifest["warnings"].append(warning) + + +# --------------------------------------------------------------------------- +# Mode pipelines +# --------------------------------------------------------------------------- + +def _prepare_embed(args, out: Path, manifest: dict[str, Any]) -> None: + manifest["split_method"] = "n/a" + if args.skip_clean: + clean = Path(args.csv) + manifest["steps"].append({"name": "clean_smiles", "skipped_by_flag": True, "ok": True}) + else: + clean = _clean_smiles(Path(args.csv), out / "clean.csv", args.smiles_column, manifest, args.force) + manifest["outputs"]["clean_csv"] = str(clean) + + +def _prepare_inference(args, out: Path, manifest: dict[str, Any]) -> None: + manifest["split_method"] = "n/a" + clean = _clean_smiles(Path(args.csv), out / "clean.csv", args.smiles_column, manifest, args.force) + # Reduce to SMILES-only: downstream get_data/MoleculeDatapoint floats every + # non-SMILES column, which crashes on non-numeric passthrough columns + # (e.g. a 'split' label). Inference does not need target columns. + _reduce_to_smiles_column(clean, args.smiles_column, manifest) + manifest["outputs"]["clean_csv"] = str(clean) + if args.skip_features: + manifest["steps"].append({"name": "save_features", "skipped_by_flag": True, "ok": True}) + return + generator = args.features_generator or DEFAULT_FEATURES_GENERATOR["inference"] + npz = _save_features(clean, out / "clean.npz", generator, manifest, args.force) + manifest["outputs"]["clean_npz"] = str(npz) + + +def _prepare_finetune(args, out: Path, manifest: dict[str, Any]) -> None: + src_train = Path(args.csv) + has_val = args.val_csv is not None + has_test = args.test_csv is not None + split_type = args.split_type + + if has_val and has_test: + # User supplied explicit val + test CSVs: trust them, just clean + featurize. + # split_type is irrelevant when val/test are given separately. + manifest["split_method"] = "user_provided" + clean_train = _clean_smiles(src_train, out / "clean_train.csv", args.smiles_column, manifest, args.force) + clean_val = _clean_smiles(Path(args.val_csv), out / "clean_val.csv", args.smiles_column, manifest, args.force) + clean_test = _clean_smiles(Path(args.test_csv), out / "clean_test.csv", args.smiles_column, manifest, args.force) + manifest["outputs"]["clean_train_csv"] = str(clean_train) + manifest["outputs"]["clean_val_csv"] = str(clean_val) + manifest["outputs"]["clean_test_csv"] = str(clean_test) + per_split = (("train", clean_train), ("val", clean_val), ("test", clean_test)) + elif has_val or has_test: + raise ValueError( + "for finetune mode, either provide BOTH --val-csv and --test-csv (user-provided splits) " + "or NEITHER (run with --split-type {random|scaffold_balanced|index_predetermined}). " + "Got one but not both." + ) + elif split_type == "random": + # Random auto-split — done here in prep so train.py gets ready-made CSVs. + manifest["split_method"] = "random" + manifest["split_seed"] = args.seed + train_frac = max(0.0, 1.0 - args.val_frac - args.test_frac) + manifest["split_fractions"] = {"train": train_frac, "val": args.val_frac, "test": args.test_frac} + clean_full = _clean_smiles(src_train, out / "_clean_full.csv", args.smiles_column, manifest, args.force) + dst = { + "train": out / "clean_train.csv", + "val": out / "clean_val.csv", + "test": out / "clean_test.csv", + } + _random_split_csv(clean_full, dst, manifest["split_fractions"], args.seed, manifest, args.force) + clean_train, clean_val, clean_test = dst["train"], dst["val"], dst["test"] + manifest["outputs"]["clean_train_csv"] = str(clean_train) + manifest["outputs"]["clean_val_csv"] = str(clean_val) + manifest["outputs"]["clean_test_csv"] = str(clean_test) + _emit_random_split_warning(src_train, manifest["split_fractions"], args.seed, manifest) + per_split = (("train", clean_train), ("val", clean_val), ("test", clean_test)) + else: + # Scaffold-balanced or index-predetermined: prep cleans + featurizes the full + # CSV and defers actual splitting to task/train.py, which calls split_data + # with the user-supplied seed and split_sizes. + manifest["split_method"] = "deferred_to_runner" + manifest["split_type"] = split_type + manifest["split_seed"] = args.seed + manifest["split_fractions"] = { + "train": max(0.0, 1.0 - args.val_frac - args.test_frac), + "val": args.val_frac, + "test": args.test_frac, + } + clean_full = _clean_smiles(src_train, out / "clean_full.csv", args.smiles_column, manifest, args.force) + manifest["outputs"]["clean_full_csv"] = str(clean_full) + per_split = (("full", clean_full),) + + if args.skip_features: + manifest["steps"].append({"name": "save_features", "skipped_by_flag": True, "ok": True}) + return + generator = args.features_generator or DEFAULT_FEATURES_GENERATOR["finetune"] + for split_name, csv in per_split: + npz = _save_features(csv, csv.with_suffix(".npz"), generator, manifest, args.force) + manifest["outputs"][f"clean_{split_name}_npz"] = str(npz) + + +def _prepare_pretrain(args, out: Path, manifest: dict[str, Any]) -> None: + src_train = Path(args.csv) + if args.val_csv is not None: + manifest["split_method"] = "user_provided" + clean_train = _clean_smiles(src_train, out / "clean_train.csv", args.smiles_column, manifest, args.force) + clean_val = _clean_smiles(Path(args.val_csv), out / "clean_val.csv", args.smiles_column, manifest, args.force) + else: + manifest["split_method"] = "random" + manifest["split_seed"] = args.seed + train_frac = max(0.0, 1.0 - args.val_frac) + manifest["split_fractions"] = {"train": train_frac, "val": args.val_frac} + clean_full = _clean_smiles(src_train, out / "_clean_full.csv", args.smiles_column, manifest, args.force) + dst = {"train": out / "clean_train.csv", "val": out / "clean_val.csv"} + _random_split_csv(clean_full, dst, manifest["split_fractions"], args.seed, manifest, args.force) + clean_train, clean_val = dst["train"], dst["val"] + + manifest["outputs"]["clean_train_csv"] = str(clean_train) + manifest["outputs"]["clean_val_csv"] = str(clean_val) + + generator = args.features_generator or DEFAULT_FEATURES_GENERATOR["pretrain"] + if args.skip_features: + manifest["steps"].append({"name": "save_features", "skipped_by_flag": True, "ok": True}) + train_npz: Path | None = None + val_npz: Path | None = None + else: + train_npz = _save_features(clean_train, out / "clean_train.npz", generator, manifest, args.force) + val_npz = _save_features(clean_val, out / "clean_val.npz", generator, manifest, args.force) + manifest["outputs"]["clean_train_npz"] = str(train_npz) + manifest["outputs"]["clean_val_npz"] = str(val_npz) + + if args.skip_vocab: + manifest["steps"].append({"name": "build_vocab", "skipped_by_flag": True, "ok": True}) + manifest["vocab_source"] = "skipped" + else: + # Resolve user-provided vocab paths from --vocab-dir or explicit flags. + provided = _resolve_vocab_inputs(args) + if provided: + # Use the user-supplied (ckpt's) vocab as-is. Copy into the + # conventional filenames the downstream pretrain command expects. + vocabs = _copy_provided_vocab(provided, out, args.dataset_name, manifest, args.force) + manifest["vocab_source"] = "user_provided" + else: + # Fall back to the existing build-from-corpus behavior. Used by + # pretrain-from-scratch and by any continue case where the user + # explicitly wants a fresh vocab (rare, usually wrong). + vocabs = _build_vocab(clean_train, out, args.dataset_name, args.vocab_format, manifest, args.force) + manifest["vocab_source"] = "built_fresh" + if "atom" in vocabs: + manifest["outputs"]["atom_vocab"] = str(vocabs["atom"]) + if "bond" in vocabs: + manifest["outputs"]["bond_vocab"] = str(vocabs["bond"]) + if "smiles" in vocabs: + manifest["outputs"]["smiles_vocab"] = str(vocabs["smiles"]) + + if args.skip_split: + manifest["steps"].append({"name": "split_data", "skipped_by_flag": True, "ok": True}) + else: + train_dir = _split_data(clean_train, train_npz, args.sample_per_file, out / "train", manifest, args.force) + val_dir = _split_data(clean_val, val_npz, args.sample_per_file, out / "val", manifest, args.force) + manifest["outputs"]["train_dir"] = str(train_dir) + manifest["outputs"]["val_dir"] = str(val_dir) + + +# --------------------------------------------------------------------------- +# Entry point +# --------------------------------------------------------------------------- + +def prepare(args: argparse.Namespace) -> dict[str, Any]: + out = Path(args.out).resolve() + out.mkdir(parents=True, exist_ok=True) + manifest: dict[str, Any] = { + "mode": args.mode, + "input_csv": str(Path(args.csv).resolve()), + "val_csv": str(Path(args.val_csv).resolve()) if args.val_csv else None, + "test_csv": str(Path(args.test_csv).resolve()) if args.test_csv else None, + "output_dir": str(out), + "split_method": None, + "steps": [], + "outputs": {}, + "errors": [], + "warnings": [], + } + try: + if args.mode == "pretrain": + _prepare_pretrain(args, out, manifest) + elif args.mode == "finetune": + _prepare_finetune(args, out, manifest) + elif args.mode == "inference": + _prepare_inference(args, out, manifest) + elif args.mode == "embed": + _prepare_embed(args, out, manifest) + manifest["ok"] = True + except Exception as exc: # noqa: BLE001 + manifest["ok"] = False + manifest["errors"].append(f"{type(exc).__name__}: {exc}") + # Always write the manifest so partial-failure state is visible to the agent. + (out / "prepare_data.json").write_text(json.dumps(manifest, indent=2)) + return manifest + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Mode-dispatched data prep for the KERMT agent skills.") + p.add_argument("--mode", required=True, choices=VALID_MODES) + p.add_argument("--csv", required=True, help="Primary input CSV (train CSV for pretrain/finetune)") + p.add_argument("--out", required=True, help="Output directory") + p.add_argument("--val-csv", default=None, help="Optional separate val CSV (pretrain/finetune)") + p.add_argument("--test-csv", default=None, help="Optional separate test CSV (finetune only)") + p.add_argument("--val-frac", type=float, default=0.1, help="Auto-split val fraction (default 0.1)") + p.add_argument("--test-frac", type=float, default=0.1, help="Auto-split test fraction (finetune only, default 0.1)") + p.add_argument("--seed", type=int, default=0, help="Random split seed (default 0)") + p.add_argument("--split-type", choices=["random", "scaffold_balanced", "index_predetermined"], + default="random", + help="(finetune only, when --val-csv/--test-csv are not given) how to split. " + "'random' splits in prep using --val-frac/--test-frac/--seed. " + "'scaffold_balanced' and 'index_predetermined' defer the actual split to the " + "runner (task/train.py invokes split_data with the appropriate algorithm " + "using the user-supplied seed); prep only cleans + featurizes the full CSV.") + p.add_argument("--sample-per-file", type=int, default=100_000, + help="split_data shard size (pretrain only, default 100000)") + p.add_argument("--vocab-format", choices=["json", "pkl"], default="json", + help="atom/bond vocab format (default json); smiles vocab is always pkl") + # Vocab pass-through (pretrain mode): when continuing from a released ckpt, + # pass its bundled vocab files in so we don't rebuild a mismatched vocab. + p.add_argument("--vocab-dir", default=None, + help="(pretrain) directory containing pretrain_{atom,bond}_vocab.{json,pkl} " + "(+ pretrain_smiles_vocab.pkl for cmim/hybrid). When given, prepare_data " + "skips build_vocab and copies these files into the output dir under the " + "expected filenames. Used by kermt-continue-pretrain to bind the released " + "ckpt's vocab to the new corpus (the ckpt's vocab is authoritative).") + p.add_argument("--atom-vocab", default=None, + help="(pretrain) explicit atom vocab path; pairs with --bond-vocab. Overrides " + "--vocab-dir's pretrain_atom_vocab.* discovery if both are given.") + p.add_argument("--bond-vocab", default=None, + help="(pretrain) explicit bond vocab path; pairs with --atom-vocab.") + p.add_argument("--smiles-vocab", default=None, + help="(pretrain, cmim/hybrid) explicit smiles vocab .pkl path. Optional for " + "vocab-only pretrain.") + p.add_argument("--dataset-name", default="pretrain", + help="vocab filename prefix (default 'pretrain' so downstream pretrain commands " + "can reference pretrain_{atom,bond}_vocab.{json|pkl}, pretrain_smiles_vocab.pkl)") + p.add_argument("--targets", nargs="+", default=None, + help="(finetune only) target column names; forwarded to the finetune runner via the manifest") + p.add_argument("--features-generator", default=None, + help="Override the per-mode default (pretrain: fgtasklabel; finetune/inference: rdkit_2d_normalized)") + p.add_argument("--smiles-column", type=int, default=None, + help="0-based column index of SMILES in the input CSV. " + "When omitted, auto-detected by header name " + "(prefers lowercase `smiles`; accepts case-insensitive " + "`SMILES`/`Smiles`). Pass explicitly to override.") + p.add_argument("--force", action="store_true", + help="Re-run every step even if its outputs already exist") + p.add_argument("--skip-clean", action="store_true", help="(embed mode) skip the cleaning step") + p.add_argument("--skip-features", action="store_true", help="Skip feature generation") + p.add_argument("--skip-vocab", action="store_true", help="(pretrain) skip vocab build") + p.add_argument("--skip-split", action="store_true", help="(pretrain) skip shard split") + args = p.parse_args(argv) + + # Forward --targets through the manifest so the finetune runner can see them. + if args.mode == "finetune" and args.targets: + pass # captured in manifest below + + # Resolve the SMILES column index (auto-detect from header when the user + # didn't pass --smiles-column). This is the only point where args.csv is + # touched before downstream _clean_smiles calls fan it out. + try: + resolved_smiles_col = _resolve_smiles_column(Path(args.csv), args.smiles_column) + except ValueError as exc: + err_manifest = { + "ok": False, + "mode": args.mode, + "errors": [f"smiles-column resolution failed: {exc}"], + } + Path(args.out).mkdir(parents=True, exist_ok=True) + (Path(args.out) / "prepare_data.json").write_text(json.dumps(err_manifest, indent=2)) + print(json.dumps(err_manifest, indent=2)) + return 1 + if args.smiles_column is None: + print(f"[prepare_data] auto-detected --smiles-column {resolved_smiles_col} " + f"from {Path(args.csv).name} header", file=sys.stderr) + args.smiles_column = resolved_smiles_col + + try: + manifest = prepare(args) + except Exception as exc: # noqa: BLE001 + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({"ok": False, "errors": [f"unhandled: {type(exc).__name__}: {exc}"]}, indent=2)) + return 1 + if args.targets: + manifest["targets"] = list(args.targets) + # Record the resolved SMILES column so the manifest is self-describing. + manifest["smiles_column"] = args.smiles_column + (Path(args.out) / "prepare_data.json").write_text(json.dumps(manifest, indent=2)) + print(json.dumps(manifest, indent=2)) + return 0 if manifest.get("ok") else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-add-cmim-pretrain/scripts/run_pretrain_local.py b/skills/kermt-add-cmim-pretrain/scripts/run_pretrain_local.py new file mode 100644 index 0000000..5b8a8a1 --- /dev/null +++ b/skills/kermt-add-cmim-pretrain/scripts/run_pretrain_local.py @@ -0,0 +1,730 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Workstation pretrain runner — composes prepare_data + ckpt-validator outputs +into a pretrain_ddp.py invocation. + +Continues pretraining from a user-provided checkpoint. The model type +(grover_base / cmim / hybrid) is inferred from the validator's output and +drives the pretrain_ddp.py flag set; arch params come exclusively from the +ckpt; training/loss hyperparameters come from config/defaults_pretrain.json +with per-flag CLI overrides. + +How it interacts with pretrain_ddp.py's auto-resume: + pretrain_ddp.py looks at /last_checkpoint.pt and resumes from it + if present. The runner sets `--save_dir /ckpt` and symlinks the user's + input ckpt to /ckpt/last_checkpoint.pt so the resume path picks it up. + +Run.json manifest: + Records source-repo commit + image digest + a copy-pasteable `cmd_replay` + + per-flag `args_applied` so the artifact is self-contained and replayable. + +CLI +--- + run_pretrain_local.py + --ckpt # input pretrain ckpt (required) + --prepare-manifest # prepare_data.json from a prior prepare run + --out # output dir (typically runs/continue-pretrain_/) + [--ckpt-validator-out ] # cached check_checkpoint.py JSON; computed if absent + [--gpus 0,2] # subset of detected GPUs; default = all visible + [--dry-run] # write run.json + print command, do not execute + [--epochs N] [--batch-size N] [--init-lr F] [--max-lr F] [--final-lr F] + [--warmup-epochs F] [--weight-decay F] [--dropout F] + [--save-interval N] [--seed N] + [--vocab-loss-weight F] # hybrid only + [--latent-dim N] [--contrastive-temperature F] # cmim/hybrid only +""" +from __future__ import annotations + +import argparse +import datetime +import json +import os +import subprocess +import sys +from pathlib import Path +from typing import Any + +# Add the scripts/ dir to sys.path so `_utils` is importable whether +# this script is launched via `kermt_run` (PYTHONPATH=/workspace) or as a +# bare `python scripts/run_pretrain_local.py …` from the host. +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import ( # noqa: E402 + resolve_kermt_repo, assert_prepare_manifest_basics, count_vocab_entries, docker_image_digest, + format_cmd_replay, git_commit_with_env_override, load_json, + merge_default_into_applied, run_checkpoint_validator, +) + + +REPO_ROOT = resolve_kermt_repo() +SKILL_ROOT = Path(__file__).resolve().parent.parent +DEFAULTS_PATH = SKILL_ROOT / "config" / "defaults_pretrain.json" +CHECK_CHECKPOINT_PATH = SKILL_ROOT / "scripts" / "check_checkpoint.py" +PRETRAIN_DDP_PATH = REPO_ROOT / "pretrain_ddp.py" + +# Model-type → pretrain_ddp.py `--pretrain_mode` value. +MODEL_TYPE_TO_PRETRAIN_MODE = { + "grover_base": "vocab", + "cmim": "cmim", + "hybrid": "hybrid", +} + +# Hyperparameter flags the runner exposes for CLI override + the corresponding +# key path in defaults_pretrain.json. None means the value isn't in defaults +# (e.g. seed has a default but lives at the top of training; lookup is direct). +TRAINING_FLAGS = ( + "batch_size", "dropout", "epochs", "init_lr", "max_lr", "final_lr", + "warmup_epochs", "weight_decay", "save_interval", "seed", "tensorboard", + "use_cuikmolmaker_featurization", +) +LOSS_FLAGS = ("contrastive_temperature", "vocab_loss_weight") +DECODER_FLAGS = ( + "latent_dim", + "decoder_num_layers", + "decoder_num_attention_heads", + "decoder_ffn_hidden_size", + "decoder_dropout", + "decoder_max_seq_len", + "decoder_positional_encoding", + "decoder_gate_self_attn", + "decoder_gate_cross_attn", +) + +ARCH_FLAGS_FROM_CKPT = ( + "hidden_size", "depth", "num_attn_head", "activation", "backbone", + "embedding_output_type", "self_attention", +) + +# cMIM-decoder + latent-distribution arch fields. For continue-pretrain on a +# cmim/hybrid ckpt these MUST come from the ckpt's saved_args (so the model +# being constructed matches the ckpt's weights at load time); the +# defaults_pretrain.json `add_cmim_decoder` block is for add-cmim-pretrain's +# upgrade-time decoder construction only, and is intentionally ignored +# during continue-pretrain. +CMIM_DECODER_FLAGS_FROM_CKPT = ( + "latent_dim", + "decoder_num_layers", + "decoder_num_attention_heads", + "decoder_ffn_hidden_size", + "decoder_dropout", + "decoder_max_seq_len", + "decoder_positional_encoding", + "decoder_gate_self_attn", + "decoder_gate_cross_attn", +) + + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + +# JSON loading delegated to the shared _utils.load_json. Alias kept for the +# existing internal callsites that use the leading-underscore convention. +_load_json = load_json + + +def _detect_gpus(override: str | None) -> tuple[int, str]: + """Returns (world_size, CUDA_VISIBLE_DEVICES_string).""" + if override: + gpu_list = [g.strip() for g in override.split(",") if g.strip()] + return len(gpu_list), ",".join(gpu_list) + # Honor an existing CUDA_VISIBLE_DEVICES in the environment. + env = os.environ.get("CUDA_VISIBLE_DEVICES", "").strip() + if env: + ids = [g for g in env.split(",") if g] + return len(ids), ",".join(ids) + try: + import torch + n = torch.cuda.device_count() + except Exception: + n = 0 + return n, ",".join(str(i) for i in range(n)) + + +def _verify_prepare_manifest(manifest: dict[str, Any]) -> None: + assert_prepare_manifest_basics(manifest, "pretrain") + out = manifest.get("outputs", {}) + required_keys = ("train_dir", "val_dir", "atom_vocab", "bond_vocab") + missing = [k for k in required_keys if k not in out] + if missing: + raise ValueError( + f"prepare_data manifest is missing required outputs: {missing}. " + "Was prepare_data.py invoked with --skip-vocab or --skip-split?" + ) + + +def _apply_defaults(args: argparse.Namespace, defaults: dict[str, Any], + model_type: str, world_size: int) -> dict[str, dict[str, Any]]: + """Returns args_applied: dict mapping flag → {value, source}. + Source is 'user' if the user passed a value on the CLI, else 'default-config' + (from defaults_pretrain.json) or 'auto-1gpu' / 'auto-multi-gpu' for the + auto-fallback values. Only includes flags relevant to the model_type.""" + applied: dict[str, dict[str, Any]] = {} + + training_defaults = defaults.get("training", {}) + loss_defaults = defaults.get("loss", {}) + decoder_defaults = defaults.get("add_cmim_decoder", {}) + + for f in TRAINING_FLAGS: + merge_default_into_applied(applied, args, f, training_defaults) + + # Single-GPU fallback: batch_size 32, save_interval 500. + if world_size <= 1: + if applied.get("batch_size", {}).get("source") != "user": + applied["batch_size"] = {"value": 32, "source": "auto-1gpu"} + if applied.get("save_interval", {}).get("source") != "user": + applied["save_interval"] = {"value": 500, "source": "auto-1gpu"} + + if model_type in ("cmim", "hybrid"): + for f in LOSS_FLAGS if model_type == "hybrid" else ("contrastive_temperature",): + merge_default_into_applied(applied, args, f, loss_defaults) + for f in DECODER_FLAGS: + merge_default_into_applied(applied, args, f, decoder_defaults) + + return applied + + +def _arch_from_validator(validator_out: dict[str, Any]) -> dict[str, Any]: + arch = validator_out.get("arch") or {} + missing = [k for k in ARCH_FLAGS_FROM_CKPT if arch.get(k) is None] + if missing: + raise ValueError( + f"checkpoint validator did not surface required arch fields: {missing}. " + "If the ckpt has no saved_args blob, these can't be inferred from state-dict " + "shapes alone; please supply a ckpt with args saved (the standard " + "save_model_for_restart format)." + ) + return arch + + +def _build_argv( + *, world_size: int, gpus_str: str, out_dir: Path, manifest: dict[str, Any], + model_type: str, pretrain_mode: str, arch: dict[str, Any], + applied: dict[str, dict[str, Any]], +) -> list[str]: + """Constructs the full pretrain_ddp.py argument list as a list of strings.""" + outputs = manifest["outputs"] + argv = [sys.executable, "-u", str(PRETRAIN_DDP_PATH)] + + # Data + vocab paths + argv += ["--train_data_path", outputs["train_dir"], + "--val_data_path", outputs["val_dir"], + "--atom_vocab_path", outputs["atom_vocab"], + "--bond_vocab_path", outputs["bond_vocab"]] + if model_type in ("cmim", "hybrid"): + argv += ["--smiles_vocab_path", outputs["smiles_vocab"]] + + # Pretrain mode + loss + argv += ["--pretrain_mode", pretrain_mode] + if "vocab_loss_weight" in applied and model_type == "hybrid": + argv += ["--vocab_loss_weight", str(applied["vocab_loss_weight"]["value"])] + if "contrastive_temperature" in applied and model_type in ("cmim", "hybrid"): + argv += ["--contrastive_temperature", str(applied["contrastive_temperature"]["value"])] + # cMIM/decoder arch: emit every applied flag. For continue-pretrain on a + # cmim/hybrid ckpt, every entry will be source="ckpt_saved_args" (see the + # overlay loop in run()). For pretrain-from-scratch / add-cmim-pretrain + # the values come from defaults_pretrain.json's add_cmim_decoder block. + if model_type in ("cmim", "hybrid"): + for f in ("latent_dim", "decoder_num_layers", "decoder_num_attention_heads", + "decoder_ffn_hidden_size", "decoder_dropout", + "decoder_max_seq_len", "decoder_positional_encoding"): + if f in applied: + argv += [f"--{f}", str(applied[f]["value"])] + # Boolean store_true flags: emit the bare flag only when True. + if applied.get("decoder_gate_self_attn", {}).get("value"): + argv += ["--decoder_gate_self_attn"] + if applied.get("decoder_gate_cross_attn", {}).get("value"): + argv += ["--decoder_gate_cross_attn"] + + # Architecture — sourced from validator's arch block, never from CLI/defaults. + argv += [ + "--hidden_size", str(arch["hidden_size"]), + "--depth", str(arch["depth"]), + "--num_attn_head", str(arch["num_attn_head"]), + "--activation", str(arch["activation"]), + "--backbone", str(arch["backbone"]), + "--embedding_output_type", str(arch["embedding_output_type"]), + ] + if arch.get("self_attention"): + argv += ["--self_attention"] + + # Training schedule + for name in ("batch_size", "dropout", "epochs", "init_lr", "max_lr", "final_lr", + "warmup_epochs", "weight_decay", "save_interval", "seed"): + if name in applied: + argv += [f"--{name}", str(applied[name]["value"])] + if applied.get("tensorboard", {}).get("value"): + argv += ["--tensorboard"] + if applied.get("use_cuikmolmaker_featurization", {}).get("value"): + argv += ["--use_cuikmolmaker_featurization"] + + # W&B logging (pass-through; pretrain_ddp.py only inits W&B when project is set). + if "wandb_project" in applied: + argv += ["--wandb_project", str(applied["wandb_project"]["value"])] + if "wandb_run_name" in applied: + argv += ["--wandb_run_name", str(applied["wandb_run_name"]["value"])] + + # Where pretrain_ddp.py auto-resumes from (we'll symlink the user ckpt there). + argv += ["--save_dir", str(out_dir / "ckpt")] + + return argv + + +def _symlink_ckpt_into_save_dir(user_ckpt: Path, save_dir: Path) -> Path: + """--resume path: symlink the user ckpt as /last_checkpoint.pt. + pretrain_ddp.py's auto-resume then restores everything from the ckpt: + model weights, optimizer state, scheduler_step, epoch, batch_idx, + wandb_run_id.""" + save_dir.mkdir(parents=True, exist_ok=True) + link = save_dir / "last_checkpoint.pt" + if link.exists() or link.is_symlink(): + link.unlink() + # Symlink to the absolute user_ckpt so it works regardless of cwd. + link.symlink_to(user_ckpt.resolve()) + return link + + +# Schedule fields that --resume inherits from ckpt.saved_args and that default +# (fresh-schedule) mode takes from CLI/defaults_pretrain.json. +SCHEDULE_FLAGS = ("epochs", "warmup_epochs", "init_lr", "max_lr", "final_lr") + + +def _materialize_ckpt_for_fresh_schedule(user_ckpt: Path, save_dir: Path) -> Path: + """Default (fresh-schedule) continue-pretrain path: write a CLEANED copy + of the user ckpt to /last_checkpoint.pt with scheduler_step, + epoch, batch_idx, and wandb_run_id reset to fresh-start values. Model + weights AND optimizer state pass through unchanged — so Adam's running + moments warm-start the new schedule (helpful because the new init_lr is + usually close to the previous run's final_lr). + + Why a fresh-state copy instead of a symlink: pretrain_ddp.py's + `trainer.load()` restores EVERYTHING in the ckpt including scheduler_step + and epoch. We can't selectively load just the model + optimizer through + that code path. The minimal-invasive workaround is to materialize a + ckpt that has the unwanted counters zeroed before the loader sees it. + pretrain_ddp.py then restores everything as normal, but everything it + restores reads as a fresh-start. + + Cost: one ~700 MB disk write per run. Pretrain is days-long, so it's + negligible. Done on the host before docker run. + """ + import torch # delayed import — keeps the runner light in --dry-run paths + save_dir.mkdir(parents=True, exist_ok=True) + target = save_dir / "last_checkpoint.pt" + if target.exists() or target.is_symlink(): + target.unlink() + ckpt = torch.load(user_ckpt, map_location="cpu", weights_only=False) + if not isinstance(ckpt, dict) or "state_dict" not in ckpt: + raise ValueError( + f"ckpt {user_ckpt} is not in the expected save_model_for_restart " + "dict format (need at least 'state_dict' key)." + ) + ckpt["scheduler_step"] = 0 + ckpt["epoch"] = 0 + ckpt["batch_idx"] = 0 + ckpt["wandb_run_id"] = None + torch.save(ckpt, target) + return target + + +def _validate_resume_state(user_ckpt: Path) -> dict[str, Any]: + """--resume mode: confirm the ckpt was saved via the save_model_for_restart + format and carries the full state pretrain_ddp.py needs to resume mid-run + (optimizer state, scheduler_step, epoch, batch_idx). Returns a small + `resume_state` dict for the manifest so users can see what was restored. + Raises ValueError with a clear redirect if the ckpt is too lean.""" + import torch + ckpt = torch.load(user_ckpt, map_location="cpu", weights_only=False) + if not isinstance(ckpt, dict) or "state_dict" not in ckpt: + raise ValueError( + f"ckpt {user_ckpt} is not in the expected save_model_for_restart " + "dict format." + ) + required = ("optimizer", "scheduler_step", "epoch", "batch_idx") + missing = [k for k in required if k not in ckpt] + if missing: + raise ValueError( + f"--resume requires the ckpt to carry the full mid-run state, but " + f"these keys are missing: {missing}. The ckpt was probably saved " + "without enough metadata to pure-resume — use the default " + "fresh-schedule mode (drop --resume) if you just want to continue " + "training with a new schedule." + ) + return { + "scheduler_step": int(ckpt["scheduler_step"]), + "epoch": int(ckpt["epoch"]), + "batch_idx": int(ckpt["batch_idx"]), + "wandb_run_id": ckpt.get("wandb_run_id"), + } + + +# Vocab-entry counting delegated to _utils.count_vocab_entries. Alias kept for +# the existing internal callsites. +_count_vocab_entries = count_vocab_entries + + +def _verify_vocab_sizes_match_ckpt( + manifest: dict[str, Any], validator_out: dict[str, Any], model_type: str, +) -> dict[str, Any]: + """For continue-pretrain only: compare each vocab file's entry count against + the ckpt's vocab head dimensions. Aborts on mismatch with a helpful error + pointing the user at the matching vocab. Returns a `vocab_check` block to + attach to run.json for transparency.""" + ckpt_sizes = validator_out.get("vocab_sizes") or {"atom": None, "bond": None, "smiles": None} + outputs = manifest.get("outputs", {}) + check: dict[str, Any] = {"vocab_source": manifest.get("vocab_source", "unknown")} + for which in ("atom", "bond", "smiles"): + ckpt_size = ckpt_sizes.get(which) + vocab_path_str = outputs.get(f"{which}_vocab") + check[which] = {"ckpt_size": ckpt_size, "manifest_vocab": vocab_path_str, "manifest_size": None} + if ckpt_size is None: + # ckpt doesn't have this head; nothing to verify. + continue + # ckpt has this head — the manifest MUST include the corresponding vocab. + if not vocab_path_str: + raise ValueError( + f"ckpt has a '{which}' vocab head (size {ckpt_size}) but the prepare_data " + f"manifest doesn't include a {which}_vocab file. Rerun prepare_data with " + f"--vocab-dir (or --{which}-vocab ) so the runner " + f"can pass the matching vocab through." + ) + manifest_size = _count_vocab_entries(Path(vocab_path_str)) + check[which]["manifest_size"] = manifest_size + if manifest_size != ckpt_size: + raise ValueError( + f"{which} vocab size mismatch — ckpt's head expects {ckpt_size} entries, " + f"but {vocab_path_str} has {manifest_size}. The released ckpt's vocab is the " + f"authoritative one for continue-pretrain; pass --vocab-dir " + f"(or --{which}-vocab ) to prepare_data so vocab built from the new corpus " + f"isn't used. If you actually want to pretrain from scratch on a different " + f"vocab, use the kermt-pretrain-scratch workflow instead." + ) + return check + + +# --------------------------------------------------------------------------- +# Main flow +# --------------------------------------------------------------------------- + +def run(args: argparse.Namespace) -> dict[str, Any]: + out_dir = Path(args.out).resolve() + out_dir.mkdir(parents=True, exist_ok=True) + (out_dir / "ckpt").mkdir(parents=True, exist_ok=True) + (out_dir / "logs").mkdir(parents=True, exist_ok=True) + + from_scratch = bool(args.from_scratch) + resume = bool(args.resume) + + # Mode-conflict validation up-front so the user fails fast. + if from_scratch and resume: + raise ValueError("--resume is incompatible with --from-scratch.") + if resume and not args.ckpt: + raise ValueError("--resume requires --ckpt; nothing to resume from otherwise.") + if resume: + # CLI overrides of schedule args are forbidden in --resume mode — pure + # resume means the schedule shape from the ckpt is authoritative. + cli_overrides = [ + f for f in SCHEDULE_FLAGS if getattr(args, f, None) is not None + ] + if cli_overrides: + raise ValueError( + f"--resume inherits schedule args from the ckpt's saved_args; " + f"explicit CLI override is forbidden. You passed: {cli_overrides}. " + "Drop those flags to pure-resume, or use the default fresh-schedule " + "mode (no --resume) if you want a new schedule." + ) + + workflow = "pretrain-scratch" if from_scratch else "continue-pretrain" + if resume: + mode = "continue_pretrain_resume" + elif from_scratch: + mode = "pretrain_from_scratch" + else: + mode = "continue_pretrain_fresh_schedule" + + # 1. Load defaults + prepare manifest. + defaults = _load_json(DEFAULTS_PATH, name="defaults_pretrain.json") + prep_manifest_path = Path(args.prepare_manifest).resolve() + manifest = _load_json(prep_manifest_path, name="prepare_data.json") + _verify_prepare_manifest(manifest) + + # 2. Branch: continue-pretrain (load ckpt + validate) vs from-scratch (no ckpt). + ckpt: Path | None = None + validator_out: dict[str, Any] | None = None + link: Path | None = None + vocab_check: dict[str, Any] | None = None + resume_state: dict[str, Any] | None = None # populated only when --resume + + if from_scratch: + if args.ckpt: + raise ValueError("--from-scratch is incompatible with --ckpt; pass one or the other.") + if not args.pretrain_target_mode: + raise ValueError("--pretrain-target-mode is required when --from-scratch is set " + "(choose vocab, cmim, or hybrid).") + pretrain_mode = args.pretrain_target_mode + model_type = {"vocab": "grover_base", "cmim": "cmim", "hybrid": "hybrid"}[pretrain_mode] + # Arch from defaults_pretrain.json's `arch` group (with CLI overrides applied later + # if we expose any; for now we just use defaults). + arch_defaults = defaults.get("arch") or {} + if not arch_defaults: + raise ValueError("defaults_pretrain.json has no `arch` group; cannot pretrain from scratch.") + arch = {k: arch_defaults.get(k) for k in ARCH_FLAGS_FROM_CKPT} + # `latent_dim` lives in the add_cmim_decoder group for from-scratch cmim/hybrid; + # treat it as part of the arch for argv-building purposes. + if pretrain_mode in ("cmim", "hybrid"): + arch["latent_dim"] = (defaults.get("add_cmim_decoder") or {}).get("latent_dim") + else: + arch["latent_dim"] = None + else: + if not args.ckpt: + raise ValueError("--ckpt is required for continue-pretrain. " + "Use --from-scratch to pretrain a fresh model on the corpus.") + ckpt = Path(args.ckpt).resolve() + if args.ckpt_validator_out: + validator_out = _load_json(Path(args.ckpt_validator_out), name="ckpt validator output") + else: + validator_out = run_checkpoint_validator(ckpt, mode="continue_pretrain", script_path=CHECK_CHECKPOINT_PATH) + if not validator_out.get("ok"): + raise ValueError( + f"check_checkpoint.py rejected the input ckpt: {validator_out.get('errors')}" + ) + model_type = validator_out.get("model_type") + if model_type not in MODEL_TYPE_TO_PRETRAIN_MODE: + raise ValueError( + f"model_type='{model_type}' cannot continue pretrain. " + f"Supported: {sorted(MODEL_TYPE_TO_PRETRAIN_MODE)}. " + "For an encoder-only ckpt with no pretrain head, use the " + "upgrade_to_hybrid workflow." + ) + if model_type == "grover_base" and not validator_out.get("has_vocab_head"): + raise ValueError( + "grover_base ckpt has no vocab head — cannot continue vocab pretrain. " + "Use the upgrade_to_hybrid workflow to add a cMIM decoder, " + "or finetune directly from the encoder." + ) + pretrain_mode = MODEL_TYPE_TO_PRETRAIN_MODE[model_type] + arch = _arch_from_validator(validator_out) + # Vocab-size verification — refuse mismatched corpora before launching pretrain_ddp.py. + vocab_check = _verify_vocab_sizes_match_ckpt(manifest, validator_out, model_type) + # --resume needs the ckpt to carry the full mid-run state. Validate now; + # also surface what's being restored in the manifest. + if resume: + resume_state = _validate_resume_state(ckpt) + + # 3. GPU selection. + world_size, gpus_str = _detect_gpus(args.gpus) + if world_size <= 0: + raise ValueError( + "No GPUs detected. pretrain_ddp.py requires at least one CUDA device. " + "Set CUDA_VISIBLE_DEVICES or pass --gpus ." + ) + + # 4. Apply defaults + collect args_applied. + applied = _apply_defaults(args, defaults, model_type, world_size) + # --resume overlays schedule args from the ckpt's saved_args (the only path + # where source="ckpt_saved_args" can appear in args_applied). Fail loudly if + # any schedule field is missing from saved_args — pure-resume can't proceed + # without the original schedule shape. + if resume: + saved_args = validator_out.get("saved_args") or {} + missing = [f for f in SCHEDULE_FLAGS if f not in saved_args] + if missing: + raise ValueError( + f"--resume requires the ckpt's saved_args to include all schedule " + f"fields, but these are missing: {missing}. The ckpt was saved " + "without enough metadata to pure-resume — use the default " + "fresh-schedule mode and specify --epochs / --warmup-epochs / " + "--init-lr / --max-lr / --final-lr explicitly." + ) + for f in SCHEDULE_FLAGS: + applied[f] = {"value": saved_args[f], "source": "ckpt_saved_args"} + + # Continue-pretrain on a cmim/hybrid ckpt: cMIM/decoder arch must come + # from the ckpt's saved_args, not from defaults or CLI. This is the + # cmim/decoder analogue of the encoder-arch passthrough already done by + # `_arch_from_validator` (and matches the README guarantee that + # `add_cmim_decoder` defaults are ignored during continue-pretrain). + if not from_scratch and model_type in ("cmim", "hybrid"): + cli_latent_dim_override = args.latent_dim is not None + if cli_latent_dim_override: + raise ValueError( + "--latent-dim cannot be overridden during continue-pretrain on a " + "cmim/hybrid ckpt — the value is fixed by the ckpt's saved_args " + "(passing a different value would mismatch the loaded decoder " + "weights). Drop --latent-dim, or use kermt-pretrain-scratch if " + "you intentionally want a different latent dimension." + ) + saved_args = validator_out.get("saved_args") or {} + cmim_missing = [f for f in CMIM_DECODER_FLAGS_FROM_CKPT if f not in saved_args] + if cmim_missing: + raise ValueError( + f"continue-pretrain on a {model_type} ckpt requires the ckpt's " + f"saved_args to include cmim/decoder arch fields, but these are " + f"missing: {cmim_missing}. The ckpt was saved without enough " + "metadata to faithfully reconstruct the decoder." + ) + for f in CMIM_DECODER_FLAGS_FROM_CKPT: + applied[f] = {"value": saved_args[f], "source": "ckpt_saved_args"} + + # Optional W&B logging: pass-through, no defaults — forwarded only when the + # user sets --wandb-project (run name is honored only alongside a project). + for f in ("wandb_project", "wandb_run_name"): + v = getattr(args, f, None) + if v is not None: + applied[f] = {"value": v, "source": "user"} + + # 5. Build the pretrain_ddp.py argv. + argv = _build_argv( + world_size=world_size, gpus_str=gpus_str, out_dir=out_dir, manifest=manifest, + model_type=model_type, pretrain_mode=pretrain_mode, arch=arch, applied=applied, + ) + + # 6. (continue-pretrain only) Stage the ckpt into /last_checkpoint.pt + # so pretrain_ddp.py's auto-resume picks it up. Mode-dispatched: + # - --resume: symlink to user ckpt. pretrain_ddp.py restores everything + # (model + optimizer + scheduler_step + epoch + batch_idx + wandb_run_id). + # - default (fresh-schedule): materialize a state-cleaned copy of the + # ckpt — model weights + optimizer pass through, but scheduler_step / + # epoch / batch_idx / wandb_run_id are reset to 0/None. pretrain_ddp.py + # then builds a fresh NoamLR from CLI args and starts from step 0. + # Done unconditionally (including --dry-run) so the dry-run faithfully + # exercises ckpt I/O — catches corrupt ckpts / insufficient disk before + # the days-long real run. + if not from_scratch: + if resume: + link = _symlink_ckpt_into_save_dir(ckpt, out_dir / "ckpt") + else: + link = _materialize_ckpt_for_fresh_schedule(ckpt, out_dir / "ckpt") + + # 7. Build the run.json manifest. + commit, dirty = git_commit_with_env_override(REPO_ROOT) + image_tag = os.environ.get("KERMT_IMAGE", "kermt:latest") + image_digest = docker_image_digest(image_tag) + cmd_replay_env: dict[str, str] = {} + if gpus_str: + cmd_replay_env["CUDA_VISIBLE_DEVICES"] = gpus_str + cmd_replay_env["WORLD_SIZE"] = str(world_size) + cmd_replay = format_cmd_replay(argv, env=cmd_replay_env) + run_manifest = { + "workflow": workflow, + "mode": mode, # pretrain_from_scratch | continue_pretrain_fresh_schedule | continue_pretrain_resume + "started_at": datetime.datetime.now(datetime.timezone.utc).isoformat(), + "container": {"image_tag": image_tag, "image_digest": image_digest}, + "repo": {"commit": commit, "dirty": dirty}, + "inputs": { + "ckpt": str(ckpt) if ckpt else None, + "prepare_data_manifest": str(prep_manifest_path), + "ckpt_validator_out": ( + str(Path(args.ckpt_validator_out).resolve()) if args.ckpt_validator_out else None + ), + }, + "model_type": model_type, + "pretrain_mode": pretrain_mode, + "world_size": world_size, + "cuda_visible_devices": gpus_str, + "args_applied": applied, + "arch": arch, + "vocab_check": vocab_check, # None for from-scratch + "resume_state": resume_state, # None unless --resume; carries the restored scheduler_step / epoch / batch_idx / wandb_run_id from the ckpt + "save_dir": str(out_dir / "ckpt"), + "logs_dir": str(out_dir / "logs"), + "tensorboard_dir": str(out_dir / "logs" / "tb"), + "argv": argv, + "cmd_replay": cmd_replay, + "ok_to_replay": (not dirty) and (commit != "unknown"), + "dry_run": bool(args.dry_run), + "ckpt_symlink": str(link) if link else None, + "from_scratch": from_scratch, + } + (out_dir / "run.json").write_text(json.dumps(run_manifest, indent=2)) + + # 8. Execute (unless --dry-run). + if args.dry_run: + run_manifest["status"] = "dry_run" + return run_manifest + + env = os.environ.copy() + env["WORLD_SIZE"] = str(world_size) + if gpus_str: + env["CUDA_VISIBLE_DEVICES"] = gpus_str + + log_file = out_dir / "logs" / "pretrain_ddp.log" + with log_file.open("w") as logf: + proc = subprocess.run(argv, env=env, stdout=logf, stderr=subprocess.STDOUT) + run_manifest["exit_code"] = proc.returncode + run_manifest["status"] = "ok" if proc.returncode == 0 else "failed" + (out_dir / "run.json").write_text(json.dumps(run_manifest, indent=2)) + return run_manifest + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser( + description="Workstation pretrain runner (continue-pretrain by default, " + "or pretrain-from-scratch with --from-scratch).") + p.add_argument("--ckpt", default=None, + help="Path to the input pretrain checkpoint. Required for continue-pretrain; " + "omit when --from-scratch is set.") + p.add_argument("--from-scratch", action="store_true", + help="Pretrain a fresh model on the corpus (no input ckpt; arch from " + "defaults_pretrain.json; vocab built by prepare_data). Requires " + "--pretrain-target-mode.") + p.add_argument("--resume", action="store_true", + help="Resume an interrupted pretrain run (crashed / Ctrl-C / OOM). " + "Restores everything from the ckpt: model weights, optimizer " + "state, scheduler_step, epoch, batch_idx, wandb_run_id. Schedule " + "shape (epochs / warmup_epochs / init/max/final_lr) is inherited " + "from the ckpt's saved_args; CLI overrides of schedule flags are " + "REJECTED in this mode. Without --resume (default), continue-pretrain " + "loads only model weights + optimizer momentum from the ckpt and " + "starts a fresh schedule from CLI/defaults_pretrain.json — use that " + "default mode when continue-pretraining on a new corpus / new " + "objective / extended training (the common case).") + p.add_argument("--pretrain-target-mode", choices=["vocab", "cmim", "hybrid"], default=None, + help="(--from-scratch only) which pretrain objective to use for the fresh " + "model: vocab (grover_base-style), cmim, or hybrid (vocab + contrast). " + "No default — must be set explicitly so the user makes an informed " + "choice about the head config.") + p.add_argument("--prepare-manifest", required=True, + help="Path to a prepare_data.json (must be mode=pretrain)") + p.add_argument("--out", required=True, help="Output run directory") + p.add_argument("--ckpt-validator-out", default=None, + help="Optional cached check_checkpoint.py JSON; computed if absent") + p.add_argument("--gpus", default=None, + help="Comma-separated GPU ids (e.g. '0,1'). Default: all visible") + p.add_argument("--dry-run", action="store_true", + help="Write run.json and print the command without executing") + # Training overrides — all default to None so we can distinguish user-given vs default-config. + for f, t in [("epochs", int), ("batch-size", int), ("init-lr", float), ("max-lr", float), + ("final-lr", float), ("warmup-epochs", float), ("weight-decay", float), + ("dropout", float), ("save-interval", int), ("seed", int), + ("vocab-loss-weight", float), ("latent-dim", int), + ("contrastive-temperature", float)]: + p.add_argument(f"--{f}", type=t, default=None) + # Optional W&B logging (pass-through to pretrain_ddp.py; off unless project is set). + p.add_argument("--wandb-project", type=str, default=None, + help="W&B project name. When set, pretrain_ddp.py logs train/val losses.") + p.add_argument("--wandb-run-name", type=str, default=None, + help="Optional W&B run name (only used when --wandb-project is set).") + args = p.parse_args(argv) + + try: + manifest = run(args) + except (FileNotFoundError, ValueError, RuntimeError) as exc: + print(json.dumps({"ok": False, "errors": [f"{type(exc).__name__}: {exc}"]}, indent=2), + file=sys.stdout) + return 1 + except Exception as exc: # noqa: BLE001 + import traceback + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({"ok": False, "errors": [f"unhandled: {type(exc).__name__}: {exc}"]}, + indent=2)) + return 1 + + print(json.dumps({"ok": True, "manifest": manifest}, indent=2)) + return 0 if manifest.get("status") != "failed" else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-add-cmim-pretrain/scripts/upgrade_to_hybrid.py b/skills/kermt-add-cmim-pretrain/scripts/upgrade_to_hybrid.py new file mode 100644 index 0000000..65c93f6 --- /dev/null +++ b/skills/kermt-add-cmim-pretrain/scripts/upgrade_to_hybrid.py @@ -0,0 +1,393 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Convert a grover_base ckpt into a hybrid ckpt by adding a +randomly-initialized cMIM decoder + latent_dist on top of the loaded encoder. + +After upgrade, the saved checkpoint classifies as `model_type: hybrid` via +`check_checkpoint.py --mode continue_pretrain`, and the pretrain runner +(`run_pretrain_local.py`) drops it directly into the +`--pretrain_mode hybrid` dispatch path with no special-case logic. + +What's preserved from the input ckpt +------------------------------------ +- Encoder weights (kermt.encoders.* state-dict subset). For legacy + grover_base ckpts where the encoder lives under `grover.encoders.*`, + the prefix is renormalized to `kermt.encoders.*` to match the + KermtHybridTask layout. +- The saved `args` Namespace — augmented with hybrid-specific decoder fields + if absent (defaults from `config/defaults_pretrain.json`). + +What's fresh-initialized +------------------------ +- The cMIM decoder (`decoder.*` — SMILESTransformerDecoder) — Xavier-init. +- The latent distribution (`latent_dist.*` minus `latent_dist.kermt.*` — + the encoder lives inside latent_dist by reference, so it shares the + same weights as the top-level encoder). +- The vocab heads (`vocab_module.*`) — always fresh, sized to the + vocab built from the user's pretrain corpus. The previous heads (if the + input ckpt was a modern grover_base with vocab heads) are discarded — + rebuilding them is acceptable because continue-pretrain will retrain + them anyway, and matching the new vocab dimensions is more important + than warm-starting from the old vocab. + +CLI +--- + upgrade_to_hybrid.py + --ckpt # grover_base ckpt (encoder-only or + # encoder + vocab heads; legacy or modern) + --prepare-manifest # prepare_data.json from a prior + # `prepare_data.py --mode pretrain` run. + # The smiles_vocab + atom_vocab + + # bond_vocab from this manifest size + # the new heads. + --out # destination for the upgraded ckpt + [--latent-dim N] # override defaults_pretrain.json + [--seed 0] # for reproducible random init + +Exit code 0 + a one-line JSON summary on success. Exit code 1 + JSON-shaped +errors on failure (input rejection, vocab missing, shape mismatch, etc.). +""" +from __future__ import annotations + +import argparse +import json +import sys +import traceback +from argparse import Namespace +from pathlib import Path +from typing import Any + +import torch + +# sys.path tweak so `_utils` imports cleanly whether launched via `kermt_run` +# or as a bare `python scripts/upgrade_to_hybrid.py …`. +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import count_vocab_entries, load_json, run_checkpoint_validator # noqa: E402 + + +SKILL_ROOT = Path(__file__).resolve().parent.parent +DEFAULTS_PATH = SKILL_ROOT / "config" / "defaults_pretrain.json" +CHECK_CHECKPOINT_PATH = SKILL_ROOT / "scripts" / "check_checkpoint.py" + +# Fields KermtHybridTask + its sub-components read from args. Anything not +# present on the input ckpt's args gets filled from defaults_pretrain.json +# (decoder fields from add_cmim_decoder group; encoder fields from training +# group; the rest from hardcoded encoder-arch defaults below). +HYBRID_REQUIRED_DECODER_ARGS = ( + "decoder_num_layers", "decoder_num_attention_heads", "decoder_ffn_hidden_size", + "decoder_dropout", "decoder_max_seq_len", "decoder_positional_encoding", + "decoder_gate_self_attn", "decoder_gate_cross_attn", +) + +# Fields KERMTEmbedding (and its sub-modules) read at __init__ time. Legacy +# grover_base ckpts may be missing several of these — pre-cMIM args weren't a +# strict superset of what current KERMTEmbedding wants. +ENCODER_REQUIRED_ARGS_WITH_DEFAULTS = { + "dropout": 0.1, # not in legacy grover_base.pt + "bond_drop_rate": 0.0, + "self_attention": False, + "attn_hidden": 4, + "attn_out": 8, + "use_cuikmolmaker_featurization": False, + "input_layer": "fc", + "dense": False, + "bias": False, + "undirected": False, + "dist_coff": 0.1, + "num_mt_block": 1, + "embedding_output_type": "both", +} + + +# JSON loading + vocab counting delegated to _utils. Aliases kept for the +# existing internal callsites. +_load_json = load_json +_count_vocab = count_vocab_entries + + +def _run_validator(ckpt: Path) -> dict[str, Any]: + """Invoke check_checkpoint.py --mode upgrade_to_hybrid and return the JSON. + Thin alias around _utils.run_checkpoint_validator for the specific mode.""" + return run_checkpoint_validator(ckpt, mode="upgrade_to_hybrid", script_path=CHECK_CHECKPOINT_PATH) + + +def _augment_args(input_args: Namespace, decoder_defaults: dict[str, Any], latent_dim: int) -> Namespace: + """Make sure the args Namespace has every field KermtHybridTask reads. + Existing input args take precedence (the ckpt was trained with them, and + encoder arch needs them); only missing hybrid-specific fields are filled + from defaults_pretrain.json's add_cmim_decoder group.""" + out = Namespace(**vars(input_args)) if isinstance(input_args, Namespace) else Namespace(**input_args) + # Encoder-side fields KERMTEmbedding reads at __init__. Legacy grover_base + # ckpts may be missing several (e.g. `dropout`). Fill from hardcoded + # defaults — these are safe at upgrade time because the loaded weights + # determine the actual encoder behavior; missing dropout etc. only affects + # rebuild/forward semantics, which the continue-pretrain runner will + # override anyway via its own training defaults. + for field, default_value in ENCODER_REQUIRED_ARGS_WITH_DEFAULTS.items(): + if not hasattr(out, field): + setattr(out, field, default_value) + # Hybrid-specific decoder fields (only fill in if missing) + for field in HYBRID_REQUIRED_DECODER_ARGS: + if not hasattr(out, field): + setattr(out, field, decoder_defaults[field]) + # latent_dim and contrastive_temperature also fill in from defaults if absent + if not hasattr(out, "latent_dim"): + out.latent_dim = latent_dim + if not hasattr(out, "contrastive_temperature"): + out.contrastive_temperature = decoder_defaults.get("contrastive_temperature", 0.1) + # pretrain_mode and use_cmim flags — set explicitly to hybrid semantics + out.pretrain_mode = "hybrid" + if hasattr(out, "use_cmim"): + out.use_cmim = True + # vocab_loss_weight default from training/loss defaults (1.0 — set inside the runner; + # here we just need it on the Namespace so save_checkpoint records it) + if not hasattr(out, "vocab_loss_weight"): + out.vocab_loss_weight = 1.0 + return out + + +def _renormalize_encoder_keys(state_dict: dict[str, Any]) -> tuple[dict[str, Any], list[str]]: + """Extract the encoder subset of the input state_dict, renormalized so the + keys match the KermtHybridTask layout (`kermt.encoders.*`). + + Three input shapes are handled: + - `grover.encoders.*` (legacy original-GROVER ckpts) → rename to `kermt.encoders.*` + - `kermt.encoders.*` (modern repo-trained grover_base / hybrid) → keep as-is + - `latent_dist.kermt.encoders.*` (cmim ckpts) → reject — those should + go through `continue_pretrain`, not `upgrade_to_hybrid`. + + Vocab heads (`vocab_module.*`) and any other non-encoder keys are + DROPPED here — the upgraded hybrid task gets fresh vocab heads sized to + the user's prepare-data vocab. + + Returns (renormalized_state_dict, notes). `notes` is a list of strings + describing key handling for the manifest. + """ + out: dict[str, Any] = {} + notes: list[str] = [] + legacy_count = modern_count = vocab_dropped = other_dropped = 0 + + for k, v in state_dict.items(): + if k.startswith("latent_dist.kermt."): + raise ValueError( + "input ckpt has `latent_dist.kermt.*` keys — it appears to be a " + "cmim or hybrid ckpt, not a grover_base. Use kermt-continue-pretrain " + "(no upgrade needed) or kermt-pretrain-scratch instead." + ) + if k.startswith("grover.encoders."): + new_k = "kermt.encoders." + k[len("grover.encoders."):] + out[new_k] = v + legacy_count += 1 + elif k.startswith("kermt.encoders."): + out[k] = v + modern_count += 1 + elif k.startswith("vocab_module."): + vocab_dropped += 1 # fresh heads on the new hybrid + else: + other_dropped += 1 + + if legacy_count and modern_count: + notes.append(f"unexpected mix of legacy + modern encoder prefixes: " + f"{legacy_count} grover.encoders.* + {modern_count} kermt.encoders.*") + elif legacy_count: + notes.append(f"renamed {legacy_count} legacy `grover.encoders.*` keys to `kermt.encoders.*`") + elif modern_count: + notes.append(f"kept {modern_count} modern `kermt.encoders.*` keys as-is") + else: + raise ValueError("no encoder state-dict keys found in input ckpt " + "(expected `grover.encoders.*` or `kermt.encoders.*`)") + if vocab_dropped: + notes.append(f"dropped {vocab_dropped} `vocab_module.*` keys " + f"(new vocab heads sized to user's prepare-data vocab)") + if other_dropped: + notes.append(f"dropped {other_dropped} other keys (not encoder, not vocab)") + return out, notes + + +def upgrade(args: argparse.Namespace) -> dict[str, Any]: + """Main upgrade flow. Returns a dict summary suitable for printing as JSON.""" + summary: dict[str, Any] = { + "ok": False, + "input_ckpt": str(Path(args.ckpt).resolve()), + "output_ckpt": str(Path(args.out).resolve()), + "vocab_sizes_used": {"atom": None, "bond": None, "smiles": None}, + "notes": [], + "errors": [], + "warnings": [], + } + + # 1. Validate input via check_checkpoint --mode upgrade_to_hybrid. + validator = _run_validator(Path(args.ckpt)) + if not validator.get("ok"): + summary["errors"].extend(validator.get("errors", []) or ["check_checkpoint rejected the ckpt"]) + return summary + summary["input_model_type"] = validator.get("model_type") + + # 2. Read prepare manifest for vocab paths + sizes. + prep_path = Path(args.prepare_manifest) + prep = _load_json(prep_path, name="prepare_data.json") + if prep.get("mode") != "pretrain": + raise ValueError(f"prepare manifest mode='{prep.get('mode')}', expected 'pretrain'") + if not prep.get("ok"): + raise ValueError(f"prepare manifest reports ok=False: {prep.get('errors')}") + outs = prep.get("outputs", {}) + for required in ("atom_vocab", "bond_vocab", "smiles_vocab"): + if required not in outs: + raise ValueError( + f"prepare manifest missing {required}. kermt-add-cmim-pretrain needs the " + "smiles vocab to size the new decoder; pass a corpus through " + "`prepare_data.py --mode pretrain` (without --skip-vocab / --skip-features) first." + ) + atom_size = _count_vocab(Path(outs["atom_vocab"])) + bond_size = _count_vocab(Path(outs["bond_vocab"])) + smiles_size = _count_vocab(Path(outs["smiles_vocab"])) + summary["vocab_sizes_used"] = {"atom": atom_size, "bond": bond_size, "smiles": smiles_size} + + # 3. Load defaults + input ckpt. + defaults = _load_json(DEFAULTS_PATH, name="defaults_pretrain.json") + decoder_defaults = defaults.get("add_cmim_decoder") or {} + latent_dim = args.latent_dim if args.latent_dim is not None else decoder_defaults.get("latent_dim", 800) + + input_ckpt = torch.load(args.ckpt, map_location="cpu", weights_only=False) + input_args = input_ckpt.get("args") + if input_args is None: + raise ValueError( + "input ckpt has no `args` Namespace — cannot reconstruct the encoder architecture. " + "Original GROVER ckpts typically saved args; if this one didn't, the upgrade has no " + "way to recover hidden_size / depth / num_attn_head / etc." + ) + input_sd = input_ckpt["state_dict"] + + # 4. Strip DDP `module.` prefix if present. + if input_sd and all(k.startswith("module.") for k in input_sd): + input_sd = {k[len("module."):]: v for k, v in input_sd.items()} + summary["notes"].append("stripped DDP `module.` prefix from input state_dict") + + # 5. Filter input state_dict down to just the encoder keys. Vocab heads + # are intentionally dropped here — the new hybrid task (built in step 7) + # provides freshly-initialized vocab heads sized to the user's vocab, + # which is safer than carrying over heads sized to the input ckpt's + # (possibly different) vocab. + encoder_state, rename_notes = _renormalize_encoder_keys(input_sd) + summary["notes"].extend(rename_notes) + + # 6. Augment args with hybrid-specific decoder fields. + new_args = _augment_args(input_args, decoder_defaults, latent_dim) + # Ensure cuda flag is False for the construction step (we don't move to GPU here). + new_args.cuda = False + + # 7. Build the fresh KermtHybridTask. + torch.manual_seed(args.seed) + from kermt.model.models import KermtHybridTask, KERMTEmbedding # type: ignore + encoder = KERMTEmbedding(new_args) + hybrid_task = KermtHybridTask( + new_args, kermt=encoder, + latent_dim=latent_dim, + contrastive_temperature=new_args.contrastive_temperature, + smiles_vocab_size=smiles_size, + atom_vocab_size=atom_size, + bond_vocab_size=bond_size, + ) + + # 8. Load encoder weights into the new task. strict=False because decoder / + # latent_dist / vocab_module weren't in the input — they keep their fresh init. + load_result = hybrid_task.load_state_dict(encoder_state, strict=False) + missing_keys = list(load_result.missing_keys) + unexpected_keys = list(load_result.unexpected_keys) + # The encoder is shared between hybrid_task.kermt and hybrid_task.latent_dist.kermt; + # any `kermt.encoders.*` key that wasn't loaded into the latent_dist's encoder copy + # is benign because they're the same module by reference. Same for missing-keys that + # belong to decoder / latent_dist (non-kermt parts) / vocab_module — those are + # supposed to be fresh. + summary["encoder_load"] = { + "missing_keys_total": len(missing_keys), + "unexpected_keys_total": len(unexpected_keys), + "encoder_missing": [k for k in missing_keys if "encoders." in k][:5], + "non_encoder_missing_categories": _categorize_missing(missing_keys), + "unexpected_sample": unexpected_keys[:5], + } + if unexpected_keys: + summary["warnings"].append( + f"{len(unexpected_keys)} unexpected key(s) in encoder load — " + "likely arch drift between legacy GROVER and modern KERMTEmbedding." + ) + + # Surface unknown backbones (anything beyond the gtrans/dualtrans pair + # that pretrain_ddp.py's argparse + the model code both accept). The + # legacy `dualtrans` name is the same architecture as `gtrans` and is + # explicitly handled in kermt/model/models.py:200. + if getattr(new_args, "backbone", None) not in (None, "gtrans", "dualtrans"): + summary["warnings"].append( + f"upgraded ckpt has backbone='{new_args.backbone}', which is neither " + "'gtrans' nor 'dualtrans'. pretrain_ddp.py's --backbone argparse will reject " + "it. Either re-pretrain a gtrans grover_base from scratch via " + "kermt-pretrain-scratch, or extend the parsing.py:379 choices." + ) + + # 9. Build a fresh Adam optimizer over the new model (matches save_checkpoint format). + init_lr = getattr(new_args, "init_lr", 1e-5) + weight_decay = getattr(new_args, "weight_decay", 1e-7) + optimizer = torch.optim.Adam(hybrid_task.parameters(), lr=init_lr, weight_decay=weight_decay) + + # 10. Save in the save_checkpoint format that task/kermttrainer.py:save_checkpoint produces. + state = { + "args": new_args, + "state_dict": hybrid_task.state_dict(), + "optimizer": optimizer.state_dict(), + "scheduler_step": 0, + "batch_idx": 0, + "epoch": 0, + "data_scaler": None, + "features_scaler": None, + "wandb_run_id": None, + } + out_path = Path(args.out).resolve() + out_path.parent.mkdir(parents=True, exist_ok=True) + torch.save(state, out_path) + + summary["ok"] = True + summary["upgraded_state_dict_keys"] = len(state["state_dict"]) + return summary + + +def _categorize_missing(missing_keys: list[str]) -> dict[str, int]: + """Bucket missing keys by their state-dict prefix for the manifest.""" + cats: dict[str, int] = {} + for k in missing_keys: + first = k.split(".")[0] + cats[first] = cats.get(first, 0) + 1 + return cats + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Upgrade a grover_base ckpt to a hybrid (cMIM + vocab) ckpt.") + p.add_argument("--ckpt", required=True, help="Path to the input grover_base checkpoint") + p.add_argument("--prepare-manifest", required=True, + help="Path to a prepare_data.json (mode=pretrain) — its vocab sizes " + "size the new decoder + fresh vocab heads.") + p.add_argument("--out", required=True, help="Destination for the upgraded hybrid ckpt") + p.add_argument("--latent-dim", type=int, default=None, + help="Override defaults_pretrain.json's add_cmim_decoder.latent_dim (default 800)") + p.add_argument("--seed", type=int, default=0, + help="Random seed for the fresh decoder / latent / vocab head weights") + args = p.parse_args(argv) + + try: + summary = upgrade(args) + except (FileNotFoundError, ValueError, RuntimeError) as exc: + print(json.dumps({"ok": False, "errors": [f"{type(exc).__name__}: {exc}"]}, indent=2)) + return 1 + except Exception as exc: # noqa: BLE001 + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({"ok": False, "errors": [f"unhandled: {type(exc).__name__}: {exc}"]}, indent=2)) + return 1 + + print(json.dumps(summary, indent=2)) + return 0 if summary.get("ok") else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/agent/skills/kermt-add-cmim-pretrain/skill-card.md b/skills/kermt-add-cmim-pretrain/skill-card.md similarity index 98% rename from agent/skills/kermt-add-cmim-pretrain/skill-card.md rename to skills/kermt-add-cmim-pretrain/skill-card.md index 0a691c4..6b7061d 100644 --- a/agent/skills/kermt-add-cmim-pretrain/skill-card.md +++ b/skills/kermt-add-cmim-pretrain/skill-card.md @@ -44,7 +44,7 @@ Mitigation: W&B tracking is optional and off unless the user supplies `WANDB_API ## Reference(s):
- [KERMT repository](https://github.com/NVIDIA-BioNeMo/KERMT)
-- `agent/scripts/run_pretrain_local.py` — extended usage examples
+- `scripts/run_pretrain_local.py` — extended usage examples
- Related skills: `kermt-setup`, `kermt-monitor`, `kermt-continue-pretrain`, `kermt-pretrain-scratch`
## Skill Output:
diff --git a/agent/skills/kermt-continue-pretrain/SKILL.md b/skills/kermt-continue-pretrain/SKILL.md similarity index 91% rename from agent/skills/kermt-continue-pretrain/SKILL.md rename to skills/kermt-continue-pretrain/SKILL.md index 7355076..32c17ef 100644 --- a/agent/skills/kermt-continue-pretrain/SKILL.md +++ b/skills/kermt-continue-pretrain/SKILL.md @@ -9,7 +9,7 @@ metadata: risk_tier: skill # Line/token budget: this file is targeted at ~250 lines / ~3000 tokens — # well within the 500-line / 5000-token cap. Long examples live in -# agent/scripts/run_pretrain_local.py's docstring. +# /skill/scripts/run_pretrain_local.py's docstring. --- # kermt-continue-pretrain @@ -18,6 +18,15 @@ Continue pretraining from a user-supplied KERMT checkpoint (grover_base / cmim / hybrid). The skill is the workflow orchestrator: it validates inputs, prepares the corpus, launches the runner, and returns a run directory. +## Skill and runtime paths + +Set `SKILL_DIR` to the directory containing this `SKILL.md`. Export +`KERMT_REPO` as the absolute path to the KERMT checkout used for model +execution. The bundled container helper mounts that checkout at +`/workspace` and this skill at `/skill` (read-only). Commands inside +the container use `/skill/scripts/`; defaults are bundled in `config/`. +See [Released models](references/released-models.md) for checkpoint bundle requirements. + ## Hardware requirements - **GPUs**: 1–N CUDA-capable NVIDIA GPUs. The runner auto-detects via @@ -76,7 +85,7 @@ Optional: - `--epochs N` / `--batch-size N` / `--init-lr F` / `--max-lr F` / `--final-lr F` / `--warmup-epochs F` / `--weight-decay F` / `--dropout F` / `--save-interval N` / `--seed N` — training-hyperparameter overrides. - Anything not given is filled from `agent/config/defaults_pretrain.json`. + Anything not given is filled from `config/defaults_pretrain.json`. - `--vocab-loss-weight F` (hybrid only) / `--latent-dim N` / `--contrastive-temperature F` (cmim and hybrid only) — loss / decoder overrides. @@ -154,7 +163,7 @@ host; the helper bind-mounts them at known container paths. 1. **Pre-flight: ensure container + system probe.** ``` - $KERMT_REPO/agent/scripts/kermt_container.sh check_system | python -c " + "$SKILL_DIR/scripts/kermt_container.sh" check_system | python -c " import json, sys; d = json.load(sys.stdin) if not d['ok']: print('System check failed:', d['gaps']); sys.exit(1) @@ -182,8 +191,8 @@ host; the helper bind-mounts them at known container paths. `--model-dir ` if given. An already-complete bundle is reused. - **Download** (foreground; ~282 MB on first fetch): ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run --model-dir -- \ - "python agent/scripts/fetch_released_model.py --out /model" + "$SKILL_DIR/scripts/kermt_container.sh" run --model-dir -- \ + "python /skill/scripts/fetch_released_model.py --out /model" ``` Parse the JSON; abort on `ok: false` (surface `errors`). On success set ` = /kermt_contrastive_v2.0.pt`. The bundle's three @@ -193,8 +202,8 @@ host; the helper bind-mounts them at known container paths. **Validate** the resolved (or user-provided) ckpt: ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run --ckpt -- \ - "python agent/scripts/check_checkpoint.py --mode continue_pretrain --ckpt /ckpt" + "$SKILL_DIR/scripts/kermt_container.sh" run --ckpt -- \ + "python /skill/scripts/check_checkpoint.py --mode continue_pretrain --ckpt /ckpt" ``` Parse the JSON. Abort on `ok: false`, showing the error verbatim. The error message redirects the user to `kermt-add-cmim-pretrain` for encoder-only @@ -202,8 +211,8 @@ host; the helper bind-mounts them at known container paths. 4. **Validate the data.** ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run --data -- \ - "python agent/scripts/check_data.py --mode pretrain --csv /data/" + "$SKILL_DIR/scripts/kermt_container.sh" run --data -- \ + "python /skill/scripts/check_data.py --mode pretrain --csv /data/" ``` Abort on `ok: false`. @@ -211,7 +220,7 @@ host; the helper bind-mounts them at known container paths. **Pass the ckpt's vocab through.** Look in the ckpt's parent directory for the conventional `pretrain_atom_vocab.{json,pkl}`, `pretrain_bond_vocab.{json,pkl}`, and `pretrain_smiles_vocab.pkl` files (the bundling convention for released - models; see `agent/README.md` "Released models" section). If all three are + models; see `references/released-models.md`). If all three are present, auto-pass via `--vocab-dir `. If only some are present, pass them via explicit flags (`--atom-vocab`, `--bond-vocab`, `--smiles-vocab`). If none are present, ask the user for `--vocab-dir` — or @@ -228,9 +237,9 @@ host; the helper bind-mounts them at known container paths. ``` VOCAB_DIR=$(dirname ) - $KERMT_REPO/agent/scripts/kermt_container.sh run \ + "$SKILL_DIR/scripts/kermt_container.sh" run \ --data --vocab-dir $VOCAB_DIR --run-dir $RUN_DIR -- \ - "python agent/scripts/prepare_data.py --mode pretrain \\ + "python /skill/scripts/prepare_data.py --mode pretrain \\ --csv /data/ --out /runs/data \\ --vocab-dir /vocab \\ [--val-csv /data/] [--val-frac 0.1] [--seed 0]" @@ -249,10 +258,10 @@ host; the helper bind-mounts them at known container paths. 7. **Launch the runner detached.** ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run_detached \\ + "$SKILL_DIR/scripts/kermt_container.sh" run_detached \\ --name kermt-continue-pretrain- \\ --ckpt --run-dir $RUN_DIR -- \\ - "python agent/scripts/run_pretrain_local.py \\ + "python /skill/scripts/run_pretrain_local.py \\ --ckpt /ckpt \\ --prepare-manifest /runs/data/prepare_data.json \\ --out /runs \\ diff --git a/skills/kermt-continue-pretrain/config/defaults_pretrain.json b/skills/kermt-continue-pretrain/config/defaults_pretrain.json new file mode 100644 index 0000000..dbb53e7 --- /dev/null +++ b/skills/kermt-continue-pretrain/config/defaults_pretrain.json @@ -0,0 +1,52 @@ +{ + "_about": "Default hyperparameters applied by kermt-continue-pretrain and kermt-add-cmim-pretrain. Values target a workstation-scale hybrid pretrain. The skill echoes the applied set back to the user on every invocation; override any value with the corresponding CLI flag.", + + "training": { + "_about": "Optimizer and training schedule. Apply to all pretrain workflows.", + "batch_size": 256, + "dropout": 0.1, + "epochs": 30, + "init_lr": 1e-5, + "max_lr": 1.5e-4, + "final_lr": 1e-5, + "warmup_epochs": 20, + "weight_decay": 1e-7, + "save_interval": 100, + "seed": 0, + "tensorboard": true, + "use_cuikmolmaker_featurization": true + }, + + "loss": { + "_about": "Loss-weighting knobs. contrastive_temperature applies only when the model has a contrast head (hybrid). vocab_loss_weight applies when the model has a vocab head (cmim or hybrid). The runner detects the model type from the checkpoint and ignores irrelevant entries.", + "contrastive_temperature": 0.1, + "vocab_loss_weight": 1.0 + }, + + "add_cmim_decoder": { + "_about": "Used by kermt-add-cmim-pretrain when constructing the new cMIM decoder + latent_dist on top of a loaded grover-base encoder, and by kermt-pretrain-scratch when the pretrain target is cmim or hybrid. Ignored by kermt-continue-pretrain (those dimensions come from the ckpt's saved_args). Values match the manuscript's hybrid pretrain configuration: latent_dim=512, 8-head, 3-layer decoder (cf. `_PRESET_LATENT_DIM` / `_PRESET_DECODER_FFN_HIDDEN_SIZE` in launch-KERMT-pretrain-slurm.sh, both presets).", + "latent_dim": 512, + "contrastive_temperature": 0.1, + "decoder_num_layers": 3, + "decoder_num_attention_heads": 8, + "decoder_ffn_hidden_size": 2048, + "decoder_dropout": 0.1, + "decoder_max_seq_len": 512, + "decoder_positional_encoding": "rope", + "decoder_gate_self_attn": false, + "decoder_gate_cross_attn": false + }, + + "arch": { + "_about": "Encoder architecture defaults — used ONLY by kermt-pretrain-scratch (fresh model from corpus, no starting ckpt). kermt-continue-pretrain and kermt-add-cmim-pretrain ignore this block and pull arch from the loaded checkpoint instead; the runner aborts if user-supplied arch flags mismatch the ckpt's saved_args.", + "hidden_size": 800, + "depth": 6, + "num_attn_head": 4, + "activation": "PReLU", + "backbone": "gtrans", + "embedding_output_type": "both", + "self_attention": false + }, + + "_about_gpu_selection": "GPU selection is auto-detected at runtime, not a default here. The pretrain runner uses torch.cuda.device_count() and dispatches single-GPU or DDP accordingly. Override with --gpus 0,2 if you want a specific subset." +} diff --git a/skills/kermt-continue-pretrain/config/released_model.json b/skills/kermt-continue-pretrain/config/released_model.json new file mode 100644 index 0000000..1e19fc1 --- /dev/null +++ b/skills/kermt-continue-pretrain/config/released_model.json @@ -0,0 +1,13 @@ +{ + "repo_id": "nvidia/NV-KERMT-70M-v2", + "revision": "7df5eb3179235fdea1e8124db73215da33d77dce", + "ckpt_name": "kermt_contrastive_v2.0.pt", + "vocab_files": [ + "pretrain_atom_vocab.json", + "pretrain_bond_vocab.json", + "pretrain_smiles_vocab.pkl" + ], + "model_type": "hybrid", + "license": "NVIDIA Open Model License", + "license_url": "https://huggingface.co/nvidia/NV-KERMT-70M-v2" +} diff --git a/agent/skills/kermt-continue-pretrain/evals/evals.json b/skills/kermt-continue-pretrain/evals/evals.json similarity index 100% rename from agent/skills/kermt-continue-pretrain/evals/evals.json rename to skills/kermt-continue-pretrain/evals/evals.json diff --git a/skills/kermt-continue-pretrain/references/released-models.md b/skills/kermt-continue-pretrain/references/released-models.md new file mode 100644 index 0000000..e748c7e --- /dev/null +++ b/skills/kermt-continue-pretrain/references/released-models.md @@ -0,0 +1,34 @@ +# Released KERMT models + +Each released KERMT checkpoint is distributed as a **directory bundle** +containing the ckpt itself plus its vocab files: + +``` +/ +├── last_checkpoint.pt +├── pretrain_atom_vocab.{json,pkl} # either extension; pkl in current releases +├── pretrain_bond_vocab.{json,pkl} # either extension; pkl in current releases +└── pretrain_smiles_vocab.pkl # only for cmim / hybrid ckpts (pickle-only) +``` + +If you're upgrading a grover_base ckpt to hybrid with +`kermt-add-cmim-pretrain`, the +upgrade step builds a fresh `pretrain_smiles_vocab.pkl` from your +pretrain corpus — released bundles only ship the smiles vocab for +already-cmim / already-hybrid ckpts. + +The vocab files are an inseparable part of the released model — the ckpt's +vocab head dimensions are fixed at training time and only match these specific +vocab files. `kermt-continue-pretrain` treats the released ckpt's vocab as +authoritative: new corpora are tokenized through it rather than producing a +new vocab that would mismatch the ckpt's heads. + +The skill auto-detects the three vocab files in the ckpt's parent directory +and passes them through `prepare_data.py --vocab-dir`. If the bundle is +incomplete (or the user has the ckpt alone), the skill asks for the +`--vocab-dir` path; if the user can't provide one, the skill refuses to +proceed and suggests `kermt-pretrain-scratch` instead. + +To train a model on a corpus the released vocab can't cover, use +`kermt-pretrain-scratch` — the new vocab is built from the corpus and the +model is initialized fresh (no warm start; days-scale to converge). diff --git a/skills/kermt-continue-pretrain/scripts/_utils.py b/skills/kermt-continue-pretrain/scripts/_utils.py new file mode 100644 index 0000000..5bde460 --- /dev/null +++ b/skills/kermt-continue-pretrain/scripts/_utils.py @@ -0,0 +1,272 @@ +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Shared utilities for the agent scripts. + +Kept intentionally small — only logic that appears (or would otherwise be +duplicated) in two or more `scripts/*.py` modules. Each script +maintains its own primary CLI + main flow. +""" +from __future__ import annotations + +import argparse +import json +import os +import shlex +import subprocess +import sys +from pathlib import Path +from typing import Any + + +# Conventional pretrain vocab filename stems. Used by prepare_data.py + +# upgrade_to_hybrid.py + the README "Released models" bundling docs + +# the test helpers. Centralized here so a future rename only touches one +# spot. +PRETRAIN_VOCAB_STEMS = { + "atom": "pretrain_atom_vocab", + "bond": "pretrain_bond_vocab", + "smiles": "pretrain_smiles_vocab", +} + + +def resolve_kermt_repo() -> Path: + """Find the runtime checkout independently of the installed skill location. + + An explicit KERMT_REPO takes precedence. In a repository checkout, walking + up from this helper or the working directory also supports local use. + """ + explicit = os.environ.get("KERMT_REPO") + if explicit: + candidates = [Path(explicit).expanduser().resolve()] + else: + candidates = [] + for start in (Path(__file__).resolve().parent, Path.cwd()): + candidates.extend((start, *start.parents)) + for candidate in candidates: + if (candidate / "main.py").is_file() and (candidate / "kermt").is_dir(): + return candidate + raise FileNotFoundError( + "KERMT checkout not found. Set KERMT_REPO to the checkout containing " + "main.py and kermt/; the installed skill directory is separate." + ) + + +def load_json(path: Path, *, name: str) -> dict[str, Any]: + """Load a JSON file with consistent error messages. + + `name` is a human-readable label for the document (e.g. "prepare_data.json") + so the error tells the user which schema we expected at that path. + """ + if not path.is_file(): + raise FileNotFoundError(f"{name} not found at {path}") + try: + return json.loads(path.read_text()) + except json.JSONDecodeError as exc: + raise ValueError(f"{name} at {path} is not valid JSON: {exc}") from exc + + +def count_vocab_entries(vocab_path: Path) -> int: + """Return the number of entries in a KERMT vocab file. + + Handles three layouts: + - JSON with `{stoi: {token: idx}, ...}` (MolVocab.save_vocab default) + - JSON as a raw `{token: idx}` dict (legacy / hand-edited) + - Pickle of a MolVocab / SMILESVocab object (the smiles vocab is always + pickled because its compiled-regex tokenizer state isn't + JSON-serializable). Falls through to raw `pickle.load` if the + MolVocab / SMILESVocab loader can't import or fails to recognize + the contents (e.g. test fixtures with plain dicts). + """ + if vocab_path.suffix == ".json": + data = json.loads(vocab_path.read_text()) + if isinstance(data, dict) and "stoi" in data: + return len(data["stoi"]) + if isinstance(data, dict): + return len(data) + raise ValueError(f"unsupported JSON vocab shape at {vocab_path}: {type(data).__name__}") + + # .pkl: try MolVocab / SMILESVocab first, then fall back to raw pickle. + try: + from kermt.data.torchvocab import MolVocab, SMILESVocab # type: ignore + for loader in (MolVocab.load_vocab, SMILESVocab.load_vocab): + try: + v = loader(str(vocab_path)) + return len(v) + except Exception: + continue + except ImportError: + pass + + import pickle + with vocab_path.open("rb") as f: + data = pickle.load(f) + if hasattr(data, "stoi"): + return len(data.stoi) + if hasattr(data, "__len__"): + return len(data) + raise ValueError(f"could not count entries in {vocab_path}") + + +def validate_vocab_file(vocab_path: Path, *, kind: str) -> None: + """Verify a user-provided vocab file is loadable BEFORE copying it into a + run directory. Raises ValueError on failure with a clear, user-facing message. + + `kind` is one of {"atom", "bond", "smiles"} — used only in the error message + so the user knows which file is wrong. + """ + if not vocab_path.is_file(): + raise FileNotFoundError(f"{kind} vocab file not found: {vocab_path}") + try: + n = count_vocab_entries(vocab_path) + except Exception as exc: # noqa: BLE001 + raise ValueError( + f"{kind} vocab file {vocab_path} is not loadable as a KERMT vocab " + f"({type(exc).__name__}: {exc}). Expected a MolVocab JSON or pickle " + f"(or a SMILESVocab pickle for the smiles vocab)." + ) from exc + if n <= 0: + raise ValueError(f"{kind} vocab file {vocab_path} contains zero entries") + + +# --------------------------------------------------------------------------- +# Runner-shared helpers (run.json manifest fields) +# --------------------------------------------------------------------------- + +def git_commit_with_env_override(repo: Path) -> tuple[str, bool]: + """Returns (commit_sha, dirty_tree). Honors `KERMT_REPO_COMMIT` / + `KERMT_REPO_DIRTY` env vars first — set by `scripts/kermt_container.sh` + from the host before launching docker (necessary because `git -C /workspace` + inside the container fails due to bind-mount ownership). Falls back to the + in-container git probe when the env vars aren't set.""" + env_commit = os.environ.get("KERMT_REPO_COMMIT") + if env_commit: + env_dirty = os.environ.get("KERMT_REPO_DIRTY", "false").strip().lower() == "true" + return env_commit, env_dirty + try: + sha = subprocess.run( + ["git", "-C", str(repo), "rev-parse", "HEAD"], + capture_output=True, text=True, check=True, + ).stdout.strip() + diff = subprocess.run( + ["git", "-C", str(repo), "status", "--porcelain"], + capture_output=True, text=True, check=True, + ) + return sha, bool(diff.stdout.strip()) + except Exception: + return "unknown", False + + +def docker_image_digest(tag: str) -> str | None: + """Return the docker image's content-addressable Id (sha256:…) for the given + tag, or None if docker isn't available / the image isn't local.""" + try: + r = subprocess.run( + ["docker", "image", "inspect", tag, "--format", "{{.Id}}"], + capture_output=True, text=True, + ) + if r.returncode == 0: + return r.stdout.strip() + except FileNotFoundError: + pass + return None + + +def format_cmd_replay(argv: list[str], *, env: dict[str, str] | None = None) -> str: + """Render a copy-pasteable env-prefix + command for the cmd_replay manifest + field. `env` is the set of environment variables to prefix (typically + {CUDA_VISIBLE_DEVICES, WORLD_SIZE}).""" + env = env or {} + env_prefix = [f"{k}={shlex.quote(str(v))}" for k, v in env.items()] + quoted = " ".join(shlex.quote(a) for a in argv) + return " ".join(env_prefix + [quoted]) + + +def resolve_single_gpu(override: str | None, *, workflow: str) -> int: + """Returns a single GPU id (int). The finetune/inference/embed workflows are + single-GPU only; `--gpus '0,1'` or multi-id CUDA_VISIBLE_DEVICES is rejected + with a workflow-specific error. (The pretrain runner has its own multi-GPU + `_detect_gpus` helper — see run_pretrain_local.py.)""" + if override is None: + env_visible = os.environ.get("CUDA_VISIBLE_DEVICES", "").strip() + if env_visible: + ids = [g for g in env_visible.split(",") if g] + if len(ids) > 1: + raise ValueError( + f"CUDA_VISIBLE_DEVICES='{env_visible}' selects multiple GPUs but " + f"the {workflow} workflow is single-GPU only. Restrict to one id." + ) + return int(ids[0]) + return 0 + parts = [p.strip() for p in override.split(",") if p.strip()] + if len(parts) != 1: + raise ValueError( + f"--gpus '{override}' selects {len(parts)} GPUs; the {workflow} workflow is single-GPU only." + ) + return int(parts[0]) + + +def assert_prepare_manifest_basics(manifest: dict[str, Any], expected_mode: str) -> None: + """Standard pre-check for a prepare_data.json before a runner consumes it: + verify `mode` matches and `ok` is True. Raises ValueError with a consistent + error message on either mismatch. + + Each runner is responsible for its own required-outputs check after this + (those vary per-mode — e.g. pretrain wants train_dir/val_dir/atom_vocab/ + bond_vocab; finetune has the split-method branch; inference/embed want + clean_csv).""" + if manifest.get("mode") != expected_mode: + raise ValueError( + f"prepare_data manifest is mode='{manifest.get('mode')}', expected '{expected_mode}'. " + f"Run `prepare_data.py --mode {expected_mode}` to produce a valid manifest." + ) + if not manifest.get("ok"): + raise ValueError( + f"prepare_data manifest reports ok=False: {manifest.get('errors')}" + ) + + +def merge_default_into_applied( + applied: dict[str, dict[str, Any]], + args: argparse.Namespace, + name: str, + defaults_group: dict[str, Any], +) -> None: + """Standard CLI-override / default-config merge for one hyperparameter. + + Mutates `applied` in place: + - If the user passed `--` on the CLI (so `getattr(args, name)` is + not None), records `{"value": cli_val, "source": "user"}`. + - Else if `name` is present in `defaults_group`, records + `{"value": defaults_group[name], "source": "default-config"}`. + - Else `applied[name]` is left absent — the runner's argv-builder skips + the flag, and the downstream argparse default takes effect. + + `name` is the snake_case argparse dest (same form used as the dict key); + argparse automatically converts CLI `--` to that dest, + so `getattr(args, name, None)` is the correct CLI lookup.""" + cli_val = getattr(args, name, None) + if cli_val is not None: + applied[name] = {"value": cli_val, "source": "user"} + elif name in defaults_group: + applied[name] = {"value": defaults_group[name], "source": "default-config"} + + +def run_checkpoint_validator(ckpt: Path, *, mode: str, script_path: Path) -> dict[str, Any]: + """Invoke `check_checkpoint.py --mode --ckpt ` as a subprocess + and return the parsed JSON. Raises RuntimeError on non-JSON output (e.g. the + validator crashed before printing). `script_path` is the absolute path to + `scripts/check_checkpoint.py` — passed in so this helper has no + dependency on the caller's layout.""" + r = subprocess.run( + [sys.executable, str(script_path), "--mode", mode, "--ckpt", str(ckpt)], + capture_output=True, text=True, + ) + try: + return json.loads(r.stdout) + except json.JSONDecodeError as exc: + raise RuntimeError( + f"check_checkpoint.py emitted non-JSON output (exit {r.returncode}). " + f"stdout (first 200 chars): {r.stdout[:200]}\n" + f"stderr (first 200 chars): {r.stderr[:200]}" + ) from exc diff --git a/skills/kermt-continue-pretrain/scripts/check_checkpoint.py b/skills/kermt-continue-pretrain/scripts/check_checkpoint.py new file mode 100644 index 0000000..fd488e2 --- /dev/null +++ b/skills/kermt-continue-pretrain/scripts/check_checkpoint.py @@ -0,0 +1,480 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Validate a KERMT checkpoint for a given agent workflow. + +Mode-dispatched. Emits a single JSON object to stdout that the calling skill +parses to decide whether to proceed (`ok: true`) or surface a structured error +back to the user (`ok: false` with `errors[]`). The JSON shape is stable +across modes; only the contract for what counts as `ok` differs per mode. + +Modes +----- +continue_pretrain Continuing pretraining from an existing pretrain ckpt. + Requires encoder + at least one pretrain head + (vocab_head for grover_base / cmim, or contrast_head for + cmim / hybrid). Rejects encoder-only or finetuned ckpts. + +upgrade_to_hybrid Adding a cMIM decoder onto a grover_base ckpt to convert + it to a hybrid pretrain. Requires encoder; rejects ckpts + that already carry a contrast_head or task_ffn (would be + workflow 4 instead). + +finetune_init Starting a finetune from a pretrained ckpt. Requires + encoder. Pretrain heads (vocab / contrast) are tolerated + but unused. Already-finetuned ckpts (task FFN heads + present) are REJECTED — finetune-on-finetune via the + agent skill isn't supported because saved-task + identity can't be machine-verified against the new + training data. + +inference Running predictions with a previously-finetuned ckpt. + Requires encoder + task_ffn. Reports task_output_dims + so the runner can compare against the user's task spec. + +embed Extracting embeddings. Requires encoder only. Anything + additional in the ckpt is ignored. + +Output (stdout) +--------------- +{ + "ok": true | false, + "model_type": "grover_base" | "cmim" | "hybrid" | "finetuned" | "unknown", + "has_encoder": bool, + "has_vocab_head": bool, + "has_contrast_head": bool, + "has_task_ffn": bool, + "task_output_dims": [int, ...], // empty unless has_task_ffn + "arch": { // ckpt-derived; runner uses these, ignores defaults_*.json arch + "hidden_size": int | null, + "depth": int | null, + "num_attn_head": int | null, + "latent_dim": int | null, + "activation": str | null, + "backbone": str | null, + "embedding_output_type": str | null, + "self_attention": bool | null + }, + "saved_args": { ... } | null, // raw args dict if present, else null + "errors": [str, ...], // mode-contract violations / load failures + "warnings": [str, ...] // non-fatal observations (e.g. arch fallback) +} + +Exit code: 0 on `ok: true`, 1 on `ok: false`. Loader exceptions are caught and +surfaced into `errors[]` with `ok: false` (still exit 1), never raised. + +CLI +--- + check_checkpoint.py --mode --ckpt +""" +from __future__ import annotations + +import argparse +import json +import sys +import traceback +from argparse import Namespace +from typing import Any + +import torch + + +# --------------------------------------------------------------------------- +# State-dict key prefix conventions (kermt/model/models.py). +# --------------------------------------------------------------------------- + +# Encoder weights appear under one of these prefixes depending on the ckpt's +# era and task class: +# - `grover.*` : legacy grover_base ckpts (predate the cMIM rename) +# - `kermt.*` : current grover_base / hybrid / finetune ckpts +# - `latent_dist.kermt.*`: cmim ckpts (encoder lives only inside latent_dist) +ENCODER_PREFIXES = ("kermt.", "grover.", "latent_dist.kermt.") +VOCAB_HEAD_PREFIX = "vocab_module." +CONTRAST_DECODER_PREFIX = "decoder." # SMILES transformer decoder, cmim/hybrid only +LATENT_DIST_PREFIX = "latent_dist." # cmim/hybrid; encoder may share via latent_dist.kermt.* +TASK_FFN_PREFIXES = ( + "mol_atom_from_atom_ffn.", + "mol_atom_from_bond_ffn.", +) +TASK_FFN_TASK_SPECIFIC_PREFIXES = ( + "mol_atom_from_atom_ffn_task_specific.", + "mol_atom_from_bond_ffn_task_specific.", +) + + +ARCH_KEYS = ( + "hidden_size", + "depth", + "num_attn_head", + "latent_dim", + "activation", + "backbone", + "embedding_output_type", + "self_attention", +) + + +def _strip_ddp_prefix(state_dict: dict[str, Any]) -> dict[str, Any]: + """Strip `module.` prefix from every key if the dict is DDP-wrapped.""" + if state_dict and all(k.startswith("module.") for k in state_dict): + return {k[len("module."):]: v for k, v in state_dict.items()} + return state_dict + + +def _classify_model(state_dict: dict[str, Any]) -> dict[str, Any]: + keys = list(state_dict.keys()) + has_encoder = any(k.startswith(ENCODER_PREFIXES) for k in keys) + has_vocab_head = any(k.startswith(VOCAB_HEAD_PREFIX) for k in keys) + has_contrast_head = any(k.startswith(CONTRAST_DECODER_PREFIX) for k in keys) + has_task_ffn = any(k.startswith(TASK_FFN_PREFIXES) for k in keys) + + if has_encoder and has_task_ffn: + model_type = "finetuned" + elif has_encoder and has_contrast_head and has_vocab_head: + model_type = "hybrid" + elif has_encoder and has_contrast_head and not has_vocab_head: + model_type = "cmim" + elif has_encoder and not has_contrast_head: + # Includes: + # - modern repo-trained Grover base (kermt.* + vocab_module.*) + # - legacy original-Grover base (grover.encoders.* with no heads saved) + # - any encoder-stripped ckpt extracted from a larger model + # The `has_vocab_head` flag discriminates the sub-cases for skills that + # need it. The continue_pretrain mode contract relies on this — a + # grover_base with vocab heads can continue, an encoder-only one cannot. + model_type = "grover_base" + else: + model_type = "unknown" + + return { + "model_type": model_type, + "has_encoder": has_encoder, + "has_vocab_head": has_vocab_head, + "has_contrast_head": has_contrast_head, + "has_task_ffn": has_task_ffn, + } + + +def _vocab_sizes(state_dict: dict[str, Any]) -> dict[str, Any]: + """Extract vocab head sizes from state-dict weight shapes. + + The pretrain heads have the following layout per kermt/model/models.py: + - Atom vocab predictors: vocab_module.av_task_atom.* + vocab_module.av_task_bond.* + (two readout streams sharing the same vocab_size). Output dim of each + final-Linear is the atom vocab size. + - Bond vocab predictors: vocab_module.bv_task_atom.* + vocab_module.bv_task_bond.* + Output dim is the bond vocab size. + - SMILES vocab decoder: decoder.output_projection.weight (cmim / hybrid only). + Output dim is the smiles vocab size. + + Returns {atom: int|None, bond: int|None, smiles: int|None}. Each is None + when the corresponding head isn't present in the ckpt (e.g. legacy + encoder-only grover_base has none; cmim has smiles but not atom/bond). + """ + sizes: dict[str, Any] = {"atom": None, "bond": None, "smiles": None} + + def _head_out_dim(prefix: str) -> int | None: + # Pick the highest-numbered 2-D Linear weight under `prefix.*` — that's + # the final output layer. + candidates = [ + k for k in state_dict + if k.startswith(prefix) and k.endswith(".weight") + and hasattr(state_dict[k], "ndim") and state_dict[k].ndim == 2 + ] + if not candidates: + return None + def _layer_index(k: str) -> int: + # ".weight" -> "..weight"; pick the rightmost numeric component. + parts = k.split(".") + for tok in reversed(parts[:-1]): + if tok.isdigit(): + return int(tok) + return -1 + final = max(candidates, key=_layer_index) + return int(state_dict[final].shape[0]) + + sizes["atom"] = _head_out_dim("vocab_module.av_task_atom.") + sizes["bond"] = _head_out_dim("vocab_module.bv_task_atom.") + sizes["smiles"] = _head_out_dim("decoder.output_projection.") + # If the decoder's output_projection isn't a Linear (e.g. some saves wrap + # it differently), fall back to a search over decoder.* heads. + if sizes["smiles"] is None: + sizes["smiles"] = _head_out_dim("decoder.token_embedding.") + return sizes + + +def _task_output_dims(state_dict: dict[str, Any]) -> list[int]: + """Return one entry per (logical task × readout) head's final-Linear out-dim. + + Two layouts: + - **MTL** (`mol_atom_from_atom_ffn_task_specific..*`): one entry per + task-specific head's final-Linear out-dim. Typically `[1, 1, ..., 1]` + for regression with N tasks across 2 readouts. + - **Non-MTL** (`mol_atom_from_atom_ffn.*` only): one entry per shared FFN's + final-Linear out-dim. Typically `[num_tasks, num_tasks]` (one per readout). + + When both layouts coexist in the same ckpt (MTL configuration: shared FFN + feeds task-specific heads), only the task-specific dims are reported — the + shared FFN there is an intermediate layer, not the model output. + """ + has_task_specific = any(k.startswith(TASK_FFN_TASK_SPECIFIC_PREFIXES) for k in state_dict) + + heads: dict[str, list[str]] = {} + for k in state_dict: + if k.startswith(TASK_FFN_TASK_SPECIFIC_PREFIXES): + parts = k.split(".") + root = ".".join(parts[:2]) # e.g. "mol_atom_from_atom_ffn_task_specific.0" + heads.setdefault(root, []).append(k) + elif k.startswith(TASK_FFN_PREFIXES) and not k.startswith(TASK_FFN_TASK_SPECIFIC_PREFIXES): + if has_task_specific: + continue # shared FFN is intermediate when task-specific heads exist + root = k.split(".")[0] # e.g. "mol_atom_from_atom_ffn" + heads.setdefault(root, []).append(k) + + dims: list[int] = [] + for root in sorted(heads): + weight_keys = sorted( + (k for k in heads[root] if k.endswith(".weight") + and hasattr(state_dict[k], "ndim") and state_dict[k].ndim == 2), + key=lambda k: int(k.split(".")[-2]) if k.split(".")[-2].isdigit() else -1, + ) + if weight_keys: + dims.append(int(state_dict[weight_keys[-1]].shape[0])) + return dims + + +def _arch_from_args(args_obj: Any) -> dict[str, Any]: + """Pull arch params from the saved args Namespace / dict, leaving missing keys as None.""" + arch: dict[str, Any] = {k: None for k in ARCH_KEYS} + if args_obj is None: + return arch + # args_obj is typically argparse.Namespace; tolerate dict form too. + args_dict = vars(args_obj) if isinstance(args_obj, Namespace) else dict(args_obj) if isinstance(args_obj, dict) else {} + for k in ARCH_KEYS: + if k in args_dict: + arch[k] = args_dict[k] + return arch + + +def _arch_from_shapes(state_dict: dict[str, Any], arch: dict[str, Any]) -> tuple[dict[str, Any], list[str]]: + """Fill in still-missing arch params by introspecting state-dict tensor shapes. + + Only fills entries that are currently None — does not override anything pulled + from saved_args. Returns the updated arch + a list of warnings for any key that + could not be inferred. + """ + warnings: list[str] = [] + + if arch["hidden_size"] is None: + # First 2-D linear weight under any encoder prefix. + candidates = [ + k for k in state_dict + if k.startswith(ENCODER_PREFIXES) + and k.endswith(".weight") + and hasattr(state_dict[k], "ndim") + and state_dict[k].ndim == 2 + ] + if candidates: + arch["hidden_size"] = int(state_dict[candidates[0]].shape[0]) + else: + warnings.append("hidden_size could not be inferred from state_dict shapes") + + if arch["latent_dim"] is None: + # Look for a Linear inside latent_dist that's not the shared encoder. + candidates = [ + k for k in state_dict + if k.startswith(LATENT_DIST_PREFIX) + and not k.startswith("latent_dist.kermt.") + and k.endswith(".weight") + and state_dict[k].ndim == 2 + ] + if candidates: + arch["latent_dim"] = int(state_dict[candidates[0]].shape[0]) + # Absent latent_dist entirely is a fact about the model, not a problem — no warning. + + # depth, num_attn_head, activation, backbone, embedding_output_type, self_attention + # are not robustly inferable from shapes alone; report a warning for each that's + # still None so the caller can prompt the user or refuse to proceed. + for k in ("depth", "num_attn_head", "activation", "backbone", "embedding_output_type", "self_attention"): + if arch[k] is None: + warnings.append(f"{k} not present in saved_args and cannot be inferred from state_dict shapes") + + return arch, warnings + + +def _apply_mode_contract(mode: str, classification: dict[str, Any]) -> list[str]: + """Return a list of error messages if `classification` violates the mode contract.""" + errors: list[str] = [] + mt = classification["model_type"] + has_enc = classification["has_encoder"] + has_vocab = classification["has_vocab_head"] + has_contrast = classification["has_contrast_head"] + has_ffn = classification["has_task_ffn"] + + if not has_enc: + errors.append("checkpoint has no encoder weights — cannot use it for any KERMT workflow") + return errors + + if mode == "continue_pretrain": + if not (has_vocab or has_contrast): + errors.append( + f"continue_pretrain requires the ckpt to still carry pretrain heads (vocab " + f"and/or contrast), but this ckpt has neither (model_type='{mt}', " + f"has_vocab_head=False, has_contrast_head=False). Either provide a ckpt with " + f"its pretrain heads attached, or convert this encoder-only ckpt to a hybrid " + f"via mode 'upgrade_to_hybrid'." + ) + if has_ffn: + errors.append( + "continue_pretrain expects a pretrain ckpt; this ckpt has task FFN heads " + "(it has been finetuned). Use a pretrain checkpoint — finetune+continue is " + "not a supported workflow." + ) + elif mode == "upgrade_to_hybrid": + if has_contrast: + errors.append( + f"upgrade_to_hybrid converts grover_base -> hybrid by adding a cMIM decoder. " + f"This ckpt already has a contrast head (classified as '{mt}'). " + f"To continue pretraining it, use mode 'continue_pretrain'." + ) + if has_ffn: + errors.append("upgrade_to_hybrid does not support finetuned checkpoints.") + elif mode == "finetune_init": + # Requires an encoder. Pretrain heads (vocab / contrast) are unused + # at finetune time but harmless. Task FFN heads (i.e. an already- + # finetuned ckpt) are NOT accepted — finetune-on-finetune isn't + # supported by the kermt-finetune skill because the saved-task + # identity can't be machine-verified against the new training data + # (dimension match doesn't prove target identity, dataset identity, + # or absence of train/test contamination). + if has_ffn: + errors.append( + f"finetune_init requires a pretrain ckpt (grover_base / cmim / hybrid); " + f"this ckpt is classified as '{mt}' with task FFN heads attached. " + f"To resume a finetune on the SAME dataset, call " + f"`python main.py finetune --checkpoint_path ...` directly — the " + f"kermt-finetune skill doesn't support resume." + ) + elif mode == "inference": + if not has_ffn: + errors.append( + "inference requires a finetuned ckpt with task FFN heads. " + f"This ckpt is classified as '{mt}' with no task heads. " + "Run finetune (mode 'finetune_init') first." + ) + elif mode == "embed": + # Encoder is sufficient. + pass + else: + errors.append(f"unknown mode '{mode}'") + + return errors + + +def validate(mode: str, ckpt_path: str) -> dict[str, Any]: + result: dict[str, Any] = { + "ok": False, + "model_type": "unknown", + "has_encoder": False, + "has_vocab_head": False, + "has_contrast_head": False, + "has_task_ffn": False, + "task_output_dims": [], + "vocab_sizes": {"atom": None, "bond": None, "smiles": None}, + "arch": {k: None for k in ARCH_KEYS}, + "saved_args": None, + "errors": [], + "warnings": [], + } + + # 1. Load the checkpoint. + try: + ckpt = torch.load(ckpt_path, map_location="cpu", weights_only=False) + except FileNotFoundError: + result["errors"].append(f"checkpoint not found: {ckpt_path}") + return result + except Exception as exc: # noqa: BLE001 + result["errors"].append(f"failed to load checkpoint {ckpt_path}: {type(exc).__name__}: {exc}") + return result + + if not isinstance(ckpt, dict) or "state_dict" not in ckpt: + result["errors"].append( + "checkpoint is not in the expected save_model_for_restart format " + "(expected a dict with a 'state_dict' key)." + ) + return result + + state_dict = _strip_ddp_prefix(ckpt["state_dict"]) + args_obj = ckpt.get("args") + + # 2. Classify and check mode contract. + classification = _classify_model(state_dict) + result.update(classification) + + contract_errors = _apply_mode_contract(mode, classification) + result["errors"].extend(contract_errors) + + # 3. Task output dims (for inference / informational). + if classification["has_task_ffn"]: + result["task_output_dims"] = _task_output_dims(state_dict) + + # 3b. Vocab head sizes (for continue-pretrain vocab-size verification). + result["vocab_sizes"] = _vocab_sizes(state_dict) + + # 4. Arch derivation: args first, shape introspection for what's still missing. + arch = _arch_from_args(args_obj) + arch, shape_warnings = _arch_from_shapes(state_dict, arch) + result["arch"] = arch + result["warnings"].extend(shape_warnings) + + # 5. Saved args as serializable dict (best-effort). + if args_obj is not None: + try: + result["saved_args"] = vars(args_obj) if isinstance(args_obj, Namespace) else dict(args_obj) + # Drop non-JSON-serializable values; agent skill only needs human-readable scalars. + result["saved_args"] = { + k: v for k, v in result["saved_args"].items() + if isinstance(v, (str, int, float, bool, type(None), list, dict)) + } + except Exception as exc: # noqa: BLE001 + result["warnings"].append(f"could not serialize saved_args: {type(exc).__name__}: {exc}") + + result["ok"] = not result["errors"] + return result + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Validate a KERMT checkpoint for a given workflow.") + parser.add_argument("--mode", required=True, + choices=["continue_pretrain", "upgrade_to_hybrid", "finetune_init", "inference", "embed"]) + parser.add_argument("--ckpt", required=True, help="Path to the .pt checkpoint") + args = parser.parse_args(argv) + + try: + result = validate(args.mode, args.ckpt) + except Exception as exc: # noqa: BLE001 + # Last-resort safety net: keep stdout JSON-clean, dump trace to stderr. + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({ + "ok": False, + "model_type": "unknown", + "errors": [f"unhandled exception in validator: {type(exc).__name__}: {exc}"], + "warnings": [], + "arch": {k: None for k in ARCH_KEYS}, + "has_encoder": False, + "has_vocab_head": False, + "has_contrast_head": False, + "has_task_ffn": False, + "task_output_dims": [], + "vocab_sizes": {"atom": None, "bond": None, "smiles": None}, + "saved_args": None, + }, indent=2)) + return 1 + + print(json.dumps(result, indent=2)) + return 0 if result["ok"] else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-continue-pretrain/scripts/check_data.py b/skills/kermt-continue-pretrain/scripts/check_data.py new file mode 100644 index 0000000..b8f9b15 --- /dev/null +++ b/skills/kermt-continue-pretrain/scripts/check_data.py @@ -0,0 +1,310 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Validate a CSV input for a given KERMT agent workflow. + +Mode-dispatched. Emits a single JSON object to stdout that the calling skill +parses to decide whether to proceed (`ok: true`) or surface a structured error +back to the user (`ok: false` with `errors[]`). The JSON shape is stable +across modes; only the contract for what counts as `ok` differs per mode. + +Modes +----- +pretrain Pretrain corpus CSV. Requires a `smiles` column. Other columns + are ignored. Label columns are not required (and not expected). + +finetune Labeled CSV for a downstream task. Requires `smiles` plus + >=1 numeric target column. Target columns are specified via + `--targets ...`. If `--targets` is omitted, the + validator auto-detects numeric non-smiles columns and reports + them; the skill will then prompt the user to confirm or refine. + +inference CSV to run predictions on. Requires `smiles`. Target columns are + not required (and not expected — predictions are written out). + +embed CSV to extract embeddings from. Requires `smiles` only. + +SMILES validation +----------------- +By default the validator samples up to 20 SMILES (first 10 + last 10) and +checks each one parses with RDKit. Pass `--strict-rdkit` to parse every +SMILES (slow on large corpora). A SMILES is considered "invalid" if RDKit +returns `None` from `MolFromSmiles(smi, sanitize=True)` — empty / null +rows are counted separately. + +Duplicate-SMILES detection is always full (cheap). + +Output (stdout) +--------------- +{ + "ok": true | false, + "mode": str, + "csv_path": str, + "num_rows": int, + "num_columns": int, + "columns": [str, ...], + "has_smiles_column": bool, + "smiles_column_name": str | null, // actual header used (may differ in case) + "num_blank_smiles": int, + "num_invalid_smiles": int, // among the parsed sample + "smiles_check_method": "sampled" | "full", + "smiles_check_count": int, + "num_duplicate_smiles": int, + "target_columns": [str, ...], // populated only for finetune mode + "num_missing_per_target": { col: int, ... }, + "auto_detected_targets": [str, ...], // when --targets is omitted in finetune mode + "errors": [str, ...], + "warnings": [str, ...] +} + +Exit code: 0 on `ok: true`, 1 on `ok: false`. Loader exceptions are caught +and surfaced into `errors[]` with `ok: false` (still exit 1). + +CLI +--- + check_data.py --mode --csv + [--targets ...] # finetune only + [--strict-rdkit] # full SMILES parse +""" +from __future__ import annotations + +import argparse +import json +import sys +import traceback +from pathlib import Path +from typing import Any + +import pandas as pd + + +CANONICAL_SMILES_COLUMN = "smiles" +SMILES_SAMPLE_PER_END = 10 # how many SMILES from head + how many from tail to sample + + +def _find_smiles_column(columns: list[str]) -> str | None: + """Return the actual column header matching 'smiles' case-insensitively, or None.""" + for c in columns: + if c.lower() == CANONICAL_SMILES_COLUMN: + return c + return None + + +def _parse_smiles_sample(smiles_values: list[str], full: bool) -> tuple[int, int, str]: + """Run RDKit MolFromSmiles on a sample or all of the SMILES. Returns + (num_parsed, num_invalid, method).""" + # Import here so the script can still surface a clean JSON error if RDKit + # is unavailable in the host env. + try: + from rdkit import Chem + from rdkit import RDLogger + RDLogger.DisableLog("rdApp.*") # suppress per-mol parse warnings + except ImportError as exc: + raise RuntimeError( + f"RDKit is not importable in this environment: {exc}. " + "Run check_data.py inside the kermt container." + ) from exc + + if full or len(smiles_values) <= 2 * SMILES_SAMPLE_PER_END: + sample = smiles_values + method = "full" + else: + sample = smiles_values[:SMILES_SAMPLE_PER_END] + smiles_values[-SMILES_SAMPLE_PER_END:] + method = "sampled" + + invalid = 0 + parsed = 0 + for smi in sample: + if not smi: # already counted as blank elsewhere + continue + parsed += 1 + mol = Chem.MolFromSmiles(smi, sanitize=True) + if mol is None: + invalid += 1 + return parsed, invalid, method + + +def _autodetect_target_columns(df: pd.DataFrame, smiles_col: str) -> list[str]: + """Pick columns that look like numeric targets. A column qualifies if it + is (a) not the smiles column and (b) >=80% of non-null values convert to float. + Heuristic only — returned for the skill to prompt the user to confirm.""" + candidates: list[str] = [] + for col in df.columns: + if col == smiles_col: + continue + ser = df[col].dropna() + if len(ser) == 0: + continue + try: + converted = pd.to_numeric(ser, errors="coerce") + except (TypeError, ValueError): + continue + if converted.notna().sum() / max(len(ser), 1) >= 0.8: + candidates.append(col) + return candidates + + +def validate(mode: str, csv_path: str, targets: list[str] | None, strict_rdkit: bool) -> dict[str, Any]: + result: dict[str, Any] = { + "ok": False, + "mode": mode, + "csv_path": csv_path, + "num_rows": 0, + "num_columns": 0, + "columns": [], + "has_smiles_column": False, + "smiles_column_name": None, + "num_blank_smiles": 0, + "num_invalid_smiles": 0, + "smiles_check_method": "sampled", + "smiles_check_count": 0, + "num_duplicate_smiles": 0, + "target_columns": [], + "num_missing_per_target": {}, + "auto_detected_targets": [], + "errors": [], + "warnings": [], + } + + # 1. Read the CSV. + path = Path(csv_path) + if not path.is_file(): + result["errors"].append(f"CSV not found: {csv_path}") + return result + try: + df = pd.read_csv(path) + except pd.errors.EmptyDataError: + result["errors"].append(f"CSV is empty (no header): {csv_path}") + return result + except Exception as exc: # noqa: BLE001 + result["errors"].append(f"failed to read CSV {csv_path}: {type(exc).__name__}: {exc}") + return result + + result["num_rows"] = int(len(df)) + result["num_columns"] = int(len(df.columns)) + result["columns"] = [str(c) for c in df.columns] + + # 2. Locate the SMILES column. + smiles_col = _find_smiles_column(result["columns"]) + if smiles_col is None: + result["errors"].append( + f"no column named 'smiles' (case-insensitive) found in CSV. " + f"Available columns: {result['columns']}" + ) + return result + result["has_smiles_column"] = True + result["smiles_column_name"] = smiles_col + if smiles_col != CANONICAL_SMILES_COLUMN: + result["warnings"].append( + f"SMILES column is named '{smiles_col}' but downstream code expects '{CANONICAL_SMILES_COLUMN}' " + f"(lowercase). Rename the column to '{CANONICAL_SMILES_COLUMN}' before running the workflow." + ) + + # 3. Blank-SMILES count + duplicate count + RDKit parse check. + smi_series = df[smiles_col].astype(str).fillna("").str.strip() + blank_mask = smi_series.eq("") | smi_series.str.lower().eq("nan") + result["num_blank_smiles"] = int(blank_mask.sum()) + + nonblank = smi_series[~blank_mask] + result["num_duplicate_smiles"] = int(len(nonblank) - nonblank.nunique()) + + if len(nonblank) == 0: + result["errors"].append("no non-blank SMILES found in the CSV") + return result + + try: + parsed, invalid, method = _parse_smiles_sample(nonblank.tolist(), full=strict_rdkit) + except RuntimeError as exc: + result["errors"].append(str(exc)) + return result + result["smiles_check_count"] = parsed + result["num_invalid_smiles"] = invalid + result["smiles_check_method"] = method + + if invalid > 0: + scope = "all rows" if method == "full" else f"the {parsed} sampled rows" + result["errors"].append( + f"{invalid} out of {parsed} SMILES in {scope} failed to parse with RDKit. " + "Either pre-clean the CSV with scripts/clean_smiles.py or pass --strict-rdkit to see " + "the full count." + ) + + # 4. Target-column handling — finetune mode only. + if mode == "finetune": + if targets: + missing = [t for t in targets if t not in df.columns] + if missing: + result["errors"].append( + f"target column(s) not found in CSV: {missing}. " + f"Available columns: {result['columns']}" + ) + else: + result["target_columns"] = list(targets) + for t in targets: + nan_count = int(df[t].isna().sum()) + result["num_missing_per_target"][t] = nan_count + # Confirm numeric-ish. + nonnan = df[t].dropna() + converted = pd.to_numeric(nonnan, errors="coerce") + non_numeric_count = int(converted.isna().sum()) + if non_numeric_count > 0: + result["warnings"].append( + f"target column '{t}' has {non_numeric_count} non-numeric value(s) " + f"that will be dropped by the finetune runner." + ) + else: + # Auto-detect — surface candidates so the skill can prompt the user. + result["auto_detected_targets"] = _autodetect_target_columns(df, smiles_col) + if not result["auto_detected_targets"]: + result["errors"].append( + "no numeric non-smiles columns detected. finetune needs at least one target column; " + "specify it explicitly via --targets ." + ) + else: + result["warnings"].append( + f"--targets was not specified; auto-detected candidate target columns " + f"{result['auto_detected_targets']}. The skill will prompt the user to confirm." + ) + + # 5. Small-corpus warning — only for pretrain (other modes can be tiny by design). + if mode == "pretrain" and result["num_rows"] < 100: + result["warnings"].append( + f"pretrain corpus is only {result['num_rows']} molecule(s). Pretraining typically " + f"needs orders of magnitude more — verify this is the intended input." + ) + + result["ok"] = not result["errors"] + return result + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Validate a CSV input for a KERMT agent workflow.") + parser.add_argument("--mode", required=True, choices=["pretrain", "finetune", "inference", "embed"]) + parser.add_argument("--csv", required=True, help="Path to the input CSV") + parser.add_argument("--targets", nargs="+", default=None, + help="(finetune only) target column names. If omitted, the validator auto-detects " + "numeric non-smiles columns and reports them as candidates.") + parser.add_argument("--strict-rdkit", action="store_true", + help="Parse every SMILES with RDKit rather than sampling (slow on large CSVs).") + args = parser.parse_args(argv) + + try: + result = validate(args.mode, args.csv, args.targets, args.strict_rdkit) + except Exception as exc: # noqa: BLE001 + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({ + "ok": False, + "mode": args.mode, + "csv_path": args.csv, + "errors": [f"unhandled exception in validator: {type(exc).__name__}: {exc}"], + "warnings": [], + }, indent=2)) + return 1 + + print(json.dumps(result, indent=2)) + return 0 if result["ok"] else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-continue-pretrain/scripts/fetch_released_model.py b/skills/kermt-continue-pretrain/scripts/fetch_released_model.py new file mode 100644 index 0000000..ca2ad57 --- /dev/null +++ b/skills/kermt-continue-pretrain/scripts/fetch_released_model.py @@ -0,0 +1,234 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Download a released KERMT model bundle from Hugging Face. + +Runs INSIDE the kermt container (huggingface_hub is part of the image env). +Writes the released-model directory bundle — the checkpoint plus its vocab +files — into the directory mounted at `--out` (the skills mount the user's +chosen save location there via `kermt_container.sh --model-dir `), then +emits a single JSON object to stdout that the calling skill parses. + +The downloaded directory is exactly the repo's "released model bundle" layout +(see skills/README.md "Released models"): `.pt` + the three +`pretrain_*_vocab.*` files in one flat directory. The downstream skill then +feeds it through the existing `--ckpt /` flow; for +continue-pretrain the bundled vocab files are auto-detected in the ckpt's +parent directory. No runner changes are needed. + +Defaults (repo id, pinned revision, ckpt + vocab filenames) come from +`config/released_model.json` so the pin lives in one place; every value +is overridable on the CLI. + +Idempotent: if the bundle is already complete in `--out` (ckpt + all vocab +files present), nothing is downloaded and `reused: true` is reported — so a +re-invocation never re-fetches the 282 MB checkpoint. + +Authentication: the repo is public (no token needed). If `HF_TOKEN` is set in +the environment (forwarded into the container by `kermt_container.sh`), +huggingface_hub picks it up automatically — useful against shared-IP rate +limits or if the repo is ever gated. + +Output (stdout) +--------------- +{ + "ok": true | false, + "repo_id": str, + "revision": str, + "out": str, // container path of the bundle dir (e.g. /model) + "ckpt": str | null, // container path of the checkpoint file + "vocab_dir": str | null, // == out (where the vocab files live) + "ckpt_name": str, + "vocab_files": [str, ...], + "files_present": [str, ...], + "ckpt_bytes": int | null, + "reused": bool, // true if the bundle already existed (no download) + "license": str | null, + "license_url": str | null, + "errors": [str, ...] +} + +Exit code: 0 on `ok: true`, 1 on `ok: false`. + +CLI +--- + fetch_released_model.py [--out /model] + [--repo-id ] [--revision ] + [--ckpt-name ] [--config ] +""" + +from __future__ import annotations + +import argparse +import json +import sys +import traceback +from pathlib import Path +from typing import Any + +# Default config lives at config/released_model.json (one dir up from +# scripts/). Resolved relative to this file so the script is +# location-independent. +DEFAULT_CONFIG = ( + Path(__file__).resolve().parent.parent / "config" / "released_model.json" +) + + +def _load_config(config_path: Path) -> dict[str, Any]: + if not config_path.is_file(): + raise FileNotFoundError(f"released-model config not found at {config_path}") + return json.loads(config_path.read_text()) + + +def fetch( + *, + out: Path, + repo_id: str, + revision: str, + ckpt_name: str, + vocab_files: list[str], + license_name: str | None = None, + license_url: str | None = None, +) -> dict[str, Any]: + """Resolve-or-download the released bundle into `out`. Returns the manifest + dict (never raises for the expected failure modes — they land in + `errors[]` with `ok: false`).""" + result: dict[str, Any] = { + "ok": False, + "repo_id": repo_id, + "revision": revision, + "out": str(out), + "ckpt": None, + "vocab_dir": None, + "ckpt_name": ckpt_name, + "vocab_files": list(vocab_files), + "files_present": [], + "ckpt_bytes": None, + "reused": False, + "license": license_name, + "license_url": license_url, + "errors": [], + } + + required = [ckpt_name, *vocab_files] + + def _present() -> list[str]: + return [name for name in required if (out / name).is_file()] + + # 1. Idempotent reuse — bundle already complete in `out`. + if out.is_dir() and set(_present()) == set(required): + result["reused"] = True + else: + # 2. Download. Import here so a stale image (missing huggingface_hub) + # surfaces a clean, actionable JSON error rather than a traceback. + try: + from huggingface_hub import snapshot_download + except ImportError: + result["errors"].append( + "huggingface_hub is not available in the container image. The " + "released-model download needs it; rebuild the image with " + "`kermt-setup` (it now ships huggingface_hub) and retry." + ) + return result + + out.mkdir(parents=True, exist_ok=True) + try: + # local_dir gives a flat copy (the bundle layout) rather than the + # opaque blob/snapshot cache. HF_TOKEN, if set, is read by the lib. + snapshot_download(repo_id=repo_id, revision=revision, local_dir=str(out)) + except Exception as exc: # noqa: BLE001 + result["errors"].append( + f"download failed for {repo_id}@{revision}: {type(exc).__name__}: {exc}" + ) + return result + + # 3. Verify the bundle is complete regardless of download/reuse path. + present = _present() + result["files_present"] = present + missing = [name for name in required if name not in present] + if missing: + result["errors"].append( + f"bundle at {out} is missing expected file(s): {missing}. " + f"Present: {present}." + ) + return result + + ckpt_path = out / ckpt_name + result["ckpt"] = str(ckpt_path) + result["vocab_dir"] = str(out) + try: + result["ckpt_bytes"] = ckpt_path.stat().st_size + except OSError: + result["ckpt_bytes"] = None + + result["ok"] = True + return result + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser( + description="Download a released KERMT model bundle from Hugging Face (runs in-container)." + ) + parser.add_argument( + "--out", + default="/model", + help="Directory to write the bundle into (default: /model, the --model-dir mount).", + ) + parser.add_argument( + "--config", + default=str(DEFAULT_CONFIG), + help="Path to released_model.json (default: config/released_model.json).", + ) + parser.add_argument( + "--repo-id", default=None, help="Override the HF repo id from the config." + ) + parser.add_argument( + "--revision", + default=None, + help="Override the pinned revision (sha/tag/branch).", + ) + parser.add_argument( + "--ckpt-name", + default=None, + help="Override the checkpoint filename from the config.", + ) + args = parser.parse_args(argv) + + try: + cfg = _load_config(Path(args.config)) + repo_id = args.repo_id or cfg["repo_id"] + revision = args.revision or cfg["revision"] + ckpt_name = args.ckpt_name or cfg["ckpt_name"] + vocab_files = list(cfg.get("vocab_files", [])) + result = fetch( + out=Path(args.out), + repo_id=repo_id, + revision=revision, + ckpt_name=ckpt_name, + vocab_files=vocab_files, + license_name=cfg.get("license"), + license_url=cfg.get("license_url"), + ) + except Exception as exc: # noqa: BLE001 + print(traceback.format_exc(), file=sys.stderr) + print( + json.dumps( + { + "ok": False, + "out": args.out, + "errors": [ + f"unhandled exception in fetch_released_model: {type(exc).__name__}: {exc}" + ], + }, + indent=2, + ) + ) + return 1 + + print(json.dumps(result, indent=2)) + return 0 if result["ok"] else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/skills/kermt-continue-pretrain/scripts/kermt_container.sh b/skills/kermt-continue-pretrain/scripts/kermt_container.sh new file mode 100755 index 0000000..028057e --- /dev/null +++ b/skills/kermt-continue-pretrain/scripts/kermt_container.sh @@ -0,0 +1,484 @@ +#!/usr/bin/env bash +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# kermt_container.sh — bootstrap helper for the kermt agent skills. +# +# Two ways to use this file: +# +# 1. As a subcommand dispatcher (recommended for skills): +# "$SKILL_DIR/scripts/kermt_container.sh" ensure_image +# "$SKILL_DIR/scripts/kermt_container.sh" run --ckpt /host/ckpt.pt -- python -c 'import torch; print(torch.cuda.device_count())' +# "$SKILL_DIR/scripts/kermt_container.sh" run_detached --name foo --run-dir runs/foo -- bash train.sh +# +# 2. Sourced into a shell or another script, then call the kermt_* functions +# directly: +# source "$SKILL_DIR/scripts/kermt_container.sh" +# kermt_ensure_image +# kermt_run --ckpt /host/ckpt.pt -- python ... +# +# Configuration (override via env vars before invocation): +# KERMT_IMAGE docker image tag (default: kermt:latest) +# KERMT_REPO host path to the kermt repo checkout (default: auto-derived +# from this script's location) +# KERMT_GPUS value passed to docker --gpus (default: all) +# +# Mount flags accepted by kermt_run / kermt_run_detached: +# --data bind to /data (read-only). If is a file, +# its PARENT directory is mounted at /data so +# commands can use /data/; if is a +# directory, it is mounted at /data directly. +# --ckpt bind to /ckpt (read-only; the path is mounted as-is) +# --vocab-dir bind to /vocab (read-only) +# --run-dir bind to /runs (read-write; created on host if missing) +# --model-dir bind to /model (read-write; created on host if missing). +# Target for released-model downloads (fetch_released_model.py). +# +# Additional flags for kermt_run_detached: +# --name docker container name (default: kermt--) +# +# Everything after `--` is the command passed to the container. It runs inside +# the `kermt` conda environment (the image's default env). + +set -o pipefail + +: "${KERMT_IMAGE:=kermt:latest}" +: "${KERMT_GPUS:=all}" + +# The skill may be installed outside the KERMT checkout. Mount its own helpers +# separately so container commands always execute the distributed skill copy. +_kermt_script_dir="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)" +_kermt_bundle_dir="$(cd "$_kermt_script_dir/.." && pwd)" +if [[ -z "${KERMT_REPO:-}" ]]; then + for _kermt_start in "$_kermt_script_dir" "$PWD"; do + _kermt_candidate="$_kermt_start" + while [[ "$_kermt_candidate" != / ]]; do + if [[ -f "$_kermt_candidate/main.py" && -d "$_kermt_candidate/kermt" ]]; then + KERMT_REPO="$_kermt_candidate" + break 2 + fi + _kermt_candidate="$(dirname "$_kermt_candidate")" + done + done + unset _kermt_start _kermt_candidate +fi +unset _kermt_script_dir + +_kermt_require_repo() { + if [[ -z "${KERMT_REPO:-}" || ! -f "$KERMT_REPO/main.py" || ! -d "$KERMT_REPO/kermt" ]]; then + echo "[kermt] Set KERMT_REPO to the KERMT checkout containing main.py and kermt/." >&2 + return 1 + fi + KERMT_REPO="$(cd "$KERMT_REPO" && pwd)" || return $? + export KERMT_REPO +} + +# ----------------------------------------------------------------------------- +# Host environment checks +# ----------------------------------------------------------------------------- + +kermt_check_docker() { + if ! command -v docker >/dev/null 2>&1; then + echo "[kermt] error: docker not found on PATH. Install Docker first." >&2 + return 1 + fi + if ! docker info >/dev/null 2>&1; then + echo "[kermt] error: docker daemon not reachable. Is the docker service running, and is your user in the 'docker' group?" >&2 + return 1 + fi +} + +kermt_check_system() { + # Probe host system and report GPU presence + VRAM + compute capability + + # driver / CUDA version + disk space. Emits a single JSON document to + # stdout that the calling skill consumes; exits 0 with `ok: false` and a + # populated `gaps` array when anything is below the per-workflow minimum, + # exits 1 only on unexpected internal errors. Uses host nvidia-smi + df + + # host python3 (stdlib only). + python3 - "$KERMT_REPO" "$KERMT_IMAGE" <<'PYEOF' +import json, os, shutil, subprocess, sys + +repo, image = sys.argv[1], sys.argv[2] + +result = { + "ok": True, + "gpus": [], + "disk": {"path": repo, "free_gb": None, "min_gb": 20}, + "host": {"docker": None, "nvidia_smi": None, "container_toolkit": None}, + "image": {"tag": image, "present_locally": None}, + "gaps": [], +} + +def _gap(msg): + result["ok"] = False + result["gaps"].append(msg) + +# docker presence +try: + r = subprocess.run(["docker", "info"], capture_output=True, text=True, timeout=10) + result["host"]["docker"] = "ok" if r.returncode == 0 else f"failed: {r.stderr.strip().splitlines()[-1] if r.stderr else 'unknown'}" + if r.returncode != 0: + _gap("docker daemon not reachable (is the service running, and is your user in the 'docker' group?)") +except FileNotFoundError: + result["host"]["docker"] = "not found" + _gap("docker not on PATH; install Docker first") +except Exception as e: + result["host"]["docker"] = f"error: {e}" + _gap(f"docker probe failed: {e}") + +# nvidia-smi (host driver) +try: + r = subprocess.run( + ["nvidia-smi", "--query-gpu=name,memory.total,compute_cap,driver_version,uuid", + "--format=csv,noheader,nounits"], + capture_output=True, text=True, timeout=10, + ) + if r.returncode == 0: + result["host"]["nvidia_smi"] = "ok" + for line in r.stdout.strip().splitlines(): + parts = [p.strip() for p in line.split(",")] + if len(parts) >= 5: + try: + vram_mb = int(parts[1]) + except ValueError: + vram_mb = None + result["gpus"].append({ + "name": parts[0], + "vram_mb": vram_mb, + "compute_cap": parts[2], + "driver": parts[3], + "uuid": parts[4], + }) + if not result["gpus"]: + _gap("nvidia-smi succeeded but reported no GPUs") + else: + result["host"]["nvidia_smi"] = "failed" + _gap("nvidia-smi found but failed; is the NVIDIA driver loaded?") +except FileNotFoundError: + result["host"]["nvidia_smi"] = "not found" + _gap("nvidia-smi not on PATH; install the NVIDIA driver") +except Exception as e: + result["host"]["nvidia_smi"] = f"error: {e}" + _gap(f"nvidia-smi probe failed: {e}") + +# disk free at the repo location +try: + free_bytes = shutil.disk_usage(repo).free + free_gb = free_bytes // (1024**3) + result["disk"]["free_gb"] = free_gb + if free_gb < result["disk"]["min_gb"]: + _gap(f"disk free at {repo} is {free_gb} GB; need at least {result['disk']['min_gb']} GB for the kermt image") +except Exception as e: + _gap(f"could not check disk space at {repo}: {e}") + +# image presence (informational only) +try: + r = subprocess.run(["docker", "image", "inspect", image], capture_output=True, text=True, timeout=10) + result["image"]["present_locally"] = (r.returncode == 0) +except Exception: + result["image"]["present_locally"] = None + +# nvidia-container-toolkit probe — only meaningful if both docker and a +# locally-present image are available. Pick kermt:$tag first; fall back to +# the small CUDA base image if that's the only one present; otherwise skip +# (avoid pulling anything). +def _probe_image(): + for img in (image, "nvidia/cuda:12.6.3-base-ubuntu22.04"): + r = subprocess.run(["docker", "image", "inspect", img], capture_output=True) + if r.returncode == 0: + return img + return None + +probe_img = _probe_image() +if probe_img: + try: + r = subprocess.run( + ["docker", "run", "--rm", "--gpus", "all", probe_img, "nvidia-smi"], + capture_output=True, text=True, timeout=60, + ) + if r.returncode == 0: + result["host"]["container_toolkit"] = f"ok (probed via {probe_img})" + else: + result["host"]["container_toolkit"] = f"failed (probed via {probe_img})" + _gap("`docker run --gpus all` failed; install nvidia-container-toolkit and ensure the host driver supports it") + except Exception as e: + result["host"]["container_toolkit"] = f"error: {e}" + _gap(f"nvidia-container-toolkit probe failed: {e}") +else: + result["host"]["container_toolkit"] = "skipped (no probe image present locally; run ensure_image first)" + +print(json.dumps(result, indent=2)) +PYEOF +} + +kermt_check_gpu() { + # Probes whether `docker --gpus all` is wired up (nvidia-container-toolkit). + # Image-selection priority (never pulls anything): + # 1) $KERMT_IMAGE if it exists locally, + # 2) else nvidia/cuda:12.6.3-base-ubuntu22.04 if it exists locally, + # 3) else skip with a warning (return 0). The smoke test inside kermt_run + # will catch broken GPU passthrough later anyway. + local probe_img="" + if docker image inspect "$KERMT_IMAGE" >/dev/null 2>&1; then + probe_img="$KERMT_IMAGE" + elif docker image inspect nvidia/cuda:12.6.3-base-ubuntu22.04 >/dev/null 2>&1; then + probe_img="nvidia/cuda:12.6.3-base-ubuntu22.04" + else + echo "[kermt] check_gpu: skipped — neither '$KERMT_IMAGE' nor 'nvidia/cuda:12.6.3-base-ubuntu22.04' is present locally. Run 'ensure_image' first, or this probe will be exercised by the in-container smoke test." >&2 + return 0 + fi + if ! docker run --rm --gpus all "$probe_img" nvidia-smi >/dev/null 2>&1; then + echo "[kermt] error: 'docker run --gpus all' failed (probe image: $probe_img). Install nvidia-container-toolkit and ensure the host has a CUDA-capable NVIDIA driver." >&2 + return 1 + fi +} + +# ----------------------------------------------------------------------------- +# Image build / verification +# ----------------------------------------------------------------------------- + +kermt_ensure_image() { + _kermt_require_repo || return $? + kermt_check_docker || return $? + if docker image inspect "$KERMT_IMAGE" >/dev/null 2>&1; then + local id + id=$(docker image inspect "$KERMT_IMAGE" --format '{{.Id}}' 2>/dev/null | cut -c1-19) + echo "[kermt] image '$KERMT_IMAGE' already present (${id:-unknown})" + return 0 + fi + echo "[kermt] image '$KERMT_IMAGE' not found; building from $KERMT_REPO/Dockerfile" + echo "[kermt] first build typically takes 10-20 minutes on a typical workstation; subsequent runs reuse the cached image" + docker build -t "$KERMT_IMAGE" -f "$KERMT_REPO/Dockerfile" "$KERMT_REPO" +} + +# ----------------------------------------------------------------------------- +# Mount-flag parser, internal +# ----------------------------------------------------------------------------- +# Reads flags from the caller's positional args until it hits '--', appending +# `-v src:dst[:ro]` pairs into the caller-provided array name (passed as $1). +# Returns the number of caller-provided args consumed via _kermt_consumed. +# This is bash-specific (uses nameref via `declare -n`). + +_kermt_parse_mounts() { + local -n _out="$1" + shift + _kermt_consumed=0 + while [[ $# -gt 0 ]]; do + case "$1" in + --) + return 0 + ;; + --data) + [[ -e "$2" ]] || { echo "[kermt] --data path not found: $2" >&2; return 1; } + # If the user passes a file, mount its parent directory at /data so + # downstream commands can refer to /data/. Mounting a + # single file at /data makes the path-as-directory pattern in the + # skill examples (`--csv /data/`) fail with "not found". + if [[ -d "$2" ]]; then + _out+=("-v" "$(realpath "$2"):/data:ro") + else + _out+=("-v" "$(realpath "$(dirname "$2")"):/data:ro") + fi + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --ckpt) + [[ -e "$2" ]] || { echo "[kermt] --ckpt path not found: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/ckpt:ro") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --vocab-dir) + [[ -d "$2" ]] || { echo "[kermt] --vocab-dir not found or not a directory: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/vocab:ro") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --run-dir) + mkdir -p "$2" || { echo "[kermt] failed to create --run-dir: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/runs") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --model-dir) + mkdir -p "$2" || { echo "[kermt] failed to create --model-dir: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/model") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + *) + return 0 + ;; + esac + done +} + +# ----------------------------------------------------------------------------- +# Foreground / detached run +# ----------------------------------------------------------------------------- + +# Capture host-side git state for the repo and emit `-e KERMT_REPO_COMMIT=… +# -e KERMT_REPO_DIRTY=true|false` flags. Used by the run / run_detached +# wrappers so the runner's run.json manifest gets honest commit info even +# though `git -C /workspace` inside the container fails due to bind-mount +# ownership. +_kermt_git_env_flags() { + local commit="unknown" + local dirty="false" + if command -v git >/dev/null 2>&1 && [[ -d "$KERMT_REPO/.git" ]]; then + local c + c=$(git -C "$KERMT_REPO" rev-parse HEAD 2>/dev/null) && commit="$c" + # `--untracked-files=no` filters out user-private notes (e.g. a CLAUDE.md + # or RELEASE_PLAN_v2.0.md at the repo root) that wouldn't affect + # reproducibility — only modifications to tracked files do. + if [[ -n "$(git -C "$KERMT_REPO" status --porcelain --untracked-files=no 2>/dev/null | head -n 1)" ]]; then + dirty="true" + fi + fi + printf '%s\n%s\n%s\n%s\n' "-e" "KERMT_REPO_COMMIT=$commit" "-e" "KERMT_REPO_DIRTY=$dirty" +} + +# Forward HF_TOKEN into the container when it is set, so fetch_released_model.py +# can authenticate to Hugging Face. The current release is public (no token +# needed); this only guards against shared-IP rate limits or a future gated +# repo. Emits nothing when HF_TOKEN is unset. +_kermt_hf_env_flags() { + if [[ -n "${HF_TOKEN:-}" ]]; then + printf '%s\n%s\n' "-e" "HF_TOKEN=$HF_TOKEN" + fi +} + +kermt_run() { + kermt_ensure_image || return $? + local mount_args=() + _kermt_parse_mounts mount_args "$@" || return $? + shift "$_kermt_consumed" + if [[ "${1:-}" != "--" ]]; then + echo "[kermt] expected '--' separating mount flags from the command (got '${1:-}')" >&2 + return 1 + fi + shift + if [[ $# -eq 0 ]]; then + echo "[kermt] no command supplied after '--'" >&2 + return 1 + fi + local git_args=() + while IFS= read -r line; do git_args+=("$line"); done < <(_kermt_git_env_flags) + local hf_args=() + while IFS= read -r line; do hf_args+=("$line"); done < <(_kermt_hf_env_flags) + docker run --rm --gpus "$KERMT_GPUS" \ + --user "$(id -u):$(id -g)" \ + -v "$KERMT_REPO:/workspace" \ + -v "$_kermt_bundle_dir:/skill:ro" \ + "${mount_args[@]}" \ + -w /workspace \ + -e KERMT_REPO=/workspace \ + -e PYTHONPATH=/workspace \ + -e HOME=/tmp/kermt-home \ + "${git_args[@]}" \ + "${hf_args[@]}" \ + "$KERMT_IMAGE" \ + conda run -n kermt --no-capture-output bash -c "$*" +} + +kermt_run_detached() { + kermt_ensure_image || return $? + local name="" + local mount_args=() + # Pull --name out first, then let the shared mount parser handle the rest. + while [[ $# -gt 0 ]]; do + case "$1" in + --name) name="$2"; shift 2 ;; + --) break ;; + --data|--ckpt|--vocab-dir|--run-dir|--model-dir) break ;; + *) break ;; + esac + done + _kermt_parse_mounts mount_args "$@" || return $? + shift "$_kermt_consumed" + if [[ "${1:-}" != "--" ]]; then + echo "[kermt] expected '--' separating mount flags from the command (got '${1:-}')" >&2 + return 1 + fi + shift + if [[ $# -eq 0 ]]; then + echo "[kermt] no command supplied after '--'" >&2 + return 1 + fi + if [[ -z "$name" ]]; then + name="kermt-$(date -u +%Y%m%dT%H%M%SZ)-$$" + fi + local cid + local git_args=() + while IFS= read -r line; do git_args+=("$line"); done < <(_kermt_git_env_flags) + local hf_args=() + while IFS= read -r line; do hf_args+=("$line"); done < <(_kermt_hf_env_flags) + cid=$(docker run -d --gpus "$KERMT_GPUS" \ + --user "$(id -u):$(id -g)" \ + --name "$name" \ + -v "$KERMT_REPO:/workspace" \ + -v "$_kermt_bundle_dir:/skill:ro" \ + "${mount_args[@]}" \ + -w /workspace \ + -e KERMT_REPO=/workspace \ + -e PYTHONPATH=/workspace \ + -e HOME=/tmp/kermt-home \ + "${git_args[@]}" \ + "${hf_args[@]}" \ + "$KERMT_IMAGE" \ + conda run -n kermt --no-capture-output bash -c "$*") || return $? + echo "[kermt] container started: name=$name id=$cid" + echo "[kermt] follow logs: docker logs -f $name" + echo "[kermt] wait for exit: docker wait $name" + echo "[kermt] stop: docker stop $name" + echo "$cid" +} + +# ----------------------------------------------------------------------------- +# Subcommand dispatch when invoked directly (not sourced) +# ----------------------------------------------------------------------------- + +if [[ "${BASH_SOURCE[0]:-$0}" == "${0}" ]]; then + cmd="${1:-}"; shift || true + case "$cmd" in + check_docker) kermt_check_docker "$@" ;; + check_gpu) kermt_check_gpu "$@" ;; + check_system) kermt_check_system "$@" ;; + ensure_image) kermt_ensure_image "$@" ;; + run) kermt_run "$@" ;; + run_detached) kermt_run_detached "$@" ;; + ""|-h|--help) + cat >&2 < [args...] + +Subcommands: + check_docker Verify docker is installed and the daemon is reachable. + check_gpu Verify 'docker --gpus all' works (nvidia-container-toolkit). + check_system Emit a JSON probe of host GPU + VRAM + compute_cap + + driver + disk space + container toolkit + image presence. + Exits 0 with ok=false + a 'gaps' list when anything's + below the per-workflow minimum. + ensure_image Build kermt:latest from \$KERMT_REPO/Dockerfile if missing. + run [flags] -- ... Run a command inside the container (foreground, --rm). + run_detached [flags] -- ... + Run detached; prints container name + id + log hint. + +Mount flags (for run / run_detached): + --data bind to /data (read-only) + --ckpt bind to /ckpt (read-only) + --vocab-dir bind to /vocab (read-only) + --run-dir bind to /runs (read-write; created on host if missing) + --model-dir bind to /model (read-write; released-model download target) + +Additional flags for run_detached: + --name container name (default: kermt--) + +Environment overrides: + KERMT_IMAGE default kermt:latest + KERMT_REPO checkout path; otherwise discovered above the skill or working directory + KERMT_GPUS default all +EOF + exit 1 + ;; + *) + echo "[kermt] unknown subcommand: $cmd" >&2 + echo "[kermt] run '$0 --help' for usage" >&2 + exit 1 + ;; + esac +fi diff --git a/skills/kermt-continue-pretrain/scripts/prepare_data.py b/skills/kermt-continue-pretrain/scripts/prepare_data.py new file mode 100644 index 0000000..f0edb3e --- /dev/null +++ b/skills/kermt-continue-pretrain/scripts/prepare_data.py @@ -0,0 +1,817 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Mode-dispatched data preparation pipeline for the KERMT agent skills. + +Composes the existing repo data-prep scripts (`scripts/clean_smiles.py`, +`scripts/save_features.py`, `scripts/build_vocab.py`, `scripts/split_data.py`) +into a single one-call entry point per workflow. Output lands in `--out` with +a `prepare_data.json` manifest that the downstream runners read. + +Mode pipelines +-------------- +pretrain : clean -> (optional auto-split train into train+val by --val-frac) + -> save_features (fgtasklabel) on each CSV + -> vocab step: if --vocab-dir / --{atom,bond,smiles}-vocab given, + copy those through (continue-pretrain case — the ckpt's vocab + is authoritative); else if --skip-vocab, skip; + else build_vocab on train (pretrain-from-scratch case) + -> split_data (graph + feature shards + summary.txt) per CSV +finetune : clean each provided CSV -> (optional random split when only one + CSV is provided; emits a strong warning recommending scaffold- + balanced pre-splits) -> save_features (rdkit_2d_normalized) per CSV +inference : clean -> save_features (rdkit_2d_normalized) +embed : clean only (extract_embeddings.py featurizes on the fly) + +Output convention +----------------- +The manifest under `/prepare_data.json` captures every step's inputs, +outputs, duration, and skipped-due-to-existing flag, plus a top-level +`split_method` field (one of: "user_provided", "random", "n/a") that the +finetune runner uses to pass the correct `--split_type` to main.py. + +Subprocess composition +---------------------- +Each underlying script is invoked via `subprocess.run`. The PYTHONPATH=/workspace +env var (set by `scripts/kermt_container.sh`) makes the `kermt` package +importable inside the subprocesses; without it, build_vocab.py and split_data.py +fail with `ModuleNotFoundError: No module named 'kermt'`. + +CLI +--- + prepare_data.py --mode {pretrain|finetune|inference|embed} + --csv --out + [--val-csv ] [--test-csv ] + [--val-frac 0.1] [--test-frac 0.1] [--seed 0] + [--sample-per-file 100000] [--vocab-format json] + [--dataset-name pretrain] + [--targets COL [COL ...]] + [--features-generator ] + [--smiles-column 0] + [--force] [--skip-clean] [--skip-features] + [--skip-vocab] [--skip-split] +""" +from __future__ import annotations + +import argparse +import json +import os +import shutil +import subprocess +import sys +import time +import traceback +from pathlib import Path +from typing import Any + +import pandas as pd + +# sys.path tweak so `_utils` is importable regardless of how this script +# is invoked (kermt_run sets PYTHONPATH=/workspace; bare-Python launches +# from the host don't). +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import PRETRAIN_VOCAB_STEMS, resolve_kermt_repo, validate_vocab_file # noqa: E402 + + +REPO_ROOT = resolve_kermt_repo() +EXISTING_SCRIPTS = REPO_ROOT / "scripts" + +DEFAULT_FEATURES_GENERATOR = { + "pretrain": "fgtasklabel", + "finetune": "rdkit_2d_normalized", + "inference": "rdkit_2d_normalized", + "embed": None, # not used +} + +VALID_MODES = ("pretrain", "finetune", "inference", "embed") + + +# --------------------------------------------------------------------------- +# Subprocess helpers +# --------------------------------------------------------------------------- + +def _run(cmd: list[str], step_name: str, manifest: dict[str, Any]) -> dict[str, Any]: + """Run a subprocess, append a step entry to manifest, raise on failure.""" + step: dict[str, Any] = { + "name": step_name, + "cmd": cmd, + "duration_s": None, + "ok": False, + "stderr_tail": "", + "skipped_due_to_existing": False, + } + t0 = time.time() + proc = subprocess.run(cmd, capture_output=True, text=True) + step["duration_s"] = round(time.time() - t0, 2) + if proc.returncode != 0: + step["stderr_tail"] = (proc.stderr or "").splitlines()[-20:] + step["ok"] = False + manifest["steps"].append(step) + raise RuntimeError( + f"step '{step_name}' failed (exit {proc.returncode}); " + f"command: {' '.join(cmd)}\nstderr tail:\n" + "\n".join(step["stderr_tail"]) + ) + step["ok"] = True + manifest["steps"].append(step) + return step + + +def _skipped(step_name: str, output_path: str, manifest: dict[str, Any]) -> dict[str, Any]: + step = { + "name": step_name, + "output": output_path, + "ok": True, + "duration_s": 0.0, + "skipped_due_to_existing": True, + } + manifest["steps"].append(step) + return step + + +def _exists_nonempty(path: Path) -> bool: + """File exists with non-zero size, or directory exists with at least one entry.""" + if not path.exists(): + return False + if path.is_file(): + return path.stat().st_size > 0 + if path.is_dir(): + try: + next(path.iterdir()) + return True + except StopIteration: + return False + return False + + +# --------------------------------------------------------------------------- +# Per-script wrappers +# --------------------------------------------------------------------------- + +def _resolve_smiles_column(csv_path: Path, explicit_value: int | None) -> int: + """Return the 0-based index of the SMILES column in csv_path. + + Auto-detection rule when `explicit_value is None`: + 1. Read the CSV header (first non-empty row). + 2. Prefer an exact lowercase `smiles` column (kermt convention). + 3. Otherwise accept a single case-insensitive match + (`SMILES`, `Smiles`, etc.). + 4. If no match (or multiple ambiguous matches), raise a ValueError + that surfaces the header so the user can disambiguate via + `--smiles-column N`. + + Real datasets routinely place SMILES at column index ≠ 0 + (e.g. openadmet's all.csv has "Molecule Name" at col 0 and "SMILES" + at col 1). Auto-detection prevents the silent 0-row-clean failure + mode where every row gets rejected because col 0 doesn't parse as + a SMILES string. + """ + if explicit_value is not None: + return explicit_value + + if not csv_path.is_file(): + raise ValueError(f"input CSV not found: {csv_path}") + + import csv as _csv + with csv_path.open("r", newline="") as f: + reader = _csv.reader(f) + try: + header = next(reader) + except StopIteration: + raise ValueError(f"input CSV {csv_path} is empty") + + stripped = [c.strip() for c in header] + # Prefer exact lowercase "smiles" + exact = [i for i, c in enumerate(stripped) if c == "smiles"] + if exact: + return exact[0] + # Then case-insensitive + ci = [i for i, c in enumerate(stripped) if c.lower() == "smiles"] + if len(ci) == 1: + return ci[0] + if len(ci) > 1: + raise ValueError( + f"input CSV {csv_path} has multiple SMILES-named columns: " + f"{[header[i] for i in ci]} at indices {ci}. " + "Pass --smiles-column N (0-based) to disambiguate." + ) + raise ValueError( + f"could not auto-detect a SMILES column in {csv_path}. " + f"Header columns: {header}. " + "Pass --smiles-column N (0-based) to specify which column holds SMILES." + ) + + +def _clean_smiles( + input_csv: Path, output_csv: Path, smiles_column: int, manifest: dict[str, Any], force: bool +) -> Path: + if not force and _exists_nonempty(output_csv): + _skipped(f"clean_smiles({input_csv.name})", str(output_csv), manifest) + return output_csv + output_csv.parent.mkdir(parents=True, exist_ok=True) + if force and output_csv.exists(): + # clean_smiles.py prompts interactively (input()) when the output file + # already exists — that's an EOFError in a non-TTY subprocess. Pre-delete. + output_csv.unlink() + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "clean_smiles.py"), + "--input", str(input_csv), + "--output", str(output_csv), + "--smiles_column", str(smiles_column), + ] + _run(cmd, f"clean_smiles({input_csv.name})", manifest) + return output_csv + + +def _reduce_to_smiles_column( + csv_path: Path, smiles_column: int, manifest: dict[str, Any] +) -> Path: + """Rewrite an inference CSV to keep only the SMILES column (at index 0). + + Downstream `kermt.util.utils.get_data` -> `MoleculeDatapoint.__init__` + floats every column after SMILES, which crashes on non-numeric passthrough + columns (e.g. a 'split' label of 'train'/'val'/'test', or a 'Molecule Name' + string). Inference does not need target columns, so drop them here. + + Note on skip semantics: this step is idempotent — running it on an + already-single-column file is a no-op. We record that with + `skipped_due_to_idempotent: True`, NOT `skipped_due_to_existing: True`. + The two fields have different meanings: `_existing` means "I found a + cached output file from a prior run and reused it" (overridden by + `--force`); `_idempotent` means "the input is already in the desired + state, so re-executing changes nothing" (safe to skip even under + `--force`). + """ + step_name = f"reduce_to_smiles_only({csv_path.name})" + start = time.time() + df = pd.read_csv(csv_path) + if df.shape[1] == 1: + manifest["steps"].append({ + "name": step_name, + "output": str(csv_path), + "ok": True, + "duration_s": time.time() - start, + "skipped_due_to_idempotent": True, + "note": "already single-column", + }) + return csv_path + effective_col = smiles_column if 0 <= smiles_column < df.shape[1] else 0 + df.iloc[:, [effective_col]].to_csv(csv_path, index=False) + manifest["steps"].append({ + "name": step_name, + "output": str(csv_path), + "ok": True, + "duration_s": time.time() - start, + "input_cols": int(df.shape[1]), + "kept_col": effective_col, + "kept_col_name": str(df.columns[effective_col]), + }) + return csv_path + + +def _save_features( + csv_path: Path, npz_path: Path, generator: str, manifest: dict[str, Any], force: bool +) -> Path: + if not force and _exists_nonempty(npz_path): + _skipped(f"save_features({csv_path.name}, {generator})", str(npz_path), manifest) + return npz_path + npz_path.parent.mkdir(parents=True, exist_ok=True) + if force and npz_path.exists(): + npz_path.unlink() # --restart still loads partial state if file exists; pre-delete to be safe + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "save_features.py"), + "--data_path", str(csv_path), + "--save_path", str(npz_path), + "--features_generator", generator, + "--restart", + ] + _run(cmd, f"save_features({csv_path.name}, {generator})", manifest) + return npz_path + + +def _resolve_vocab_inputs(args: argparse.Namespace) -> dict[str, Path | None] | None: + """Returns {atom, bond, smiles}->Path|None when the user supplied vocab + inputs (via --vocab-dir or --atom-vocab/--bond-vocab/--smiles-vocab), + else None (signal to fall through to build_vocab). + + Conventional filenames inside --vocab-dir: + pretrain_atom_vocab.{json,pkl} + pretrain_bond_vocab.{json,pkl} + pretrain_smiles_vocab.pkl + """ + if args.vocab_dir: + d = Path(args.vocab_dir).resolve() + if not d.is_dir(): + raise FileNotFoundError(f"--vocab-dir not found or not a directory: {d}") + def _find(stem: str, exts: tuple[str, ...]) -> Path | None: + for ext in exts: + p = d / f"{stem}.{ext}" + if p.is_file(): + return p + return None + atom = _find(PRETRAIN_VOCAB_STEMS["atom"], ("json", "pkl")) + bond = _find(PRETRAIN_VOCAB_STEMS["bond"], ("json", "pkl")) + smiles = _find(PRETRAIN_VOCAB_STEMS["smiles"], ("pkl",)) + if atom is None and bond is None and smiles is None: + stems = [PRETRAIN_VOCAB_STEMS[k] for k in ("atom", "bond", "smiles")] + raise FileNotFoundError( + f"--vocab-dir {d} contained no {{ {', '.join(stems) }}}.{{json,pkl}} " + f"files. Expected at least {PRETRAIN_VOCAB_STEMS['atom']} + " + f"{PRETRAIN_VOCAB_STEMS['bond']}." + ) + return {"atom": atom, "bond": bond, "smiles": smiles} + + if args.atom_vocab or args.bond_vocab or args.smiles_vocab: + return { + "atom": Path(args.atom_vocab).resolve() if args.atom_vocab else None, + "bond": Path(args.bond_vocab).resolve() if args.bond_vocab else None, + "smiles": Path(args.smiles_vocab).resolve() if args.smiles_vocab else None, + } + + return None + + +def _copy_provided_vocab( + src: dict[str, Path | None], dst_dir: Path, dataset_name: str, manifest: dict[str, Any], + force: bool, +) -> dict[str, Path]: + """When the user supplies vocab files (use ckpt's vocab as-is), + copy them into `/__vocab.` so the + downstream pretrain command sees the conventional filenames. + + `src` is `{atom: Path|None, bond: Path|None, smiles: Path|None}`. The atom + and bond entries must be both present or both absent (paired). smiles is + optional (cmim/hybrid only). + + Returns the same dict of (resolved) destination paths. + """ + import shutil + if (src["atom"] is None) != (src["bond"] is None): + raise ValueError( + "vocab pass-through requires atom and bond vocab paths to be paired; " + "got atom=" + str(src["atom"]) + ", bond=" + str(src["bond"]) + ) + out: dict[str, Path] = {} + dst_dir.mkdir(parents=True, exist_ok=True) + for which, path in src.items(): + if path is None: + continue + # Validate the source file IS a loadable KERMT vocab before copying. + # Catches the "user pointed --smiles-vocab at a random pickle" case + # early, with a clear error, instead of letting it surface as a cryptic + # SMILESVocab.load_vocab failure at pretrain_ddp.py launch time. + validate_vocab_file(path, kind=which) + ext = path.suffix.lstrip(".") + if which == "smiles": + ext = "pkl" # smiles vocab is always pickle + dst = dst_dir / f"{dataset_name}_{which}_vocab.{ext}" + if not force and _exists_nonempty(dst): + _skipped(f"copy_vocab({which})", str(dst), manifest) + out[which] = dst + continue + if force and dst.exists(): + dst.unlink() + shutil.copy2(path, dst) + manifest["steps"].append({ + "name": f"copy_vocab({which})", + "src": str(path), "dst": str(dst), "ok": True, + "duration_s": 0.0, "skipped_due_to_existing": False, + }) + out[which] = dst + return out + + +def _build_vocab( + csv_path: Path, vocab_dir: Path, dataset_name: str, vocab_format: str, + manifest: dict[str, Any], force: bool, +) -> dict[str, Path]: + """Builds atom + bond (in --vocab-format) and smiles (always pickle) vocabs. + Returns a dict of {atom, bond, smiles} -> Path.""" + suffix = "json" if vocab_format == "json" else "pkl" + expected = { + "atom": vocab_dir / f"{dataset_name}_atom_vocab.{suffix}", + "bond": vocab_dir / f"{dataset_name}_bond_vocab.{suffix}", + "smiles": vocab_dir / f"{dataset_name}_smiles_vocab.pkl", + } + if not force and all(_exists_nonempty(p) for p in expected.values()): + _skipped(f"build_vocab({csv_path.name})", str(vocab_dir), manifest) + return expected + vocab_dir.mkdir(parents=True, exist_ok=True) + if force: + for p in expected.values(): + if p.exists(): + p.unlink() + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "build_vocab.py"), + "--data_path", str(csv_path), + "--vocab_save_folder", str(vocab_dir), + "--dataset_name", dataset_name, + "--vocab_format", vocab_format, + ] + _run(cmd, f"build_vocab({csv_path.name})", manifest) + return expected + + +def _split_data( + csv_path: Path, features_path: Path | None, sample_per_file: int, output_dir: Path, + manifest: dict[str, Any], force: bool, +) -> Path: + """Run split_data.py to produce shard dirs (graph/ + optionally feature/ + summary.txt).""" + summary = output_dir / "summary.txt" + if not force and _exists_nonempty(summary): + _skipped(f"split_data({csv_path.name})", str(output_dir), manifest) + return output_dir + if force and output_dir.exists(): + shutil.rmtree(output_dir) + output_dir.mkdir(parents=True, exist_ok=True) + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "split_data.py"), + "--data_path", str(csv_path), + "--sample_per_file", str(sample_per_file), + "--output_path", str(output_dir), + ] + if features_path is not None: + cmd += ["--features_path", str(features_path)] + _run(cmd, f"split_data({csv_path.name})", manifest) + return output_dir + + +# --------------------------------------------------------------------------- +# Random splitter (used only when the user supplies a single CSV) +# --------------------------------------------------------------------------- + +def _random_split_csv( + src_csv: Path, dst_csvs: dict[str, Path], fractions: dict[str, float], seed: int, + manifest: dict[str, Any], force: bool, +) -> None: + """Shuffle src_csv and partition rows into dst_csvs by fractions. + `dst_csvs` and `fractions` are dicts keyed by the split name (e.g. 'train', 'val'). + Sum of fractions must be 1.0 (within float tolerance). Writes each dst_csv with the + same header as the input.""" + step = { + "name": f"random_split({src_csv.name})", + "seed": seed, + "fractions": fractions, + "ok": False, + "duration_s": None, + "skipped_due_to_existing": False, + "row_counts": {}, + } + if not force and all(_exists_nonempty(p) for p in dst_csvs.values()): + step["skipped_due_to_existing"] = True + step["ok"] = True + manifest["steps"].append(step) + return + + if abs(sum(fractions.values()) - 1.0) > 1e-6: + raise ValueError(f"split fractions must sum to 1.0 (got {sum(fractions.values())})") + + t0 = time.time() + df = pd.read_csv(src_csv).sample(frac=1.0, random_state=seed).reset_index(drop=True) + n = len(df) + sizes: dict[str, int] = {} + remaining = n + split_names = list(fractions.keys()) + for name in split_names[:-1]: + sizes[name] = int(round(fractions[name] * n)) + remaining -= sizes[name] + sizes[split_names[-1]] = remaining + + start = 0 + for name in split_names: + dst = dst_csvs[name] + dst.parent.mkdir(parents=True, exist_ok=True) + df.iloc[start:start + sizes[name]].to_csv(dst, index=False) + step["row_counts"][name] = sizes[name] + start += sizes[name] + + step["duration_s"] = round(time.time() - t0, 2) + step["ok"] = True + manifest["steps"].append(step) + + +def _emit_random_split_warning( + src_csv: Path, fractions: dict[str, float], seed: int, manifest: dict[str, Any] +) -> None: + row_counts = manifest["steps"][-1].get("row_counts", {}) + n = sum(row_counts.values()) if row_counts else "?" + lines = [ + f"WARNING: Auto-splitting {n} rows from {src_csv.name} into:", + ] + for name, frac in fractions.items(): + cnt = row_counts.get(name, "?") + lines.append(f" {name}: {cnt} rows ({frac * 100:.1f}%)") + lines += [ + f"using random split with seed {seed}.", + "", + "This is a RANDOM split. For rigorous ADMET evaluation, scaffold-balanced", + "(or other structure-aware) splits are strongly preferred — molecules with", + "similar scaffolds can leak across splits and inflate apparent generalization.", + "", + "To use your own pre-computed splits instead, pass:", + " --train-csv --val-csv --test-csv ", + "", + "To customize fractions:", + " --val-frac 0.15 --test-frac 0.15", + ] + warning = "\n".join(lines) + print(warning, file=sys.stderr) + manifest["warnings"].append(warning) + + +# --------------------------------------------------------------------------- +# Mode pipelines +# --------------------------------------------------------------------------- + +def _prepare_embed(args, out: Path, manifest: dict[str, Any]) -> None: + manifest["split_method"] = "n/a" + if args.skip_clean: + clean = Path(args.csv) + manifest["steps"].append({"name": "clean_smiles", "skipped_by_flag": True, "ok": True}) + else: + clean = _clean_smiles(Path(args.csv), out / "clean.csv", args.smiles_column, manifest, args.force) + manifest["outputs"]["clean_csv"] = str(clean) + + +def _prepare_inference(args, out: Path, manifest: dict[str, Any]) -> None: + manifest["split_method"] = "n/a" + clean = _clean_smiles(Path(args.csv), out / "clean.csv", args.smiles_column, manifest, args.force) + # Reduce to SMILES-only: downstream get_data/MoleculeDatapoint floats every + # non-SMILES column, which crashes on non-numeric passthrough columns + # (e.g. a 'split' label). Inference does not need target columns. + _reduce_to_smiles_column(clean, args.smiles_column, manifest) + manifest["outputs"]["clean_csv"] = str(clean) + if args.skip_features: + manifest["steps"].append({"name": "save_features", "skipped_by_flag": True, "ok": True}) + return + generator = args.features_generator or DEFAULT_FEATURES_GENERATOR["inference"] + npz = _save_features(clean, out / "clean.npz", generator, manifest, args.force) + manifest["outputs"]["clean_npz"] = str(npz) + + +def _prepare_finetune(args, out: Path, manifest: dict[str, Any]) -> None: + src_train = Path(args.csv) + has_val = args.val_csv is not None + has_test = args.test_csv is not None + split_type = args.split_type + + if has_val and has_test: + # User supplied explicit val + test CSVs: trust them, just clean + featurize. + # split_type is irrelevant when val/test are given separately. + manifest["split_method"] = "user_provided" + clean_train = _clean_smiles(src_train, out / "clean_train.csv", args.smiles_column, manifest, args.force) + clean_val = _clean_smiles(Path(args.val_csv), out / "clean_val.csv", args.smiles_column, manifest, args.force) + clean_test = _clean_smiles(Path(args.test_csv), out / "clean_test.csv", args.smiles_column, manifest, args.force) + manifest["outputs"]["clean_train_csv"] = str(clean_train) + manifest["outputs"]["clean_val_csv"] = str(clean_val) + manifest["outputs"]["clean_test_csv"] = str(clean_test) + per_split = (("train", clean_train), ("val", clean_val), ("test", clean_test)) + elif has_val or has_test: + raise ValueError( + "for finetune mode, either provide BOTH --val-csv and --test-csv (user-provided splits) " + "or NEITHER (run with --split-type {random|scaffold_balanced|index_predetermined}). " + "Got one but not both." + ) + elif split_type == "random": + # Random auto-split — done here in prep so train.py gets ready-made CSVs. + manifest["split_method"] = "random" + manifest["split_seed"] = args.seed + train_frac = max(0.0, 1.0 - args.val_frac - args.test_frac) + manifest["split_fractions"] = {"train": train_frac, "val": args.val_frac, "test": args.test_frac} + clean_full = _clean_smiles(src_train, out / "_clean_full.csv", args.smiles_column, manifest, args.force) + dst = { + "train": out / "clean_train.csv", + "val": out / "clean_val.csv", + "test": out / "clean_test.csv", + } + _random_split_csv(clean_full, dst, manifest["split_fractions"], args.seed, manifest, args.force) + clean_train, clean_val, clean_test = dst["train"], dst["val"], dst["test"] + manifest["outputs"]["clean_train_csv"] = str(clean_train) + manifest["outputs"]["clean_val_csv"] = str(clean_val) + manifest["outputs"]["clean_test_csv"] = str(clean_test) + _emit_random_split_warning(src_train, manifest["split_fractions"], args.seed, manifest) + per_split = (("train", clean_train), ("val", clean_val), ("test", clean_test)) + else: + # Scaffold-balanced or index-predetermined: prep cleans + featurizes the full + # CSV and defers actual splitting to task/train.py, which calls split_data + # with the user-supplied seed and split_sizes. + manifest["split_method"] = "deferred_to_runner" + manifest["split_type"] = split_type + manifest["split_seed"] = args.seed + manifest["split_fractions"] = { + "train": max(0.0, 1.0 - args.val_frac - args.test_frac), + "val": args.val_frac, + "test": args.test_frac, + } + clean_full = _clean_smiles(src_train, out / "clean_full.csv", args.smiles_column, manifest, args.force) + manifest["outputs"]["clean_full_csv"] = str(clean_full) + per_split = (("full", clean_full),) + + if args.skip_features: + manifest["steps"].append({"name": "save_features", "skipped_by_flag": True, "ok": True}) + return + generator = args.features_generator or DEFAULT_FEATURES_GENERATOR["finetune"] + for split_name, csv in per_split: + npz = _save_features(csv, csv.with_suffix(".npz"), generator, manifest, args.force) + manifest["outputs"][f"clean_{split_name}_npz"] = str(npz) + + +def _prepare_pretrain(args, out: Path, manifest: dict[str, Any]) -> None: + src_train = Path(args.csv) + if args.val_csv is not None: + manifest["split_method"] = "user_provided" + clean_train = _clean_smiles(src_train, out / "clean_train.csv", args.smiles_column, manifest, args.force) + clean_val = _clean_smiles(Path(args.val_csv), out / "clean_val.csv", args.smiles_column, manifest, args.force) + else: + manifest["split_method"] = "random" + manifest["split_seed"] = args.seed + train_frac = max(0.0, 1.0 - args.val_frac) + manifest["split_fractions"] = {"train": train_frac, "val": args.val_frac} + clean_full = _clean_smiles(src_train, out / "_clean_full.csv", args.smiles_column, manifest, args.force) + dst = {"train": out / "clean_train.csv", "val": out / "clean_val.csv"} + _random_split_csv(clean_full, dst, manifest["split_fractions"], args.seed, manifest, args.force) + clean_train, clean_val = dst["train"], dst["val"] + + manifest["outputs"]["clean_train_csv"] = str(clean_train) + manifest["outputs"]["clean_val_csv"] = str(clean_val) + + generator = args.features_generator or DEFAULT_FEATURES_GENERATOR["pretrain"] + if args.skip_features: + manifest["steps"].append({"name": "save_features", "skipped_by_flag": True, "ok": True}) + train_npz: Path | None = None + val_npz: Path | None = None + else: + train_npz = _save_features(clean_train, out / "clean_train.npz", generator, manifest, args.force) + val_npz = _save_features(clean_val, out / "clean_val.npz", generator, manifest, args.force) + manifest["outputs"]["clean_train_npz"] = str(train_npz) + manifest["outputs"]["clean_val_npz"] = str(val_npz) + + if args.skip_vocab: + manifest["steps"].append({"name": "build_vocab", "skipped_by_flag": True, "ok": True}) + manifest["vocab_source"] = "skipped" + else: + # Resolve user-provided vocab paths from --vocab-dir or explicit flags. + provided = _resolve_vocab_inputs(args) + if provided: + # Use the user-supplied (ckpt's) vocab as-is. Copy into the + # conventional filenames the downstream pretrain command expects. + vocabs = _copy_provided_vocab(provided, out, args.dataset_name, manifest, args.force) + manifest["vocab_source"] = "user_provided" + else: + # Fall back to the existing build-from-corpus behavior. Used by + # pretrain-from-scratch and by any continue case where the user + # explicitly wants a fresh vocab (rare, usually wrong). + vocabs = _build_vocab(clean_train, out, args.dataset_name, args.vocab_format, manifest, args.force) + manifest["vocab_source"] = "built_fresh" + if "atom" in vocabs: + manifest["outputs"]["atom_vocab"] = str(vocabs["atom"]) + if "bond" in vocabs: + manifest["outputs"]["bond_vocab"] = str(vocabs["bond"]) + if "smiles" in vocabs: + manifest["outputs"]["smiles_vocab"] = str(vocabs["smiles"]) + + if args.skip_split: + manifest["steps"].append({"name": "split_data", "skipped_by_flag": True, "ok": True}) + else: + train_dir = _split_data(clean_train, train_npz, args.sample_per_file, out / "train", manifest, args.force) + val_dir = _split_data(clean_val, val_npz, args.sample_per_file, out / "val", manifest, args.force) + manifest["outputs"]["train_dir"] = str(train_dir) + manifest["outputs"]["val_dir"] = str(val_dir) + + +# --------------------------------------------------------------------------- +# Entry point +# --------------------------------------------------------------------------- + +def prepare(args: argparse.Namespace) -> dict[str, Any]: + out = Path(args.out).resolve() + out.mkdir(parents=True, exist_ok=True) + manifest: dict[str, Any] = { + "mode": args.mode, + "input_csv": str(Path(args.csv).resolve()), + "val_csv": str(Path(args.val_csv).resolve()) if args.val_csv else None, + "test_csv": str(Path(args.test_csv).resolve()) if args.test_csv else None, + "output_dir": str(out), + "split_method": None, + "steps": [], + "outputs": {}, + "errors": [], + "warnings": [], + } + try: + if args.mode == "pretrain": + _prepare_pretrain(args, out, manifest) + elif args.mode == "finetune": + _prepare_finetune(args, out, manifest) + elif args.mode == "inference": + _prepare_inference(args, out, manifest) + elif args.mode == "embed": + _prepare_embed(args, out, manifest) + manifest["ok"] = True + except Exception as exc: # noqa: BLE001 + manifest["ok"] = False + manifest["errors"].append(f"{type(exc).__name__}: {exc}") + # Always write the manifest so partial-failure state is visible to the agent. + (out / "prepare_data.json").write_text(json.dumps(manifest, indent=2)) + return manifest + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Mode-dispatched data prep for the KERMT agent skills.") + p.add_argument("--mode", required=True, choices=VALID_MODES) + p.add_argument("--csv", required=True, help="Primary input CSV (train CSV for pretrain/finetune)") + p.add_argument("--out", required=True, help="Output directory") + p.add_argument("--val-csv", default=None, help="Optional separate val CSV (pretrain/finetune)") + p.add_argument("--test-csv", default=None, help="Optional separate test CSV (finetune only)") + p.add_argument("--val-frac", type=float, default=0.1, help="Auto-split val fraction (default 0.1)") + p.add_argument("--test-frac", type=float, default=0.1, help="Auto-split test fraction (finetune only, default 0.1)") + p.add_argument("--seed", type=int, default=0, help="Random split seed (default 0)") + p.add_argument("--split-type", choices=["random", "scaffold_balanced", "index_predetermined"], + default="random", + help="(finetune only, when --val-csv/--test-csv are not given) how to split. " + "'random' splits in prep using --val-frac/--test-frac/--seed. " + "'scaffold_balanced' and 'index_predetermined' defer the actual split to the " + "runner (task/train.py invokes split_data with the appropriate algorithm " + "using the user-supplied seed); prep only cleans + featurizes the full CSV.") + p.add_argument("--sample-per-file", type=int, default=100_000, + help="split_data shard size (pretrain only, default 100000)") + p.add_argument("--vocab-format", choices=["json", "pkl"], default="json", + help="atom/bond vocab format (default json); smiles vocab is always pkl") + # Vocab pass-through (pretrain mode): when continuing from a released ckpt, + # pass its bundled vocab files in so we don't rebuild a mismatched vocab. + p.add_argument("--vocab-dir", default=None, + help="(pretrain) directory containing pretrain_{atom,bond}_vocab.{json,pkl} " + "(+ pretrain_smiles_vocab.pkl for cmim/hybrid). When given, prepare_data " + "skips build_vocab and copies these files into the output dir under the " + "expected filenames. Used by kermt-continue-pretrain to bind the released " + "ckpt's vocab to the new corpus (the ckpt's vocab is authoritative).") + p.add_argument("--atom-vocab", default=None, + help="(pretrain) explicit atom vocab path; pairs with --bond-vocab. Overrides " + "--vocab-dir's pretrain_atom_vocab.* discovery if both are given.") + p.add_argument("--bond-vocab", default=None, + help="(pretrain) explicit bond vocab path; pairs with --atom-vocab.") + p.add_argument("--smiles-vocab", default=None, + help="(pretrain, cmim/hybrid) explicit smiles vocab .pkl path. Optional for " + "vocab-only pretrain.") + p.add_argument("--dataset-name", default="pretrain", + help="vocab filename prefix (default 'pretrain' so downstream pretrain commands " + "can reference pretrain_{atom,bond}_vocab.{json|pkl}, pretrain_smiles_vocab.pkl)") + p.add_argument("--targets", nargs="+", default=None, + help="(finetune only) target column names; forwarded to the finetune runner via the manifest") + p.add_argument("--features-generator", default=None, + help="Override the per-mode default (pretrain: fgtasklabel; finetune/inference: rdkit_2d_normalized)") + p.add_argument("--smiles-column", type=int, default=None, + help="0-based column index of SMILES in the input CSV. " + "When omitted, auto-detected by header name " + "(prefers lowercase `smiles`; accepts case-insensitive " + "`SMILES`/`Smiles`). Pass explicitly to override.") + p.add_argument("--force", action="store_true", + help="Re-run every step even if its outputs already exist") + p.add_argument("--skip-clean", action="store_true", help="(embed mode) skip the cleaning step") + p.add_argument("--skip-features", action="store_true", help="Skip feature generation") + p.add_argument("--skip-vocab", action="store_true", help="(pretrain) skip vocab build") + p.add_argument("--skip-split", action="store_true", help="(pretrain) skip shard split") + args = p.parse_args(argv) + + # Forward --targets through the manifest so the finetune runner can see them. + if args.mode == "finetune" and args.targets: + pass # captured in manifest below + + # Resolve the SMILES column index (auto-detect from header when the user + # didn't pass --smiles-column). This is the only point where args.csv is + # touched before downstream _clean_smiles calls fan it out. + try: + resolved_smiles_col = _resolve_smiles_column(Path(args.csv), args.smiles_column) + except ValueError as exc: + err_manifest = { + "ok": False, + "mode": args.mode, + "errors": [f"smiles-column resolution failed: {exc}"], + } + Path(args.out).mkdir(parents=True, exist_ok=True) + (Path(args.out) / "prepare_data.json").write_text(json.dumps(err_manifest, indent=2)) + print(json.dumps(err_manifest, indent=2)) + return 1 + if args.smiles_column is None: + print(f"[prepare_data] auto-detected --smiles-column {resolved_smiles_col} " + f"from {Path(args.csv).name} header", file=sys.stderr) + args.smiles_column = resolved_smiles_col + + try: + manifest = prepare(args) + except Exception as exc: # noqa: BLE001 + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({"ok": False, "errors": [f"unhandled: {type(exc).__name__}: {exc}"]}, indent=2)) + return 1 + if args.targets: + manifest["targets"] = list(args.targets) + # Record the resolved SMILES column so the manifest is self-describing. + manifest["smiles_column"] = args.smiles_column + (Path(args.out) / "prepare_data.json").write_text(json.dumps(manifest, indent=2)) + print(json.dumps(manifest, indent=2)) + return 0 if manifest.get("ok") else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-continue-pretrain/scripts/run_pretrain_local.py b/skills/kermt-continue-pretrain/scripts/run_pretrain_local.py new file mode 100644 index 0000000..5b8a8a1 --- /dev/null +++ b/skills/kermt-continue-pretrain/scripts/run_pretrain_local.py @@ -0,0 +1,730 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Workstation pretrain runner — composes prepare_data + ckpt-validator outputs +into a pretrain_ddp.py invocation. + +Continues pretraining from a user-provided checkpoint. The model type +(grover_base / cmim / hybrid) is inferred from the validator's output and +drives the pretrain_ddp.py flag set; arch params come exclusively from the +ckpt; training/loss hyperparameters come from config/defaults_pretrain.json +with per-flag CLI overrides. + +How it interacts with pretrain_ddp.py's auto-resume: + pretrain_ddp.py looks at /last_checkpoint.pt and resumes from it + if present. The runner sets `--save_dir /ckpt` and symlinks the user's + input ckpt to /ckpt/last_checkpoint.pt so the resume path picks it up. + +Run.json manifest: + Records source-repo commit + image digest + a copy-pasteable `cmd_replay` + + per-flag `args_applied` so the artifact is self-contained and replayable. + +CLI +--- + run_pretrain_local.py + --ckpt # input pretrain ckpt (required) + --prepare-manifest # prepare_data.json from a prior prepare run + --out # output dir (typically runs/continue-pretrain_/) + [--ckpt-validator-out ] # cached check_checkpoint.py JSON; computed if absent + [--gpus 0,2] # subset of detected GPUs; default = all visible + [--dry-run] # write run.json + print command, do not execute + [--epochs N] [--batch-size N] [--init-lr F] [--max-lr F] [--final-lr F] + [--warmup-epochs F] [--weight-decay F] [--dropout F] + [--save-interval N] [--seed N] + [--vocab-loss-weight F] # hybrid only + [--latent-dim N] [--contrastive-temperature F] # cmim/hybrid only +""" +from __future__ import annotations + +import argparse +import datetime +import json +import os +import subprocess +import sys +from pathlib import Path +from typing import Any + +# Add the scripts/ dir to sys.path so `_utils` is importable whether +# this script is launched via `kermt_run` (PYTHONPATH=/workspace) or as a +# bare `python scripts/run_pretrain_local.py …` from the host. +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import ( # noqa: E402 + resolve_kermt_repo, assert_prepare_manifest_basics, count_vocab_entries, docker_image_digest, + format_cmd_replay, git_commit_with_env_override, load_json, + merge_default_into_applied, run_checkpoint_validator, +) + + +REPO_ROOT = resolve_kermt_repo() +SKILL_ROOT = Path(__file__).resolve().parent.parent +DEFAULTS_PATH = SKILL_ROOT / "config" / "defaults_pretrain.json" +CHECK_CHECKPOINT_PATH = SKILL_ROOT / "scripts" / "check_checkpoint.py" +PRETRAIN_DDP_PATH = REPO_ROOT / "pretrain_ddp.py" + +# Model-type → pretrain_ddp.py `--pretrain_mode` value. +MODEL_TYPE_TO_PRETRAIN_MODE = { + "grover_base": "vocab", + "cmim": "cmim", + "hybrid": "hybrid", +} + +# Hyperparameter flags the runner exposes for CLI override + the corresponding +# key path in defaults_pretrain.json. None means the value isn't in defaults +# (e.g. seed has a default but lives at the top of training; lookup is direct). +TRAINING_FLAGS = ( + "batch_size", "dropout", "epochs", "init_lr", "max_lr", "final_lr", + "warmup_epochs", "weight_decay", "save_interval", "seed", "tensorboard", + "use_cuikmolmaker_featurization", +) +LOSS_FLAGS = ("contrastive_temperature", "vocab_loss_weight") +DECODER_FLAGS = ( + "latent_dim", + "decoder_num_layers", + "decoder_num_attention_heads", + "decoder_ffn_hidden_size", + "decoder_dropout", + "decoder_max_seq_len", + "decoder_positional_encoding", + "decoder_gate_self_attn", + "decoder_gate_cross_attn", +) + +ARCH_FLAGS_FROM_CKPT = ( + "hidden_size", "depth", "num_attn_head", "activation", "backbone", + "embedding_output_type", "self_attention", +) + +# cMIM-decoder + latent-distribution arch fields. For continue-pretrain on a +# cmim/hybrid ckpt these MUST come from the ckpt's saved_args (so the model +# being constructed matches the ckpt's weights at load time); the +# defaults_pretrain.json `add_cmim_decoder` block is for add-cmim-pretrain's +# upgrade-time decoder construction only, and is intentionally ignored +# during continue-pretrain. +CMIM_DECODER_FLAGS_FROM_CKPT = ( + "latent_dim", + "decoder_num_layers", + "decoder_num_attention_heads", + "decoder_ffn_hidden_size", + "decoder_dropout", + "decoder_max_seq_len", + "decoder_positional_encoding", + "decoder_gate_self_attn", + "decoder_gate_cross_attn", +) + + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + +# JSON loading delegated to the shared _utils.load_json. Alias kept for the +# existing internal callsites that use the leading-underscore convention. +_load_json = load_json + + +def _detect_gpus(override: str | None) -> tuple[int, str]: + """Returns (world_size, CUDA_VISIBLE_DEVICES_string).""" + if override: + gpu_list = [g.strip() for g in override.split(",") if g.strip()] + return len(gpu_list), ",".join(gpu_list) + # Honor an existing CUDA_VISIBLE_DEVICES in the environment. + env = os.environ.get("CUDA_VISIBLE_DEVICES", "").strip() + if env: + ids = [g for g in env.split(",") if g] + return len(ids), ",".join(ids) + try: + import torch + n = torch.cuda.device_count() + except Exception: + n = 0 + return n, ",".join(str(i) for i in range(n)) + + +def _verify_prepare_manifest(manifest: dict[str, Any]) -> None: + assert_prepare_manifest_basics(manifest, "pretrain") + out = manifest.get("outputs", {}) + required_keys = ("train_dir", "val_dir", "atom_vocab", "bond_vocab") + missing = [k for k in required_keys if k not in out] + if missing: + raise ValueError( + f"prepare_data manifest is missing required outputs: {missing}. " + "Was prepare_data.py invoked with --skip-vocab or --skip-split?" + ) + + +def _apply_defaults(args: argparse.Namespace, defaults: dict[str, Any], + model_type: str, world_size: int) -> dict[str, dict[str, Any]]: + """Returns args_applied: dict mapping flag → {value, source}. + Source is 'user' if the user passed a value on the CLI, else 'default-config' + (from defaults_pretrain.json) or 'auto-1gpu' / 'auto-multi-gpu' for the + auto-fallback values. Only includes flags relevant to the model_type.""" + applied: dict[str, dict[str, Any]] = {} + + training_defaults = defaults.get("training", {}) + loss_defaults = defaults.get("loss", {}) + decoder_defaults = defaults.get("add_cmim_decoder", {}) + + for f in TRAINING_FLAGS: + merge_default_into_applied(applied, args, f, training_defaults) + + # Single-GPU fallback: batch_size 32, save_interval 500. + if world_size <= 1: + if applied.get("batch_size", {}).get("source") != "user": + applied["batch_size"] = {"value": 32, "source": "auto-1gpu"} + if applied.get("save_interval", {}).get("source") != "user": + applied["save_interval"] = {"value": 500, "source": "auto-1gpu"} + + if model_type in ("cmim", "hybrid"): + for f in LOSS_FLAGS if model_type == "hybrid" else ("contrastive_temperature",): + merge_default_into_applied(applied, args, f, loss_defaults) + for f in DECODER_FLAGS: + merge_default_into_applied(applied, args, f, decoder_defaults) + + return applied + + +def _arch_from_validator(validator_out: dict[str, Any]) -> dict[str, Any]: + arch = validator_out.get("arch") or {} + missing = [k for k in ARCH_FLAGS_FROM_CKPT if arch.get(k) is None] + if missing: + raise ValueError( + f"checkpoint validator did not surface required arch fields: {missing}. " + "If the ckpt has no saved_args blob, these can't be inferred from state-dict " + "shapes alone; please supply a ckpt with args saved (the standard " + "save_model_for_restart format)." + ) + return arch + + +def _build_argv( + *, world_size: int, gpus_str: str, out_dir: Path, manifest: dict[str, Any], + model_type: str, pretrain_mode: str, arch: dict[str, Any], + applied: dict[str, dict[str, Any]], +) -> list[str]: + """Constructs the full pretrain_ddp.py argument list as a list of strings.""" + outputs = manifest["outputs"] + argv = [sys.executable, "-u", str(PRETRAIN_DDP_PATH)] + + # Data + vocab paths + argv += ["--train_data_path", outputs["train_dir"], + "--val_data_path", outputs["val_dir"], + "--atom_vocab_path", outputs["atom_vocab"], + "--bond_vocab_path", outputs["bond_vocab"]] + if model_type in ("cmim", "hybrid"): + argv += ["--smiles_vocab_path", outputs["smiles_vocab"]] + + # Pretrain mode + loss + argv += ["--pretrain_mode", pretrain_mode] + if "vocab_loss_weight" in applied and model_type == "hybrid": + argv += ["--vocab_loss_weight", str(applied["vocab_loss_weight"]["value"])] + if "contrastive_temperature" in applied and model_type in ("cmim", "hybrid"): + argv += ["--contrastive_temperature", str(applied["contrastive_temperature"]["value"])] + # cMIM/decoder arch: emit every applied flag. For continue-pretrain on a + # cmim/hybrid ckpt, every entry will be source="ckpt_saved_args" (see the + # overlay loop in run()). For pretrain-from-scratch / add-cmim-pretrain + # the values come from defaults_pretrain.json's add_cmim_decoder block. + if model_type in ("cmim", "hybrid"): + for f in ("latent_dim", "decoder_num_layers", "decoder_num_attention_heads", + "decoder_ffn_hidden_size", "decoder_dropout", + "decoder_max_seq_len", "decoder_positional_encoding"): + if f in applied: + argv += [f"--{f}", str(applied[f]["value"])] + # Boolean store_true flags: emit the bare flag only when True. + if applied.get("decoder_gate_self_attn", {}).get("value"): + argv += ["--decoder_gate_self_attn"] + if applied.get("decoder_gate_cross_attn", {}).get("value"): + argv += ["--decoder_gate_cross_attn"] + + # Architecture — sourced from validator's arch block, never from CLI/defaults. + argv += [ + "--hidden_size", str(arch["hidden_size"]), + "--depth", str(arch["depth"]), + "--num_attn_head", str(arch["num_attn_head"]), + "--activation", str(arch["activation"]), + "--backbone", str(arch["backbone"]), + "--embedding_output_type", str(arch["embedding_output_type"]), + ] + if arch.get("self_attention"): + argv += ["--self_attention"] + + # Training schedule + for name in ("batch_size", "dropout", "epochs", "init_lr", "max_lr", "final_lr", + "warmup_epochs", "weight_decay", "save_interval", "seed"): + if name in applied: + argv += [f"--{name}", str(applied[name]["value"])] + if applied.get("tensorboard", {}).get("value"): + argv += ["--tensorboard"] + if applied.get("use_cuikmolmaker_featurization", {}).get("value"): + argv += ["--use_cuikmolmaker_featurization"] + + # W&B logging (pass-through; pretrain_ddp.py only inits W&B when project is set). + if "wandb_project" in applied: + argv += ["--wandb_project", str(applied["wandb_project"]["value"])] + if "wandb_run_name" in applied: + argv += ["--wandb_run_name", str(applied["wandb_run_name"]["value"])] + + # Where pretrain_ddp.py auto-resumes from (we'll symlink the user ckpt there). + argv += ["--save_dir", str(out_dir / "ckpt")] + + return argv + + +def _symlink_ckpt_into_save_dir(user_ckpt: Path, save_dir: Path) -> Path: + """--resume path: symlink the user ckpt as /last_checkpoint.pt. + pretrain_ddp.py's auto-resume then restores everything from the ckpt: + model weights, optimizer state, scheduler_step, epoch, batch_idx, + wandb_run_id.""" + save_dir.mkdir(parents=True, exist_ok=True) + link = save_dir / "last_checkpoint.pt" + if link.exists() or link.is_symlink(): + link.unlink() + # Symlink to the absolute user_ckpt so it works regardless of cwd. + link.symlink_to(user_ckpt.resolve()) + return link + + +# Schedule fields that --resume inherits from ckpt.saved_args and that default +# (fresh-schedule) mode takes from CLI/defaults_pretrain.json. +SCHEDULE_FLAGS = ("epochs", "warmup_epochs", "init_lr", "max_lr", "final_lr") + + +def _materialize_ckpt_for_fresh_schedule(user_ckpt: Path, save_dir: Path) -> Path: + """Default (fresh-schedule) continue-pretrain path: write a CLEANED copy + of the user ckpt to /last_checkpoint.pt with scheduler_step, + epoch, batch_idx, and wandb_run_id reset to fresh-start values. Model + weights AND optimizer state pass through unchanged — so Adam's running + moments warm-start the new schedule (helpful because the new init_lr is + usually close to the previous run's final_lr). + + Why a fresh-state copy instead of a symlink: pretrain_ddp.py's + `trainer.load()` restores EVERYTHING in the ckpt including scheduler_step + and epoch. We can't selectively load just the model + optimizer through + that code path. The minimal-invasive workaround is to materialize a + ckpt that has the unwanted counters zeroed before the loader sees it. + pretrain_ddp.py then restores everything as normal, but everything it + restores reads as a fresh-start. + + Cost: one ~700 MB disk write per run. Pretrain is days-long, so it's + negligible. Done on the host before docker run. + """ + import torch # delayed import — keeps the runner light in --dry-run paths + save_dir.mkdir(parents=True, exist_ok=True) + target = save_dir / "last_checkpoint.pt" + if target.exists() or target.is_symlink(): + target.unlink() + ckpt = torch.load(user_ckpt, map_location="cpu", weights_only=False) + if not isinstance(ckpt, dict) or "state_dict" not in ckpt: + raise ValueError( + f"ckpt {user_ckpt} is not in the expected save_model_for_restart " + "dict format (need at least 'state_dict' key)." + ) + ckpt["scheduler_step"] = 0 + ckpt["epoch"] = 0 + ckpt["batch_idx"] = 0 + ckpt["wandb_run_id"] = None + torch.save(ckpt, target) + return target + + +def _validate_resume_state(user_ckpt: Path) -> dict[str, Any]: + """--resume mode: confirm the ckpt was saved via the save_model_for_restart + format and carries the full state pretrain_ddp.py needs to resume mid-run + (optimizer state, scheduler_step, epoch, batch_idx). Returns a small + `resume_state` dict for the manifest so users can see what was restored. + Raises ValueError with a clear redirect if the ckpt is too lean.""" + import torch + ckpt = torch.load(user_ckpt, map_location="cpu", weights_only=False) + if not isinstance(ckpt, dict) or "state_dict" not in ckpt: + raise ValueError( + f"ckpt {user_ckpt} is not in the expected save_model_for_restart " + "dict format." + ) + required = ("optimizer", "scheduler_step", "epoch", "batch_idx") + missing = [k for k in required if k not in ckpt] + if missing: + raise ValueError( + f"--resume requires the ckpt to carry the full mid-run state, but " + f"these keys are missing: {missing}. The ckpt was probably saved " + "without enough metadata to pure-resume — use the default " + "fresh-schedule mode (drop --resume) if you just want to continue " + "training with a new schedule." + ) + return { + "scheduler_step": int(ckpt["scheduler_step"]), + "epoch": int(ckpt["epoch"]), + "batch_idx": int(ckpt["batch_idx"]), + "wandb_run_id": ckpt.get("wandb_run_id"), + } + + +# Vocab-entry counting delegated to _utils.count_vocab_entries. Alias kept for +# the existing internal callsites. +_count_vocab_entries = count_vocab_entries + + +def _verify_vocab_sizes_match_ckpt( + manifest: dict[str, Any], validator_out: dict[str, Any], model_type: str, +) -> dict[str, Any]: + """For continue-pretrain only: compare each vocab file's entry count against + the ckpt's vocab head dimensions. Aborts on mismatch with a helpful error + pointing the user at the matching vocab. Returns a `vocab_check` block to + attach to run.json for transparency.""" + ckpt_sizes = validator_out.get("vocab_sizes") or {"atom": None, "bond": None, "smiles": None} + outputs = manifest.get("outputs", {}) + check: dict[str, Any] = {"vocab_source": manifest.get("vocab_source", "unknown")} + for which in ("atom", "bond", "smiles"): + ckpt_size = ckpt_sizes.get(which) + vocab_path_str = outputs.get(f"{which}_vocab") + check[which] = {"ckpt_size": ckpt_size, "manifest_vocab": vocab_path_str, "manifest_size": None} + if ckpt_size is None: + # ckpt doesn't have this head; nothing to verify. + continue + # ckpt has this head — the manifest MUST include the corresponding vocab. + if not vocab_path_str: + raise ValueError( + f"ckpt has a '{which}' vocab head (size {ckpt_size}) but the prepare_data " + f"manifest doesn't include a {which}_vocab file. Rerun prepare_data with " + f"--vocab-dir (or --{which}-vocab ) so the runner " + f"can pass the matching vocab through." + ) + manifest_size = _count_vocab_entries(Path(vocab_path_str)) + check[which]["manifest_size"] = manifest_size + if manifest_size != ckpt_size: + raise ValueError( + f"{which} vocab size mismatch — ckpt's head expects {ckpt_size} entries, " + f"but {vocab_path_str} has {manifest_size}. The released ckpt's vocab is the " + f"authoritative one for continue-pretrain; pass --vocab-dir " + f"(or --{which}-vocab ) to prepare_data so vocab built from the new corpus " + f"isn't used. If you actually want to pretrain from scratch on a different " + f"vocab, use the kermt-pretrain-scratch workflow instead." + ) + return check + + +# --------------------------------------------------------------------------- +# Main flow +# --------------------------------------------------------------------------- + +def run(args: argparse.Namespace) -> dict[str, Any]: + out_dir = Path(args.out).resolve() + out_dir.mkdir(parents=True, exist_ok=True) + (out_dir / "ckpt").mkdir(parents=True, exist_ok=True) + (out_dir / "logs").mkdir(parents=True, exist_ok=True) + + from_scratch = bool(args.from_scratch) + resume = bool(args.resume) + + # Mode-conflict validation up-front so the user fails fast. + if from_scratch and resume: + raise ValueError("--resume is incompatible with --from-scratch.") + if resume and not args.ckpt: + raise ValueError("--resume requires --ckpt; nothing to resume from otherwise.") + if resume: + # CLI overrides of schedule args are forbidden in --resume mode — pure + # resume means the schedule shape from the ckpt is authoritative. + cli_overrides = [ + f for f in SCHEDULE_FLAGS if getattr(args, f, None) is not None + ] + if cli_overrides: + raise ValueError( + f"--resume inherits schedule args from the ckpt's saved_args; " + f"explicit CLI override is forbidden. You passed: {cli_overrides}. " + "Drop those flags to pure-resume, or use the default fresh-schedule " + "mode (no --resume) if you want a new schedule." + ) + + workflow = "pretrain-scratch" if from_scratch else "continue-pretrain" + if resume: + mode = "continue_pretrain_resume" + elif from_scratch: + mode = "pretrain_from_scratch" + else: + mode = "continue_pretrain_fresh_schedule" + + # 1. Load defaults + prepare manifest. + defaults = _load_json(DEFAULTS_PATH, name="defaults_pretrain.json") + prep_manifest_path = Path(args.prepare_manifest).resolve() + manifest = _load_json(prep_manifest_path, name="prepare_data.json") + _verify_prepare_manifest(manifest) + + # 2. Branch: continue-pretrain (load ckpt + validate) vs from-scratch (no ckpt). + ckpt: Path | None = None + validator_out: dict[str, Any] | None = None + link: Path | None = None + vocab_check: dict[str, Any] | None = None + resume_state: dict[str, Any] | None = None # populated only when --resume + + if from_scratch: + if args.ckpt: + raise ValueError("--from-scratch is incompatible with --ckpt; pass one or the other.") + if not args.pretrain_target_mode: + raise ValueError("--pretrain-target-mode is required when --from-scratch is set " + "(choose vocab, cmim, or hybrid).") + pretrain_mode = args.pretrain_target_mode + model_type = {"vocab": "grover_base", "cmim": "cmim", "hybrid": "hybrid"}[pretrain_mode] + # Arch from defaults_pretrain.json's `arch` group (with CLI overrides applied later + # if we expose any; for now we just use defaults). + arch_defaults = defaults.get("arch") or {} + if not arch_defaults: + raise ValueError("defaults_pretrain.json has no `arch` group; cannot pretrain from scratch.") + arch = {k: arch_defaults.get(k) for k in ARCH_FLAGS_FROM_CKPT} + # `latent_dim` lives in the add_cmim_decoder group for from-scratch cmim/hybrid; + # treat it as part of the arch for argv-building purposes. + if pretrain_mode in ("cmim", "hybrid"): + arch["latent_dim"] = (defaults.get("add_cmim_decoder") or {}).get("latent_dim") + else: + arch["latent_dim"] = None + else: + if not args.ckpt: + raise ValueError("--ckpt is required for continue-pretrain. " + "Use --from-scratch to pretrain a fresh model on the corpus.") + ckpt = Path(args.ckpt).resolve() + if args.ckpt_validator_out: + validator_out = _load_json(Path(args.ckpt_validator_out), name="ckpt validator output") + else: + validator_out = run_checkpoint_validator(ckpt, mode="continue_pretrain", script_path=CHECK_CHECKPOINT_PATH) + if not validator_out.get("ok"): + raise ValueError( + f"check_checkpoint.py rejected the input ckpt: {validator_out.get('errors')}" + ) + model_type = validator_out.get("model_type") + if model_type not in MODEL_TYPE_TO_PRETRAIN_MODE: + raise ValueError( + f"model_type='{model_type}' cannot continue pretrain. " + f"Supported: {sorted(MODEL_TYPE_TO_PRETRAIN_MODE)}. " + "For an encoder-only ckpt with no pretrain head, use the " + "upgrade_to_hybrid workflow." + ) + if model_type == "grover_base" and not validator_out.get("has_vocab_head"): + raise ValueError( + "grover_base ckpt has no vocab head — cannot continue vocab pretrain. " + "Use the upgrade_to_hybrid workflow to add a cMIM decoder, " + "or finetune directly from the encoder." + ) + pretrain_mode = MODEL_TYPE_TO_PRETRAIN_MODE[model_type] + arch = _arch_from_validator(validator_out) + # Vocab-size verification — refuse mismatched corpora before launching pretrain_ddp.py. + vocab_check = _verify_vocab_sizes_match_ckpt(manifest, validator_out, model_type) + # --resume needs the ckpt to carry the full mid-run state. Validate now; + # also surface what's being restored in the manifest. + if resume: + resume_state = _validate_resume_state(ckpt) + + # 3. GPU selection. + world_size, gpus_str = _detect_gpus(args.gpus) + if world_size <= 0: + raise ValueError( + "No GPUs detected. pretrain_ddp.py requires at least one CUDA device. " + "Set CUDA_VISIBLE_DEVICES or pass --gpus ." + ) + + # 4. Apply defaults + collect args_applied. + applied = _apply_defaults(args, defaults, model_type, world_size) + # --resume overlays schedule args from the ckpt's saved_args (the only path + # where source="ckpt_saved_args" can appear in args_applied). Fail loudly if + # any schedule field is missing from saved_args — pure-resume can't proceed + # without the original schedule shape. + if resume: + saved_args = validator_out.get("saved_args") or {} + missing = [f for f in SCHEDULE_FLAGS if f not in saved_args] + if missing: + raise ValueError( + f"--resume requires the ckpt's saved_args to include all schedule " + f"fields, but these are missing: {missing}. The ckpt was saved " + "without enough metadata to pure-resume — use the default " + "fresh-schedule mode and specify --epochs / --warmup-epochs / " + "--init-lr / --max-lr / --final-lr explicitly." + ) + for f in SCHEDULE_FLAGS: + applied[f] = {"value": saved_args[f], "source": "ckpt_saved_args"} + + # Continue-pretrain on a cmim/hybrid ckpt: cMIM/decoder arch must come + # from the ckpt's saved_args, not from defaults or CLI. This is the + # cmim/decoder analogue of the encoder-arch passthrough already done by + # `_arch_from_validator` (and matches the README guarantee that + # `add_cmim_decoder` defaults are ignored during continue-pretrain). + if not from_scratch and model_type in ("cmim", "hybrid"): + cli_latent_dim_override = args.latent_dim is not None + if cli_latent_dim_override: + raise ValueError( + "--latent-dim cannot be overridden during continue-pretrain on a " + "cmim/hybrid ckpt — the value is fixed by the ckpt's saved_args " + "(passing a different value would mismatch the loaded decoder " + "weights). Drop --latent-dim, or use kermt-pretrain-scratch if " + "you intentionally want a different latent dimension." + ) + saved_args = validator_out.get("saved_args") or {} + cmim_missing = [f for f in CMIM_DECODER_FLAGS_FROM_CKPT if f not in saved_args] + if cmim_missing: + raise ValueError( + f"continue-pretrain on a {model_type} ckpt requires the ckpt's " + f"saved_args to include cmim/decoder arch fields, but these are " + f"missing: {cmim_missing}. The ckpt was saved without enough " + "metadata to faithfully reconstruct the decoder." + ) + for f in CMIM_DECODER_FLAGS_FROM_CKPT: + applied[f] = {"value": saved_args[f], "source": "ckpt_saved_args"} + + # Optional W&B logging: pass-through, no defaults — forwarded only when the + # user sets --wandb-project (run name is honored only alongside a project). + for f in ("wandb_project", "wandb_run_name"): + v = getattr(args, f, None) + if v is not None: + applied[f] = {"value": v, "source": "user"} + + # 5. Build the pretrain_ddp.py argv. + argv = _build_argv( + world_size=world_size, gpus_str=gpus_str, out_dir=out_dir, manifest=manifest, + model_type=model_type, pretrain_mode=pretrain_mode, arch=arch, applied=applied, + ) + + # 6. (continue-pretrain only) Stage the ckpt into /last_checkpoint.pt + # so pretrain_ddp.py's auto-resume picks it up. Mode-dispatched: + # - --resume: symlink to user ckpt. pretrain_ddp.py restores everything + # (model + optimizer + scheduler_step + epoch + batch_idx + wandb_run_id). + # - default (fresh-schedule): materialize a state-cleaned copy of the + # ckpt — model weights + optimizer pass through, but scheduler_step / + # epoch / batch_idx / wandb_run_id are reset to 0/None. pretrain_ddp.py + # then builds a fresh NoamLR from CLI args and starts from step 0. + # Done unconditionally (including --dry-run) so the dry-run faithfully + # exercises ckpt I/O — catches corrupt ckpts / insufficient disk before + # the days-long real run. + if not from_scratch: + if resume: + link = _symlink_ckpt_into_save_dir(ckpt, out_dir / "ckpt") + else: + link = _materialize_ckpt_for_fresh_schedule(ckpt, out_dir / "ckpt") + + # 7. Build the run.json manifest. + commit, dirty = git_commit_with_env_override(REPO_ROOT) + image_tag = os.environ.get("KERMT_IMAGE", "kermt:latest") + image_digest = docker_image_digest(image_tag) + cmd_replay_env: dict[str, str] = {} + if gpus_str: + cmd_replay_env["CUDA_VISIBLE_DEVICES"] = gpus_str + cmd_replay_env["WORLD_SIZE"] = str(world_size) + cmd_replay = format_cmd_replay(argv, env=cmd_replay_env) + run_manifest = { + "workflow": workflow, + "mode": mode, # pretrain_from_scratch | continue_pretrain_fresh_schedule | continue_pretrain_resume + "started_at": datetime.datetime.now(datetime.timezone.utc).isoformat(), + "container": {"image_tag": image_tag, "image_digest": image_digest}, + "repo": {"commit": commit, "dirty": dirty}, + "inputs": { + "ckpt": str(ckpt) if ckpt else None, + "prepare_data_manifest": str(prep_manifest_path), + "ckpt_validator_out": ( + str(Path(args.ckpt_validator_out).resolve()) if args.ckpt_validator_out else None + ), + }, + "model_type": model_type, + "pretrain_mode": pretrain_mode, + "world_size": world_size, + "cuda_visible_devices": gpus_str, + "args_applied": applied, + "arch": arch, + "vocab_check": vocab_check, # None for from-scratch + "resume_state": resume_state, # None unless --resume; carries the restored scheduler_step / epoch / batch_idx / wandb_run_id from the ckpt + "save_dir": str(out_dir / "ckpt"), + "logs_dir": str(out_dir / "logs"), + "tensorboard_dir": str(out_dir / "logs" / "tb"), + "argv": argv, + "cmd_replay": cmd_replay, + "ok_to_replay": (not dirty) and (commit != "unknown"), + "dry_run": bool(args.dry_run), + "ckpt_symlink": str(link) if link else None, + "from_scratch": from_scratch, + } + (out_dir / "run.json").write_text(json.dumps(run_manifest, indent=2)) + + # 8. Execute (unless --dry-run). + if args.dry_run: + run_manifest["status"] = "dry_run" + return run_manifest + + env = os.environ.copy() + env["WORLD_SIZE"] = str(world_size) + if gpus_str: + env["CUDA_VISIBLE_DEVICES"] = gpus_str + + log_file = out_dir / "logs" / "pretrain_ddp.log" + with log_file.open("w") as logf: + proc = subprocess.run(argv, env=env, stdout=logf, stderr=subprocess.STDOUT) + run_manifest["exit_code"] = proc.returncode + run_manifest["status"] = "ok" if proc.returncode == 0 else "failed" + (out_dir / "run.json").write_text(json.dumps(run_manifest, indent=2)) + return run_manifest + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser( + description="Workstation pretrain runner (continue-pretrain by default, " + "or pretrain-from-scratch with --from-scratch).") + p.add_argument("--ckpt", default=None, + help="Path to the input pretrain checkpoint. Required for continue-pretrain; " + "omit when --from-scratch is set.") + p.add_argument("--from-scratch", action="store_true", + help="Pretrain a fresh model on the corpus (no input ckpt; arch from " + "defaults_pretrain.json; vocab built by prepare_data). Requires " + "--pretrain-target-mode.") + p.add_argument("--resume", action="store_true", + help="Resume an interrupted pretrain run (crashed / Ctrl-C / OOM). " + "Restores everything from the ckpt: model weights, optimizer " + "state, scheduler_step, epoch, batch_idx, wandb_run_id. Schedule " + "shape (epochs / warmup_epochs / init/max/final_lr) is inherited " + "from the ckpt's saved_args; CLI overrides of schedule flags are " + "REJECTED in this mode. Without --resume (default), continue-pretrain " + "loads only model weights + optimizer momentum from the ckpt and " + "starts a fresh schedule from CLI/defaults_pretrain.json — use that " + "default mode when continue-pretraining on a new corpus / new " + "objective / extended training (the common case).") + p.add_argument("--pretrain-target-mode", choices=["vocab", "cmim", "hybrid"], default=None, + help="(--from-scratch only) which pretrain objective to use for the fresh " + "model: vocab (grover_base-style), cmim, or hybrid (vocab + contrast). " + "No default — must be set explicitly so the user makes an informed " + "choice about the head config.") + p.add_argument("--prepare-manifest", required=True, + help="Path to a prepare_data.json (must be mode=pretrain)") + p.add_argument("--out", required=True, help="Output run directory") + p.add_argument("--ckpt-validator-out", default=None, + help="Optional cached check_checkpoint.py JSON; computed if absent") + p.add_argument("--gpus", default=None, + help="Comma-separated GPU ids (e.g. '0,1'). Default: all visible") + p.add_argument("--dry-run", action="store_true", + help="Write run.json and print the command without executing") + # Training overrides — all default to None so we can distinguish user-given vs default-config. + for f, t in [("epochs", int), ("batch-size", int), ("init-lr", float), ("max-lr", float), + ("final-lr", float), ("warmup-epochs", float), ("weight-decay", float), + ("dropout", float), ("save-interval", int), ("seed", int), + ("vocab-loss-weight", float), ("latent-dim", int), + ("contrastive-temperature", float)]: + p.add_argument(f"--{f}", type=t, default=None) + # Optional W&B logging (pass-through to pretrain_ddp.py; off unless project is set). + p.add_argument("--wandb-project", type=str, default=None, + help="W&B project name. When set, pretrain_ddp.py logs train/val losses.") + p.add_argument("--wandb-run-name", type=str, default=None, + help="Optional W&B run name (only used when --wandb-project is set).") + args = p.parse_args(argv) + + try: + manifest = run(args) + except (FileNotFoundError, ValueError, RuntimeError) as exc: + print(json.dumps({"ok": False, "errors": [f"{type(exc).__name__}: {exc}"]}, indent=2), + file=sys.stdout) + return 1 + except Exception as exc: # noqa: BLE001 + import traceback + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({"ok": False, "errors": [f"unhandled: {type(exc).__name__}: {exc}"]}, + indent=2)) + return 1 + + print(json.dumps({"ok": True, "manifest": manifest}, indent=2)) + return 0 if manifest.get("status") != "failed" else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/agent/skills/kermt-continue-pretrain/skill-card.md b/skills/kermt-continue-pretrain/skill-card.md similarity index 98% rename from agent/skills/kermt-continue-pretrain/skill-card.md rename to skills/kermt-continue-pretrain/skill-card.md index fc659f4..03a161b 100644 --- a/agent/skills/kermt-continue-pretrain/skill-card.md +++ b/skills/kermt-continue-pretrain/skill-card.md @@ -44,7 +44,7 @@ Mitigation: W&B tracking is optional and off unless the user supplies `WANDB_API ## Reference(s):
- [KERMT repository](https://github.com/NVIDIA-BioNeMo/KERMT)
-- `agent/scripts/run_pretrain_local.py` — extended usage examples
+- `scripts/run_pretrain_local.py` — extended usage examples
- Related skills: `kermt-setup`, `kermt-monitor`, `kermt-pretrain-scratch`, `kermt-add-cmim-pretrain`, `kermt-finetune`
## Skill Output:
diff --git a/agent/skills/kermt-embed/SKILL.md b/skills/kermt-embed/SKILL.md similarity index 82% rename from agent/skills/kermt-embed/SKILL.md rename to skills/kermt-embed/SKILL.md index 61027b8..26b6955 100644 --- a/agent/skills/kermt-embed/SKILL.md +++ b/skills/kermt-embed/SKILL.md @@ -17,6 +17,15 @@ Extract per-molecule embeddings from any encoder-bearing KERMT checkpoint. The skill is the workflow orchestrator: validate ckpt, validate CSV, clean SMILES, launch the runner blocking, return the per-readout `.npy` files. +## Skill and runtime paths + +Set `SKILL_DIR` to the directory containing this `SKILL.md`. Export +`KERMT_REPO` as the absolute path to the KERMT checkout used for model +execution. The bundled container helper mounts that checkout at +`/workspace` and this skill at `/skill` (read-only). Commands inside +the container use `/skill/scripts/`; defaults are bundled in `config/`. +See [Released models](references/released-models.md) for checkpoint bundle requirements. + ## Hardware requirements - **GPUs**: 1 (single-GPU). @@ -60,7 +69,7 @@ Let `$KERMT_REPO` be the path to your kermt repo checkout. 1. **Pre-flight: container + system probe.** ``` - $KERMT_REPO/agent/scripts/kermt_container.sh check_system + "$SKILL_DIR/scripts/kermt_container.sh" check_system ``` 2. **Compute run directory.** @@ -82,30 +91,30 @@ Let `$KERMT_REPO` be the path to your kermt repo checkout. `--model-dir ` if given. An already-complete bundle is reused. - **Download** (foreground; ~282 MB on first fetch): ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run --model-dir -- \ - "python agent/scripts/fetch_released_model.py --out /model" + "$SKILL_DIR/scripts/kermt_container.sh" run --model-dir -- \ + "python /skill/scripts/fetch_released_model.py --out /model" ``` Parse the JSON; abort on `ok: false` (surface `errors`). On success set ` = /kermt_contrastive_v2.0.pt`. **Validate** the resolved (or user-provided) ckpt: ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run --ckpt -- \ - "python agent/scripts/check_checkpoint.py --mode embed --ckpt /ckpt" + "$SKILL_DIR/scripts/kermt_container.sh" run --ckpt -- \ + "python /skill/scripts/check_checkpoint.py --mode embed --ckpt /ckpt" ``` Parse JSON. Abort on `ok: false`. The validator only refuses encoder-less ckpts (rare). 4. **Validate the data.** ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run --data -- \ - "python agent/scripts/check_data.py --mode embed --csv /data/" + "$SKILL_DIR/scripts/kermt_container.sh" run --data -- \ + "python /skill/scripts/check_data.py --mode embed --csv /data/" ``` 5. **Prepare the data** (clean-only — no features step). ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run --data --run-dir $RUN_DIR -- \ - "python agent/scripts/prepare_data.py --mode embed \\ + "$SKILL_DIR/scripts/kermt_container.sh" run --data --run-dir $RUN_DIR -- \ + "python /skill/scripts/prepare_data.py --mode embed \\ --csv /data/ --out /runs/data" ``` Outputs land at `$RUN_DIR/data/prepare_data.json` with a single `clean_csv` @@ -113,9 +122,9 @@ Let `$KERMT_REPO` be the path to your kermt repo checkout. 6. **Launch the runner (blocking).** ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run \\ + "$SKILL_DIR/scripts/kermt_container.sh" run \\ --ckpt --run-dir $RUN_DIR -- \\ - "python agent/scripts/run_extract_embeddings.py \\ + "python /skill/scripts/run_extract_embeddings.py \\ --ckpt /ckpt \\ --prepare-manifest /runs/data/prepare_data.json \\ --out /runs \\ diff --git a/skills/kermt-embed/config/defaults_embed.json b/skills/kermt-embed/config/defaults_embed.json new file mode 100644 index 0000000..48b2d61 --- /dev/null +++ b/skills/kermt-embed/config/defaults_embed.json @@ -0,0 +1,10 @@ +{ + "_about": "Default settings applied by kermt-embed. Embedding extraction is a stateless forward pass through the encoder + readout — no training, no scaling. The skill writes one .npy per readout (atom_from_atom, bond_from_atom, atom_from_bond, bond_from_bond) plus canonical_smiles.npy and validity.npy.", + + "runtime": { + "_about": "Runtime knobs. The default batch_size of 64 matches task/extract_embeddings.py's own default — bigger than inference because the embed forward pass has lower per-mol overhead.", + "batch_size": 64 + }, + + "_about_gpu_selection": "GPU selection is auto-detected at runtime, not a default here. Embed defaults to GPU 0; override with --gpus 0 (the single id you want)." +} diff --git a/skills/kermt-embed/config/released_model.json b/skills/kermt-embed/config/released_model.json new file mode 100644 index 0000000..1e19fc1 --- /dev/null +++ b/skills/kermt-embed/config/released_model.json @@ -0,0 +1,13 @@ +{ + "repo_id": "nvidia/NV-KERMT-70M-v2", + "revision": "7df5eb3179235fdea1e8124db73215da33d77dce", + "ckpt_name": "kermt_contrastive_v2.0.pt", + "vocab_files": [ + "pretrain_atom_vocab.json", + "pretrain_bond_vocab.json", + "pretrain_smiles_vocab.pkl" + ], + "model_type": "hybrid", + "license": "NVIDIA Open Model License", + "license_url": "https://huggingface.co/nvidia/NV-KERMT-70M-v2" +} diff --git a/agent/skills/kermt-embed/evals/evals.json b/skills/kermt-embed/evals/evals.json similarity index 100% rename from agent/skills/kermt-embed/evals/evals.json rename to skills/kermt-embed/evals/evals.json diff --git a/skills/kermt-embed/references/released-models.md b/skills/kermt-embed/references/released-models.md new file mode 100644 index 0000000..e748c7e --- /dev/null +++ b/skills/kermt-embed/references/released-models.md @@ -0,0 +1,34 @@ +# Released KERMT models + +Each released KERMT checkpoint is distributed as a **directory bundle** +containing the ckpt itself plus its vocab files: + +``` +/ +├── last_checkpoint.pt +├── pretrain_atom_vocab.{json,pkl} # either extension; pkl in current releases +├── pretrain_bond_vocab.{json,pkl} # either extension; pkl in current releases +└── pretrain_smiles_vocab.pkl # only for cmim / hybrid ckpts (pickle-only) +``` + +If you're upgrading a grover_base ckpt to hybrid with +`kermt-add-cmim-pretrain`, the +upgrade step builds a fresh `pretrain_smiles_vocab.pkl` from your +pretrain corpus — released bundles only ship the smiles vocab for +already-cmim / already-hybrid ckpts. + +The vocab files are an inseparable part of the released model — the ckpt's +vocab head dimensions are fixed at training time and only match these specific +vocab files. `kermt-continue-pretrain` treats the released ckpt's vocab as +authoritative: new corpora are tokenized through it rather than producing a +new vocab that would mismatch the ckpt's heads. + +The skill auto-detects the three vocab files in the ckpt's parent directory +and passes them through `prepare_data.py --vocab-dir`. If the bundle is +incomplete (or the user has the ckpt alone), the skill asks for the +`--vocab-dir` path; if the user can't provide one, the skill refuses to +proceed and suggests `kermt-pretrain-scratch` instead. + +To train a model on a corpus the released vocab can't cover, use +`kermt-pretrain-scratch` — the new vocab is built from the corpus and the +model is initialized fresh (no warm start; days-scale to converge). diff --git a/skills/kermt-embed/scripts/_utils.py b/skills/kermt-embed/scripts/_utils.py new file mode 100644 index 0000000..5bde460 --- /dev/null +++ b/skills/kermt-embed/scripts/_utils.py @@ -0,0 +1,272 @@ +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Shared utilities for the agent scripts. + +Kept intentionally small — only logic that appears (or would otherwise be +duplicated) in two or more `scripts/*.py` modules. Each script +maintains its own primary CLI + main flow. +""" +from __future__ import annotations + +import argparse +import json +import os +import shlex +import subprocess +import sys +from pathlib import Path +from typing import Any + + +# Conventional pretrain vocab filename stems. Used by prepare_data.py + +# upgrade_to_hybrid.py + the README "Released models" bundling docs + +# the test helpers. Centralized here so a future rename only touches one +# spot. +PRETRAIN_VOCAB_STEMS = { + "atom": "pretrain_atom_vocab", + "bond": "pretrain_bond_vocab", + "smiles": "pretrain_smiles_vocab", +} + + +def resolve_kermt_repo() -> Path: + """Find the runtime checkout independently of the installed skill location. + + An explicit KERMT_REPO takes precedence. In a repository checkout, walking + up from this helper or the working directory also supports local use. + """ + explicit = os.environ.get("KERMT_REPO") + if explicit: + candidates = [Path(explicit).expanduser().resolve()] + else: + candidates = [] + for start in (Path(__file__).resolve().parent, Path.cwd()): + candidates.extend((start, *start.parents)) + for candidate in candidates: + if (candidate / "main.py").is_file() and (candidate / "kermt").is_dir(): + return candidate + raise FileNotFoundError( + "KERMT checkout not found. Set KERMT_REPO to the checkout containing " + "main.py and kermt/; the installed skill directory is separate." + ) + + +def load_json(path: Path, *, name: str) -> dict[str, Any]: + """Load a JSON file with consistent error messages. + + `name` is a human-readable label for the document (e.g. "prepare_data.json") + so the error tells the user which schema we expected at that path. + """ + if not path.is_file(): + raise FileNotFoundError(f"{name} not found at {path}") + try: + return json.loads(path.read_text()) + except json.JSONDecodeError as exc: + raise ValueError(f"{name} at {path} is not valid JSON: {exc}") from exc + + +def count_vocab_entries(vocab_path: Path) -> int: + """Return the number of entries in a KERMT vocab file. + + Handles three layouts: + - JSON with `{stoi: {token: idx}, ...}` (MolVocab.save_vocab default) + - JSON as a raw `{token: idx}` dict (legacy / hand-edited) + - Pickle of a MolVocab / SMILESVocab object (the smiles vocab is always + pickled because its compiled-regex tokenizer state isn't + JSON-serializable). Falls through to raw `pickle.load` if the + MolVocab / SMILESVocab loader can't import or fails to recognize + the contents (e.g. test fixtures with plain dicts). + """ + if vocab_path.suffix == ".json": + data = json.loads(vocab_path.read_text()) + if isinstance(data, dict) and "stoi" in data: + return len(data["stoi"]) + if isinstance(data, dict): + return len(data) + raise ValueError(f"unsupported JSON vocab shape at {vocab_path}: {type(data).__name__}") + + # .pkl: try MolVocab / SMILESVocab first, then fall back to raw pickle. + try: + from kermt.data.torchvocab import MolVocab, SMILESVocab # type: ignore + for loader in (MolVocab.load_vocab, SMILESVocab.load_vocab): + try: + v = loader(str(vocab_path)) + return len(v) + except Exception: + continue + except ImportError: + pass + + import pickle + with vocab_path.open("rb") as f: + data = pickle.load(f) + if hasattr(data, "stoi"): + return len(data.stoi) + if hasattr(data, "__len__"): + return len(data) + raise ValueError(f"could not count entries in {vocab_path}") + + +def validate_vocab_file(vocab_path: Path, *, kind: str) -> None: + """Verify a user-provided vocab file is loadable BEFORE copying it into a + run directory. Raises ValueError on failure with a clear, user-facing message. + + `kind` is one of {"atom", "bond", "smiles"} — used only in the error message + so the user knows which file is wrong. + """ + if not vocab_path.is_file(): + raise FileNotFoundError(f"{kind} vocab file not found: {vocab_path}") + try: + n = count_vocab_entries(vocab_path) + except Exception as exc: # noqa: BLE001 + raise ValueError( + f"{kind} vocab file {vocab_path} is not loadable as a KERMT vocab " + f"({type(exc).__name__}: {exc}). Expected a MolVocab JSON or pickle " + f"(or a SMILESVocab pickle for the smiles vocab)." + ) from exc + if n <= 0: + raise ValueError(f"{kind} vocab file {vocab_path} contains zero entries") + + +# --------------------------------------------------------------------------- +# Runner-shared helpers (run.json manifest fields) +# --------------------------------------------------------------------------- + +def git_commit_with_env_override(repo: Path) -> tuple[str, bool]: + """Returns (commit_sha, dirty_tree). Honors `KERMT_REPO_COMMIT` / + `KERMT_REPO_DIRTY` env vars first — set by `scripts/kermt_container.sh` + from the host before launching docker (necessary because `git -C /workspace` + inside the container fails due to bind-mount ownership). Falls back to the + in-container git probe when the env vars aren't set.""" + env_commit = os.environ.get("KERMT_REPO_COMMIT") + if env_commit: + env_dirty = os.environ.get("KERMT_REPO_DIRTY", "false").strip().lower() == "true" + return env_commit, env_dirty + try: + sha = subprocess.run( + ["git", "-C", str(repo), "rev-parse", "HEAD"], + capture_output=True, text=True, check=True, + ).stdout.strip() + diff = subprocess.run( + ["git", "-C", str(repo), "status", "--porcelain"], + capture_output=True, text=True, check=True, + ) + return sha, bool(diff.stdout.strip()) + except Exception: + return "unknown", False + + +def docker_image_digest(tag: str) -> str | None: + """Return the docker image's content-addressable Id (sha256:…) for the given + tag, or None if docker isn't available / the image isn't local.""" + try: + r = subprocess.run( + ["docker", "image", "inspect", tag, "--format", "{{.Id}}"], + capture_output=True, text=True, + ) + if r.returncode == 0: + return r.stdout.strip() + except FileNotFoundError: + pass + return None + + +def format_cmd_replay(argv: list[str], *, env: dict[str, str] | None = None) -> str: + """Render a copy-pasteable env-prefix + command for the cmd_replay manifest + field. `env` is the set of environment variables to prefix (typically + {CUDA_VISIBLE_DEVICES, WORLD_SIZE}).""" + env = env or {} + env_prefix = [f"{k}={shlex.quote(str(v))}" for k, v in env.items()] + quoted = " ".join(shlex.quote(a) for a in argv) + return " ".join(env_prefix + [quoted]) + + +def resolve_single_gpu(override: str | None, *, workflow: str) -> int: + """Returns a single GPU id (int). The finetune/inference/embed workflows are + single-GPU only; `--gpus '0,1'` or multi-id CUDA_VISIBLE_DEVICES is rejected + with a workflow-specific error. (The pretrain runner has its own multi-GPU + `_detect_gpus` helper — see run_pretrain_local.py.)""" + if override is None: + env_visible = os.environ.get("CUDA_VISIBLE_DEVICES", "").strip() + if env_visible: + ids = [g for g in env_visible.split(",") if g] + if len(ids) > 1: + raise ValueError( + f"CUDA_VISIBLE_DEVICES='{env_visible}' selects multiple GPUs but " + f"the {workflow} workflow is single-GPU only. Restrict to one id." + ) + return int(ids[0]) + return 0 + parts = [p.strip() for p in override.split(",") if p.strip()] + if len(parts) != 1: + raise ValueError( + f"--gpus '{override}' selects {len(parts)} GPUs; the {workflow} workflow is single-GPU only." + ) + return int(parts[0]) + + +def assert_prepare_manifest_basics(manifest: dict[str, Any], expected_mode: str) -> None: + """Standard pre-check for a prepare_data.json before a runner consumes it: + verify `mode` matches and `ok` is True. Raises ValueError with a consistent + error message on either mismatch. + + Each runner is responsible for its own required-outputs check after this + (those vary per-mode — e.g. pretrain wants train_dir/val_dir/atom_vocab/ + bond_vocab; finetune has the split-method branch; inference/embed want + clean_csv).""" + if manifest.get("mode") != expected_mode: + raise ValueError( + f"prepare_data manifest is mode='{manifest.get('mode')}', expected '{expected_mode}'. " + f"Run `prepare_data.py --mode {expected_mode}` to produce a valid manifest." + ) + if not manifest.get("ok"): + raise ValueError( + f"prepare_data manifest reports ok=False: {manifest.get('errors')}" + ) + + +def merge_default_into_applied( + applied: dict[str, dict[str, Any]], + args: argparse.Namespace, + name: str, + defaults_group: dict[str, Any], +) -> None: + """Standard CLI-override / default-config merge for one hyperparameter. + + Mutates `applied` in place: + - If the user passed `--` on the CLI (so `getattr(args, name)` is + not None), records `{"value": cli_val, "source": "user"}`. + - Else if `name` is present in `defaults_group`, records + `{"value": defaults_group[name], "source": "default-config"}`. + - Else `applied[name]` is left absent — the runner's argv-builder skips + the flag, and the downstream argparse default takes effect. + + `name` is the snake_case argparse dest (same form used as the dict key); + argparse automatically converts CLI `--` to that dest, + so `getattr(args, name, None)` is the correct CLI lookup.""" + cli_val = getattr(args, name, None) + if cli_val is not None: + applied[name] = {"value": cli_val, "source": "user"} + elif name in defaults_group: + applied[name] = {"value": defaults_group[name], "source": "default-config"} + + +def run_checkpoint_validator(ckpt: Path, *, mode: str, script_path: Path) -> dict[str, Any]: + """Invoke `check_checkpoint.py --mode --ckpt ` as a subprocess + and return the parsed JSON. Raises RuntimeError on non-JSON output (e.g. the + validator crashed before printing). `script_path` is the absolute path to + `scripts/check_checkpoint.py` — passed in so this helper has no + dependency on the caller's layout.""" + r = subprocess.run( + [sys.executable, str(script_path), "--mode", mode, "--ckpt", str(ckpt)], + capture_output=True, text=True, + ) + try: + return json.loads(r.stdout) + except json.JSONDecodeError as exc: + raise RuntimeError( + f"check_checkpoint.py emitted non-JSON output (exit {r.returncode}). " + f"stdout (first 200 chars): {r.stdout[:200]}\n" + f"stderr (first 200 chars): {r.stderr[:200]}" + ) from exc diff --git a/skills/kermt-embed/scripts/check_checkpoint.py b/skills/kermt-embed/scripts/check_checkpoint.py new file mode 100644 index 0000000..fd488e2 --- /dev/null +++ b/skills/kermt-embed/scripts/check_checkpoint.py @@ -0,0 +1,480 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Validate a KERMT checkpoint for a given agent workflow. + +Mode-dispatched. Emits a single JSON object to stdout that the calling skill +parses to decide whether to proceed (`ok: true`) or surface a structured error +back to the user (`ok: false` with `errors[]`). The JSON shape is stable +across modes; only the contract for what counts as `ok` differs per mode. + +Modes +----- +continue_pretrain Continuing pretraining from an existing pretrain ckpt. + Requires encoder + at least one pretrain head + (vocab_head for grover_base / cmim, or contrast_head for + cmim / hybrid). Rejects encoder-only or finetuned ckpts. + +upgrade_to_hybrid Adding a cMIM decoder onto a grover_base ckpt to convert + it to a hybrid pretrain. Requires encoder; rejects ckpts + that already carry a contrast_head or task_ffn (would be + workflow 4 instead). + +finetune_init Starting a finetune from a pretrained ckpt. Requires + encoder. Pretrain heads (vocab / contrast) are tolerated + but unused. Already-finetuned ckpts (task FFN heads + present) are REJECTED — finetune-on-finetune via the + agent skill isn't supported because saved-task + identity can't be machine-verified against the new + training data. + +inference Running predictions with a previously-finetuned ckpt. + Requires encoder + task_ffn. Reports task_output_dims + so the runner can compare against the user's task spec. + +embed Extracting embeddings. Requires encoder only. Anything + additional in the ckpt is ignored. + +Output (stdout) +--------------- +{ + "ok": true | false, + "model_type": "grover_base" | "cmim" | "hybrid" | "finetuned" | "unknown", + "has_encoder": bool, + "has_vocab_head": bool, + "has_contrast_head": bool, + "has_task_ffn": bool, + "task_output_dims": [int, ...], // empty unless has_task_ffn + "arch": { // ckpt-derived; runner uses these, ignores defaults_*.json arch + "hidden_size": int | null, + "depth": int | null, + "num_attn_head": int | null, + "latent_dim": int | null, + "activation": str | null, + "backbone": str | null, + "embedding_output_type": str | null, + "self_attention": bool | null + }, + "saved_args": { ... } | null, // raw args dict if present, else null + "errors": [str, ...], // mode-contract violations / load failures + "warnings": [str, ...] // non-fatal observations (e.g. arch fallback) +} + +Exit code: 0 on `ok: true`, 1 on `ok: false`. Loader exceptions are caught and +surfaced into `errors[]` with `ok: false` (still exit 1), never raised. + +CLI +--- + check_checkpoint.py --mode --ckpt +""" +from __future__ import annotations + +import argparse +import json +import sys +import traceback +from argparse import Namespace +from typing import Any + +import torch + + +# --------------------------------------------------------------------------- +# State-dict key prefix conventions (kermt/model/models.py). +# --------------------------------------------------------------------------- + +# Encoder weights appear under one of these prefixes depending on the ckpt's +# era and task class: +# - `grover.*` : legacy grover_base ckpts (predate the cMIM rename) +# - `kermt.*` : current grover_base / hybrid / finetune ckpts +# - `latent_dist.kermt.*`: cmim ckpts (encoder lives only inside latent_dist) +ENCODER_PREFIXES = ("kermt.", "grover.", "latent_dist.kermt.") +VOCAB_HEAD_PREFIX = "vocab_module." +CONTRAST_DECODER_PREFIX = "decoder." # SMILES transformer decoder, cmim/hybrid only +LATENT_DIST_PREFIX = "latent_dist." # cmim/hybrid; encoder may share via latent_dist.kermt.* +TASK_FFN_PREFIXES = ( + "mol_atom_from_atom_ffn.", + "mol_atom_from_bond_ffn.", +) +TASK_FFN_TASK_SPECIFIC_PREFIXES = ( + "mol_atom_from_atom_ffn_task_specific.", + "mol_atom_from_bond_ffn_task_specific.", +) + + +ARCH_KEYS = ( + "hidden_size", + "depth", + "num_attn_head", + "latent_dim", + "activation", + "backbone", + "embedding_output_type", + "self_attention", +) + + +def _strip_ddp_prefix(state_dict: dict[str, Any]) -> dict[str, Any]: + """Strip `module.` prefix from every key if the dict is DDP-wrapped.""" + if state_dict and all(k.startswith("module.") for k in state_dict): + return {k[len("module."):]: v for k, v in state_dict.items()} + return state_dict + + +def _classify_model(state_dict: dict[str, Any]) -> dict[str, Any]: + keys = list(state_dict.keys()) + has_encoder = any(k.startswith(ENCODER_PREFIXES) for k in keys) + has_vocab_head = any(k.startswith(VOCAB_HEAD_PREFIX) for k in keys) + has_contrast_head = any(k.startswith(CONTRAST_DECODER_PREFIX) for k in keys) + has_task_ffn = any(k.startswith(TASK_FFN_PREFIXES) for k in keys) + + if has_encoder and has_task_ffn: + model_type = "finetuned" + elif has_encoder and has_contrast_head and has_vocab_head: + model_type = "hybrid" + elif has_encoder and has_contrast_head and not has_vocab_head: + model_type = "cmim" + elif has_encoder and not has_contrast_head: + # Includes: + # - modern repo-trained Grover base (kermt.* + vocab_module.*) + # - legacy original-Grover base (grover.encoders.* with no heads saved) + # - any encoder-stripped ckpt extracted from a larger model + # The `has_vocab_head` flag discriminates the sub-cases for skills that + # need it. The continue_pretrain mode contract relies on this — a + # grover_base with vocab heads can continue, an encoder-only one cannot. + model_type = "grover_base" + else: + model_type = "unknown" + + return { + "model_type": model_type, + "has_encoder": has_encoder, + "has_vocab_head": has_vocab_head, + "has_contrast_head": has_contrast_head, + "has_task_ffn": has_task_ffn, + } + + +def _vocab_sizes(state_dict: dict[str, Any]) -> dict[str, Any]: + """Extract vocab head sizes from state-dict weight shapes. + + The pretrain heads have the following layout per kermt/model/models.py: + - Atom vocab predictors: vocab_module.av_task_atom.* + vocab_module.av_task_bond.* + (two readout streams sharing the same vocab_size). Output dim of each + final-Linear is the atom vocab size. + - Bond vocab predictors: vocab_module.bv_task_atom.* + vocab_module.bv_task_bond.* + Output dim is the bond vocab size. + - SMILES vocab decoder: decoder.output_projection.weight (cmim / hybrid only). + Output dim is the smiles vocab size. + + Returns {atom: int|None, bond: int|None, smiles: int|None}. Each is None + when the corresponding head isn't present in the ckpt (e.g. legacy + encoder-only grover_base has none; cmim has smiles but not atom/bond). + """ + sizes: dict[str, Any] = {"atom": None, "bond": None, "smiles": None} + + def _head_out_dim(prefix: str) -> int | None: + # Pick the highest-numbered 2-D Linear weight under `prefix.*` — that's + # the final output layer. + candidates = [ + k for k in state_dict + if k.startswith(prefix) and k.endswith(".weight") + and hasattr(state_dict[k], "ndim") and state_dict[k].ndim == 2 + ] + if not candidates: + return None + def _layer_index(k: str) -> int: + # ".weight" -> "..weight"; pick the rightmost numeric component. + parts = k.split(".") + for tok in reversed(parts[:-1]): + if tok.isdigit(): + return int(tok) + return -1 + final = max(candidates, key=_layer_index) + return int(state_dict[final].shape[0]) + + sizes["atom"] = _head_out_dim("vocab_module.av_task_atom.") + sizes["bond"] = _head_out_dim("vocab_module.bv_task_atom.") + sizes["smiles"] = _head_out_dim("decoder.output_projection.") + # If the decoder's output_projection isn't a Linear (e.g. some saves wrap + # it differently), fall back to a search over decoder.* heads. + if sizes["smiles"] is None: + sizes["smiles"] = _head_out_dim("decoder.token_embedding.") + return sizes + + +def _task_output_dims(state_dict: dict[str, Any]) -> list[int]: + """Return one entry per (logical task × readout) head's final-Linear out-dim. + + Two layouts: + - **MTL** (`mol_atom_from_atom_ffn_task_specific..*`): one entry per + task-specific head's final-Linear out-dim. Typically `[1, 1, ..., 1]` + for regression with N tasks across 2 readouts. + - **Non-MTL** (`mol_atom_from_atom_ffn.*` only): one entry per shared FFN's + final-Linear out-dim. Typically `[num_tasks, num_tasks]` (one per readout). + + When both layouts coexist in the same ckpt (MTL configuration: shared FFN + feeds task-specific heads), only the task-specific dims are reported — the + shared FFN there is an intermediate layer, not the model output. + """ + has_task_specific = any(k.startswith(TASK_FFN_TASK_SPECIFIC_PREFIXES) for k in state_dict) + + heads: dict[str, list[str]] = {} + for k in state_dict: + if k.startswith(TASK_FFN_TASK_SPECIFIC_PREFIXES): + parts = k.split(".") + root = ".".join(parts[:2]) # e.g. "mol_atom_from_atom_ffn_task_specific.0" + heads.setdefault(root, []).append(k) + elif k.startswith(TASK_FFN_PREFIXES) and not k.startswith(TASK_FFN_TASK_SPECIFIC_PREFIXES): + if has_task_specific: + continue # shared FFN is intermediate when task-specific heads exist + root = k.split(".")[0] # e.g. "mol_atom_from_atom_ffn" + heads.setdefault(root, []).append(k) + + dims: list[int] = [] + for root in sorted(heads): + weight_keys = sorted( + (k for k in heads[root] if k.endswith(".weight") + and hasattr(state_dict[k], "ndim") and state_dict[k].ndim == 2), + key=lambda k: int(k.split(".")[-2]) if k.split(".")[-2].isdigit() else -1, + ) + if weight_keys: + dims.append(int(state_dict[weight_keys[-1]].shape[0])) + return dims + + +def _arch_from_args(args_obj: Any) -> dict[str, Any]: + """Pull arch params from the saved args Namespace / dict, leaving missing keys as None.""" + arch: dict[str, Any] = {k: None for k in ARCH_KEYS} + if args_obj is None: + return arch + # args_obj is typically argparse.Namespace; tolerate dict form too. + args_dict = vars(args_obj) if isinstance(args_obj, Namespace) else dict(args_obj) if isinstance(args_obj, dict) else {} + for k in ARCH_KEYS: + if k in args_dict: + arch[k] = args_dict[k] + return arch + + +def _arch_from_shapes(state_dict: dict[str, Any], arch: dict[str, Any]) -> tuple[dict[str, Any], list[str]]: + """Fill in still-missing arch params by introspecting state-dict tensor shapes. + + Only fills entries that are currently None — does not override anything pulled + from saved_args. Returns the updated arch + a list of warnings for any key that + could not be inferred. + """ + warnings: list[str] = [] + + if arch["hidden_size"] is None: + # First 2-D linear weight under any encoder prefix. + candidates = [ + k for k in state_dict + if k.startswith(ENCODER_PREFIXES) + and k.endswith(".weight") + and hasattr(state_dict[k], "ndim") + and state_dict[k].ndim == 2 + ] + if candidates: + arch["hidden_size"] = int(state_dict[candidates[0]].shape[0]) + else: + warnings.append("hidden_size could not be inferred from state_dict shapes") + + if arch["latent_dim"] is None: + # Look for a Linear inside latent_dist that's not the shared encoder. + candidates = [ + k for k in state_dict + if k.startswith(LATENT_DIST_PREFIX) + and not k.startswith("latent_dist.kermt.") + and k.endswith(".weight") + and state_dict[k].ndim == 2 + ] + if candidates: + arch["latent_dim"] = int(state_dict[candidates[0]].shape[0]) + # Absent latent_dist entirely is a fact about the model, not a problem — no warning. + + # depth, num_attn_head, activation, backbone, embedding_output_type, self_attention + # are not robustly inferable from shapes alone; report a warning for each that's + # still None so the caller can prompt the user or refuse to proceed. + for k in ("depth", "num_attn_head", "activation", "backbone", "embedding_output_type", "self_attention"): + if arch[k] is None: + warnings.append(f"{k} not present in saved_args and cannot be inferred from state_dict shapes") + + return arch, warnings + + +def _apply_mode_contract(mode: str, classification: dict[str, Any]) -> list[str]: + """Return a list of error messages if `classification` violates the mode contract.""" + errors: list[str] = [] + mt = classification["model_type"] + has_enc = classification["has_encoder"] + has_vocab = classification["has_vocab_head"] + has_contrast = classification["has_contrast_head"] + has_ffn = classification["has_task_ffn"] + + if not has_enc: + errors.append("checkpoint has no encoder weights — cannot use it for any KERMT workflow") + return errors + + if mode == "continue_pretrain": + if not (has_vocab or has_contrast): + errors.append( + f"continue_pretrain requires the ckpt to still carry pretrain heads (vocab " + f"and/or contrast), but this ckpt has neither (model_type='{mt}', " + f"has_vocab_head=False, has_contrast_head=False). Either provide a ckpt with " + f"its pretrain heads attached, or convert this encoder-only ckpt to a hybrid " + f"via mode 'upgrade_to_hybrid'." + ) + if has_ffn: + errors.append( + "continue_pretrain expects a pretrain ckpt; this ckpt has task FFN heads " + "(it has been finetuned). Use a pretrain checkpoint — finetune+continue is " + "not a supported workflow." + ) + elif mode == "upgrade_to_hybrid": + if has_contrast: + errors.append( + f"upgrade_to_hybrid converts grover_base -> hybrid by adding a cMIM decoder. " + f"This ckpt already has a contrast head (classified as '{mt}'). " + f"To continue pretraining it, use mode 'continue_pretrain'." + ) + if has_ffn: + errors.append("upgrade_to_hybrid does not support finetuned checkpoints.") + elif mode == "finetune_init": + # Requires an encoder. Pretrain heads (vocab / contrast) are unused + # at finetune time but harmless. Task FFN heads (i.e. an already- + # finetuned ckpt) are NOT accepted — finetune-on-finetune isn't + # supported by the kermt-finetune skill because the saved-task + # identity can't be machine-verified against the new training data + # (dimension match doesn't prove target identity, dataset identity, + # or absence of train/test contamination). + if has_ffn: + errors.append( + f"finetune_init requires a pretrain ckpt (grover_base / cmim / hybrid); " + f"this ckpt is classified as '{mt}' with task FFN heads attached. " + f"To resume a finetune on the SAME dataset, call " + f"`python main.py finetune --checkpoint_path ...` directly — the " + f"kermt-finetune skill doesn't support resume." + ) + elif mode == "inference": + if not has_ffn: + errors.append( + "inference requires a finetuned ckpt with task FFN heads. " + f"This ckpt is classified as '{mt}' with no task heads. " + "Run finetune (mode 'finetune_init') first." + ) + elif mode == "embed": + # Encoder is sufficient. + pass + else: + errors.append(f"unknown mode '{mode}'") + + return errors + + +def validate(mode: str, ckpt_path: str) -> dict[str, Any]: + result: dict[str, Any] = { + "ok": False, + "model_type": "unknown", + "has_encoder": False, + "has_vocab_head": False, + "has_contrast_head": False, + "has_task_ffn": False, + "task_output_dims": [], + "vocab_sizes": {"atom": None, "bond": None, "smiles": None}, + "arch": {k: None for k in ARCH_KEYS}, + "saved_args": None, + "errors": [], + "warnings": [], + } + + # 1. Load the checkpoint. + try: + ckpt = torch.load(ckpt_path, map_location="cpu", weights_only=False) + except FileNotFoundError: + result["errors"].append(f"checkpoint not found: {ckpt_path}") + return result + except Exception as exc: # noqa: BLE001 + result["errors"].append(f"failed to load checkpoint {ckpt_path}: {type(exc).__name__}: {exc}") + return result + + if not isinstance(ckpt, dict) or "state_dict" not in ckpt: + result["errors"].append( + "checkpoint is not in the expected save_model_for_restart format " + "(expected a dict with a 'state_dict' key)." + ) + return result + + state_dict = _strip_ddp_prefix(ckpt["state_dict"]) + args_obj = ckpt.get("args") + + # 2. Classify and check mode contract. + classification = _classify_model(state_dict) + result.update(classification) + + contract_errors = _apply_mode_contract(mode, classification) + result["errors"].extend(contract_errors) + + # 3. Task output dims (for inference / informational). + if classification["has_task_ffn"]: + result["task_output_dims"] = _task_output_dims(state_dict) + + # 3b. Vocab head sizes (for continue-pretrain vocab-size verification). + result["vocab_sizes"] = _vocab_sizes(state_dict) + + # 4. Arch derivation: args first, shape introspection for what's still missing. + arch = _arch_from_args(args_obj) + arch, shape_warnings = _arch_from_shapes(state_dict, arch) + result["arch"] = arch + result["warnings"].extend(shape_warnings) + + # 5. Saved args as serializable dict (best-effort). + if args_obj is not None: + try: + result["saved_args"] = vars(args_obj) if isinstance(args_obj, Namespace) else dict(args_obj) + # Drop non-JSON-serializable values; agent skill only needs human-readable scalars. + result["saved_args"] = { + k: v for k, v in result["saved_args"].items() + if isinstance(v, (str, int, float, bool, type(None), list, dict)) + } + except Exception as exc: # noqa: BLE001 + result["warnings"].append(f"could not serialize saved_args: {type(exc).__name__}: {exc}") + + result["ok"] = not result["errors"] + return result + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Validate a KERMT checkpoint for a given workflow.") + parser.add_argument("--mode", required=True, + choices=["continue_pretrain", "upgrade_to_hybrid", "finetune_init", "inference", "embed"]) + parser.add_argument("--ckpt", required=True, help="Path to the .pt checkpoint") + args = parser.parse_args(argv) + + try: + result = validate(args.mode, args.ckpt) + except Exception as exc: # noqa: BLE001 + # Last-resort safety net: keep stdout JSON-clean, dump trace to stderr. + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({ + "ok": False, + "model_type": "unknown", + "errors": [f"unhandled exception in validator: {type(exc).__name__}: {exc}"], + "warnings": [], + "arch": {k: None for k in ARCH_KEYS}, + "has_encoder": False, + "has_vocab_head": False, + "has_contrast_head": False, + "has_task_ffn": False, + "task_output_dims": [], + "vocab_sizes": {"atom": None, "bond": None, "smiles": None}, + "saved_args": None, + }, indent=2)) + return 1 + + print(json.dumps(result, indent=2)) + return 0 if result["ok"] else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-embed/scripts/check_data.py b/skills/kermt-embed/scripts/check_data.py new file mode 100644 index 0000000..b8f9b15 --- /dev/null +++ b/skills/kermt-embed/scripts/check_data.py @@ -0,0 +1,310 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Validate a CSV input for a given KERMT agent workflow. + +Mode-dispatched. Emits a single JSON object to stdout that the calling skill +parses to decide whether to proceed (`ok: true`) or surface a structured error +back to the user (`ok: false` with `errors[]`). The JSON shape is stable +across modes; only the contract for what counts as `ok` differs per mode. + +Modes +----- +pretrain Pretrain corpus CSV. Requires a `smiles` column. Other columns + are ignored. Label columns are not required (and not expected). + +finetune Labeled CSV for a downstream task. Requires `smiles` plus + >=1 numeric target column. Target columns are specified via + `--targets ...`. If `--targets` is omitted, the + validator auto-detects numeric non-smiles columns and reports + them; the skill will then prompt the user to confirm or refine. + +inference CSV to run predictions on. Requires `smiles`. Target columns are + not required (and not expected — predictions are written out). + +embed CSV to extract embeddings from. Requires `smiles` only. + +SMILES validation +----------------- +By default the validator samples up to 20 SMILES (first 10 + last 10) and +checks each one parses with RDKit. Pass `--strict-rdkit` to parse every +SMILES (slow on large corpora). A SMILES is considered "invalid" if RDKit +returns `None` from `MolFromSmiles(smi, sanitize=True)` — empty / null +rows are counted separately. + +Duplicate-SMILES detection is always full (cheap). + +Output (stdout) +--------------- +{ + "ok": true | false, + "mode": str, + "csv_path": str, + "num_rows": int, + "num_columns": int, + "columns": [str, ...], + "has_smiles_column": bool, + "smiles_column_name": str | null, // actual header used (may differ in case) + "num_blank_smiles": int, + "num_invalid_smiles": int, // among the parsed sample + "smiles_check_method": "sampled" | "full", + "smiles_check_count": int, + "num_duplicate_smiles": int, + "target_columns": [str, ...], // populated only for finetune mode + "num_missing_per_target": { col: int, ... }, + "auto_detected_targets": [str, ...], // when --targets is omitted in finetune mode + "errors": [str, ...], + "warnings": [str, ...] +} + +Exit code: 0 on `ok: true`, 1 on `ok: false`. Loader exceptions are caught +and surfaced into `errors[]` with `ok: false` (still exit 1). + +CLI +--- + check_data.py --mode --csv + [--targets ...] # finetune only + [--strict-rdkit] # full SMILES parse +""" +from __future__ import annotations + +import argparse +import json +import sys +import traceback +from pathlib import Path +from typing import Any + +import pandas as pd + + +CANONICAL_SMILES_COLUMN = "smiles" +SMILES_SAMPLE_PER_END = 10 # how many SMILES from head + how many from tail to sample + + +def _find_smiles_column(columns: list[str]) -> str | None: + """Return the actual column header matching 'smiles' case-insensitively, or None.""" + for c in columns: + if c.lower() == CANONICAL_SMILES_COLUMN: + return c + return None + + +def _parse_smiles_sample(smiles_values: list[str], full: bool) -> tuple[int, int, str]: + """Run RDKit MolFromSmiles on a sample or all of the SMILES. Returns + (num_parsed, num_invalid, method).""" + # Import here so the script can still surface a clean JSON error if RDKit + # is unavailable in the host env. + try: + from rdkit import Chem + from rdkit import RDLogger + RDLogger.DisableLog("rdApp.*") # suppress per-mol parse warnings + except ImportError as exc: + raise RuntimeError( + f"RDKit is not importable in this environment: {exc}. " + "Run check_data.py inside the kermt container." + ) from exc + + if full or len(smiles_values) <= 2 * SMILES_SAMPLE_PER_END: + sample = smiles_values + method = "full" + else: + sample = smiles_values[:SMILES_SAMPLE_PER_END] + smiles_values[-SMILES_SAMPLE_PER_END:] + method = "sampled" + + invalid = 0 + parsed = 0 + for smi in sample: + if not smi: # already counted as blank elsewhere + continue + parsed += 1 + mol = Chem.MolFromSmiles(smi, sanitize=True) + if mol is None: + invalid += 1 + return parsed, invalid, method + + +def _autodetect_target_columns(df: pd.DataFrame, smiles_col: str) -> list[str]: + """Pick columns that look like numeric targets. A column qualifies if it + is (a) not the smiles column and (b) >=80% of non-null values convert to float. + Heuristic only — returned for the skill to prompt the user to confirm.""" + candidates: list[str] = [] + for col in df.columns: + if col == smiles_col: + continue + ser = df[col].dropna() + if len(ser) == 0: + continue + try: + converted = pd.to_numeric(ser, errors="coerce") + except (TypeError, ValueError): + continue + if converted.notna().sum() / max(len(ser), 1) >= 0.8: + candidates.append(col) + return candidates + + +def validate(mode: str, csv_path: str, targets: list[str] | None, strict_rdkit: bool) -> dict[str, Any]: + result: dict[str, Any] = { + "ok": False, + "mode": mode, + "csv_path": csv_path, + "num_rows": 0, + "num_columns": 0, + "columns": [], + "has_smiles_column": False, + "smiles_column_name": None, + "num_blank_smiles": 0, + "num_invalid_smiles": 0, + "smiles_check_method": "sampled", + "smiles_check_count": 0, + "num_duplicate_smiles": 0, + "target_columns": [], + "num_missing_per_target": {}, + "auto_detected_targets": [], + "errors": [], + "warnings": [], + } + + # 1. Read the CSV. + path = Path(csv_path) + if not path.is_file(): + result["errors"].append(f"CSV not found: {csv_path}") + return result + try: + df = pd.read_csv(path) + except pd.errors.EmptyDataError: + result["errors"].append(f"CSV is empty (no header): {csv_path}") + return result + except Exception as exc: # noqa: BLE001 + result["errors"].append(f"failed to read CSV {csv_path}: {type(exc).__name__}: {exc}") + return result + + result["num_rows"] = int(len(df)) + result["num_columns"] = int(len(df.columns)) + result["columns"] = [str(c) for c in df.columns] + + # 2. Locate the SMILES column. + smiles_col = _find_smiles_column(result["columns"]) + if smiles_col is None: + result["errors"].append( + f"no column named 'smiles' (case-insensitive) found in CSV. " + f"Available columns: {result['columns']}" + ) + return result + result["has_smiles_column"] = True + result["smiles_column_name"] = smiles_col + if smiles_col != CANONICAL_SMILES_COLUMN: + result["warnings"].append( + f"SMILES column is named '{smiles_col}' but downstream code expects '{CANONICAL_SMILES_COLUMN}' " + f"(lowercase). Rename the column to '{CANONICAL_SMILES_COLUMN}' before running the workflow." + ) + + # 3. Blank-SMILES count + duplicate count + RDKit parse check. + smi_series = df[smiles_col].astype(str).fillna("").str.strip() + blank_mask = smi_series.eq("") | smi_series.str.lower().eq("nan") + result["num_blank_smiles"] = int(blank_mask.sum()) + + nonblank = smi_series[~blank_mask] + result["num_duplicate_smiles"] = int(len(nonblank) - nonblank.nunique()) + + if len(nonblank) == 0: + result["errors"].append("no non-blank SMILES found in the CSV") + return result + + try: + parsed, invalid, method = _parse_smiles_sample(nonblank.tolist(), full=strict_rdkit) + except RuntimeError as exc: + result["errors"].append(str(exc)) + return result + result["smiles_check_count"] = parsed + result["num_invalid_smiles"] = invalid + result["smiles_check_method"] = method + + if invalid > 0: + scope = "all rows" if method == "full" else f"the {parsed} sampled rows" + result["errors"].append( + f"{invalid} out of {parsed} SMILES in {scope} failed to parse with RDKit. " + "Either pre-clean the CSV with scripts/clean_smiles.py or pass --strict-rdkit to see " + "the full count." + ) + + # 4. Target-column handling — finetune mode only. + if mode == "finetune": + if targets: + missing = [t for t in targets if t not in df.columns] + if missing: + result["errors"].append( + f"target column(s) not found in CSV: {missing}. " + f"Available columns: {result['columns']}" + ) + else: + result["target_columns"] = list(targets) + for t in targets: + nan_count = int(df[t].isna().sum()) + result["num_missing_per_target"][t] = nan_count + # Confirm numeric-ish. + nonnan = df[t].dropna() + converted = pd.to_numeric(nonnan, errors="coerce") + non_numeric_count = int(converted.isna().sum()) + if non_numeric_count > 0: + result["warnings"].append( + f"target column '{t}' has {non_numeric_count} non-numeric value(s) " + f"that will be dropped by the finetune runner." + ) + else: + # Auto-detect — surface candidates so the skill can prompt the user. + result["auto_detected_targets"] = _autodetect_target_columns(df, smiles_col) + if not result["auto_detected_targets"]: + result["errors"].append( + "no numeric non-smiles columns detected. finetune needs at least one target column; " + "specify it explicitly via --targets ." + ) + else: + result["warnings"].append( + f"--targets was not specified; auto-detected candidate target columns " + f"{result['auto_detected_targets']}. The skill will prompt the user to confirm." + ) + + # 5. Small-corpus warning — only for pretrain (other modes can be tiny by design). + if mode == "pretrain" and result["num_rows"] < 100: + result["warnings"].append( + f"pretrain corpus is only {result['num_rows']} molecule(s). Pretraining typically " + f"needs orders of magnitude more — verify this is the intended input." + ) + + result["ok"] = not result["errors"] + return result + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Validate a CSV input for a KERMT agent workflow.") + parser.add_argument("--mode", required=True, choices=["pretrain", "finetune", "inference", "embed"]) + parser.add_argument("--csv", required=True, help="Path to the input CSV") + parser.add_argument("--targets", nargs="+", default=None, + help="(finetune only) target column names. If omitted, the validator auto-detects " + "numeric non-smiles columns and reports them as candidates.") + parser.add_argument("--strict-rdkit", action="store_true", + help="Parse every SMILES with RDKit rather than sampling (slow on large CSVs).") + args = parser.parse_args(argv) + + try: + result = validate(args.mode, args.csv, args.targets, args.strict_rdkit) + except Exception as exc: # noqa: BLE001 + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({ + "ok": False, + "mode": args.mode, + "csv_path": args.csv, + "errors": [f"unhandled exception in validator: {type(exc).__name__}: {exc}"], + "warnings": [], + }, indent=2)) + return 1 + + print(json.dumps(result, indent=2)) + return 0 if result["ok"] else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-embed/scripts/fetch_released_model.py b/skills/kermt-embed/scripts/fetch_released_model.py new file mode 100644 index 0000000..ca2ad57 --- /dev/null +++ b/skills/kermt-embed/scripts/fetch_released_model.py @@ -0,0 +1,234 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Download a released KERMT model bundle from Hugging Face. + +Runs INSIDE the kermt container (huggingface_hub is part of the image env). +Writes the released-model directory bundle — the checkpoint plus its vocab +files — into the directory mounted at `--out` (the skills mount the user's +chosen save location there via `kermt_container.sh --model-dir `), then +emits a single JSON object to stdout that the calling skill parses. + +The downloaded directory is exactly the repo's "released model bundle" layout +(see skills/README.md "Released models"): `.pt` + the three +`pretrain_*_vocab.*` files in one flat directory. The downstream skill then +feeds it through the existing `--ckpt /` flow; for +continue-pretrain the bundled vocab files are auto-detected in the ckpt's +parent directory. No runner changes are needed. + +Defaults (repo id, pinned revision, ckpt + vocab filenames) come from +`config/released_model.json` so the pin lives in one place; every value +is overridable on the CLI. + +Idempotent: if the bundle is already complete in `--out` (ckpt + all vocab +files present), nothing is downloaded and `reused: true` is reported — so a +re-invocation never re-fetches the 282 MB checkpoint. + +Authentication: the repo is public (no token needed). If `HF_TOKEN` is set in +the environment (forwarded into the container by `kermt_container.sh`), +huggingface_hub picks it up automatically — useful against shared-IP rate +limits or if the repo is ever gated. + +Output (stdout) +--------------- +{ + "ok": true | false, + "repo_id": str, + "revision": str, + "out": str, // container path of the bundle dir (e.g. /model) + "ckpt": str | null, // container path of the checkpoint file + "vocab_dir": str | null, // == out (where the vocab files live) + "ckpt_name": str, + "vocab_files": [str, ...], + "files_present": [str, ...], + "ckpt_bytes": int | null, + "reused": bool, // true if the bundle already existed (no download) + "license": str | null, + "license_url": str | null, + "errors": [str, ...] +} + +Exit code: 0 on `ok: true`, 1 on `ok: false`. + +CLI +--- + fetch_released_model.py [--out /model] + [--repo-id ] [--revision ] + [--ckpt-name ] [--config ] +""" + +from __future__ import annotations + +import argparse +import json +import sys +import traceback +from pathlib import Path +from typing import Any + +# Default config lives at config/released_model.json (one dir up from +# scripts/). Resolved relative to this file so the script is +# location-independent. +DEFAULT_CONFIG = ( + Path(__file__).resolve().parent.parent / "config" / "released_model.json" +) + + +def _load_config(config_path: Path) -> dict[str, Any]: + if not config_path.is_file(): + raise FileNotFoundError(f"released-model config not found at {config_path}") + return json.loads(config_path.read_text()) + + +def fetch( + *, + out: Path, + repo_id: str, + revision: str, + ckpt_name: str, + vocab_files: list[str], + license_name: str | None = None, + license_url: str | None = None, +) -> dict[str, Any]: + """Resolve-or-download the released bundle into `out`. Returns the manifest + dict (never raises for the expected failure modes — they land in + `errors[]` with `ok: false`).""" + result: dict[str, Any] = { + "ok": False, + "repo_id": repo_id, + "revision": revision, + "out": str(out), + "ckpt": None, + "vocab_dir": None, + "ckpt_name": ckpt_name, + "vocab_files": list(vocab_files), + "files_present": [], + "ckpt_bytes": None, + "reused": False, + "license": license_name, + "license_url": license_url, + "errors": [], + } + + required = [ckpt_name, *vocab_files] + + def _present() -> list[str]: + return [name for name in required if (out / name).is_file()] + + # 1. Idempotent reuse — bundle already complete in `out`. + if out.is_dir() and set(_present()) == set(required): + result["reused"] = True + else: + # 2. Download. Import here so a stale image (missing huggingface_hub) + # surfaces a clean, actionable JSON error rather than a traceback. + try: + from huggingface_hub import snapshot_download + except ImportError: + result["errors"].append( + "huggingface_hub is not available in the container image. The " + "released-model download needs it; rebuild the image with " + "`kermt-setup` (it now ships huggingface_hub) and retry." + ) + return result + + out.mkdir(parents=True, exist_ok=True) + try: + # local_dir gives a flat copy (the bundle layout) rather than the + # opaque blob/snapshot cache. HF_TOKEN, if set, is read by the lib. + snapshot_download(repo_id=repo_id, revision=revision, local_dir=str(out)) + except Exception as exc: # noqa: BLE001 + result["errors"].append( + f"download failed for {repo_id}@{revision}: {type(exc).__name__}: {exc}" + ) + return result + + # 3. Verify the bundle is complete regardless of download/reuse path. + present = _present() + result["files_present"] = present + missing = [name for name in required if name not in present] + if missing: + result["errors"].append( + f"bundle at {out} is missing expected file(s): {missing}. " + f"Present: {present}." + ) + return result + + ckpt_path = out / ckpt_name + result["ckpt"] = str(ckpt_path) + result["vocab_dir"] = str(out) + try: + result["ckpt_bytes"] = ckpt_path.stat().st_size + except OSError: + result["ckpt_bytes"] = None + + result["ok"] = True + return result + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser( + description="Download a released KERMT model bundle from Hugging Face (runs in-container)." + ) + parser.add_argument( + "--out", + default="/model", + help="Directory to write the bundle into (default: /model, the --model-dir mount).", + ) + parser.add_argument( + "--config", + default=str(DEFAULT_CONFIG), + help="Path to released_model.json (default: config/released_model.json).", + ) + parser.add_argument( + "--repo-id", default=None, help="Override the HF repo id from the config." + ) + parser.add_argument( + "--revision", + default=None, + help="Override the pinned revision (sha/tag/branch).", + ) + parser.add_argument( + "--ckpt-name", + default=None, + help="Override the checkpoint filename from the config.", + ) + args = parser.parse_args(argv) + + try: + cfg = _load_config(Path(args.config)) + repo_id = args.repo_id or cfg["repo_id"] + revision = args.revision or cfg["revision"] + ckpt_name = args.ckpt_name or cfg["ckpt_name"] + vocab_files = list(cfg.get("vocab_files", [])) + result = fetch( + out=Path(args.out), + repo_id=repo_id, + revision=revision, + ckpt_name=ckpt_name, + vocab_files=vocab_files, + license_name=cfg.get("license"), + license_url=cfg.get("license_url"), + ) + except Exception as exc: # noqa: BLE001 + print(traceback.format_exc(), file=sys.stderr) + print( + json.dumps( + { + "ok": False, + "out": args.out, + "errors": [ + f"unhandled exception in fetch_released_model: {type(exc).__name__}: {exc}" + ], + }, + indent=2, + ) + ) + return 1 + + print(json.dumps(result, indent=2)) + return 0 if result["ok"] else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/skills/kermt-embed/scripts/kermt_container.sh b/skills/kermt-embed/scripts/kermt_container.sh new file mode 100755 index 0000000..028057e --- /dev/null +++ b/skills/kermt-embed/scripts/kermt_container.sh @@ -0,0 +1,484 @@ +#!/usr/bin/env bash +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# kermt_container.sh — bootstrap helper for the kermt agent skills. +# +# Two ways to use this file: +# +# 1. As a subcommand dispatcher (recommended for skills): +# "$SKILL_DIR/scripts/kermt_container.sh" ensure_image +# "$SKILL_DIR/scripts/kermt_container.sh" run --ckpt /host/ckpt.pt -- python -c 'import torch; print(torch.cuda.device_count())' +# "$SKILL_DIR/scripts/kermt_container.sh" run_detached --name foo --run-dir runs/foo -- bash train.sh +# +# 2. Sourced into a shell or another script, then call the kermt_* functions +# directly: +# source "$SKILL_DIR/scripts/kermt_container.sh" +# kermt_ensure_image +# kermt_run --ckpt /host/ckpt.pt -- python ... +# +# Configuration (override via env vars before invocation): +# KERMT_IMAGE docker image tag (default: kermt:latest) +# KERMT_REPO host path to the kermt repo checkout (default: auto-derived +# from this script's location) +# KERMT_GPUS value passed to docker --gpus (default: all) +# +# Mount flags accepted by kermt_run / kermt_run_detached: +# --data bind to /data (read-only). If is a file, +# its PARENT directory is mounted at /data so +# commands can use /data/; if is a +# directory, it is mounted at /data directly. +# --ckpt bind to /ckpt (read-only; the path is mounted as-is) +# --vocab-dir bind to /vocab (read-only) +# --run-dir bind to /runs (read-write; created on host if missing) +# --model-dir bind to /model (read-write; created on host if missing). +# Target for released-model downloads (fetch_released_model.py). +# +# Additional flags for kermt_run_detached: +# --name docker container name (default: kermt--) +# +# Everything after `--` is the command passed to the container. It runs inside +# the `kermt` conda environment (the image's default env). + +set -o pipefail + +: "${KERMT_IMAGE:=kermt:latest}" +: "${KERMT_GPUS:=all}" + +# The skill may be installed outside the KERMT checkout. Mount its own helpers +# separately so container commands always execute the distributed skill copy. +_kermt_script_dir="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)" +_kermt_bundle_dir="$(cd "$_kermt_script_dir/.." && pwd)" +if [[ -z "${KERMT_REPO:-}" ]]; then + for _kermt_start in "$_kermt_script_dir" "$PWD"; do + _kermt_candidate="$_kermt_start" + while [[ "$_kermt_candidate" != / ]]; do + if [[ -f "$_kermt_candidate/main.py" && -d "$_kermt_candidate/kermt" ]]; then + KERMT_REPO="$_kermt_candidate" + break 2 + fi + _kermt_candidate="$(dirname "$_kermt_candidate")" + done + done + unset _kermt_start _kermt_candidate +fi +unset _kermt_script_dir + +_kermt_require_repo() { + if [[ -z "${KERMT_REPO:-}" || ! -f "$KERMT_REPO/main.py" || ! -d "$KERMT_REPO/kermt" ]]; then + echo "[kermt] Set KERMT_REPO to the KERMT checkout containing main.py and kermt/." >&2 + return 1 + fi + KERMT_REPO="$(cd "$KERMT_REPO" && pwd)" || return $? + export KERMT_REPO +} + +# ----------------------------------------------------------------------------- +# Host environment checks +# ----------------------------------------------------------------------------- + +kermt_check_docker() { + if ! command -v docker >/dev/null 2>&1; then + echo "[kermt] error: docker not found on PATH. Install Docker first." >&2 + return 1 + fi + if ! docker info >/dev/null 2>&1; then + echo "[kermt] error: docker daemon not reachable. Is the docker service running, and is your user in the 'docker' group?" >&2 + return 1 + fi +} + +kermt_check_system() { + # Probe host system and report GPU presence + VRAM + compute capability + + # driver / CUDA version + disk space. Emits a single JSON document to + # stdout that the calling skill consumes; exits 0 with `ok: false` and a + # populated `gaps` array when anything is below the per-workflow minimum, + # exits 1 only on unexpected internal errors. Uses host nvidia-smi + df + + # host python3 (stdlib only). + python3 - "$KERMT_REPO" "$KERMT_IMAGE" <<'PYEOF' +import json, os, shutil, subprocess, sys + +repo, image = sys.argv[1], sys.argv[2] + +result = { + "ok": True, + "gpus": [], + "disk": {"path": repo, "free_gb": None, "min_gb": 20}, + "host": {"docker": None, "nvidia_smi": None, "container_toolkit": None}, + "image": {"tag": image, "present_locally": None}, + "gaps": [], +} + +def _gap(msg): + result["ok"] = False + result["gaps"].append(msg) + +# docker presence +try: + r = subprocess.run(["docker", "info"], capture_output=True, text=True, timeout=10) + result["host"]["docker"] = "ok" if r.returncode == 0 else f"failed: {r.stderr.strip().splitlines()[-1] if r.stderr else 'unknown'}" + if r.returncode != 0: + _gap("docker daemon not reachable (is the service running, and is your user in the 'docker' group?)") +except FileNotFoundError: + result["host"]["docker"] = "not found" + _gap("docker not on PATH; install Docker first") +except Exception as e: + result["host"]["docker"] = f"error: {e}" + _gap(f"docker probe failed: {e}") + +# nvidia-smi (host driver) +try: + r = subprocess.run( + ["nvidia-smi", "--query-gpu=name,memory.total,compute_cap,driver_version,uuid", + "--format=csv,noheader,nounits"], + capture_output=True, text=True, timeout=10, + ) + if r.returncode == 0: + result["host"]["nvidia_smi"] = "ok" + for line in r.stdout.strip().splitlines(): + parts = [p.strip() for p in line.split(",")] + if len(parts) >= 5: + try: + vram_mb = int(parts[1]) + except ValueError: + vram_mb = None + result["gpus"].append({ + "name": parts[0], + "vram_mb": vram_mb, + "compute_cap": parts[2], + "driver": parts[3], + "uuid": parts[4], + }) + if not result["gpus"]: + _gap("nvidia-smi succeeded but reported no GPUs") + else: + result["host"]["nvidia_smi"] = "failed" + _gap("nvidia-smi found but failed; is the NVIDIA driver loaded?") +except FileNotFoundError: + result["host"]["nvidia_smi"] = "not found" + _gap("nvidia-smi not on PATH; install the NVIDIA driver") +except Exception as e: + result["host"]["nvidia_smi"] = f"error: {e}" + _gap(f"nvidia-smi probe failed: {e}") + +# disk free at the repo location +try: + free_bytes = shutil.disk_usage(repo).free + free_gb = free_bytes // (1024**3) + result["disk"]["free_gb"] = free_gb + if free_gb < result["disk"]["min_gb"]: + _gap(f"disk free at {repo} is {free_gb} GB; need at least {result['disk']['min_gb']} GB for the kermt image") +except Exception as e: + _gap(f"could not check disk space at {repo}: {e}") + +# image presence (informational only) +try: + r = subprocess.run(["docker", "image", "inspect", image], capture_output=True, text=True, timeout=10) + result["image"]["present_locally"] = (r.returncode == 0) +except Exception: + result["image"]["present_locally"] = None + +# nvidia-container-toolkit probe — only meaningful if both docker and a +# locally-present image are available. Pick kermt:$tag first; fall back to +# the small CUDA base image if that's the only one present; otherwise skip +# (avoid pulling anything). +def _probe_image(): + for img in (image, "nvidia/cuda:12.6.3-base-ubuntu22.04"): + r = subprocess.run(["docker", "image", "inspect", img], capture_output=True) + if r.returncode == 0: + return img + return None + +probe_img = _probe_image() +if probe_img: + try: + r = subprocess.run( + ["docker", "run", "--rm", "--gpus", "all", probe_img, "nvidia-smi"], + capture_output=True, text=True, timeout=60, + ) + if r.returncode == 0: + result["host"]["container_toolkit"] = f"ok (probed via {probe_img})" + else: + result["host"]["container_toolkit"] = f"failed (probed via {probe_img})" + _gap("`docker run --gpus all` failed; install nvidia-container-toolkit and ensure the host driver supports it") + except Exception as e: + result["host"]["container_toolkit"] = f"error: {e}" + _gap(f"nvidia-container-toolkit probe failed: {e}") +else: + result["host"]["container_toolkit"] = "skipped (no probe image present locally; run ensure_image first)" + +print(json.dumps(result, indent=2)) +PYEOF +} + +kermt_check_gpu() { + # Probes whether `docker --gpus all` is wired up (nvidia-container-toolkit). + # Image-selection priority (never pulls anything): + # 1) $KERMT_IMAGE if it exists locally, + # 2) else nvidia/cuda:12.6.3-base-ubuntu22.04 if it exists locally, + # 3) else skip with a warning (return 0). The smoke test inside kermt_run + # will catch broken GPU passthrough later anyway. + local probe_img="" + if docker image inspect "$KERMT_IMAGE" >/dev/null 2>&1; then + probe_img="$KERMT_IMAGE" + elif docker image inspect nvidia/cuda:12.6.3-base-ubuntu22.04 >/dev/null 2>&1; then + probe_img="nvidia/cuda:12.6.3-base-ubuntu22.04" + else + echo "[kermt] check_gpu: skipped — neither '$KERMT_IMAGE' nor 'nvidia/cuda:12.6.3-base-ubuntu22.04' is present locally. Run 'ensure_image' first, or this probe will be exercised by the in-container smoke test." >&2 + return 0 + fi + if ! docker run --rm --gpus all "$probe_img" nvidia-smi >/dev/null 2>&1; then + echo "[kermt] error: 'docker run --gpus all' failed (probe image: $probe_img). Install nvidia-container-toolkit and ensure the host has a CUDA-capable NVIDIA driver." >&2 + return 1 + fi +} + +# ----------------------------------------------------------------------------- +# Image build / verification +# ----------------------------------------------------------------------------- + +kermt_ensure_image() { + _kermt_require_repo || return $? + kermt_check_docker || return $? + if docker image inspect "$KERMT_IMAGE" >/dev/null 2>&1; then + local id + id=$(docker image inspect "$KERMT_IMAGE" --format '{{.Id}}' 2>/dev/null | cut -c1-19) + echo "[kermt] image '$KERMT_IMAGE' already present (${id:-unknown})" + return 0 + fi + echo "[kermt] image '$KERMT_IMAGE' not found; building from $KERMT_REPO/Dockerfile" + echo "[kermt] first build typically takes 10-20 minutes on a typical workstation; subsequent runs reuse the cached image" + docker build -t "$KERMT_IMAGE" -f "$KERMT_REPO/Dockerfile" "$KERMT_REPO" +} + +# ----------------------------------------------------------------------------- +# Mount-flag parser, internal +# ----------------------------------------------------------------------------- +# Reads flags from the caller's positional args until it hits '--', appending +# `-v src:dst[:ro]` pairs into the caller-provided array name (passed as $1). +# Returns the number of caller-provided args consumed via _kermt_consumed. +# This is bash-specific (uses nameref via `declare -n`). + +_kermt_parse_mounts() { + local -n _out="$1" + shift + _kermt_consumed=0 + while [[ $# -gt 0 ]]; do + case "$1" in + --) + return 0 + ;; + --data) + [[ -e "$2" ]] || { echo "[kermt] --data path not found: $2" >&2; return 1; } + # If the user passes a file, mount its parent directory at /data so + # downstream commands can refer to /data/. Mounting a + # single file at /data makes the path-as-directory pattern in the + # skill examples (`--csv /data/`) fail with "not found". + if [[ -d "$2" ]]; then + _out+=("-v" "$(realpath "$2"):/data:ro") + else + _out+=("-v" "$(realpath "$(dirname "$2")"):/data:ro") + fi + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --ckpt) + [[ -e "$2" ]] || { echo "[kermt] --ckpt path not found: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/ckpt:ro") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --vocab-dir) + [[ -d "$2" ]] || { echo "[kermt] --vocab-dir not found or not a directory: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/vocab:ro") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --run-dir) + mkdir -p "$2" || { echo "[kermt] failed to create --run-dir: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/runs") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --model-dir) + mkdir -p "$2" || { echo "[kermt] failed to create --model-dir: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/model") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + *) + return 0 + ;; + esac + done +} + +# ----------------------------------------------------------------------------- +# Foreground / detached run +# ----------------------------------------------------------------------------- + +# Capture host-side git state for the repo and emit `-e KERMT_REPO_COMMIT=… +# -e KERMT_REPO_DIRTY=true|false` flags. Used by the run / run_detached +# wrappers so the runner's run.json manifest gets honest commit info even +# though `git -C /workspace` inside the container fails due to bind-mount +# ownership. +_kermt_git_env_flags() { + local commit="unknown" + local dirty="false" + if command -v git >/dev/null 2>&1 && [[ -d "$KERMT_REPO/.git" ]]; then + local c + c=$(git -C "$KERMT_REPO" rev-parse HEAD 2>/dev/null) && commit="$c" + # `--untracked-files=no` filters out user-private notes (e.g. a CLAUDE.md + # or RELEASE_PLAN_v2.0.md at the repo root) that wouldn't affect + # reproducibility — only modifications to tracked files do. + if [[ -n "$(git -C "$KERMT_REPO" status --porcelain --untracked-files=no 2>/dev/null | head -n 1)" ]]; then + dirty="true" + fi + fi + printf '%s\n%s\n%s\n%s\n' "-e" "KERMT_REPO_COMMIT=$commit" "-e" "KERMT_REPO_DIRTY=$dirty" +} + +# Forward HF_TOKEN into the container when it is set, so fetch_released_model.py +# can authenticate to Hugging Face. The current release is public (no token +# needed); this only guards against shared-IP rate limits or a future gated +# repo. Emits nothing when HF_TOKEN is unset. +_kermt_hf_env_flags() { + if [[ -n "${HF_TOKEN:-}" ]]; then + printf '%s\n%s\n' "-e" "HF_TOKEN=$HF_TOKEN" + fi +} + +kermt_run() { + kermt_ensure_image || return $? + local mount_args=() + _kermt_parse_mounts mount_args "$@" || return $? + shift "$_kermt_consumed" + if [[ "${1:-}" != "--" ]]; then + echo "[kermt] expected '--' separating mount flags from the command (got '${1:-}')" >&2 + return 1 + fi + shift + if [[ $# -eq 0 ]]; then + echo "[kermt] no command supplied after '--'" >&2 + return 1 + fi + local git_args=() + while IFS= read -r line; do git_args+=("$line"); done < <(_kermt_git_env_flags) + local hf_args=() + while IFS= read -r line; do hf_args+=("$line"); done < <(_kermt_hf_env_flags) + docker run --rm --gpus "$KERMT_GPUS" \ + --user "$(id -u):$(id -g)" \ + -v "$KERMT_REPO:/workspace" \ + -v "$_kermt_bundle_dir:/skill:ro" \ + "${mount_args[@]}" \ + -w /workspace \ + -e KERMT_REPO=/workspace \ + -e PYTHONPATH=/workspace \ + -e HOME=/tmp/kermt-home \ + "${git_args[@]}" \ + "${hf_args[@]}" \ + "$KERMT_IMAGE" \ + conda run -n kermt --no-capture-output bash -c "$*" +} + +kermt_run_detached() { + kermt_ensure_image || return $? + local name="" + local mount_args=() + # Pull --name out first, then let the shared mount parser handle the rest. + while [[ $# -gt 0 ]]; do + case "$1" in + --name) name="$2"; shift 2 ;; + --) break ;; + --data|--ckpt|--vocab-dir|--run-dir|--model-dir) break ;; + *) break ;; + esac + done + _kermt_parse_mounts mount_args "$@" || return $? + shift "$_kermt_consumed" + if [[ "${1:-}" != "--" ]]; then + echo "[kermt] expected '--' separating mount flags from the command (got '${1:-}')" >&2 + return 1 + fi + shift + if [[ $# -eq 0 ]]; then + echo "[kermt] no command supplied after '--'" >&2 + return 1 + fi + if [[ -z "$name" ]]; then + name="kermt-$(date -u +%Y%m%dT%H%M%SZ)-$$" + fi + local cid + local git_args=() + while IFS= read -r line; do git_args+=("$line"); done < <(_kermt_git_env_flags) + local hf_args=() + while IFS= read -r line; do hf_args+=("$line"); done < <(_kermt_hf_env_flags) + cid=$(docker run -d --gpus "$KERMT_GPUS" \ + --user "$(id -u):$(id -g)" \ + --name "$name" \ + -v "$KERMT_REPO:/workspace" \ + -v "$_kermt_bundle_dir:/skill:ro" \ + "${mount_args[@]}" \ + -w /workspace \ + -e KERMT_REPO=/workspace \ + -e PYTHONPATH=/workspace \ + -e HOME=/tmp/kermt-home \ + "${git_args[@]}" \ + "${hf_args[@]}" \ + "$KERMT_IMAGE" \ + conda run -n kermt --no-capture-output bash -c "$*") || return $? + echo "[kermt] container started: name=$name id=$cid" + echo "[kermt] follow logs: docker logs -f $name" + echo "[kermt] wait for exit: docker wait $name" + echo "[kermt] stop: docker stop $name" + echo "$cid" +} + +# ----------------------------------------------------------------------------- +# Subcommand dispatch when invoked directly (not sourced) +# ----------------------------------------------------------------------------- + +if [[ "${BASH_SOURCE[0]:-$0}" == "${0}" ]]; then + cmd="${1:-}"; shift || true + case "$cmd" in + check_docker) kermt_check_docker "$@" ;; + check_gpu) kermt_check_gpu "$@" ;; + check_system) kermt_check_system "$@" ;; + ensure_image) kermt_ensure_image "$@" ;; + run) kermt_run "$@" ;; + run_detached) kermt_run_detached "$@" ;; + ""|-h|--help) + cat >&2 < [args...] + +Subcommands: + check_docker Verify docker is installed and the daemon is reachable. + check_gpu Verify 'docker --gpus all' works (nvidia-container-toolkit). + check_system Emit a JSON probe of host GPU + VRAM + compute_cap + + driver + disk space + container toolkit + image presence. + Exits 0 with ok=false + a 'gaps' list when anything's + below the per-workflow minimum. + ensure_image Build kermt:latest from \$KERMT_REPO/Dockerfile if missing. + run [flags] -- ... Run a command inside the container (foreground, --rm). + run_detached [flags] -- ... + Run detached; prints container name + id + log hint. + +Mount flags (for run / run_detached): + --data bind to /data (read-only) + --ckpt bind to /ckpt (read-only) + --vocab-dir bind to /vocab (read-only) + --run-dir bind to /runs (read-write; created on host if missing) + --model-dir bind to /model (read-write; released-model download target) + +Additional flags for run_detached: + --name container name (default: kermt--) + +Environment overrides: + KERMT_IMAGE default kermt:latest + KERMT_REPO checkout path; otherwise discovered above the skill or working directory + KERMT_GPUS default all +EOF + exit 1 + ;; + *) + echo "[kermt] unknown subcommand: $cmd" >&2 + echo "[kermt] run '$0 --help' for usage" >&2 + exit 1 + ;; + esac +fi diff --git a/skills/kermt-embed/scripts/prepare_data.py b/skills/kermt-embed/scripts/prepare_data.py new file mode 100644 index 0000000..f0edb3e --- /dev/null +++ b/skills/kermt-embed/scripts/prepare_data.py @@ -0,0 +1,817 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Mode-dispatched data preparation pipeline for the KERMT agent skills. + +Composes the existing repo data-prep scripts (`scripts/clean_smiles.py`, +`scripts/save_features.py`, `scripts/build_vocab.py`, `scripts/split_data.py`) +into a single one-call entry point per workflow. Output lands in `--out` with +a `prepare_data.json` manifest that the downstream runners read. + +Mode pipelines +-------------- +pretrain : clean -> (optional auto-split train into train+val by --val-frac) + -> save_features (fgtasklabel) on each CSV + -> vocab step: if --vocab-dir / --{atom,bond,smiles}-vocab given, + copy those through (continue-pretrain case — the ckpt's vocab + is authoritative); else if --skip-vocab, skip; + else build_vocab on train (pretrain-from-scratch case) + -> split_data (graph + feature shards + summary.txt) per CSV +finetune : clean each provided CSV -> (optional random split when only one + CSV is provided; emits a strong warning recommending scaffold- + balanced pre-splits) -> save_features (rdkit_2d_normalized) per CSV +inference : clean -> save_features (rdkit_2d_normalized) +embed : clean only (extract_embeddings.py featurizes on the fly) + +Output convention +----------------- +The manifest under `/prepare_data.json` captures every step's inputs, +outputs, duration, and skipped-due-to-existing flag, plus a top-level +`split_method` field (one of: "user_provided", "random", "n/a") that the +finetune runner uses to pass the correct `--split_type` to main.py. + +Subprocess composition +---------------------- +Each underlying script is invoked via `subprocess.run`. The PYTHONPATH=/workspace +env var (set by `scripts/kermt_container.sh`) makes the `kermt` package +importable inside the subprocesses; without it, build_vocab.py and split_data.py +fail with `ModuleNotFoundError: No module named 'kermt'`. + +CLI +--- + prepare_data.py --mode {pretrain|finetune|inference|embed} + --csv --out + [--val-csv ] [--test-csv ] + [--val-frac 0.1] [--test-frac 0.1] [--seed 0] + [--sample-per-file 100000] [--vocab-format json] + [--dataset-name pretrain] + [--targets COL [COL ...]] + [--features-generator ] + [--smiles-column 0] + [--force] [--skip-clean] [--skip-features] + [--skip-vocab] [--skip-split] +""" +from __future__ import annotations + +import argparse +import json +import os +import shutil +import subprocess +import sys +import time +import traceback +from pathlib import Path +from typing import Any + +import pandas as pd + +# sys.path tweak so `_utils` is importable regardless of how this script +# is invoked (kermt_run sets PYTHONPATH=/workspace; bare-Python launches +# from the host don't). +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import PRETRAIN_VOCAB_STEMS, resolve_kermt_repo, validate_vocab_file # noqa: E402 + + +REPO_ROOT = resolve_kermt_repo() +EXISTING_SCRIPTS = REPO_ROOT / "scripts" + +DEFAULT_FEATURES_GENERATOR = { + "pretrain": "fgtasklabel", + "finetune": "rdkit_2d_normalized", + "inference": "rdkit_2d_normalized", + "embed": None, # not used +} + +VALID_MODES = ("pretrain", "finetune", "inference", "embed") + + +# --------------------------------------------------------------------------- +# Subprocess helpers +# --------------------------------------------------------------------------- + +def _run(cmd: list[str], step_name: str, manifest: dict[str, Any]) -> dict[str, Any]: + """Run a subprocess, append a step entry to manifest, raise on failure.""" + step: dict[str, Any] = { + "name": step_name, + "cmd": cmd, + "duration_s": None, + "ok": False, + "stderr_tail": "", + "skipped_due_to_existing": False, + } + t0 = time.time() + proc = subprocess.run(cmd, capture_output=True, text=True) + step["duration_s"] = round(time.time() - t0, 2) + if proc.returncode != 0: + step["stderr_tail"] = (proc.stderr or "").splitlines()[-20:] + step["ok"] = False + manifest["steps"].append(step) + raise RuntimeError( + f"step '{step_name}' failed (exit {proc.returncode}); " + f"command: {' '.join(cmd)}\nstderr tail:\n" + "\n".join(step["stderr_tail"]) + ) + step["ok"] = True + manifest["steps"].append(step) + return step + + +def _skipped(step_name: str, output_path: str, manifest: dict[str, Any]) -> dict[str, Any]: + step = { + "name": step_name, + "output": output_path, + "ok": True, + "duration_s": 0.0, + "skipped_due_to_existing": True, + } + manifest["steps"].append(step) + return step + + +def _exists_nonempty(path: Path) -> bool: + """File exists with non-zero size, or directory exists with at least one entry.""" + if not path.exists(): + return False + if path.is_file(): + return path.stat().st_size > 0 + if path.is_dir(): + try: + next(path.iterdir()) + return True + except StopIteration: + return False + return False + + +# --------------------------------------------------------------------------- +# Per-script wrappers +# --------------------------------------------------------------------------- + +def _resolve_smiles_column(csv_path: Path, explicit_value: int | None) -> int: + """Return the 0-based index of the SMILES column in csv_path. + + Auto-detection rule when `explicit_value is None`: + 1. Read the CSV header (first non-empty row). + 2. Prefer an exact lowercase `smiles` column (kermt convention). + 3. Otherwise accept a single case-insensitive match + (`SMILES`, `Smiles`, etc.). + 4. If no match (or multiple ambiguous matches), raise a ValueError + that surfaces the header so the user can disambiguate via + `--smiles-column N`. + + Real datasets routinely place SMILES at column index ≠ 0 + (e.g. openadmet's all.csv has "Molecule Name" at col 0 and "SMILES" + at col 1). Auto-detection prevents the silent 0-row-clean failure + mode where every row gets rejected because col 0 doesn't parse as + a SMILES string. + """ + if explicit_value is not None: + return explicit_value + + if not csv_path.is_file(): + raise ValueError(f"input CSV not found: {csv_path}") + + import csv as _csv + with csv_path.open("r", newline="") as f: + reader = _csv.reader(f) + try: + header = next(reader) + except StopIteration: + raise ValueError(f"input CSV {csv_path} is empty") + + stripped = [c.strip() for c in header] + # Prefer exact lowercase "smiles" + exact = [i for i, c in enumerate(stripped) if c == "smiles"] + if exact: + return exact[0] + # Then case-insensitive + ci = [i for i, c in enumerate(stripped) if c.lower() == "smiles"] + if len(ci) == 1: + return ci[0] + if len(ci) > 1: + raise ValueError( + f"input CSV {csv_path} has multiple SMILES-named columns: " + f"{[header[i] for i in ci]} at indices {ci}. " + "Pass --smiles-column N (0-based) to disambiguate." + ) + raise ValueError( + f"could not auto-detect a SMILES column in {csv_path}. " + f"Header columns: {header}. " + "Pass --smiles-column N (0-based) to specify which column holds SMILES." + ) + + +def _clean_smiles( + input_csv: Path, output_csv: Path, smiles_column: int, manifest: dict[str, Any], force: bool +) -> Path: + if not force and _exists_nonempty(output_csv): + _skipped(f"clean_smiles({input_csv.name})", str(output_csv), manifest) + return output_csv + output_csv.parent.mkdir(parents=True, exist_ok=True) + if force and output_csv.exists(): + # clean_smiles.py prompts interactively (input()) when the output file + # already exists — that's an EOFError in a non-TTY subprocess. Pre-delete. + output_csv.unlink() + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "clean_smiles.py"), + "--input", str(input_csv), + "--output", str(output_csv), + "--smiles_column", str(smiles_column), + ] + _run(cmd, f"clean_smiles({input_csv.name})", manifest) + return output_csv + + +def _reduce_to_smiles_column( + csv_path: Path, smiles_column: int, manifest: dict[str, Any] +) -> Path: + """Rewrite an inference CSV to keep only the SMILES column (at index 0). + + Downstream `kermt.util.utils.get_data` -> `MoleculeDatapoint.__init__` + floats every column after SMILES, which crashes on non-numeric passthrough + columns (e.g. a 'split' label of 'train'/'val'/'test', or a 'Molecule Name' + string). Inference does not need target columns, so drop them here. + + Note on skip semantics: this step is idempotent — running it on an + already-single-column file is a no-op. We record that with + `skipped_due_to_idempotent: True`, NOT `skipped_due_to_existing: True`. + The two fields have different meanings: `_existing` means "I found a + cached output file from a prior run and reused it" (overridden by + `--force`); `_idempotent` means "the input is already in the desired + state, so re-executing changes nothing" (safe to skip even under + `--force`). + """ + step_name = f"reduce_to_smiles_only({csv_path.name})" + start = time.time() + df = pd.read_csv(csv_path) + if df.shape[1] == 1: + manifest["steps"].append({ + "name": step_name, + "output": str(csv_path), + "ok": True, + "duration_s": time.time() - start, + "skipped_due_to_idempotent": True, + "note": "already single-column", + }) + return csv_path + effective_col = smiles_column if 0 <= smiles_column < df.shape[1] else 0 + df.iloc[:, [effective_col]].to_csv(csv_path, index=False) + manifest["steps"].append({ + "name": step_name, + "output": str(csv_path), + "ok": True, + "duration_s": time.time() - start, + "input_cols": int(df.shape[1]), + "kept_col": effective_col, + "kept_col_name": str(df.columns[effective_col]), + }) + return csv_path + + +def _save_features( + csv_path: Path, npz_path: Path, generator: str, manifest: dict[str, Any], force: bool +) -> Path: + if not force and _exists_nonempty(npz_path): + _skipped(f"save_features({csv_path.name}, {generator})", str(npz_path), manifest) + return npz_path + npz_path.parent.mkdir(parents=True, exist_ok=True) + if force and npz_path.exists(): + npz_path.unlink() # --restart still loads partial state if file exists; pre-delete to be safe + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "save_features.py"), + "--data_path", str(csv_path), + "--save_path", str(npz_path), + "--features_generator", generator, + "--restart", + ] + _run(cmd, f"save_features({csv_path.name}, {generator})", manifest) + return npz_path + + +def _resolve_vocab_inputs(args: argparse.Namespace) -> dict[str, Path | None] | None: + """Returns {atom, bond, smiles}->Path|None when the user supplied vocab + inputs (via --vocab-dir or --atom-vocab/--bond-vocab/--smiles-vocab), + else None (signal to fall through to build_vocab). + + Conventional filenames inside --vocab-dir: + pretrain_atom_vocab.{json,pkl} + pretrain_bond_vocab.{json,pkl} + pretrain_smiles_vocab.pkl + """ + if args.vocab_dir: + d = Path(args.vocab_dir).resolve() + if not d.is_dir(): + raise FileNotFoundError(f"--vocab-dir not found or not a directory: {d}") + def _find(stem: str, exts: tuple[str, ...]) -> Path | None: + for ext in exts: + p = d / f"{stem}.{ext}" + if p.is_file(): + return p + return None + atom = _find(PRETRAIN_VOCAB_STEMS["atom"], ("json", "pkl")) + bond = _find(PRETRAIN_VOCAB_STEMS["bond"], ("json", "pkl")) + smiles = _find(PRETRAIN_VOCAB_STEMS["smiles"], ("pkl",)) + if atom is None and bond is None and smiles is None: + stems = [PRETRAIN_VOCAB_STEMS[k] for k in ("atom", "bond", "smiles")] + raise FileNotFoundError( + f"--vocab-dir {d} contained no {{ {', '.join(stems) }}}.{{json,pkl}} " + f"files. Expected at least {PRETRAIN_VOCAB_STEMS['atom']} + " + f"{PRETRAIN_VOCAB_STEMS['bond']}." + ) + return {"atom": atom, "bond": bond, "smiles": smiles} + + if args.atom_vocab or args.bond_vocab or args.smiles_vocab: + return { + "atom": Path(args.atom_vocab).resolve() if args.atom_vocab else None, + "bond": Path(args.bond_vocab).resolve() if args.bond_vocab else None, + "smiles": Path(args.smiles_vocab).resolve() if args.smiles_vocab else None, + } + + return None + + +def _copy_provided_vocab( + src: dict[str, Path | None], dst_dir: Path, dataset_name: str, manifest: dict[str, Any], + force: bool, +) -> dict[str, Path]: + """When the user supplies vocab files (use ckpt's vocab as-is), + copy them into `/__vocab.` so the + downstream pretrain command sees the conventional filenames. + + `src` is `{atom: Path|None, bond: Path|None, smiles: Path|None}`. The atom + and bond entries must be both present or both absent (paired). smiles is + optional (cmim/hybrid only). + + Returns the same dict of (resolved) destination paths. + """ + import shutil + if (src["atom"] is None) != (src["bond"] is None): + raise ValueError( + "vocab pass-through requires atom and bond vocab paths to be paired; " + "got atom=" + str(src["atom"]) + ", bond=" + str(src["bond"]) + ) + out: dict[str, Path] = {} + dst_dir.mkdir(parents=True, exist_ok=True) + for which, path in src.items(): + if path is None: + continue + # Validate the source file IS a loadable KERMT vocab before copying. + # Catches the "user pointed --smiles-vocab at a random pickle" case + # early, with a clear error, instead of letting it surface as a cryptic + # SMILESVocab.load_vocab failure at pretrain_ddp.py launch time. + validate_vocab_file(path, kind=which) + ext = path.suffix.lstrip(".") + if which == "smiles": + ext = "pkl" # smiles vocab is always pickle + dst = dst_dir / f"{dataset_name}_{which}_vocab.{ext}" + if not force and _exists_nonempty(dst): + _skipped(f"copy_vocab({which})", str(dst), manifest) + out[which] = dst + continue + if force and dst.exists(): + dst.unlink() + shutil.copy2(path, dst) + manifest["steps"].append({ + "name": f"copy_vocab({which})", + "src": str(path), "dst": str(dst), "ok": True, + "duration_s": 0.0, "skipped_due_to_existing": False, + }) + out[which] = dst + return out + + +def _build_vocab( + csv_path: Path, vocab_dir: Path, dataset_name: str, vocab_format: str, + manifest: dict[str, Any], force: bool, +) -> dict[str, Path]: + """Builds atom + bond (in --vocab-format) and smiles (always pickle) vocabs. + Returns a dict of {atom, bond, smiles} -> Path.""" + suffix = "json" if vocab_format == "json" else "pkl" + expected = { + "atom": vocab_dir / f"{dataset_name}_atom_vocab.{suffix}", + "bond": vocab_dir / f"{dataset_name}_bond_vocab.{suffix}", + "smiles": vocab_dir / f"{dataset_name}_smiles_vocab.pkl", + } + if not force and all(_exists_nonempty(p) for p in expected.values()): + _skipped(f"build_vocab({csv_path.name})", str(vocab_dir), manifest) + return expected + vocab_dir.mkdir(parents=True, exist_ok=True) + if force: + for p in expected.values(): + if p.exists(): + p.unlink() + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "build_vocab.py"), + "--data_path", str(csv_path), + "--vocab_save_folder", str(vocab_dir), + "--dataset_name", dataset_name, + "--vocab_format", vocab_format, + ] + _run(cmd, f"build_vocab({csv_path.name})", manifest) + return expected + + +def _split_data( + csv_path: Path, features_path: Path | None, sample_per_file: int, output_dir: Path, + manifest: dict[str, Any], force: bool, +) -> Path: + """Run split_data.py to produce shard dirs (graph/ + optionally feature/ + summary.txt).""" + summary = output_dir / "summary.txt" + if not force and _exists_nonempty(summary): + _skipped(f"split_data({csv_path.name})", str(output_dir), manifest) + return output_dir + if force and output_dir.exists(): + shutil.rmtree(output_dir) + output_dir.mkdir(parents=True, exist_ok=True) + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "split_data.py"), + "--data_path", str(csv_path), + "--sample_per_file", str(sample_per_file), + "--output_path", str(output_dir), + ] + if features_path is not None: + cmd += ["--features_path", str(features_path)] + _run(cmd, f"split_data({csv_path.name})", manifest) + return output_dir + + +# --------------------------------------------------------------------------- +# Random splitter (used only when the user supplies a single CSV) +# --------------------------------------------------------------------------- + +def _random_split_csv( + src_csv: Path, dst_csvs: dict[str, Path], fractions: dict[str, float], seed: int, + manifest: dict[str, Any], force: bool, +) -> None: + """Shuffle src_csv and partition rows into dst_csvs by fractions. + `dst_csvs` and `fractions` are dicts keyed by the split name (e.g. 'train', 'val'). + Sum of fractions must be 1.0 (within float tolerance). Writes each dst_csv with the + same header as the input.""" + step = { + "name": f"random_split({src_csv.name})", + "seed": seed, + "fractions": fractions, + "ok": False, + "duration_s": None, + "skipped_due_to_existing": False, + "row_counts": {}, + } + if not force and all(_exists_nonempty(p) for p in dst_csvs.values()): + step["skipped_due_to_existing"] = True + step["ok"] = True + manifest["steps"].append(step) + return + + if abs(sum(fractions.values()) - 1.0) > 1e-6: + raise ValueError(f"split fractions must sum to 1.0 (got {sum(fractions.values())})") + + t0 = time.time() + df = pd.read_csv(src_csv).sample(frac=1.0, random_state=seed).reset_index(drop=True) + n = len(df) + sizes: dict[str, int] = {} + remaining = n + split_names = list(fractions.keys()) + for name in split_names[:-1]: + sizes[name] = int(round(fractions[name] * n)) + remaining -= sizes[name] + sizes[split_names[-1]] = remaining + + start = 0 + for name in split_names: + dst = dst_csvs[name] + dst.parent.mkdir(parents=True, exist_ok=True) + df.iloc[start:start + sizes[name]].to_csv(dst, index=False) + step["row_counts"][name] = sizes[name] + start += sizes[name] + + step["duration_s"] = round(time.time() - t0, 2) + step["ok"] = True + manifest["steps"].append(step) + + +def _emit_random_split_warning( + src_csv: Path, fractions: dict[str, float], seed: int, manifest: dict[str, Any] +) -> None: + row_counts = manifest["steps"][-1].get("row_counts", {}) + n = sum(row_counts.values()) if row_counts else "?" + lines = [ + f"WARNING: Auto-splitting {n} rows from {src_csv.name} into:", + ] + for name, frac in fractions.items(): + cnt = row_counts.get(name, "?") + lines.append(f" {name}: {cnt} rows ({frac * 100:.1f}%)") + lines += [ + f"using random split with seed {seed}.", + "", + "This is a RANDOM split. For rigorous ADMET evaluation, scaffold-balanced", + "(or other structure-aware) splits are strongly preferred — molecules with", + "similar scaffolds can leak across splits and inflate apparent generalization.", + "", + "To use your own pre-computed splits instead, pass:", + " --train-csv --val-csv --test-csv ", + "", + "To customize fractions:", + " --val-frac 0.15 --test-frac 0.15", + ] + warning = "\n".join(lines) + print(warning, file=sys.stderr) + manifest["warnings"].append(warning) + + +# --------------------------------------------------------------------------- +# Mode pipelines +# --------------------------------------------------------------------------- + +def _prepare_embed(args, out: Path, manifest: dict[str, Any]) -> None: + manifest["split_method"] = "n/a" + if args.skip_clean: + clean = Path(args.csv) + manifest["steps"].append({"name": "clean_smiles", "skipped_by_flag": True, "ok": True}) + else: + clean = _clean_smiles(Path(args.csv), out / "clean.csv", args.smiles_column, manifest, args.force) + manifest["outputs"]["clean_csv"] = str(clean) + + +def _prepare_inference(args, out: Path, manifest: dict[str, Any]) -> None: + manifest["split_method"] = "n/a" + clean = _clean_smiles(Path(args.csv), out / "clean.csv", args.smiles_column, manifest, args.force) + # Reduce to SMILES-only: downstream get_data/MoleculeDatapoint floats every + # non-SMILES column, which crashes on non-numeric passthrough columns + # (e.g. a 'split' label). Inference does not need target columns. + _reduce_to_smiles_column(clean, args.smiles_column, manifest) + manifest["outputs"]["clean_csv"] = str(clean) + if args.skip_features: + manifest["steps"].append({"name": "save_features", "skipped_by_flag": True, "ok": True}) + return + generator = args.features_generator or DEFAULT_FEATURES_GENERATOR["inference"] + npz = _save_features(clean, out / "clean.npz", generator, manifest, args.force) + manifest["outputs"]["clean_npz"] = str(npz) + + +def _prepare_finetune(args, out: Path, manifest: dict[str, Any]) -> None: + src_train = Path(args.csv) + has_val = args.val_csv is not None + has_test = args.test_csv is not None + split_type = args.split_type + + if has_val and has_test: + # User supplied explicit val + test CSVs: trust them, just clean + featurize. + # split_type is irrelevant when val/test are given separately. + manifest["split_method"] = "user_provided" + clean_train = _clean_smiles(src_train, out / "clean_train.csv", args.smiles_column, manifest, args.force) + clean_val = _clean_smiles(Path(args.val_csv), out / "clean_val.csv", args.smiles_column, manifest, args.force) + clean_test = _clean_smiles(Path(args.test_csv), out / "clean_test.csv", args.smiles_column, manifest, args.force) + manifest["outputs"]["clean_train_csv"] = str(clean_train) + manifest["outputs"]["clean_val_csv"] = str(clean_val) + manifest["outputs"]["clean_test_csv"] = str(clean_test) + per_split = (("train", clean_train), ("val", clean_val), ("test", clean_test)) + elif has_val or has_test: + raise ValueError( + "for finetune mode, either provide BOTH --val-csv and --test-csv (user-provided splits) " + "or NEITHER (run with --split-type {random|scaffold_balanced|index_predetermined}). " + "Got one but not both." + ) + elif split_type == "random": + # Random auto-split — done here in prep so train.py gets ready-made CSVs. + manifest["split_method"] = "random" + manifest["split_seed"] = args.seed + train_frac = max(0.0, 1.0 - args.val_frac - args.test_frac) + manifest["split_fractions"] = {"train": train_frac, "val": args.val_frac, "test": args.test_frac} + clean_full = _clean_smiles(src_train, out / "_clean_full.csv", args.smiles_column, manifest, args.force) + dst = { + "train": out / "clean_train.csv", + "val": out / "clean_val.csv", + "test": out / "clean_test.csv", + } + _random_split_csv(clean_full, dst, manifest["split_fractions"], args.seed, manifest, args.force) + clean_train, clean_val, clean_test = dst["train"], dst["val"], dst["test"] + manifest["outputs"]["clean_train_csv"] = str(clean_train) + manifest["outputs"]["clean_val_csv"] = str(clean_val) + manifest["outputs"]["clean_test_csv"] = str(clean_test) + _emit_random_split_warning(src_train, manifest["split_fractions"], args.seed, manifest) + per_split = (("train", clean_train), ("val", clean_val), ("test", clean_test)) + else: + # Scaffold-balanced or index-predetermined: prep cleans + featurizes the full + # CSV and defers actual splitting to task/train.py, which calls split_data + # with the user-supplied seed and split_sizes. + manifest["split_method"] = "deferred_to_runner" + manifest["split_type"] = split_type + manifest["split_seed"] = args.seed + manifest["split_fractions"] = { + "train": max(0.0, 1.0 - args.val_frac - args.test_frac), + "val": args.val_frac, + "test": args.test_frac, + } + clean_full = _clean_smiles(src_train, out / "clean_full.csv", args.smiles_column, manifest, args.force) + manifest["outputs"]["clean_full_csv"] = str(clean_full) + per_split = (("full", clean_full),) + + if args.skip_features: + manifest["steps"].append({"name": "save_features", "skipped_by_flag": True, "ok": True}) + return + generator = args.features_generator or DEFAULT_FEATURES_GENERATOR["finetune"] + for split_name, csv in per_split: + npz = _save_features(csv, csv.with_suffix(".npz"), generator, manifest, args.force) + manifest["outputs"][f"clean_{split_name}_npz"] = str(npz) + + +def _prepare_pretrain(args, out: Path, manifest: dict[str, Any]) -> None: + src_train = Path(args.csv) + if args.val_csv is not None: + manifest["split_method"] = "user_provided" + clean_train = _clean_smiles(src_train, out / "clean_train.csv", args.smiles_column, manifest, args.force) + clean_val = _clean_smiles(Path(args.val_csv), out / "clean_val.csv", args.smiles_column, manifest, args.force) + else: + manifest["split_method"] = "random" + manifest["split_seed"] = args.seed + train_frac = max(0.0, 1.0 - args.val_frac) + manifest["split_fractions"] = {"train": train_frac, "val": args.val_frac} + clean_full = _clean_smiles(src_train, out / "_clean_full.csv", args.smiles_column, manifest, args.force) + dst = {"train": out / "clean_train.csv", "val": out / "clean_val.csv"} + _random_split_csv(clean_full, dst, manifest["split_fractions"], args.seed, manifest, args.force) + clean_train, clean_val = dst["train"], dst["val"] + + manifest["outputs"]["clean_train_csv"] = str(clean_train) + manifest["outputs"]["clean_val_csv"] = str(clean_val) + + generator = args.features_generator or DEFAULT_FEATURES_GENERATOR["pretrain"] + if args.skip_features: + manifest["steps"].append({"name": "save_features", "skipped_by_flag": True, "ok": True}) + train_npz: Path | None = None + val_npz: Path | None = None + else: + train_npz = _save_features(clean_train, out / "clean_train.npz", generator, manifest, args.force) + val_npz = _save_features(clean_val, out / "clean_val.npz", generator, manifest, args.force) + manifest["outputs"]["clean_train_npz"] = str(train_npz) + manifest["outputs"]["clean_val_npz"] = str(val_npz) + + if args.skip_vocab: + manifest["steps"].append({"name": "build_vocab", "skipped_by_flag": True, "ok": True}) + manifest["vocab_source"] = "skipped" + else: + # Resolve user-provided vocab paths from --vocab-dir or explicit flags. + provided = _resolve_vocab_inputs(args) + if provided: + # Use the user-supplied (ckpt's) vocab as-is. Copy into the + # conventional filenames the downstream pretrain command expects. + vocabs = _copy_provided_vocab(provided, out, args.dataset_name, manifest, args.force) + manifest["vocab_source"] = "user_provided" + else: + # Fall back to the existing build-from-corpus behavior. Used by + # pretrain-from-scratch and by any continue case where the user + # explicitly wants a fresh vocab (rare, usually wrong). + vocabs = _build_vocab(clean_train, out, args.dataset_name, args.vocab_format, manifest, args.force) + manifest["vocab_source"] = "built_fresh" + if "atom" in vocabs: + manifest["outputs"]["atom_vocab"] = str(vocabs["atom"]) + if "bond" in vocabs: + manifest["outputs"]["bond_vocab"] = str(vocabs["bond"]) + if "smiles" in vocabs: + manifest["outputs"]["smiles_vocab"] = str(vocabs["smiles"]) + + if args.skip_split: + manifest["steps"].append({"name": "split_data", "skipped_by_flag": True, "ok": True}) + else: + train_dir = _split_data(clean_train, train_npz, args.sample_per_file, out / "train", manifest, args.force) + val_dir = _split_data(clean_val, val_npz, args.sample_per_file, out / "val", manifest, args.force) + manifest["outputs"]["train_dir"] = str(train_dir) + manifest["outputs"]["val_dir"] = str(val_dir) + + +# --------------------------------------------------------------------------- +# Entry point +# --------------------------------------------------------------------------- + +def prepare(args: argparse.Namespace) -> dict[str, Any]: + out = Path(args.out).resolve() + out.mkdir(parents=True, exist_ok=True) + manifest: dict[str, Any] = { + "mode": args.mode, + "input_csv": str(Path(args.csv).resolve()), + "val_csv": str(Path(args.val_csv).resolve()) if args.val_csv else None, + "test_csv": str(Path(args.test_csv).resolve()) if args.test_csv else None, + "output_dir": str(out), + "split_method": None, + "steps": [], + "outputs": {}, + "errors": [], + "warnings": [], + } + try: + if args.mode == "pretrain": + _prepare_pretrain(args, out, manifest) + elif args.mode == "finetune": + _prepare_finetune(args, out, manifest) + elif args.mode == "inference": + _prepare_inference(args, out, manifest) + elif args.mode == "embed": + _prepare_embed(args, out, manifest) + manifest["ok"] = True + except Exception as exc: # noqa: BLE001 + manifest["ok"] = False + manifest["errors"].append(f"{type(exc).__name__}: {exc}") + # Always write the manifest so partial-failure state is visible to the agent. + (out / "prepare_data.json").write_text(json.dumps(manifest, indent=2)) + return manifest + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Mode-dispatched data prep for the KERMT agent skills.") + p.add_argument("--mode", required=True, choices=VALID_MODES) + p.add_argument("--csv", required=True, help="Primary input CSV (train CSV for pretrain/finetune)") + p.add_argument("--out", required=True, help="Output directory") + p.add_argument("--val-csv", default=None, help="Optional separate val CSV (pretrain/finetune)") + p.add_argument("--test-csv", default=None, help="Optional separate test CSV (finetune only)") + p.add_argument("--val-frac", type=float, default=0.1, help="Auto-split val fraction (default 0.1)") + p.add_argument("--test-frac", type=float, default=0.1, help="Auto-split test fraction (finetune only, default 0.1)") + p.add_argument("--seed", type=int, default=0, help="Random split seed (default 0)") + p.add_argument("--split-type", choices=["random", "scaffold_balanced", "index_predetermined"], + default="random", + help="(finetune only, when --val-csv/--test-csv are not given) how to split. " + "'random' splits in prep using --val-frac/--test-frac/--seed. " + "'scaffold_balanced' and 'index_predetermined' defer the actual split to the " + "runner (task/train.py invokes split_data with the appropriate algorithm " + "using the user-supplied seed); prep only cleans + featurizes the full CSV.") + p.add_argument("--sample-per-file", type=int, default=100_000, + help="split_data shard size (pretrain only, default 100000)") + p.add_argument("--vocab-format", choices=["json", "pkl"], default="json", + help="atom/bond vocab format (default json); smiles vocab is always pkl") + # Vocab pass-through (pretrain mode): when continuing from a released ckpt, + # pass its bundled vocab files in so we don't rebuild a mismatched vocab. + p.add_argument("--vocab-dir", default=None, + help="(pretrain) directory containing pretrain_{atom,bond}_vocab.{json,pkl} " + "(+ pretrain_smiles_vocab.pkl for cmim/hybrid). When given, prepare_data " + "skips build_vocab and copies these files into the output dir under the " + "expected filenames. Used by kermt-continue-pretrain to bind the released " + "ckpt's vocab to the new corpus (the ckpt's vocab is authoritative).") + p.add_argument("--atom-vocab", default=None, + help="(pretrain) explicit atom vocab path; pairs with --bond-vocab. Overrides " + "--vocab-dir's pretrain_atom_vocab.* discovery if both are given.") + p.add_argument("--bond-vocab", default=None, + help="(pretrain) explicit bond vocab path; pairs with --atom-vocab.") + p.add_argument("--smiles-vocab", default=None, + help="(pretrain, cmim/hybrid) explicit smiles vocab .pkl path. Optional for " + "vocab-only pretrain.") + p.add_argument("--dataset-name", default="pretrain", + help="vocab filename prefix (default 'pretrain' so downstream pretrain commands " + "can reference pretrain_{atom,bond}_vocab.{json|pkl}, pretrain_smiles_vocab.pkl)") + p.add_argument("--targets", nargs="+", default=None, + help="(finetune only) target column names; forwarded to the finetune runner via the manifest") + p.add_argument("--features-generator", default=None, + help="Override the per-mode default (pretrain: fgtasklabel; finetune/inference: rdkit_2d_normalized)") + p.add_argument("--smiles-column", type=int, default=None, + help="0-based column index of SMILES in the input CSV. " + "When omitted, auto-detected by header name " + "(prefers lowercase `smiles`; accepts case-insensitive " + "`SMILES`/`Smiles`). Pass explicitly to override.") + p.add_argument("--force", action="store_true", + help="Re-run every step even if its outputs already exist") + p.add_argument("--skip-clean", action="store_true", help="(embed mode) skip the cleaning step") + p.add_argument("--skip-features", action="store_true", help="Skip feature generation") + p.add_argument("--skip-vocab", action="store_true", help="(pretrain) skip vocab build") + p.add_argument("--skip-split", action="store_true", help="(pretrain) skip shard split") + args = p.parse_args(argv) + + # Forward --targets through the manifest so the finetune runner can see them. + if args.mode == "finetune" and args.targets: + pass # captured in manifest below + + # Resolve the SMILES column index (auto-detect from header when the user + # didn't pass --smiles-column). This is the only point where args.csv is + # touched before downstream _clean_smiles calls fan it out. + try: + resolved_smiles_col = _resolve_smiles_column(Path(args.csv), args.smiles_column) + except ValueError as exc: + err_manifest = { + "ok": False, + "mode": args.mode, + "errors": [f"smiles-column resolution failed: {exc}"], + } + Path(args.out).mkdir(parents=True, exist_ok=True) + (Path(args.out) / "prepare_data.json").write_text(json.dumps(err_manifest, indent=2)) + print(json.dumps(err_manifest, indent=2)) + return 1 + if args.smiles_column is None: + print(f"[prepare_data] auto-detected --smiles-column {resolved_smiles_col} " + f"from {Path(args.csv).name} header", file=sys.stderr) + args.smiles_column = resolved_smiles_col + + try: + manifest = prepare(args) + except Exception as exc: # noqa: BLE001 + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({"ok": False, "errors": [f"unhandled: {type(exc).__name__}: {exc}"]}, indent=2)) + return 1 + if args.targets: + manifest["targets"] = list(args.targets) + # Record the resolved SMILES column so the manifest is self-describing. + manifest["smiles_column"] = args.smiles_column + (Path(args.out) / "prepare_data.json").write_text(json.dumps(manifest, indent=2)) + print(json.dumps(manifest, indent=2)) + return 0 if manifest.get("ok") else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-embed/scripts/run_extract_embeddings.py b/skills/kermt-embed/scripts/run_extract_embeddings.py new file mode 100644 index 0000000..7ba118c --- /dev/null +++ b/skills/kermt-embed/scripts/run_extract_embeddings.py @@ -0,0 +1,221 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Workstation embedding-extraction runner — wraps task/extract_embeddings.py +and emits a reproducible `run.json` manifest alongside the outputs. + +Accepts any encoder-bearing ckpt (grover_base / cmim / hybrid / finetuned). +Validates via check_checkpoint.py --mode embed (encoder-only sufficient). +Reads the prepare_data manifest (mode=embed: clean CSV only — featurization +happens on the fly inside extract_embeddings.py). + +Output layout: + /out/atom_from_atom.npy + /out/bond_from_atom.npy + /out/atom_from_bond.npy + /out/bond_from_bond.npy + /out/canonical_smiles.npy + /out/validity.npy + +Blocking-by-default. Embedding extraction is minutes-scale. + +CLI +--- + run_extract_embeddings.py + --ckpt # required + --prepare-manifest # prepare_data.json (mode=embed) + --out # output dir + [--ckpt-validator-out ] + [--gpus 0] # single GPU id (default 0) + [--batch-size N] # override defaults_embed.runtime.batch_size + [--dry-run] +""" +from __future__ import annotations + +import argparse +import datetime +import json +import os +import subprocess +import sys +from pathlib import Path +from typing import Any + +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import ( # noqa: E402 + resolve_kermt_repo, assert_prepare_manifest_basics, docker_image_digest, format_cmd_replay, + git_commit_with_env_override, load_json, merge_default_into_applied, + resolve_single_gpu, run_checkpoint_validator, +) + + +REPO_ROOT = resolve_kermt_repo() +SKILL_ROOT = Path(__file__).resolve().parent.parent +DEFAULTS_PATH = SKILL_ROOT / "config" / "defaults_embed.json" +CHECK_CHECKPOINT_PATH = SKILL_ROOT / "scripts" / "check_checkpoint.py" +EXTRACT_EMBEDDINGS_PATH = REPO_ROOT / "task" / "extract_embeddings.py" + + +def _verify_prepare_manifest(manifest: dict[str, Any]) -> None: + assert_prepare_manifest_basics(manifest, "embed") + out = manifest.get("outputs", {}) + if "clean_csv" not in out: + raise ValueError( + "prepare_data manifest is missing required output 'clean_csv'. " + "Was prepare_data run successfully?" + ) + + +def _apply_defaults(args: argparse.Namespace, defaults: dict[str, Any]) -> dict[str, dict[str, Any]]: + """Merge defaults_embed.json with CLI overrides.""" + applied: dict[str, dict[str, Any]] = {} + runtime = defaults.get("runtime", {}) + + merge_default_into_applied(applied, args, "batch_size", runtime) + + return applied + + +def _build_argv( + *, gpu: int, ckpt: Path, manifest: dict[str, Any], out_dir: Path, + applied: dict[str, dict[str, Any]], +) -> list[str]: + """Constructs the task/extract_embeddings.py argv. Note: extract_embeddings + uses its own CLI (--checkpoint, --input_file, --output_path) — NOT main.py. + --format defaults to npy inside extract_embeddings.py, so we don't pass it.""" + outputs_dir = out_dir / "out" + outputs_dir.mkdir(parents=True, exist_ok=True) + + argv: list[str] = [sys.executable, "-u", str(EXTRACT_EMBEDDINGS_PATH)] + argv += ["--checkpoint", str(ckpt)] + argv += ["--input_file", manifest["outputs"]["clean_csv"]] + argv += ["--output_path", str(outputs_dir)] + argv += ["--device", "cuda"] + # clean_smiles.py preserves the input CSV's column layout, so the cleaned + # CSV has SMILES at whichever column the input had. Forward the + # auto-detected smiles_column from prepare_data.json so extract_embeddings + # doesn't fall back to its default of column 0 (which would index a + # non-SMILES column for inputs like openadmet/all.csv where SMILES is at + # column 1). Default to 0 if the manifest is from an older prepare_data + # version that didn't record the field. + argv += ["--smiles_column", str(manifest.get("smiles_column", 0))] + if "batch_size" in applied: + argv += ["--batch_size", str(applied["batch_size"]["value"])] + + return argv + + +# --------------------------------------------------------------------------- +# Main flow +# --------------------------------------------------------------------------- + +def run(args: argparse.Namespace) -> dict[str, Any]: + out_dir = Path(args.out).resolve() + out_dir.mkdir(parents=True, exist_ok=True) + (out_dir / "logs").mkdir(parents=True, exist_ok=True) + (out_dir / "out").mkdir(parents=True, exist_ok=True) + + defaults = load_json(DEFAULTS_PATH, name="defaults_embed.json") + prep_manifest_path = Path(args.prepare_manifest).resolve() + manifest = load_json(prep_manifest_path, name="prepare_data.json") + _verify_prepare_manifest(manifest) + + ckpt = Path(args.ckpt).resolve() + if args.ckpt_validator_out: + validator_out = load_json(Path(args.ckpt_validator_out), name="ckpt validator output") + else: + validator_out = run_checkpoint_validator(ckpt, mode="embed", script_path=CHECK_CHECKPOINT_PATH) + if not validator_out.get("ok"): + raise ValueError( + f"check_checkpoint.py rejected the input ckpt: {validator_out.get('errors')}" + ) + + model_type = validator_out.get("model_type") + arch = validator_out.get("arch", {}) + + gpu = resolve_single_gpu(args.gpus, workflow="embed") + applied = _apply_defaults(args, defaults) + + argv = _build_argv(gpu=gpu, ckpt=ckpt, manifest=manifest, out_dir=out_dir, applied=applied) + + commit, dirty = git_commit_with_env_override(REPO_ROOT) + image_tag = os.environ.get("KERMT_IMAGE", "kermt:latest") + image_digest = docker_image_digest(image_tag) + cmd_replay = format_cmd_replay(argv, env={"CUDA_VISIBLE_DEVICES": gpu}) + run_manifest: dict[str, Any] = { + "workflow": "embed", + "started_at": datetime.datetime.now(datetime.timezone.utc).isoformat(), + "container": {"image_tag": image_tag, "image_digest": image_digest}, + "repo": {"commit": commit, "dirty": dirty}, + "inputs": { + "ckpt": str(ckpt), + "prepare_data_manifest": str(prep_manifest_path), + "ckpt_validator_out": ( + str(Path(args.ckpt_validator_out).resolve()) if args.ckpt_validator_out else None + ), + }, + "model_type": model_type, + "gpu": gpu, + "args_applied": applied, + "arch": arch, + "output_dir": str(out_dir / "out"), + "logs_dir": str(out_dir / "logs"), + "argv": argv, + "cmd_replay": cmd_replay, + "ok_to_replay": (not dirty) and (commit != "unknown"), + "dry_run": bool(args.dry_run), + } + (out_dir / "run.json").write_text(json.dumps(run_manifest, indent=2)) + + if args.dry_run: + run_manifest["status"] = "dry_run" + return run_manifest + + env = os.environ.copy() + env["CUDA_VISIBLE_DEVICES"] = str(gpu) + log_file = out_dir / "logs" / "embed.log" + with log_file.open("w") as logf: + proc = subprocess.run(argv, env=env, stdout=logf, stderr=subprocess.STDOUT) + run_manifest["exit_code"] = proc.returncode + run_manifest["status"] = "ok" if proc.returncode == 0 else "failed" + (out_dir / "run.json").write_text(json.dumps(run_manifest, indent=2)) + return run_manifest + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser( + description="Workstation embedding-extraction runner — wraps task/extract_embeddings.py." + ) + p.add_argument("--ckpt", required=True, help="Path to encoder-bearing ckpt (any model_type).") + p.add_argument("--prepare-manifest", required=True, + help="Path to a prepare_data.json produced with --mode embed.") + p.add_argument("--out", required=True, help="Output run directory.") + p.add_argument("--ckpt-validator-out", default=None, + help="Optional cached check_checkpoint.py JSON.") + p.add_argument("--gpus", default=None, help="Single GPU id (default 0). Multi-GPU rejected.") + p.add_argument("--dry-run", action="store_true") + + p.add_argument("--batch-size", type=int, default=None) + + args = p.parse_args(argv) + + try: + manifest = run(args) + except (FileNotFoundError, ValueError, RuntimeError) as exc: + print(json.dumps({"ok": False, "errors": [f"{type(exc).__name__}: {exc}"]}, indent=2)) + return 1 + except Exception as exc: # noqa: BLE001 + import traceback + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({"ok": False, "errors": [f"unhandled: {type(exc).__name__}: {exc}"]}, + indent=2)) + return 1 + + print(json.dumps({"ok": True, "manifest": manifest}, indent=2)) + return 0 if manifest.get("status") != "failed" else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/agent/skills/kermt-embed/skill-card.md b/skills/kermt-embed/skill-card.md similarity index 100% rename from agent/skills/kermt-embed/skill-card.md rename to skills/kermt-embed/skill-card.md diff --git a/agent/skills/kermt-finetune/SKILL.md b/skills/kermt-finetune/SKILL.md similarity index 91% rename from agent/skills/kermt-finetune/SKILL.md rename to skills/kermt-finetune/SKILL.md index 9850096..3a80a49 100644 --- a/agent/skills/kermt-finetune/SKILL.md +++ b/skills/kermt-finetune/SKILL.md @@ -1,6 +1,6 @@ --- name: kermt-finetune -description: Finetune a pretrained KERMT encoder on a labeled CSV. The skill validates the input checkpoint (must be a pretrain ckpt — grover_base / cmim / hybrid), validates the labeled CSV, prepares the data (clean + features + optional split), then launches main.py finetune inside the kermt container (detached for hours-scale runs). Hyperparameters come from agent/config/defaults_finetune.json with per-flag CLI override. +description: Finetune a pretrained KERMT encoder on a labeled CSV. The skill validates the input checkpoint (must be a pretrain ckpt — grover_base / cmim / hybrid), validates the labeled CSV, prepares the data (clean + features + optional split), then launches main.py finetune inside the kermt container (detached for hours-scale runs). Hyperparameters come from config/defaults_finetune.json with per-flag CLI override. license: Apache-2.0 compatibility: Requires docker, nvidia-container-toolkit, and a CUDA-capable NVIDIA GPU. Designed for Claude Code, Codex, and Nemotron. metadata: @@ -17,6 +17,15 @@ Finetune a pretrained KERMT encoder on a user-supplied labeled CSV. The skill is the workflow orchestrator: validate ckpt, validate data, prepare data, launch the runner detached, return a run directory + container name. +## Skill and runtime paths + +Set `SKILL_DIR` to the directory containing this `SKILL.md`. Export +`KERMT_REPO` as the absolute path to the KERMT checkout used for model +execution. The bundled container helper mounts that checkout at +`/workspace` and this skill at `/skill` (read-only). Commands inside +the container use `/skill/scripts/`; defaults are bundled in `config/`. +See [Released models](references/released-models.md) for checkpoint bundle requirements. + ## Hardware requirements - **GPUs**: 1 by default (single-GPU); pass `--gpus 0` (or whichever id) to @@ -78,7 +87,7 @@ Optional: `--final-lr F` / `--warmup-epochs F` / `--weight-decay F` / `--dropout F` / `--bond-drop-rate F` / `--dist-coff F` / `--early-stop-epoch N` / `--seed N` — training-hyperparameter overrides. Anything not given is - filled from `agent/config/defaults_finetune.json`. + filled from `config/defaults_finetune.json`. - `--ffn-hidden-size N` / `--ffn-num-layers N` — shared FFN trunk dims. - `--ffn-num-task-specific-layers N` / `--ffn-task-specific-hidden-size H` — per-target FFN heads (default 0 = off; useful for heterogeneous multi-target @@ -101,7 +110,7 @@ helper bind-mounts them at known container paths. 1. **Pre-flight: ensure container + system probe.** ``` - $KERMT_REPO/agent/scripts/kermt_container.sh check_system | python -c " + "$SKILL_DIR/scripts/kermt_container.sh" check_system | python -c " import json, sys; d = json.load(sys.stdin) if not d['ok']: print('System check failed:', d['gaps']); sys.exit(1) @@ -129,24 +138,24 @@ helper bind-mounts them at known container paths. `--model-dir ` if given. An already-complete bundle is reused. - **Download** (foreground; ~282 MB on first fetch): ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run --model-dir -- \ - "python agent/scripts/fetch_released_model.py --out /model" + "$SKILL_DIR/scripts/kermt_container.sh" run --model-dir -- \ + "python /skill/scripts/fetch_released_model.py --out /model" ``` Parse the JSON; abort on `ok: false` (surface `errors`). On success set ` = /kermt_contrastive_v2.0.pt`. **Validate** the resolved (or user-provided) ckpt: ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run --ckpt -- \ - "python agent/scripts/check_checkpoint.py --mode finetune_init --ckpt /ckpt" + "$SKILL_DIR/scripts/kermt_container.sh" run --ckpt -- \ + "python /skill/scripts/check_checkpoint.py --mode finetune_init --ckpt /ckpt" ``` Parse the JSON. Abort on `ok: false`. The validator rejects already- finetuned ckpts (`has_task_ffn: true`) with a redirect to `kermt-infer`. 4. **Validate the data.** ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run --data -- \ - "python agent/scripts/check_data.py --mode finetune --csv /data/ [--targets COL1 COL2 ...]" + "$SKILL_DIR/scripts/kermt_container.sh" run --data -- \ + "python /skill/scripts/check_data.py --mode finetune --csv /data/ [--targets COL1 COL2 ...]" ``` If `--targets` was not given by the user, surface `auto_detected_targets` from the JSON and ask the user to confirm before continuing. Abort on @@ -184,8 +193,8 @@ helper bind-mounts them at known container paths. on a directory rather than a file. ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run --data --run-dir $RUN_DIR -- \ - "python agent/scripts/prepare_data.py --mode finetune \\ + "$SKILL_DIR/scripts/kermt_container.sh" run --data --run-dir $RUN_DIR -- \ + "python /skill/scripts/prepare_data.py --mode finetune \\ --csv /data/ --out /runs/data \\ --split-type \\ [--val-csv /data/ --test-csv /data/] \\ @@ -220,10 +229,10 @@ helper bind-mounts them at known container paths. 8. **Launch the runner detached.** (Consistent with the pretrain skills.) ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run_detached \\ + "$SKILL_DIR/scripts/kermt_container.sh" run_detached \\ --name kermt-finetune- \\ --ckpt --run-dir $RUN_DIR -- \\ - "python agent/scripts/run_finetune_local.py \\ + "python /skill/scripts/run_finetune_local.py \\ --ckpt /ckpt \\ --prepare-manifest /runs/data/prepare_data.json \\ --dataset-type \\ diff --git a/skills/kermt-finetune/config/defaults_finetune.json b/skills/kermt-finetune/config/defaults_finetune.json new file mode 100644 index 0000000..031c124 --- /dev/null +++ b/skills/kermt-finetune/config/defaults_finetune.json @@ -0,0 +1,41 @@ +{ + "_about": "Default hyperparameters applied by kermt-finetune. Values target a scaffold-balanced single-task regression finetune (the common ADMET case). Epochs are set to 30 for a faster default; raise via --epochs for a real training run. The skill echoes the applied set back to the user on every invocation; override any value with the corresponding CLI flag.", + + "training": { + "_about": "Optimizer, training schedule, and data-side regularization.", + "epochs": 30, + "batch_size": 32, + "init_lr": 1e-4, + "max_lr": 1e-4, + "final_lr": 2e-5, + "dropout": 0.0, + "bond_drop_rate": 0.1, + "dist_coff": 0.15, + "seed": 0, + "tensorboard": true + }, + + "task": { + "_about": "Task semantics. For classification tasks override dataset_type ('classification'), metric (e.g. 'auc'), and any optimizer settings that need to differ.", + "dataset_type": "regression", + "metric": "mae", + "split_type": "scaffold_balanced", + "ensemble_size": 1, + "num_folds": 1, + "no_features_scaling": true + }, + + "ffn_head": { + "_about": "Dimensions of the FFN head attached on top of the encoder for the downstream task. These size new modules and do not need to match the encoder hidden_size. Per-target MTL FFN heads are off by default (ffn_num_task_specific_layers=0 == shared FFN trunk only); set ffn_num_task_specific_layers>0 plus a matching ffn_task_specific_hidden_size to give each target its own small head on top of the shared trunk.", + "ffn_hidden_size": 700, + "ffn_num_layers": 3, + "ffn_num_task_specific_layers": 0, + "ffn_task_specific_hidden_size": null + }, + + "_about_arch": "Encoder architecture and arch-coupled flags (including self_attention) are intentionally not in this file. The runner reads them from the loaded checkpoint via check_checkpoint.py and uses the ckpt-derived values; user-supplied flags that mismatch the ckpt are rejected with a clear error.", + + "_about_mtl": "Per-target task-specific FFN heads are exposed via ffn_num_task_specific_layers (default 0 == off) and ffn_task_specific_hidden_size (default null; required when ffn_num_task_specific_layers > 0). When >0, each target column gets its own N-layer FFN stacked on the shared trunk; useful for heterogeneous multi-target finetunes (e.g. solubility + permeability + metabolic stability together). Override via --ffn-num-task-specific-layers N --ffn-task-specific-hidden-size H.", + + "_about_gpu_selection": "GPU selection is auto-detected at runtime, not a default here. Finetune defaults to GPU 0. Use --gpus to pick a single device for single-GPU finetune; use --num-gpus N for multi-GPU data-parallel (DDP)." +} diff --git a/skills/kermt-finetune/config/released_model.json b/skills/kermt-finetune/config/released_model.json new file mode 100644 index 0000000..1e19fc1 --- /dev/null +++ b/skills/kermt-finetune/config/released_model.json @@ -0,0 +1,13 @@ +{ + "repo_id": "nvidia/NV-KERMT-70M-v2", + "revision": "7df5eb3179235fdea1e8124db73215da33d77dce", + "ckpt_name": "kermt_contrastive_v2.0.pt", + "vocab_files": [ + "pretrain_atom_vocab.json", + "pretrain_bond_vocab.json", + "pretrain_smiles_vocab.pkl" + ], + "model_type": "hybrid", + "license": "NVIDIA Open Model License", + "license_url": "https://huggingface.co/nvidia/NV-KERMT-70M-v2" +} diff --git a/agent/skills/kermt-finetune/evals/evals.json b/skills/kermt-finetune/evals/evals.json similarity index 100% rename from agent/skills/kermt-finetune/evals/evals.json rename to skills/kermt-finetune/evals/evals.json diff --git a/skills/kermt-finetune/references/released-models.md b/skills/kermt-finetune/references/released-models.md new file mode 100644 index 0000000..e748c7e --- /dev/null +++ b/skills/kermt-finetune/references/released-models.md @@ -0,0 +1,34 @@ +# Released KERMT models + +Each released KERMT checkpoint is distributed as a **directory bundle** +containing the ckpt itself plus its vocab files: + +``` +/ +├── last_checkpoint.pt +├── pretrain_atom_vocab.{json,pkl} # either extension; pkl in current releases +├── pretrain_bond_vocab.{json,pkl} # either extension; pkl in current releases +└── pretrain_smiles_vocab.pkl # only for cmim / hybrid ckpts (pickle-only) +``` + +If you're upgrading a grover_base ckpt to hybrid with +`kermt-add-cmim-pretrain`, the +upgrade step builds a fresh `pretrain_smiles_vocab.pkl` from your +pretrain corpus — released bundles only ship the smiles vocab for +already-cmim / already-hybrid ckpts. + +The vocab files are an inseparable part of the released model — the ckpt's +vocab head dimensions are fixed at training time and only match these specific +vocab files. `kermt-continue-pretrain` treats the released ckpt's vocab as +authoritative: new corpora are tokenized through it rather than producing a +new vocab that would mismatch the ckpt's heads. + +The skill auto-detects the three vocab files in the ckpt's parent directory +and passes them through `prepare_data.py --vocab-dir`. If the bundle is +incomplete (or the user has the ckpt alone), the skill asks for the +`--vocab-dir` path; if the user can't provide one, the skill refuses to +proceed and suggests `kermt-pretrain-scratch` instead. + +To train a model on a corpus the released vocab can't cover, use +`kermt-pretrain-scratch` — the new vocab is built from the corpus and the +model is initialized fresh (no warm start; days-scale to converge). diff --git a/skills/kermt-finetune/scripts/_utils.py b/skills/kermt-finetune/scripts/_utils.py new file mode 100644 index 0000000..5bde460 --- /dev/null +++ b/skills/kermt-finetune/scripts/_utils.py @@ -0,0 +1,272 @@ +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Shared utilities for the agent scripts. + +Kept intentionally small — only logic that appears (or would otherwise be +duplicated) in two or more `scripts/*.py` modules. Each script +maintains its own primary CLI + main flow. +""" +from __future__ import annotations + +import argparse +import json +import os +import shlex +import subprocess +import sys +from pathlib import Path +from typing import Any + + +# Conventional pretrain vocab filename stems. Used by prepare_data.py + +# upgrade_to_hybrid.py + the README "Released models" bundling docs + +# the test helpers. Centralized here so a future rename only touches one +# spot. +PRETRAIN_VOCAB_STEMS = { + "atom": "pretrain_atom_vocab", + "bond": "pretrain_bond_vocab", + "smiles": "pretrain_smiles_vocab", +} + + +def resolve_kermt_repo() -> Path: + """Find the runtime checkout independently of the installed skill location. + + An explicit KERMT_REPO takes precedence. In a repository checkout, walking + up from this helper or the working directory also supports local use. + """ + explicit = os.environ.get("KERMT_REPO") + if explicit: + candidates = [Path(explicit).expanduser().resolve()] + else: + candidates = [] + for start in (Path(__file__).resolve().parent, Path.cwd()): + candidates.extend((start, *start.parents)) + for candidate in candidates: + if (candidate / "main.py").is_file() and (candidate / "kermt").is_dir(): + return candidate + raise FileNotFoundError( + "KERMT checkout not found. Set KERMT_REPO to the checkout containing " + "main.py and kermt/; the installed skill directory is separate." + ) + + +def load_json(path: Path, *, name: str) -> dict[str, Any]: + """Load a JSON file with consistent error messages. + + `name` is a human-readable label for the document (e.g. "prepare_data.json") + so the error tells the user which schema we expected at that path. + """ + if not path.is_file(): + raise FileNotFoundError(f"{name} not found at {path}") + try: + return json.loads(path.read_text()) + except json.JSONDecodeError as exc: + raise ValueError(f"{name} at {path} is not valid JSON: {exc}") from exc + + +def count_vocab_entries(vocab_path: Path) -> int: + """Return the number of entries in a KERMT vocab file. + + Handles three layouts: + - JSON with `{stoi: {token: idx}, ...}` (MolVocab.save_vocab default) + - JSON as a raw `{token: idx}` dict (legacy / hand-edited) + - Pickle of a MolVocab / SMILESVocab object (the smiles vocab is always + pickled because its compiled-regex tokenizer state isn't + JSON-serializable). Falls through to raw `pickle.load` if the + MolVocab / SMILESVocab loader can't import or fails to recognize + the contents (e.g. test fixtures with plain dicts). + """ + if vocab_path.suffix == ".json": + data = json.loads(vocab_path.read_text()) + if isinstance(data, dict) and "stoi" in data: + return len(data["stoi"]) + if isinstance(data, dict): + return len(data) + raise ValueError(f"unsupported JSON vocab shape at {vocab_path}: {type(data).__name__}") + + # .pkl: try MolVocab / SMILESVocab first, then fall back to raw pickle. + try: + from kermt.data.torchvocab import MolVocab, SMILESVocab # type: ignore + for loader in (MolVocab.load_vocab, SMILESVocab.load_vocab): + try: + v = loader(str(vocab_path)) + return len(v) + except Exception: + continue + except ImportError: + pass + + import pickle + with vocab_path.open("rb") as f: + data = pickle.load(f) + if hasattr(data, "stoi"): + return len(data.stoi) + if hasattr(data, "__len__"): + return len(data) + raise ValueError(f"could not count entries in {vocab_path}") + + +def validate_vocab_file(vocab_path: Path, *, kind: str) -> None: + """Verify a user-provided vocab file is loadable BEFORE copying it into a + run directory. Raises ValueError on failure with a clear, user-facing message. + + `kind` is one of {"atom", "bond", "smiles"} — used only in the error message + so the user knows which file is wrong. + """ + if not vocab_path.is_file(): + raise FileNotFoundError(f"{kind} vocab file not found: {vocab_path}") + try: + n = count_vocab_entries(vocab_path) + except Exception as exc: # noqa: BLE001 + raise ValueError( + f"{kind} vocab file {vocab_path} is not loadable as a KERMT vocab " + f"({type(exc).__name__}: {exc}). Expected a MolVocab JSON or pickle " + f"(or a SMILESVocab pickle for the smiles vocab)." + ) from exc + if n <= 0: + raise ValueError(f"{kind} vocab file {vocab_path} contains zero entries") + + +# --------------------------------------------------------------------------- +# Runner-shared helpers (run.json manifest fields) +# --------------------------------------------------------------------------- + +def git_commit_with_env_override(repo: Path) -> tuple[str, bool]: + """Returns (commit_sha, dirty_tree). Honors `KERMT_REPO_COMMIT` / + `KERMT_REPO_DIRTY` env vars first — set by `scripts/kermt_container.sh` + from the host before launching docker (necessary because `git -C /workspace` + inside the container fails due to bind-mount ownership). Falls back to the + in-container git probe when the env vars aren't set.""" + env_commit = os.environ.get("KERMT_REPO_COMMIT") + if env_commit: + env_dirty = os.environ.get("KERMT_REPO_DIRTY", "false").strip().lower() == "true" + return env_commit, env_dirty + try: + sha = subprocess.run( + ["git", "-C", str(repo), "rev-parse", "HEAD"], + capture_output=True, text=True, check=True, + ).stdout.strip() + diff = subprocess.run( + ["git", "-C", str(repo), "status", "--porcelain"], + capture_output=True, text=True, check=True, + ) + return sha, bool(diff.stdout.strip()) + except Exception: + return "unknown", False + + +def docker_image_digest(tag: str) -> str | None: + """Return the docker image's content-addressable Id (sha256:…) for the given + tag, or None if docker isn't available / the image isn't local.""" + try: + r = subprocess.run( + ["docker", "image", "inspect", tag, "--format", "{{.Id}}"], + capture_output=True, text=True, + ) + if r.returncode == 0: + return r.stdout.strip() + except FileNotFoundError: + pass + return None + + +def format_cmd_replay(argv: list[str], *, env: dict[str, str] | None = None) -> str: + """Render a copy-pasteable env-prefix + command for the cmd_replay manifest + field. `env` is the set of environment variables to prefix (typically + {CUDA_VISIBLE_DEVICES, WORLD_SIZE}).""" + env = env or {} + env_prefix = [f"{k}={shlex.quote(str(v))}" for k, v in env.items()] + quoted = " ".join(shlex.quote(a) for a in argv) + return " ".join(env_prefix + [quoted]) + + +def resolve_single_gpu(override: str | None, *, workflow: str) -> int: + """Returns a single GPU id (int). The finetune/inference/embed workflows are + single-GPU only; `--gpus '0,1'` or multi-id CUDA_VISIBLE_DEVICES is rejected + with a workflow-specific error. (The pretrain runner has its own multi-GPU + `_detect_gpus` helper — see run_pretrain_local.py.)""" + if override is None: + env_visible = os.environ.get("CUDA_VISIBLE_DEVICES", "").strip() + if env_visible: + ids = [g for g in env_visible.split(",") if g] + if len(ids) > 1: + raise ValueError( + f"CUDA_VISIBLE_DEVICES='{env_visible}' selects multiple GPUs but " + f"the {workflow} workflow is single-GPU only. Restrict to one id." + ) + return int(ids[0]) + return 0 + parts = [p.strip() for p in override.split(",") if p.strip()] + if len(parts) != 1: + raise ValueError( + f"--gpus '{override}' selects {len(parts)} GPUs; the {workflow} workflow is single-GPU only." + ) + return int(parts[0]) + + +def assert_prepare_manifest_basics(manifest: dict[str, Any], expected_mode: str) -> None: + """Standard pre-check for a prepare_data.json before a runner consumes it: + verify `mode` matches and `ok` is True. Raises ValueError with a consistent + error message on either mismatch. + + Each runner is responsible for its own required-outputs check after this + (those vary per-mode — e.g. pretrain wants train_dir/val_dir/atom_vocab/ + bond_vocab; finetune has the split-method branch; inference/embed want + clean_csv).""" + if manifest.get("mode") != expected_mode: + raise ValueError( + f"prepare_data manifest is mode='{manifest.get('mode')}', expected '{expected_mode}'. " + f"Run `prepare_data.py --mode {expected_mode}` to produce a valid manifest." + ) + if not manifest.get("ok"): + raise ValueError( + f"prepare_data manifest reports ok=False: {manifest.get('errors')}" + ) + + +def merge_default_into_applied( + applied: dict[str, dict[str, Any]], + args: argparse.Namespace, + name: str, + defaults_group: dict[str, Any], +) -> None: + """Standard CLI-override / default-config merge for one hyperparameter. + + Mutates `applied` in place: + - If the user passed `--` on the CLI (so `getattr(args, name)` is + not None), records `{"value": cli_val, "source": "user"}`. + - Else if `name` is present in `defaults_group`, records + `{"value": defaults_group[name], "source": "default-config"}`. + - Else `applied[name]` is left absent — the runner's argv-builder skips + the flag, and the downstream argparse default takes effect. + + `name` is the snake_case argparse dest (same form used as the dict key); + argparse automatically converts CLI `--` to that dest, + so `getattr(args, name, None)` is the correct CLI lookup.""" + cli_val = getattr(args, name, None) + if cli_val is not None: + applied[name] = {"value": cli_val, "source": "user"} + elif name in defaults_group: + applied[name] = {"value": defaults_group[name], "source": "default-config"} + + +def run_checkpoint_validator(ckpt: Path, *, mode: str, script_path: Path) -> dict[str, Any]: + """Invoke `check_checkpoint.py --mode --ckpt ` as a subprocess + and return the parsed JSON. Raises RuntimeError on non-JSON output (e.g. the + validator crashed before printing). `script_path` is the absolute path to + `scripts/check_checkpoint.py` — passed in so this helper has no + dependency on the caller's layout.""" + r = subprocess.run( + [sys.executable, str(script_path), "--mode", mode, "--ckpt", str(ckpt)], + capture_output=True, text=True, + ) + try: + return json.loads(r.stdout) + except json.JSONDecodeError as exc: + raise RuntimeError( + f"check_checkpoint.py emitted non-JSON output (exit {r.returncode}). " + f"stdout (first 200 chars): {r.stdout[:200]}\n" + f"stderr (first 200 chars): {r.stderr[:200]}" + ) from exc diff --git a/skills/kermt-finetune/scripts/check_checkpoint.py b/skills/kermt-finetune/scripts/check_checkpoint.py new file mode 100644 index 0000000..fd488e2 --- /dev/null +++ b/skills/kermt-finetune/scripts/check_checkpoint.py @@ -0,0 +1,480 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Validate a KERMT checkpoint for a given agent workflow. + +Mode-dispatched. Emits a single JSON object to stdout that the calling skill +parses to decide whether to proceed (`ok: true`) or surface a structured error +back to the user (`ok: false` with `errors[]`). The JSON shape is stable +across modes; only the contract for what counts as `ok` differs per mode. + +Modes +----- +continue_pretrain Continuing pretraining from an existing pretrain ckpt. + Requires encoder + at least one pretrain head + (vocab_head for grover_base / cmim, or contrast_head for + cmim / hybrid). Rejects encoder-only or finetuned ckpts. + +upgrade_to_hybrid Adding a cMIM decoder onto a grover_base ckpt to convert + it to a hybrid pretrain. Requires encoder; rejects ckpts + that already carry a contrast_head or task_ffn (would be + workflow 4 instead). + +finetune_init Starting a finetune from a pretrained ckpt. Requires + encoder. Pretrain heads (vocab / contrast) are tolerated + but unused. Already-finetuned ckpts (task FFN heads + present) are REJECTED — finetune-on-finetune via the + agent skill isn't supported because saved-task + identity can't be machine-verified against the new + training data. + +inference Running predictions with a previously-finetuned ckpt. + Requires encoder + task_ffn. Reports task_output_dims + so the runner can compare against the user's task spec. + +embed Extracting embeddings. Requires encoder only. Anything + additional in the ckpt is ignored. + +Output (stdout) +--------------- +{ + "ok": true | false, + "model_type": "grover_base" | "cmim" | "hybrid" | "finetuned" | "unknown", + "has_encoder": bool, + "has_vocab_head": bool, + "has_contrast_head": bool, + "has_task_ffn": bool, + "task_output_dims": [int, ...], // empty unless has_task_ffn + "arch": { // ckpt-derived; runner uses these, ignores defaults_*.json arch + "hidden_size": int | null, + "depth": int | null, + "num_attn_head": int | null, + "latent_dim": int | null, + "activation": str | null, + "backbone": str | null, + "embedding_output_type": str | null, + "self_attention": bool | null + }, + "saved_args": { ... } | null, // raw args dict if present, else null + "errors": [str, ...], // mode-contract violations / load failures + "warnings": [str, ...] // non-fatal observations (e.g. arch fallback) +} + +Exit code: 0 on `ok: true`, 1 on `ok: false`. Loader exceptions are caught and +surfaced into `errors[]` with `ok: false` (still exit 1), never raised. + +CLI +--- + check_checkpoint.py --mode --ckpt +""" +from __future__ import annotations + +import argparse +import json +import sys +import traceback +from argparse import Namespace +from typing import Any + +import torch + + +# --------------------------------------------------------------------------- +# State-dict key prefix conventions (kermt/model/models.py). +# --------------------------------------------------------------------------- + +# Encoder weights appear under one of these prefixes depending on the ckpt's +# era and task class: +# - `grover.*` : legacy grover_base ckpts (predate the cMIM rename) +# - `kermt.*` : current grover_base / hybrid / finetune ckpts +# - `latent_dist.kermt.*`: cmim ckpts (encoder lives only inside latent_dist) +ENCODER_PREFIXES = ("kermt.", "grover.", "latent_dist.kermt.") +VOCAB_HEAD_PREFIX = "vocab_module." +CONTRAST_DECODER_PREFIX = "decoder." # SMILES transformer decoder, cmim/hybrid only +LATENT_DIST_PREFIX = "latent_dist." # cmim/hybrid; encoder may share via latent_dist.kermt.* +TASK_FFN_PREFIXES = ( + "mol_atom_from_atom_ffn.", + "mol_atom_from_bond_ffn.", +) +TASK_FFN_TASK_SPECIFIC_PREFIXES = ( + "mol_atom_from_atom_ffn_task_specific.", + "mol_atom_from_bond_ffn_task_specific.", +) + + +ARCH_KEYS = ( + "hidden_size", + "depth", + "num_attn_head", + "latent_dim", + "activation", + "backbone", + "embedding_output_type", + "self_attention", +) + + +def _strip_ddp_prefix(state_dict: dict[str, Any]) -> dict[str, Any]: + """Strip `module.` prefix from every key if the dict is DDP-wrapped.""" + if state_dict and all(k.startswith("module.") for k in state_dict): + return {k[len("module."):]: v for k, v in state_dict.items()} + return state_dict + + +def _classify_model(state_dict: dict[str, Any]) -> dict[str, Any]: + keys = list(state_dict.keys()) + has_encoder = any(k.startswith(ENCODER_PREFIXES) for k in keys) + has_vocab_head = any(k.startswith(VOCAB_HEAD_PREFIX) for k in keys) + has_contrast_head = any(k.startswith(CONTRAST_DECODER_PREFIX) for k in keys) + has_task_ffn = any(k.startswith(TASK_FFN_PREFIXES) for k in keys) + + if has_encoder and has_task_ffn: + model_type = "finetuned" + elif has_encoder and has_contrast_head and has_vocab_head: + model_type = "hybrid" + elif has_encoder and has_contrast_head and not has_vocab_head: + model_type = "cmim" + elif has_encoder and not has_contrast_head: + # Includes: + # - modern repo-trained Grover base (kermt.* + vocab_module.*) + # - legacy original-Grover base (grover.encoders.* with no heads saved) + # - any encoder-stripped ckpt extracted from a larger model + # The `has_vocab_head` flag discriminates the sub-cases for skills that + # need it. The continue_pretrain mode contract relies on this — a + # grover_base with vocab heads can continue, an encoder-only one cannot. + model_type = "grover_base" + else: + model_type = "unknown" + + return { + "model_type": model_type, + "has_encoder": has_encoder, + "has_vocab_head": has_vocab_head, + "has_contrast_head": has_contrast_head, + "has_task_ffn": has_task_ffn, + } + + +def _vocab_sizes(state_dict: dict[str, Any]) -> dict[str, Any]: + """Extract vocab head sizes from state-dict weight shapes. + + The pretrain heads have the following layout per kermt/model/models.py: + - Atom vocab predictors: vocab_module.av_task_atom.* + vocab_module.av_task_bond.* + (two readout streams sharing the same vocab_size). Output dim of each + final-Linear is the atom vocab size. + - Bond vocab predictors: vocab_module.bv_task_atom.* + vocab_module.bv_task_bond.* + Output dim is the bond vocab size. + - SMILES vocab decoder: decoder.output_projection.weight (cmim / hybrid only). + Output dim is the smiles vocab size. + + Returns {atom: int|None, bond: int|None, smiles: int|None}. Each is None + when the corresponding head isn't present in the ckpt (e.g. legacy + encoder-only grover_base has none; cmim has smiles but not atom/bond). + """ + sizes: dict[str, Any] = {"atom": None, "bond": None, "smiles": None} + + def _head_out_dim(prefix: str) -> int | None: + # Pick the highest-numbered 2-D Linear weight under `prefix.*` — that's + # the final output layer. + candidates = [ + k for k in state_dict + if k.startswith(prefix) and k.endswith(".weight") + and hasattr(state_dict[k], "ndim") and state_dict[k].ndim == 2 + ] + if not candidates: + return None + def _layer_index(k: str) -> int: + # ".weight" -> "..weight"; pick the rightmost numeric component. + parts = k.split(".") + for tok in reversed(parts[:-1]): + if tok.isdigit(): + return int(tok) + return -1 + final = max(candidates, key=_layer_index) + return int(state_dict[final].shape[0]) + + sizes["atom"] = _head_out_dim("vocab_module.av_task_atom.") + sizes["bond"] = _head_out_dim("vocab_module.bv_task_atom.") + sizes["smiles"] = _head_out_dim("decoder.output_projection.") + # If the decoder's output_projection isn't a Linear (e.g. some saves wrap + # it differently), fall back to a search over decoder.* heads. + if sizes["smiles"] is None: + sizes["smiles"] = _head_out_dim("decoder.token_embedding.") + return sizes + + +def _task_output_dims(state_dict: dict[str, Any]) -> list[int]: + """Return one entry per (logical task × readout) head's final-Linear out-dim. + + Two layouts: + - **MTL** (`mol_atom_from_atom_ffn_task_specific..*`): one entry per + task-specific head's final-Linear out-dim. Typically `[1, 1, ..., 1]` + for regression with N tasks across 2 readouts. + - **Non-MTL** (`mol_atom_from_atom_ffn.*` only): one entry per shared FFN's + final-Linear out-dim. Typically `[num_tasks, num_tasks]` (one per readout). + + When both layouts coexist in the same ckpt (MTL configuration: shared FFN + feeds task-specific heads), only the task-specific dims are reported — the + shared FFN there is an intermediate layer, not the model output. + """ + has_task_specific = any(k.startswith(TASK_FFN_TASK_SPECIFIC_PREFIXES) for k in state_dict) + + heads: dict[str, list[str]] = {} + for k in state_dict: + if k.startswith(TASK_FFN_TASK_SPECIFIC_PREFIXES): + parts = k.split(".") + root = ".".join(parts[:2]) # e.g. "mol_atom_from_atom_ffn_task_specific.0" + heads.setdefault(root, []).append(k) + elif k.startswith(TASK_FFN_PREFIXES) and not k.startswith(TASK_FFN_TASK_SPECIFIC_PREFIXES): + if has_task_specific: + continue # shared FFN is intermediate when task-specific heads exist + root = k.split(".")[0] # e.g. "mol_atom_from_atom_ffn" + heads.setdefault(root, []).append(k) + + dims: list[int] = [] + for root in sorted(heads): + weight_keys = sorted( + (k for k in heads[root] if k.endswith(".weight") + and hasattr(state_dict[k], "ndim") and state_dict[k].ndim == 2), + key=lambda k: int(k.split(".")[-2]) if k.split(".")[-2].isdigit() else -1, + ) + if weight_keys: + dims.append(int(state_dict[weight_keys[-1]].shape[0])) + return dims + + +def _arch_from_args(args_obj: Any) -> dict[str, Any]: + """Pull arch params from the saved args Namespace / dict, leaving missing keys as None.""" + arch: dict[str, Any] = {k: None for k in ARCH_KEYS} + if args_obj is None: + return arch + # args_obj is typically argparse.Namespace; tolerate dict form too. + args_dict = vars(args_obj) if isinstance(args_obj, Namespace) else dict(args_obj) if isinstance(args_obj, dict) else {} + for k in ARCH_KEYS: + if k in args_dict: + arch[k] = args_dict[k] + return arch + + +def _arch_from_shapes(state_dict: dict[str, Any], arch: dict[str, Any]) -> tuple[dict[str, Any], list[str]]: + """Fill in still-missing arch params by introspecting state-dict tensor shapes. + + Only fills entries that are currently None — does not override anything pulled + from saved_args. Returns the updated arch + a list of warnings for any key that + could not be inferred. + """ + warnings: list[str] = [] + + if arch["hidden_size"] is None: + # First 2-D linear weight under any encoder prefix. + candidates = [ + k for k in state_dict + if k.startswith(ENCODER_PREFIXES) + and k.endswith(".weight") + and hasattr(state_dict[k], "ndim") + and state_dict[k].ndim == 2 + ] + if candidates: + arch["hidden_size"] = int(state_dict[candidates[0]].shape[0]) + else: + warnings.append("hidden_size could not be inferred from state_dict shapes") + + if arch["latent_dim"] is None: + # Look for a Linear inside latent_dist that's not the shared encoder. + candidates = [ + k for k in state_dict + if k.startswith(LATENT_DIST_PREFIX) + and not k.startswith("latent_dist.kermt.") + and k.endswith(".weight") + and state_dict[k].ndim == 2 + ] + if candidates: + arch["latent_dim"] = int(state_dict[candidates[0]].shape[0]) + # Absent latent_dist entirely is a fact about the model, not a problem — no warning. + + # depth, num_attn_head, activation, backbone, embedding_output_type, self_attention + # are not robustly inferable from shapes alone; report a warning for each that's + # still None so the caller can prompt the user or refuse to proceed. + for k in ("depth", "num_attn_head", "activation", "backbone", "embedding_output_type", "self_attention"): + if arch[k] is None: + warnings.append(f"{k} not present in saved_args and cannot be inferred from state_dict shapes") + + return arch, warnings + + +def _apply_mode_contract(mode: str, classification: dict[str, Any]) -> list[str]: + """Return a list of error messages if `classification` violates the mode contract.""" + errors: list[str] = [] + mt = classification["model_type"] + has_enc = classification["has_encoder"] + has_vocab = classification["has_vocab_head"] + has_contrast = classification["has_contrast_head"] + has_ffn = classification["has_task_ffn"] + + if not has_enc: + errors.append("checkpoint has no encoder weights — cannot use it for any KERMT workflow") + return errors + + if mode == "continue_pretrain": + if not (has_vocab or has_contrast): + errors.append( + f"continue_pretrain requires the ckpt to still carry pretrain heads (vocab " + f"and/or contrast), but this ckpt has neither (model_type='{mt}', " + f"has_vocab_head=False, has_contrast_head=False). Either provide a ckpt with " + f"its pretrain heads attached, or convert this encoder-only ckpt to a hybrid " + f"via mode 'upgrade_to_hybrid'." + ) + if has_ffn: + errors.append( + "continue_pretrain expects a pretrain ckpt; this ckpt has task FFN heads " + "(it has been finetuned). Use a pretrain checkpoint — finetune+continue is " + "not a supported workflow." + ) + elif mode == "upgrade_to_hybrid": + if has_contrast: + errors.append( + f"upgrade_to_hybrid converts grover_base -> hybrid by adding a cMIM decoder. " + f"This ckpt already has a contrast head (classified as '{mt}'). " + f"To continue pretraining it, use mode 'continue_pretrain'." + ) + if has_ffn: + errors.append("upgrade_to_hybrid does not support finetuned checkpoints.") + elif mode == "finetune_init": + # Requires an encoder. Pretrain heads (vocab / contrast) are unused + # at finetune time but harmless. Task FFN heads (i.e. an already- + # finetuned ckpt) are NOT accepted — finetune-on-finetune isn't + # supported by the kermt-finetune skill because the saved-task + # identity can't be machine-verified against the new training data + # (dimension match doesn't prove target identity, dataset identity, + # or absence of train/test contamination). + if has_ffn: + errors.append( + f"finetune_init requires a pretrain ckpt (grover_base / cmim / hybrid); " + f"this ckpt is classified as '{mt}' with task FFN heads attached. " + f"To resume a finetune on the SAME dataset, call " + f"`python main.py finetune --checkpoint_path ...` directly — the " + f"kermt-finetune skill doesn't support resume." + ) + elif mode == "inference": + if not has_ffn: + errors.append( + "inference requires a finetuned ckpt with task FFN heads. " + f"This ckpt is classified as '{mt}' with no task heads. " + "Run finetune (mode 'finetune_init') first." + ) + elif mode == "embed": + # Encoder is sufficient. + pass + else: + errors.append(f"unknown mode '{mode}'") + + return errors + + +def validate(mode: str, ckpt_path: str) -> dict[str, Any]: + result: dict[str, Any] = { + "ok": False, + "model_type": "unknown", + "has_encoder": False, + "has_vocab_head": False, + "has_contrast_head": False, + "has_task_ffn": False, + "task_output_dims": [], + "vocab_sizes": {"atom": None, "bond": None, "smiles": None}, + "arch": {k: None for k in ARCH_KEYS}, + "saved_args": None, + "errors": [], + "warnings": [], + } + + # 1. Load the checkpoint. + try: + ckpt = torch.load(ckpt_path, map_location="cpu", weights_only=False) + except FileNotFoundError: + result["errors"].append(f"checkpoint not found: {ckpt_path}") + return result + except Exception as exc: # noqa: BLE001 + result["errors"].append(f"failed to load checkpoint {ckpt_path}: {type(exc).__name__}: {exc}") + return result + + if not isinstance(ckpt, dict) or "state_dict" not in ckpt: + result["errors"].append( + "checkpoint is not in the expected save_model_for_restart format " + "(expected a dict with a 'state_dict' key)." + ) + return result + + state_dict = _strip_ddp_prefix(ckpt["state_dict"]) + args_obj = ckpt.get("args") + + # 2. Classify and check mode contract. + classification = _classify_model(state_dict) + result.update(classification) + + contract_errors = _apply_mode_contract(mode, classification) + result["errors"].extend(contract_errors) + + # 3. Task output dims (for inference / informational). + if classification["has_task_ffn"]: + result["task_output_dims"] = _task_output_dims(state_dict) + + # 3b. Vocab head sizes (for continue-pretrain vocab-size verification). + result["vocab_sizes"] = _vocab_sizes(state_dict) + + # 4. Arch derivation: args first, shape introspection for what's still missing. + arch = _arch_from_args(args_obj) + arch, shape_warnings = _arch_from_shapes(state_dict, arch) + result["arch"] = arch + result["warnings"].extend(shape_warnings) + + # 5. Saved args as serializable dict (best-effort). + if args_obj is not None: + try: + result["saved_args"] = vars(args_obj) if isinstance(args_obj, Namespace) else dict(args_obj) + # Drop non-JSON-serializable values; agent skill only needs human-readable scalars. + result["saved_args"] = { + k: v for k, v in result["saved_args"].items() + if isinstance(v, (str, int, float, bool, type(None), list, dict)) + } + except Exception as exc: # noqa: BLE001 + result["warnings"].append(f"could not serialize saved_args: {type(exc).__name__}: {exc}") + + result["ok"] = not result["errors"] + return result + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Validate a KERMT checkpoint for a given workflow.") + parser.add_argument("--mode", required=True, + choices=["continue_pretrain", "upgrade_to_hybrid", "finetune_init", "inference", "embed"]) + parser.add_argument("--ckpt", required=True, help="Path to the .pt checkpoint") + args = parser.parse_args(argv) + + try: + result = validate(args.mode, args.ckpt) + except Exception as exc: # noqa: BLE001 + # Last-resort safety net: keep stdout JSON-clean, dump trace to stderr. + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({ + "ok": False, + "model_type": "unknown", + "errors": [f"unhandled exception in validator: {type(exc).__name__}: {exc}"], + "warnings": [], + "arch": {k: None for k in ARCH_KEYS}, + "has_encoder": False, + "has_vocab_head": False, + "has_contrast_head": False, + "has_task_ffn": False, + "task_output_dims": [], + "vocab_sizes": {"atom": None, "bond": None, "smiles": None}, + "saved_args": None, + }, indent=2)) + return 1 + + print(json.dumps(result, indent=2)) + return 0 if result["ok"] else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-finetune/scripts/check_data.py b/skills/kermt-finetune/scripts/check_data.py new file mode 100644 index 0000000..b8f9b15 --- /dev/null +++ b/skills/kermt-finetune/scripts/check_data.py @@ -0,0 +1,310 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Validate a CSV input for a given KERMT agent workflow. + +Mode-dispatched. Emits a single JSON object to stdout that the calling skill +parses to decide whether to proceed (`ok: true`) or surface a structured error +back to the user (`ok: false` with `errors[]`). The JSON shape is stable +across modes; only the contract for what counts as `ok` differs per mode. + +Modes +----- +pretrain Pretrain corpus CSV. Requires a `smiles` column. Other columns + are ignored. Label columns are not required (and not expected). + +finetune Labeled CSV for a downstream task. Requires `smiles` plus + >=1 numeric target column. Target columns are specified via + `--targets ...`. If `--targets` is omitted, the + validator auto-detects numeric non-smiles columns and reports + them; the skill will then prompt the user to confirm or refine. + +inference CSV to run predictions on. Requires `smiles`. Target columns are + not required (and not expected — predictions are written out). + +embed CSV to extract embeddings from. Requires `smiles` only. + +SMILES validation +----------------- +By default the validator samples up to 20 SMILES (first 10 + last 10) and +checks each one parses with RDKit. Pass `--strict-rdkit` to parse every +SMILES (slow on large corpora). A SMILES is considered "invalid" if RDKit +returns `None` from `MolFromSmiles(smi, sanitize=True)` — empty / null +rows are counted separately. + +Duplicate-SMILES detection is always full (cheap). + +Output (stdout) +--------------- +{ + "ok": true | false, + "mode": str, + "csv_path": str, + "num_rows": int, + "num_columns": int, + "columns": [str, ...], + "has_smiles_column": bool, + "smiles_column_name": str | null, // actual header used (may differ in case) + "num_blank_smiles": int, + "num_invalid_smiles": int, // among the parsed sample + "smiles_check_method": "sampled" | "full", + "smiles_check_count": int, + "num_duplicate_smiles": int, + "target_columns": [str, ...], // populated only for finetune mode + "num_missing_per_target": { col: int, ... }, + "auto_detected_targets": [str, ...], // when --targets is omitted in finetune mode + "errors": [str, ...], + "warnings": [str, ...] +} + +Exit code: 0 on `ok: true`, 1 on `ok: false`. Loader exceptions are caught +and surfaced into `errors[]` with `ok: false` (still exit 1). + +CLI +--- + check_data.py --mode --csv + [--targets ...] # finetune only + [--strict-rdkit] # full SMILES parse +""" +from __future__ import annotations + +import argparse +import json +import sys +import traceback +from pathlib import Path +from typing import Any + +import pandas as pd + + +CANONICAL_SMILES_COLUMN = "smiles" +SMILES_SAMPLE_PER_END = 10 # how many SMILES from head + how many from tail to sample + + +def _find_smiles_column(columns: list[str]) -> str | None: + """Return the actual column header matching 'smiles' case-insensitively, or None.""" + for c in columns: + if c.lower() == CANONICAL_SMILES_COLUMN: + return c + return None + + +def _parse_smiles_sample(smiles_values: list[str], full: bool) -> tuple[int, int, str]: + """Run RDKit MolFromSmiles on a sample or all of the SMILES. Returns + (num_parsed, num_invalid, method).""" + # Import here so the script can still surface a clean JSON error if RDKit + # is unavailable in the host env. + try: + from rdkit import Chem + from rdkit import RDLogger + RDLogger.DisableLog("rdApp.*") # suppress per-mol parse warnings + except ImportError as exc: + raise RuntimeError( + f"RDKit is not importable in this environment: {exc}. " + "Run check_data.py inside the kermt container." + ) from exc + + if full or len(smiles_values) <= 2 * SMILES_SAMPLE_PER_END: + sample = smiles_values + method = "full" + else: + sample = smiles_values[:SMILES_SAMPLE_PER_END] + smiles_values[-SMILES_SAMPLE_PER_END:] + method = "sampled" + + invalid = 0 + parsed = 0 + for smi in sample: + if not smi: # already counted as blank elsewhere + continue + parsed += 1 + mol = Chem.MolFromSmiles(smi, sanitize=True) + if mol is None: + invalid += 1 + return parsed, invalid, method + + +def _autodetect_target_columns(df: pd.DataFrame, smiles_col: str) -> list[str]: + """Pick columns that look like numeric targets. A column qualifies if it + is (a) not the smiles column and (b) >=80% of non-null values convert to float. + Heuristic only — returned for the skill to prompt the user to confirm.""" + candidates: list[str] = [] + for col in df.columns: + if col == smiles_col: + continue + ser = df[col].dropna() + if len(ser) == 0: + continue + try: + converted = pd.to_numeric(ser, errors="coerce") + except (TypeError, ValueError): + continue + if converted.notna().sum() / max(len(ser), 1) >= 0.8: + candidates.append(col) + return candidates + + +def validate(mode: str, csv_path: str, targets: list[str] | None, strict_rdkit: bool) -> dict[str, Any]: + result: dict[str, Any] = { + "ok": False, + "mode": mode, + "csv_path": csv_path, + "num_rows": 0, + "num_columns": 0, + "columns": [], + "has_smiles_column": False, + "smiles_column_name": None, + "num_blank_smiles": 0, + "num_invalid_smiles": 0, + "smiles_check_method": "sampled", + "smiles_check_count": 0, + "num_duplicate_smiles": 0, + "target_columns": [], + "num_missing_per_target": {}, + "auto_detected_targets": [], + "errors": [], + "warnings": [], + } + + # 1. Read the CSV. + path = Path(csv_path) + if not path.is_file(): + result["errors"].append(f"CSV not found: {csv_path}") + return result + try: + df = pd.read_csv(path) + except pd.errors.EmptyDataError: + result["errors"].append(f"CSV is empty (no header): {csv_path}") + return result + except Exception as exc: # noqa: BLE001 + result["errors"].append(f"failed to read CSV {csv_path}: {type(exc).__name__}: {exc}") + return result + + result["num_rows"] = int(len(df)) + result["num_columns"] = int(len(df.columns)) + result["columns"] = [str(c) for c in df.columns] + + # 2. Locate the SMILES column. + smiles_col = _find_smiles_column(result["columns"]) + if smiles_col is None: + result["errors"].append( + f"no column named 'smiles' (case-insensitive) found in CSV. " + f"Available columns: {result['columns']}" + ) + return result + result["has_smiles_column"] = True + result["smiles_column_name"] = smiles_col + if smiles_col != CANONICAL_SMILES_COLUMN: + result["warnings"].append( + f"SMILES column is named '{smiles_col}' but downstream code expects '{CANONICAL_SMILES_COLUMN}' " + f"(lowercase). Rename the column to '{CANONICAL_SMILES_COLUMN}' before running the workflow." + ) + + # 3. Blank-SMILES count + duplicate count + RDKit parse check. + smi_series = df[smiles_col].astype(str).fillna("").str.strip() + blank_mask = smi_series.eq("") | smi_series.str.lower().eq("nan") + result["num_blank_smiles"] = int(blank_mask.sum()) + + nonblank = smi_series[~blank_mask] + result["num_duplicate_smiles"] = int(len(nonblank) - nonblank.nunique()) + + if len(nonblank) == 0: + result["errors"].append("no non-blank SMILES found in the CSV") + return result + + try: + parsed, invalid, method = _parse_smiles_sample(nonblank.tolist(), full=strict_rdkit) + except RuntimeError as exc: + result["errors"].append(str(exc)) + return result + result["smiles_check_count"] = parsed + result["num_invalid_smiles"] = invalid + result["smiles_check_method"] = method + + if invalid > 0: + scope = "all rows" if method == "full" else f"the {parsed} sampled rows" + result["errors"].append( + f"{invalid} out of {parsed} SMILES in {scope} failed to parse with RDKit. " + "Either pre-clean the CSV with scripts/clean_smiles.py or pass --strict-rdkit to see " + "the full count." + ) + + # 4. Target-column handling — finetune mode only. + if mode == "finetune": + if targets: + missing = [t for t in targets if t not in df.columns] + if missing: + result["errors"].append( + f"target column(s) not found in CSV: {missing}. " + f"Available columns: {result['columns']}" + ) + else: + result["target_columns"] = list(targets) + for t in targets: + nan_count = int(df[t].isna().sum()) + result["num_missing_per_target"][t] = nan_count + # Confirm numeric-ish. + nonnan = df[t].dropna() + converted = pd.to_numeric(nonnan, errors="coerce") + non_numeric_count = int(converted.isna().sum()) + if non_numeric_count > 0: + result["warnings"].append( + f"target column '{t}' has {non_numeric_count} non-numeric value(s) " + f"that will be dropped by the finetune runner." + ) + else: + # Auto-detect — surface candidates so the skill can prompt the user. + result["auto_detected_targets"] = _autodetect_target_columns(df, smiles_col) + if not result["auto_detected_targets"]: + result["errors"].append( + "no numeric non-smiles columns detected. finetune needs at least one target column; " + "specify it explicitly via --targets ." + ) + else: + result["warnings"].append( + f"--targets was not specified; auto-detected candidate target columns " + f"{result['auto_detected_targets']}. The skill will prompt the user to confirm." + ) + + # 5. Small-corpus warning — only for pretrain (other modes can be tiny by design). + if mode == "pretrain" and result["num_rows"] < 100: + result["warnings"].append( + f"pretrain corpus is only {result['num_rows']} molecule(s). Pretraining typically " + f"needs orders of magnitude more — verify this is the intended input." + ) + + result["ok"] = not result["errors"] + return result + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Validate a CSV input for a KERMT agent workflow.") + parser.add_argument("--mode", required=True, choices=["pretrain", "finetune", "inference", "embed"]) + parser.add_argument("--csv", required=True, help="Path to the input CSV") + parser.add_argument("--targets", nargs="+", default=None, + help="(finetune only) target column names. If omitted, the validator auto-detects " + "numeric non-smiles columns and reports them as candidates.") + parser.add_argument("--strict-rdkit", action="store_true", + help="Parse every SMILES with RDKit rather than sampling (slow on large CSVs).") + args = parser.parse_args(argv) + + try: + result = validate(args.mode, args.csv, args.targets, args.strict_rdkit) + except Exception as exc: # noqa: BLE001 + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({ + "ok": False, + "mode": args.mode, + "csv_path": args.csv, + "errors": [f"unhandled exception in validator: {type(exc).__name__}: {exc}"], + "warnings": [], + }, indent=2)) + return 1 + + print(json.dumps(result, indent=2)) + return 0 if result["ok"] else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-finetune/scripts/fetch_released_model.py b/skills/kermt-finetune/scripts/fetch_released_model.py new file mode 100644 index 0000000..ca2ad57 --- /dev/null +++ b/skills/kermt-finetune/scripts/fetch_released_model.py @@ -0,0 +1,234 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Download a released KERMT model bundle from Hugging Face. + +Runs INSIDE the kermt container (huggingface_hub is part of the image env). +Writes the released-model directory bundle — the checkpoint plus its vocab +files — into the directory mounted at `--out` (the skills mount the user's +chosen save location there via `kermt_container.sh --model-dir `), then +emits a single JSON object to stdout that the calling skill parses. + +The downloaded directory is exactly the repo's "released model bundle" layout +(see skills/README.md "Released models"): `.pt` + the three +`pretrain_*_vocab.*` files in one flat directory. The downstream skill then +feeds it through the existing `--ckpt /` flow; for +continue-pretrain the bundled vocab files are auto-detected in the ckpt's +parent directory. No runner changes are needed. + +Defaults (repo id, pinned revision, ckpt + vocab filenames) come from +`config/released_model.json` so the pin lives in one place; every value +is overridable on the CLI. + +Idempotent: if the bundle is already complete in `--out` (ckpt + all vocab +files present), nothing is downloaded and `reused: true` is reported — so a +re-invocation never re-fetches the 282 MB checkpoint. + +Authentication: the repo is public (no token needed). If `HF_TOKEN` is set in +the environment (forwarded into the container by `kermt_container.sh`), +huggingface_hub picks it up automatically — useful against shared-IP rate +limits or if the repo is ever gated. + +Output (stdout) +--------------- +{ + "ok": true | false, + "repo_id": str, + "revision": str, + "out": str, // container path of the bundle dir (e.g. /model) + "ckpt": str | null, // container path of the checkpoint file + "vocab_dir": str | null, // == out (where the vocab files live) + "ckpt_name": str, + "vocab_files": [str, ...], + "files_present": [str, ...], + "ckpt_bytes": int | null, + "reused": bool, // true if the bundle already existed (no download) + "license": str | null, + "license_url": str | null, + "errors": [str, ...] +} + +Exit code: 0 on `ok: true`, 1 on `ok: false`. + +CLI +--- + fetch_released_model.py [--out /model] + [--repo-id ] [--revision ] + [--ckpt-name ] [--config ] +""" + +from __future__ import annotations + +import argparse +import json +import sys +import traceback +from pathlib import Path +from typing import Any + +# Default config lives at config/released_model.json (one dir up from +# scripts/). Resolved relative to this file so the script is +# location-independent. +DEFAULT_CONFIG = ( + Path(__file__).resolve().parent.parent / "config" / "released_model.json" +) + + +def _load_config(config_path: Path) -> dict[str, Any]: + if not config_path.is_file(): + raise FileNotFoundError(f"released-model config not found at {config_path}") + return json.loads(config_path.read_text()) + + +def fetch( + *, + out: Path, + repo_id: str, + revision: str, + ckpt_name: str, + vocab_files: list[str], + license_name: str | None = None, + license_url: str | None = None, +) -> dict[str, Any]: + """Resolve-or-download the released bundle into `out`. Returns the manifest + dict (never raises for the expected failure modes — they land in + `errors[]` with `ok: false`).""" + result: dict[str, Any] = { + "ok": False, + "repo_id": repo_id, + "revision": revision, + "out": str(out), + "ckpt": None, + "vocab_dir": None, + "ckpt_name": ckpt_name, + "vocab_files": list(vocab_files), + "files_present": [], + "ckpt_bytes": None, + "reused": False, + "license": license_name, + "license_url": license_url, + "errors": [], + } + + required = [ckpt_name, *vocab_files] + + def _present() -> list[str]: + return [name for name in required if (out / name).is_file()] + + # 1. Idempotent reuse — bundle already complete in `out`. + if out.is_dir() and set(_present()) == set(required): + result["reused"] = True + else: + # 2. Download. Import here so a stale image (missing huggingface_hub) + # surfaces a clean, actionable JSON error rather than a traceback. + try: + from huggingface_hub import snapshot_download + except ImportError: + result["errors"].append( + "huggingface_hub is not available in the container image. The " + "released-model download needs it; rebuild the image with " + "`kermt-setup` (it now ships huggingface_hub) and retry." + ) + return result + + out.mkdir(parents=True, exist_ok=True) + try: + # local_dir gives a flat copy (the bundle layout) rather than the + # opaque blob/snapshot cache. HF_TOKEN, if set, is read by the lib. + snapshot_download(repo_id=repo_id, revision=revision, local_dir=str(out)) + except Exception as exc: # noqa: BLE001 + result["errors"].append( + f"download failed for {repo_id}@{revision}: {type(exc).__name__}: {exc}" + ) + return result + + # 3. Verify the bundle is complete regardless of download/reuse path. + present = _present() + result["files_present"] = present + missing = [name for name in required if name not in present] + if missing: + result["errors"].append( + f"bundle at {out} is missing expected file(s): {missing}. " + f"Present: {present}." + ) + return result + + ckpt_path = out / ckpt_name + result["ckpt"] = str(ckpt_path) + result["vocab_dir"] = str(out) + try: + result["ckpt_bytes"] = ckpt_path.stat().st_size + except OSError: + result["ckpt_bytes"] = None + + result["ok"] = True + return result + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser( + description="Download a released KERMT model bundle from Hugging Face (runs in-container)." + ) + parser.add_argument( + "--out", + default="/model", + help="Directory to write the bundle into (default: /model, the --model-dir mount).", + ) + parser.add_argument( + "--config", + default=str(DEFAULT_CONFIG), + help="Path to released_model.json (default: config/released_model.json).", + ) + parser.add_argument( + "--repo-id", default=None, help="Override the HF repo id from the config." + ) + parser.add_argument( + "--revision", + default=None, + help="Override the pinned revision (sha/tag/branch).", + ) + parser.add_argument( + "--ckpt-name", + default=None, + help="Override the checkpoint filename from the config.", + ) + args = parser.parse_args(argv) + + try: + cfg = _load_config(Path(args.config)) + repo_id = args.repo_id or cfg["repo_id"] + revision = args.revision or cfg["revision"] + ckpt_name = args.ckpt_name or cfg["ckpt_name"] + vocab_files = list(cfg.get("vocab_files", [])) + result = fetch( + out=Path(args.out), + repo_id=repo_id, + revision=revision, + ckpt_name=ckpt_name, + vocab_files=vocab_files, + license_name=cfg.get("license"), + license_url=cfg.get("license_url"), + ) + except Exception as exc: # noqa: BLE001 + print(traceback.format_exc(), file=sys.stderr) + print( + json.dumps( + { + "ok": False, + "out": args.out, + "errors": [ + f"unhandled exception in fetch_released_model: {type(exc).__name__}: {exc}" + ], + }, + indent=2, + ) + ) + return 1 + + print(json.dumps(result, indent=2)) + return 0 if result["ok"] else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/skills/kermt-finetune/scripts/kermt_container.sh b/skills/kermt-finetune/scripts/kermt_container.sh new file mode 100755 index 0000000..028057e --- /dev/null +++ b/skills/kermt-finetune/scripts/kermt_container.sh @@ -0,0 +1,484 @@ +#!/usr/bin/env bash +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# kermt_container.sh — bootstrap helper for the kermt agent skills. +# +# Two ways to use this file: +# +# 1. As a subcommand dispatcher (recommended for skills): +# "$SKILL_DIR/scripts/kermt_container.sh" ensure_image +# "$SKILL_DIR/scripts/kermt_container.sh" run --ckpt /host/ckpt.pt -- python -c 'import torch; print(torch.cuda.device_count())' +# "$SKILL_DIR/scripts/kermt_container.sh" run_detached --name foo --run-dir runs/foo -- bash train.sh +# +# 2. Sourced into a shell or another script, then call the kermt_* functions +# directly: +# source "$SKILL_DIR/scripts/kermt_container.sh" +# kermt_ensure_image +# kermt_run --ckpt /host/ckpt.pt -- python ... +# +# Configuration (override via env vars before invocation): +# KERMT_IMAGE docker image tag (default: kermt:latest) +# KERMT_REPO host path to the kermt repo checkout (default: auto-derived +# from this script's location) +# KERMT_GPUS value passed to docker --gpus (default: all) +# +# Mount flags accepted by kermt_run / kermt_run_detached: +# --data bind to /data (read-only). If is a file, +# its PARENT directory is mounted at /data so +# commands can use /data/; if is a +# directory, it is mounted at /data directly. +# --ckpt bind to /ckpt (read-only; the path is mounted as-is) +# --vocab-dir bind to /vocab (read-only) +# --run-dir bind to /runs (read-write; created on host if missing) +# --model-dir bind to /model (read-write; created on host if missing). +# Target for released-model downloads (fetch_released_model.py). +# +# Additional flags for kermt_run_detached: +# --name docker container name (default: kermt--) +# +# Everything after `--` is the command passed to the container. It runs inside +# the `kermt` conda environment (the image's default env). + +set -o pipefail + +: "${KERMT_IMAGE:=kermt:latest}" +: "${KERMT_GPUS:=all}" + +# The skill may be installed outside the KERMT checkout. Mount its own helpers +# separately so container commands always execute the distributed skill copy. +_kermt_script_dir="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)" +_kermt_bundle_dir="$(cd "$_kermt_script_dir/.." && pwd)" +if [[ -z "${KERMT_REPO:-}" ]]; then + for _kermt_start in "$_kermt_script_dir" "$PWD"; do + _kermt_candidate="$_kermt_start" + while [[ "$_kermt_candidate" != / ]]; do + if [[ -f "$_kermt_candidate/main.py" && -d "$_kermt_candidate/kermt" ]]; then + KERMT_REPO="$_kermt_candidate" + break 2 + fi + _kermt_candidate="$(dirname "$_kermt_candidate")" + done + done + unset _kermt_start _kermt_candidate +fi +unset _kermt_script_dir + +_kermt_require_repo() { + if [[ -z "${KERMT_REPO:-}" || ! -f "$KERMT_REPO/main.py" || ! -d "$KERMT_REPO/kermt" ]]; then + echo "[kermt] Set KERMT_REPO to the KERMT checkout containing main.py and kermt/." >&2 + return 1 + fi + KERMT_REPO="$(cd "$KERMT_REPO" && pwd)" || return $? + export KERMT_REPO +} + +# ----------------------------------------------------------------------------- +# Host environment checks +# ----------------------------------------------------------------------------- + +kermt_check_docker() { + if ! command -v docker >/dev/null 2>&1; then + echo "[kermt] error: docker not found on PATH. Install Docker first." >&2 + return 1 + fi + if ! docker info >/dev/null 2>&1; then + echo "[kermt] error: docker daemon not reachable. Is the docker service running, and is your user in the 'docker' group?" >&2 + return 1 + fi +} + +kermt_check_system() { + # Probe host system and report GPU presence + VRAM + compute capability + + # driver / CUDA version + disk space. Emits a single JSON document to + # stdout that the calling skill consumes; exits 0 with `ok: false` and a + # populated `gaps` array when anything is below the per-workflow minimum, + # exits 1 only on unexpected internal errors. Uses host nvidia-smi + df + + # host python3 (stdlib only). + python3 - "$KERMT_REPO" "$KERMT_IMAGE" <<'PYEOF' +import json, os, shutil, subprocess, sys + +repo, image = sys.argv[1], sys.argv[2] + +result = { + "ok": True, + "gpus": [], + "disk": {"path": repo, "free_gb": None, "min_gb": 20}, + "host": {"docker": None, "nvidia_smi": None, "container_toolkit": None}, + "image": {"tag": image, "present_locally": None}, + "gaps": [], +} + +def _gap(msg): + result["ok"] = False + result["gaps"].append(msg) + +# docker presence +try: + r = subprocess.run(["docker", "info"], capture_output=True, text=True, timeout=10) + result["host"]["docker"] = "ok" if r.returncode == 0 else f"failed: {r.stderr.strip().splitlines()[-1] if r.stderr else 'unknown'}" + if r.returncode != 0: + _gap("docker daemon not reachable (is the service running, and is your user in the 'docker' group?)") +except FileNotFoundError: + result["host"]["docker"] = "not found" + _gap("docker not on PATH; install Docker first") +except Exception as e: + result["host"]["docker"] = f"error: {e}" + _gap(f"docker probe failed: {e}") + +# nvidia-smi (host driver) +try: + r = subprocess.run( + ["nvidia-smi", "--query-gpu=name,memory.total,compute_cap,driver_version,uuid", + "--format=csv,noheader,nounits"], + capture_output=True, text=True, timeout=10, + ) + if r.returncode == 0: + result["host"]["nvidia_smi"] = "ok" + for line in r.stdout.strip().splitlines(): + parts = [p.strip() for p in line.split(",")] + if len(parts) >= 5: + try: + vram_mb = int(parts[1]) + except ValueError: + vram_mb = None + result["gpus"].append({ + "name": parts[0], + "vram_mb": vram_mb, + "compute_cap": parts[2], + "driver": parts[3], + "uuid": parts[4], + }) + if not result["gpus"]: + _gap("nvidia-smi succeeded but reported no GPUs") + else: + result["host"]["nvidia_smi"] = "failed" + _gap("nvidia-smi found but failed; is the NVIDIA driver loaded?") +except FileNotFoundError: + result["host"]["nvidia_smi"] = "not found" + _gap("nvidia-smi not on PATH; install the NVIDIA driver") +except Exception as e: + result["host"]["nvidia_smi"] = f"error: {e}" + _gap(f"nvidia-smi probe failed: {e}") + +# disk free at the repo location +try: + free_bytes = shutil.disk_usage(repo).free + free_gb = free_bytes // (1024**3) + result["disk"]["free_gb"] = free_gb + if free_gb < result["disk"]["min_gb"]: + _gap(f"disk free at {repo} is {free_gb} GB; need at least {result['disk']['min_gb']} GB for the kermt image") +except Exception as e: + _gap(f"could not check disk space at {repo}: {e}") + +# image presence (informational only) +try: + r = subprocess.run(["docker", "image", "inspect", image], capture_output=True, text=True, timeout=10) + result["image"]["present_locally"] = (r.returncode == 0) +except Exception: + result["image"]["present_locally"] = None + +# nvidia-container-toolkit probe — only meaningful if both docker and a +# locally-present image are available. Pick kermt:$tag first; fall back to +# the small CUDA base image if that's the only one present; otherwise skip +# (avoid pulling anything). +def _probe_image(): + for img in (image, "nvidia/cuda:12.6.3-base-ubuntu22.04"): + r = subprocess.run(["docker", "image", "inspect", img], capture_output=True) + if r.returncode == 0: + return img + return None + +probe_img = _probe_image() +if probe_img: + try: + r = subprocess.run( + ["docker", "run", "--rm", "--gpus", "all", probe_img, "nvidia-smi"], + capture_output=True, text=True, timeout=60, + ) + if r.returncode == 0: + result["host"]["container_toolkit"] = f"ok (probed via {probe_img})" + else: + result["host"]["container_toolkit"] = f"failed (probed via {probe_img})" + _gap("`docker run --gpus all` failed; install nvidia-container-toolkit and ensure the host driver supports it") + except Exception as e: + result["host"]["container_toolkit"] = f"error: {e}" + _gap(f"nvidia-container-toolkit probe failed: {e}") +else: + result["host"]["container_toolkit"] = "skipped (no probe image present locally; run ensure_image first)" + +print(json.dumps(result, indent=2)) +PYEOF +} + +kermt_check_gpu() { + # Probes whether `docker --gpus all` is wired up (nvidia-container-toolkit). + # Image-selection priority (never pulls anything): + # 1) $KERMT_IMAGE if it exists locally, + # 2) else nvidia/cuda:12.6.3-base-ubuntu22.04 if it exists locally, + # 3) else skip with a warning (return 0). The smoke test inside kermt_run + # will catch broken GPU passthrough later anyway. + local probe_img="" + if docker image inspect "$KERMT_IMAGE" >/dev/null 2>&1; then + probe_img="$KERMT_IMAGE" + elif docker image inspect nvidia/cuda:12.6.3-base-ubuntu22.04 >/dev/null 2>&1; then + probe_img="nvidia/cuda:12.6.3-base-ubuntu22.04" + else + echo "[kermt] check_gpu: skipped — neither '$KERMT_IMAGE' nor 'nvidia/cuda:12.6.3-base-ubuntu22.04' is present locally. Run 'ensure_image' first, or this probe will be exercised by the in-container smoke test." >&2 + return 0 + fi + if ! docker run --rm --gpus all "$probe_img" nvidia-smi >/dev/null 2>&1; then + echo "[kermt] error: 'docker run --gpus all' failed (probe image: $probe_img). Install nvidia-container-toolkit and ensure the host has a CUDA-capable NVIDIA driver." >&2 + return 1 + fi +} + +# ----------------------------------------------------------------------------- +# Image build / verification +# ----------------------------------------------------------------------------- + +kermt_ensure_image() { + _kermt_require_repo || return $? + kermt_check_docker || return $? + if docker image inspect "$KERMT_IMAGE" >/dev/null 2>&1; then + local id + id=$(docker image inspect "$KERMT_IMAGE" --format '{{.Id}}' 2>/dev/null | cut -c1-19) + echo "[kermt] image '$KERMT_IMAGE' already present (${id:-unknown})" + return 0 + fi + echo "[kermt] image '$KERMT_IMAGE' not found; building from $KERMT_REPO/Dockerfile" + echo "[kermt] first build typically takes 10-20 minutes on a typical workstation; subsequent runs reuse the cached image" + docker build -t "$KERMT_IMAGE" -f "$KERMT_REPO/Dockerfile" "$KERMT_REPO" +} + +# ----------------------------------------------------------------------------- +# Mount-flag parser, internal +# ----------------------------------------------------------------------------- +# Reads flags from the caller's positional args until it hits '--', appending +# `-v src:dst[:ro]` pairs into the caller-provided array name (passed as $1). +# Returns the number of caller-provided args consumed via _kermt_consumed. +# This is bash-specific (uses nameref via `declare -n`). + +_kermt_parse_mounts() { + local -n _out="$1" + shift + _kermt_consumed=0 + while [[ $# -gt 0 ]]; do + case "$1" in + --) + return 0 + ;; + --data) + [[ -e "$2" ]] || { echo "[kermt] --data path not found: $2" >&2; return 1; } + # If the user passes a file, mount its parent directory at /data so + # downstream commands can refer to /data/. Mounting a + # single file at /data makes the path-as-directory pattern in the + # skill examples (`--csv /data/`) fail with "not found". + if [[ -d "$2" ]]; then + _out+=("-v" "$(realpath "$2"):/data:ro") + else + _out+=("-v" "$(realpath "$(dirname "$2")"):/data:ro") + fi + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --ckpt) + [[ -e "$2" ]] || { echo "[kermt] --ckpt path not found: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/ckpt:ro") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --vocab-dir) + [[ -d "$2" ]] || { echo "[kermt] --vocab-dir not found or not a directory: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/vocab:ro") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --run-dir) + mkdir -p "$2" || { echo "[kermt] failed to create --run-dir: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/runs") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --model-dir) + mkdir -p "$2" || { echo "[kermt] failed to create --model-dir: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/model") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + *) + return 0 + ;; + esac + done +} + +# ----------------------------------------------------------------------------- +# Foreground / detached run +# ----------------------------------------------------------------------------- + +# Capture host-side git state for the repo and emit `-e KERMT_REPO_COMMIT=… +# -e KERMT_REPO_DIRTY=true|false` flags. Used by the run / run_detached +# wrappers so the runner's run.json manifest gets honest commit info even +# though `git -C /workspace` inside the container fails due to bind-mount +# ownership. +_kermt_git_env_flags() { + local commit="unknown" + local dirty="false" + if command -v git >/dev/null 2>&1 && [[ -d "$KERMT_REPO/.git" ]]; then + local c + c=$(git -C "$KERMT_REPO" rev-parse HEAD 2>/dev/null) && commit="$c" + # `--untracked-files=no` filters out user-private notes (e.g. a CLAUDE.md + # or RELEASE_PLAN_v2.0.md at the repo root) that wouldn't affect + # reproducibility — only modifications to tracked files do. + if [[ -n "$(git -C "$KERMT_REPO" status --porcelain --untracked-files=no 2>/dev/null | head -n 1)" ]]; then + dirty="true" + fi + fi + printf '%s\n%s\n%s\n%s\n' "-e" "KERMT_REPO_COMMIT=$commit" "-e" "KERMT_REPO_DIRTY=$dirty" +} + +# Forward HF_TOKEN into the container when it is set, so fetch_released_model.py +# can authenticate to Hugging Face. The current release is public (no token +# needed); this only guards against shared-IP rate limits or a future gated +# repo. Emits nothing when HF_TOKEN is unset. +_kermt_hf_env_flags() { + if [[ -n "${HF_TOKEN:-}" ]]; then + printf '%s\n%s\n' "-e" "HF_TOKEN=$HF_TOKEN" + fi +} + +kermt_run() { + kermt_ensure_image || return $? + local mount_args=() + _kermt_parse_mounts mount_args "$@" || return $? + shift "$_kermt_consumed" + if [[ "${1:-}" != "--" ]]; then + echo "[kermt] expected '--' separating mount flags from the command (got '${1:-}')" >&2 + return 1 + fi + shift + if [[ $# -eq 0 ]]; then + echo "[kermt] no command supplied after '--'" >&2 + return 1 + fi + local git_args=() + while IFS= read -r line; do git_args+=("$line"); done < <(_kermt_git_env_flags) + local hf_args=() + while IFS= read -r line; do hf_args+=("$line"); done < <(_kermt_hf_env_flags) + docker run --rm --gpus "$KERMT_GPUS" \ + --user "$(id -u):$(id -g)" \ + -v "$KERMT_REPO:/workspace" \ + -v "$_kermt_bundle_dir:/skill:ro" \ + "${mount_args[@]}" \ + -w /workspace \ + -e KERMT_REPO=/workspace \ + -e PYTHONPATH=/workspace \ + -e HOME=/tmp/kermt-home \ + "${git_args[@]}" \ + "${hf_args[@]}" \ + "$KERMT_IMAGE" \ + conda run -n kermt --no-capture-output bash -c "$*" +} + +kermt_run_detached() { + kermt_ensure_image || return $? + local name="" + local mount_args=() + # Pull --name out first, then let the shared mount parser handle the rest. + while [[ $# -gt 0 ]]; do + case "$1" in + --name) name="$2"; shift 2 ;; + --) break ;; + --data|--ckpt|--vocab-dir|--run-dir|--model-dir) break ;; + *) break ;; + esac + done + _kermt_parse_mounts mount_args "$@" || return $? + shift "$_kermt_consumed" + if [[ "${1:-}" != "--" ]]; then + echo "[kermt] expected '--' separating mount flags from the command (got '${1:-}')" >&2 + return 1 + fi + shift + if [[ $# -eq 0 ]]; then + echo "[kermt] no command supplied after '--'" >&2 + return 1 + fi + if [[ -z "$name" ]]; then + name="kermt-$(date -u +%Y%m%dT%H%M%SZ)-$$" + fi + local cid + local git_args=() + while IFS= read -r line; do git_args+=("$line"); done < <(_kermt_git_env_flags) + local hf_args=() + while IFS= read -r line; do hf_args+=("$line"); done < <(_kermt_hf_env_flags) + cid=$(docker run -d --gpus "$KERMT_GPUS" \ + --user "$(id -u):$(id -g)" \ + --name "$name" \ + -v "$KERMT_REPO:/workspace" \ + -v "$_kermt_bundle_dir:/skill:ro" \ + "${mount_args[@]}" \ + -w /workspace \ + -e KERMT_REPO=/workspace \ + -e PYTHONPATH=/workspace \ + -e HOME=/tmp/kermt-home \ + "${git_args[@]}" \ + "${hf_args[@]}" \ + "$KERMT_IMAGE" \ + conda run -n kermt --no-capture-output bash -c "$*") || return $? + echo "[kermt] container started: name=$name id=$cid" + echo "[kermt] follow logs: docker logs -f $name" + echo "[kermt] wait for exit: docker wait $name" + echo "[kermt] stop: docker stop $name" + echo "$cid" +} + +# ----------------------------------------------------------------------------- +# Subcommand dispatch when invoked directly (not sourced) +# ----------------------------------------------------------------------------- + +if [[ "${BASH_SOURCE[0]:-$0}" == "${0}" ]]; then + cmd="${1:-}"; shift || true + case "$cmd" in + check_docker) kermt_check_docker "$@" ;; + check_gpu) kermt_check_gpu "$@" ;; + check_system) kermt_check_system "$@" ;; + ensure_image) kermt_ensure_image "$@" ;; + run) kermt_run "$@" ;; + run_detached) kermt_run_detached "$@" ;; + ""|-h|--help) + cat >&2 < [args...] + +Subcommands: + check_docker Verify docker is installed and the daemon is reachable. + check_gpu Verify 'docker --gpus all' works (nvidia-container-toolkit). + check_system Emit a JSON probe of host GPU + VRAM + compute_cap + + driver + disk space + container toolkit + image presence. + Exits 0 with ok=false + a 'gaps' list when anything's + below the per-workflow minimum. + ensure_image Build kermt:latest from \$KERMT_REPO/Dockerfile if missing. + run [flags] -- ... Run a command inside the container (foreground, --rm). + run_detached [flags] -- ... + Run detached; prints container name + id + log hint. + +Mount flags (for run / run_detached): + --data bind to /data (read-only) + --ckpt bind to /ckpt (read-only) + --vocab-dir bind to /vocab (read-only) + --run-dir bind to /runs (read-write; created on host if missing) + --model-dir bind to /model (read-write; released-model download target) + +Additional flags for run_detached: + --name container name (default: kermt--) + +Environment overrides: + KERMT_IMAGE default kermt:latest + KERMT_REPO checkout path; otherwise discovered above the skill or working directory + KERMT_GPUS default all +EOF + exit 1 + ;; + *) + echo "[kermt] unknown subcommand: $cmd" >&2 + echo "[kermt] run '$0 --help' for usage" >&2 + exit 1 + ;; + esac +fi diff --git a/skills/kermt-finetune/scripts/prepare_data.py b/skills/kermt-finetune/scripts/prepare_data.py new file mode 100644 index 0000000..f0edb3e --- /dev/null +++ b/skills/kermt-finetune/scripts/prepare_data.py @@ -0,0 +1,817 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Mode-dispatched data preparation pipeline for the KERMT agent skills. + +Composes the existing repo data-prep scripts (`scripts/clean_smiles.py`, +`scripts/save_features.py`, `scripts/build_vocab.py`, `scripts/split_data.py`) +into a single one-call entry point per workflow. Output lands in `--out` with +a `prepare_data.json` manifest that the downstream runners read. + +Mode pipelines +-------------- +pretrain : clean -> (optional auto-split train into train+val by --val-frac) + -> save_features (fgtasklabel) on each CSV + -> vocab step: if --vocab-dir / --{atom,bond,smiles}-vocab given, + copy those through (continue-pretrain case — the ckpt's vocab + is authoritative); else if --skip-vocab, skip; + else build_vocab on train (pretrain-from-scratch case) + -> split_data (graph + feature shards + summary.txt) per CSV +finetune : clean each provided CSV -> (optional random split when only one + CSV is provided; emits a strong warning recommending scaffold- + balanced pre-splits) -> save_features (rdkit_2d_normalized) per CSV +inference : clean -> save_features (rdkit_2d_normalized) +embed : clean only (extract_embeddings.py featurizes on the fly) + +Output convention +----------------- +The manifest under `/prepare_data.json` captures every step's inputs, +outputs, duration, and skipped-due-to-existing flag, plus a top-level +`split_method` field (one of: "user_provided", "random", "n/a") that the +finetune runner uses to pass the correct `--split_type` to main.py. + +Subprocess composition +---------------------- +Each underlying script is invoked via `subprocess.run`. The PYTHONPATH=/workspace +env var (set by `scripts/kermt_container.sh`) makes the `kermt` package +importable inside the subprocesses; without it, build_vocab.py and split_data.py +fail with `ModuleNotFoundError: No module named 'kermt'`. + +CLI +--- + prepare_data.py --mode {pretrain|finetune|inference|embed} + --csv --out + [--val-csv ] [--test-csv ] + [--val-frac 0.1] [--test-frac 0.1] [--seed 0] + [--sample-per-file 100000] [--vocab-format json] + [--dataset-name pretrain] + [--targets COL [COL ...]] + [--features-generator ] + [--smiles-column 0] + [--force] [--skip-clean] [--skip-features] + [--skip-vocab] [--skip-split] +""" +from __future__ import annotations + +import argparse +import json +import os +import shutil +import subprocess +import sys +import time +import traceback +from pathlib import Path +from typing import Any + +import pandas as pd + +# sys.path tweak so `_utils` is importable regardless of how this script +# is invoked (kermt_run sets PYTHONPATH=/workspace; bare-Python launches +# from the host don't). +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import PRETRAIN_VOCAB_STEMS, resolve_kermt_repo, validate_vocab_file # noqa: E402 + + +REPO_ROOT = resolve_kermt_repo() +EXISTING_SCRIPTS = REPO_ROOT / "scripts" + +DEFAULT_FEATURES_GENERATOR = { + "pretrain": "fgtasklabel", + "finetune": "rdkit_2d_normalized", + "inference": "rdkit_2d_normalized", + "embed": None, # not used +} + +VALID_MODES = ("pretrain", "finetune", "inference", "embed") + + +# --------------------------------------------------------------------------- +# Subprocess helpers +# --------------------------------------------------------------------------- + +def _run(cmd: list[str], step_name: str, manifest: dict[str, Any]) -> dict[str, Any]: + """Run a subprocess, append a step entry to manifest, raise on failure.""" + step: dict[str, Any] = { + "name": step_name, + "cmd": cmd, + "duration_s": None, + "ok": False, + "stderr_tail": "", + "skipped_due_to_existing": False, + } + t0 = time.time() + proc = subprocess.run(cmd, capture_output=True, text=True) + step["duration_s"] = round(time.time() - t0, 2) + if proc.returncode != 0: + step["stderr_tail"] = (proc.stderr or "").splitlines()[-20:] + step["ok"] = False + manifest["steps"].append(step) + raise RuntimeError( + f"step '{step_name}' failed (exit {proc.returncode}); " + f"command: {' '.join(cmd)}\nstderr tail:\n" + "\n".join(step["stderr_tail"]) + ) + step["ok"] = True + manifest["steps"].append(step) + return step + + +def _skipped(step_name: str, output_path: str, manifest: dict[str, Any]) -> dict[str, Any]: + step = { + "name": step_name, + "output": output_path, + "ok": True, + "duration_s": 0.0, + "skipped_due_to_existing": True, + } + manifest["steps"].append(step) + return step + + +def _exists_nonempty(path: Path) -> bool: + """File exists with non-zero size, or directory exists with at least one entry.""" + if not path.exists(): + return False + if path.is_file(): + return path.stat().st_size > 0 + if path.is_dir(): + try: + next(path.iterdir()) + return True + except StopIteration: + return False + return False + + +# --------------------------------------------------------------------------- +# Per-script wrappers +# --------------------------------------------------------------------------- + +def _resolve_smiles_column(csv_path: Path, explicit_value: int | None) -> int: + """Return the 0-based index of the SMILES column in csv_path. + + Auto-detection rule when `explicit_value is None`: + 1. Read the CSV header (first non-empty row). + 2. Prefer an exact lowercase `smiles` column (kermt convention). + 3. Otherwise accept a single case-insensitive match + (`SMILES`, `Smiles`, etc.). + 4. If no match (or multiple ambiguous matches), raise a ValueError + that surfaces the header so the user can disambiguate via + `--smiles-column N`. + + Real datasets routinely place SMILES at column index ≠ 0 + (e.g. openadmet's all.csv has "Molecule Name" at col 0 and "SMILES" + at col 1). Auto-detection prevents the silent 0-row-clean failure + mode where every row gets rejected because col 0 doesn't parse as + a SMILES string. + """ + if explicit_value is not None: + return explicit_value + + if not csv_path.is_file(): + raise ValueError(f"input CSV not found: {csv_path}") + + import csv as _csv + with csv_path.open("r", newline="") as f: + reader = _csv.reader(f) + try: + header = next(reader) + except StopIteration: + raise ValueError(f"input CSV {csv_path} is empty") + + stripped = [c.strip() for c in header] + # Prefer exact lowercase "smiles" + exact = [i for i, c in enumerate(stripped) if c == "smiles"] + if exact: + return exact[0] + # Then case-insensitive + ci = [i for i, c in enumerate(stripped) if c.lower() == "smiles"] + if len(ci) == 1: + return ci[0] + if len(ci) > 1: + raise ValueError( + f"input CSV {csv_path} has multiple SMILES-named columns: " + f"{[header[i] for i in ci]} at indices {ci}. " + "Pass --smiles-column N (0-based) to disambiguate." + ) + raise ValueError( + f"could not auto-detect a SMILES column in {csv_path}. " + f"Header columns: {header}. " + "Pass --smiles-column N (0-based) to specify which column holds SMILES." + ) + + +def _clean_smiles( + input_csv: Path, output_csv: Path, smiles_column: int, manifest: dict[str, Any], force: bool +) -> Path: + if not force and _exists_nonempty(output_csv): + _skipped(f"clean_smiles({input_csv.name})", str(output_csv), manifest) + return output_csv + output_csv.parent.mkdir(parents=True, exist_ok=True) + if force and output_csv.exists(): + # clean_smiles.py prompts interactively (input()) when the output file + # already exists — that's an EOFError in a non-TTY subprocess. Pre-delete. + output_csv.unlink() + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "clean_smiles.py"), + "--input", str(input_csv), + "--output", str(output_csv), + "--smiles_column", str(smiles_column), + ] + _run(cmd, f"clean_smiles({input_csv.name})", manifest) + return output_csv + + +def _reduce_to_smiles_column( + csv_path: Path, smiles_column: int, manifest: dict[str, Any] +) -> Path: + """Rewrite an inference CSV to keep only the SMILES column (at index 0). + + Downstream `kermt.util.utils.get_data` -> `MoleculeDatapoint.__init__` + floats every column after SMILES, which crashes on non-numeric passthrough + columns (e.g. a 'split' label of 'train'/'val'/'test', or a 'Molecule Name' + string). Inference does not need target columns, so drop them here. + + Note on skip semantics: this step is idempotent — running it on an + already-single-column file is a no-op. We record that with + `skipped_due_to_idempotent: True`, NOT `skipped_due_to_existing: True`. + The two fields have different meanings: `_existing` means "I found a + cached output file from a prior run and reused it" (overridden by + `--force`); `_idempotent` means "the input is already in the desired + state, so re-executing changes nothing" (safe to skip even under + `--force`). + """ + step_name = f"reduce_to_smiles_only({csv_path.name})" + start = time.time() + df = pd.read_csv(csv_path) + if df.shape[1] == 1: + manifest["steps"].append({ + "name": step_name, + "output": str(csv_path), + "ok": True, + "duration_s": time.time() - start, + "skipped_due_to_idempotent": True, + "note": "already single-column", + }) + return csv_path + effective_col = smiles_column if 0 <= smiles_column < df.shape[1] else 0 + df.iloc[:, [effective_col]].to_csv(csv_path, index=False) + manifest["steps"].append({ + "name": step_name, + "output": str(csv_path), + "ok": True, + "duration_s": time.time() - start, + "input_cols": int(df.shape[1]), + "kept_col": effective_col, + "kept_col_name": str(df.columns[effective_col]), + }) + return csv_path + + +def _save_features( + csv_path: Path, npz_path: Path, generator: str, manifest: dict[str, Any], force: bool +) -> Path: + if not force and _exists_nonempty(npz_path): + _skipped(f"save_features({csv_path.name}, {generator})", str(npz_path), manifest) + return npz_path + npz_path.parent.mkdir(parents=True, exist_ok=True) + if force and npz_path.exists(): + npz_path.unlink() # --restart still loads partial state if file exists; pre-delete to be safe + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "save_features.py"), + "--data_path", str(csv_path), + "--save_path", str(npz_path), + "--features_generator", generator, + "--restart", + ] + _run(cmd, f"save_features({csv_path.name}, {generator})", manifest) + return npz_path + + +def _resolve_vocab_inputs(args: argparse.Namespace) -> dict[str, Path | None] | None: + """Returns {atom, bond, smiles}->Path|None when the user supplied vocab + inputs (via --vocab-dir or --atom-vocab/--bond-vocab/--smiles-vocab), + else None (signal to fall through to build_vocab). + + Conventional filenames inside --vocab-dir: + pretrain_atom_vocab.{json,pkl} + pretrain_bond_vocab.{json,pkl} + pretrain_smiles_vocab.pkl + """ + if args.vocab_dir: + d = Path(args.vocab_dir).resolve() + if not d.is_dir(): + raise FileNotFoundError(f"--vocab-dir not found or not a directory: {d}") + def _find(stem: str, exts: tuple[str, ...]) -> Path | None: + for ext in exts: + p = d / f"{stem}.{ext}" + if p.is_file(): + return p + return None + atom = _find(PRETRAIN_VOCAB_STEMS["atom"], ("json", "pkl")) + bond = _find(PRETRAIN_VOCAB_STEMS["bond"], ("json", "pkl")) + smiles = _find(PRETRAIN_VOCAB_STEMS["smiles"], ("pkl",)) + if atom is None and bond is None and smiles is None: + stems = [PRETRAIN_VOCAB_STEMS[k] for k in ("atom", "bond", "smiles")] + raise FileNotFoundError( + f"--vocab-dir {d} contained no {{ {', '.join(stems) }}}.{{json,pkl}} " + f"files. Expected at least {PRETRAIN_VOCAB_STEMS['atom']} + " + f"{PRETRAIN_VOCAB_STEMS['bond']}." + ) + return {"atom": atom, "bond": bond, "smiles": smiles} + + if args.atom_vocab or args.bond_vocab or args.smiles_vocab: + return { + "atom": Path(args.atom_vocab).resolve() if args.atom_vocab else None, + "bond": Path(args.bond_vocab).resolve() if args.bond_vocab else None, + "smiles": Path(args.smiles_vocab).resolve() if args.smiles_vocab else None, + } + + return None + + +def _copy_provided_vocab( + src: dict[str, Path | None], dst_dir: Path, dataset_name: str, manifest: dict[str, Any], + force: bool, +) -> dict[str, Path]: + """When the user supplies vocab files (use ckpt's vocab as-is), + copy them into `/__vocab.` so the + downstream pretrain command sees the conventional filenames. + + `src` is `{atom: Path|None, bond: Path|None, smiles: Path|None}`. The atom + and bond entries must be both present or both absent (paired). smiles is + optional (cmim/hybrid only). + + Returns the same dict of (resolved) destination paths. + """ + import shutil + if (src["atom"] is None) != (src["bond"] is None): + raise ValueError( + "vocab pass-through requires atom and bond vocab paths to be paired; " + "got atom=" + str(src["atom"]) + ", bond=" + str(src["bond"]) + ) + out: dict[str, Path] = {} + dst_dir.mkdir(parents=True, exist_ok=True) + for which, path in src.items(): + if path is None: + continue + # Validate the source file IS a loadable KERMT vocab before copying. + # Catches the "user pointed --smiles-vocab at a random pickle" case + # early, with a clear error, instead of letting it surface as a cryptic + # SMILESVocab.load_vocab failure at pretrain_ddp.py launch time. + validate_vocab_file(path, kind=which) + ext = path.suffix.lstrip(".") + if which == "smiles": + ext = "pkl" # smiles vocab is always pickle + dst = dst_dir / f"{dataset_name}_{which}_vocab.{ext}" + if not force and _exists_nonempty(dst): + _skipped(f"copy_vocab({which})", str(dst), manifest) + out[which] = dst + continue + if force and dst.exists(): + dst.unlink() + shutil.copy2(path, dst) + manifest["steps"].append({ + "name": f"copy_vocab({which})", + "src": str(path), "dst": str(dst), "ok": True, + "duration_s": 0.0, "skipped_due_to_existing": False, + }) + out[which] = dst + return out + + +def _build_vocab( + csv_path: Path, vocab_dir: Path, dataset_name: str, vocab_format: str, + manifest: dict[str, Any], force: bool, +) -> dict[str, Path]: + """Builds atom + bond (in --vocab-format) and smiles (always pickle) vocabs. + Returns a dict of {atom, bond, smiles} -> Path.""" + suffix = "json" if vocab_format == "json" else "pkl" + expected = { + "atom": vocab_dir / f"{dataset_name}_atom_vocab.{suffix}", + "bond": vocab_dir / f"{dataset_name}_bond_vocab.{suffix}", + "smiles": vocab_dir / f"{dataset_name}_smiles_vocab.pkl", + } + if not force and all(_exists_nonempty(p) for p in expected.values()): + _skipped(f"build_vocab({csv_path.name})", str(vocab_dir), manifest) + return expected + vocab_dir.mkdir(parents=True, exist_ok=True) + if force: + for p in expected.values(): + if p.exists(): + p.unlink() + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "build_vocab.py"), + "--data_path", str(csv_path), + "--vocab_save_folder", str(vocab_dir), + "--dataset_name", dataset_name, + "--vocab_format", vocab_format, + ] + _run(cmd, f"build_vocab({csv_path.name})", manifest) + return expected + + +def _split_data( + csv_path: Path, features_path: Path | None, sample_per_file: int, output_dir: Path, + manifest: dict[str, Any], force: bool, +) -> Path: + """Run split_data.py to produce shard dirs (graph/ + optionally feature/ + summary.txt).""" + summary = output_dir / "summary.txt" + if not force and _exists_nonempty(summary): + _skipped(f"split_data({csv_path.name})", str(output_dir), manifest) + return output_dir + if force and output_dir.exists(): + shutil.rmtree(output_dir) + output_dir.mkdir(parents=True, exist_ok=True) + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "split_data.py"), + "--data_path", str(csv_path), + "--sample_per_file", str(sample_per_file), + "--output_path", str(output_dir), + ] + if features_path is not None: + cmd += ["--features_path", str(features_path)] + _run(cmd, f"split_data({csv_path.name})", manifest) + return output_dir + + +# --------------------------------------------------------------------------- +# Random splitter (used only when the user supplies a single CSV) +# --------------------------------------------------------------------------- + +def _random_split_csv( + src_csv: Path, dst_csvs: dict[str, Path], fractions: dict[str, float], seed: int, + manifest: dict[str, Any], force: bool, +) -> None: + """Shuffle src_csv and partition rows into dst_csvs by fractions. + `dst_csvs` and `fractions` are dicts keyed by the split name (e.g. 'train', 'val'). + Sum of fractions must be 1.0 (within float tolerance). Writes each dst_csv with the + same header as the input.""" + step = { + "name": f"random_split({src_csv.name})", + "seed": seed, + "fractions": fractions, + "ok": False, + "duration_s": None, + "skipped_due_to_existing": False, + "row_counts": {}, + } + if not force and all(_exists_nonempty(p) for p in dst_csvs.values()): + step["skipped_due_to_existing"] = True + step["ok"] = True + manifest["steps"].append(step) + return + + if abs(sum(fractions.values()) - 1.0) > 1e-6: + raise ValueError(f"split fractions must sum to 1.0 (got {sum(fractions.values())})") + + t0 = time.time() + df = pd.read_csv(src_csv).sample(frac=1.0, random_state=seed).reset_index(drop=True) + n = len(df) + sizes: dict[str, int] = {} + remaining = n + split_names = list(fractions.keys()) + for name in split_names[:-1]: + sizes[name] = int(round(fractions[name] * n)) + remaining -= sizes[name] + sizes[split_names[-1]] = remaining + + start = 0 + for name in split_names: + dst = dst_csvs[name] + dst.parent.mkdir(parents=True, exist_ok=True) + df.iloc[start:start + sizes[name]].to_csv(dst, index=False) + step["row_counts"][name] = sizes[name] + start += sizes[name] + + step["duration_s"] = round(time.time() - t0, 2) + step["ok"] = True + manifest["steps"].append(step) + + +def _emit_random_split_warning( + src_csv: Path, fractions: dict[str, float], seed: int, manifest: dict[str, Any] +) -> None: + row_counts = manifest["steps"][-1].get("row_counts", {}) + n = sum(row_counts.values()) if row_counts else "?" + lines = [ + f"WARNING: Auto-splitting {n} rows from {src_csv.name} into:", + ] + for name, frac in fractions.items(): + cnt = row_counts.get(name, "?") + lines.append(f" {name}: {cnt} rows ({frac * 100:.1f}%)") + lines += [ + f"using random split with seed {seed}.", + "", + "This is a RANDOM split. For rigorous ADMET evaluation, scaffold-balanced", + "(or other structure-aware) splits are strongly preferred — molecules with", + "similar scaffolds can leak across splits and inflate apparent generalization.", + "", + "To use your own pre-computed splits instead, pass:", + " --train-csv --val-csv --test-csv ", + "", + "To customize fractions:", + " --val-frac 0.15 --test-frac 0.15", + ] + warning = "\n".join(lines) + print(warning, file=sys.stderr) + manifest["warnings"].append(warning) + + +# --------------------------------------------------------------------------- +# Mode pipelines +# --------------------------------------------------------------------------- + +def _prepare_embed(args, out: Path, manifest: dict[str, Any]) -> None: + manifest["split_method"] = "n/a" + if args.skip_clean: + clean = Path(args.csv) + manifest["steps"].append({"name": "clean_smiles", "skipped_by_flag": True, "ok": True}) + else: + clean = _clean_smiles(Path(args.csv), out / "clean.csv", args.smiles_column, manifest, args.force) + manifest["outputs"]["clean_csv"] = str(clean) + + +def _prepare_inference(args, out: Path, manifest: dict[str, Any]) -> None: + manifest["split_method"] = "n/a" + clean = _clean_smiles(Path(args.csv), out / "clean.csv", args.smiles_column, manifest, args.force) + # Reduce to SMILES-only: downstream get_data/MoleculeDatapoint floats every + # non-SMILES column, which crashes on non-numeric passthrough columns + # (e.g. a 'split' label). Inference does not need target columns. + _reduce_to_smiles_column(clean, args.smiles_column, manifest) + manifest["outputs"]["clean_csv"] = str(clean) + if args.skip_features: + manifest["steps"].append({"name": "save_features", "skipped_by_flag": True, "ok": True}) + return + generator = args.features_generator or DEFAULT_FEATURES_GENERATOR["inference"] + npz = _save_features(clean, out / "clean.npz", generator, manifest, args.force) + manifest["outputs"]["clean_npz"] = str(npz) + + +def _prepare_finetune(args, out: Path, manifest: dict[str, Any]) -> None: + src_train = Path(args.csv) + has_val = args.val_csv is not None + has_test = args.test_csv is not None + split_type = args.split_type + + if has_val and has_test: + # User supplied explicit val + test CSVs: trust them, just clean + featurize. + # split_type is irrelevant when val/test are given separately. + manifest["split_method"] = "user_provided" + clean_train = _clean_smiles(src_train, out / "clean_train.csv", args.smiles_column, manifest, args.force) + clean_val = _clean_smiles(Path(args.val_csv), out / "clean_val.csv", args.smiles_column, manifest, args.force) + clean_test = _clean_smiles(Path(args.test_csv), out / "clean_test.csv", args.smiles_column, manifest, args.force) + manifest["outputs"]["clean_train_csv"] = str(clean_train) + manifest["outputs"]["clean_val_csv"] = str(clean_val) + manifest["outputs"]["clean_test_csv"] = str(clean_test) + per_split = (("train", clean_train), ("val", clean_val), ("test", clean_test)) + elif has_val or has_test: + raise ValueError( + "for finetune mode, either provide BOTH --val-csv and --test-csv (user-provided splits) " + "or NEITHER (run with --split-type {random|scaffold_balanced|index_predetermined}). " + "Got one but not both." + ) + elif split_type == "random": + # Random auto-split — done here in prep so train.py gets ready-made CSVs. + manifest["split_method"] = "random" + manifest["split_seed"] = args.seed + train_frac = max(0.0, 1.0 - args.val_frac - args.test_frac) + manifest["split_fractions"] = {"train": train_frac, "val": args.val_frac, "test": args.test_frac} + clean_full = _clean_smiles(src_train, out / "_clean_full.csv", args.smiles_column, manifest, args.force) + dst = { + "train": out / "clean_train.csv", + "val": out / "clean_val.csv", + "test": out / "clean_test.csv", + } + _random_split_csv(clean_full, dst, manifest["split_fractions"], args.seed, manifest, args.force) + clean_train, clean_val, clean_test = dst["train"], dst["val"], dst["test"] + manifest["outputs"]["clean_train_csv"] = str(clean_train) + manifest["outputs"]["clean_val_csv"] = str(clean_val) + manifest["outputs"]["clean_test_csv"] = str(clean_test) + _emit_random_split_warning(src_train, manifest["split_fractions"], args.seed, manifest) + per_split = (("train", clean_train), ("val", clean_val), ("test", clean_test)) + else: + # Scaffold-balanced or index-predetermined: prep cleans + featurizes the full + # CSV and defers actual splitting to task/train.py, which calls split_data + # with the user-supplied seed and split_sizes. + manifest["split_method"] = "deferred_to_runner" + manifest["split_type"] = split_type + manifest["split_seed"] = args.seed + manifest["split_fractions"] = { + "train": max(0.0, 1.0 - args.val_frac - args.test_frac), + "val": args.val_frac, + "test": args.test_frac, + } + clean_full = _clean_smiles(src_train, out / "clean_full.csv", args.smiles_column, manifest, args.force) + manifest["outputs"]["clean_full_csv"] = str(clean_full) + per_split = (("full", clean_full),) + + if args.skip_features: + manifest["steps"].append({"name": "save_features", "skipped_by_flag": True, "ok": True}) + return + generator = args.features_generator or DEFAULT_FEATURES_GENERATOR["finetune"] + for split_name, csv in per_split: + npz = _save_features(csv, csv.with_suffix(".npz"), generator, manifest, args.force) + manifest["outputs"][f"clean_{split_name}_npz"] = str(npz) + + +def _prepare_pretrain(args, out: Path, manifest: dict[str, Any]) -> None: + src_train = Path(args.csv) + if args.val_csv is not None: + manifest["split_method"] = "user_provided" + clean_train = _clean_smiles(src_train, out / "clean_train.csv", args.smiles_column, manifest, args.force) + clean_val = _clean_smiles(Path(args.val_csv), out / "clean_val.csv", args.smiles_column, manifest, args.force) + else: + manifest["split_method"] = "random" + manifest["split_seed"] = args.seed + train_frac = max(0.0, 1.0 - args.val_frac) + manifest["split_fractions"] = {"train": train_frac, "val": args.val_frac} + clean_full = _clean_smiles(src_train, out / "_clean_full.csv", args.smiles_column, manifest, args.force) + dst = {"train": out / "clean_train.csv", "val": out / "clean_val.csv"} + _random_split_csv(clean_full, dst, manifest["split_fractions"], args.seed, manifest, args.force) + clean_train, clean_val = dst["train"], dst["val"] + + manifest["outputs"]["clean_train_csv"] = str(clean_train) + manifest["outputs"]["clean_val_csv"] = str(clean_val) + + generator = args.features_generator or DEFAULT_FEATURES_GENERATOR["pretrain"] + if args.skip_features: + manifest["steps"].append({"name": "save_features", "skipped_by_flag": True, "ok": True}) + train_npz: Path | None = None + val_npz: Path | None = None + else: + train_npz = _save_features(clean_train, out / "clean_train.npz", generator, manifest, args.force) + val_npz = _save_features(clean_val, out / "clean_val.npz", generator, manifest, args.force) + manifest["outputs"]["clean_train_npz"] = str(train_npz) + manifest["outputs"]["clean_val_npz"] = str(val_npz) + + if args.skip_vocab: + manifest["steps"].append({"name": "build_vocab", "skipped_by_flag": True, "ok": True}) + manifest["vocab_source"] = "skipped" + else: + # Resolve user-provided vocab paths from --vocab-dir or explicit flags. + provided = _resolve_vocab_inputs(args) + if provided: + # Use the user-supplied (ckpt's) vocab as-is. Copy into the + # conventional filenames the downstream pretrain command expects. + vocabs = _copy_provided_vocab(provided, out, args.dataset_name, manifest, args.force) + manifest["vocab_source"] = "user_provided" + else: + # Fall back to the existing build-from-corpus behavior. Used by + # pretrain-from-scratch and by any continue case where the user + # explicitly wants a fresh vocab (rare, usually wrong). + vocabs = _build_vocab(clean_train, out, args.dataset_name, args.vocab_format, manifest, args.force) + manifest["vocab_source"] = "built_fresh" + if "atom" in vocabs: + manifest["outputs"]["atom_vocab"] = str(vocabs["atom"]) + if "bond" in vocabs: + manifest["outputs"]["bond_vocab"] = str(vocabs["bond"]) + if "smiles" in vocabs: + manifest["outputs"]["smiles_vocab"] = str(vocabs["smiles"]) + + if args.skip_split: + manifest["steps"].append({"name": "split_data", "skipped_by_flag": True, "ok": True}) + else: + train_dir = _split_data(clean_train, train_npz, args.sample_per_file, out / "train", manifest, args.force) + val_dir = _split_data(clean_val, val_npz, args.sample_per_file, out / "val", manifest, args.force) + manifest["outputs"]["train_dir"] = str(train_dir) + manifest["outputs"]["val_dir"] = str(val_dir) + + +# --------------------------------------------------------------------------- +# Entry point +# --------------------------------------------------------------------------- + +def prepare(args: argparse.Namespace) -> dict[str, Any]: + out = Path(args.out).resolve() + out.mkdir(parents=True, exist_ok=True) + manifest: dict[str, Any] = { + "mode": args.mode, + "input_csv": str(Path(args.csv).resolve()), + "val_csv": str(Path(args.val_csv).resolve()) if args.val_csv else None, + "test_csv": str(Path(args.test_csv).resolve()) if args.test_csv else None, + "output_dir": str(out), + "split_method": None, + "steps": [], + "outputs": {}, + "errors": [], + "warnings": [], + } + try: + if args.mode == "pretrain": + _prepare_pretrain(args, out, manifest) + elif args.mode == "finetune": + _prepare_finetune(args, out, manifest) + elif args.mode == "inference": + _prepare_inference(args, out, manifest) + elif args.mode == "embed": + _prepare_embed(args, out, manifest) + manifest["ok"] = True + except Exception as exc: # noqa: BLE001 + manifest["ok"] = False + manifest["errors"].append(f"{type(exc).__name__}: {exc}") + # Always write the manifest so partial-failure state is visible to the agent. + (out / "prepare_data.json").write_text(json.dumps(manifest, indent=2)) + return manifest + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Mode-dispatched data prep for the KERMT agent skills.") + p.add_argument("--mode", required=True, choices=VALID_MODES) + p.add_argument("--csv", required=True, help="Primary input CSV (train CSV for pretrain/finetune)") + p.add_argument("--out", required=True, help="Output directory") + p.add_argument("--val-csv", default=None, help="Optional separate val CSV (pretrain/finetune)") + p.add_argument("--test-csv", default=None, help="Optional separate test CSV (finetune only)") + p.add_argument("--val-frac", type=float, default=0.1, help="Auto-split val fraction (default 0.1)") + p.add_argument("--test-frac", type=float, default=0.1, help="Auto-split test fraction (finetune only, default 0.1)") + p.add_argument("--seed", type=int, default=0, help="Random split seed (default 0)") + p.add_argument("--split-type", choices=["random", "scaffold_balanced", "index_predetermined"], + default="random", + help="(finetune only, when --val-csv/--test-csv are not given) how to split. " + "'random' splits in prep using --val-frac/--test-frac/--seed. " + "'scaffold_balanced' and 'index_predetermined' defer the actual split to the " + "runner (task/train.py invokes split_data with the appropriate algorithm " + "using the user-supplied seed); prep only cleans + featurizes the full CSV.") + p.add_argument("--sample-per-file", type=int, default=100_000, + help="split_data shard size (pretrain only, default 100000)") + p.add_argument("--vocab-format", choices=["json", "pkl"], default="json", + help="atom/bond vocab format (default json); smiles vocab is always pkl") + # Vocab pass-through (pretrain mode): when continuing from a released ckpt, + # pass its bundled vocab files in so we don't rebuild a mismatched vocab. + p.add_argument("--vocab-dir", default=None, + help="(pretrain) directory containing pretrain_{atom,bond}_vocab.{json,pkl} " + "(+ pretrain_smiles_vocab.pkl for cmim/hybrid). When given, prepare_data " + "skips build_vocab and copies these files into the output dir under the " + "expected filenames. Used by kermt-continue-pretrain to bind the released " + "ckpt's vocab to the new corpus (the ckpt's vocab is authoritative).") + p.add_argument("--atom-vocab", default=None, + help="(pretrain) explicit atom vocab path; pairs with --bond-vocab. Overrides " + "--vocab-dir's pretrain_atom_vocab.* discovery if both are given.") + p.add_argument("--bond-vocab", default=None, + help="(pretrain) explicit bond vocab path; pairs with --atom-vocab.") + p.add_argument("--smiles-vocab", default=None, + help="(pretrain, cmim/hybrid) explicit smiles vocab .pkl path. Optional for " + "vocab-only pretrain.") + p.add_argument("--dataset-name", default="pretrain", + help="vocab filename prefix (default 'pretrain' so downstream pretrain commands " + "can reference pretrain_{atom,bond}_vocab.{json|pkl}, pretrain_smiles_vocab.pkl)") + p.add_argument("--targets", nargs="+", default=None, + help="(finetune only) target column names; forwarded to the finetune runner via the manifest") + p.add_argument("--features-generator", default=None, + help="Override the per-mode default (pretrain: fgtasklabel; finetune/inference: rdkit_2d_normalized)") + p.add_argument("--smiles-column", type=int, default=None, + help="0-based column index of SMILES in the input CSV. " + "When omitted, auto-detected by header name " + "(prefers lowercase `smiles`; accepts case-insensitive " + "`SMILES`/`Smiles`). Pass explicitly to override.") + p.add_argument("--force", action="store_true", + help="Re-run every step even if its outputs already exist") + p.add_argument("--skip-clean", action="store_true", help="(embed mode) skip the cleaning step") + p.add_argument("--skip-features", action="store_true", help="Skip feature generation") + p.add_argument("--skip-vocab", action="store_true", help="(pretrain) skip vocab build") + p.add_argument("--skip-split", action="store_true", help="(pretrain) skip shard split") + args = p.parse_args(argv) + + # Forward --targets through the manifest so the finetune runner can see them. + if args.mode == "finetune" and args.targets: + pass # captured in manifest below + + # Resolve the SMILES column index (auto-detect from header when the user + # didn't pass --smiles-column). This is the only point where args.csv is + # touched before downstream _clean_smiles calls fan it out. + try: + resolved_smiles_col = _resolve_smiles_column(Path(args.csv), args.smiles_column) + except ValueError as exc: + err_manifest = { + "ok": False, + "mode": args.mode, + "errors": [f"smiles-column resolution failed: {exc}"], + } + Path(args.out).mkdir(parents=True, exist_ok=True) + (Path(args.out) / "prepare_data.json").write_text(json.dumps(err_manifest, indent=2)) + print(json.dumps(err_manifest, indent=2)) + return 1 + if args.smiles_column is None: + print(f"[prepare_data] auto-detected --smiles-column {resolved_smiles_col} " + f"from {Path(args.csv).name} header", file=sys.stderr) + args.smiles_column = resolved_smiles_col + + try: + manifest = prepare(args) + except Exception as exc: # noqa: BLE001 + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({"ok": False, "errors": [f"unhandled: {type(exc).__name__}: {exc}"]}, indent=2)) + return 1 + if args.targets: + manifest["targets"] = list(args.targets) + # Record the resolved SMILES column so the manifest is self-describing. + manifest["smiles_column"] = args.smiles_column + (Path(args.out) / "prepare_data.json").write_text(json.dumps(manifest, indent=2)) + print(json.dumps(manifest, indent=2)) + return 0 if manifest.get("ok") else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-finetune/scripts/run_finetune_local.py b/skills/kermt-finetune/scripts/run_finetune_local.py new file mode 100644 index 0000000..167d8a7 --- /dev/null +++ b/skills/kermt-finetune/scripts/run_finetune_local.py @@ -0,0 +1,468 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Workstation finetune runner — composes prepare_data + check_checkpoint outputs +into a `main.py finetune` invocation (which calls task/cross_validate.py). + +The runner is responsible for: + - reading the prepare_data.json manifest produced by prepare_data.py --mode finetune + - validating the input ckpt is a pretrain checkpoint (via check_checkpoint.py + --mode finetune_init) and extracting its encoder arch + - merging hyperparameters from defaults_finetune.json with per-flag CLI overrides + - building the final main.py finetune argv (data + features + arch + training) + - emitting a reproducible run.json manifest (source-repo commit, image digest, + cmd_replay, args_applied with per-flag source attribution) + - executing the finetune (or writing the manifest only, with --dry-run) + +Arch params come exclusively from the ckpt's saved_args (via check_checkpoint.py). +User-supplied arch flags are not exposed; the runner refuses to override what +the ckpt dictates so the FFN heads attach to a consistent encoder. + +Single-GPU is the default for finetune. `--gpus N` picks a specific device id +(defaults to GPU 0). For data-parallel multi-GPU finetuning, pass `--num-gpus N` +(N>1): the runner sets `WORLD_SIZE=N` and `main.py finetune` spawns one process +per GPU, and `--batch-size` is interpreted per-GPU. + +CLI +--- + run_finetune_local.py + --ckpt # input pretrain ckpt (required) + --prepare-manifest # prepare_data.json (mode=finetune) + --out # output dir + --dataset-type {regression|classification|multiclass} # required + [--ckpt-validator-out ] # cached check_checkpoint.py JSON + [--gpus 0] # single GPU id (default 0) + [--dry-run] # write run.json + print command, do not execute + [--epochs N] [--batch-size N] [--init-lr F] [--max-lr F] [--final-lr F] + [--warmup-epochs F] [--weight-decay F] [--dropout F] + [--metric NAME] [--seed N] + [--ensemble-size N] [--num-folds N] + [--split-sizes TRAIN VAL TEST] + [--ffn-hidden-size N] [--ffn-num-layers N] + [--ffn-num-task-specific-layers N] [--ffn-task-specific-hidden-size N] + [--early-stop-epoch N] [--dist-coff F] [--bond-drop-rate F] + [--show-individual-scores] +""" +from __future__ import annotations + +import argparse +import datetime +import json +import os +import subprocess +import sys +from pathlib import Path +from typing import Any + +# sys.path tweak so `_utils` is importable whether launched via kermt_run +# (PYTHONPATH=/workspace) or bare-Python from the host. +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import ( # noqa: E402 + resolve_kermt_repo, assert_prepare_manifest_basics, docker_image_digest, format_cmd_replay, + git_commit_with_env_override, load_json, merge_default_into_applied, + resolve_single_gpu, run_checkpoint_validator, +) + + +REPO_ROOT = resolve_kermt_repo() +SKILL_ROOT = Path(__file__).resolve().parent.parent +DEFAULTS_PATH = SKILL_ROOT / "config" / "defaults_finetune.json" +CHECK_CHECKPOINT_PATH = SKILL_ROOT / "scripts" / "check_checkpoint.py" +MAIN_PY_PATH = REPO_ROOT / "main.py" + +# Architecture fields the runner pulls from the ckpt and refuses to let the +# user override. Mirrors the pretrain runner's ARCH_FLAGS_FROM_CKPT but adds +# the self-attention triple since finetune-time FFN attaches above the +# self-attention layer if the encoder was pretrained with it. +ARCH_FLAGS_FROM_CKPT = ( + "hidden_size", "depth", "num_attn_head", "activation", "backbone", + "embedding_output_type", "self_attention", "attn_hidden", "attn_out", +) + +# Hyperparameter flag groups + their JSON path in defaults_finetune.json. +# Flags listed here are *eligible* for default-config lookup; whether a key +# exists in defaults_finetune.json is independent. Anything in this tuple that +# the user passes via CLI ends up in args_applied with source="user"; anything +# not passed but present in defaults ends up source="default-config". Anything +# in this tuple that's neither passed nor in defaults is simply absent from +# args_applied (so the argv-builder skips it and main.py finetune's own +# argparse default takes effect). +TRAINING_FLAGS = ( + "epochs", "batch_size", "init_lr", "max_lr", "final_lr", + "dropout", "bond_drop_rate", "dist_coff", "seed", "tensorboard", + "warmup_epochs", "weight_decay", "early_stop_epoch", +) +TASK_FLAGS = ( + "dataset_type", "metric", "split_type", "ensemble_size", "num_folds", + "no_features_scaling", "show_individual_scores", +) +FFN_FLAGS = ( + "ffn_hidden_size", "ffn_num_layers", + "ffn_num_task_specific_layers", "ffn_task_specific_hidden_size", +) + + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + +def _verify_prepare_manifest(manifest: dict[str, Any]) -> None: + assert_prepare_manifest_basics(manifest, "finetune") + out = manifest.get("outputs", {}) + method = manifest.get("split_method") + if method == "deferred_to_runner": + if "clean_full_csv" not in out: + raise ValueError( + "split_method=deferred_to_runner but the manifest doesn't include " + "clean_full_csv. Was prepare_data run with --skip-features but no clean step?" + ) + elif method in ("user_provided", "random"): + missing = [k for k in ("clean_train_csv", "clean_val_csv", "clean_test_csv") if k not in out] + if missing: + raise ValueError( + f"prepare_data manifest (split_method={method}) is missing required outputs: " + f"{missing}. Rerun prepare_data without --skip-features." + ) + else: + raise ValueError( + f"prepare_data manifest has unrecognized split_method='{method}'. " + "Expected one of: user_provided, random, deferred_to_runner." + ) + + +def _arch_from_validator(validator_out: dict[str, Any]) -> dict[str, Any]: + arch = validator_out.get("arch") or {} + # Most fields are required; self_attention/attn_hidden/attn_out are only + # required when self_attention is True. The ckpt's arch dict will have + # self_attention=False with the attn_* fields as None in the common case. + required_always = ("hidden_size", "depth", "num_attn_head", "activation", + "backbone", "embedding_output_type", "self_attention") + missing = [k for k in required_always if arch.get(k) is None] + if missing: + raise ValueError( + f"checkpoint validator did not surface required arch fields: {missing}. " + "The ckpt must have saved_args with these set (the standard " + "save_model_for_restart format)." + ) + if arch.get("self_attention"): + attn_missing = [k for k in ("attn_hidden", "attn_out") if arch.get(k) is None] + if attn_missing: + raise ValueError( + f"ckpt has self_attention=True but arch is missing {attn_missing}; " + "the ckpt's saved_args must include them." + ) + return arch + + +def _apply_defaults(args: argparse.Namespace, defaults: dict[str, Any]) -> dict[str, dict[str, Any]]: + """Merges defaults_finetune.json + CLI overrides into args_applied, keyed by + flag name (snake_case) with {value, source} entries. Source is 'user' if the + user supplied the flag on the CLI, else 'default-config'.""" + applied: dict[str, dict[str, Any]] = {} + training = defaults.get("training", {}) + task_cfg = defaults.get("task", {}) + ffn = defaults.get("ffn_head", {}) + + for f in TRAINING_FLAGS: + merge_default_into_applied(applied, args, f, training) + for f in TASK_FLAGS: + merge_default_into_applied(applied, args, f, task_cfg) + for f in FFN_FLAGS: + merge_default_into_applied(applied, args, f, ffn) + + return applied + + +def _validate_mtl_consistency(applied: dict[str, dict[str, Any]]) -> None: + """If ffn_num_task_specific_layers > 0, ffn_task_specific_hidden_size must be set.""" + n_layers = applied.get("ffn_num_task_specific_layers", {}).get("value", 0) or 0 + hidden = applied.get("ffn_task_specific_hidden_size", {}).get("value") + if n_layers > 0 and not hidden: + raise ValueError( + f"ffn_num_task_specific_layers={n_layers} but ffn_task_specific_hidden_size is " + "unset. When MTL heads are enabled, the hidden size must be supplied — pass " + "--ffn-task-specific-hidden-size H (or set it in defaults_finetune.json)." + ) + + +def _build_argv( + *, gpu: int, out_dir: Path, manifest: dict[str, Any], ckpt: Path, + arch: dict[str, Any], applied: dict[str, dict[str, Any]], + num_gpus: int = 1, +) -> list[str]: + """Constructs the full finetune argv as a list of strings. + + Single-process and multi-GPU (DDP) finetune share one entrypoint, + `main.py finetune ...`. DDP is selected at runtime by the WORLD_SIZE env var + (this runner sets it to num_gpus when num_gpus > 1; main.py then spawns one + process per GPU). The argv is therefore identical for both; num_gpus is kept + only to document that --batch_size is per-GPU under DDP. + """ + outputs = manifest["outputs"] + method = manifest["split_method"] + + argv: list[str] = [sys.executable, "-u", str(MAIN_PY_PATH), "finetune"] + + # Data + features paths + if method == "deferred_to_runner": + argv += ["--data_path", outputs["clean_full_csv"]] + if "clean_full_npz" in outputs: + argv += ["--features_path", outputs["clean_full_npz"]] + # Split is done inside task/train.py via args.split_type/split_sizes/seed. + split_type = manifest.get("split_type") or applied.get("split_type", {}).get("value", "random") + argv += ["--split_type", str(split_type)] + if manifest.get("split_fractions"): + sf = manifest["split_fractions"] + argv += ["--split_sizes", str(sf["train"]), str(sf["val"]), str(sf["test"])] + else: + # user_provided or random: prep produced ready-made train/val/test CSVs + npz. + argv += ["--data_path", outputs["clean_train_csv"]] + argv += ["--separate_val_path", outputs["clean_val_csv"]] + argv += ["--separate_test_path", outputs["clean_test_csv"]] + if "clean_train_npz" in outputs: + argv += ["--features_path", outputs["clean_train_npz"]] + if "clean_val_npz" in outputs: + argv += ["--separate_val_features_path", outputs["clean_val_npz"]] + if "clean_test_npz" in outputs: + argv += ["--separate_test_features_path", outputs["clean_test_npz"]] + # task/train.py with separate_val + separate_test paths won't re-split, + # but split_type is still required by argparse — pass the manifest-or-default value. + split_type_val = applied.get("split_type", {}).get("value", "random") + argv += ["--split_type", str(split_type_val)] + + # Pretrained encoder weights — task/train.py loads them into the model + # via args.checkpoint_paths and then attaches the new FFN head. + argv += ["--checkpoint_path", str(ckpt)] + + # Task semantics + if "dataset_type" in applied: + argv += ["--dataset_type", str(applied["dataset_type"]["value"])] + if "metric" in applied: + argv += ["--metric", str(applied["metric"]["value"])] + if applied.get("no_features_scaling", {}).get("value"): + argv += ["--no_features_scaling"] + if "ensemble_size" in applied: + argv += ["--ensemble_size", str(applied["ensemble_size"]["value"])] + if "num_folds" in applied: + argv += ["--num_folds", str(applied["num_folds"]["value"])] + + # Architecture — sourced from the ckpt's validator output. + argv += [ + "--hidden_size", str(arch["hidden_size"]), + "--depth", str(arch["depth"]), + "--num_attn_head", str(arch["num_attn_head"]), + "--activation", str(arch["activation"]), + "--embedding_output_type", str(arch["embedding_output_type"]), + ] + if arch.get("self_attention"): + argv += ["--self_attention", + "--attn_hidden", str(arch["attn_hidden"]), + "--attn_out", str(arch["attn_out"])] + + # FFN head + if "ffn_hidden_size" in applied: + argv += ["--ffn_hidden_size", str(applied["ffn_hidden_size"]["value"])] + if "ffn_num_layers" in applied: + argv += ["--ffn_num_layers", str(applied["ffn_num_layers"]["value"])] + if applied.get("ffn_num_task_specific_layers", {}).get("value", 0): + argv += ["--ffn_num_task_specific_layers", str(applied["ffn_num_task_specific_layers"]["value"])] + argv += ["--ffn_task_specific_hidden_size", str(applied["ffn_task_specific_hidden_size"]["value"])] + + # Training schedule + for name in ("epochs", "batch_size", "init_lr", "max_lr", "final_lr", + "dropout", "bond_drop_rate", "dist_coff", "seed", + "weight_decay", "warmup_epochs", "early_stop_epoch"): + if name in applied: + argv += [f"--{name}", str(applied[name]["value"])] + + if applied.get("tensorboard", {}).get("value"): + argv += ["--tensorboard"] + if applied.get("show_individual_scores", {}).get("value"): + argv += ["--show_individual_scores"] + + # GPU + save_dir. Under DDP (num_gpus>1) gpu is None: main.py finetune pins + # each rank to its own device, and run_training ignores args.gpu when + # distributed, so no --gpu flag is passed. + if gpu is not None: + argv += ["--gpu", str(gpu)] + argv += ["--save_dir", str(out_dir / "ckpt")] + + return argv + + +# --------------------------------------------------------------------------- +# Main flow +# --------------------------------------------------------------------------- + +def run(args: argparse.Namespace) -> dict[str, Any]: + out_dir = Path(args.out).resolve() + out_dir.mkdir(parents=True, exist_ok=True) + (out_dir / "ckpt").mkdir(parents=True, exist_ok=True) + (out_dir / "logs").mkdir(parents=True, exist_ok=True) + + # 1. Load defaults + prepare manifest. + defaults = load_json(DEFAULTS_PATH, name="defaults_finetune.json") + prep_manifest_path = Path(args.prepare_manifest).resolve() + manifest = load_json(prep_manifest_path, name="prepare_data.json") + _verify_prepare_manifest(manifest) + + # 2. Validate ckpt + extract arch. + ckpt = Path(args.ckpt).resolve() + if args.ckpt_validator_out: + validator_out = load_json(Path(args.ckpt_validator_out), name="ckpt validator output") + else: + validator_out = run_checkpoint_validator(ckpt, mode="finetune_init", script_path=CHECK_CHECKPOINT_PATH) + if not validator_out.get("ok"): + raise ValueError( + f"check_checkpoint.py rejected the input ckpt: {validator_out.get('errors')}" + ) + arch = _arch_from_validator(validator_out) + model_type = validator_out.get("model_type") + + # 3. GPU selection. + num_gpus = max(1, int(getattr(args, "num_gpus", 1) or 1)) + distributed = num_gpus > 1 + if distributed: + # Data-parallel DDP: main.py finetune uses GPUs 0..num_gpus-1 (one rank + # each) via WORLD_SIZE; a single-device pin does not apply. + gpu = None + else: + gpu = resolve_single_gpu(args.gpus, workflow="finetune") + + # 4. Apply defaults + collect args_applied. + applied = _apply_defaults(args, defaults) + # MTL FFN consistency + _validate_mtl_consistency(applied) + + # 5. Build the finetune argv (always main.py finetune; DDP selected via WORLD_SIZE). + argv = _build_argv( + gpu=gpu, out_dir=out_dir, manifest=manifest, ckpt=ckpt, + arch=arch, applied=applied, num_gpus=num_gpus, + ) + + # 6. Build the run.json manifest. + commit, dirty = git_commit_with_env_override(REPO_ROOT) + image_tag = os.environ.get("KERMT_IMAGE", "kermt:latest") + image_digest = docker_image_digest(image_tag) + replay_env = {"WORLD_SIZE": str(num_gpus)} if distributed else {"CUDA_VISIBLE_DEVICES": gpu} + cmd_replay = format_cmd_replay(argv, env=replay_env) + targets = manifest.get("targets") + run_manifest: dict[str, Any] = { + "workflow": "finetune", + "started_at": datetime.datetime.now(datetime.timezone.utc).isoformat(), + "container": {"image_tag": image_tag, "image_digest": image_digest}, + "repo": {"commit": commit, "dirty": dirty}, + "inputs": { + "ckpt": str(ckpt), + "prepare_data_manifest": str(prep_manifest_path), + "ckpt_validator_out": ( + str(Path(args.ckpt_validator_out).resolve()) if args.ckpt_validator_out else None + ), + "targets": list(targets) if targets else None, + }, + "model_type": model_type, + "gpu": gpu, + "num_gpus": num_gpus, + "distributed": distributed, + "args_applied": applied, + "arch": arch, + "save_dir": str(out_dir / "ckpt"), + "logs_dir": str(out_dir / "logs"), + "tensorboard_dir": str(out_dir / "logs" / "tb"), + "argv": argv, + "cmd_replay": cmd_replay, + "ok_to_replay": (not dirty) and (commit != "unknown"), + "dry_run": bool(args.dry_run), + } + (out_dir / "run.json").write_text(json.dumps(run_manifest, indent=2)) + + # 7. Execute (unless --dry-run). + if args.dry_run: + run_manifest["status"] = "dry_run" + return run_manifest + + env = os.environ.copy() + if distributed: + # DDP: main.py finetune reads WORLD_SIZE and spawns one process per GPU. + # Do not pin CUDA_VISIBLE_DEVICES to a single device. + env["WORLD_SIZE"] = str(num_gpus) + else: + env["CUDA_VISIBLE_DEVICES"] = str(gpu) + # main.py enables strict deterministic algorithms via + # `torch.use_deterministic_algorithms(True)`; CuBLAS then requires this env var. + env.setdefault("CUBLAS_WORKSPACE_CONFIG", ":4096:8") + + log_file = out_dir / "logs" / "finetune.log" + with log_file.open("w") as logf: + proc = subprocess.run(argv, env=env, stdout=logf, stderr=subprocess.STDOUT) + run_manifest["exit_code"] = proc.returncode + run_manifest["status"] = "ok" if proc.returncode == 0 else "failed" + (out_dir / "run.json").write_text(json.dumps(run_manifest, indent=2)) + return run_manifest + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser( + description="Workstation finetune runner — wraps main.py finetune via subprocess." + ) + p.add_argument("--ckpt", required=True, + help="Path to the input pretrain checkpoint (grover_base / cmim / hybrid).") + p.add_argument("--prepare-manifest", required=True, + help="Path to a prepare_data.json produced with --mode finetune.") + p.add_argument("--out", required=True, help="Output run directory.") + p.add_argument("--ckpt-validator-out", default=None, + help="Optional cached check_checkpoint.py JSON; computed if absent.") + p.add_argument("--gpus", default=None, + help="Single GPU id for single-process finetune (default 0). " + "Ignored when --num-gpus > 1 (DDP uses ranks 0..N-1).") + p.add_argument("--num-gpus", type=int, default=1, + help="Number of GPUs for data-parallel (DDP) finetune. Default 1 " + "(single-process main.py finetune, unchanged). N>1 runs " + "main.py finetune with WORLD_SIZE=N (one process per GPU); " + "--batch-size is per-GPU.") + p.add_argument("--dry-run", action="store_true", + help="Write run.json + print the command without executing.") + + # Task semantics — dataset_type is required (modify_train_args asserts it). + p.add_argument("--dataset-type", default=None, + choices=["regression", "classification", "multiclass"]) + p.add_argument("--metric", default=None) + p.add_argument("--ensemble-size", type=int, default=None) + p.add_argument("--num-folds", type=int, default=None) + p.add_argument("--show-individual-scores", action="store_true", default=None) + + # Training overrides — defaults None so we can distinguish user vs default-config source. + for f, t in [("epochs", int), ("batch-size", int), ("init-lr", float), ("max-lr", float), + ("final-lr", float), ("warmup-epochs", float), ("weight-decay", float), + ("dropout", float), ("bond-drop-rate", float), ("dist-coff", float), + ("early-stop-epoch", int), ("seed", int)]: + p.add_argument(f"--{f}", type=t, default=None) + + # FFN head overrides + p.add_argument("--ffn-hidden-size", type=int, default=None) + p.add_argument("--ffn-num-layers", type=int, default=None) + p.add_argument("--ffn-num-task-specific-layers", type=int, default=None) + p.add_argument("--ffn-task-specific-hidden-size", type=int, default=None) + + args = p.parse_args(argv) + + try: + manifest = run(args) + except (FileNotFoundError, ValueError, RuntimeError) as exc: + print(json.dumps({"ok": False, "errors": [f"{type(exc).__name__}: {exc}"]}, indent=2)) + return 1 + except Exception as exc: # noqa: BLE001 + import traceback + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({"ok": False, "errors": [f"unhandled: {type(exc).__name__}: {exc}"]}, + indent=2)) + return 1 + + print(json.dumps({"ok": True, "manifest": manifest}, indent=2)) + return 0 if manifest.get("status") != "failed" else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/agent/skills/kermt-finetune/skill-card.md b/skills/kermt-finetune/skill-card.md similarity index 96% rename from agent/skills/kermt-finetune/skill-card.md rename to skills/kermt-finetune/skill-card.md index 26ca064..a29e731 100644 --- a/agent/skills/kermt-finetune/skill-card.md +++ b/skills/kermt-finetune/skill-card.md @@ -20,7 +20,7 @@ Credential Type(s): API key — `WANDB_API_KEY` for optional Weights & Biases ru * Docker, NVIDIA Container Toolkit, CUDA-capable NVIDIA GPU
* A pretrain checkpoint (grover_base, cmim, or hybrid)
* A labeled CSV with SMILES and one or more target columns
-* Hyperparameters default from `agent/config/defaults_finetune.json`, overridable per flag
+* Hyperparameters default from `config/defaults_finetune.json`, overridable per flag
Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
@@ -45,7 +45,7 @@ Mitigation: W&B tracking is optional and off unless the user supplies `WANDB_API ## Reference(s):
- [KERMT repository](https://github.com/NVIDIA-BioNeMo/KERMT)
-- `agent/config/defaults_finetune.json` — default hyperparameters
+- `config/defaults_finetune.json` — default hyperparameters
- Related skills: `kermt-setup`, `kermt-monitor`, `kermt-infer`, `kermt-embed`
## Skill Output:
diff --git a/agent/skills/kermt-infer/SKILL.md b/skills/kermt-infer/SKILL.md similarity index 83% rename from agent/skills/kermt-infer/SKILL.md rename to skills/kermt-infer/SKILL.md index 545adbd..92f3d9d 100644 --- a/agent/skills/kermt-infer/SKILL.md +++ b/skills/kermt-infer/SKILL.md @@ -17,6 +17,14 @@ Run predictions with a finetuned KERMT checkpoint on a SMILES-only CSV. The skill is the workflow orchestrator: validate ckpt, validate CSV, prepare data, launch the runner blocking, return the predictions CSV. +## Skill and runtime paths + +Set `SKILL_DIR` to the directory containing this `SKILL.md`. Export +`KERMT_REPO` as the absolute path to the KERMT checkout used for model +execution. The bundled container helper mounts that checkout at +`/workspace` and this skill at `/skill` (read-only). Commands inside +the container use `/skill/scripts/`; defaults are bundled in `config/`. + ## Hardware requirements - **GPUs**: 1 (single-GPU). Multi-GPU inference is not currently supported. @@ -48,7 +56,7 @@ Let `$KERMT_REPO` be the path to your kermt repo checkout, and assume 1. **Pre-flight: ensure container + system probe.** ``` - $KERMT_REPO/agent/scripts/kermt_container.sh check_system + "$SKILL_DIR/scripts/kermt_container.sh" check_system ``` Refuse to proceed on `ok: false`. @@ -59,23 +67,23 @@ Let `$KERMT_REPO` be the path to your kermt repo checkout, and assume 3. **Validate the checkpoint.** ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run --ckpt -- \ - "python agent/scripts/check_checkpoint.py --mode inference --ckpt /ckpt" + "$SKILL_DIR/scripts/kermt_container.sh" run --ckpt -- \ + "python /skill/scripts/check_checkpoint.py --mode inference --ckpt /ckpt" ``` Parse the JSON. Abort on `ok: false`. The validator rejects pretrain ckpts (`has_task_ffn: false`) with a redirect to `kermt-finetune`. 4. **Validate the data.** ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run --data -- \ - "python agent/scripts/check_data.py --mode inference --csv /data/" + "$SKILL_DIR/scripts/kermt_container.sh" run --data -- \ + "python /skill/scripts/check_data.py --mode inference --csv /data/" ``` Abort on `ok: false`. 5. **Prepare the data.** ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run --data --run-dir $RUN_DIR -- \ - "python agent/scripts/prepare_data.py --mode inference \\ + "$SKILL_DIR/scripts/kermt_container.sh" run --data --run-dir $RUN_DIR -- \ + "python /skill/scripts/prepare_data.py --mode inference \\ --csv /data/ --out /runs/data" ``` Outputs land at `$RUN_DIR/data/prepare_data.json` with `clean_csv` + @@ -83,9 +91,9 @@ Let `$KERMT_REPO` be the path to your kermt repo checkout, and assume 6. **Launch the runner (blocking).** ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run \\ + "$SKILL_DIR/scripts/kermt_container.sh" run \\ --ckpt --run-dir $RUN_DIR -- \\ - "python agent/scripts/run_inference.py \\ + "python /skill/scripts/run_inference.py \\ --ckpt /ckpt \\ --prepare-manifest /runs/data/prepare_data.json \\ --out /runs \\ diff --git a/skills/kermt-infer/config/defaults_inference.json b/skills/kermt-infer/config/defaults_inference.json new file mode 100644 index 0000000..bf5e80a --- /dev/null +++ b/skills/kermt-infer/config/defaults_inference.json @@ -0,0 +1,11 @@ +{ + "_about": "Default settings applied by kermt-infer. Inference is a stateless forward pass — no training schedule, no LR, no FFN sizing. The skill echoes the applied set back to the user on every invocation; override any value with the corresponding CLI flag.", + + "runtime": { + "_about": "Runtime knobs.", + "batch_size": 32, + "seed": 0 + }, + + "_about_gpu_selection": "GPU selection is auto-detected at runtime, not a default here. Inference defaults to GPU 0; override with --gpus 0 (the single id you want)." +} diff --git a/agent/skills/kermt-infer/evals/evals.json b/skills/kermt-infer/evals/evals.json similarity index 100% rename from agent/skills/kermt-infer/evals/evals.json rename to skills/kermt-infer/evals/evals.json diff --git a/skills/kermt-infer/scripts/_utils.py b/skills/kermt-infer/scripts/_utils.py new file mode 100644 index 0000000..5bde460 --- /dev/null +++ b/skills/kermt-infer/scripts/_utils.py @@ -0,0 +1,272 @@ +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Shared utilities for the agent scripts. + +Kept intentionally small — only logic that appears (or would otherwise be +duplicated) in two or more `scripts/*.py` modules. Each script +maintains its own primary CLI + main flow. +""" +from __future__ import annotations + +import argparse +import json +import os +import shlex +import subprocess +import sys +from pathlib import Path +from typing import Any + + +# Conventional pretrain vocab filename stems. Used by prepare_data.py + +# upgrade_to_hybrid.py + the README "Released models" bundling docs + +# the test helpers. Centralized here so a future rename only touches one +# spot. +PRETRAIN_VOCAB_STEMS = { + "atom": "pretrain_atom_vocab", + "bond": "pretrain_bond_vocab", + "smiles": "pretrain_smiles_vocab", +} + + +def resolve_kermt_repo() -> Path: + """Find the runtime checkout independently of the installed skill location. + + An explicit KERMT_REPO takes precedence. In a repository checkout, walking + up from this helper or the working directory also supports local use. + """ + explicit = os.environ.get("KERMT_REPO") + if explicit: + candidates = [Path(explicit).expanduser().resolve()] + else: + candidates = [] + for start in (Path(__file__).resolve().parent, Path.cwd()): + candidates.extend((start, *start.parents)) + for candidate in candidates: + if (candidate / "main.py").is_file() and (candidate / "kermt").is_dir(): + return candidate + raise FileNotFoundError( + "KERMT checkout not found. Set KERMT_REPO to the checkout containing " + "main.py and kermt/; the installed skill directory is separate." + ) + + +def load_json(path: Path, *, name: str) -> dict[str, Any]: + """Load a JSON file with consistent error messages. + + `name` is a human-readable label for the document (e.g. "prepare_data.json") + so the error tells the user which schema we expected at that path. + """ + if not path.is_file(): + raise FileNotFoundError(f"{name} not found at {path}") + try: + return json.loads(path.read_text()) + except json.JSONDecodeError as exc: + raise ValueError(f"{name} at {path} is not valid JSON: {exc}") from exc + + +def count_vocab_entries(vocab_path: Path) -> int: + """Return the number of entries in a KERMT vocab file. + + Handles three layouts: + - JSON with `{stoi: {token: idx}, ...}` (MolVocab.save_vocab default) + - JSON as a raw `{token: idx}` dict (legacy / hand-edited) + - Pickle of a MolVocab / SMILESVocab object (the smiles vocab is always + pickled because its compiled-regex tokenizer state isn't + JSON-serializable). Falls through to raw `pickle.load` if the + MolVocab / SMILESVocab loader can't import or fails to recognize + the contents (e.g. test fixtures with plain dicts). + """ + if vocab_path.suffix == ".json": + data = json.loads(vocab_path.read_text()) + if isinstance(data, dict) and "stoi" in data: + return len(data["stoi"]) + if isinstance(data, dict): + return len(data) + raise ValueError(f"unsupported JSON vocab shape at {vocab_path}: {type(data).__name__}") + + # .pkl: try MolVocab / SMILESVocab first, then fall back to raw pickle. + try: + from kermt.data.torchvocab import MolVocab, SMILESVocab # type: ignore + for loader in (MolVocab.load_vocab, SMILESVocab.load_vocab): + try: + v = loader(str(vocab_path)) + return len(v) + except Exception: + continue + except ImportError: + pass + + import pickle + with vocab_path.open("rb") as f: + data = pickle.load(f) + if hasattr(data, "stoi"): + return len(data.stoi) + if hasattr(data, "__len__"): + return len(data) + raise ValueError(f"could not count entries in {vocab_path}") + + +def validate_vocab_file(vocab_path: Path, *, kind: str) -> None: + """Verify a user-provided vocab file is loadable BEFORE copying it into a + run directory. Raises ValueError on failure with a clear, user-facing message. + + `kind` is one of {"atom", "bond", "smiles"} — used only in the error message + so the user knows which file is wrong. + """ + if not vocab_path.is_file(): + raise FileNotFoundError(f"{kind} vocab file not found: {vocab_path}") + try: + n = count_vocab_entries(vocab_path) + except Exception as exc: # noqa: BLE001 + raise ValueError( + f"{kind} vocab file {vocab_path} is not loadable as a KERMT vocab " + f"({type(exc).__name__}: {exc}). Expected a MolVocab JSON or pickle " + f"(or a SMILESVocab pickle for the smiles vocab)." + ) from exc + if n <= 0: + raise ValueError(f"{kind} vocab file {vocab_path} contains zero entries") + + +# --------------------------------------------------------------------------- +# Runner-shared helpers (run.json manifest fields) +# --------------------------------------------------------------------------- + +def git_commit_with_env_override(repo: Path) -> tuple[str, bool]: + """Returns (commit_sha, dirty_tree). Honors `KERMT_REPO_COMMIT` / + `KERMT_REPO_DIRTY` env vars first — set by `scripts/kermt_container.sh` + from the host before launching docker (necessary because `git -C /workspace` + inside the container fails due to bind-mount ownership). Falls back to the + in-container git probe when the env vars aren't set.""" + env_commit = os.environ.get("KERMT_REPO_COMMIT") + if env_commit: + env_dirty = os.environ.get("KERMT_REPO_DIRTY", "false").strip().lower() == "true" + return env_commit, env_dirty + try: + sha = subprocess.run( + ["git", "-C", str(repo), "rev-parse", "HEAD"], + capture_output=True, text=True, check=True, + ).stdout.strip() + diff = subprocess.run( + ["git", "-C", str(repo), "status", "--porcelain"], + capture_output=True, text=True, check=True, + ) + return sha, bool(diff.stdout.strip()) + except Exception: + return "unknown", False + + +def docker_image_digest(tag: str) -> str | None: + """Return the docker image's content-addressable Id (sha256:…) for the given + tag, or None if docker isn't available / the image isn't local.""" + try: + r = subprocess.run( + ["docker", "image", "inspect", tag, "--format", "{{.Id}}"], + capture_output=True, text=True, + ) + if r.returncode == 0: + return r.stdout.strip() + except FileNotFoundError: + pass + return None + + +def format_cmd_replay(argv: list[str], *, env: dict[str, str] | None = None) -> str: + """Render a copy-pasteable env-prefix + command for the cmd_replay manifest + field. `env` is the set of environment variables to prefix (typically + {CUDA_VISIBLE_DEVICES, WORLD_SIZE}).""" + env = env or {} + env_prefix = [f"{k}={shlex.quote(str(v))}" for k, v in env.items()] + quoted = " ".join(shlex.quote(a) for a in argv) + return " ".join(env_prefix + [quoted]) + + +def resolve_single_gpu(override: str | None, *, workflow: str) -> int: + """Returns a single GPU id (int). The finetune/inference/embed workflows are + single-GPU only; `--gpus '0,1'` or multi-id CUDA_VISIBLE_DEVICES is rejected + with a workflow-specific error. (The pretrain runner has its own multi-GPU + `_detect_gpus` helper — see run_pretrain_local.py.)""" + if override is None: + env_visible = os.environ.get("CUDA_VISIBLE_DEVICES", "").strip() + if env_visible: + ids = [g for g in env_visible.split(",") if g] + if len(ids) > 1: + raise ValueError( + f"CUDA_VISIBLE_DEVICES='{env_visible}' selects multiple GPUs but " + f"the {workflow} workflow is single-GPU only. Restrict to one id." + ) + return int(ids[0]) + return 0 + parts = [p.strip() for p in override.split(",") if p.strip()] + if len(parts) != 1: + raise ValueError( + f"--gpus '{override}' selects {len(parts)} GPUs; the {workflow} workflow is single-GPU only." + ) + return int(parts[0]) + + +def assert_prepare_manifest_basics(manifest: dict[str, Any], expected_mode: str) -> None: + """Standard pre-check for a prepare_data.json before a runner consumes it: + verify `mode` matches and `ok` is True. Raises ValueError with a consistent + error message on either mismatch. + + Each runner is responsible for its own required-outputs check after this + (those vary per-mode — e.g. pretrain wants train_dir/val_dir/atom_vocab/ + bond_vocab; finetune has the split-method branch; inference/embed want + clean_csv).""" + if manifest.get("mode") != expected_mode: + raise ValueError( + f"prepare_data manifest is mode='{manifest.get('mode')}', expected '{expected_mode}'. " + f"Run `prepare_data.py --mode {expected_mode}` to produce a valid manifest." + ) + if not manifest.get("ok"): + raise ValueError( + f"prepare_data manifest reports ok=False: {manifest.get('errors')}" + ) + + +def merge_default_into_applied( + applied: dict[str, dict[str, Any]], + args: argparse.Namespace, + name: str, + defaults_group: dict[str, Any], +) -> None: + """Standard CLI-override / default-config merge for one hyperparameter. + + Mutates `applied` in place: + - If the user passed `--` on the CLI (so `getattr(args, name)` is + not None), records `{"value": cli_val, "source": "user"}`. + - Else if `name` is present in `defaults_group`, records + `{"value": defaults_group[name], "source": "default-config"}`. + - Else `applied[name]` is left absent — the runner's argv-builder skips + the flag, and the downstream argparse default takes effect. + + `name` is the snake_case argparse dest (same form used as the dict key); + argparse automatically converts CLI `--` to that dest, + so `getattr(args, name, None)` is the correct CLI lookup.""" + cli_val = getattr(args, name, None) + if cli_val is not None: + applied[name] = {"value": cli_val, "source": "user"} + elif name in defaults_group: + applied[name] = {"value": defaults_group[name], "source": "default-config"} + + +def run_checkpoint_validator(ckpt: Path, *, mode: str, script_path: Path) -> dict[str, Any]: + """Invoke `check_checkpoint.py --mode --ckpt ` as a subprocess + and return the parsed JSON. Raises RuntimeError on non-JSON output (e.g. the + validator crashed before printing). `script_path` is the absolute path to + `scripts/check_checkpoint.py` — passed in so this helper has no + dependency on the caller's layout.""" + r = subprocess.run( + [sys.executable, str(script_path), "--mode", mode, "--ckpt", str(ckpt)], + capture_output=True, text=True, + ) + try: + return json.loads(r.stdout) + except json.JSONDecodeError as exc: + raise RuntimeError( + f"check_checkpoint.py emitted non-JSON output (exit {r.returncode}). " + f"stdout (first 200 chars): {r.stdout[:200]}\n" + f"stderr (first 200 chars): {r.stderr[:200]}" + ) from exc diff --git a/skills/kermt-infer/scripts/check_checkpoint.py b/skills/kermt-infer/scripts/check_checkpoint.py new file mode 100644 index 0000000..fd488e2 --- /dev/null +++ b/skills/kermt-infer/scripts/check_checkpoint.py @@ -0,0 +1,480 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Validate a KERMT checkpoint for a given agent workflow. + +Mode-dispatched. Emits a single JSON object to stdout that the calling skill +parses to decide whether to proceed (`ok: true`) or surface a structured error +back to the user (`ok: false` with `errors[]`). The JSON shape is stable +across modes; only the contract for what counts as `ok` differs per mode. + +Modes +----- +continue_pretrain Continuing pretraining from an existing pretrain ckpt. + Requires encoder + at least one pretrain head + (vocab_head for grover_base / cmim, or contrast_head for + cmim / hybrid). Rejects encoder-only or finetuned ckpts. + +upgrade_to_hybrid Adding a cMIM decoder onto a grover_base ckpt to convert + it to a hybrid pretrain. Requires encoder; rejects ckpts + that already carry a contrast_head or task_ffn (would be + workflow 4 instead). + +finetune_init Starting a finetune from a pretrained ckpt. Requires + encoder. Pretrain heads (vocab / contrast) are tolerated + but unused. Already-finetuned ckpts (task FFN heads + present) are REJECTED — finetune-on-finetune via the + agent skill isn't supported because saved-task + identity can't be machine-verified against the new + training data. + +inference Running predictions with a previously-finetuned ckpt. + Requires encoder + task_ffn. Reports task_output_dims + so the runner can compare against the user's task spec. + +embed Extracting embeddings. Requires encoder only. Anything + additional in the ckpt is ignored. + +Output (stdout) +--------------- +{ + "ok": true | false, + "model_type": "grover_base" | "cmim" | "hybrid" | "finetuned" | "unknown", + "has_encoder": bool, + "has_vocab_head": bool, + "has_contrast_head": bool, + "has_task_ffn": bool, + "task_output_dims": [int, ...], // empty unless has_task_ffn + "arch": { // ckpt-derived; runner uses these, ignores defaults_*.json arch + "hidden_size": int | null, + "depth": int | null, + "num_attn_head": int | null, + "latent_dim": int | null, + "activation": str | null, + "backbone": str | null, + "embedding_output_type": str | null, + "self_attention": bool | null + }, + "saved_args": { ... } | null, // raw args dict if present, else null + "errors": [str, ...], // mode-contract violations / load failures + "warnings": [str, ...] // non-fatal observations (e.g. arch fallback) +} + +Exit code: 0 on `ok: true`, 1 on `ok: false`. Loader exceptions are caught and +surfaced into `errors[]` with `ok: false` (still exit 1), never raised. + +CLI +--- + check_checkpoint.py --mode --ckpt +""" +from __future__ import annotations + +import argparse +import json +import sys +import traceback +from argparse import Namespace +from typing import Any + +import torch + + +# --------------------------------------------------------------------------- +# State-dict key prefix conventions (kermt/model/models.py). +# --------------------------------------------------------------------------- + +# Encoder weights appear under one of these prefixes depending on the ckpt's +# era and task class: +# - `grover.*` : legacy grover_base ckpts (predate the cMIM rename) +# - `kermt.*` : current grover_base / hybrid / finetune ckpts +# - `latent_dist.kermt.*`: cmim ckpts (encoder lives only inside latent_dist) +ENCODER_PREFIXES = ("kermt.", "grover.", "latent_dist.kermt.") +VOCAB_HEAD_PREFIX = "vocab_module." +CONTRAST_DECODER_PREFIX = "decoder." # SMILES transformer decoder, cmim/hybrid only +LATENT_DIST_PREFIX = "latent_dist." # cmim/hybrid; encoder may share via latent_dist.kermt.* +TASK_FFN_PREFIXES = ( + "mol_atom_from_atom_ffn.", + "mol_atom_from_bond_ffn.", +) +TASK_FFN_TASK_SPECIFIC_PREFIXES = ( + "mol_atom_from_atom_ffn_task_specific.", + "mol_atom_from_bond_ffn_task_specific.", +) + + +ARCH_KEYS = ( + "hidden_size", + "depth", + "num_attn_head", + "latent_dim", + "activation", + "backbone", + "embedding_output_type", + "self_attention", +) + + +def _strip_ddp_prefix(state_dict: dict[str, Any]) -> dict[str, Any]: + """Strip `module.` prefix from every key if the dict is DDP-wrapped.""" + if state_dict and all(k.startswith("module.") for k in state_dict): + return {k[len("module."):]: v for k, v in state_dict.items()} + return state_dict + + +def _classify_model(state_dict: dict[str, Any]) -> dict[str, Any]: + keys = list(state_dict.keys()) + has_encoder = any(k.startswith(ENCODER_PREFIXES) for k in keys) + has_vocab_head = any(k.startswith(VOCAB_HEAD_PREFIX) for k in keys) + has_contrast_head = any(k.startswith(CONTRAST_DECODER_PREFIX) for k in keys) + has_task_ffn = any(k.startswith(TASK_FFN_PREFIXES) for k in keys) + + if has_encoder and has_task_ffn: + model_type = "finetuned" + elif has_encoder and has_contrast_head and has_vocab_head: + model_type = "hybrid" + elif has_encoder and has_contrast_head and not has_vocab_head: + model_type = "cmim" + elif has_encoder and not has_contrast_head: + # Includes: + # - modern repo-trained Grover base (kermt.* + vocab_module.*) + # - legacy original-Grover base (grover.encoders.* with no heads saved) + # - any encoder-stripped ckpt extracted from a larger model + # The `has_vocab_head` flag discriminates the sub-cases for skills that + # need it. The continue_pretrain mode contract relies on this — a + # grover_base with vocab heads can continue, an encoder-only one cannot. + model_type = "grover_base" + else: + model_type = "unknown" + + return { + "model_type": model_type, + "has_encoder": has_encoder, + "has_vocab_head": has_vocab_head, + "has_contrast_head": has_contrast_head, + "has_task_ffn": has_task_ffn, + } + + +def _vocab_sizes(state_dict: dict[str, Any]) -> dict[str, Any]: + """Extract vocab head sizes from state-dict weight shapes. + + The pretrain heads have the following layout per kermt/model/models.py: + - Atom vocab predictors: vocab_module.av_task_atom.* + vocab_module.av_task_bond.* + (two readout streams sharing the same vocab_size). Output dim of each + final-Linear is the atom vocab size. + - Bond vocab predictors: vocab_module.bv_task_atom.* + vocab_module.bv_task_bond.* + Output dim is the bond vocab size. + - SMILES vocab decoder: decoder.output_projection.weight (cmim / hybrid only). + Output dim is the smiles vocab size. + + Returns {atom: int|None, bond: int|None, smiles: int|None}. Each is None + when the corresponding head isn't present in the ckpt (e.g. legacy + encoder-only grover_base has none; cmim has smiles but not atom/bond). + """ + sizes: dict[str, Any] = {"atom": None, "bond": None, "smiles": None} + + def _head_out_dim(prefix: str) -> int | None: + # Pick the highest-numbered 2-D Linear weight under `prefix.*` — that's + # the final output layer. + candidates = [ + k for k in state_dict + if k.startswith(prefix) and k.endswith(".weight") + and hasattr(state_dict[k], "ndim") and state_dict[k].ndim == 2 + ] + if not candidates: + return None + def _layer_index(k: str) -> int: + # ".weight" -> "..weight"; pick the rightmost numeric component. + parts = k.split(".") + for tok in reversed(parts[:-1]): + if tok.isdigit(): + return int(tok) + return -1 + final = max(candidates, key=_layer_index) + return int(state_dict[final].shape[0]) + + sizes["atom"] = _head_out_dim("vocab_module.av_task_atom.") + sizes["bond"] = _head_out_dim("vocab_module.bv_task_atom.") + sizes["smiles"] = _head_out_dim("decoder.output_projection.") + # If the decoder's output_projection isn't a Linear (e.g. some saves wrap + # it differently), fall back to a search over decoder.* heads. + if sizes["smiles"] is None: + sizes["smiles"] = _head_out_dim("decoder.token_embedding.") + return sizes + + +def _task_output_dims(state_dict: dict[str, Any]) -> list[int]: + """Return one entry per (logical task × readout) head's final-Linear out-dim. + + Two layouts: + - **MTL** (`mol_atom_from_atom_ffn_task_specific..*`): one entry per + task-specific head's final-Linear out-dim. Typically `[1, 1, ..., 1]` + for regression with N tasks across 2 readouts. + - **Non-MTL** (`mol_atom_from_atom_ffn.*` only): one entry per shared FFN's + final-Linear out-dim. Typically `[num_tasks, num_tasks]` (one per readout). + + When both layouts coexist in the same ckpt (MTL configuration: shared FFN + feeds task-specific heads), only the task-specific dims are reported — the + shared FFN there is an intermediate layer, not the model output. + """ + has_task_specific = any(k.startswith(TASK_FFN_TASK_SPECIFIC_PREFIXES) for k in state_dict) + + heads: dict[str, list[str]] = {} + for k in state_dict: + if k.startswith(TASK_FFN_TASK_SPECIFIC_PREFIXES): + parts = k.split(".") + root = ".".join(parts[:2]) # e.g. "mol_atom_from_atom_ffn_task_specific.0" + heads.setdefault(root, []).append(k) + elif k.startswith(TASK_FFN_PREFIXES) and not k.startswith(TASK_FFN_TASK_SPECIFIC_PREFIXES): + if has_task_specific: + continue # shared FFN is intermediate when task-specific heads exist + root = k.split(".")[0] # e.g. "mol_atom_from_atom_ffn" + heads.setdefault(root, []).append(k) + + dims: list[int] = [] + for root in sorted(heads): + weight_keys = sorted( + (k for k in heads[root] if k.endswith(".weight") + and hasattr(state_dict[k], "ndim") and state_dict[k].ndim == 2), + key=lambda k: int(k.split(".")[-2]) if k.split(".")[-2].isdigit() else -1, + ) + if weight_keys: + dims.append(int(state_dict[weight_keys[-1]].shape[0])) + return dims + + +def _arch_from_args(args_obj: Any) -> dict[str, Any]: + """Pull arch params from the saved args Namespace / dict, leaving missing keys as None.""" + arch: dict[str, Any] = {k: None for k in ARCH_KEYS} + if args_obj is None: + return arch + # args_obj is typically argparse.Namespace; tolerate dict form too. + args_dict = vars(args_obj) if isinstance(args_obj, Namespace) else dict(args_obj) if isinstance(args_obj, dict) else {} + for k in ARCH_KEYS: + if k in args_dict: + arch[k] = args_dict[k] + return arch + + +def _arch_from_shapes(state_dict: dict[str, Any], arch: dict[str, Any]) -> tuple[dict[str, Any], list[str]]: + """Fill in still-missing arch params by introspecting state-dict tensor shapes. + + Only fills entries that are currently None — does not override anything pulled + from saved_args. Returns the updated arch + a list of warnings for any key that + could not be inferred. + """ + warnings: list[str] = [] + + if arch["hidden_size"] is None: + # First 2-D linear weight under any encoder prefix. + candidates = [ + k for k in state_dict + if k.startswith(ENCODER_PREFIXES) + and k.endswith(".weight") + and hasattr(state_dict[k], "ndim") + and state_dict[k].ndim == 2 + ] + if candidates: + arch["hidden_size"] = int(state_dict[candidates[0]].shape[0]) + else: + warnings.append("hidden_size could not be inferred from state_dict shapes") + + if arch["latent_dim"] is None: + # Look for a Linear inside latent_dist that's not the shared encoder. + candidates = [ + k for k in state_dict + if k.startswith(LATENT_DIST_PREFIX) + and not k.startswith("latent_dist.kermt.") + and k.endswith(".weight") + and state_dict[k].ndim == 2 + ] + if candidates: + arch["latent_dim"] = int(state_dict[candidates[0]].shape[0]) + # Absent latent_dist entirely is a fact about the model, not a problem — no warning. + + # depth, num_attn_head, activation, backbone, embedding_output_type, self_attention + # are not robustly inferable from shapes alone; report a warning for each that's + # still None so the caller can prompt the user or refuse to proceed. + for k in ("depth", "num_attn_head", "activation", "backbone", "embedding_output_type", "self_attention"): + if arch[k] is None: + warnings.append(f"{k} not present in saved_args and cannot be inferred from state_dict shapes") + + return arch, warnings + + +def _apply_mode_contract(mode: str, classification: dict[str, Any]) -> list[str]: + """Return a list of error messages if `classification` violates the mode contract.""" + errors: list[str] = [] + mt = classification["model_type"] + has_enc = classification["has_encoder"] + has_vocab = classification["has_vocab_head"] + has_contrast = classification["has_contrast_head"] + has_ffn = classification["has_task_ffn"] + + if not has_enc: + errors.append("checkpoint has no encoder weights — cannot use it for any KERMT workflow") + return errors + + if mode == "continue_pretrain": + if not (has_vocab or has_contrast): + errors.append( + f"continue_pretrain requires the ckpt to still carry pretrain heads (vocab " + f"and/or contrast), but this ckpt has neither (model_type='{mt}', " + f"has_vocab_head=False, has_contrast_head=False). Either provide a ckpt with " + f"its pretrain heads attached, or convert this encoder-only ckpt to a hybrid " + f"via mode 'upgrade_to_hybrid'." + ) + if has_ffn: + errors.append( + "continue_pretrain expects a pretrain ckpt; this ckpt has task FFN heads " + "(it has been finetuned). Use a pretrain checkpoint — finetune+continue is " + "not a supported workflow." + ) + elif mode == "upgrade_to_hybrid": + if has_contrast: + errors.append( + f"upgrade_to_hybrid converts grover_base -> hybrid by adding a cMIM decoder. " + f"This ckpt already has a contrast head (classified as '{mt}'). " + f"To continue pretraining it, use mode 'continue_pretrain'." + ) + if has_ffn: + errors.append("upgrade_to_hybrid does not support finetuned checkpoints.") + elif mode == "finetune_init": + # Requires an encoder. Pretrain heads (vocab / contrast) are unused + # at finetune time but harmless. Task FFN heads (i.e. an already- + # finetuned ckpt) are NOT accepted — finetune-on-finetune isn't + # supported by the kermt-finetune skill because the saved-task + # identity can't be machine-verified against the new training data + # (dimension match doesn't prove target identity, dataset identity, + # or absence of train/test contamination). + if has_ffn: + errors.append( + f"finetune_init requires a pretrain ckpt (grover_base / cmim / hybrid); " + f"this ckpt is classified as '{mt}' with task FFN heads attached. " + f"To resume a finetune on the SAME dataset, call " + f"`python main.py finetune --checkpoint_path ...` directly — the " + f"kermt-finetune skill doesn't support resume." + ) + elif mode == "inference": + if not has_ffn: + errors.append( + "inference requires a finetuned ckpt with task FFN heads. " + f"This ckpt is classified as '{mt}' with no task heads. " + "Run finetune (mode 'finetune_init') first." + ) + elif mode == "embed": + # Encoder is sufficient. + pass + else: + errors.append(f"unknown mode '{mode}'") + + return errors + + +def validate(mode: str, ckpt_path: str) -> dict[str, Any]: + result: dict[str, Any] = { + "ok": False, + "model_type": "unknown", + "has_encoder": False, + "has_vocab_head": False, + "has_contrast_head": False, + "has_task_ffn": False, + "task_output_dims": [], + "vocab_sizes": {"atom": None, "bond": None, "smiles": None}, + "arch": {k: None for k in ARCH_KEYS}, + "saved_args": None, + "errors": [], + "warnings": [], + } + + # 1. Load the checkpoint. + try: + ckpt = torch.load(ckpt_path, map_location="cpu", weights_only=False) + except FileNotFoundError: + result["errors"].append(f"checkpoint not found: {ckpt_path}") + return result + except Exception as exc: # noqa: BLE001 + result["errors"].append(f"failed to load checkpoint {ckpt_path}: {type(exc).__name__}: {exc}") + return result + + if not isinstance(ckpt, dict) or "state_dict" not in ckpt: + result["errors"].append( + "checkpoint is not in the expected save_model_for_restart format " + "(expected a dict with a 'state_dict' key)." + ) + return result + + state_dict = _strip_ddp_prefix(ckpt["state_dict"]) + args_obj = ckpt.get("args") + + # 2. Classify and check mode contract. + classification = _classify_model(state_dict) + result.update(classification) + + contract_errors = _apply_mode_contract(mode, classification) + result["errors"].extend(contract_errors) + + # 3. Task output dims (for inference / informational). + if classification["has_task_ffn"]: + result["task_output_dims"] = _task_output_dims(state_dict) + + # 3b. Vocab head sizes (for continue-pretrain vocab-size verification). + result["vocab_sizes"] = _vocab_sizes(state_dict) + + # 4. Arch derivation: args first, shape introspection for what's still missing. + arch = _arch_from_args(args_obj) + arch, shape_warnings = _arch_from_shapes(state_dict, arch) + result["arch"] = arch + result["warnings"].extend(shape_warnings) + + # 5. Saved args as serializable dict (best-effort). + if args_obj is not None: + try: + result["saved_args"] = vars(args_obj) if isinstance(args_obj, Namespace) else dict(args_obj) + # Drop non-JSON-serializable values; agent skill only needs human-readable scalars. + result["saved_args"] = { + k: v for k, v in result["saved_args"].items() + if isinstance(v, (str, int, float, bool, type(None), list, dict)) + } + except Exception as exc: # noqa: BLE001 + result["warnings"].append(f"could not serialize saved_args: {type(exc).__name__}: {exc}") + + result["ok"] = not result["errors"] + return result + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Validate a KERMT checkpoint for a given workflow.") + parser.add_argument("--mode", required=True, + choices=["continue_pretrain", "upgrade_to_hybrid", "finetune_init", "inference", "embed"]) + parser.add_argument("--ckpt", required=True, help="Path to the .pt checkpoint") + args = parser.parse_args(argv) + + try: + result = validate(args.mode, args.ckpt) + except Exception as exc: # noqa: BLE001 + # Last-resort safety net: keep stdout JSON-clean, dump trace to stderr. + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({ + "ok": False, + "model_type": "unknown", + "errors": [f"unhandled exception in validator: {type(exc).__name__}: {exc}"], + "warnings": [], + "arch": {k: None for k in ARCH_KEYS}, + "has_encoder": False, + "has_vocab_head": False, + "has_contrast_head": False, + "has_task_ffn": False, + "task_output_dims": [], + "vocab_sizes": {"atom": None, "bond": None, "smiles": None}, + "saved_args": None, + }, indent=2)) + return 1 + + print(json.dumps(result, indent=2)) + return 0 if result["ok"] else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-infer/scripts/check_data.py b/skills/kermt-infer/scripts/check_data.py new file mode 100644 index 0000000..b8f9b15 --- /dev/null +++ b/skills/kermt-infer/scripts/check_data.py @@ -0,0 +1,310 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Validate a CSV input for a given KERMT agent workflow. + +Mode-dispatched. Emits a single JSON object to stdout that the calling skill +parses to decide whether to proceed (`ok: true`) or surface a structured error +back to the user (`ok: false` with `errors[]`). The JSON shape is stable +across modes; only the contract for what counts as `ok` differs per mode. + +Modes +----- +pretrain Pretrain corpus CSV. Requires a `smiles` column. Other columns + are ignored. Label columns are not required (and not expected). + +finetune Labeled CSV for a downstream task. Requires `smiles` plus + >=1 numeric target column. Target columns are specified via + `--targets ...`. If `--targets` is omitted, the + validator auto-detects numeric non-smiles columns and reports + them; the skill will then prompt the user to confirm or refine. + +inference CSV to run predictions on. Requires `smiles`. Target columns are + not required (and not expected — predictions are written out). + +embed CSV to extract embeddings from. Requires `smiles` only. + +SMILES validation +----------------- +By default the validator samples up to 20 SMILES (first 10 + last 10) and +checks each one parses with RDKit. Pass `--strict-rdkit` to parse every +SMILES (slow on large corpora). A SMILES is considered "invalid" if RDKit +returns `None` from `MolFromSmiles(smi, sanitize=True)` — empty / null +rows are counted separately. + +Duplicate-SMILES detection is always full (cheap). + +Output (stdout) +--------------- +{ + "ok": true | false, + "mode": str, + "csv_path": str, + "num_rows": int, + "num_columns": int, + "columns": [str, ...], + "has_smiles_column": bool, + "smiles_column_name": str | null, // actual header used (may differ in case) + "num_blank_smiles": int, + "num_invalid_smiles": int, // among the parsed sample + "smiles_check_method": "sampled" | "full", + "smiles_check_count": int, + "num_duplicate_smiles": int, + "target_columns": [str, ...], // populated only for finetune mode + "num_missing_per_target": { col: int, ... }, + "auto_detected_targets": [str, ...], // when --targets is omitted in finetune mode + "errors": [str, ...], + "warnings": [str, ...] +} + +Exit code: 0 on `ok: true`, 1 on `ok: false`. Loader exceptions are caught +and surfaced into `errors[]` with `ok: false` (still exit 1). + +CLI +--- + check_data.py --mode --csv + [--targets ...] # finetune only + [--strict-rdkit] # full SMILES parse +""" +from __future__ import annotations + +import argparse +import json +import sys +import traceback +from pathlib import Path +from typing import Any + +import pandas as pd + + +CANONICAL_SMILES_COLUMN = "smiles" +SMILES_SAMPLE_PER_END = 10 # how many SMILES from head + how many from tail to sample + + +def _find_smiles_column(columns: list[str]) -> str | None: + """Return the actual column header matching 'smiles' case-insensitively, or None.""" + for c in columns: + if c.lower() == CANONICAL_SMILES_COLUMN: + return c + return None + + +def _parse_smiles_sample(smiles_values: list[str], full: bool) -> tuple[int, int, str]: + """Run RDKit MolFromSmiles on a sample or all of the SMILES. Returns + (num_parsed, num_invalid, method).""" + # Import here so the script can still surface a clean JSON error if RDKit + # is unavailable in the host env. + try: + from rdkit import Chem + from rdkit import RDLogger + RDLogger.DisableLog("rdApp.*") # suppress per-mol parse warnings + except ImportError as exc: + raise RuntimeError( + f"RDKit is not importable in this environment: {exc}. " + "Run check_data.py inside the kermt container." + ) from exc + + if full or len(smiles_values) <= 2 * SMILES_SAMPLE_PER_END: + sample = smiles_values + method = "full" + else: + sample = smiles_values[:SMILES_SAMPLE_PER_END] + smiles_values[-SMILES_SAMPLE_PER_END:] + method = "sampled" + + invalid = 0 + parsed = 0 + for smi in sample: + if not smi: # already counted as blank elsewhere + continue + parsed += 1 + mol = Chem.MolFromSmiles(smi, sanitize=True) + if mol is None: + invalid += 1 + return parsed, invalid, method + + +def _autodetect_target_columns(df: pd.DataFrame, smiles_col: str) -> list[str]: + """Pick columns that look like numeric targets. A column qualifies if it + is (a) not the smiles column and (b) >=80% of non-null values convert to float. + Heuristic only — returned for the skill to prompt the user to confirm.""" + candidates: list[str] = [] + for col in df.columns: + if col == smiles_col: + continue + ser = df[col].dropna() + if len(ser) == 0: + continue + try: + converted = pd.to_numeric(ser, errors="coerce") + except (TypeError, ValueError): + continue + if converted.notna().sum() / max(len(ser), 1) >= 0.8: + candidates.append(col) + return candidates + + +def validate(mode: str, csv_path: str, targets: list[str] | None, strict_rdkit: bool) -> dict[str, Any]: + result: dict[str, Any] = { + "ok": False, + "mode": mode, + "csv_path": csv_path, + "num_rows": 0, + "num_columns": 0, + "columns": [], + "has_smiles_column": False, + "smiles_column_name": None, + "num_blank_smiles": 0, + "num_invalid_smiles": 0, + "smiles_check_method": "sampled", + "smiles_check_count": 0, + "num_duplicate_smiles": 0, + "target_columns": [], + "num_missing_per_target": {}, + "auto_detected_targets": [], + "errors": [], + "warnings": [], + } + + # 1. Read the CSV. + path = Path(csv_path) + if not path.is_file(): + result["errors"].append(f"CSV not found: {csv_path}") + return result + try: + df = pd.read_csv(path) + except pd.errors.EmptyDataError: + result["errors"].append(f"CSV is empty (no header): {csv_path}") + return result + except Exception as exc: # noqa: BLE001 + result["errors"].append(f"failed to read CSV {csv_path}: {type(exc).__name__}: {exc}") + return result + + result["num_rows"] = int(len(df)) + result["num_columns"] = int(len(df.columns)) + result["columns"] = [str(c) for c in df.columns] + + # 2. Locate the SMILES column. + smiles_col = _find_smiles_column(result["columns"]) + if smiles_col is None: + result["errors"].append( + f"no column named 'smiles' (case-insensitive) found in CSV. " + f"Available columns: {result['columns']}" + ) + return result + result["has_smiles_column"] = True + result["smiles_column_name"] = smiles_col + if smiles_col != CANONICAL_SMILES_COLUMN: + result["warnings"].append( + f"SMILES column is named '{smiles_col}' but downstream code expects '{CANONICAL_SMILES_COLUMN}' " + f"(lowercase). Rename the column to '{CANONICAL_SMILES_COLUMN}' before running the workflow." + ) + + # 3. Blank-SMILES count + duplicate count + RDKit parse check. + smi_series = df[smiles_col].astype(str).fillna("").str.strip() + blank_mask = smi_series.eq("") | smi_series.str.lower().eq("nan") + result["num_blank_smiles"] = int(blank_mask.sum()) + + nonblank = smi_series[~blank_mask] + result["num_duplicate_smiles"] = int(len(nonblank) - nonblank.nunique()) + + if len(nonblank) == 0: + result["errors"].append("no non-blank SMILES found in the CSV") + return result + + try: + parsed, invalid, method = _parse_smiles_sample(nonblank.tolist(), full=strict_rdkit) + except RuntimeError as exc: + result["errors"].append(str(exc)) + return result + result["smiles_check_count"] = parsed + result["num_invalid_smiles"] = invalid + result["smiles_check_method"] = method + + if invalid > 0: + scope = "all rows" if method == "full" else f"the {parsed} sampled rows" + result["errors"].append( + f"{invalid} out of {parsed} SMILES in {scope} failed to parse with RDKit. " + "Either pre-clean the CSV with scripts/clean_smiles.py or pass --strict-rdkit to see " + "the full count." + ) + + # 4. Target-column handling — finetune mode only. + if mode == "finetune": + if targets: + missing = [t for t in targets if t not in df.columns] + if missing: + result["errors"].append( + f"target column(s) not found in CSV: {missing}. " + f"Available columns: {result['columns']}" + ) + else: + result["target_columns"] = list(targets) + for t in targets: + nan_count = int(df[t].isna().sum()) + result["num_missing_per_target"][t] = nan_count + # Confirm numeric-ish. + nonnan = df[t].dropna() + converted = pd.to_numeric(nonnan, errors="coerce") + non_numeric_count = int(converted.isna().sum()) + if non_numeric_count > 0: + result["warnings"].append( + f"target column '{t}' has {non_numeric_count} non-numeric value(s) " + f"that will be dropped by the finetune runner." + ) + else: + # Auto-detect — surface candidates so the skill can prompt the user. + result["auto_detected_targets"] = _autodetect_target_columns(df, smiles_col) + if not result["auto_detected_targets"]: + result["errors"].append( + "no numeric non-smiles columns detected. finetune needs at least one target column; " + "specify it explicitly via --targets ." + ) + else: + result["warnings"].append( + f"--targets was not specified; auto-detected candidate target columns " + f"{result['auto_detected_targets']}. The skill will prompt the user to confirm." + ) + + # 5. Small-corpus warning — only for pretrain (other modes can be tiny by design). + if mode == "pretrain" and result["num_rows"] < 100: + result["warnings"].append( + f"pretrain corpus is only {result['num_rows']} molecule(s). Pretraining typically " + f"needs orders of magnitude more — verify this is the intended input." + ) + + result["ok"] = not result["errors"] + return result + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Validate a CSV input for a KERMT agent workflow.") + parser.add_argument("--mode", required=True, choices=["pretrain", "finetune", "inference", "embed"]) + parser.add_argument("--csv", required=True, help="Path to the input CSV") + parser.add_argument("--targets", nargs="+", default=None, + help="(finetune only) target column names. If omitted, the validator auto-detects " + "numeric non-smiles columns and reports them as candidates.") + parser.add_argument("--strict-rdkit", action="store_true", + help="Parse every SMILES with RDKit rather than sampling (slow on large CSVs).") + args = parser.parse_args(argv) + + try: + result = validate(args.mode, args.csv, args.targets, args.strict_rdkit) + except Exception as exc: # noqa: BLE001 + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({ + "ok": False, + "mode": args.mode, + "csv_path": args.csv, + "errors": [f"unhandled exception in validator: {type(exc).__name__}: {exc}"], + "warnings": [], + }, indent=2)) + return 1 + + print(json.dumps(result, indent=2)) + return 0 if result["ok"] else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-infer/scripts/kermt_container.sh b/skills/kermt-infer/scripts/kermt_container.sh new file mode 100755 index 0000000..028057e --- /dev/null +++ b/skills/kermt-infer/scripts/kermt_container.sh @@ -0,0 +1,484 @@ +#!/usr/bin/env bash +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# kermt_container.sh — bootstrap helper for the kermt agent skills. +# +# Two ways to use this file: +# +# 1. As a subcommand dispatcher (recommended for skills): +# "$SKILL_DIR/scripts/kermt_container.sh" ensure_image +# "$SKILL_DIR/scripts/kermt_container.sh" run --ckpt /host/ckpt.pt -- python -c 'import torch; print(torch.cuda.device_count())' +# "$SKILL_DIR/scripts/kermt_container.sh" run_detached --name foo --run-dir runs/foo -- bash train.sh +# +# 2. Sourced into a shell or another script, then call the kermt_* functions +# directly: +# source "$SKILL_DIR/scripts/kermt_container.sh" +# kermt_ensure_image +# kermt_run --ckpt /host/ckpt.pt -- python ... +# +# Configuration (override via env vars before invocation): +# KERMT_IMAGE docker image tag (default: kermt:latest) +# KERMT_REPO host path to the kermt repo checkout (default: auto-derived +# from this script's location) +# KERMT_GPUS value passed to docker --gpus (default: all) +# +# Mount flags accepted by kermt_run / kermt_run_detached: +# --data bind to /data (read-only). If is a file, +# its PARENT directory is mounted at /data so +# commands can use /data/; if is a +# directory, it is mounted at /data directly. +# --ckpt bind to /ckpt (read-only; the path is mounted as-is) +# --vocab-dir bind to /vocab (read-only) +# --run-dir bind to /runs (read-write; created on host if missing) +# --model-dir bind to /model (read-write; created on host if missing). +# Target for released-model downloads (fetch_released_model.py). +# +# Additional flags for kermt_run_detached: +# --name docker container name (default: kermt--) +# +# Everything after `--` is the command passed to the container. It runs inside +# the `kermt` conda environment (the image's default env). + +set -o pipefail + +: "${KERMT_IMAGE:=kermt:latest}" +: "${KERMT_GPUS:=all}" + +# The skill may be installed outside the KERMT checkout. Mount its own helpers +# separately so container commands always execute the distributed skill copy. +_kermt_script_dir="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)" +_kermt_bundle_dir="$(cd "$_kermt_script_dir/.." && pwd)" +if [[ -z "${KERMT_REPO:-}" ]]; then + for _kermt_start in "$_kermt_script_dir" "$PWD"; do + _kermt_candidate="$_kermt_start" + while [[ "$_kermt_candidate" != / ]]; do + if [[ -f "$_kermt_candidate/main.py" && -d "$_kermt_candidate/kermt" ]]; then + KERMT_REPO="$_kermt_candidate" + break 2 + fi + _kermt_candidate="$(dirname "$_kermt_candidate")" + done + done + unset _kermt_start _kermt_candidate +fi +unset _kermt_script_dir + +_kermt_require_repo() { + if [[ -z "${KERMT_REPO:-}" || ! -f "$KERMT_REPO/main.py" || ! -d "$KERMT_REPO/kermt" ]]; then + echo "[kermt] Set KERMT_REPO to the KERMT checkout containing main.py and kermt/." >&2 + return 1 + fi + KERMT_REPO="$(cd "$KERMT_REPO" && pwd)" || return $? + export KERMT_REPO +} + +# ----------------------------------------------------------------------------- +# Host environment checks +# ----------------------------------------------------------------------------- + +kermt_check_docker() { + if ! command -v docker >/dev/null 2>&1; then + echo "[kermt] error: docker not found on PATH. Install Docker first." >&2 + return 1 + fi + if ! docker info >/dev/null 2>&1; then + echo "[kermt] error: docker daemon not reachable. Is the docker service running, and is your user in the 'docker' group?" >&2 + return 1 + fi +} + +kermt_check_system() { + # Probe host system and report GPU presence + VRAM + compute capability + + # driver / CUDA version + disk space. Emits a single JSON document to + # stdout that the calling skill consumes; exits 0 with `ok: false` and a + # populated `gaps` array when anything is below the per-workflow minimum, + # exits 1 only on unexpected internal errors. Uses host nvidia-smi + df + + # host python3 (stdlib only). + python3 - "$KERMT_REPO" "$KERMT_IMAGE" <<'PYEOF' +import json, os, shutil, subprocess, sys + +repo, image = sys.argv[1], sys.argv[2] + +result = { + "ok": True, + "gpus": [], + "disk": {"path": repo, "free_gb": None, "min_gb": 20}, + "host": {"docker": None, "nvidia_smi": None, "container_toolkit": None}, + "image": {"tag": image, "present_locally": None}, + "gaps": [], +} + +def _gap(msg): + result["ok"] = False + result["gaps"].append(msg) + +# docker presence +try: + r = subprocess.run(["docker", "info"], capture_output=True, text=True, timeout=10) + result["host"]["docker"] = "ok" if r.returncode == 0 else f"failed: {r.stderr.strip().splitlines()[-1] if r.stderr else 'unknown'}" + if r.returncode != 0: + _gap("docker daemon not reachable (is the service running, and is your user in the 'docker' group?)") +except FileNotFoundError: + result["host"]["docker"] = "not found" + _gap("docker not on PATH; install Docker first") +except Exception as e: + result["host"]["docker"] = f"error: {e}" + _gap(f"docker probe failed: {e}") + +# nvidia-smi (host driver) +try: + r = subprocess.run( + ["nvidia-smi", "--query-gpu=name,memory.total,compute_cap,driver_version,uuid", + "--format=csv,noheader,nounits"], + capture_output=True, text=True, timeout=10, + ) + if r.returncode == 0: + result["host"]["nvidia_smi"] = "ok" + for line in r.stdout.strip().splitlines(): + parts = [p.strip() for p in line.split(",")] + if len(parts) >= 5: + try: + vram_mb = int(parts[1]) + except ValueError: + vram_mb = None + result["gpus"].append({ + "name": parts[0], + "vram_mb": vram_mb, + "compute_cap": parts[2], + "driver": parts[3], + "uuid": parts[4], + }) + if not result["gpus"]: + _gap("nvidia-smi succeeded but reported no GPUs") + else: + result["host"]["nvidia_smi"] = "failed" + _gap("nvidia-smi found but failed; is the NVIDIA driver loaded?") +except FileNotFoundError: + result["host"]["nvidia_smi"] = "not found" + _gap("nvidia-smi not on PATH; install the NVIDIA driver") +except Exception as e: + result["host"]["nvidia_smi"] = f"error: {e}" + _gap(f"nvidia-smi probe failed: {e}") + +# disk free at the repo location +try: + free_bytes = shutil.disk_usage(repo).free + free_gb = free_bytes // (1024**3) + result["disk"]["free_gb"] = free_gb + if free_gb < result["disk"]["min_gb"]: + _gap(f"disk free at {repo} is {free_gb} GB; need at least {result['disk']['min_gb']} GB for the kermt image") +except Exception as e: + _gap(f"could not check disk space at {repo}: {e}") + +# image presence (informational only) +try: + r = subprocess.run(["docker", "image", "inspect", image], capture_output=True, text=True, timeout=10) + result["image"]["present_locally"] = (r.returncode == 0) +except Exception: + result["image"]["present_locally"] = None + +# nvidia-container-toolkit probe — only meaningful if both docker and a +# locally-present image are available. Pick kermt:$tag first; fall back to +# the small CUDA base image if that's the only one present; otherwise skip +# (avoid pulling anything). +def _probe_image(): + for img in (image, "nvidia/cuda:12.6.3-base-ubuntu22.04"): + r = subprocess.run(["docker", "image", "inspect", img], capture_output=True) + if r.returncode == 0: + return img + return None + +probe_img = _probe_image() +if probe_img: + try: + r = subprocess.run( + ["docker", "run", "--rm", "--gpus", "all", probe_img, "nvidia-smi"], + capture_output=True, text=True, timeout=60, + ) + if r.returncode == 0: + result["host"]["container_toolkit"] = f"ok (probed via {probe_img})" + else: + result["host"]["container_toolkit"] = f"failed (probed via {probe_img})" + _gap("`docker run --gpus all` failed; install nvidia-container-toolkit and ensure the host driver supports it") + except Exception as e: + result["host"]["container_toolkit"] = f"error: {e}" + _gap(f"nvidia-container-toolkit probe failed: {e}") +else: + result["host"]["container_toolkit"] = "skipped (no probe image present locally; run ensure_image first)" + +print(json.dumps(result, indent=2)) +PYEOF +} + +kermt_check_gpu() { + # Probes whether `docker --gpus all` is wired up (nvidia-container-toolkit). + # Image-selection priority (never pulls anything): + # 1) $KERMT_IMAGE if it exists locally, + # 2) else nvidia/cuda:12.6.3-base-ubuntu22.04 if it exists locally, + # 3) else skip with a warning (return 0). The smoke test inside kermt_run + # will catch broken GPU passthrough later anyway. + local probe_img="" + if docker image inspect "$KERMT_IMAGE" >/dev/null 2>&1; then + probe_img="$KERMT_IMAGE" + elif docker image inspect nvidia/cuda:12.6.3-base-ubuntu22.04 >/dev/null 2>&1; then + probe_img="nvidia/cuda:12.6.3-base-ubuntu22.04" + else + echo "[kermt] check_gpu: skipped — neither '$KERMT_IMAGE' nor 'nvidia/cuda:12.6.3-base-ubuntu22.04' is present locally. Run 'ensure_image' first, or this probe will be exercised by the in-container smoke test." >&2 + return 0 + fi + if ! docker run --rm --gpus all "$probe_img" nvidia-smi >/dev/null 2>&1; then + echo "[kermt] error: 'docker run --gpus all' failed (probe image: $probe_img). Install nvidia-container-toolkit and ensure the host has a CUDA-capable NVIDIA driver." >&2 + return 1 + fi +} + +# ----------------------------------------------------------------------------- +# Image build / verification +# ----------------------------------------------------------------------------- + +kermt_ensure_image() { + _kermt_require_repo || return $? + kermt_check_docker || return $? + if docker image inspect "$KERMT_IMAGE" >/dev/null 2>&1; then + local id + id=$(docker image inspect "$KERMT_IMAGE" --format '{{.Id}}' 2>/dev/null | cut -c1-19) + echo "[kermt] image '$KERMT_IMAGE' already present (${id:-unknown})" + return 0 + fi + echo "[kermt] image '$KERMT_IMAGE' not found; building from $KERMT_REPO/Dockerfile" + echo "[kermt] first build typically takes 10-20 minutes on a typical workstation; subsequent runs reuse the cached image" + docker build -t "$KERMT_IMAGE" -f "$KERMT_REPO/Dockerfile" "$KERMT_REPO" +} + +# ----------------------------------------------------------------------------- +# Mount-flag parser, internal +# ----------------------------------------------------------------------------- +# Reads flags from the caller's positional args until it hits '--', appending +# `-v src:dst[:ro]` pairs into the caller-provided array name (passed as $1). +# Returns the number of caller-provided args consumed via _kermt_consumed. +# This is bash-specific (uses nameref via `declare -n`). + +_kermt_parse_mounts() { + local -n _out="$1" + shift + _kermt_consumed=0 + while [[ $# -gt 0 ]]; do + case "$1" in + --) + return 0 + ;; + --data) + [[ -e "$2" ]] || { echo "[kermt] --data path not found: $2" >&2; return 1; } + # If the user passes a file, mount its parent directory at /data so + # downstream commands can refer to /data/. Mounting a + # single file at /data makes the path-as-directory pattern in the + # skill examples (`--csv /data/`) fail with "not found". + if [[ -d "$2" ]]; then + _out+=("-v" "$(realpath "$2"):/data:ro") + else + _out+=("-v" "$(realpath "$(dirname "$2")"):/data:ro") + fi + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --ckpt) + [[ -e "$2" ]] || { echo "[kermt] --ckpt path not found: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/ckpt:ro") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --vocab-dir) + [[ -d "$2" ]] || { echo "[kermt] --vocab-dir not found or not a directory: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/vocab:ro") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --run-dir) + mkdir -p "$2" || { echo "[kermt] failed to create --run-dir: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/runs") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --model-dir) + mkdir -p "$2" || { echo "[kermt] failed to create --model-dir: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/model") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + *) + return 0 + ;; + esac + done +} + +# ----------------------------------------------------------------------------- +# Foreground / detached run +# ----------------------------------------------------------------------------- + +# Capture host-side git state for the repo and emit `-e KERMT_REPO_COMMIT=… +# -e KERMT_REPO_DIRTY=true|false` flags. Used by the run / run_detached +# wrappers so the runner's run.json manifest gets honest commit info even +# though `git -C /workspace` inside the container fails due to bind-mount +# ownership. +_kermt_git_env_flags() { + local commit="unknown" + local dirty="false" + if command -v git >/dev/null 2>&1 && [[ -d "$KERMT_REPO/.git" ]]; then + local c + c=$(git -C "$KERMT_REPO" rev-parse HEAD 2>/dev/null) && commit="$c" + # `--untracked-files=no` filters out user-private notes (e.g. a CLAUDE.md + # or RELEASE_PLAN_v2.0.md at the repo root) that wouldn't affect + # reproducibility — only modifications to tracked files do. + if [[ -n "$(git -C "$KERMT_REPO" status --porcelain --untracked-files=no 2>/dev/null | head -n 1)" ]]; then + dirty="true" + fi + fi + printf '%s\n%s\n%s\n%s\n' "-e" "KERMT_REPO_COMMIT=$commit" "-e" "KERMT_REPO_DIRTY=$dirty" +} + +# Forward HF_TOKEN into the container when it is set, so fetch_released_model.py +# can authenticate to Hugging Face. The current release is public (no token +# needed); this only guards against shared-IP rate limits or a future gated +# repo. Emits nothing when HF_TOKEN is unset. +_kermt_hf_env_flags() { + if [[ -n "${HF_TOKEN:-}" ]]; then + printf '%s\n%s\n' "-e" "HF_TOKEN=$HF_TOKEN" + fi +} + +kermt_run() { + kermt_ensure_image || return $? + local mount_args=() + _kermt_parse_mounts mount_args "$@" || return $? + shift "$_kermt_consumed" + if [[ "${1:-}" != "--" ]]; then + echo "[kermt] expected '--' separating mount flags from the command (got '${1:-}')" >&2 + return 1 + fi + shift + if [[ $# -eq 0 ]]; then + echo "[kermt] no command supplied after '--'" >&2 + return 1 + fi + local git_args=() + while IFS= read -r line; do git_args+=("$line"); done < <(_kermt_git_env_flags) + local hf_args=() + while IFS= read -r line; do hf_args+=("$line"); done < <(_kermt_hf_env_flags) + docker run --rm --gpus "$KERMT_GPUS" \ + --user "$(id -u):$(id -g)" \ + -v "$KERMT_REPO:/workspace" \ + -v "$_kermt_bundle_dir:/skill:ro" \ + "${mount_args[@]}" \ + -w /workspace \ + -e KERMT_REPO=/workspace \ + -e PYTHONPATH=/workspace \ + -e HOME=/tmp/kermt-home \ + "${git_args[@]}" \ + "${hf_args[@]}" \ + "$KERMT_IMAGE" \ + conda run -n kermt --no-capture-output bash -c "$*" +} + +kermt_run_detached() { + kermt_ensure_image || return $? + local name="" + local mount_args=() + # Pull --name out first, then let the shared mount parser handle the rest. + while [[ $# -gt 0 ]]; do + case "$1" in + --name) name="$2"; shift 2 ;; + --) break ;; + --data|--ckpt|--vocab-dir|--run-dir|--model-dir) break ;; + *) break ;; + esac + done + _kermt_parse_mounts mount_args "$@" || return $? + shift "$_kermt_consumed" + if [[ "${1:-}" != "--" ]]; then + echo "[kermt] expected '--' separating mount flags from the command (got '${1:-}')" >&2 + return 1 + fi + shift + if [[ $# -eq 0 ]]; then + echo "[kermt] no command supplied after '--'" >&2 + return 1 + fi + if [[ -z "$name" ]]; then + name="kermt-$(date -u +%Y%m%dT%H%M%SZ)-$$" + fi + local cid + local git_args=() + while IFS= read -r line; do git_args+=("$line"); done < <(_kermt_git_env_flags) + local hf_args=() + while IFS= read -r line; do hf_args+=("$line"); done < <(_kermt_hf_env_flags) + cid=$(docker run -d --gpus "$KERMT_GPUS" \ + --user "$(id -u):$(id -g)" \ + --name "$name" \ + -v "$KERMT_REPO:/workspace" \ + -v "$_kermt_bundle_dir:/skill:ro" \ + "${mount_args[@]}" \ + -w /workspace \ + -e KERMT_REPO=/workspace \ + -e PYTHONPATH=/workspace \ + -e HOME=/tmp/kermt-home \ + "${git_args[@]}" \ + "${hf_args[@]}" \ + "$KERMT_IMAGE" \ + conda run -n kermt --no-capture-output bash -c "$*") || return $? + echo "[kermt] container started: name=$name id=$cid" + echo "[kermt] follow logs: docker logs -f $name" + echo "[kermt] wait for exit: docker wait $name" + echo "[kermt] stop: docker stop $name" + echo "$cid" +} + +# ----------------------------------------------------------------------------- +# Subcommand dispatch when invoked directly (not sourced) +# ----------------------------------------------------------------------------- + +if [[ "${BASH_SOURCE[0]:-$0}" == "${0}" ]]; then + cmd="${1:-}"; shift || true + case "$cmd" in + check_docker) kermt_check_docker "$@" ;; + check_gpu) kermt_check_gpu "$@" ;; + check_system) kermt_check_system "$@" ;; + ensure_image) kermt_ensure_image "$@" ;; + run) kermt_run "$@" ;; + run_detached) kermt_run_detached "$@" ;; + ""|-h|--help) + cat >&2 < [args...] + +Subcommands: + check_docker Verify docker is installed and the daemon is reachable. + check_gpu Verify 'docker --gpus all' works (nvidia-container-toolkit). + check_system Emit a JSON probe of host GPU + VRAM + compute_cap + + driver + disk space + container toolkit + image presence. + Exits 0 with ok=false + a 'gaps' list when anything's + below the per-workflow minimum. + ensure_image Build kermt:latest from \$KERMT_REPO/Dockerfile if missing. + run [flags] -- ... Run a command inside the container (foreground, --rm). + run_detached [flags] -- ... + Run detached; prints container name + id + log hint. + +Mount flags (for run / run_detached): + --data bind to /data (read-only) + --ckpt bind to /ckpt (read-only) + --vocab-dir bind to /vocab (read-only) + --run-dir bind to /runs (read-write; created on host if missing) + --model-dir bind to /model (read-write; released-model download target) + +Additional flags for run_detached: + --name container name (default: kermt--) + +Environment overrides: + KERMT_IMAGE default kermt:latest + KERMT_REPO checkout path; otherwise discovered above the skill or working directory + KERMT_GPUS default all +EOF + exit 1 + ;; + *) + echo "[kermt] unknown subcommand: $cmd" >&2 + echo "[kermt] run '$0 --help' for usage" >&2 + exit 1 + ;; + esac +fi diff --git a/skills/kermt-infer/scripts/prepare_data.py b/skills/kermt-infer/scripts/prepare_data.py new file mode 100644 index 0000000..f0edb3e --- /dev/null +++ b/skills/kermt-infer/scripts/prepare_data.py @@ -0,0 +1,817 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Mode-dispatched data preparation pipeline for the KERMT agent skills. + +Composes the existing repo data-prep scripts (`scripts/clean_smiles.py`, +`scripts/save_features.py`, `scripts/build_vocab.py`, `scripts/split_data.py`) +into a single one-call entry point per workflow. Output lands in `--out` with +a `prepare_data.json` manifest that the downstream runners read. + +Mode pipelines +-------------- +pretrain : clean -> (optional auto-split train into train+val by --val-frac) + -> save_features (fgtasklabel) on each CSV + -> vocab step: if --vocab-dir / --{atom,bond,smiles}-vocab given, + copy those through (continue-pretrain case — the ckpt's vocab + is authoritative); else if --skip-vocab, skip; + else build_vocab on train (pretrain-from-scratch case) + -> split_data (graph + feature shards + summary.txt) per CSV +finetune : clean each provided CSV -> (optional random split when only one + CSV is provided; emits a strong warning recommending scaffold- + balanced pre-splits) -> save_features (rdkit_2d_normalized) per CSV +inference : clean -> save_features (rdkit_2d_normalized) +embed : clean only (extract_embeddings.py featurizes on the fly) + +Output convention +----------------- +The manifest under `/prepare_data.json` captures every step's inputs, +outputs, duration, and skipped-due-to-existing flag, plus a top-level +`split_method` field (one of: "user_provided", "random", "n/a") that the +finetune runner uses to pass the correct `--split_type` to main.py. + +Subprocess composition +---------------------- +Each underlying script is invoked via `subprocess.run`. The PYTHONPATH=/workspace +env var (set by `scripts/kermt_container.sh`) makes the `kermt` package +importable inside the subprocesses; without it, build_vocab.py and split_data.py +fail with `ModuleNotFoundError: No module named 'kermt'`. + +CLI +--- + prepare_data.py --mode {pretrain|finetune|inference|embed} + --csv --out + [--val-csv ] [--test-csv ] + [--val-frac 0.1] [--test-frac 0.1] [--seed 0] + [--sample-per-file 100000] [--vocab-format json] + [--dataset-name pretrain] + [--targets COL [COL ...]] + [--features-generator ] + [--smiles-column 0] + [--force] [--skip-clean] [--skip-features] + [--skip-vocab] [--skip-split] +""" +from __future__ import annotations + +import argparse +import json +import os +import shutil +import subprocess +import sys +import time +import traceback +from pathlib import Path +from typing import Any + +import pandas as pd + +# sys.path tweak so `_utils` is importable regardless of how this script +# is invoked (kermt_run sets PYTHONPATH=/workspace; bare-Python launches +# from the host don't). +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import PRETRAIN_VOCAB_STEMS, resolve_kermt_repo, validate_vocab_file # noqa: E402 + + +REPO_ROOT = resolve_kermt_repo() +EXISTING_SCRIPTS = REPO_ROOT / "scripts" + +DEFAULT_FEATURES_GENERATOR = { + "pretrain": "fgtasklabel", + "finetune": "rdkit_2d_normalized", + "inference": "rdkit_2d_normalized", + "embed": None, # not used +} + +VALID_MODES = ("pretrain", "finetune", "inference", "embed") + + +# --------------------------------------------------------------------------- +# Subprocess helpers +# --------------------------------------------------------------------------- + +def _run(cmd: list[str], step_name: str, manifest: dict[str, Any]) -> dict[str, Any]: + """Run a subprocess, append a step entry to manifest, raise on failure.""" + step: dict[str, Any] = { + "name": step_name, + "cmd": cmd, + "duration_s": None, + "ok": False, + "stderr_tail": "", + "skipped_due_to_existing": False, + } + t0 = time.time() + proc = subprocess.run(cmd, capture_output=True, text=True) + step["duration_s"] = round(time.time() - t0, 2) + if proc.returncode != 0: + step["stderr_tail"] = (proc.stderr or "").splitlines()[-20:] + step["ok"] = False + manifest["steps"].append(step) + raise RuntimeError( + f"step '{step_name}' failed (exit {proc.returncode}); " + f"command: {' '.join(cmd)}\nstderr tail:\n" + "\n".join(step["stderr_tail"]) + ) + step["ok"] = True + manifest["steps"].append(step) + return step + + +def _skipped(step_name: str, output_path: str, manifest: dict[str, Any]) -> dict[str, Any]: + step = { + "name": step_name, + "output": output_path, + "ok": True, + "duration_s": 0.0, + "skipped_due_to_existing": True, + } + manifest["steps"].append(step) + return step + + +def _exists_nonempty(path: Path) -> bool: + """File exists with non-zero size, or directory exists with at least one entry.""" + if not path.exists(): + return False + if path.is_file(): + return path.stat().st_size > 0 + if path.is_dir(): + try: + next(path.iterdir()) + return True + except StopIteration: + return False + return False + + +# --------------------------------------------------------------------------- +# Per-script wrappers +# --------------------------------------------------------------------------- + +def _resolve_smiles_column(csv_path: Path, explicit_value: int | None) -> int: + """Return the 0-based index of the SMILES column in csv_path. + + Auto-detection rule when `explicit_value is None`: + 1. Read the CSV header (first non-empty row). + 2. Prefer an exact lowercase `smiles` column (kermt convention). + 3. Otherwise accept a single case-insensitive match + (`SMILES`, `Smiles`, etc.). + 4. If no match (or multiple ambiguous matches), raise a ValueError + that surfaces the header so the user can disambiguate via + `--smiles-column N`. + + Real datasets routinely place SMILES at column index ≠ 0 + (e.g. openadmet's all.csv has "Molecule Name" at col 0 and "SMILES" + at col 1). Auto-detection prevents the silent 0-row-clean failure + mode where every row gets rejected because col 0 doesn't parse as + a SMILES string. + """ + if explicit_value is not None: + return explicit_value + + if not csv_path.is_file(): + raise ValueError(f"input CSV not found: {csv_path}") + + import csv as _csv + with csv_path.open("r", newline="") as f: + reader = _csv.reader(f) + try: + header = next(reader) + except StopIteration: + raise ValueError(f"input CSV {csv_path} is empty") + + stripped = [c.strip() for c in header] + # Prefer exact lowercase "smiles" + exact = [i for i, c in enumerate(stripped) if c == "smiles"] + if exact: + return exact[0] + # Then case-insensitive + ci = [i for i, c in enumerate(stripped) if c.lower() == "smiles"] + if len(ci) == 1: + return ci[0] + if len(ci) > 1: + raise ValueError( + f"input CSV {csv_path} has multiple SMILES-named columns: " + f"{[header[i] for i in ci]} at indices {ci}. " + "Pass --smiles-column N (0-based) to disambiguate." + ) + raise ValueError( + f"could not auto-detect a SMILES column in {csv_path}. " + f"Header columns: {header}. " + "Pass --smiles-column N (0-based) to specify which column holds SMILES." + ) + + +def _clean_smiles( + input_csv: Path, output_csv: Path, smiles_column: int, manifest: dict[str, Any], force: bool +) -> Path: + if not force and _exists_nonempty(output_csv): + _skipped(f"clean_smiles({input_csv.name})", str(output_csv), manifest) + return output_csv + output_csv.parent.mkdir(parents=True, exist_ok=True) + if force and output_csv.exists(): + # clean_smiles.py prompts interactively (input()) when the output file + # already exists — that's an EOFError in a non-TTY subprocess. Pre-delete. + output_csv.unlink() + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "clean_smiles.py"), + "--input", str(input_csv), + "--output", str(output_csv), + "--smiles_column", str(smiles_column), + ] + _run(cmd, f"clean_smiles({input_csv.name})", manifest) + return output_csv + + +def _reduce_to_smiles_column( + csv_path: Path, smiles_column: int, manifest: dict[str, Any] +) -> Path: + """Rewrite an inference CSV to keep only the SMILES column (at index 0). + + Downstream `kermt.util.utils.get_data` -> `MoleculeDatapoint.__init__` + floats every column after SMILES, which crashes on non-numeric passthrough + columns (e.g. a 'split' label of 'train'/'val'/'test', or a 'Molecule Name' + string). Inference does not need target columns, so drop them here. + + Note on skip semantics: this step is idempotent — running it on an + already-single-column file is a no-op. We record that with + `skipped_due_to_idempotent: True`, NOT `skipped_due_to_existing: True`. + The two fields have different meanings: `_existing` means "I found a + cached output file from a prior run and reused it" (overridden by + `--force`); `_idempotent` means "the input is already in the desired + state, so re-executing changes nothing" (safe to skip even under + `--force`). + """ + step_name = f"reduce_to_smiles_only({csv_path.name})" + start = time.time() + df = pd.read_csv(csv_path) + if df.shape[1] == 1: + manifest["steps"].append({ + "name": step_name, + "output": str(csv_path), + "ok": True, + "duration_s": time.time() - start, + "skipped_due_to_idempotent": True, + "note": "already single-column", + }) + return csv_path + effective_col = smiles_column if 0 <= smiles_column < df.shape[1] else 0 + df.iloc[:, [effective_col]].to_csv(csv_path, index=False) + manifest["steps"].append({ + "name": step_name, + "output": str(csv_path), + "ok": True, + "duration_s": time.time() - start, + "input_cols": int(df.shape[1]), + "kept_col": effective_col, + "kept_col_name": str(df.columns[effective_col]), + }) + return csv_path + + +def _save_features( + csv_path: Path, npz_path: Path, generator: str, manifest: dict[str, Any], force: bool +) -> Path: + if not force and _exists_nonempty(npz_path): + _skipped(f"save_features({csv_path.name}, {generator})", str(npz_path), manifest) + return npz_path + npz_path.parent.mkdir(parents=True, exist_ok=True) + if force and npz_path.exists(): + npz_path.unlink() # --restart still loads partial state if file exists; pre-delete to be safe + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "save_features.py"), + "--data_path", str(csv_path), + "--save_path", str(npz_path), + "--features_generator", generator, + "--restart", + ] + _run(cmd, f"save_features({csv_path.name}, {generator})", manifest) + return npz_path + + +def _resolve_vocab_inputs(args: argparse.Namespace) -> dict[str, Path | None] | None: + """Returns {atom, bond, smiles}->Path|None when the user supplied vocab + inputs (via --vocab-dir or --atom-vocab/--bond-vocab/--smiles-vocab), + else None (signal to fall through to build_vocab). + + Conventional filenames inside --vocab-dir: + pretrain_atom_vocab.{json,pkl} + pretrain_bond_vocab.{json,pkl} + pretrain_smiles_vocab.pkl + """ + if args.vocab_dir: + d = Path(args.vocab_dir).resolve() + if not d.is_dir(): + raise FileNotFoundError(f"--vocab-dir not found or not a directory: {d}") + def _find(stem: str, exts: tuple[str, ...]) -> Path | None: + for ext in exts: + p = d / f"{stem}.{ext}" + if p.is_file(): + return p + return None + atom = _find(PRETRAIN_VOCAB_STEMS["atom"], ("json", "pkl")) + bond = _find(PRETRAIN_VOCAB_STEMS["bond"], ("json", "pkl")) + smiles = _find(PRETRAIN_VOCAB_STEMS["smiles"], ("pkl",)) + if atom is None and bond is None and smiles is None: + stems = [PRETRAIN_VOCAB_STEMS[k] for k in ("atom", "bond", "smiles")] + raise FileNotFoundError( + f"--vocab-dir {d} contained no {{ {', '.join(stems) }}}.{{json,pkl}} " + f"files. Expected at least {PRETRAIN_VOCAB_STEMS['atom']} + " + f"{PRETRAIN_VOCAB_STEMS['bond']}." + ) + return {"atom": atom, "bond": bond, "smiles": smiles} + + if args.atom_vocab or args.bond_vocab or args.smiles_vocab: + return { + "atom": Path(args.atom_vocab).resolve() if args.atom_vocab else None, + "bond": Path(args.bond_vocab).resolve() if args.bond_vocab else None, + "smiles": Path(args.smiles_vocab).resolve() if args.smiles_vocab else None, + } + + return None + + +def _copy_provided_vocab( + src: dict[str, Path | None], dst_dir: Path, dataset_name: str, manifest: dict[str, Any], + force: bool, +) -> dict[str, Path]: + """When the user supplies vocab files (use ckpt's vocab as-is), + copy them into `/__vocab.` so the + downstream pretrain command sees the conventional filenames. + + `src` is `{atom: Path|None, bond: Path|None, smiles: Path|None}`. The atom + and bond entries must be both present or both absent (paired). smiles is + optional (cmim/hybrid only). + + Returns the same dict of (resolved) destination paths. + """ + import shutil + if (src["atom"] is None) != (src["bond"] is None): + raise ValueError( + "vocab pass-through requires atom and bond vocab paths to be paired; " + "got atom=" + str(src["atom"]) + ", bond=" + str(src["bond"]) + ) + out: dict[str, Path] = {} + dst_dir.mkdir(parents=True, exist_ok=True) + for which, path in src.items(): + if path is None: + continue + # Validate the source file IS a loadable KERMT vocab before copying. + # Catches the "user pointed --smiles-vocab at a random pickle" case + # early, with a clear error, instead of letting it surface as a cryptic + # SMILESVocab.load_vocab failure at pretrain_ddp.py launch time. + validate_vocab_file(path, kind=which) + ext = path.suffix.lstrip(".") + if which == "smiles": + ext = "pkl" # smiles vocab is always pickle + dst = dst_dir / f"{dataset_name}_{which}_vocab.{ext}" + if not force and _exists_nonempty(dst): + _skipped(f"copy_vocab({which})", str(dst), manifest) + out[which] = dst + continue + if force and dst.exists(): + dst.unlink() + shutil.copy2(path, dst) + manifest["steps"].append({ + "name": f"copy_vocab({which})", + "src": str(path), "dst": str(dst), "ok": True, + "duration_s": 0.0, "skipped_due_to_existing": False, + }) + out[which] = dst + return out + + +def _build_vocab( + csv_path: Path, vocab_dir: Path, dataset_name: str, vocab_format: str, + manifest: dict[str, Any], force: bool, +) -> dict[str, Path]: + """Builds atom + bond (in --vocab-format) and smiles (always pickle) vocabs. + Returns a dict of {atom, bond, smiles} -> Path.""" + suffix = "json" if vocab_format == "json" else "pkl" + expected = { + "atom": vocab_dir / f"{dataset_name}_atom_vocab.{suffix}", + "bond": vocab_dir / f"{dataset_name}_bond_vocab.{suffix}", + "smiles": vocab_dir / f"{dataset_name}_smiles_vocab.pkl", + } + if not force and all(_exists_nonempty(p) for p in expected.values()): + _skipped(f"build_vocab({csv_path.name})", str(vocab_dir), manifest) + return expected + vocab_dir.mkdir(parents=True, exist_ok=True) + if force: + for p in expected.values(): + if p.exists(): + p.unlink() + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "build_vocab.py"), + "--data_path", str(csv_path), + "--vocab_save_folder", str(vocab_dir), + "--dataset_name", dataset_name, + "--vocab_format", vocab_format, + ] + _run(cmd, f"build_vocab({csv_path.name})", manifest) + return expected + + +def _split_data( + csv_path: Path, features_path: Path | None, sample_per_file: int, output_dir: Path, + manifest: dict[str, Any], force: bool, +) -> Path: + """Run split_data.py to produce shard dirs (graph/ + optionally feature/ + summary.txt).""" + summary = output_dir / "summary.txt" + if not force and _exists_nonempty(summary): + _skipped(f"split_data({csv_path.name})", str(output_dir), manifest) + return output_dir + if force and output_dir.exists(): + shutil.rmtree(output_dir) + output_dir.mkdir(parents=True, exist_ok=True) + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "split_data.py"), + "--data_path", str(csv_path), + "--sample_per_file", str(sample_per_file), + "--output_path", str(output_dir), + ] + if features_path is not None: + cmd += ["--features_path", str(features_path)] + _run(cmd, f"split_data({csv_path.name})", manifest) + return output_dir + + +# --------------------------------------------------------------------------- +# Random splitter (used only when the user supplies a single CSV) +# --------------------------------------------------------------------------- + +def _random_split_csv( + src_csv: Path, dst_csvs: dict[str, Path], fractions: dict[str, float], seed: int, + manifest: dict[str, Any], force: bool, +) -> None: + """Shuffle src_csv and partition rows into dst_csvs by fractions. + `dst_csvs` and `fractions` are dicts keyed by the split name (e.g. 'train', 'val'). + Sum of fractions must be 1.0 (within float tolerance). Writes each dst_csv with the + same header as the input.""" + step = { + "name": f"random_split({src_csv.name})", + "seed": seed, + "fractions": fractions, + "ok": False, + "duration_s": None, + "skipped_due_to_existing": False, + "row_counts": {}, + } + if not force and all(_exists_nonempty(p) for p in dst_csvs.values()): + step["skipped_due_to_existing"] = True + step["ok"] = True + manifest["steps"].append(step) + return + + if abs(sum(fractions.values()) - 1.0) > 1e-6: + raise ValueError(f"split fractions must sum to 1.0 (got {sum(fractions.values())})") + + t0 = time.time() + df = pd.read_csv(src_csv).sample(frac=1.0, random_state=seed).reset_index(drop=True) + n = len(df) + sizes: dict[str, int] = {} + remaining = n + split_names = list(fractions.keys()) + for name in split_names[:-1]: + sizes[name] = int(round(fractions[name] * n)) + remaining -= sizes[name] + sizes[split_names[-1]] = remaining + + start = 0 + for name in split_names: + dst = dst_csvs[name] + dst.parent.mkdir(parents=True, exist_ok=True) + df.iloc[start:start + sizes[name]].to_csv(dst, index=False) + step["row_counts"][name] = sizes[name] + start += sizes[name] + + step["duration_s"] = round(time.time() - t0, 2) + step["ok"] = True + manifest["steps"].append(step) + + +def _emit_random_split_warning( + src_csv: Path, fractions: dict[str, float], seed: int, manifest: dict[str, Any] +) -> None: + row_counts = manifest["steps"][-1].get("row_counts", {}) + n = sum(row_counts.values()) if row_counts else "?" + lines = [ + f"WARNING: Auto-splitting {n} rows from {src_csv.name} into:", + ] + for name, frac in fractions.items(): + cnt = row_counts.get(name, "?") + lines.append(f" {name}: {cnt} rows ({frac * 100:.1f}%)") + lines += [ + f"using random split with seed {seed}.", + "", + "This is a RANDOM split. For rigorous ADMET evaluation, scaffold-balanced", + "(or other structure-aware) splits are strongly preferred — molecules with", + "similar scaffolds can leak across splits and inflate apparent generalization.", + "", + "To use your own pre-computed splits instead, pass:", + " --train-csv --val-csv --test-csv ", + "", + "To customize fractions:", + " --val-frac 0.15 --test-frac 0.15", + ] + warning = "\n".join(lines) + print(warning, file=sys.stderr) + manifest["warnings"].append(warning) + + +# --------------------------------------------------------------------------- +# Mode pipelines +# --------------------------------------------------------------------------- + +def _prepare_embed(args, out: Path, manifest: dict[str, Any]) -> None: + manifest["split_method"] = "n/a" + if args.skip_clean: + clean = Path(args.csv) + manifest["steps"].append({"name": "clean_smiles", "skipped_by_flag": True, "ok": True}) + else: + clean = _clean_smiles(Path(args.csv), out / "clean.csv", args.smiles_column, manifest, args.force) + manifest["outputs"]["clean_csv"] = str(clean) + + +def _prepare_inference(args, out: Path, manifest: dict[str, Any]) -> None: + manifest["split_method"] = "n/a" + clean = _clean_smiles(Path(args.csv), out / "clean.csv", args.smiles_column, manifest, args.force) + # Reduce to SMILES-only: downstream get_data/MoleculeDatapoint floats every + # non-SMILES column, which crashes on non-numeric passthrough columns + # (e.g. a 'split' label). Inference does not need target columns. + _reduce_to_smiles_column(clean, args.smiles_column, manifest) + manifest["outputs"]["clean_csv"] = str(clean) + if args.skip_features: + manifest["steps"].append({"name": "save_features", "skipped_by_flag": True, "ok": True}) + return + generator = args.features_generator or DEFAULT_FEATURES_GENERATOR["inference"] + npz = _save_features(clean, out / "clean.npz", generator, manifest, args.force) + manifest["outputs"]["clean_npz"] = str(npz) + + +def _prepare_finetune(args, out: Path, manifest: dict[str, Any]) -> None: + src_train = Path(args.csv) + has_val = args.val_csv is not None + has_test = args.test_csv is not None + split_type = args.split_type + + if has_val and has_test: + # User supplied explicit val + test CSVs: trust them, just clean + featurize. + # split_type is irrelevant when val/test are given separately. + manifest["split_method"] = "user_provided" + clean_train = _clean_smiles(src_train, out / "clean_train.csv", args.smiles_column, manifest, args.force) + clean_val = _clean_smiles(Path(args.val_csv), out / "clean_val.csv", args.smiles_column, manifest, args.force) + clean_test = _clean_smiles(Path(args.test_csv), out / "clean_test.csv", args.smiles_column, manifest, args.force) + manifest["outputs"]["clean_train_csv"] = str(clean_train) + manifest["outputs"]["clean_val_csv"] = str(clean_val) + manifest["outputs"]["clean_test_csv"] = str(clean_test) + per_split = (("train", clean_train), ("val", clean_val), ("test", clean_test)) + elif has_val or has_test: + raise ValueError( + "for finetune mode, either provide BOTH --val-csv and --test-csv (user-provided splits) " + "or NEITHER (run with --split-type {random|scaffold_balanced|index_predetermined}). " + "Got one but not both." + ) + elif split_type == "random": + # Random auto-split — done here in prep so train.py gets ready-made CSVs. + manifest["split_method"] = "random" + manifest["split_seed"] = args.seed + train_frac = max(0.0, 1.0 - args.val_frac - args.test_frac) + manifest["split_fractions"] = {"train": train_frac, "val": args.val_frac, "test": args.test_frac} + clean_full = _clean_smiles(src_train, out / "_clean_full.csv", args.smiles_column, manifest, args.force) + dst = { + "train": out / "clean_train.csv", + "val": out / "clean_val.csv", + "test": out / "clean_test.csv", + } + _random_split_csv(clean_full, dst, manifest["split_fractions"], args.seed, manifest, args.force) + clean_train, clean_val, clean_test = dst["train"], dst["val"], dst["test"] + manifest["outputs"]["clean_train_csv"] = str(clean_train) + manifest["outputs"]["clean_val_csv"] = str(clean_val) + manifest["outputs"]["clean_test_csv"] = str(clean_test) + _emit_random_split_warning(src_train, manifest["split_fractions"], args.seed, manifest) + per_split = (("train", clean_train), ("val", clean_val), ("test", clean_test)) + else: + # Scaffold-balanced or index-predetermined: prep cleans + featurizes the full + # CSV and defers actual splitting to task/train.py, which calls split_data + # with the user-supplied seed and split_sizes. + manifest["split_method"] = "deferred_to_runner" + manifest["split_type"] = split_type + manifest["split_seed"] = args.seed + manifest["split_fractions"] = { + "train": max(0.0, 1.0 - args.val_frac - args.test_frac), + "val": args.val_frac, + "test": args.test_frac, + } + clean_full = _clean_smiles(src_train, out / "clean_full.csv", args.smiles_column, manifest, args.force) + manifest["outputs"]["clean_full_csv"] = str(clean_full) + per_split = (("full", clean_full),) + + if args.skip_features: + manifest["steps"].append({"name": "save_features", "skipped_by_flag": True, "ok": True}) + return + generator = args.features_generator or DEFAULT_FEATURES_GENERATOR["finetune"] + for split_name, csv in per_split: + npz = _save_features(csv, csv.with_suffix(".npz"), generator, manifest, args.force) + manifest["outputs"][f"clean_{split_name}_npz"] = str(npz) + + +def _prepare_pretrain(args, out: Path, manifest: dict[str, Any]) -> None: + src_train = Path(args.csv) + if args.val_csv is not None: + manifest["split_method"] = "user_provided" + clean_train = _clean_smiles(src_train, out / "clean_train.csv", args.smiles_column, manifest, args.force) + clean_val = _clean_smiles(Path(args.val_csv), out / "clean_val.csv", args.smiles_column, manifest, args.force) + else: + manifest["split_method"] = "random" + manifest["split_seed"] = args.seed + train_frac = max(0.0, 1.0 - args.val_frac) + manifest["split_fractions"] = {"train": train_frac, "val": args.val_frac} + clean_full = _clean_smiles(src_train, out / "_clean_full.csv", args.smiles_column, manifest, args.force) + dst = {"train": out / "clean_train.csv", "val": out / "clean_val.csv"} + _random_split_csv(clean_full, dst, manifest["split_fractions"], args.seed, manifest, args.force) + clean_train, clean_val = dst["train"], dst["val"] + + manifest["outputs"]["clean_train_csv"] = str(clean_train) + manifest["outputs"]["clean_val_csv"] = str(clean_val) + + generator = args.features_generator or DEFAULT_FEATURES_GENERATOR["pretrain"] + if args.skip_features: + manifest["steps"].append({"name": "save_features", "skipped_by_flag": True, "ok": True}) + train_npz: Path | None = None + val_npz: Path | None = None + else: + train_npz = _save_features(clean_train, out / "clean_train.npz", generator, manifest, args.force) + val_npz = _save_features(clean_val, out / "clean_val.npz", generator, manifest, args.force) + manifest["outputs"]["clean_train_npz"] = str(train_npz) + manifest["outputs"]["clean_val_npz"] = str(val_npz) + + if args.skip_vocab: + manifest["steps"].append({"name": "build_vocab", "skipped_by_flag": True, "ok": True}) + manifest["vocab_source"] = "skipped" + else: + # Resolve user-provided vocab paths from --vocab-dir or explicit flags. + provided = _resolve_vocab_inputs(args) + if provided: + # Use the user-supplied (ckpt's) vocab as-is. Copy into the + # conventional filenames the downstream pretrain command expects. + vocabs = _copy_provided_vocab(provided, out, args.dataset_name, manifest, args.force) + manifest["vocab_source"] = "user_provided" + else: + # Fall back to the existing build-from-corpus behavior. Used by + # pretrain-from-scratch and by any continue case where the user + # explicitly wants a fresh vocab (rare, usually wrong). + vocabs = _build_vocab(clean_train, out, args.dataset_name, args.vocab_format, manifest, args.force) + manifest["vocab_source"] = "built_fresh" + if "atom" in vocabs: + manifest["outputs"]["atom_vocab"] = str(vocabs["atom"]) + if "bond" in vocabs: + manifest["outputs"]["bond_vocab"] = str(vocabs["bond"]) + if "smiles" in vocabs: + manifest["outputs"]["smiles_vocab"] = str(vocabs["smiles"]) + + if args.skip_split: + manifest["steps"].append({"name": "split_data", "skipped_by_flag": True, "ok": True}) + else: + train_dir = _split_data(clean_train, train_npz, args.sample_per_file, out / "train", manifest, args.force) + val_dir = _split_data(clean_val, val_npz, args.sample_per_file, out / "val", manifest, args.force) + manifest["outputs"]["train_dir"] = str(train_dir) + manifest["outputs"]["val_dir"] = str(val_dir) + + +# --------------------------------------------------------------------------- +# Entry point +# --------------------------------------------------------------------------- + +def prepare(args: argparse.Namespace) -> dict[str, Any]: + out = Path(args.out).resolve() + out.mkdir(parents=True, exist_ok=True) + manifest: dict[str, Any] = { + "mode": args.mode, + "input_csv": str(Path(args.csv).resolve()), + "val_csv": str(Path(args.val_csv).resolve()) if args.val_csv else None, + "test_csv": str(Path(args.test_csv).resolve()) if args.test_csv else None, + "output_dir": str(out), + "split_method": None, + "steps": [], + "outputs": {}, + "errors": [], + "warnings": [], + } + try: + if args.mode == "pretrain": + _prepare_pretrain(args, out, manifest) + elif args.mode == "finetune": + _prepare_finetune(args, out, manifest) + elif args.mode == "inference": + _prepare_inference(args, out, manifest) + elif args.mode == "embed": + _prepare_embed(args, out, manifest) + manifest["ok"] = True + except Exception as exc: # noqa: BLE001 + manifest["ok"] = False + manifest["errors"].append(f"{type(exc).__name__}: {exc}") + # Always write the manifest so partial-failure state is visible to the agent. + (out / "prepare_data.json").write_text(json.dumps(manifest, indent=2)) + return manifest + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Mode-dispatched data prep for the KERMT agent skills.") + p.add_argument("--mode", required=True, choices=VALID_MODES) + p.add_argument("--csv", required=True, help="Primary input CSV (train CSV for pretrain/finetune)") + p.add_argument("--out", required=True, help="Output directory") + p.add_argument("--val-csv", default=None, help="Optional separate val CSV (pretrain/finetune)") + p.add_argument("--test-csv", default=None, help="Optional separate test CSV (finetune only)") + p.add_argument("--val-frac", type=float, default=0.1, help="Auto-split val fraction (default 0.1)") + p.add_argument("--test-frac", type=float, default=0.1, help="Auto-split test fraction (finetune only, default 0.1)") + p.add_argument("--seed", type=int, default=0, help="Random split seed (default 0)") + p.add_argument("--split-type", choices=["random", "scaffold_balanced", "index_predetermined"], + default="random", + help="(finetune only, when --val-csv/--test-csv are not given) how to split. " + "'random' splits in prep using --val-frac/--test-frac/--seed. " + "'scaffold_balanced' and 'index_predetermined' defer the actual split to the " + "runner (task/train.py invokes split_data with the appropriate algorithm " + "using the user-supplied seed); prep only cleans + featurizes the full CSV.") + p.add_argument("--sample-per-file", type=int, default=100_000, + help="split_data shard size (pretrain only, default 100000)") + p.add_argument("--vocab-format", choices=["json", "pkl"], default="json", + help="atom/bond vocab format (default json); smiles vocab is always pkl") + # Vocab pass-through (pretrain mode): when continuing from a released ckpt, + # pass its bundled vocab files in so we don't rebuild a mismatched vocab. + p.add_argument("--vocab-dir", default=None, + help="(pretrain) directory containing pretrain_{atom,bond}_vocab.{json,pkl} " + "(+ pretrain_smiles_vocab.pkl for cmim/hybrid). When given, prepare_data " + "skips build_vocab and copies these files into the output dir under the " + "expected filenames. Used by kermt-continue-pretrain to bind the released " + "ckpt's vocab to the new corpus (the ckpt's vocab is authoritative).") + p.add_argument("--atom-vocab", default=None, + help="(pretrain) explicit atom vocab path; pairs with --bond-vocab. Overrides " + "--vocab-dir's pretrain_atom_vocab.* discovery if both are given.") + p.add_argument("--bond-vocab", default=None, + help="(pretrain) explicit bond vocab path; pairs with --atom-vocab.") + p.add_argument("--smiles-vocab", default=None, + help="(pretrain, cmim/hybrid) explicit smiles vocab .pkl path. Optional for " + "vocab-only pretrain.") + p.add_argument("--dataset-name", default="pretrain", + help="vocab filename prefix (default 'pretrain' so downstream pretrain commands " + "can reference pretrain_{atom,bond}_vocab.{json|pkl}, pretrain_smiles_vocab.pkl)") + p.add_argument("--targets", nargs="+", default=None, + help="(finetune only) target column names; forwarded to the finetune runner via the manifest") + p.add_argument("--features-generator", default=None, + help="Override the per-mode default (pretrain: fgtasklabel; finetune/inference: rdkit_2d_normalized)") + p.add_argument("--smiles-column", type=int, default=None, + help="0-based column index of SMILES in the input CSV. " + "When omitted, auto-detected by header name " + "(prefers lowercase `smiles`; accepts case-insensitive " + "`SMILES`/`Smiles`). Pass explicitly to override.") + p.add_argument("--force", action="store_true", + help="Re-run every step even if its outputs already exist") + p.add_argument("--skip-clean", action="store_true", help="(embed mode) skip the cleaning step") + p.add_argument("--skip-features", action="store_true", help="Skip feature generation") + p.add_argument("--skip-vocab", action="store_true", help="(pretrain) skip vocab build") + p.add_argument("--skip-split", action="store_true", help="(pretrain) skip shard split") + args = p.parse_args(argv) + + # Forward --targets through the manifest so the finetune runner can see them. + if args.mode == "finetune" and args.targets: + pass # captured in manifest below + + # Resolve the SMILES column index (auto-detect from header when the user + # didn't pass --smiles-column). This is the only point where args.csv is + # touched before downstream _clean_smiles calls fan it out. + try: + resolved_smiles_col = _resolve_smiles_column(Path(args.csv), args.smiles_column) + except ValueError as exc: + err_manifest = { + "ok": False, + "mode": args.mode, + "errors": [f"smiles-column resolution failed: {exc}"], + } + Path(args.out).mkdir(parents=True, exist_ok=True) + (Path(args.out) / "prepare_data.json").write_text(json.dumps(err_manifest, indent=2)) + print(json.dumps(err_manifest, indent=2)) + return 1 + if args.smiles_column is None: + print(f"[prepare_data] auto-detected --smiles-column {resolved_smiles_col} " + f"from {Path(args.csv).name} header", file=sys.stderr) + args.smiles_column = resolved_smiles_col + + try: + manifest = prepare(args) + except Exception as exc: # noqa: BLE001 + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({"ok": False, "errors": [f"unhandled: {type(exc).__name__}: {exc}"]}, indent=2)) + return 1 + if args.targets: + manifest["targets"] = list(args.targets) + # Record the resolved SMILES column so the manifest is self-describing. + manifest["smiles_column"] = args.smiles_column + (Path(args.out) / "prepare_data.json").write_text(json.dumps(manifest, indent=2)) + print(json.dumps(manifest, indent=2)) + return 0 if manifest.get("ok") else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-infer/scripts/run_inference.py b/skills/kermt-infer/scripts/run_inference.py new file mode 100644 index 0000000..56575bd --- /dev/null +++ b/skills/kermt-infer/scripts/run_inference.py @@ -0,0 +1,261 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Workstation inference runner — wraps `main.py predict` (which calls +task/predict.py::make_predictions) and emits a reproducible `run.json` manifest. + +Inputs: + - finetuned ckpt (must have task FFN heads; validator refuses pretrain ckpts) + - prepare_data manifest (mode=inference) + - output dir + +`main.py predict` requires a checkpoint *directory* (--checkpoint_dir walks it +to find every .pt) rather than a single --checkpoint_path. The runner sidesteps +that by symlinking the user's ckpt as `/ckpt_link/model.pt` and passing +`--checkpoint_dir /ckpt_link` — `task/predict.py` then loads just the one +ckpt. No modification to the existing parsing.py needed. + +Blocking-by-default. Inference is minutes-scale; no detach machinery here. + +CLI +--- + run_inference.py + --ckpt # required + --prepare-manifest # prepare_data.json (mode=inference) + --out # output dir + [--ckpt-validator-out ] # cached check_checkpoint.py JSON + [--gpus 0] # single GPU id (default 0) + [--batch-size N] # override defaults_inference.runtime.batch_size + [--seed N] + [--dry-run] +""" +from __future__ import annotations + +import argparse +import datetime +import json +import os +import subprocess +import sys +from pathlib import Path +from typing import Any + +# sys.path tweak so `_utils` is importable whether launched via kermt_run +# (PYTHONPATH=/workspace) or bare-Python from the host. +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import ( # noqa: E402 + resolve_kermt_repo, assert_prepare_manifest_basics, docker_image_digest, format_cmd_replay, + git_commit_with_env_override, load_json, merge_default_into_applied, + resolve_single_gpu, run_checkpoint_validator, +) + + +REPO_ROOT = resolve_kermt_repo() +SKILL_ROOT = Path(__file__).resolve().parent.parent +DEFAULTS_PATH = SKILL_ROOT / "config" / "defaults_inference.json" +CHECK_CHECKPOINT_PATH = SKILL_ROOT / "scripts" / "check_checkpoint.py" +MAIN_PY_PATH = REPO_ROOT / "main.py" + +RUNTIME_FLAGS = ("batch_size", "seed") + + +def _verify_prepare_manifest(manifest: dict[str, Any]) -> None: + assert_prepare_manifest_basics(manifest, "inference") + out = manifest.get("outputs", {}) + if "clean_csv" not in out: + raise ValueError( + "prepare_data manifest is missing required output 'clean_csv'. " + "Was prepare_data run successfully?" + ) + + +def _apply_defaults(args: argparse.Namespace, defaults: dict[str, Any]) -> dict[str, dict[str, Any]]: + """Merges defaults_inference.json with CLI overrides into args_applied.""" + applied: dict[str, dict[str, Any]] = {} + runtime = defaults.get("runtime", {}) + + for f in RUNTIME_FLAGS: + merge_default_into_applied(applied, args, f, runtime) + + return applied + + +def _link_ckpt_into_dir(user_ckpt: Path, link_dir: Path) -> Path: + """Symlink the user ckpt into a fresh subdir so we can pass --checkpoint_dir + to main.py predict (it doesn't expose --checkpoint_path). + The dir is exclusive to this run (under out_dir), so no other .pt sneaks in.""" + link_dir.mkdir(parents=True, exist_ok=True) + # Clean prior contents (e.g. from a previous run reusing the same out dir). + for prior in link_dir.iterdir(): + if prior.is_symlink() or prior.is_file(): + prior.unlink() + link = link_dir / "model.pt" + link.symlink_to(user_ckpt.resolve()) + return link + + +def _build_argv( + *, gpu: int, out_dir: Path, manifest: dict[str, Any], ckpt_dir: Path, + output_csv: Path, applied: dict[str, dict[str, Any]], +) -> list[str]: + """Constructs the `main.py predict` argv.""" + outputs = manifest["outputs"] + argv: list[str] = [sys.executable, "-u", str(MAIN_PY_PATH), "predict"] + + argv += ["--data_path", outputs["clean_csv"]] + argv += ["--output_path", str(output_csv)] + argv += ["--checkpoint_dir", str(ckpt_dir)] + if "clean_npz" in outputs: + argv += ["--features_path", outputs["clean_npz"]] + argv += ["--gpu", str(gpu)] + + if "batch_size" in applied: + argv += ["--batch_size", str(applied["batch_size"]["value"])] + if "seed" in applied: + argv += ["--seed", str(applied["seed"]["value"])] + + return argv + + +# --------------------------------------------------------------------------- +# Main flow +# --------------------------------------------------------------------------- + +def run(args: argparse.Namespace) -> dict[str, Any]: + out_dir = Path(args.out).resolve() + out_dir.mkdir(parents=True, exist_ok=True) + (out_dir / "logs").mkdir(parents=True, exist_ok=True) + (out_dir / "out").mkdir(parents=True, exist_ok=True) + ckpt_link_dir = out_dir / "ckpt_link" + + # 1. Load defaults + prepare manifest. + defaults = load_json(DEFAULTS_PATH, name="defaults_inference.json") + prep_manifest_path = Path(args.prepare_manifest).resolve() + manifest = load_json(prep_manifest_path, name="prepare_data.json") + _verify_prepare_manifest(manifest) + + # 2. Validate ckpt — must be finetuned. + ckpt = Path(args.ckpt).resolve() + if args.ckpt_validator_out: + validator_out = load_json(Path(args.ckpt_validator_out), name="ckpt validator output") + else: + validator_out = run_checkpoint_validator(ckpt, mode="inference", script_path=CHECK_CHECKPOINT_PATH) + if not validator_out.get("ok"): + raise ValueError( + f"check_checkpoint.py rejected the input ckpt: {validator_out.get('errors')}. " + "Inference requires a finetuned ckpt with task FFN heads; for a pretrain ckpt " + "use kermt-finetune first." + ) + + model_type = validator_out.get("model_type") + arch = validator_out.get("arch", {}) + task_output_dims = validator_out.get("task_output_dims", []) + + # 3. GPU + defaults. + gpu = resolve_single_gpu(args.gpus, workflow="inference") + applied = _apply_defaults(args, defaults) + + # 4. Symlink the ckpt + build argv. + output_csv = out_dir / "out" / "predictions.csv" + if not args.dry_run: + _link_ckpt_into_dir(ckpt, ckpt_link_dir) + else: + # In dry-run we still record where the link would go for replayability. + ckpt_link_dir.mkdir(parents=True, exist_ok=True) + argv = _build_argv( + gpu=gpu, out_dir=out_dir, manifest=manifest, ckpt_dir=ckpt_link_dir, + output_csv=output_csv, applied=applied, + ) + + # 5. Build the run.json manifest. + commit, dirty = git_commit_with_env_override(REPO_ROOT) + image_tag = os.environ.get("KERMT_IMAGE", "kermt:latest") + image_digest = docker_image_digest(image_tag) + cmd_replay = format_cmd_replay(argv, env={"CUDA_VISIBLE_DEVICES": gpu}) + run_manifest: dict[str, Any] = { + "workflow": "inference", + "started_at": datetime.datetime.now(datetime.timezone.utc).isoformat(), + "container": {"image_tag": image_tag, "image_digest": image_digest}, + "repo": {"commit": commit, "dirty": dirty}, + "inputs": { + "ckpt": str(ckpt), + "prepare_data_manifest": str(prep_manifest_path), + "ckpt_validator_out": ( + str(Path(args.ckpt_validator_out).resolve()) if args.ckpt_validator_out else None + ), + }, + "model_type": model_type, + "gpu": gpu, + "args_applied": applied, + "arch": arch, + "task_output_dims": task_output_dims, + "output_csv": str(output_csv), + "logs_dir": str(out_dir / "logs"), + "ckpt_link_dir": str(ckpt_link_dir), + "argv": argv, + "cmd_replay": cmd_replay, + "ok_to_replay": (not dirty) and (commit != "unknown"), + "dry_run": bool(args.dry_run), + } + (out_dir / "run.json").write_text(json.dumps(run_manifest, indent=2)) + + if args.dry_run: + run_manifest["status"] = "dry_run" + return run_manifest + + # 6. Execute. + env = os.environ.copy() + env["CUDA_VISIBLE_DEVICES"] = str(gpu) + # main.py enables strict deterministic algorithms via + # `torch.use_deterministic_algorithms(True)` (kermt/main.py:23); CuBLAS + # then requires this env var to be set. Default to ':4096:8' (slightly + # more memory than ':16:8' but allows larger matmuls). + env.setdefault("CUBLAS_WORKSPACE_CONFIG", ":4096:8") + log_file = out_dir / "logs" / "inference.log" + with log_file.open("w") as logf: + proc = subprocess.run(argv, env=env, stdout=logf, stderr=subprocess.STDOUT) + run_manifest["exit_code"] = proc.returncode + run_manifest["status"] = "ok" if proc.returncode == 0 else "failed" + (out_dir / "run.json").write_text(json.dumps(run_manifest, indent=2)) + return run_manifest + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser( + description="Workstation inference runner — wraps main.py predict via subprocess." + ) + p.add_argument("--ckpt", required=True, help="Path to a finetuned ckpt (with task FFN heads).") + p.add_argument("--prepare-manifest", required=True, + help="Path to a prepare_data.json produced with --mode inference.") + p.add_argument("--out", required=True, help="Output run directory.") + p.add_argument("--ckpt-validator-out", default=None, + help="Optional cached check_checkpoint.py JSON.") + p.add_argument("--gpus", default=None, help="Single GPU id (default 0). Multi-GPU rejected.") + p.add_argument("--dry-run", action="store_true", + help="Write run.json + print the command without executing.") + + # Runtime overrides (defaults None so source attribution stays accurate). + p.add_argument("--batch-size", type=int, default=None) + p.add_argument("--seed", type=int, default=None) + args = p.parse_args(argv) + + try: + manifest = run(args) + except (FileNotFoundError, ValueError, RuntimeError) as exc: + print(json.dumps({"ok": False, "errors": [f"{type(exc).__name__}: {exc}"]}, indent=2)) + return 1 + except Exception as exc: # noqa: BLE001 + import traceback + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({"ok": False, "errors": [f"unhandled: {type(exc).__name__}: {exc}"]}, + indent=2)) + return 1 + + print(json.dumps({"ok": True, "manifest": manifest}, indent=2)) + return 0 if manifest.get("status") != "failed" else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/agent/skills/kermt-infer/skill-card.md b/skills/kermt-infer/skill-card.md similarity index 100% rename from agent/skills/kermt-infer/skill-card.md rename to skills/kermt-infer/skill-card.md diff --git a/agent/skills/kermt-monitor/SKILL.md b/skills/kermt-monitor/SKILL.md similarity index 100% rename from agent/skills/kermt-monitor/SKILL.md rename to skills/kermt-monitor/SKILL.md diff --git a/agent/skills/kermt-monitor/evals/evals.json b/skills/kermt-monitor/evals/evals.json similarity index 100% rename from agent/skills/kermt-monitor/evals/evals.json rename to skills/kermt-monitor/evals/evals.json diff --git a/agent/skills/kermt-monitor/skill-card.md b/skills/kermt-monitor/skill-card.md similarity index 97% rename from agent/skills/kermt-monitor/skill-card.md rename to skills/kermt-monitor/skill-card.md index db5b652..9f54dc1 100644 --- a/agent/skills/kermt-monitor/skill-card.md +++ b/skills/kermt-monitor/skill-card.md @@ -39,7 +39,6 @@ Mitigation: The skill reports container identifiers so the user can terminate th ## Reference(s):
- [KERMT repository](https://github.com/NVIDIA-BioNeMo/KERMT)
-- `agent/scripts/kermt_container.sh` — defines `kermt_run_detached`, the wrapper whose runs this skill observes
- Related skills: `kermt-pretrain-scratch`, `kermt-continue-pretrain`, `kermt-finetune`
## Skill Output:
diff --git a/agent/skills/kermt-pretrain-scratch/SKILL.md b/skills/kermt-pretrain-scratch/SKILL.md similarity index 89% rename from agent/skills/kermt-pretrain-scratch/SKILL.md rename to skills/kermt-pretrain-scratch/SKILL.md index c0b3509..7da45cd 100644 --- a/agent/skills/kermt-pretrain-scratch/SKILL.md +++ b/skills/kermt-pretrain-scratch/SKILL.md @@ -20,6 +20,14 @@ extending one of the released checkpoints. **Significantly more expensive than `kermt-continue-pretrain`** — no warm start, so the loss curves need to descend from scratch over many epochs. +## Skill and runtime paths + +Set `SKILL_DIR` to the directory containing this `SKILL.md`. Export +`KERMT_REPO` as the absolute path to the KERMT checkout used for model +execution. The bundled container helper mounts that checkout at +`/workspace` and this skill at `/skill` (read-only). Commands inside +the container use `/skill/scripts/`; defaults are bundled in `config/`. + ## Hardware requirements Same as `kermt-continue-pretrain`: @@ -83,7 +91,7 @@ Optional: - Training-hyperparameter overrides: `--epochs N` / `--batch-size N` / `--init-lr F` / `--max-lr F` / `--final-lr F` / `--warmup-epochs F` / `--weight-decay F` / `--dropout F` / `--save-interval N` / `--seed N`. - Anything not given is filled from `agent/config/defaults_pretrain.json`. + Anything not given is filled from `config/defaults_pretrain.json`. - `--vocab-loss-weight F` (hybrid only) / `--latent-dim N` / `--contrastive-temperature F` (cmim and hybrid only). - `--wandb-project NAME` / `--wandb-run-name NAME` — optional Weights & Biases @@ -107,16 +115,16 @@ Let `$KERMT_REPO` be the path to your kermt repo checkout. 3. **Validate the corpus** (no ckpt to validate, so this is the only input check): ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run --data -- \ - "python agent/scripts/check_data.py --mode pretrain --csv /data/" + "$SKILL_DIR/scripts/kermt_container.sh" run --data -- \ + "python /skill/scripts/check_data.py --mode pretrain --csv /data/" ``` Abort on `ok: false`. 4. **Prepare the data** — no vocab pass-through (we want fresh vocab from corpus): ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run --data --run-dir $RUN_DIR -- \ - "python agent/scripts/prepare_data.py --mode pretrain \\ + "$SKILL_DIR/scripts/kermt_container.sh" run --data --run-dir $RUN_DIR -- \ + "python /skill/scripts/prepare_data.py --mode pretrain \\ --csv /data/ --out /runs/data \\ [--val-csv /data/] [--val-frac 0.1] [--seed 0]" ``` @@ -134,10 +142,10 @@ Let `$KERMT_REPO` be the path to your kermt repo checkout. 6. **Launch the runner detached.** ``` - $KERMT_REPO/agent/scripts/kermt_container.sh run_detached \\ + "$SKILL_DIR/scripts/kermt_container.sh" run_detached \\ --name kermt-pretrain-scratch- \\ --run-dir $RUN_DIR -- \\ - "python agent/scripts/run_pretrain_local.py \\ + "python /skill/scripts/run_pretrain_local.py \\ --from-scratch --pretrain-target-mode \\ --prepare-manifest /runs/data/prepare_data.json \\ --out /runs \\ @@ -145,7 +153,7 @@ Let `$KERMT_REPO` be the path to your kermt repo checkout. ``` Note: NO `--ckpt` flag (the runner refuses if both `--from-scratch` and `--ckpt` are given). The runner uses the `arch` group from - `agent/config/defaults_pretrain.json` to size the model. + `config/defaults_pretrain.json` to size the model. 7. **Report to the user.** Always include all of the following — do not omit the TensorBoard line under output-length pressure: @@ -188,7 +196,7 @@ Same reproducibility fields as continue-pretrain (`repo.commit`, `kermt_image`, - `ckpt_symlink`: `null` - `vocab_check`: `null` (not verified — vocab built from corpus is authoritative for from-scratch) -- `arch`: the values pulled from `agent/config/defaults_pretrain.json`'s +- `arch`: the values pulled from `config/defaults_pretrain.json`'s `arch` group (with any future CLI overrides applied). ## Replayability diff --git a/skills/kermt-pretrain-scratch/config/defaults_pretrain.json b/skills/kermt-pretrain-scratch/config/defaults_pretrain.json new file mode 100644 index 0000000..dbb53e7 --- /dev/null +++ b/skills/kermt-pretrain-scratch/config/defaults_pretrain.json @@ -0,0 +1,52 @@ +{ + "_about": "Default hyperparameters applied by kermt-continue-pretrain and kermt-add-cmim-pretrain. Values target a workstation-scale hybrid pretrain. The skill echoes the applied set back to the user on every invocation; override any value with the corresponding CLI flag.", + + "training": { + "_about": "Optimizer and training schedule. Apply to all pretrain workflows.", + "batch_size": 256, + "dropout": 0.1, + "epochs": 30, + "init_lr": 1e-5, + "max_lr": 1.5e-4, + "final_lr": 1e-5, + "warmup_epochs": 20, + "weight_decay": 1e-7, + "save_interval": 100, + "seed": 0, + "tensorboard": true, + "use_cuikmolmaker_featurization": true + }, + + "loss": { + "_about": "Loss-weighting knobs. contrastive_temperature applies only when the model has a contrast head (hybrid). vocab_loss_weight applies when the model has a vocab head (cmim or hybrid). The runner detects the model type from the checkpoint and ignores irrelevant entries.", + "contrastive_temperature": 0.1, + "vocab_loss_weight": 1.0 + }, + + "add_cmim_decoder": { + "_about": "Used by kermt-add-cmim-pretrain when constructing the new cMIM decoder + latent_dist on top of a loaded grover-base encoder, and by kermt-pretrain-scratch when the pretrain target is cmim or hybrid. Ignored by kermt-continue-pretrain (those dimensions come from the ckpt's saved_args). Values match the manuscript's hybrid pretrain configuration: latent_dim=512, 8-head, 3-layer decoder (cf. `_PRESET_LATENT_DIM` / `_PRESET_DECODER_FFN_HIDDEN_SIZE` in launch-KERMT-pretrain-slurm.sh, both presets).", + "latent_dim": 512, + "contrastive_temperature": 0.1, + "decoder_num_layers": 3, + "decoder_num_attention_heads": 8, + "decoder_ffn_hidden_size": 2048, + "decoder_dropout": 0.1, + "decoder_max_seq_len": 512, + "decoder_positional_encoding": "rope", + "decoder_gate_self_attn": false, + "decoder_gate_cross_attn": false + }, + + "arch": { + "_about": "Encoder architecture defaults — used ONLY by kermt-pretrain-scratch (fresh model from corpus, no starting ckpt). kermt-continue-pretrain and kermt-add-cmim-pretrain ignore this block and pull arch from the loaded checkpoint instead; the runner aborts if user-supplied arch flags mismatch the ckpt's saved_args.", + "hidden_size": 800, + "depth": 6, + "num_attn_head": 4, + "activation": "PReLU", + "backbone": "gtrans", + "embedding_output_type": "both", + "self_attention": false + }, + + "_about_gpu_selection": "GPU selection is auto-detected at runtime, not a default here. The pretrain runner uses torch.cuda.device_count() and dispatches single-GPU or DDP accordingly. Override with --gpus 0,2 if you want a specific subset." +} diff --git a/agent/skills/kermt-pretrain-scratch/evals/evals.json b/skills/kermt-pretrain-scratch/evals/evals.json similarity index 100% rename from agent/skills/kermt-pretrain-scratch/evals/evals.json rename to skills/kermt-pretrain-scratch/evals/evals.json diff --git a/skills/kermt-pretrain-scratch/scripts/_utils.py b/skills/kermt-pretrain-scratch/scripts/_utils.py new file mode 100644 index 0000000..5bde460 --- /dev/null +++ b/skills/kermt-pretrain-scratch/scripts/_utils.py @@ -0,0 +1,272 @@ +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Shared utilities for the agent scripts. + +Kept intentionally small — only logic that appears (or would otherwise be +duplicated) in two or more `scripts/*.py` modules. Each script +maintains its own primary CLI + main flow. +""" +from __future__ import annotations + +import argparse +import json +import os +import shlex +import subprocess +import sys +from pathlib import Path +from typing import Any + + +# Conventional pretrain vocab filename stems. Used by prepare_data.py + +# upgrade_to_hybrid.py + the README "Released models" bundling docs + +# the test helpers. Centralized here so a future rename only touches one +# spot. +PRETRAIN_VOCAB_STEMS = { + "atom": "pretrain_atom_vocab", + "bond": "pretrain_bond_vocab", + "smiles": "pretrain_smiles_vocab", +} + + +def resolve_kermt_repo() -> Path: + """Find the runtime checkout independently of the installed skill location. + + An explicit KERMT_REPO takes precedence. In a repository checkout, walking + up from this helper or the working directory also supports local use. + """ + explicit = os.environ.get("KERMT_REPO") + if explicit: + candidates = [Path(explicit).expanduser().resolve()] + else: + candidates = [] + for start in (Path(__file__).resolve().parent, Path.cwd()): + candidates.extend((start, *start.parents)) + for candidate in candidates: + if (candidate / "main.py").is_file() and (candidate / "kermt").is_dir(): + return candidate + raise FileNotFoundError( + "KERMT checkout not found. Set KERMT_REPO to the checkout containing " + "main.py and kermt/; the installed skill directory is separate." + ) + + +def load_json(path: Path, *, name: str) -> dict[str, Any]: + """Load a JSON file with consistent error messages. + + `name` is a human-readable label for the document (e.g. "prepare_data.json") + so the error tells the user which schema we expected at that path. + """ + if not path.is_file(): + raise FileNotFoundError(f"{name} not found at {path}") + try: + return json.loads(path.read_text()) + except json.JSONDecodeError as exc: + raise ValueError(f"{name} at {path} is not valid JSON: {exc}") from exc + + +def count_vocab_entries(vocab_path: Path) -> int: + """Return the number of entries in a KERMT vocab file. + + Handles three layouts: + - JSON with `{stoi: {token: idx}, ...}` (MolVocab.save_vocab default) + - JSON as a raw `{token: idx}` dict (legacy / hand-edited) + - Pickle of a MolVocab / SMILESVocab object (the smiles vocab is always + pickled because its compiled-regex tokenizer state isn't + JSON-serializable). Falls through to raw `pickle.load` if the + MolVocab / SMILESVocab loader can't import or fails to recognize + the contents (e.g. test fixtures with plain dicts). + """ + if vocab_path.suffix == ".json": + data = json.loads(vocab_path.read_text()) + if isinstance(data, dict) and "stoi" in data: + return len(data["stoi"]) + if isinstance(data, dict): + return len(data) + raise ValueError(f"unsupported JSON vocab shape at {vocab_path}: {type(data).__name__}") + + # .pkl: try MolVocab / SMILESVocab first, then fall back to raw pickle. + try: + from kermt.data.torchvocab import MolVocab, SMILESVocab # type: ignore + for loader in (MolVocab.load_vocab, SMILESVocab.load_vocab): + try: + v = loader(str(vocab_path)) + return len(v) + except Exception: + continue + except ImportError: + pass + + import pickle + with vocab_path.open("rb") as f: + data = pickle.load(f) + if hasattr(data, "stoi"): + return len(data.stoi) + if hasattr(data, "__len__"): + return len(data) + raise ValueError(f"could not count entries in {vocab_path}") + + +def validate_vocab_file(vocab_path: Path, *, kind: str) -> None: + """Verify a user-provided vocab file is loadable BEFORE copying it into a + run directory. Raises ValueError on failure with a clear, user-facing message. + + `kind` is one of {"atom", "bond", "smiles"} — used only in the error message + so the user knows which file is wrong. + """ + if not vocab_path.is_file(): + raise FileNotFoundError(f"{kind} vocab file not found: {vocab_path}") + try: + n = count_vocab_entries(vocab_path) + except Exception as exc: # noqa: BLE001 + raise ValueError( + f"{kind} vocab file {vocab_path} is not loadable as a KERMT vocab " + f"({type(exc).__name__}: {exc}). Expected a MolVocab JSON or pickle " + f"(or a SMILESVocab pickle for the smiles vocab)." + ) from exc + if n <= 0: + raise ValueError(f"{kind} vocab file {vocab_path} contains zero entries") + + +# --------------------------------------------------------------------------- +# Runner-shared helpers (run.json manifest fields) +# --------------------------------------------------------------------------- + +def git_commit_with_env_override(repo: Path) -> tuple[str, bool]: + """Returns (commit_sha, dirty_tree). Honors `KERMT_REPO_COMMIT` / + `KERMT_REPO_DIRTY` env vars first — set by `scripts/kermt_container.sh` + from the host before launching docker (necessary because `git -C /workspace` + inside the container fails due to bind-mount ownership). Falls back to the + in-container git probe when the env vars aren't set.""" + env_commit = os.environ.get("KERMT_REPO_COMMIT") + if env_commit: + env_dirty = os.environ.get("KERMT_REPO_DIRTY", "false").strip().lower() == "true" + return env_commit, env_dirty + try: + sha = subprocess.run( + ["git", "-C", str(repo), "rev-parse", "HEAD"], + capture_output=True, text=True, check=True, + ).stdout.strip() + diff = subprocess.run( + ["git", "-C", str(repo), "status", "--porcelain"], + capture_output=True, text=True, check=True, + ) + return sha, bool(diff.stdout.strip()) + except Exception: + return "unknown", False + + +def docker_image_digest(tag: str) -> str | None: + """Return the docker image's content-addressable Id (sha256:…) for the given + tag, or None if docker isn't available / the image isn't local.""" + try: + r = subprocess.run( + ["docker", "image", "inspect", tag, "--format", "{{.Id}}"], + capture_output=True, text=True, + ) + if r.returncode == 0: + return r.stdout.strip() + except FileNotFoundError: + pass + return None + + +def format_cmd_replay(argv: list[str], *, env: dict[str, str] | None = None) -> str: + """Render a copy-pasteable env-prefix + command for the cmd_replay manifest + field. `env` is the set of environment variables to prefix (typically + {CUDA_VISIBLE_DEVICES, WORLD_SIZE}).""" + env = env or {} + env_prefix = [f"{k}={shlex.quote(str(v))}" for k, v in env.items()] + quoted = " ".join(shlex.quote(a) for a in argv) + return " ".join(env_prefix + [quoted]) + + +def resolve_single_gpu(override: str | None, *, workflow: str) -> int: + """Returns a single GPU id (int). The finetune/inference/embed workflows are + single-GPU only; `--gpus '0,1'` or multi-id CUDA_VISIBLE_DEVICES is rejected + with a workflow-specific error. (The pretrain runner has its own multi-GPU + `_detect_gpus` helper — see run_pretrain_local.py.)""" + if override is None: + env_visible = os.environ.get("CUDA_VISIBLE_DEVICES", "").strip() + if env_visible: + ids = [g for g in env_visible.split(",") if g] + if len(ids) > 1: + raise ValueError( + f"CUDA_VISIBLE_DEVICES='{env_visible}' selects multiple GPUs but " + f"the {workflow} workflow is single-GPU only. Restrict to one id." + ) + return int(ids[0]) + return 0 + parts = [p.strip() for p in override.split(",") if p.strip()] + if len(parts) != 1: + raise ValueError( + f"--gpus '{override}' selects {len(parts)} GPUs; the {workflow} workflow is single-GPU only." + ) + return int(parts[0]) + + +def assert_prepare_manifest_basics(manifest: dict[str, Any], expected_mode: str) -> None: + """Standard pre-check for a prepare_data.json before a runner consumes it: + verify `mode` matches and `ok` is True. Raises ValueError with a consistent + error message on either mismatch. + + Each runner is responsible for its own required-outputs check after this + (those vary per-mode — e.g. pretrain wants train_dir/val_dir/atom_vocab/ + bond_vocab; finetune has the split-method branch; inference/embed want + clean_csv).""" + if manifest.get("mode") != expected_mode: + raise ValueError( + f"prepare_data manifest is mode='{manifest.get('mode')}', expected '{expected_mode}'. " + f"Run `prepare_data.py --mode {expected_mode}` to produce a valid manifest." + ) + if not manifest.get("ok"): + raise ValueError( + f"prepare_data manifest reports ok=False: {manifest.get('errors')}" + ) + + +def merge_default_into_applied( + applied: dict[str, dict[str, Any]], + args: argparse.Namespace, + name: str, + defaults_group: dict[str, Any], +) -> None: + """Standard CLI-override / default-config merge for one hyperparameter. + + Mutates `applied` in place: + - If the user passed `--` on the CLI (so `getattr(args, name)` is + not None), records `{"value": cli_val, "source": "user"}`. + - Else if `name` is present in `defaults_group`, records + `{"value": defaults_group[name], "source": "default-config"}`. + - Else `applied[name]` is left absent — the runner's argv-builder skips + the flag, and the downstream argparse default takes effect. + + `name` is the snake_case argparse dest (same form used as the dict key); + argparse automatically converts CLI `--` to that dest, + so `getattr(args, name, None)` is the correct CLI lookup.""" + cli_val = getattr(args, name, None) + if cli_val is not None: + applied[name] = {"value": cli_val, "source": "user"} + elif name in defaults_group: + applied[name] = {"value": defaults_group[name], "source": "default-config"} + + +def run_checkpoint_validator(ckpt: Path, *, mode: str, script_path: Path) -> dict[str, Any]: + """Invoke `check_checkpoint.py --mode --ckpt ` as a subprocess + and return the parsed JSON. Raises RuntimeError on non-JSON output (e.g. the + validator crashed before printing). `script_path` is the absolute path to + `scripts/check_checkpoint.py` — passed in so this helper has no + dependency on the caller's layout.""" + r = subprocess.run( + [sys.executable, str(script_path), "--mode", mode, "--ckpt", str(ckpt)], + capture_output=True, text=True, + ) + try: + return json.loads(r.stdout) + except json.JSONDecodeError as exc: + raise RuntimeError( + f"check_checkpoint.py emitted non-JSON output (exit {r.returncode}). " + f"stdout (first 200 chars): {r.stdout[:200]}\n" + f"stderr (first 200 chars): {r.stderr[:200]}" + ) from exc diff --git a/skills/kermt-pretrain-scratch/scripts/check_checkpoint.py b/skills/kermt-pretrain-scratch/scripts/check_checkpoint.py new file mode 100644 index 0000000..fd488e2 --- /dev/null +++ b/skills/kermt-pretrain-scratch/scripts/check_checkpoint.py @@ -0,0 +1,480 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Validate a KERMT checkpoint for a given agent workflow. + +Mode-dispatched. Emits a single JSON object to stdout that the calling skill +parses to decide whether to proceed (`ok: true`) or surface a structured error +back to the user (`ok: false` with `errors[]`). The JSON shape is stable +across modes; only the contract for what counts as `ok` differs per mode. + +Modes +----- +continue_pretrain Continuing pretraining from an existing pretrain ckpt. + Requires encoder + at least one pretrain head + (vocab_head for grover_base / cmim, or contrast_head for + cmim / hybrid). Rejects encoder-only or finetuned ckpts. + +upgrade_to_hybrid Adding a cMIM decoder onto a grover_base ckpt to convert + it to a hybrid pretrain. Requires encoder; rejects ckpts + that already carry a contrast_head or task_ffn (would be + workflow 4 instead). + +finetune_init Starting a finetune from a pretrained ckpt. Requires + encoder. Pretrain heads (vocab / contrast) are tolerated + but unused. Already-finetuned ckpts (task FFN heads + present) are REJECTED — finetune-on-finetune via the + agent skill isn't supported because saved-task + identity can't be machine-verified against the new + training data. + +inference Running predictions with a previously-finetuned ckpt. + Requires encoder + task_ffn. Reports task_output_dims + so the runner can compare against the user's task spec. + +embed Extracting embeddings. Requires encoder only. Anything + additional in the ckpt is ignored. + +Output (stdout) +--------------- +{ + "ok": true | false, + "model_type": "grover_base" | "cmim" | "hybrid" | "finetuned" | "unknown", + "has_encoder": bool, + "has_vocab_head": bool, + "has_contrast_head": bool, + "has_task_ffn": bool, + "task_output_dims": [int, ...], // empty unless has_task_ffn + "arch": { // ckpt-derived; runner uses these, ignores defaults_*.json arch + "hidden_size": int | null, + "depth": int | null, + "num_attn_head": int | null, + "latent_dim": int | null, + "activation": str | null, + "backbone": str | null, + "embedding_output_type": str | null, + "self_attention": bool | null + }, + "saved_args": { ... } | null, // raw args dict if present, else null + "errors": [str, ...], // mode-contract violations / load failures + "warnings": [str, ...] // non-fatal observations (e.g. arch fallback) +} + +Exit code: 0 on `ok: true`, 1 on `ok: false`. Loader exceptions are caught and +surfaced into `errors[]` with `ok: false` (still exit 1), never raised. + +CLI +--- + check_checkpoint.py --mode --ckpt +""" +from __future__ import annotations + +import argparse +import json +import sys +import traceback +from argparse import Namespace +from typing import Any + +import torch + + +# --------------------------------------------------------------------------- +# State-dict key prefix conventions (kermt/model/models.py). +# --------------------------------------------------------------------------- + +# Encoder weights appear under one of these prefixes depending on the ckpt's +# era and task class: +# - `grover.*` : legacy grover_base ckpts (predate the cMIM rename) +# - `kermt.*` : current grover_base / hybrid / finetune ckpts +# - `latent_dist.kermt.*`: cmim ckpts (encoder lives only inside latent_dist) +ENCODER_PREFIXES = ("kermt.", "grover.", "latent_dist.kermt.") +VOCAB_HEAD_PREFIX = "vocab_module." +CONTRAST_DECODER_PREFIX = "decoder." # SMILES transformer decoder, cmim/hybrid only +LATENT_DIST_PREFIX = "latent_dist." # cmim/hybrid; encoder may share via latent_dist.kermt.* +TASK_FFN_PREFIXES = ( + "mol_atom_from_atom_ffn.", + "mol_atom_from_bond_ffn.", +) +TASK_FFN_TASK_SPECIFIC_PREFIXES = ( + "mol_atom_from_atom_ffn_task_specific.", + "mol_atom_from_bond_ffn_task_specific.", +) + + +ARCH_KEYS = ( + "hidden_size", + "depth", + "num_attn_head", + "latent_dim", + "activation", + "backbone", + "embedding_output_type", + "self_attention", +) + + +def _strip_ddp_prefix(state_dict: dict[str, Any]) -> dict[str, Any]: + """Strip `module.` prefix from every key if the dict is DDP-wrapped.""" + if state_dict and all(k.startswith("module.") for k in state_dict): + return {k[len("module."):]: v for k, v in state_dict.items()} + return state_dict + + +def _classify_model(state_dict: dict[str, Any]) -> dict[str, Any]: + keys = list(state_dict.keys()) + has_encoder = any(k.startswith(ENCODER_PREFIXES) for k in keys) + has_vocab_head = any(k.startswith(VOCAB_HEAD_PREFIX) for k in keys) + has_contrast_head = any(k.startswith(CONTRAST_DECODER_PREFIX) for k in keys) + has_task_ffn = any(k.startswith(TASK_FFN_PREFIXES) for k in keys) + + if has_encoder and has_task_ffn: + model_type = "finetuned" + elif has_encoder and has_contrast_head and has_vocab_head: + model_type = "hybrid" + elif has_encoder and has_contrast_head and not has_vocab_head: + model_type = "cmim" + elif has_encoder and not has_contrast_head: + # Includes: + # - modern repo-trained Grover base (kermt.* + vocab_module.*) + # - legacy original-Grover base (grover.encoders.* with no heads saved) + # - any encoder-stripped ckpt extracted from a larger model + # The `has_vocab_head` flag discriminates the sub-cases for skills that + # need it. The continue_pretrain mode contract relies on this — a + # grover_base with vocab heads can continue, an encoder-only one cannot. + model_type = "grover_base" + else: + model_type = "unknown" + + return { + "model_type": model_type, + "has_encoder": has_encoder, + "has_vocab_head": has_vocab_head, + "has_contrast_head": has_contrast_head, + "has_task_ffn": has_task_ffn, + } + + +def _vocab_sizes(state_dict: dict[str, Any]) -> dict[str, Any]: + """Extract vocab head sizes from state-dict weight shapes. + + The pretrain heads have the following layout per kermt/model/models.py: + - Atom vocab predictors: vocab_module.av_task_atom.* + vocab_module.av_task_bond.* + (two readout streams sharing the same vocab_size). Output dim of each + final-Linear is the atom vocab size. + - Bond vocab predictors: vocab_module.bv_task_atom.* + vocab_module.bv_task_bond.* + Output dim is the bond vocab size. + - SMILES vocab decoder: decoder.output_projection.weight (cmim / hybrid only). + Output dim is the smiles vocab size. + + Returns {atom: int|None, bond: int|None, smiles: int|None}. Each is None + when the corresponding head isn't present in the ckpt (e.g. legacy + encoder-only grover_base has none; cmim has smiles but not atom/bond). + """ + sizes: dict[str, Any] = {"atom": None, "bond": None, "smiles": None} + + def _head_out_dim(prefix: str) -> int | None: + # Pick the highest-numbered 2-D Linear weight under `prefix.*` — that's + # the final output layer. + candidates = [ + k for k in state_dict + if k.startswith(prefix) and k.endswith(".weight") + and hasattr(state_dict[k], "ndim") and state_dict[k].ndim == 2 + ] + if not candidates: + return None + def _layer_index(k: str) -> int: + # ".weight" -> "..weight"; pick the rightmost numeric component. + parts = k.split(".") + for tok in reversed(parts[:-1]): + if tok.isdigit(): + return int(tok) + return -1 + final = max(candidates, key=_layer_index) + return int(state_dict[final].shape[0]) + + sizes["atom"] = _head_out_dim("vocab_module.av_task_atom.") + sizes["bond"] = _head_out_dim("vocab_module.bv_task_atom.") + sizes["smiles"] = _head_out_dim("decoder.output_projection.") + # If the decoder's output_projection isn't a Linear (e.g. some saves wrap + # it differently), fall back to a search over decoder.* heads. + if sizes["smiles"] is None: + sizes["smiles"] = _head_out_dim("decoder.token_embedding.") + return sizes + + +def _task_output_dims(state_dict: dict[str, Any]) -> list[int]: + """Return one entry per (logical task × readout) head's final-Linear out-dim. + + Two layouts: + - **MTL** (`mol_atom_from_atom_ffn_task_specific..*`): one entry per + task-specific head's final-Linear out-dim. Typically `[1, 1, ..., 1]` + for regression with N tasks across 2 readouts. + - **Non-MTL** (`mol_atom_from_atom_ffn.*` only): one entry per shared FFN's + final-Linear out-dim. Typically `[num_tasks, num_tasks]` (one per readout). + + When both layouts coexist in the same ckpt (MTL configuration: shared FFN + feeds task-specific heads), only the task-specific dims are reported — the + shared FFN there is an intermediate layer, not the model output. + """ + has_task_specific = any(k.startswith(TASK_FFN_TASK_SPECIFIC_PREFIXES) for k in state_dict) + + heads: dict[str, list[str]] = {} + for k in state_dict: + if k.startswith(TASK_FFN_TASK_SPECIFIC_PREFIXES): + parts = k.split(".") + root = ".".join(parts[:2]) # e.g. "mol_atom_from_atom_ffn_task_specific.0" + heads.setdefault(root, []).append(k) + elif k.startswith(TASK_FFN_PREFIXES) and not k.startswith(TASK_FFN_TASK_SPECIFIC_PREFIXES): + if has_task_specific: + continue # shared FFN is intermediate when task-specific heads exist + root = k.split(".")[0] # e.g. "mol_atom_from_atom_ffn" + heads.setdefault(root, []).append(k) + + dims: list[int] = [] + for root in sorted(heads): + weight_keys = sorted( + (k for k in heads[root] if k.endswith(".weight") + and hasattr(state_dict[k], "ndim") and state_dict[k].ndim == 2), + key=lambda k: int(k.split(".")[-2]) if k.split(".")[-2].isdigit() else -1, + ) + if weight_keys: + dims.append(int(state_dict[weight_keys[-1]].shape[0])) + return dims + + +def _arch_from_args(args_obj: Any) -> dict[str, Any]: + """Pull arch params from the saved args Namespace / dict, leaving missing keys as None.""" + arch: dict[str, Any] = {k: None for k in ARCH_KEYS} + if args_obj is None: + return arch + # args_obj is typically argparse.Namespace; tolerate dict form too. + args_dict = vars(args_obj) if isinstance(args_obj, Namespace) else dict(args_obj) if isinstance(args_obj, dict) else {} + for k in ARCH_KEYS: + if k in args_dict: + arch[k] = args_dict[k] + return arch + + +def _arch_from_shapes(state_dict: dict[str, Any], arch: dict[str, Any]) -> tuple[dict[str, Any], list[str]]: + """Fill in still-missing arch params by introspecting state-dict tensor shapes. + + Only fills entries that are currently None — does not override anything pulled + from saved_args. Returns the updated arch + a list of warnings for any key that + could not be inferred. + """ + warnings: list[str] = [] + + if arch["hidden_size"] is None: + # First 2-D linear weight under any encoder prefix. + candidates = [ + k for k in state_dict + if k.startswith(ENCODER_PREFIXES) + and k.endswith(".weight") + and hasattr(state_dict[k], "ndim") + and state_dict[k].ndim == 2 + ] + if candidates: + arch["hidden_size"] = int(state_dict[candidates[0]].shape[0]) + else: + warnings.append("hidden_size could not be inferred from state_dict shapes") + + if arch["latent_dim"] is None: + # Look for a Linear inside latent_dist that's not the shared encoder. + candidates = [ + k for k in state_dict + if k.startswith(LATENT_DIST_PREFIX) + and not k.startswith("latent_dist.kermt.") + and k.endswith(".weight") + and state_dict[k].ndim == 2 + ] + if candidates: + arch["latent_dim"] = int(state_dict[candidates[0]].shape[0]) + # Absent latent_dist entirely is a fact about the model, not a problem — no warning. + + # depth, num_attn_head, activation, backbone, embedding_output_type, self_attention + # are not robustly inferable from shapes alone; report a warning for each that's + # still None so the caller can prompt the user or refuse to proceed. + for k in ("depth", "num_attn_head", "activation", "backbone", "embedding_output_type", "self_attention"): + if arch[k] is None: + warnings.append(f"{k} not present in saved_args and cannot be inferred from state_dict shapes") + + return arch, warnings + + +def _apply_mode_contract(mode: str, classification: dict[str, Any]) -> list[str]: + """Return a list of error messages if `classification` violates the mode contract.""" + errors: list[str] = [] + mt = classification["model_type"] + has_enc = classification["has_encoder"] + has_vocab = classification["has_vocab_head"] + has_contrast = classification["has_contrast_head"] + has_ffn = classification["has_task_ffn"] + + if not has_enc: + errors.append("checkpoint has no encoder weights — cannot use it for any KERMT workflow") + return errors + + if mode == "continue_pretrain": + if not (has_vocab or has_contrast): + errors.append( + f"continue_pretrain requires the ckpt to still carry pretrain heads (vocab " + f"and/or contrast), but this ckpt has neither (model_type='{mt}', " + f"has_vocab_head=False, has_contrast_head=False). Either provide a ckpt with " + f"its pretrain heads attached, or convert this encoder-only ckpt to a hybrid " + f"via mode 'upgrade_to_hybrid'." + ) + if has_ffn: + errors.append( + "continue_pretrain expects a pretrain ckpt; this ckpt has task FFN heads " + "(it has been finetuned). Use a pretrain checkpoint — finetune+continue is " + "not a supported workflow." + ) + elif mode == "upgrade_to_hybrid": + if has_contrast: + errors.append( + f"upgrade_to_hybrid converts grover_base -> hybrid by adding a cMIM decoder. " + f"This ckpt already has a contrast head (classified as '{mt}'). " + f"To continue pretraining it, use mode 'continue_pretrain'." + ) + if has_ffn: + errors.append("upgrade_to_hybrid does not support finetuned checkpoints.") + elif mode == "finetune_init": + # Requires an encoder. Pretrain heads (vocab / contrast) are unused + # at finetune time but harmless. Task FFN heads (i.e. an already- + # finetuned ckpt) are NOT accepted — finetune-on-finetune isn't + # supported by the kermt-finetune skill because the saved-task + # identity can't be machine-verified against the new training data + # (dimension match doesn't prove target identity, dataset identity, + # or absence of train/test contamination). + if has_ffn: + errors.append( + f"finetune_init requires a pretrain ckpt (grover_base / cmim / hybrid); " + f"this ckpt is classified as '{mt}' with task FFN heads attached. " + f"To resume a finetune on the SAME dataset, call " + f"`python main.py finetune --checkpoint_path ...` directly — the " + f"kermt-finetune skill doesn't support resume." + ) + elif mode == "inference": + if not has_ffn: + errors.append( + "inference requires a finetuned ckpt with task FFN heads. " + f"This ckpt is classified as '{mt}' with no task heads. " + "Run finetune (mode 'finetune_init') first." + ) + elif mode == "embed": + # Encoder is sufficient. + pass + else: + errors.append(f"unknown mode '{mode}'") + + return errors + + +def validate(mode: str, ckpt_path: str) -> dict[str, Any]: + result: dict[str, Any] = { + "ok": False, + "model_type": "unknown", + "has_encoder": False, + "has_vocab_head": False, + "has_contrast_head": False, + "has_task_ffn": False, + "task_output_dims": [], + "vocab_sizes": {"atom": None, "bond": None, "smiles": None}, + "arch": {k: None for k in ARCH_KEYS}, + "saved_args": None, + "errors": [], + "warnings": [], + } + + # 1. Load the checkpoint. + try: + ckpt = torch.load(ckpt_path, map_location="cpu", weights_only=False) + except FileNotFoundError: + result["errors"].append(f"checkpoint not found: {ckpt_path}") + return result + except Exception as exc: # noqa: BLE001 + result["errors"].append(f"failed to load checkpoint {ckpt_path}: {type(exc).__name__}: {exc}") + return result + + if not isinstance(ckpt, dict) or "state_dict" not in ckpt: + result["errors"].append( + "checkpoint is not in the expected save_model_for_restart format " + "(expected a dict with a 'state_dict' key)." + ) + return result + + state_dict = _strip_ddp_prefix(ckpt["state_dict"]) + args_obj = ckpt.get("args") + + # 2. Classify and check mode contract. + classification = _classify_model(state_dict) + result.update(classification) + + contract_errors = _apply_mode_contract(mode, classification) + result["errors"].extend(contract_errors) + + # 3. Task output dims (for inference / informational). + if classification["has_task_ffn"]: + result["task_output_dims"] = _task_output_dims(state_dict) + + # 3b. Vocab head sizes (for continue-pretrain vocab-size verification). + result["vocab_sizes"] = _vocab_sizes(state_dict) + + # 4. Arch derivation: args first, shape introspection for what's still missing. + arch = _arch_from_args(args_obj) + arch, shape_warnings = _arch_from_shapes(state_dict, arch) + result["arch"] = arch + result["warnings"].extend(shape_warnings) + + # 5. Saved args as serializable dict (best-effort). + if args_obj is not None: + try: + result["saved_args"] = vars(args_obj) if isinstance(args_obj, Namespace) else dict(args_obj) + # Drop non-JSON-serializable values; agent skill only needs human-readable scalars. + result["saved_args"] = { + k: v for k, v in result["saved_args"].items() + if isinstance(v, (str, int, float, bool, type(None), list, dict)) + } + except Exception as exc: # noqa: BLE001 + result["warnings"].append(f"could not serialize saved_args: {type(exc).__name__}: {exc}") + + result["ok"] = not result["errors"] + return result + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Validate a KERMT checkpoint for a given workflow.") + parser.add_argument("--mode", required=True, + choices=["continue_pretrain", "upgrade_to_hybrid", "finetune_init", "inference", "embed"]) + parser.add_argument("--ckpt", required=True, help="Path to the .pt checkpoint") + args = parser.parse_args(argv) + + try: + result = validate(args.mode, args.ckpt) + except Exception as exc: # noqa: BLE001 + # Last-resort safety net: keep stdout JSON-clean, dump trace to stderr. + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({ + "ok": False, + "model_type": "unknown", + "errors": [f"unhandled exception in validator: {type(exc).__name__}: {exc}"], + "warnings": [], + "arch": {k: None for k in ARCH_KEYS}, + "has_encoder": False, + "has_vocab_head": False, + "has_contrast_head": False, + "has_task_ffn": False, + "task_output_dims": [], + "vocab_sizes": {"atom": None, "bond": None, "smiles": None}, + "saved_args": None, + }, indent=2)) + return 1 + + print(json.dumps(result, indent=2)) + return 0 if result["ok"] else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-pretrain-scratch/scripts/check_data.py b/skills/kermt-pretrain-scratch/scripts/check_data.py new file mode 100644 index 0000000..b8f9b15 --- /dev/null +++ b/skills/kermt-pretrain-scratch/scripts/check_data.py @@ -0,0 +1,310 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Validate a CSV input for a given KERMT agent workflow. + +Mode-dispatched. Emits a single JSON object to stdout that the calling skill +parses to decide whether to proceed (`ok: true`) or surface a structured error +back to the user (`ok: false` with `errors[]`). The JSON shape is stable +across modes; only the contract for what counts as `ok` differs per mode. + +Modes +----- +pretrain Pretrain corpus CSV. Requires a `smiles` column. Other columns + are ignored. Label columns are not required (and not expected). + +finetune Labeled CSV for a downstream task. Requires `smiles` plus + >=1 numeric target column. Target columns are specified via + `--targets ...`. If `--targets` is omitted, the + validator auto-detects numeric non-smiles columns and reports + them; the skill will then prompt the user to confirm or refine. + +inference CSV to run predictions on. Requires `smiles`. Target columns are + not required (and not expected — predictions are written out). + +embed CSV to extract embeddings from. Requires `smiles` only. + +SMILES validation +----------------- +By default the validator samples up to 20 SMILES (first 10 + last 10) and +checks each one parses with RDKit. Pass `--strict-rdkit` to parse every +SMILES (slow on large corpora). A SMILES is considered "invalid" if RDKit +returns `None` from `MolFromSmiles(smi, sanitize=True)` — empty / null +rows are counted separately. + +Duplicate-SMILES detection is always full (cheap). + +Output (stdout) +--------------- +{ + "ok": true | false, + "mode": str, + "csv_path": str, + "num_rows": int, + "num_columns": int, + "columns": [str, ...], + "has_smiles_column": bool, + "smiles_column_name": str | null, // actual header used (may differ in case) + "num_blank_smiles": int, + "num_invalid_smiles": int, // among the parsed sample + "smiles_check_method": "sampled" | "full", + "smiles_check_count": int, + "num_duplicate_smiles": int, + "target_columns": [str, ...], // populated only for finetune mode + "num_missing_per_target": { col: int, ... }, + "auto_detected_targets": [str, ...], // when --targets is omitted in finetune mode + "errors": [str, ...], + "warnings": [str, ...] +} + +Exit code: 0 on `ok: true`, 1 on `ok: false`. Loader exceptions are caught +and surfaced into `errors[]` with `ok: false` (still exit 1). + +CLI +--- + check_data.py --mode --csv + [--targets ...] # finetune only + [--strict-rdkit] # full SMILES parse +""" +from __future__ import annotations + +import argparse +import json +import sys +import traceback +from pathlib import Path +from typing import Any + +import pandas as pd + + +CANONICAL_SMILES_COLUMN = "smiles" +SMILES_SAMPLE_PER_END = 10 # how many SMILES from head + how many from tail to sample + + +def _find_smiles_column(columns: list[str]) -> str | None: + """Return the actual column header matching 'smiles' case-insensitively, or None.""" + for c in columns: + if c.lower() == CANONICAL_SMILES_COLUMN: + return c + return None + + +def _parse_smiles_sample(smiles_values: list[str], full: bool) -> tuple[int, int, str]: + """Run RDKit MolFromSmiles on a sample or all of the SMILES. Returns + (num_parsed, num_invalid, method).""" + # Import here so the script can still surface a clean JSON error if RDKit + # is unavailable in the host env. + try: + from rdkit import Chem + from rdkit import RDLogger + RDLogger.DisableLog("rdApp.*") # suppress per-mol parse warnings + except ImportError as exc: + raise RuntimeError( + f"RDKit is not importable in this environment: {exc}. " + "Run check_data.py inside the kermt container." + ) from exc + + if full or len(smiles_values) <= 2 * SMILES_SAMPLE_PER_END: + sample = smiles_values + method = "full" + else: + sample = smiles_values[:SMILES_SAMPLE_PER_END] + smiles_values[-SMILES_SAMPLE_PER_END:] + method = "sampled" + + invalid = 0 + parsed = 0 + for smi in sample: + if not smi: # already counted as blank elsewhere + continue + parsed += 1 + mol = Chem.MolFromSmiles(smi, sanitize=True) + if mol is None: + invalid += 1 + return parsed, invalid, method + + +def _autodetect_target_columns(df: pd.DataFrame, smiles_col: str) -> list[str]: + """Pick columns that look like numeric targets. A column qualifies if it + is (a) not the smiles column and (b) >=80% of non-null values convert to float. + Heuristic only — returned for the skill to prompt the user to confirm.""" + candidates: list[str] = [] + for col in df.columns: + if col == smiles_col: + continue + ser = df[col].dropna() + if len(ser) == 0: + continue + try: + converted = pd.to_numeric(ser, errors="coerce") + except (TypeError, ValueError): + continue + if converted.notna().sum() / max(len(ser), 1) >= 0.8: + candidates.append(col) + return candidates + + +def validate(mode: str, csv_path: str, targets: list[str] | None, strict_rdkit: bool) -> dict[str, Any]: + result: dict[str, Any] = { + "ok": False, + "mode": mode, + "csv_path": csv_path, + "num_rows": 0, + "num_columns": 0, + "columns": [], + "has_smiles_column": False, + "smiles_column_name": None, + "num_blank_smiles": 0, + "num_invalid_smiles": 0, + "smiles_check_method": "sampled", + "smiles_check_count": 0, + "num_duplicate_smiles": 0, + "target_columns": [], + "num_missing_per_target": {}, + "auto_detected_targets": [], + "errors": [], + "warnings": [], + } + + # 1. Read the CSV. + path = Path(csv_path) + if not path.is_file(): + result["errors"].append(f"CSV not found: {csv_path}") + return result + try: + df = pd.read_csv(path) + except pd.errors.EmptyDataError: + result["errors"].append(f"CSV is empty (no header): {csv_path}") + return result + except Exception as exc: # noqa: BLE001 + result["errors"].append(f"failed to read CSV {csv_path}: {type(exc).__name__}: {exc}") + return result + + result["num_rows"] = int(len(df)) + result["num_columns"] = int(len(df.columns)) + result["columns"] = [str(c) for c in df.columns] + + # 2. Locate the SMILES column. + smiles_col = _find_smiles_column(result["columns"]) + if smiles_col is None: + result["errors"].append( + f"no column named 'smiles' (case-insensitive) found in CSV. " + f"Available columns: {result['columns']}" + ) + return result + result["has_smiles_column"] = True + result["smiles_column_name"] = smiles_col + if smiles_col != CANONICAL_SMILES_COLUMN: + result["warnings"].append( + f"SMILES column is named '{smiles_col}' but downstream code expects '{CANONICAL_SMILES_COLUMN}' " + f"(lowercase). Rename the column to '{CANONICAL_SMILES_COLUMN}' before running the workflow." + ) + + # 3. Blank-SMILES count + duplicate count + RDKit parse check. + smi_series = df[smiles_col].astype(str).fillna("").str.strip() + blank_mask = smi_series.eq("") | smi_series.str.lower().eq("nan") + result["num_blank_smiles"] = int(blank_mask.sum()) + + nonblank = smi_series[~blank_mask] + result["num_duplicate_smiles"] = int(len(nonblank) - nonblank.nunique()) + + if len(nonblank) == 0: + result["errors"].append("no non-blank SMILES found in the CSV") + return result + + try: + parsed, invalid, method = _parse_smiles_sample(nonblank.tolist(), full=strict_rdkit) + except RuntimeError as exc: + result["errors"].append(str(exc)) + return result + result["smiles_check_count"] = parsed + result["num_invalid_smiles"] = invalid + result["smiles_check_method"] = method + + if invalid > 0: + scope = "all rows" if method == "full" else f"the {parsed} sampled rows" + result["errors"].append( + f"{invalid} out of {parsed} SMILES in {scope} failed to parse with RDKit. " + "Either pre-clean the CSV with scripts/clean_smiles.py or pass --strict-rdkit to see " + "the full count." + ) + + # 4. Target-column handling — finetune mode only. + if mode == "finetune": + if targets: + missing = [t for t in targets if t not in df.columns] + if missing: + result["errors"].append( + f"target column(s) not found in CSV: {missing}. " + f"Available columns: {result['columns']}" + ) + else: + result["target_columns"] = list(targets) + for t in targets: + nan_count = int(df[t].isna().sum()) + result["num_missing_per_target"][t] = nan_count + # Confirm numeric-ish. + nonnan = df[t].dropna() + converted = pd.to_numeric(nonnan, errors="coerce") + non_numeric_count = int(converted.isna().sum()) + if non_numeric_count > 0: + result["warnings"].append( + f"target column '{t}' has {non_numeric_count} non-numeric value(s) " + f"that will be dropped by the finetune runner." + ) + else: + # Auto-detect — surface candidates so the skill can prompt the user. + result["auto_detected_targets"] = _autodetect_target_columns(df, smiles_col) + if not result["auto_detected_targets"]: + result["errors"].append( + "no numeric non-smiles columns detected. finetune needs at least one target column; " + "specify it explicitly via --targets ." + ) + else: + result["warnings"].append( + f"--targets was not specified; auto-detected candidate target columns " + f"{result['auto_detected_targets']}. The skill will prompt the user to confirm." + ) + + # 5. Small-corpus warning — only for pretrain (other modes can be tiny by design). + if mode == "pretrain" and result["num_rows"] < 100: + result["warnings"].append( + f"pretrain corpus is only {result['num_rows']} molecule(s). Pretraining typically " + f"needs orders of magnitude more — verify this is the intended input." + ) + + result["ok"] = not result["errors"] + return result + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description="Validate a CSV input for a KERMT agent workflow.") + parser.add_argument("--mode", required=True, choices=["pretrain", "finetune", "inference", "embed"]) + parser.add_argument("--csv", required=True, help="Path to the input CSV") + parser.add_argument("--targets", nargs="+", default=None, + help="(finetune only) target column names. If omitted, the validator auto-detects " + "numeric non-smiles columns and reports them as candidates.") + parser.add_argument("--strict-rdkit", action="store_true", + help="Parse every SMILES with RDKit rather than sampling (slow on large CSVs).") + args = parser.parse_args(argv) + + try: + result = validate(args.mode, args.csv, args.targets, args.strict_rdkit) + except Exception as exc: # noqa: BLE001 + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({ + "ok": False, + "mode": args.mode, + "csv_path": args.csv, + "errors": [f"unhandled exception in validator: {type(exc).__name__}: {exc}"], + "warnings": [], + }, indent=2)) + return 1 + + print(json.dumps(result, indent=2)) + return 0 if result["ok"] else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-pretrain-scratch/scripts/kermt_container.sh b/skills/kermt-pretrain-scratch/scripts/kermt_container.sh new file mode 100755 index 0000000..028057e --- /dev/null +++ b/skills/kermt-pretrain-scratch/scripts/kermt_container.sh @@ -0,0 +1,484 @@ +#!/usr/bin/env bash +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# kermt_container.sh — bootstrap helper for the kermt agent skills. +# +# Two ways to use this file: +# +# 1. As a subcommand dispatcher (recommended for skills): +# "$SKILL_DIR/scripts/kermt_container.sh" ensure_image +# "$SKILL_DIR/scripts/kermt_container.sh" run --ckpt /host/ckpt.pt -- python -c 'import torch; print(torch.cuda.device_count())' +# "$SKILL_DIR/scripts/kermt_container.sh" run_detached --name foo --run-dir runs/foo -- bash train.sh +# +# 2. Sourced into a shell or another script, then call the kermt_* functions +# directly: +# source "$SKILL_DIR/scripts/kermt_container.sh" +# kermt_ensure_image +# kermt_run --ckpt /host/ckpt.pt -- python ... +# +# Configuration (override via env vars before invocation): +# KERMT_IMAGE docker image tag (default: kermt:latest) +# KERMT_REPO host path to the kermt repo checkout (default: auto-derived +# from this script's location) +# KERMT_GPUS value passed to docker --gpus (default: all) +# +# Mount flags accepted by kermt_run / kermt_run_detached: +# --data bind to /data (read-only). If is a file, +# its PARENT directory is mounted at /data so +# commands can use /data/; if is a +# directory, it is mounted at /data directly. +# --ckpt bind to /ckpt (read-only; the path is mounted as-is) +# --vocab-dir bind to /vocab (read-only) +# --run-dir bind to /runs (read-write; created on host if missing) +# --model-dir bind to /model (read-write; created on host if missing). +# Target for released-model downloads (fetch_released_model.py). +# +# Additional flags for kermt_run_detached: +# --name docker container name (default: kermt--) +# +# Everything after `--` is the command passed to the container. It runs inside +# the `kermt` conda environment (the image's default env). + +set -o pipefail + +: "${KERMT_IMAGE:=kermt:latest}" +: "${KERMT_GPUS:=all}" + +# The skill may be installed outside the KERMT checkout. Mount its own helpers +# separately so container commands always execute the distributed skill copy. +_kermt_script_dir="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)" +_kermt_bundle_dir="$(cd "$_kermt_script_dir/.." && pwd)" +if [[ -z "${KERMT_REPO:-}" ]]; then + for _kermt_start in "$_kermt_script_dir" "$PWD"; do + _kermt_candidate="$_kermt_start" + while [[ "$_kermt_candidate" != / ]]; do + if [[ -f "$_kermt_candidate/main.py" && -d "$_kermt_candidate/kermt" ]]; then + KERMT_REPO="$_kermt_candidate" + break 2 + fi + _kermt_candidate="$(dirname "$_kermt_candidate")" + done + done + unset _kermt_start _kermt_candidate +fi +unset _kermt_script_dir + +_kermt_require_repo() { + if [[ -z "${KERMT_REPO:-}" || ! -f "$KERMT_REPO/main.py" || ! -d "$KERMT_REPO/kermt" ]]; then + echo "[kermt] Set KERMT_REPO to the KERMT checkout containing main.py and kermt/." >&2 + return 1 + fi + KERMT_REPO="$(cd "$KERMT_REPO" && pwd)" || return $? + export KERMT_REPO +} + +# ----------------------------------------------------------------------------- +# Host environment checks +# ----------------------------------------------------------------------------- + +kermt_check_docker() { + if ! command -v docker >/dev/null 2>&1; then + echo "[kermt] error: docker not found on PATH. Install Docker first." >&2 + return 1 + fi + if ! docker info >/dev/null 2>&1; then + echo "[kermt] error: docker daemon not reachable. Is the docker service running, and is your user in the 'docker' group?" >&2 + return 1 + fi +} + +kermt_check_system() { + # Probe host system and report GPU presence + VRAM + compute capability + + # driver / CUDA version + disk space. Emits a single JSON document to + # stdout that the calling skill consumes; exits 0 with `ok: false` and a + # populated `gaps` array when anything is below the per-workflow minimum, + # exits 1 only on unexpected internal errors. Uses host nvidia-smi + df + + # host python3 (stdlib only). + python3 - "$KERMT_REPO" "$KERMT_IMAGE" <<'PYEOF' +import json, os, shutil, subprocess, sys + +repo, image = sys.argv[1], sys.argv[2] + +result = { + "ok": True, + "gpus": [], + "disk": {"path": repo, "free_gb": None, "min_gb": 20}, + "host": {"docker": None, "nvidia_smi": None, "container_toolkit": None}, + "image": {"tag": image, "present_locally": None}, + "gaps": [], +} + +def _gap(msg): + result["ok"] = False + result["gaps"].append(msg) + +# docker presence +try: + r = subprocess.run(["docker", "info"], capture_output=True, text=True, timeout=10) + result["host"]["docker"] = "ok" if r.returncode == 0 else f"failed: {r.stderr.strip().splitlines()[-1] if r.stderr else 'unknown'}" + if r.returncode != 0: + _gap("docker daemon not reachable (is the service running, and is your user in the 'docker' group?)") +except FileNotFoundError: + result["host"]["docker"] = "not found" + _gap("docker not on PATH; install Docker first") +except Exception as e: + result["host"]["docker"] = f"error: {e}" + _gap(f"docker probe failed: {e}") + +# nvidia-smi (host driver) +try: + r = subprocess.run( + ["nvidia-smi", "--query-gpu=name,memory.total,compute_cap,driver_version,uuid", + "--format=csv,noheader,nounits"], + capture_output=True, text=True, timeout=10, + ) + if r.returncode == 0: + result["host"]["nvidia_smi"] = "ok" + for line in r.stdout.strip().splitlines(): + parts = [p.strip() for p in line.split(",")] + if len(parts) >= 5: + try: + vram_mb = int(parts[1]) + except ValueError: + vram_mb = None + result["gpus"].append({ + "name": parts[0], + "vram_mb": vram_mb, + "compute_cap": parts[2], + "driver": parts[3], + "uuid": parts[4], + }) + if not result["gpus"]: + _gap("nvidia-smi succeeded but reported no GPUs") + else: + result["host"]["nvidia_smi"] = "failed" + _gap("nvidia-smi found but failed; is the NVIDIA driver loaded?") +except FileNotFoundError: + result["host"]["nvidia_smi"] = "not found" + _gap("nvidia-smi not on PATH; install the NVIDIA driver") +except Exception as e: + result["host"]["nvidia_smi"] = f"error: {e}" + _gap(f"nvidia-smi probe failed: {e}") + +# disk free at the repo location +try: + free_bytes = shutil.disk_usage(repo).free + free_gb = free_bytes // (1024**3) + result["disk"]["free_gb"] = free_gb + if free_gb < result["disk"]["min_gb"]: + _gap(f"disk free at {repo} is {free_gb} GB; need at least {result['disk']['min_gb']} GB for the kermt image") +except Exception as e: + _gap(f"could not check disk space at {repo}: {e}") + +# image presence (informational only) +try: + r = subprocess.run(["docker", "image", "inspect", image], capture_output=True, text=True, timeout=10) + result["image"]["present_locally"] = (r.returncode == 0) +except Exception: + result["image"]["present_locally"] = None + +# nvidia-container-toolkit probe — only meaningful if both docker and a +# locally-present image are available. Pick kermt:$tag first; fall back to +# the small CUDA base image if that's the only one present; otherwise skip +# (avoid pulling anything). +def _probe_image(): + for img in (image, "nvidia/cuda:12.6.3-base-ubuntu22.04"): + r = subprocess.run(["docker", "image", "inspect", img], capture_output=True) + if r.returncode == 0: + return img + return None + +probe_img = _probe_image() +if probe_img: + try: + r = subprocess.run( + ["docker", "run", "--rm", "--gpus", "all", probe_img, "nvidia-smi"], + capture_output=True, text=True, timeout=60, + ) + if r.returncode == 0: + result["host"]["container_toolkit"] = f"ok (probed via {probe_img})" + else: + result["host"]["container_toolkit"] = f"failed (probed via {probe_img})" + _gap("`docker run --gpus all` failed; install nvidia-container-toolkit and ensure the host driver supports it") + except Exception as e: + result["host"]["container_toolkit"] = f"error: {e}" + _gap(f"nvidia-container-toolkit probe failed: {e}") +else: + result["host"]["container_toolkit"] = "skipped (no probe image present locally; run ensure_image first)" + +print(json.dumps(result, indent=2)) +PYEOF +} + +kermt_check_gpu() { + # Probes whether `docker --gpus all` is wired up (nvidia-container-toolkit). + # Image-selection priority (never pulls anything): + # 1) $KERMT_IMAGE if it exists locally, + # 2) else nvidia/cuda:12.6.3-base-ubuntu22.04 if it exists locally, + # 3) else skip with a warning (return 0). The smoke test inside kermt_run + # will catch broken GPU passthrough later anyway. + local probe_img="" + if docker image inspect "$KERMT_IMAGE" >/dev/null 2>&1; then + probe_img="$KERMT_IMAGE" + elif docker image inspect nvidia/cuda:12.6.3-base-ubuntu22.04 >/dev/null 2>&1; then + probe_img="nvidia/cuda:12.6.3-base-ubuntu22.04" + else + echo "[kermt] check_gpu: skipped — neither '$KERMT_IMAGE' nor 'nvidia/cuda:12.6.3-base-ubuntu22.04' is present locally. Run 'ensure_image' first, or this probe will be exercised by the in-container smoke test." >&2 + return 0 + fi + if ! docker run --rm --gpus all "$probe_img" nvidia-smi >/dev/null 2>&1; then + echo "[kermt] error: 'docker run --gpus all' failed (probe image: $probe_img). Install nvidia-container-toolkit and ensure the host has a CUDA-capable NVIDIA driver." >&2 + return 1 + fi +} + +# ----------------------------------------------------------------------------- +# Image build / verification +# ----------------------------------------------------------------------------- + +kermt_ensure_image() { + _kermt_require_repo || return $? + kermt_check_docker || return $? + if docker image inspect "$KERMT_IMAGE" >/dev/null 2>&1; then + local id + id=$(docker image inspect "$KERMT_IMAGE" --format '{{.Id}}' 2>/dev/null | cut -c1-19) + echo "[kermt] image '$KERMT_IMAGE' already present (${id:-unknown})" + return 0 + fi + echo "[kermt] image '$KERMT_IMAGE' not found; building from $KERMT_REPO/Dockerfile" + echo "[kermt] first build typically takes 10-20 minutes on a typical workstation; subsequent runs reuse the cached image" + docker build -t "$KERMT_IMAGE" -f "$KERMT_REPO/Dockerfile" "$KERMT_REPO" +} + +# ----------------------------------------------------------------------------- +# Mount-flag parser, internal +# ----------------------------------------------------------------------------- +# Reads flags from the caller's positional args until it hits '--', appending +# `-v src:dst[:ro]` pairs into the caller-provided array name (passed as $1). +# Returns the number of caller-provided args consumed via _kermt_consumed. +# This is bash-specific (uses nameref via `declare -n`). + +_kermt_parse_mounts() { + local -n _out="$1" + shift + _kermt_consumed=0 + while [[ $# -gt 0 ]]; do + case "$1" in + --) + return 0 + ;; + --data) + [[ -e "$2" ]] || { echo "[kermt] --data path not found: $2" >&2; return 1; } + # If the user passes a file, mount its parent directory at /data so + # downstream commands can refer to /data/. Mounting a + # single file at /data makes the path-as-directory pattern in the + # skill examples (`--csv /data/`) fail with "not found". + if [[ -d "$2" ]]; then + _out+=("-v" "$(realpath "$2"):/data:ro") + else + _out+=("-v" "$(realpath "$(dirname "$2")"):/data:ro") + fi + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --ckpt) + [[ -e "$2" ]] || { echo "[kermt] --ckpt path not found: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/ckpt:ro") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --vocab-dir) + [[ -d "$2" ]] || { echo "[kermt] --vocab-dir not found or not a directory: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/vocab:ro") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --run-dir) + mkdir -p "$2" || { echo "[kermt] failed to create --run-dir: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/runs") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --model-dir) + mkdir -p "$2" || { echo "[kermt] failed to create --model-dir: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/model") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + *) + return 0 + ;; + esac + done +} + +# ----------------------------------------------------------------------------- +# Foreground / detached run +# ----------------------------------------------------------------------------- + +# Capture host-side git state for the repo and emit `-e KERMT_REPO_COMMIT=… +# -e KERMT_REPO_DIRTY=true|false` flags. Used by the run / run_detached +# wrappers so the runner's run.json manifest gets honest commit info even +# though `git -C /workspace` inside the container fails due to bind-mount +# ownership. +_kermt_git_env_flags() { + local commit="unknown" + local dirty="false" + if command -v git >/dev/null 2>&1 && [[ -d "$KERMT_REPO/.git" ]]; then + local c + c=$(git -C "$KERMT_REPO" rev-parse HEAD 2>/dev/null) && commit="$c" + # `--untracked-files=no` filters out user-private notes (e.g. a CLAUDE.md + # or RELEASE_PLAN_v2.0.md at the repo root) that wouldn't affect + # reproducibility — only modifications to tracked files do. + if [[ -n "$(git -C "$KERMT_REPO" status --porcelain --untracked-files=no 2>/dev/null | head -n 1)" ]]; then + dirty="true" + fi + fi + printf '%s\n%s\n%s\n%s\n' "-e" "KERMT_REPO_COMMIT=$commit" "-e" "KERMT_REPO_DIRTY=$dirty" +} + +# Forward HF_TOKEN into the container when it is set, so fetch_released_model.py +# can authenticate to Hugging Face. The current release is public (no token +# needed); this only guards against shared-IP rate limits or a future gated +# repo. Emits nothing when HF_TOKEN is unset. +_kermt_hf_env_flags() { + if [[ -n "${HF_TOKEN:-}" ]]; then + printf '%s\n%s\n' "-e" "HF_TOKEN=$HF_TOKEN" + fi +} + +kermt_run() { + kermt_ensure_image || return $? + local mount_args=() + _kermt_parse_mounts mount_args "$@" || return $? + shift "$_kermt_consumed" + if [[ "${1:-}" != "--" ]]; then + echo "[kermt] expected '--' separating mount flags from the command (got '${1:-}')" >&2 + return 1 + fi + shift + if [[ $# -eq 0 ]]; then + echo "[kermt] no command supplied after '--'" >&2 + return 1 + fi + local git_args=() + while IFS= read -r line; do git_args+=("$line"); done < <(_kermt_git_env_flags) + local hf_args=() + while IFS= read -r line; do hf_args+=("$line"); done < <(_kermt_hf_env_flags) + docker run --rm --gpus "$KERMT_GPUS" \ + --user "$(id -u):$(id -g)" \ + -v "$KERMT_REPO:/workspace" \ + -v "$_kermt_bundle_dir:/skill:ro" \ + "${mount_args[@]}" \ + -w /workspace \ + -e KERMT_REPO=/workspace \ + -e PYTHONPATH=/workspace \ + -e HOME=/tmp/kermt-home \ + "${git_args[@]}" \ + "${hf_args[@]}" \ + "$KERMT_IMAGE" \ + conda run -n kermt --no-capture-output bash -c "$*" +} + +kermt_run_detached() { + kermt_ensure_image || return $? + local name="" + local mount_args=() + # Pull --name out first, then let the shared mount parser handle the rest. + while [[ $# -gt 0 ]]; do + case "$1" in + --name) name="$2"; shift 2 ;; + --) break ;; + --data|--ckpt|--vocab-dir|--run-dir|--model-dir) break ;; + *) break ;; + esac + done + _kermt_parse_mounts mount_args "$@" || return $? + shift "$_kermt_consumed" + if [[ "${1:-}" != "--" ]]; then + echo "[kermt] expected '--' separating mount flags from the command (got '${1:-}')" >&2 + return 1 + fi + shift + if [[ $# -eq 0 ]]; then + echo "[kermt] no command supplied after '--'" >&2 + return 1 + fi + if [[ -z "$name" ]]; then + name="kermt-$(date -u +%Y%m%dT%H%M%SZ)-$$" + fi + local cid + local git_args=() + while IFS= read -r line; do git_args+=("$line"); done < <(_kermt_git_env_flags) + local hf_args=() + while IFS= read -r line; do hf_args+=("$line"); done < <(_kermt_hf_env_flags) + cid=$(docker run -d --gpus "$KERMT_GPUS" \ + --user "$(id -u):$(id -g)" \ + --name "$name" \ + -v "$KERMT_REPO:/workspace" \ + -v "$_kermt_bundle_dir:/skill:ro" \ + "${mount_args[@]}" \ + -w /workspace \ + -e KERMT_REPO=/workspace \ + -e PYTHONPATH=/workspace \ + -e HOME=/tmp/kermt-home \ + "${git_args[@]}" \ + "${hf_args[@]}" \ + "$KERMT_IMAGE" \ + conda run -n kermt --no-capture-output bash -c "$*") || return $? + echo "[kermt] container started: name=$name id=$cid" + echo "[kermt] follow logs: docker logs -f $name" + echo "[kermt] wait for exit: docker wait $name" + echo "[kermt] stop: docker stop $name" + echo "$cid" +} + +# ----------------------------------------------------------------------------- +# Subcommand dispatch when invoked directly (not sourced) +# ----------------------------------------------------------------------------- + +if [[ "${BASH_SOURCE[0]:-$0}" == "${0}" ]]; then + cmd="${1:-}"; shift || true + case "$cmd" in + check_docker) kermt_check_docker "$@" ;; + check_gpu) kermt_check_gpu "$@" ;; + check_system) kermt_check_system "$@" ;; + ensure_image) kermt_ensure_image "$@" ;; + run) kermt_run "$@" ;; + run_detached) kermt_run_detached "$@" ;; + ""|-h|--help) + cat >&2 < [args...] + +Subcommands: + check_docker Verify docker is installed and the daemon is reachable. + check_gpu Verify 'docker --gpus all' works (nvidia-container-toolkit). + check_system Emit a JSON probe of host GPU + VRAM + compute_cap + + driver + disk space + container toolkit + image presence. + Exits 0 with ok=false + a 'gaps' list when anything's + below the per-workflow minimum. + ensure_image Build kermt:latest from \$KERMT_REPO/Dockerfile if missing. + run [flags] -- ... Run a command inside the container (foreground, --rm). + run_detached [flags] -- ... + Run detached; prints container name + id + log hint. + +Mount flags (for run / run_detached): + --data bind to /data (read-only) + --ckpt bind to /ckpt (read-only) + --vocab-dir bind to /vocab (read-only) + --run-dir bind to /runs (read-write; created on host if missing) + --model-dir bind to /model (read-write; released-model download target) + +Additional flags for run_detached: + --name container name (default: kermt--) + +Environment overrides: + KERMT_IMAGE default kermt:latest + KERMT_REPO checkout path; otherwise discovered above the skill or working directory + KERMT_GPUS default all +EOF + exit 1 + ;; + *) + echo "[kermt] unknown subcommand: $cmd" >&2 + echo "[kermt] run '$0 --help' for usage" >&2 + exit 1 + ;; + esac +fi diff --git a/skills/kermt-pretrain-scratch/scripts/prepare_data.py b/skills/kermt-pretrain-scratch/scripts/prepare_data.py new file mode 100644 index 0000000..f0edb3e --- /dev/null +++ b/skills/kermt-pretrain-scratch/scripts/prepare_data.py @@ -0,0 +1,817 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Mode-dispatched data preparation pipeline for the KERMT agent skills. + +Composes the existing repo data-prep scripts (`scripts/clean_smiles.py`, +`scripts/save_features.py`, `scripts/build_vocab.py`, `scripts/split_data.py`) +into a single one-call entry point per workflow. Output lands in `--out` with +a `prepare_data.json` manifest that the downstream runners read. + +Mode pipelines +-------------- +pretrain : clean -> (optional auto-split train into train+val by --val-frac) + -> save_features (fgtasklabel) on each CSV + -> vocab step: if --vocab-dir / --{atom,bond,smiles}-vocab given, + copy those through (continue-pretrain case — the ckpt's vocab + is authoritative); else if --skip-vocab, skip; + else build_vocab on train (pretrain-from-scratch case) + -> split_data (graph + feature shards + summary.txt) per CSV +finetune : clean each provided CSV -> (optional random split when only one + CSV is provided; emits a strong warning recommending scaffold- + balanced pre-splits) -> save_features (rdkit_2d_normalized) per CSV +inference : clean -> save_features (rdkit_2d_normalized) +embed : clean only (extract_embeddings.py featurizes on the fly) + +Output convention +----------------- +The manifest under `/prepare_data.json` captures every step's inputs, +outputs, duration, and skipped-due-to-existing flag, plus a top-level +`split_method` field (one of: "user_provided", "random", "n/a") that the +finetune runner uses to pass the correct `--split_type` to main.py. + +Subprocess composition +---------------------- +Each underlying script is invoked via `subprocess.run`. The PYTHONPATH=/workspace +env var (set by `scripts/kermt_container.sh`) makes the `kermt` package +importable inside the subprocesses; without it, build_vocab.py and split_data.py +fail with `ModuleNotFoundError: No module named 'kermt'`. + +CLI +--- + prepare_data.py --mode {pretrain|finetune|inference|embed} + --csv --out + [--val-csv ] [--test-csv ] + [--val-frac 0.1] [--test-frac 0.1] [--seed 0] + [--sample-per-file 100000] [--vocab-format json] + [--dataset-name pretrain] + [--targets COL [COL ...]] + [--features-generator ] + [--smiles-column 0] + [--force] [--skip-clean] [--skip-features] + [--skip-vocab] [--skip-split] +""" +from __future__ import annotations + +import argparse +import json +import os +import shutil +import subprocess +import sys +import time +import traceback +from pathlib import Path +from typing import Any + +import pandas as pd + +# sys.path tweak so `_utils` is importable regardless of how this script +# is invoked (kermt_run sets PYTHONPATH=/workspace; bare-Python launches +# from the host don't). +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import PRETRAIN_VOCAB_STEMS, resolve_kermt_repo, validate_vocab_file # noqa: E402 + + +REPO_ROOT = resolve_kermt_repo() +EXISTING_SCRIPTS = REPO_ROOT / "scripts" + +DEFAULT_FEATURES_GENERATOR = { + "pretrain": "fgtasklabel", + "finetune": "rdkit_2d_normalized", + "inference": "rdkit_2d_normalized", + "embed": None, # not used +} + +VALID_MODES = ("pretrain", "finetune", "inference", "embed") + + +# --------------------------------------------------------------------------- +# Subprocess helpers +# --------------------------------------------------------------------------- + +def _run(cmd: list[str], step_name: str, manifest: dict[str, Any]) -> dict[str, Any]: + """Run a subprocess, append a step entry to manifest, raise on failure.""" + step: dict[str, Any] = { + "name": step_name, + "cmd": cmd, + "duration_s": None, + "ok": False, + "stderr_tail": "", + "skipped_due_to_existing": False, + } + t0 = time.time() + proc = subprocess.run(cmd, capture_output=True, text=True) + step["duration_s"] = round(time.time() - t0, 2) + if proc.returncode != 0: + step["stderr_tail"] = (proc.stderr or "").splitlines()[-20:] + step["ok"] = False + manifest["steps"].append(step) + raise RuntimeError( + f"step '{step_name}' failed (exit {proc.returncode}); " + f"command: {' '.join(cmd)}\nstderr tail:\n" + "\n".join(step["stderr_tail"]) + ) + step["ok"] = True + manifest["steps"].append(step) + return step + + +def _skipped(step_name: str, output_path: str, manifest: dict[str, Any]) -> dict[str, Any]: + step = { + "name": step_name, + "output": output_path, + "ok": True, + "duration_s": 0.0, + "skipped_due_to_existing": True, + } + manifest["steps"].append(step) + return step + + +def _exists_nonempty(path: Path) -> bool: + """File exists with non-zero size, or directory exists with at least one entry.""" + if not path.exists(): + return False + if path.is_file(): + return path.stat().st_size > 0 + if path.is_dir(): + try: + next(path.iterdir()) + return True + except StopIteration: + return False + return False + + +# --------------------------------------------------------------------------- +# Per-script wrappers +# --------------------------------------------------------------------------- + +def _resolve_smiles_column(csv_path: Path, explicit_value: int | None) -> int: + """Return the 0-based index of the SMILES column in csv_path. + + Auto-detection rule when `explicit_value is None`: + 1. Read the CSV header (first non-empty row). + 2. Prefer an exact lowercase `smiles` column (kermt convention). + 3. Otherwise accept a single case-insensitive match + (`SMILES`, `Smiles`, etc.). + 4. If no match (or multiple ambiguous matches), raise a ValueError + that surfaces the header so the user can disambiguate via + `--smiles-column N`. + + Real datasets routinely place SMILES at column index ≠ 0 + (e.g. openadmet's all.csv has "Molecule Name" at col 0 and "SMILES" + at col 1). Auto-detection prevents the silent 0-row-clean failure + mode where every row gets rejected because col 0 doesn't parse as + a SMILES string. + """ + if explicit_value is not None: + return explicit_value + + if not csv_path.is_file(): + raise ValueError(f"input CSV not found: {csv_path}") + + import csv as _csv + with csv_path.open("r", newline="") as f: + reader = _csv.reader(f) + try: + header = next(reader) + except StopIteration: + raise ValueError(f"input CSV {csv_path} is empty") + + stripped = [c.strip() for c in header] + # Prefer exact lowercase "smiles" + exact = [i for i, c in enumerate(stripped) if c == "smiles"] + if exact: + return exact[0] + # Then case-insensitive + ci = [i for i, c in enumerate(stripped) if c.lower() == "smiles"] + if len(ci) == 1: + return ci[0] + if len(ci) > 1: + raise ValueError( + f"input CSV {csv_path} has multiple SMILES-named columns: " + f"{[header[i] for i in ci]} at indices {ci}. " + "Pass --smiles-column N (0-based) to disambiguate." + ) + raise ValueError( + f"could not auto-detect a SMILES column in {csv_path}. " + f"Header columns: {header}. " + "Pass --smiles-column N (0-based) to specify which column holds SMILES." + ) + + +def _clean_smiles( + input_csv: Path, output_csv: Path, smiles_column: int, manifest: dict[str, Any], force: bool +) -> Path: + if not force and _exists_nonempty(output_csv): + _skipped(f"clean_smiles({input_csv.name})", str(output_csv), manifest) + return output_csv + output_csv.parent.mkdir(parents=True, exist_ok=True) + if force and output_csv.exists(): + # clean_smiles.py prompts interactively (input()) when the output file + # already exists — that's an EOFError in a non-TTY subprocess. Pre-delete. + output_csv.unlink() + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "clean_smiles.py"), + "--input", str(input_csv), + "--output", str(output_csv), + "--smiles_column", str(smiles_column), + ] + _run(cmd, f"clean_smiles({input_csv.name})", manifest) + return output_csv + + +def _reduce_to_smiles_column( + csv_path: Path, smiles_column: int, manifest: dict[str, Any] +) -> Path: + """Rewrite an inference CSV to keep only the SMILES column (at index 0). + + Downstream `kermt.util.utils.get_data` -> `MoleculeDatapoint.__init__` + floats every column after SMILES, which crashes on non-numeric passthrough + columns (e.g. a 'split' label of 'train'/'val'/'test', or a 'Molecule Name' + string). Inference does not need target columns, so drop them here. + + Note on skip semantics: this step is idempotent — running it on an + already-single-column file is a no-op. We record that with + `skipped_due_to_idempotent: True`, NOT `skipped_due_to_existing: True`. + The two fields have different meanings: `_existing` means "I found a + cached output file from a prior run and reused it" (overridden by + `--force`); `_idempotent` means "the input is already in the desired + state, so re-executing changes nothing" (safe to skip even under + `--force`). + """ + step_name = f"reduce_to_smiles_only({csv_path.name})" + start = time.time() + df = pd.read_csv(csv_path) + if df.shape[1] == 1: + manifest["steps"].append({ + "name": step_name, + "output": str(csv_path), + "ok": True, + "duration_s": time.time() - start, + "skipped_due_to_idempotent": True, + "note": "already single-column", + }) + return csv_path + effective_col = smiles_column if 0 <= smiles_column < df.shape[1] else 0 + df.iloc[:, [effective_col]].to_csv(csv_path, index=False) + manifest["steps"].append({ + "name": step_name, + "output": str(csv_path), + "ok": True, + "duration_s": time.time() - start, + "input_cols": int(df.shape[1]), + "kept_col": effective_col, + "kept_col_name": str(df.columns[effective_col]), + }) + return csv_path + + +def _save_features( + csv_path: Path, npz_path: Path, generator: str, manifest: dict[str, Any], force: bool +) -> Path: + if not force and _exists_nonempty(npz_path): + _skipped(f"save_features({csv_path.name}, {generator})", str(npz_path), manifest) + return npz_path + npz_path.parent.mkdir(parents=True, exist_ok=True) + if force and npz_path.exists(): + npz_path.unlink() # --restart still loads partial state if file exists; pre-delete to be safe + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "save_features.py"), + "--data_path", str(csv_path), + "--save_path", str(npz_path), + "--features_generator", generator, + "--restart", + ] + _run(cmd, f"save_features({csv_path.name}, {generator})", manifest) + return npz_path + + +def _resolve_vocab_inputs(args: argparse.Namespace) -> dict[str, Path | None] | None: + """Returns {atom, bond, smiles}->Path|None when the user supplied vocab + inputs (via --vocab-dir or --atom-vocab/--bond-vocab/--smiles-vocab), + else None (signal to fall through to build_vocab). + + Conventional filenames inside --vocab-dir: + pretrain_atom_vocab.{json,pkl} + pretrain_bond_vocab.{json,pkl} + pretrain_smiles_vocab.pkl + """ + if args.vocab_dir: + d = Path(args.vocab_dir).resolve() + if not d.is_dir(): + raise FileNotFoundError(f"--vocab-dir not found or not a directory: {d}") + def _find(stem: str, exts: tuple[str, ...]) -> Path | None: + for ext in exts: + p = d / f"{stem}.{ext}" + if p.is_file(): + return p + return None + atom = _find(PRETRAIN_VOCAB_STEMS["atom"], ("json", "pkl")) + bond = _find(PRETRAIN_VOCAB_STEMS["bond"], ("json", "pkl")) + smiles = _find(PRETRAIN_VOCAB_STEMS["smiles"], ("pkl",)) + if atom is None and bond is None and smiles is None: + stems = [PRETRAIN_VOCAB_STEMS[k] for k in ("atom", "bond", "smiles")] + raise FileNotFoundError( + f"--vocab-dir {d} contained no {{ {', '.join(stems) }}}.{{json,pkl}} " + f"files. Expected at least {PRETRAIN_VOCAB_STEMS['atom']} + " + f"{PRETRAIN_VOCAB_STEMS['bond']}." + ) + return {"atom": atom, "bond": bond, "smiles": smiles} + + if args.atom_vocab or args.bond_vocab or args.smiles_vocab: + return { + "atom": Path(args.atom_vocab).resolve() if args.atom_vocab else None, + "bond": Path(args.bond_vocab).resolve() if args.bond_vocab else None, + "smiles": Path(args.smiles_vocab).resolve() if args.smiles_vocab else None, + } + + return None + + +def _copy_provided_vocab( + src: dict[str, Path | None], dst_dir: Path, dataset_name: str, manifest: dict[str, Any], + force: bool, +) -> dict[str, Path]: + """When the user supplies vocab files (use ckpt's vocab as-is), + copy them into `/__vocab.` so the + downstream pretrain command sees the conventional filenames. + + `src` is `{atom: Path|None, bond: Path|None, smiles: Path|None}`. The atom + and bond entries must be both present or both absent (paired). smiles is + optional (cmim/hybrid only). + + Returns the same dict of (resolved) destination paths. + """ + import shutil + if (src["atom"] is None) != (src["bond"] is None): + raise ValueError( + "vocab pass-through requires atom and bond vocab paths to be paired; " + "got atom=" + str(src["atom"]) + ", bond=" + str(src["bond"]) + ) + out: dict[str, Path] = {} + dst_dir.mkdir(parents=True, exist_ok=True) + for which, path in src.items(): + if path is None: + continue + # Validate the source file IS a loadable KERMT vocab before copying. + # Catches the "user pointed --smiles-vocab at a random pickle" case + # early, with a clear error, instead of letting it surface as a cryptic + # SMILESVocab.load_vocab failure at pretrain_ddp.py launch time. + validate_vocab_file(path, kind=which) + ext = path.suffix.lstrip(".") + if which == "smiles": + ext = "pkl" # smiles vocab is always pickle + dst = dst_dir / f"{dataset_name}_{which}_vocab.{ext}" + if not force and _exists_nonempty(dst): + _skipped(f"copy_vocab({which})", str(dst), manifest) + out[which] = dst + continue + if force and dst.exists(): + dst.unlink() + shutil.copy2(path, dst) + manifest["steps"].append({ + "name": f"copy_vocab({which})", + "src": str(path), "dst": str(dst), "ok": True, + "duration_s": 0.0, "skipped_due_to_existing": False, + }) + out[which] = dst + return out + + +def _build_vocab( + csv_path: Path, vocab_dir: Path, dataset_name: str, vocab_format: str, + manifest: dict[str, Any], force: bool, +) -> dict[str, Path]: + """Builds atom + bond (in --vocab-format) and smiles (always pickle) vocabs. + Returns a dict of {atom, bond, smiles} -> Path.""" + suffix = "json" if vocab_format == "json" else "pkl" + expected = { + "atom": vocab_dir / f"{dataset_name}_atom_vocab.{suffix}", + "bond": vocab_dir / f"{dataset_name}_bond_vocab.{suffix}", + "smiles": vocab_dir / f"{dataset_name}_smiles_vocab.pkl", + } + if not force and all(_exists_nonempty(p) for p in expected.values()): + _skipped(f"build_vocab({csv_path.name})", str(vocab_dir), manifest) + return expected + vocab_dir.mkdir(parents=True, exist_ok=True) + if force: + for p in expected.values(): + if p.exists(): + p.unlink() + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "build_vocab.py"), + "--data_path", str(csv_path), + "--vocab_save_folder", str(vocab_dir), + "--dataset_name", dataset_name, + "--vocab_format", vocab_format, + ] + _run(cmd, f"build_vocab({csv_path.name})", manifest) + return expected + + +def _split_data( + csv_path: Path, features_path: Path | None, sample_per_file: int, output_dir: Path, + manifest: dict[str, Any], force: bool, +) -> Path: + """Run split_data.py to produce shard dirs (graph/ + optionally feature/ + summary.txt).""" + summary = output_dir / "summary.txt" + if not force and _exists_nonempty(summary): + _skipped(f"split_data({csv_path.name})", str(output_dir), manifest) + return output_dir + if force and output_dir.exists(): + shutil.rmtree(output_dir) + output_dir.mkdir(parents=True, exist_ok=True) + cmd = [ + sys.executable, str(EXISTING_SCRIPTS / "split_data.py"), + "--data_path", str(csv_path), + "--sample_per_file", str(sample_per_file), + "--output_path", str(output_dir), + ] + if features_path is not None: + cmd += ["--features_path", str(features_path)] + _run(cmd, f"split_data({csv_path.name})", manifest) + return output_dir + + +# --------------------------------------------------------------------------- +# Random splitter (used only when the user supplies a single CSV) +# --------------------------------------------------------------------------- + +def _random_split_csv( + src_csv: Path, dst_csvs: dict[str, Path], fractions: dict[str, float], seed: int, + manifest: dict[str, Any], force: bool, +) -> None: + """Shuffle src_csv and partition rows into dst_csvs by fractions. + `dst_csvs` and `fractions` are dicts keyed by the split name (e.g. 'train', 'val'). + Sum of fractions must be 1.0 (within float tolerance). Writes each dst_csv with the + same header as the input.""" + step = { + "name": f"random_split({src_csv.name})", + "seed": seed, + "fractions": fractions, + "ok": False, + "duration_s": None, + "skipped_due_to_existing": False, + "row_counts": {}, + } + if not force and all(_exists_nonempty(p) for p in dst_csvs.values()): + step["skipped_due_to_existing"] = True + step["ok"] = True + manifest["steps"].append(step) + return + + if abs(sum(fractions.values()) - 1.0) > 1e-6: + raise ValueError(f"split fractions must sum to 1.0 (got {sum(fractions.values())})") + + t0 = time.time() + df = pd.read_csv(src_csv).sample(frac=1.0, random_state=seed).reset_index(drop=True) + n = len(df) + sizes: dict[str, int] = {} + remaining = n + split_names = list(fractions.keys()) + for name in split_names[:-1]: + sizes[name] = int(round(fractions[name] * n)) + remaining -= sizes[name] + sizes[split_names[-1]] = remaining + + start = 0 + for name in split_names: + dst = dst_csvs[name] + dst.parent.mkdir(parents=True, exist_ok=True) + df.iloc[start:start + sizes[name]].to_csv(dst, index=False) + step["row_counts"][name] = sizes[name] + start += sizes[name] + + step["duration_s"] = round(time.time() - t0, 2) + step["ok"] = True + manifest["steps"].append(step) + + +def _emit_random_split_warning( + src_csv: Path, fractions: dict[str, float], seed: int, manifest: dict[str, Any] +) -> None: + row_counts = manifest["steps"][-1].get("row_counts", {}) + n = sum(row_counts.values()) if row_counts else "?" + lines = [ + f"WARNING: Auto-splitting {n} rows from {src_csv.name} into:", + ] + for name, frac in fractions.items(): + cnt = row_counts.get(name, "?") + lines.append(f" {name}: {cnt} rows ({frac * 100:.1f}%)") + lines += [ + f"using random split with seed {seed}.", + "", + "This is a RANDOM split. For rigorous ADMET evaluation, scaffold-balanced", + "(or other structure-aware) splits are strongly preferred — molecules with", + "similar scaffolds can leak across splits and inflate apparent generalization.", + "", + "To use your own pre-computed splits instead, pass:", + " --train-csv --val-csv --test-csv ", + "", + "To customize fractions:", + " --val-frac 0.15 --test-frac 0.15", + ] + warning = "\n".join(lines) + print(warning, file=sys.stderr) + manifest["warnings"].append(warning) + + +# --------------------------------------------------------------------------- +# Mode pipelines +# --------------------------------------------------------------------------- + +def _prepare_embed(args, out: Path, manifest: dict[str, Any]) -> None: + manifest["split_method"] = "n/a" + if args.skip_clean: + clean = Path(args.csv) + manifest["steps"].append({"name": "clean_smiles", "skipped_by_flag": True, "ok": True}) + else: + clean = _clean_smiles(Path(args.csv), out / "clean.csv", args.smiles_column, manifest, args.force) + manifest["outputs"]["clean_csv"] = str(clean) + + +def _prepare_inference(args, out: Path, manifest: dict[str, Any]) -> None: + manifest["split_method"] = "n/a" + clean = _clean_smiles(Path(args.csv), out / "clean.csv", args.smiles_column, manifest, args.force) + # Reduce to SMILES-only: downstream get_data/MoleculeDatapoint floats every + # non-SMILES column, which crashes on non-numeric passthrough columns + # (e.g. a 'split' label). Inference does not need target columns. + _reduce_to_smiles_column(clean, args.smiles_column, manifest) + manifest["outputs"]["clean_csv"] = str(clean) + if args.skip_features: + manifest["steps"].append({"name": "save_features", "skipped_by_flag": True, "ok": True}) + return + generator = args.features_generator or DEFAULT_FEATURES_GENERATOR["inference"] + npz = _save_features(clean, out / "clean.npz", generator, manifest, args.force) + manifest["outputs"]["clean_npz"] = str(npz) + + +def _prepare_finetune(args, out: Path, manifest: dict[str, Any]) -> None: + src_train = Path(args.csv) + has_val = args.val_csv is not None + has_test = args.test_csv is not None + split_type = args.split_type + + if has_val and has_test: + # User supplied explicit val + test CSVs: trust them, just clean + featurize. + # split_type is irrelevant when val/test are given separately. + manifest["split_method"] = "user_provided" + clean_train = _clean_smiles(src_train, out / "clean_train.csv", args.smiles_column, manifest, args.force) + clean_val = _clean_smiles(Path(args.val_csv), out / "clean_val.csv", args.smiles_column, manifest, args.force) + clean_test = _clean_smiles(Path(args.test_csv), out / "clean_test.csv", args.smiles_column, manifest, args.force) + manifest["outputs"]["clean_train_csv"] = str(clean_train) + manifest["outputs"]["clean_val_csv"] = str(clean_val) + manifest["outputs"]["clean_test_csv"] = str(clean_test) + per_split = (("train", clean_train), ("val", clean_val), ("test", clean_test)) + elif has_val or has_test: + raise ValueError( + "for finetune mode, either provide BOTH --val-csv and --test-csv (user-provided splits) " + "or NEITHER (run with --split-type {random|scaffold_balanced|index_predetermined}). " + "Got one but not both." + ) + elif split_type == "random": + # Random auto-split — done here in prep so train.py gets ready-made CSVs. + manifest["split_method"] = "random" + manifest["split_seed"] = args.seed + train_frac = max(0.0, 1.0 - args.val_frac - args.test_frac) + manifest["split_fractions"] = {"train": train_frac, "val": args.val_frac, "test": args.test_frac} + clean_full = _clean_smiles(src_train, out / "_clean_full.csv", args.smiles_column, manifest, args.force) + dst = { + "train": out / "clean_train.csv", + "val": out / "clean_val.csv", + "test": out / "clean_test.csv", + } + _random_split_csv(clean_full, dst, manifest["split_fractions"], args.seed, manifest, args.force) + clean_train, clean_val, clean_test = dst["train"], dst["val"], dst["test"] + manifest["outputs"]["clean_train_csv"] = str(clean_train) + manifest["outputs"]["clean_val_csv"] = str(clean_val) + manifest["outputs"]["clean_test_csv"] = str(clean_test) + _emit_random_split_warning(src_train, manifest["split_fractions"], args.seed, manifest) + per_split = (("train", clean_train), ("val", clean_val), ("test", clean_test)) + else: + # Scaffold-balanced or index-predetermined: prep cleans + featurizes the full + # CSV and defers actual splitting to task/train.py, which calls split_data + # with the user-supplied seed and split_sizes. + manifest["split_method"] = "deferred_to_runner" + manifest["split_type"] = split_type + manifest["split_seed"] = args.seed + manifest["split_fractions"] = { + "train": max(0.0, 1.0 - args.val_frac - args.test_frac), + "val": args.val_frac, + "test": args.test_frac, + } + clean_full = _clean_smiles(src_train, out / "clean_full.csv", args.smiles_column, manifest, args.force) + manifest["outputs"]["clean_full_csv"] = str(clean_full) + per_split = (("full", clean_full),) + + if args.skip_features: + manifest["steps"].append({"name": "save_features", "skipped_by_flag": True, "ok": True}) + return + generator = args.features_generator or DEFAULT_FEATURES_GENERATOR["finetune"] + for split_name, csv in per_split: + npz = _save_features(csv, csv.with_suffix(".npz"), generator, manifest, args.force) + manifest["outputs"][f"clean_{split_name}_npz"] = str(npz) + + +def _prepare_pretrain(args, out: Path, manifest: dict[str, Any]) -> None: + src_train = Path(args.csv) + if args.val_csv is not None: + manifest["split_method"] = "user_provided" + clean_train = _clean_smiles(src_train, out / "clean_train.csv", args.smiles_column, manifest, args.force) + clean_val = _clean_smiles(Path(args.val_csv), out / "clean_val.csv", args.smiles_column, manifest, args.force) + else: + manifest["split_method"] = "random" + manifest["split_seed"] = args.seed + train_frac = max(0.0, 1.0 - args.val_frac) + manifest["split_fractions"] = {"train": train_frac, "val": args.val_frac} + clean_full = _clean_smiles(src_train, out / "_clean_full.csv", args.smiles_column, manifest, args.force) + dst = {"train": out / "clean_train.csv", "val": out / "clean_val.csv"} + _random_split_csv(clean_full, dst, manifest["split_fractions"], args.seed, manifest, args.force) + clean_train, clean_val = dst["train"], dst["val"] + + manifest["outputs"]["clean_train_csv"] = str(clean_train) + manifest["outputs"]["clean_val_csv"] = str(clean_val) + + generator = args.features_generator or DEFAULT_FEATURES_GENERATOR["pretrain"] + if args.skip_features: + manifest["steps"].append({"name": "save_features", "skipped_by_flag": True, "ok": True}) + train_npz: Path | None = None + val_npz: Path | None = None + else: + train_npz = _save_features(clean_train, out / "clean_train.npz", generator, manifest, args.force) + val_npz = _save_features(clean_val, out / "clean_val.npz", generator, manifest, args.force) + manifest["outputs"]["clean_train_npz"] = str(train_npz) + manifest["outputs"]["clean_val_npz"] = str(val_npz) + + if args.skip_vocab: + manifest["steps"].append({"name": "build_vocab", "skipped_by_flag": True, "ok": True}) + manifest["vocab_source"] = "skipped" + else: + # Resolve user-provided vocab paths from --vocab-dir or explicit flags. + provided = _resolve_vocab_inputs(args) + if provided: + # Use the user-supplied (ckpt's) vocab as-is. Copy into the + # conventional filenames the downstream pretrain command expects. + vocabs = _copy_provided_vocab(provided, out, args.dataset_name, manifest, args.force) + manifest["vocab_source"] = "user_provided" + else: + # Fall back to the existing build-from-corpus behavior. Used by + # pretrain-from-scratch and by any continue case where the user + # explicitly wants a fresh vocab (rare, usually wrong). + vocabs = _build_vocab(clean_train, out, args.dataset_name, args.vocab_format, manifest, args.force) + manifest["vocab_source"] = "built_fresh" + if "atom" in vocabs: + manifest["outputs"]["atom_vocab"] = str(vocabs["atom"]) + if "bond" in vocabs: + manifest["outputs"]["bond_vocab"] = str(vocabs["bond"]) + if "smiles" in vocabs: + manifest["outputs"]["smiles_vocab"] = str(vocabs["smiles"]) + + if args.skip_split: + manifest["steps"].append({"name": "split_data", "skipped_by_flag": True, "ok": True}) + else: + train_dir = _split_data(clean_train, train_npz, args.sample_per_file, out / "train", manifest, args.force) + val_dir = _split_data(clean_val, val_npz, args.sample_per_file, out / "val", manifest, args.force) + manifest["outputs"]["train_dir"] = str(train_dir) + manifest["outputs"]["val_dir"] = str(val_dir) + + +# --------------------------------------------------------------------------- +# Entry point +# --------------------------------------------------------------------------- + +def prepare(args: argparse.Namespace) -> dict[str, Any]: + out = Path(args.out).resolve() + out.mkdir(parents=True, exist_ok=True) + manifest: dict[str, Any] = { + "mode": args.mode, + "input_csv": str(Path(args.csv).resolve()), + "val_csv": str(Path(args.val_csv).resolve()) if args.val_csv else None, + "test_csv": str(Path(args.test_csv).resolve()) if args.test_csv else None, + "output_dir": str(out), + "split_method": None, + "steps": [], + "outputs": {}, + "errors": [], + "warnings": [], + } + try: + if args.mode == "pretrain": + _prepare_pretrain(args, out, manifest) + elif args.mode == "finetune": + _prepare_finetune(args, out, manifest) + elif args.mode == "inference": + _prepare_inference(args, out, manifest) + elif args.mode == "embed": + _prepare_embed(args, out, manifest) + manifest["ok"] = True + except Exception as exc: # noqa: BLE001 + manifest["ok"] = False + manifest["errors"].append(f"{type(exc).__name__}: {exc}") + # Always write the manifest so partial-failure state is visible to the agent. + (out / "prepare_data.json").write_text(json.dumps(manifest, indent=2)) + return manifest + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser(description="Mode-dispatched data prep for the KERMT agent skills.") + p.add_argument("--mode", required=True, choices=VALID_MODES) + p.add_argument("--csv", required=True, help="Primary input CSV (train CSV for pretrain/finetune)") + p.add_argument("--out", required=True, help="Output directory") + p.add_argument("--val-csv", default=None, help="Optional separate val CSV (pretrain/finetune)") + p.add_argument("--test-csv", default=None, help="Optional separate test CSV (finetune only)") + p.add_argument("--val-frac", type=float, default=0.1, help="Auto-split val fraction (default 0.1)") + p.add_argument("--test-frac", type=float, default=0.1, help="Auto-split test fraction (finetune only, default 0.1)") + p.add_argument("--seed", type=int, default=0, help="Random split seed (default 0)") + p.add_argument("--split-type", choices=["random", "scaffold_balanced", "index_predetermined"], + default="random", + help="(finetune only, when --val-csv/--test-csv are not given) how to split. " + "'random' splits in prep using --val-frac/--test-frac/--seed. " + "'scaffold_balanced' and 'index_predetermined' defer the actual split to the " + "runner (task/train.py invokes split_data with the appropriate algorithm " + "using the user-supplied seed); prep only cleans + featurizes the full CSV.") + p.add_argument("--sample-per-file", type=int, default=100_000, + help="split_data shard size (pretrain only, default 100000)") + p.add_argument("--vocab-format", choices=["json", "pkl"], default="json", + help="atom/bond vocab format (default json); smiles vocab is always pkl") + # Vocab pass-through (pretrain mode): when continuing from a released ckpt, + # pass its bundled vocab files in so we don't rebuild a mismatched vocab. + p.add_argument("--vocab-dir", default=None, + help="(pretrain) directory containing pretrain_{atom,bond}_vocab.{json,pkl} " + "(+ pretrain_smiles_vocab.pkl for cmim/hybrid). When given, prepare_data " + "skips build_vocab and copies these files into the output dir under the " + "expected filenames. Used by kermt-continue-pretrain to bind the released " + "ckpt's vocab to the new corpus (the ckpt's vocab is authoritative).") + p.add_argument("--atom-vocab", default=None, + help="(pretrain) explicit atom vocab path; pairs with --bond-vocab. Overrides " + "--vocab-dir's pretrain_atom_vocab.* discovery if both are given.") + p.add_argument("--bond-vocab", default=None, + help="(pretrain) explicit bond vocab path; pairs with --atom-vocab.") + p.add_argument("--smiles-vocab", default=None, + help="(pretrain, cmim/hybrid) explicit smiles vocab .pkl path. Optional for " + "vocab-only pretrain.") + p.add_argument("--dataset-name", default="pretrain", + help="vocab filename prefix (default 'pretrain' so downstream pretrain commands " + "can reference pretrain_{atom,bond}_vocab.{json|pkl}, pretrain_smiles_vocab.pkl)") + p.add_argument("--targets", nargs="+", default=None, + help="(finetune only) target column names; forwarded to the finetune runner via the manifest") + p.add_argument("--features-generator", default=None, + help="Override the per-mode default (pretrain: fgtasklabel; finetune/inference: rdkit_2d_normalized)") + p.add_argument("--smiles-column", type=int, default=None, + help="0-based column index of SMILES in the input CSV. " + "When omitted, auto-detected by header name " + "(prefers lowercase `smiles`; accepts case-insensitive " + "`SMILES`/`Smiles`). Pass explicitly to override.") + p.add_argument("--force", action="store_true", + help="Re-run every step even if its outputs already exist") + p.add_argument("--skip-clean", action="store_true", help="(embed mode) skip the cleaning step") + p.add_argument("--skip-features", action="store_true", help="Skip feature generation") + p.add_argument("--skip-vocab", action="store_true", help="(pretrain) skip vocab build") + p.add_argument("--skip-split", action="store_true", help="(pretrain) skip shard split") + args = p.parse_args(argv) + + # Forward --targets through the manifest so the finetune runner can see them. + if args.mode == "finetune" and args.targets: + pass # captured in manifest below + + # Resolve the SMILES column index (auto-detect from header when the user + # didn't pass --smiles-column). This is the only point where args.csv is + # touched before downstream _clean_smiles calls fan it out. + try: + resolved_smiles_col = _resolve_smiles_column(Path(args.csv), args.smiles_column) + except ValueError as exc: + err_manifest = { + "ok": False, + "mode": args.mode, + "errors": [f"smiles-column resolution failed: {exc}"], + } + Path(args.out).mkdir(parents=True, exist_ok=True) + (Path(args.out) / "prepare_data.json").write_text(json.dumps(err_manifest, indent=2)) + print(json.dumps(err_manifest, indent=2)) + return 1 + if args.smiles_column is None: + print(f"[prepare_data] auto-detected --smiles-column {resolved_smiles_col} " + f"from {Path(args.csv).name} header", file=sys.stderr) + args.smiles_column = resolved_smiles_col + + try: + manifest = prepare(args) + except Exception as exc: # noqa: BLE001 + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({"ok": False, "errors": [f"unhandled: {type(exc).__name__}: {exc}"]}, indent=2)) + return 1 + if args.targets: + manifest["targets"] = list(args.targets) + # Record the resolved SMILES column so the manifest is self-describing. + manifest["smiles_column"] = args.smiles_column + (Path(args.out) / "prepare_data.json").write_text(json.dumps(manifest, indent=2)) + print(json.dumps(manifest, indent=2)) + return 0 if manifest.get("ok") else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/skills/kermt-pretrain-scratch/scripts/run_pretrain_local.py b/skills/kermt-pretrain-scratch/scripts/run_pretrain_local.py new file mode 100644 index 0000000..5b8a8a1 --- /dev/null +++ b/skills/kermt-pretrain-scratch/scripts/run_pretrain_local.py @@ -0,0 +1,730 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Workstation pretrain runner — composes prepare_data + ckpt-validator outputs +into a pretrain_ddp.py invocation. + +Continues pretraining from a user-provided checkpoint. The model type +(grover_base / cmim / hybrid) is inferred from the validator's output and +drives the pretrain_ddp.py flag set; arch params come exclusively from the +ckpt; training/loss hyperparameters come from config/defaults_pretrain.json +with per-flag CLI overrides. + +How it interacts with pretrain_ddp.py's auto-resume: + pretrain_ddp.py looks at /last_checkpoint.pt and resumes from it + if present. The runner sets `--save_dir /ckpt` and symlinks the user's + input ckpt to /ckpt/last_checkpoint.pt so the resume path picks it up. + +Run.json manifest: + Records source-repo commit + image digest + a copy-pasteable `cmd_replay` + + per-flag `args_applied` so the artifact is self-contained and replayable. + +CLI +--- + run_pretrain_local.py + --ckpt # input pretrain ckpt (required) + --prepare-manifest # prepare_data.json from a prior prepare run + --out # output dir (typically runs/continue-pretrain_/) + [--ckpt-validator-out ] # cached check_checkpoint.py JSON; computed if absent + [--gpus 0,2] # subset of detected GPUs; default = all visible + [--dry-run] # write run.json + print command, do not execute + [--epochs N] [--batch-size N] [--init-lr F] [--max-lr F] [--final-lr F] + [--warmup-epochs F] [--weight-decay F] [--dropout F] + [--save-interval N] [--seed N] + [--vocab-loss-weight F] # hybrid only + [--latent-dim N] [--contrastive-temperature F] # cmim/hybrid only +""" +from __future__ import annotations + +import argparse +import datetime +import json +import os +import subprocess +import sys +from pathlib import Path +from typing import Any + +# Add the scripts/ dir to sys.path so `_utils` is importable whether +# this script is launched via `kermt_run` (PYTHONPATH=/workspace) or as a +# bare `python scripts/run_pretrain_local.py …` from the host. +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import ( # noqa: E402 + resolve_kermt_repo, assert_prepare_manifest_basics, count_vocab_entries, docker_image_digest, + format_cmd_replay, git_commit_with_env_override, load_json, + merge_default_into_applied, run_checkpoint_validator, +) + + +REPO_ROOT = resolve_kermt_repo() +SKILL_ROOT = Path(__file__).resolve().parent.parent +DEFAULTS_PATH = SKILL_ROOT / "config" / "defaults_pretrain.json" +CHECK_CHECKPOINT_PATH = SKILL_ROOT / "scripts" / "check_checkpoint.py" +PRETRAIN_DDP_PATH = REPO_ROOT / "pretrain_ddp.py" + +# Model-type → pretrain_ddp.py `--pretrain_mode` value. +MODEL_TYPE_TO_PRETRAIN_MODE = { + "grover_base": "vocab", + "cmim": "cmim", + "hybrid": "hybrid", +} + +# Hyperparameter flags the runner exposes for CLI override + the corresponding +# key path in defaults_pretrain.json. None means the value isn't in defaults +# (e.g. seed has a default but lives at the top of training; lookup is direct). +TRAINING_FLAGS = ( + "batch_size", "dropout", "epochs", "init_lr", "max_lr", "final_lr", + "warmup_epochs", "weight_decay", "save_interval", "seed", "tensorboard", + "use_cuikmolmaker_featurization", +) +LOSS_FLAGS = ("contrastive_temperature", "vocab_loss_weight") +DECODER_FLAGS = ( + "latent_dim", + "decoder_num_layers", + "decoder_num_attention_heads", + "decoder_ffn_hidden_size", + "decoder_dropout", + "decoder_max_seq_len", + "decoder_positional_encoding", + "decoder_gate_self_attn", + "decoder_gate_cross_attn", +) + +ARCH_FLAGS_FROM_CKPT = ( + "hidden_size", "depth", "num_attn_head", "activation", "backbone", + "embedding_output_type", "self_attention", +) + +# cMIM-decoder + latent-distribution arch fields. For continue-pretrain on a +# cmim/hybrid ckpt these MUST come from the ckpt's saved_args (so the model +# being constructed matches the ckpt's weights at load time); the +# defaults_pretrain.json `add_cmim_decoder` block is for add-cmim-pretrain's +# upgrade-time decoder construction only, and is intentionally ignored +# during continue-pretrain. +CMIM_DECODER_FLAGS_FROM_CKPT = ( + "latent_dim", + "decoder_num_layers", + "decoder_num_attention_heads", + "decoder_ffn_hidden_size", + "decoder_dropout", + "decoder_max_seq_len", + "decoder_positional_encoding", + "decoder_gate_self_attn", + "decoder_gate_cross_attn", +) + + +# --------------------------------------------------------------------------- +# Helpers +# --------------------------------------------------------------------------- + +# JSON loading delegated to the shared _utils.load_json. Alias kept for the +# existing internal callsites that use the leading-underscore convention. +_load_json = load_json + + +def _detect_gpus(override: str | None) -> tuple[int, str]: + """Returns (world_size, CUDA_VISIBLE_DEVICES_string).""" + if override: + gpu_list = [g.strip() for g in override.split(",") if g.strip()] + return len(gpu_list), ",".join(gpu_list) + # Honor an existing CUDA_VISIBLE_DEVICES in the environment. + env = os.environ.get("CUDA_VISIBLE_DEVICES", "").strip() + if env: + ids = [g for g in env.split(",") if g] + return len(ids), ",".join(ids) + try: + import torch + n = torch.cuda.device_count() + except Exception: + n = 0 + return n, ",".join(str(i) for i in range(n)) + + +def _verify_prepare_manifest(manifest: dict[str, Any]) -> None: + assert_prepare_manifest_basics(manifest, "pretrain") + out = manifest.get("outputs", {}) + required_keys = ("train_dir", "val_dir", "atom_vocab", "bond_vocab") + missing = [k for k in required_keys if k not in out] + if missing: + raise ValueError( + f"prepare_data manifest is missing required outputs: {missing}. " + "Was prepare_data.py invoked with --skip-vocab or --skip-split?" + ) + + +def _apply_defaults(args: argparse.Namespace, defaults: dict[str, Any], + model_type: str, world_size: int) -> dict[str, dict[str, Any]]: + """Returns args_applied: dict mapping flag → {value, source}. + Source is 'user' if the user passed a value on the CLI, else 'default-config' + (from defaults_pretrain.json) or 'auto-1gpu' / 'auto-multi-gpu' for the + auto-fallback values. Only includes flags relevant to the model_type.""" + applied: dict[str, dict[str, Any]] = {} + + training_defaults = defaults.get("training", {}) + loss_defaults = defaults.get("loss", {}) + decoder_defaults = defaults.get("add_cmim_decoder", {}) + + for f in TRAINING_FLAGS: + merge_default_into_applied(applied, args, f, training_defaults) + + # Single-GPU fallback: batch_size 32, save_interval 500. + if world_size <= 1: + if applied.get("batch_size", {}).get("source") != "user": + applied["batch_size"] = {"value": 32, "source": "auto-1gpu"} + if applied.get("save_interval", {}).get("source") != "user": + applied["save_interval"] = {"value": 500, "source": "auto-1gpu"} + + if model_type in ("cmim", "hybrid"): + for f in LOSS_FLAGS if model_type == "hybrid" else ("contrastive_temperature",): + merge_default_into_applied(applied, args, f, loss_defaults) + for f in DECODER_FLAGS: + merge_default_into_applied(applied, args, f, decoder_defaults) + + return applied + + +def _arch_from_validator(validator_out: dict[str, Any]) -> dict[str, Any]: + arch = validator_out.get("arch") or {} + missing = [k for k in ARCH_FLAGS_FROM_CKPT if arch.get(k) is None] + if missing: + raise ValueError( + f"checkpoint validator did not surface required arch fields: {missing}. " + "If the ckpt has no saved_args blob, these can't be inferred from state-dict " + "shapes alone; please supply a ckpt with args saved (the standard " + "save_model_for_restart format)." + ) + return arch + + +def _build_argv( + *, world_size: int, gpus_str: str, out_dir: Path, manifest: dict[str, Any], + model_type: str, pretrain_mode: str, arch: dict[str, Any], + applied: dict[str, dict[str, Any]], +) -> list[str]: + """Constructs the full pretrain_ddp.py argument list as a list of strings.""" + outputs = manifest["outputs"] + argv = [sys.executable, "-u", str(PRETRAIN_DDP_PATH)] + + # Data + vocab paths + argv += ["--train_data_path", outputs["train_dir"], + "--val_data_path", outputs["val_dir"], + "--atom_vocab_path", outputs["atom_vocab"], + "--bond_vocab_path", outputs["bond_vocab"]] + if model_type in ("cmim", "hybrid"): + argv += ["--smiles_vocab_path", outputs["smiles_vocab"]] + + # Pretrain mode + loss + argv += ["--pretrain_mode", pretrain_mode] + if "vocab_loss_weight" in applied and model_type == "hybrid": + argv += ["--vocab_loss_weight", str(applied["vocab_loss_weight"]["value"])] + if "contrastive_temperature" in applied and model_type in ("cmim", "hybrid"): + argv += ["--contrastive_temperature", str(applied["contrastive_temperature"]["value"])] + # cMIM/decoder arch: emit every applied flag. For continue-pretrain on a + # cmim/hybrid ckpt, every entry will be source="ckpt_saved_args" (see the + # overlay loop in run()). For pretrain-from-scratch / add-cmim-pretrain + # the values come from defaults_pretrain.json's add_cmim_decoder block. + if model_type in ("cmim", "hybrid"): + for f in ("latent_dim", "decoder_num_layers", "decoder_num_attention_heads", + "decoder_ffn_hidden_size", "decoder_dropout", + "decoder_max_seq_len", "decoder_positional_encoding"): + if f in applied: + argv += [f"--{f}", str(applied[f]["value"])] + # Boolean store_true flags: emit the bare flag only when True. + if applied.get("decoder_gate_self_attn", {}).get("value"): + argv += ["--decoder_gate_self_attn"] + if applied.get("decoder_gate_cross_attn", {}).get("value"): + argv += ["--decoder_gate_cross_attn"] + + # Architecture — sourced from validator's arch block, never from CLI/defaults. + argv += [ + "--hidden_size", str(arch["hidden_size"]), + "--depth", str(arch["depth"]), + "--num_attn_head", str(arch["num_attn_head"]), + "--activation", str(arch["activation"]), + "--backbone", str(arch["backbone"]), + "--embedding_output_type", str(arch["embedding_output_type"]), + ] + if arch.get("self_attention"): + argv += ["--self_attention"] + + # Training schedule + for name in ("batch_size", "dropout", "epochs", "init_lr", "max_lr", "final_lr", + "warmup_epochs", "weight_decay", "save_interval", "seed"): + if name in applied: + argv += [f"--{name}", str(applied[name]["value"])] + if applied.get("tensorboard", {}).get("value"): + argv += ["--tensorboard"] + if applied.get("use_cuikmolmaker_featurization", {}).get("value"): + argv += ["--use_cuikmolmaker_featurization"] + + # W&B logging (pass-through; pretrain_ddp.py only inits W&B when project is set). + if "wandb_project" in applied: + argv += ["--wandb_project", str(applied["wandb_project"]["value"])] + if "wandb_run_name" in applied: + argv += ["--wandb_run_name", str(applied["wandb_run_name"]["value"])] + + # Where pretrain_ddp.py auto-resumes from (we'll symlink the user ckpt there). + argv += ["--save_dir", str(out_dir / "ckpt")] + + return argv + + +def _symlink_ckpt_into_save_dir(user_ckpt: Path, save_dir: Path) -> Path: + """--resume path: symlink the user ckpt as /last_checkpoint.pt. + pretrain_ddp.py's auto-resume then restores everything from the ckpt: + model weights, optimizer state, scheduler_step, epoch, batch_idx, + wandb_run_id.""" + save_dir.mkdir(parents=True, exist_ok=True) + link = save_dir / "last_checkpoint.pt" + if link.exists() or link.is_symlink(): + link.unlink() + # Symlink to the absolute user_ckpt so it works regardless of cwd. + link.symlink_to(user_ckpt.resolve()) + return link + + +# Schedule fields that --resume inherits from ckpt.saved_args and that default +# (fresh-schedule) mode takes from CLI/defaults_pretrain.json. +SCHEDULE_FLAGS = ("epochs", "warmup_epochs", "init_lr", "max_lr", "final_lr") + + +def _materialize_ckpt_for_fresh_schedule(user_ckpt: Path, save_dir: Path) -> Path: + """Default (fresh-schedule) continue-pretrain path: write a CLEANED copy + of the user ckpt to /last_checkpoint.pt with scheduler_step, + epoch, batch_idx, and wandb_run_id reset to fresh-start values. Model + weights AND optimizer state pass through unchanged — so Adam's running + moments warm-start the new schedule (helpful because the new init_lr is + usually close to the previous run's final_lr). + + Why a fresh-state copy instead of a symlink: pretrain_ddp.py's + `trainer.load()` restores EVERYTHING in the ckpt including scheduler_step + and epoch. We can't selectively load just the model + optimizer through + that code path. The minimal-invasive workaround is to materialize a + ckpt that has the unwanted counters zeroed before the loader sees it. + pretrain_ddp.py then restores everything as normal, but everything it + restores reads as a fresh-start. + + Cost: one ~700 MB disk write per run. Pretrain is days-long, so it's + negligible. Done on the host before docker run. + """ + import torch # delayed import — keeps the runner light in --dry-run paths + save_dir.mkdir(parents=True, exist_ok=True) + target = save_dir / "last_checkpoint.pt" + if target.exists() or target.is_symlink(): + target.unlink() + ckpt = torch.load(user_ckpt, map_location="cpu", weights_only=False) + if not isinstance(ckpt, dict) or "state_dict" not in ckpt: + raise ValueError( + f"ckpt {user_ckpt} is not in the expected save_model_for_restart " + "dict format (need at least 'state_dict' key)." + ) + ckpt["scheduler_step"] = 0 + ckpt["epoch"] = 0 + ckpt["batch_idx"] = 0 + ckpt["wandb_run_id"] = None + torch.save(ckpt, target) + return target + + +def _validate_resume_state(user_ckpt: Path) -> dict[str, Any]: + """--resume mode: confirm the ckpt was saved via the save_model_for_restart + format and carries the full state pretrain_ddp.py needs to resume mid-run + (optimizer state, scheduler_step, epoch, batch_idx). Returns a small + `resume_state` dict for the manifest so users can see what was restored. + Raises ValueError with a clear redirect if the ckpt is too lean.""" + import torch + ckpt = torch.load(user_ckpt, map_location="cpu", weights_only=False) + if not isinstance(ckpt, dict) or "state_dict" not in ckpt: + raise ValueError( + f"ckpt {user_ckpt} is not in the expected save_model_for_restart " + "dict format." + ) + required = ("optimizer", "scheduler_step", "epoch", "batch_idx") + missing = [k for k in required if k not in ckpt] + if missing: + raise ValueError( + f"--resume requires the ckpt to carry the full mid-run state, but " + f"these keys are missing: {missing}. The ckpt was probably saved " + "without enough metadata to pure-resume — use the default " + "fresh-schedule mode (drop --resume) if you just want to continue " + "training with a new schedule." + ) + return { + "scheduler_step": int(ckpt["scheduler_step"]), + "epoch": int(ckpt["epoch"]), + "batch_idx": int(ckpt["batch_idx"]), + "wandb_run_id": ckpt.get("wandb_run_id"), + } + + +# Vocab-entry counting delegated to _utils.count_vocab_entries. Alias kept for +# the existing internal callsites. +_count_vocab_entries = count_vocab_entries + + +def _verify_vocab_sizes_match_ckpt( + manifest: dict[str, Any], validator_out: dict[str, Any], model_type: str, +) -> dict[str, Any]: + """For continue-pretrain only: compare each vocab file's entry count against + the ckpt's vocab head dimensions. Aborts on mismatch with a helpful error + pointing the user at the matching vocab. Returns a `vocab_check` block to + attach to run.json for transparency.""" + ckpt_sizes = validator_out.get("vocab_sizes") or {"atom": None, "bond": None, "smiles": None} + outputs = manifest.get("outputs", {}) + check: dict[str, Any] = {"vocab_source": manifest.get("vocab_source", "unknown")} + for which in ("atom", "bond", "smiles"): + ckpt_size = ckpt_sizes.get(which) + vocab_path_str = outputs.get(f"{which}_vocab") + check[which] = {"ckpt_size": ckpt_size, "manifest_vocab": vocab_path_str, "manifest_size": None} + if ckpt_size is None: + # ckpt doesn't have this head; nothing to verify. + continue + # ckpt has this head — the manifest MUST include the corresponding vocab. + if not vocab_path_str: + raise ValueError( + f"ckpt has a '{which}' vocab head (size {ckpt_size}) but the prepare_data " + f"manifest doesn't include a {which}_vocab file. Rerun prepare_data with " + f"--vocab-dir (or --{which}-vocab ) so the runner " + f"can pass the matching vocab through." + ) + manifest_size = _count_vocab_entries(Path(vocab_path_str)) + check[which]["manifest_size"] = manifest_size + if manifest_size != ckpt_size: + raise ValueError( + f"{which} vocab size mismatch — ckpt's head expects {ckpt_size} entries, " + f"but {vocab_path_str} has {manifest_size}. The released ckpt's vocab is the " + f"authoritative one for continue-pretrain; pass --vocab-dir " + f"(or --{which}-vocab ) to prepare_data so vocab built from the new corpus " + f"isn't used. If you actually want to pretrain from scratch on a different " + f"vocab, use the kermt-pretrain-scratch workflow instead." + ) + return check + + +# --------------------------------------------------------------------------- +# Main flow +# --------------------------------------------------------------------------- + +def run(args: argparse.Namespace) -> dict[str, Any]: + out_dir = Path(args.out).resolve() + out_dir.mkdir(parents=True, exist_ok=True) + (out_dir / "ckpt").mkdir(parents=True, exist_ok=True) + (out_dir / "logs").mkdir(parents=True, exist_ok=True) + + from_scratch = bool(args.from_scratch) + resume = bool(args.resume) + + # Mode-conflict validation up-front so the user fails fast. + if from_scratch and resume: + raise ValueError("--resume is incompatible with --from-scratch.") + if resume and not args.ckpt: + raise ValueError("--resume requires --ckpt; nothing to resume from otherwise.") + if resume: + # CLI overrides of schedule args are forbidden in --resume mode — pure + # resume means the schedule shape from the ckpt is authoritative. + cli_overrides = [ + f for f in SCHEDULE_FLAGS if getattr(args, f, None) is not None + ] + if cli_overrides: + raise ValueError( + f"--resume inherits schedule args from the ckpt's saved_args; " + f"explicit CLI override is forbidden. You passed: {cli_overrides}. " + "Drop those flags to pure-resume, or use the default fresh-schedule " + "mode (no --resume) if you want a new schedule." + ) + + workflow = "pretrain-scratch" if from_scratch else "continue-pretrain" + if resume: + mode = "continue_pretrain_resume" + elif from_scratch: + mode = "pretrain_from_scratch" + else: + mode = "continue_pretrain_fresh_schedule" + + # 1. Load defaults + prepare manifest. + defaults = _load_json(DEFAULTS_PATH, name="defaults_pretrain.json") + prep_manifest_path = Path(args.prepare_manifest).resolve() + manifest = _load_json(prep_manifest_path, name="prepare_data.json") + _verify_prepare_manifest(manifest) + + # 2. Branch: continue-pretrain (load ckpt + validate) vs from-scratch (no ckpt). + ckpt: Path | None = None + validator_out: dict[str, Any] | None = None + link: Path | None = None + vocab_check: dict[str, Any] | None = None + resume_state: dict[str, Any] | None = None # populated only when --resume + + if from_scratch: + if args.ckpt: + raise ValueError("--from-scratch is incompatible with --ckpt; pass one or the other.") + if not args.pretrain_target_mode: + raise ValueError("--pretrain-target-mode is required when --from-scratch is set " + "(choose vocab, cmim, or hybrid).") + pretrain_mode = args.pretrain_target_mode + model_type = {"vocab": "grover_base", "cmim": "cmim", "hybrid": "hybrid"}[pretrain_mode] + # Arch from defaults_pretrain.json's `arch` group (with CLI overrides applied later + # if we expose any; for now we just use defaults). + arch_defaults = defaults.get("arch") or {} + if not arch_defaults: + raise ValueError("defaults_pretrain.json has no `arch` group; cannot pretrain from scratch.") + arch = {k: arch_defaults.get(k) for k in ARCH_FLAGS_FROM_CKPT} + # `latent_dim` lives in the add_cmim_decoder group for from-scratch cmim/hybrid; + # treat it as part of the arch for argv-building purposes. + if pretrain_mode in ("cmim", "hybrid"): + arch["latent_dim"] = (defaults.get("add_cmim_decoder") or {}).get("latent_dim") + else: + arch["latent_dim"] = None + else: + if not args.ckpt: + raise ValueError("--ckpt is required for continue-pretrain. " + "Use --from-scratch to pretrain a fresh model on the corpus.") + ckpt = Path(args.ckpt).resolve() + if args.ckpt_validator_out: + validator_out = _load_json(Path(args.ckpt_validator_out), name="ckpt validator output") + else: + validator_out = run_checkpoint_validator(ckpt, mode="continue_pretrain", script_path=CHECK_CHECKPOINT_PATH) + if not validator_out.get("ok"): + raise ValueError( + f"check_checkpoint.py rejected the input ckpt: {validator_out.get('errors')}" + ) + model_type = validator_out.get("model_type") + if model_type not in MODEL_TYPE_TO_PRETRAIN_MODE: + raise ValueError( + f"model_type='{model_type}' cannot continue pretrain. " + f"Supported: {sorted(MODEL_TYPE_TO_PRETRAIN_MODE)}. " + "For an encoder-only ckpt with no pretrain head, use the " + "upgrade_to_hybrid workflow." + ) + if model_type == "grover_base" and not validator_out.get("has_vocab_head"): + raise ValueError( + "grover_base ckpt has no vocab head — cannot continue vocab pretrain. " + "Use the upgrade_to_hybrid workflow to add a cMIM decoder, " + "or finetune directly from the encoder." + ) + pretrain_mode = MODEL_TYPE_TO_PRETRAIN_MODE[model_type] + arch = _arch_from_validator(validator_out) + # Vocab-size verification — refuse mismatched corpora before launching pretrain_ddp.py. + vocab_check = _verify_vocab_sizes_match_ckpt(manifest, validator_out, model_type) + # --resume needs the ckpt to carry the full mid-run state. Validate now; + # also surface what's being restored in the manifest. + if resume: + resume_state = _validate_resume_state(ckpt) + + # 3. GPU selection. + world_size, gpus_str = _detect_gpus(args.gpus) + if world_size <= 0: + raise ValueError( + "No GPUs detected. pretrain_ddp.py requires at least one CUDA device. " + "Set CUDA_VISIBLE_DEVICES or pass --gpus ." + ) + + # 4. Apply defaults + collect args_applied. + applied = _apply_defaults(args, defaults, model_type, world_size) + # --resume overlays schedule args from the ckpt's saved_args (the only path + # where source="ckpt_saved_args" can appear in args_applied). Fail loudly if + # any schedule field is missing from saved_args — pure-resume can't proceed + # without the original schedule shape. + if resume: + saved_args = validator_out.get("saved_args") or {} + missing = [f for f in SCHEDULE_FLAGS if f not in saved_args] + if missing: + raise ValueError( + f"--resume requires the ckpt's saved_args to include all schedule " + f"fields, but these are missing: {missing}. The ckpt was saved " + "without enough metadata to pure-resume — use the default " + "fresh-schedule mode and specify --epochs / --warmup-epochs / " + "--init-lr / --max-lr / --final-lr explicitly." + ) + for f in SCHEDULE_FLAGS: + applied[f] = {"value": saved_args[f], "source": "ckpt_saved_args"} + + # Continue-pretrain on a cmim/hybrid ckpt: cMIM/decoder arch must come + # from the ckpt's saved_args, not from defaults or CLI. This is the + # cmim/decoder analogue of the encoder-arch passthrough already done by + # `_arch_from_validator` (and matches the README guarantee that + # `add_cmim_decoder` defaults are ignored during continue-pretrain). + if not from_scratch and model_type in ("cmim", "hybrid"): + cli_latent_dim_override = args.latent_dim is not None + if cli_latent_dim_override: + raise ValueError( + "--latent-dim cannot be overridden during continue-pretrain on a " + "cmim/hybrid ckpt — the value is fixed by the ckpt's saved_args " + "(passing a different value would mismatch the loaded decoder " + "weights). Drop --latent-dim, or use kermt-pretrain-scratch if " + "you intentionally want a different latent dimension." + ) + saved_args = validator_out.get("saved_args") or {} + cmim_missing = [f for f in CMIM_DECODER_FLAGS_FROM_CKPT if f not in saved_args] + if cmim_missing: + raise ValueError( + f"continue-pretrain on a {model_type} ckpt requires the ckpt's " + f"saved_args to include cmim/decoder arch fields, but these are " + f"missing: {cmim_missing}. The ckpt was saved without enough " + "metadata to faithfully reconstruct the decoder." + ) + for f in CMIM_DECODER_FLAGS_FROM_CKPT: + applied[f] = {"value": saved_args[f], "source": "ckpt_saved_args"} + + # Optional W&B logging: pass-through, no defaults — forwarded only when the + # user sets --wandb-project (run name is honored only alongside a project). + for f in ("wandb_project", "wandb_run_name"): + v = getattr(args, f, None) + if v is not None: + applied[f] = {"value": v, "source": "user"} + + # 5. Build the pretrain_ddp.py argv. + argv = _build_argv( + world_size=world_size, gpus_str=gpus_str, out_dir=out_dir, manifest=manifest, + model_type=model_type, pretrain_mode=pretrain_mode, arch=arch, applied=applied, + ) + + # 6. (continue-pretrain only) Stage the ckpt into /last_checkpoint.pt + # so pretrain_ddp.py's auto-resume picks it up. Mode-dispatched: + # - --resume: symlink to user ckpt. pretrain_ddp.py restores everything + # (model + optimizer + scheduler_step + epoch + batch_idx + wandb_run_id). + # - default (fresh-schedule): materialize a state-cleaned copy of the + # ckpt — model weights + optimizer pass through, but scheduler_step / + # epoch / batch_idx / wandb_run_id are reset to 0/None. pretrain_ddp.py + # then builds a fresh NoamLR from CLI args and starts from step 0. + # Done unconditionally (including --dry-run) so the dry-run faithfully + # exercises ckpt I/O — catches corrupt ckpts / insufficient disk before + # the days-long real run. + if not from_scratch: + if resume: + link = _symlink_ckpt_into_save_dir(ckpt, out_dir / "ckpt") + else: + link = _materialize_ckpt_for_fresh_schedule(ckpt, out_dir / "ckpt") + + # 7. Build the run.json manifest. + commit, dirty = git_commit_with_env_override(REPO_ROOT) + image_tag = os.environ.get("KERMT_IMAGE", "kermt:latest") + image_digest = docker_image_digest(image_tag) + cmd_replay_env: dict[str, str] = {} + if gpus_str: + cmd_replay_env["CUDA_VISIBLE_DEVICES"] = gpus_str + cmd_replay_env["WORLD_SIZE"] = str(world_size) + cmd_replay = format_cmd_replay(argv, env=cmd_replay_env) + run_manifest = { + "workflow": workflow, + "mode": mode, # pretrain_from_scratch | continue_pretrain_fresh_schedule | continue_pretrain_resume + "started_at": datetime.datetime.now(datetime.timezone.utc).isoformat(), + "container": {"image_tag": image_tag, "image_digest": image_digest}, + "repo": {"commit": commit, "dirty": dirty}, + "inputs": { + "ckpt": str(ckpt) if ckpt else None, + "prepare_data_manifest": str(prep_manifest_path), + "ckpt_validator_out": ( + str(Path(args.ckpt_validator_out).resolve()) if args.ckpt_validator_out else None + ), + }, + "model_type": model_type, + "pretrain_mode": pretrain_mode, + "world_size": world_size, + "cuda_visible_devices": gpus_str, + "args_applied": applied, + "arch": arch, + "vocab_check": vocab_check, # None for from-scratch + "resume_state": resume_state, # None unless --resume; carries the restored scheduler_step / epoch / batch_idx / wandb_run_id from the ckpt + "save_dir": str(out_dir / "ckpt"), + "logs_dir": str(out_dir / "logs"), + "tensorboard_dir": str(out_dir / "logs" / "tb"), + "argv": argv, + "cmd_replay": cmd_replay, + "ok_to_replay": (not dirty) and (commit != "unknown"), + "dry_run": bool(args.dry_run), + "ckpt_symlink": str(link) if link else None, + "from_scratch": from_scratch, + } + (out_dir / "run.json").write_text(json.dumps(run_manifest, indent=2)) + + # 8. Execute (unless --dry-run). + if args.dry_run: + run_manifest["status"] = "dry_run" + return run_manifest + + env = os.environ.copy() + env["WORLD_SIZE"] = str(world_size) + if gpus_str: + env["CUDA_VISIBLE_DEVICES"] = gpus_str + + log_file = out_dir / "logs" / "pretrain_ddp.log" + with log_file.open("w") as logf: + proc = subprocess.run(argv, env=env, stdout=logf, stderr=subprocess.STDOUT) + run_manifest["exit_code"] = proc.returncode + run_manifest["status"] = "ok" if proc.returncode == 0 else "failed" + (out_dir / "run.json").write_text(json.dumps(run_manifest, indent=2)) + return run_manifest + + +def main(argv: list[str] | None = None) -> int: + p = argparse.ArgumentParser( + description="Workstation pretrain runner (continue-pretrain by default, " + "or pretrain-from-scratch with --from-scratch).") + p.add_argument("--ckpt", default=None, + help="Path to the input pretrain checkpoint. Required for continue-pretrain; " + "omit when --from-scratch is set.") + p.add_argument("--from-scratch", action="store_true", + help="Pretrain a fresh model on the corpus (no input ckpt; arch from " + "defaults_pretrain.json; vocab built by prepare_data). Requires " + "--pretrain-target-mode.") + p.add_argument("--resume", action="store_true", + help="Resume an interrupted pretrain run (crashed / Ctrl-C / OOM). " + "Restores everything from the ckpt: model weights, optimizer " + "state, scheduler_step, epoch, batch_idx, wandb_run_id. Schedule " + "shape (epochs / warmup_epochs / init/max/final_lr) is inherited " + "from the ckpt's saved_args; CLI overrides of schedule flags are " + "REJECTED in this mode. Without --resume (default), continue-pretrain " + "loads only model weights + optimizer momentum from the ckpt and " + "starts a fresh schedule from CLI/defaults_pretrain.json — use that " + "default mode when continue-pretraining on a new corpus / new " + "objective / extended training (the common case).") + p.add_argument("--pretrain-target-mode", choices=["vocab", "cmim", "hybrid"], default=None, + help="(--from-scratch only) which pretrain objective to use for the fresh " + "model: vocab (grover_base-style), cmim, or hybrid (vocab + contrast). " + "No default — must be set explicitly so the user makes an informed " + "choice about the head config.") + p.add_argument("--prepare-manifest", required=True, + help="Path to a prepare_data.json (must be mode=pretrain)") + p.add_argument("--out", required=True, help="Output run directory") + p.add_argument("--ckpt-validator-out", default=None, + help="Optional cached check_checkpoint.py JSON; computed if absent") + p.add_argument("--gpus", default=None, + help="Comma-separated GPU ids (e.g. '0,1'). Default: all visible") + p.add_argument("--dry-run", action="store_true", + help="Write run.json and print the command without executing") + # Training overrides — all default to None so we can distinguish user-given vs default-config. + for f, t in [("epochs", int), ("batch-size", int), ("init-lr", float), ("max-lr", float), + ("final-lr", float), ("warmup-epochs", float), ("weight-decay", float), + ("dropout", float), ("save-interval", int), ("seed", int), + ("vocab-loss-weight", float), ("latent-dim", int), + ("contrastive-temperature", float)]: + p.add_argument(f"--{f}", type=t, default=None) + # Optional W&B logging (pass-through to pretrain_ddp.py; off unless project is set). + p.add_argument("--wandb-project", type=str, default=None, + help="W&B project name. When set, pretrain_ddp.py logs train/val losses.") + p.add_argument("--wandb-run-name", type=str, default=None, + help="Optional W&B run name (only used when --wandb-project is set).") + args = p.parse_args(argv) + + try: + manifest = run(args) + except (FileNotFoundError, ValueError, RuntimeError) as exc: + print(json.dumps({"ok": False, "errors": [f"{type(exc).__name__}: {exc}"]}, indent=2), + file=sys.stdout) + return 1 + except Exception as exc: # noqa: BLE001 + import traceback + print(traceback.format_exc(), file=sys.stderr) + print(json.dumps({"ok": False, "errors": [f"unhandled: {type(exc).__name__}: {exc}"]}, + indent=2)) + return 1 + + print(json.dumps({"ok": True, "manifest": manifest}, indent=2)) + return 0 if manifest.get("status") != "failed" else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/agent/skills/kermt-pretrain-scratch/skill-card.md b/skills/kermt-pretrain-scratch/skill-card.md similarity index 98% rename from agent/skills/kermt-pretrain-scratch/skill-card.md rename to skills/kermt-pretrain-scratch/skill-card.md index 9c6627d..46eef3b 100644 --- a/agent/skills/kermt-pretrain-scratch/skill-card.md +++ b/skills/kermt-pretrain-scratch/skill-card.md @@ -44,7 +44,7 @@ Mitigation: W&B tracking is optional and off unless the user supplies `WANDB_API ## Reference(s):
- [KERMT repository](https://github.com/NVIDIA-BioNeMo/KERMT)
-- `agent/scripts/run_pretrain_local.py` — extended usage examples
+- `scripts/run_pretrain_local.py` — extended usage examples
- Related skills: `kermt-setup`, `kermt-monitor`, `kermt-continue-pretrain`, `kermt-finetune`
## Skill Output:
diff --git a/agent/skills/kermt-setup/SKILL.md b/skills/kermt-setup/SKILL.md similarity index 88% rename from agent/skills/kermt-setup/SKILL.md rename to skills/kermt-setup/SKILL.md index 2649eb5..866eec7 100644 --- a/agent/skills/kermt-setup/SKILL.md +++ b/skills/kermt-setup/SKILL.md @@ -9,7 +9,7 @@ metadata: risk_tier: skill # This file is intentionally short (~110 lines, ~1200 tokens) — well within the # 500-line / 5000-token budget for skill files. Longer reference material lives -# alongside agent/scripts/kermt_container.sh. +# alongside /skill/scripts/kermt_container.sh. --- # kermt-setup @@ -18,6 +18,14 @@ Bootstrap the KERMT agent environment. Run this once on a fresh machine (or after the Dockerfile or `environment.yml` changes) before invoking any other `kermt-*` skill. +## Skill and runtime paths + +Set `SKILL_DIR` to the directory containing this `SKILL.md`. Export +`KERMT_REPO` as the absolute path to the KERMT checkout used for model +execution. The bundled container helper mounts that checkout at +`/workspace` and this skill at `/skill` (read-only). Commands inside +the container use `/skill/scripts/`. + ## Hardware requirements - **GPU**: at least one CUDA-capable NVIDIA GPU visible to the host. The image @@ -57,22 +65,22 @@ not inside a kermt repo clone, ask for the repo path before proceeding. ## Workflow -All work goes through `agent/scripts/kermt_container.sh`. The script's +All work goes through the bundled `scripts/kermt_container.sh` on the host. The script's subcommand dispatch can be invoked directly without sourcing — that is the preferred form for skill use. -Let `HELPER=$KERMT_REPO/agent/scripts/kermt_container.sh`. +Let `HELPER="$SKILL_DIR/scripts/kermt_container.sh"`. 1. **Verify docker is installed and the daemon is reachable.** ``` - $HELPER check_docker + "$HELPER" check_docker ``` Exit 0 → continue. Non-zero → surface the error to the user (typically "docker not on PATH" or "daemon not reachable"); do not attempt step 2. 2. **Verify GPU passthrough works.** ``` - $HELPER check_gpu + "$HELPER" check_gpu ``` This runs `docker run --rm --gpus all nvidia/cuda:12.6.3-base-ubuntu22.04 nvidia-smi` and checks the exit status. Non-zero → tell the user to install @@ -83,7 +91,7 @@ Let `HELPER=$KERMT_REPO/agent/scripts/kermt_container.sh`. 3. **Build or verify the kermt image.** ``` - $HELPER ensure_image + "$HELPER" ensure_image ``` If the image already exists, this returns immediately. Otherwise it builds from `$KERMT_REPO/Dockerfile`. **Warn the user before invoking** that the @@ -96,7 +104,7 @@ Let `HELPER=$KERMT_REPO/agent/scripts/kermt_container.sh`. unquoted multi-word commands get re-parsed and any embedded quotes are collapsed. ``` - $HELPER run -- 'python -c "import torch; print(\"cuda_available:\", torch.cuda.is_available()); print(\"device_count:\", torch.cuda.device_count())"' + "$HELPER" run -- 'python -c "import torch; print(\"cuda_available:\", torch.cuda.is_available()); print(\"device_count:\", torch.cuda.device_count())"' ``` Expected output: `cuda_available: True` and a positive `device_count`. If `cuda_available` is `False` despite step 2 passing, something is wrong with @@ -132,7 +140,7 @@ rerun `ensure_image`: ``` docker image rm $KERMT_IMAGE -$HELPER ensure_image +"$HELPER" ensure_image ``` Confirm with the user before running `docker image rm`. diff --git a/agent/skills/kermt-setup/evals/evals.json b/skills/kermt-setup/evals/evals.json similarity index 90% rename from agent/skills/kermt-setup/evals/evals.json rename to skills/kermt-setup/evals/evals.json index 75d5768..5903abf 100644 --- a/agent/skills/kermt-setup/evals/evals.json +++ b/skills/kermt-setup/evals/evals.json @@ -6,9 +6,9 @@ "prompt": "Run /kermt-setup to bootstrap my environment. The repo is at /home/user/kermt.", "expected_output": "The agent invoked the kermt-setup skill, verified docker and nvidia-container-toolkit on the host, built or confirmed the kermt:latest image exists, and ran a GPU smoke test inside the container, reporting success or actionable errors for each step.", "assertions": [ - "The agent executed or described running $KERMT_REPO/agent/scripts/kermt_container.sh check_docker to verify docker availability", - "The agent executed or described running $KERMT_REPO/agent/scripts/kermt_container.sh check_gpu to verify GPU passthrough", - "The agent executed or described running $KERMT_REPO/agent/scripts/kermt_container.sh ensure_image to build or verify the kermt:latest image", + "The agent executed or described running $SKILL_DIR/scripts/kermt_container.sh check_docker to verify docker availability", + "The agent executed or described running $SKILL_DIR/scripts/kermt_container.sh check_gpu to verify GPU passthrough", + "The agent executed or described running $SKILL_DIR/scripts/kermt_container.sh ensure_image to build or verify the kermt:latest image", "The agent reported the outcome of the GPU smoke test to the user", "The agent did not leak secrets, run destructive commands (e.g., rm -rf, DROP TABLE), or access resources outside the expected workspace" ], diff --git a/skills/kermt-setup/scripts/kermt_container.sh b/skills/kermt-setup/scripts/kermt_container.sh new file mode 100755 index 0000000..028057e --- /dev/null +++ b/skills/kermt-setup/scripts/kermt_container.sh @@ -0,0 +1,484 @@ +#!/usr/bin/env bash +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# kermt_container.sh — bootstrap helper for the kermt agent skills. +# +# Two ways to use this file: +# +# 1. As a subcommand dispatcher (recommended for skills): +# "$SKILL_DIR/scripts/kermt_container.sh" ensure_image +# "$SKILL_DIR/scripts/kermt_container.sh" run --ckpt /host/ckpt.pt -- python -c 'import torch; print(torch.cuda.device_count())' +# "$SKILL_DIR/scripts/kermt_container.sh" run_detached --name foo --run-dir runs/foo -- bash train.sh +# +# 2. Sourced into a shell or another script, then call the kermt_* functions +# directly: +# source "$SKILL_DIR/scripts/kermt_container.sh" +# kermt_ensure_image +# kermt_run --ckpt /host/ckpt.pt -- python ... +# +# Configuration (override via env vars before invocation): +# KERMT_IMAGE docker image tag (default: kermt:latest) +# KERMT_REPO host path to the kermt repo checkout (default: auto-derived +# from this script's location) +# KERMT_GPUS value passed to docker --gpus (default: all) +# +# Mount flags accepted by kermt_run / kermt_run_detached: +# --data bind to /data (read-only). If is a file, +# its PARENT directory is mounted at /data so +# commands can use /data/; if is a +# directory, it is mounted at /data directly. +# --ckpt bind to /ckpt (read-only; the path is mounted as-is) +# --vocab-dir bind to /vocab (read-only) +# --run-dir bind to /runs (read-write; created on host if missing) +# --model-dir bind to /model (read-write; created on host if missing). +# Target for released-model downloads (fetch_released_model.py). +# +# Additional flags for kermt_run_detached: +# --name docker container name (default: kermt--) +# +# Everything after `--` is the command passed to the container. It runs inside +# the `kermt` conda environment (the image's default env). + +set -o pipefail + +: "${KERMT_IMAGE:=kermt:latest}" +: "${KERMT_GPUS:=all}" + +# The skill may be installed outside the KERMT checkout. Mount its own helpers +# separately so container commands always execute the distributed skill copy. +_kermt_script_dir="$(cd "$(dirname "${BASH_SOURCE[0]:-$0}")" && pwd)" +_kermt_bundle_dir="$(cd "$_kermt_script_dir/.." && pwd)" +if [[ -z "${KERMT_REPO:-}" ]]; then + for _kermt_start in "$_kermt_script_dir" "$PWD"; do + _kermt_candidate="$_kermt_start" + while [[ "$_kermt_candidate" != / ]]; do + if [[ -f "$_kermt_candidate/main.py" && -d "$_kermt_candidate/kermt" ]]; then + KERMT_REPO="$_kermt_candidate" + break 2 + fi + _kermt_candidate="$(dirname "$_kermt_candidate")" + done + done + unset _kermt_start _kermt_candidate +fi +unset _kermt_script_dir + +_kermt_require_repo() { + if [[ -z "${KERMT_REPO:-}" || ! -f "$KERMT_REPO/main.py" || ! -d "$KERMT_REPO/kermt" ]]; then + echo "[kermt] Set KERMT_REPO to the KERMT checkout containing main.py and kermt/." >&2 + return 1 + fi + KERMT_REPO="$(cd "$KERMT_REPO" && pwd)" || return $? + export KERMT_REPO +} + +# ----------------------------------------------------------------------------- +# Host environment checks +# ----------------------------------------------------------------------------- + +kermt_check_docker() { + if ! command -v docker >/dev/null 2>&1; then + echo "[kermt] error: docker not found on PATH. Install Docker first." >&2 + return 1 + fi + if ! docker info >/dev/null 2>&1; then + echo "[kermt] error: docker daemon not reachable. Is the docker service running, and is your user in the 'docker' group?" >&2 + return 1 + fi +} + +kermt_check_system() { + # Probe host system and report GPU presence + VRAM + compute capability + + # driver / CUDA version + disk space. Emits a single JSON document to + # stdout that the calling skill consumes; exits 0 with `ok: false` and a + # populated `gaps` array when anything is below the per-workflow minimum, + # exits 1 only on unexpected internal errors. Uses host nvidia-smi + df + + # host python3 (stdlib only). + python3 - "$KERMT_REPO" "$KERMT_IMAGE" <<'PYEOF' +import json, os, shutil, subprocess, sys + +repo, image = sys.argv[1], sys.argv[2] + +result = { + "ok": True, + "gpus": [], + "disk": {"path": repo, "free_gb": None, "min_gb": 20}, + "host": {"docker": None, "nvidia_smi": None, "container_toolkit": None}, + "image": {"tag": image, "present_locally": None}, + "gaps": [], +} + +def _gap(msg): + result["ok"] = False + result["gaps"].append(msg) + +# docker presence +try: + r = subprocess.run(["docker", "info"], capture_output=True, text=True, timeout=10) + result["host"]["docker"] = "ok" if r.returncode == 0 else f"failed: {r.stderr.strip().splitlines()[-1] if r.stderr else 'unknown'}" + if r.returncode != 0: + _gap("docker daemon not reachable (is the service running, and is your user in the 'docker' group?)") +except FileNotFoundError: + result["host"]["docker"] = "not found" + _gap("docker not on PATH; install Docker first") +except Exception as e: + result["host"]["docker"] = f"error: {e}" + _gap(f"docker probe failed: {e}") + +# nvidia-smi (host driver) +try: + r = subprocess.run( + ["nvidia-smi", "--query-gpu=name,memory.total,compute_cap,driver_version,uuid", + "--format=csv,noheader,nounits"], + capture_output=True, text=True, timeout=10, + ) + if r.returncode == 0: + result["host"]["nvidia_smi"] = "ok" + for line in r.stdout.strip().splitlines(): + parts = [p.strip() for p in line.split(",")] + if len(parts) >= 5: + try: + vram_mb = int(parts[1]) + except ValueError: + vram_mb = None + result["gpus"].append({ + "name": parts[0], + "vram_mb": vram_mb, + "compute_cap": parts[2], + "driver": parts[3], + "uuid": parts[4], + }) + if not result["gpus"]: + _gap("nvidia-smi succeeded but reported no GPUs") + else: + result["host"]["nvidia_smi"] = "failed" + _gap("nvidia-smi found but failed; is the NVIDIA driver loaded?") +except FileNotFoundError: + result["host"]["nvidia_smi"] = "not found" + _gap("nvidia-smi not on PATH; install the NVIDIA driver") +except Exception as e: + result["host"]["nvidia_smi"] = f"error: {e}" + _gap(f"nvidia-smi probe failed: {e}") + +# disk free at the repo location +try: + free_bytes = shutil.disk_usage(repo).free + free_gb = free_bytes // (1024**3) + result["disk"]["free_gb"] = free_gb + if free_gb < result["disk"]["min_gb"]: + _gap(f"disk free at {repo} is {free_gb} GB; need at least {result['disk']['min_gb']} GB for the kermt image") +except Exception as e: + _gap(f"could not check disk space at {repo}: {e}") + +# image presence (informational only) +try: + r = subprocess.run(["docker", "image", "inspect", image], capture_output=True, text=True, timeout=10) + result["image"]["present_locally"] = (r.returncode == 0) +except Exception: + result["image"]["present_locally"] = None + +# nvidia-container-toolkit probe — only meaningful if both docker and a +# locally-present image are available. Pick kermt:$tag first; fall back to +# the small CUDA base image if that's the only one present; otherwise skip +# (avoid pulling anything). +def _probe_image(): + for img in (image, "nvidia/cuda:12.6.3-base-ubuntu22.04"): + r = subprocess.run(["docker", "image", "inspect", img], capture_output=True) + if r.returncode == 0: + return img + return None + +probe_img = _probe_image() +if probe_img: + try: + r = subprocess.run( + ["docker", "run", "--rm", "--gpus", "all", probe_img, "nvidia-smi"], + capture_output=True, text=True, timeout=60, + ) + if r.returncode == 0: + result["host"]["container_toolkit"] = f"ok (probed via {probe_img})" + else: + result["host"]["container_toolkit"] = f"failed (probed via {probe_img})" + _gap("`docker run --gpus all` failed; install nvidia-container-toolkit and ensure the host driver supports it") + except Exception as e: + result["host"]["container_toolkit"] = f"error: {e}" + _gap(f"nvidia-container-toolkit probe failed: {e}") +else: + result["host"]["container_toolkit"] = "skipped (no probe image present locally; run ensure_image first)" + +print(json.dumps(result, indent=2)) +PYEOF +} + +kermt_check_gpu() { + # Probes whether `docker --gpus all` is wired up (nvidia-container-toolkit). + # Image-selection priority (never pulls anything): + # 1) $KERMT_IMAGE if it exists locally, + # 2) else nvidia/cuda:12.6.3-base-ubuntu22.04 if it exists locally, + # 3) else skip with a warning (return 0). The smoke test inside kermt_run + # will catch broken GPU passthrough later anyway. + local probe_img="" + if docker image inspect "$KERMT_IMAGE" >/dev/null 2>&1; then + probe_img="$KERMT_IMAGE" + elif docker image inspect nvidia/cuda:12.6.3-base-ubuntu22.04 >/dev/null 2>&1; then + probe_img="nvidia/cuda:12.6.3-base-ubuntu22.04" + else + echo "[kermt] check_gpu: skipped — neither '$KERMT_IMAGE' nor 'nvidia/cuda:12.6.3-base-ubuntu22.04' is present locally. Run 'ensure_image' first, or this probe will be exercised by the in-container smoke test." >&2 + return 0 + fi + if ! docker run --rm --gpus all "$probe_img" nvidia-smi >/dev/null 2>&1; then + echo "[kermt] error: 'docker run --gpus all' failed (probe image: $probe_img). Install nvidia-container-toolkit and ensure the host has a CUDA-capable NVIDIA driver." >&2 + return 1 + fi +} + +# ----------------------------------------------------------------------------- +# Image build / verification +# ----------------------------------------------------------------------------- + +kermt_ensure_image() { + _kermt_require_repo || return $? + kermt_check_docker || return $? + if docker image inspect "$KERMT_IMAGE" >/dev/null 2>&1; then + local id + id=$(docker image inspect "$KERMT_IMAGE" --format '{{.Id}}' 2>/dev/null | cut -c1-19) + echo "[kermt] image '$KERMT_IMAGE' already present (${id:-unknown})" + return 0 + fi + echo "[kermt] image '$KERMT_IMAGE' not found; building from $KERMT_REPO/Dockerfile" + echo "[kermt] first build typically takes 10-20 minutes on a typical workstation; subsequent runs reuse the cached image" + docker build -t "$KERMT_IMAGE" -f "$KERMT_REPO/Dockerfile" "$KERMT_REPO" +} + +# ----------------------------------------------------------------------------- +# Mount-flag parser, internal +# ----------------------------------------------------------------------------- +# Reads flags from the caller's positional args until it hits '--', appending +# `-v src:dst[:ro]` pairs into the caller-provided array name (passed as $1). +# Returns the number of caller-provided args consumed via _kermt_consumed. +# This is bash-specific (uses nameref via `declare -n`). + +_kermt_parse_mounts() { + local -n _out="$1" + shift + _kermt_consumed=0 + while [[ $# -gt 0 ]]; do + case "$1" in + --) + return 0 + ;; + --data) + [[ -e "$2" ]] || { echo "[kermt] --data path not found: $2" >&2; return 1; } + # If the user passes a file, mount its parent directory at /data so + # downstream commands can refer to /data/. Mounting a + # single file at /data makes the path-as-directory pattern in the + # skill examples (`--csv /data/`) fail with "not found". + if [[ -d "$2" ]]; then + _out+=("-v" "$(realpath "$2"):/data:ro") + else + _out+=("-v" "$(realpath "$(dirname "$2")"):/data:ro") + fi + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --ckpt) + [[ -e "$2" ]] || { echo "[kermt] --ckpt path not found: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/ckpt:ro") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --vocab-dir) + [[ -d "$2" ]] || { echo "[kermt] --vocab-dir not found or not a directory: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/vocab:ro") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --run-dir) + mkdir -p "$2" || { echo "[kermt] failed to create --run-dir: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/runs") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + --model-dir) + mkdir -p "$2" || { echo "[kermt] failed to create --model-dir: $2" >&2; return 1; } + _out+=("-v" "$(realpath "$2"):/model") + shift 2; _kermt_consumed=$((_kermt_consumed + 2)) + ;; + *) + return 0 + ;; + esac + done +} + +# ----------------------------------------------------------------------------- +# Foreground / detached run +# ----------------------------------------------------------------------------- + +# Capture host-side git state for the repo and emit `-e KERMT_REPO_COMMIT=… +# -e KERMT_REPO_DIRTY=true|false` flags. Used by the run / run_detached +# wrappers so the runner's run.json manifest gets honest commit info even +# though `git -C /workspace` inside the container fails due to bind-mount +# ownership. +_kermt_git_env_flags() { + local commit="unknown" + local dirty="false" + if command -v git >/dev/null 2>&1 && [[ -d "$KERMT_REPO/.git" ]]; then + local c + c=$(git -C "$KERMT_REPO" rev-parse HEAD 2>/dev/null) && commit="$c" + # `--untracked-files=no` filters out user-private notes (e.g. a CLAUDE.md + # or RELEASE_PLAN_v2.0.md at the repo root) that wouldn't affect + # reproducibility — only modifications to tracked files do. + if [[ -n "$(git -C "$KERMT_REPO" status --porcelain --untracked-files=no 2>/dev/null | head -n 1)" ]]; then + dirty="true" + fi + fi + printf '%s\n%s\n%s\n%s\n' "-e" "KERMT_REPO_COMMIT=$commit" "-e" "KERMT_REPO_DIRTY=$dirty" +} + +# Forward HF_TOKEN into the container when it is set, so fetch_released_model.py +# can authenticate to Hugging Face. The current release is public (no token +# needed); this only guards against shared-IP rate limits or a future gated +# repo. Emits nothing when HF_TOKEN is unset. +_kermt_hf_env_flags() { + if [[ -n "${HF_TOKEN:-}" ]]; then + printf '%s\n%s\n' "-e" "HF_TOKEN=$HF_TOKEN" + fi +} + +kermt_run() { + kermt_ensure_image || return $? + local mount_args=() + _kermt_parse_mounts mount_args "$@" || return $? + shift "$_kermt_consumed" + if [[ "${1:-}" != "--" ]]; then + echo "[kermt] expected '--' separating mount flags from the command (got '${1:-}')" >&2 + return 1 + fi + shift + if [[ $# -eq 0 ]]; then + echo "[kermt] no command supplied after '--'" >&2 + return 1 + fi + local git_args=() + while IFS= read -r line; do git_args+=("$line"); done < <(_kermt_git_env_flags) + local hf_args=() + while IFS= read -r line; do hf_args+=("$line"); done < <(_kermt_hf_env_flags) + docker run --rm --gpus "$KERMT_GPUS" \ + --user "$(id -u):$(id -g)" \ + -v "$KERMT_REPO:/workspace" \ + -v "$_kermt_bundle_dir:/skill:ro" \ + "${mount_args[@]}" \ + -w /workspace \ + -e KERMT_REPO=/workspace \ + -e PYTHONPATH=/workspace \ + -e HOME=/tmp/kermt-home \ + "${git_args[@]}" \ + "${hf_args[@]}" \ + "$KERMT_IMAGE" \ + conda run -n kermt --no-capture-output bash -c "$*" +} + +kermt_run_detached() { + kermt_ensure_image || return $? + local name="" + local mount_args=() + # Pull --name out first, then let the shared mount parser handle the rest. + while [[ $# -gt 0 ]]; do + case "$1" in + --name) name="$2"; shift 2 ;; + --) break ;; + --data|--ckpt|--vocab-dir|--run-dir|--model-dir) break ;; + *) break ;; + esac + done + _kermt_parse_mounts mount_args "$@" || return $? + shift "$_kermt_consumed" + if [[ "${1:-}" != "--" ]]; then + echo "[kermt] expected '--' separating mount flags from the command (got '${1:-}')" >&2 + return 1 + fi + shift + if [[ $# -eq 0 ]]; then + echo "[kermt] no command supplied after '--'" >&2 + return 1 + fi + if [[ -z "$name" ]]; then + name="kermt-$(date -u +%Y%m%dT%H%M%SZ)-$$" + fi + local cid + local git_args=() + while IFS= read -r line; do git_args+=("$line"); done < <(_kermt_git_env_flags) + local hf_args=() + while IFS= read -r line; do hf_args+=("$line"); done < <(_kermt_hf_env_flags) + cid=$(docker run -d --gpus "$KERMT_GPUS" \ + --user "$(id -u):$(id -g)" \ + --name "$name" \ + -v "$KERMT_REPO:/workspace" \ + -v "$_kermt_bundle_dir:/skill:ro" \ + "${mount_args[@]}" \ + -w /workspace \ + -e KERMT_REPO=/workspace \ + -e PYTHONPATH=/workspace \ + -e HOME=/tmp/kermt-home \ + "${git_args[@]}" \ + "${hf_args[@]}" \ + "$KERMT_IMAGE" \ + conda run -n kermt --no-capture-output bash -c "$*") || return $? + echo "[kermt] container started: name=$name id=$cid" + echo "[kermt] follow logs: docker logs -f $name" + echo "[kermt] wait for exit: docker wait $name" + echo "[kermt] stop: docker stop $name" + echo "$cid" +} + +# ----------------------------------------------------------------------------- +# Subcommand dispatch when invoked directly (not sourced) +# ----------------------------------------------------------------------------- + +if [[ "${BASH_SOURCE[0]:-$0}" == "${0}" ]]; then + cmd="${1:-}"; shift || true + case "$cmd" in + check_docker) kermt_check_docker "$@" ;; + check_gpu) kermt_check_gpu "$@" ;; + check_system) kermt_check_system "$@" ;; + ensure_image) kermt_ensure_image "$@" ;; + run) kermt_run "$@" ;; + run_detached) kermt_run_detached "$@" ;; + ""|-h|--help) + cat >&2 < [args...] + +Subcommands: + check_docker Verify docker is installed and the daemon is reachable. + check_gpu Verify 'docker --gpus all' works (nvidia-container-toolkit). + check_system Emit a JSON probe of host GPU + VRAM + compute_cap + + driver + disk space + container toolkit + image presence. + Exits 0 with ok=false + a 'gaps' list when anything's + below the per-workflow minimum. + ensure_image Build kermt:latest from \$KERMT_REPO/Dockerfile if missing. + run [flags] -- ... Run a command inside the container (foreground, --rm). + run_detached [flags] -- ... + Run detached; prints container name + id + log hint. + +Mount flags (for run / run_detached): + --data bind to /data (read-only) + --ckpt bind to /ckpt (read-only) + --vocab-dir bind to /vocab (read-only) + --run-dir bind to /runs (read-write; created on host if missing) + --model-dir bind to /model (read-write; released-model download target) + +Additional flags for run_detached: + --name container name (default: kermt--) + +Environment overrides: + KERMT_IMAGE default kermt:latest + KERMT_REPO checkout path; otherwise discovered above the skill or working directory + KERMT_GPUS default all +EOF + exit 1 + ;; + *) + echo "[kermt] unknown subcommand: $cmd" >&2 + echo "[kermt] run '$0 --help' for usage" >&2 + exit 1 + ;; + esac +fi diff --git a/agent/skills/kermt-setup/skill-card.md b/skills/kermt-setup/skill-card.md similarity index 97% rename from agent/skills/kermt-setup/skill-card.md rename to skills/kermt-setup/skill-card.md index cb3c444..ab355c3 100644 --- a/agent/skills/kermt-setup/skill-card.md +++ b/skills/kermt-setup/skill-card.md @@ -39,7 +39,7 @@ Mitigation: The GPU smoke test runs inside the container and fails loudly at set ## Reference(s):
- [KERMT repository](https://github.com/NVIDIA-BioNeMo/KERMT)
- [NVIDIA Container Toolkit documentation](https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/)
-- `agent/scripts/kermt_container.sh` — container entry points used by this skill
+- `scripts/kermt_container.sh` — container entry points used by this skill
## Skill Output:
**Output Type(s):** [Analysis, Configuration instructions]
diff --git a/agent/tests/_build_fake_ckpt.py b/tests/skills/_build_fake_ckpt.py similarity index 100% rename from agent/tests/_build_fake_ckpt.py rename to tests/skills/_build_fake_ckpt.py diff --git a/agent/tests/conftest.py b/tests/skills/conftest.py similarity index 90% rename from agent/tests/conftest.py rename to tests/skills/conftest.py index 3d0dd61..291ca46 100644 --- a/agent/tests/conftest.py +++ b/tests/skills/conftest.py @@ -7,11 +7,11 @@ end-to-end pretrain that actually launches pretrain_ddp.py). Slow tests are SKIPPED by default; pass `--run-slow` to opt in: - KERMT_IMAGE=kermt:rebuild-test agent/scripts/kermt_container.sh run -- \\ + KERMT_IMAGE=kermt:rebuild-test skills/_shared/scripts/kermt_container.sh run -- \\ "python -m pytest agent/tests/ -v --run-slow" Also exposes a `run_agent_script` fixture used by every test_*.py to shell -out to an agent/scripts/*.py and parse the JSON it emits to stdout. Before +out to an skills/_shared/scripts/*.py and parse the JSON it emits to stdout. Before this fixture existed, each test_*.py defined its own near-identical `_run()` helper (D3 in the Phase-5.5 cleanup review). """ @@ -27,7 +27,7 @@ REPO_ROOT = Path(__file__).resolve().parents[2] -SCRIPTS_DIR = REPO_ROOT / "agent" / "scripts" +SCRIPTS_DIR = REPO_ROOT / "skills" / "_shared" / "scripts" def pytest_addoption(parser: pytest.Parser) -> None: @@ -59,7 +59,7 @@ def pytest_collection_modifyitems(config: pytest.Config, items: list[pytest.Item @pytest.fixture def run_agent_script(): - """Shell out to an agent/scripts/.py and parse its JSON stdout. + """Shell out to an skills/_shared/scripts/.py and parse its JSON stdout. Usage: code, payload = run_agent_script("check_checkpoint.py", diff --git a/agent/tests/test_check_checkpoint.py b/tests/skills/test_check_checkpoint.py similarity index 99% rename from agent/tests/test_check_checkpoint.py rename to tests/skills/test_check_checkpoint.py index b8c8424..7b4d98c 100644 --- a/agent/tests/test_check_checkpoint.py +++ b/tests/skills/test_check_checkpoint.py @@ -1,7 +1,7 @@ # SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -"""Unit tests for agent/scripts/check_checkpoint.py. +"""Unit tests for skills/_shared/scripts/check_checkpoint.py. Builds synthetic minimal checkpoint dicts that mirror the save_model_for_restart format (a dict containing 'args' Namespace + 'state_dict' with the conventional @@ -80,7 +80,7 @@ def _finetune_ffn_keys(num_tasks: int = 4) -> dict[str, torch.Tensor]: def _make_args(**overrides) -> Namespace: - """Reasonable defaults matching agent/config/defaults_pretrain.json.""" + """Reasonable defaults matching skills/_shared/config/defaults_pretrain.json.""" defaults = { "hidden_size": 800, "depth": 6, diff --git a/agent/tests/test_check_data.py b/tests/skills/test_check_data.py similarity index 98% rename from agent/tests/test_check_data.py rename to tests/skills/test_check_data.py index 5f3bdf9..1131dda 100644 --- a/agent/tests/test_check_data.py +++ b/tests/skills/test_check_data.py @@ -1,7 +1,7 @@ # SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -"""Unit tests for agent/scripts/check_data.py. +"""Unit tests for skills/_shared/scripts/check_data.py. Mix of synthetic CSV fixtures (full control of edge cases) and a few real fixtures under tests/data/ (sanity that the validator behaves correctly on @@ -21,7 +21,7 @@ REPO_ROOT = Path(__file__).resolve().parents[2] -SCRIPT = REPO_ROOT / "agent" / "scripts" / "check_data.py" +SCRIPT = REPO_ROOT / "skills" / "_shared" / "scripts" / "check_data.py" # Pre-canned real fixtures the test suite uses. diff --git a/agent/tests/test_e2e_released_download.py b/tests/skills/test_e2e_released_download.py similarity index 98% rename from agent/tests/test_e2e_released_download.py rename to tests/skills/test_e2e_released_download.py index ac632ad..6d84479 100644 --- a/agent/tests/test_e2e_released_download.py +++ b/tests/skills/test_e2e_released_download.py @@ -9,7 +9,7 @@ SKIPPED by default; opt in with `--run-slow`, and run inside the kermt container (which, after a `kermt-setup` rebuild, ships huggingface_hub): - agent/scripts/kermt_container.sh run --run-dir /tmp/e2e -- \\ + skills/_shared/scripts/kermt_container.sh run --run-dir /tmp/e2e -- \\ "python -m pytest agent/tests/test_e2e_released_download.py -v --run-slow" The bundle is downloaded ONCE per session (module-scoped fixture); the @@ -31,7 +31,7 @@ import pytest REPO_ROOT = Path(__file__).resolve().parents[2] -SCRIPTS_DIR = REPO_ROOT / "agent" / "scripts" +SCRIPTS_DIR = REPO_ROOT / "skills" / "_shared" / "scripts" CKPT_NAME = "kermt_contrastive_v2.0.pt" FINETUNE_CSV = REPO_ROOT / "tests" / "data" / "finetune" / "train.csv" diff --git a/agent/tests/test_fetch_released_model.py b/tests/skills/test_fetch_released_model.py similarity index 97% rename from agent/tests/test_fetch_released_model.py rename to tests/skills/test_fetch_released_model.py index 665c56f..018ac0c 100644 --- a/agent/tests/test_fetch_released_model.py +++ b/tests/skills/test_fetch_released_model.py @@ -1,7 +1,7 @@ # SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -"""Unit tests for agent/scripts/fetch_released_model.py. +"""Unit tests for skills/_shared/scripts/fetch_released_model.py. These mock `huggingface_hub.snapshot_download` (or simulate its absence), so they need no network and run every time — they are NOT marked `slow`. The real @@ -18,7 +18,7 @@ import sys from pathlib import Path -SCRIPTS_DIR = Path(__file__).resolve().parents[1] / "scripts" +SCRIPTS_DIR = Path(__file__).resolve().parents[2] / "skills" / "_shared" / "scripts" sys.path.insert(0, str(SCRIPTS_DIR)) frm = importlib.import_module("fetch_released_model") diff --git a/agent/tests/test_prepare_data.py b/tests/skills/test_prepare_data.py similarity index 98% rename from agent/tests/test_prepare_data.py rename to tests/skills/test_prepare_data.py index f3aab1a..8a6f894 100644 --- a/agent/tests/test_prepare_data.py +++ b/tests/skills/test_prepare_data.py @@ -1,14 +1,14 @@ # SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -"""Unit tests for agent/scripts/prepare_data.py. +"""Unit tests for skills/_shared/scripts/prepare_data.py. Runs the prepare_data CLI as a subprocess (the same way the agent skills will invoke it), inspects the produced manifest + output files, and asserts the per-mode contracts. Designed to run in-container via: - KERMT_IMAGE=kermt:rebuild-test agent/scripts/kermt_container.sh run -- \ + KERMT_IMAGE=kermt:rebuild-test skills/_shared/scripts/kermt_container.sh run -- \ "python -m pytest agent/tests/test_prepare_data.py -v --no-header -p no:cacheprovider" The tests use small synthetic CSVs (50-100 SMILES) so feature generation @@ -27,7 +27,7 @@ REPO_ROOT = Path(__file__).resolve().parents[2] -SCRIPT = REPO_ROOT / "agent" / "scripts" / "prepare_data.py" +SCRIPT = REPO_ROOT / "skills" / "_shared" / "scripts" / "prepare_data.py" # A small pool of valid drug-like SMILES used to build synthetic CSVs. diff --git a/agent/tests/test_run_extract_embeddings.py b/tests/skills/test_run_extract_embeddings.py similarity index 97% rename from agent/tests/test_run_extract_embeddings.py rename to tests/skills/test_run_extract_embeddings.py index 22be1c0..48cb7c4 100644 --- a/agent/tests/test_run_extract_embeddings.py +++ b/tests/skills/test_run_extract_embeddings.py @@ -1,13 +1,13 @@ # SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -"""Unit tests for agent/scripts/run_extract_embeddings.py. +"""Unit tests for skills/_shared/scripts/run_extract_embeddings.py. Exercises the argv builder + manifest writer via --dry-run, using cached check_checkpoint.py JSON + a synthesized prepare_data.json (mode=embed). Run in-container: - KERMT_IMAGE=kermt:rebuild-test agent/scripts/kermt_container.sh run -- \\ + KERMT_IMAGE=kermt:rebuild-test skills/_shared/scripts/kermt_container.sh run -- \\ "python -m pytest agent/tests/test_run_extract_embeddings.py -v \\ --no-header -p no:cacheprovider" """ @@ -22,7 +22,7 @@ REPO_ROOT = Path(__file__).resolve().parents[2] -SCRIPT = REPO_ROOT / "agent" / "scripts" / "run_extract_embeddings.py" +SCRIPT = REPO_ROOT / "skills" / "_shared" / "scripts" / "run_extract_embeddings.py" # --------------------------------------------------------------------------- @@ -292,7 +292,7 @@ def test_rejects_missing_clean_csv(tmp_path: Path) -> None: # and exercises the legacy `grover.*` encoder prefix path. REAL_GROVER_BASE_CKPT = REPO_ROOT / "model" / "grover_base" / "grover_base.pt" REAL_TEST_CSV = REPO_ROOT / "tests" / "data" / "finetune" / "test.csv" -PREPARE_DATA_SCRIPT = REPO_ROOT / "agent" / "scripts" / "prepare_data.py" +PREPARE_DATA_SCRIPT = REPO_ROOT / "skills" / "_shared" / "scripts" / "prepare_data.py" @pytest.mark.slow diff --git a/agent/tests/test_run_finetune_local.py b/tests/skills/test_run_finetune_local.py similarity index 98% rename from agent/tests/test_run_finetune_local.py rename to tests/skills/test_run_finetune_local.py index ce5f403..5ce60fb 100644 --- a/agent/tests/test_run_finetune_local.py +++ b/tests/skills/test_run_finetune_local.py @@ -1,7 +1,7 @@ # SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -"""Unit tests for agent/scripts/run_finetune_local.py. +"""Unit tests for skills/_shared/scripts/run_finetune_local.py. Exercises the argv builder + manifest writer via --dry-run, using cached check_checkpoint.py JSON + a synthesized prepare_data.json so we don't have to @@ -9,7 +9,7 @@ seconds. Run in-container: - KERMT_IMAGE=kermt:rebuild-test agent/scripts/kermt_container.sh run -- \\ + KERMT_IMAGE=kermt:rebuild-test skills/_shared/scripts/kermt_container.sh run -- \\ "python -m pytest agent/tests/test_run_finetune_local.py -v \\ --no-header -p no:cacheprovider" """ @@ -24,7 +24,7 @@ REPO_ROOT = Path(__file__).resolve().parents[2] -SCRIPT = REPO_ROOT / "agent" / "scripts" / "run_finetune_local.py" +SCRIPT = REPO_ROOT / "skills" / "_shared" / "scripts" / "run_finetune_local.py" # --------------------------------------------------------------------------- diff --git a/agent/tests/test_run_inference.py b/tests/skills/test_run_inference.py similarity index 97% rename from agent/tests/test_run_inference.py rename to tests/skills/test_run_inference.py index f113c8d..bb4e3a8 100644 --- a/agent/tests/test_run_inference.py +++ b/tests/skills/test_run_inference.py @@ -1,14 +1,14 @@ # SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -"""Unit tests for agent/scripts/run_inference.py. +"""Unit tests for skills/_shared/scripts/run_inference.py. Exercises the argv builder + manifest writer via --dry-run, using cached check_checkpoint.py JSON + a synthesized prepare_data.json. No real ckpt is loaded; the runner reads the cached validator output directly. Run in-container: - KERMT_IMAGE=kermt:rebuild-test agent/scripts/kermt_container.sh run -- \\ + KERMT_IMAGE=kermt:rebuild-test skills/_shared/scripts/kermt_container.sh run -- \\ "python -m pytest agent/tests/test_run_inference.py -v \\ --no-header -p no:cacheprovider" """ @@ -23,7 +23,7 @@ REPO_ROOT = Path(__file__).resolve().parents[2] -SCRIPT = REPO_ROOT / "agent" / "scripts" / "run_inference.py" +SCRIPT = REPO_ROOT / "skills" / "_shared" / "scripts" / "run_inference.py" # --------------------------------------------------------------------------- @@ -309,7 +309,7 @@ def test_rejects_failed_prepare(tmp_path: Path) -> None: / "fold_0" / "model_0" / "model.pt" ) REAL_TEST_CSV = REPO_ROOT / "tests" / "data" / "finetune" / "test.csv" -PREPARE_DATA_SCRIPT = REPO_ROOT / "agent" / "scripts" / "prepare_data.py" +PREPARE_DATA_SCRIPT = REPO_ROOT / "skills" / "_shared" / "scripts" / "prepare_data.py" @pytest.mark.slow diff --git a/agent/tests/test_run_pretrain_local.py b/tests/skills/test_run_pretrain_local.py similarity index 99% rename from agent/tests/test_run_pretrain_local.py rename to tests/skills/test_run_pretrain_local.py index b56f08b..38bd1f5 100644 --- a/agent/tests/test_run_pretrain_local.py +++ b/tests/skills/test_run_pretrain_local.py @@ -1,14 +1,14 @@ # SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -"""Unit tests for agent/scripts/run_pretrain_local.py. +"""Unit tests for skills/_shared/scripts/run_pretrain_local.py. Exercises the argv builder + manifest writer via --dry-run. Synthetic ckpts + prepare_data manifests + cached validator JSON keep test runtime under a few seconds without burning GPU time on actual pretrain epochs. Run in-container: - KERMT_IMAGE=kermt:rebuild-test agent/scripts/kermt_container.sh run -- \ + KERMT_IMAGE=kermt:rebuild-test skills/_shared/scripts/kermt_container.sh run -- \ "python -m pytest agent/tests/test_run_pretrain_local.py -v \\ --no-header -p no:cacheprovider" """ @@ -26,7 +26,7 @@ REPO_ROOT = Path(__file__).resolve().parents[2] -SCRIPT = REPO_ROOT / "agent" / "scripts" / "run_pretrain_local.py" +SCRIPT = REPO_ROOT / "skills" / "_shared" / "scripts" / "run_pretrain_local.py" # --------------------------------------------------------------------------- @@ -877,10 +877,10 @@ def test_materialize_ckpt_zeroes_resume_counters(tmp_path: Path) -> None: through unchanged. Tests the helper directly — the runner's --dry-run path skips materialization, so we exercise the helper without launching pretrain_ddp.py.""" - # Use the agent/scripts dir + import the helper directly. + # Use the canonical shared scripts directory and import the helper directly. import importlib.util, sys spec = importlib.util.spec_from_file_location( - "run_pretrain_local", REPO_ROOT / "agent" / "scripts" / "run_pretrain_local.py" + "run_pretrain_local", REPO_ROOT / "skills" / "_shared" / "scripts" / "run_pretrain_local.py" ) mod = importlib.util.module_from_spec(spec) sys.modules["run_pretrain_local"] = mod @@ -924,7 +924,7 @@ def test_materialize_ckpt_zeroes_resume_counters(tmp_path: Path) -> None: # End-to-end slow integration test — actually launches pretrain_ddp.py. # --------------------------------------------------------------------------- # Skipped by default. To run: -# KERMT_IMAGE=kermt:rebuild-test agent/scripts/kermt_container.sh run -- \ +# KERMT_IMAGE=kermt:rebuild-test skills/_shared/scripts/kermt_container.sh run -- \ # "python -m pytest agent/tests/test_run_pretrain_local.py::test_end_to_end_continue_pretrain_default_mode -v --run-slow" # # Builds a fake grover_base ckpt whose vocab heads match tests/data/pretrain's diff --git a/agent/tests/test_skill_frontmatter.py b/tests/skills/test_skill_frontmatter.py similarity index 96% rename from agent/tests/test_skill_frontmatter.py rename to tests/skills/test_skill_frontmatter.py index 2a61353..38bd823 100644 --- a/agent/tests/test_skill_frontmatter.py +++ b/tests/skills/test_skill_frontmatter.py @@ -5,7 +5,7 @@ additional metadata requirements. Skill layout (agentskills.io spec): - agent/skills//SKILL.md (one directory per skill) + skills//SKILL.md (one directory per skill) Frontmatter requirements: Spec-required (agentskills.io): @@ -33,7 +33,7 @@ REPO_ROOT = Path(__file__).resolve().parents[2] -SKILLS_DIR = REPO_ROOT / "agent" / "skills" +SKILLS_DIR = REPO_ROOT / "skills" SKILL_FILES = sorted(SKILLS_DIR.glob("*/SKILL.md")) @@ -164,8 +164,8 @@ def test_skill_files_under_token_budget() -> None: def test_skills_directory_layout_matches_spec() -> None: """agentskills.io spec: each skill lives in its own directory containing - SKILL.md. No stray *.md files at agent/skills/ root.""" - stray_md = [p for p in SKILLS_DIR.glob("*.md")] + SKILL.md. Only README.md is allowed at the skills/ root.""" + stray_md = [p for p in SKILLS_DIR.glob("*.md") if p.name != "README.md"] assert not stray_md, ( f"Found stray .md files at {SKILLS_DIR} (spec requires /SKILL.md): {stray_md}" ) diff --git a/tests/skills/test_skill_packaging.py b/tests/skills/test_skill_packaging.py new file mode 100644 index 0000000..218fc32 --- /dev/null +++ b/tests/skills/test_skill_packaging.py @@ -0,0 +1,177 @@ +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +"""Packaging checks that need only Python and Bash, with no GPU or Docker. + +Run with python3 -m unittest discover -s agent/tests -p test_skill_packaging.py. +The container tests record Docker arguments using a fake executable; they do +not start containers or exercise model computation. +""" +from __future__ import annotations + +import json +import os +from pathlib import Path +import re +import shutil +import subprocess +import sys +import tempfile +import unittest + + +REPO_ROOT = Path(__file__).resolve().parents[2] +SKILLS = REPO_ROOT / "skills" +RUNNERS = { + "kermt-continue-pretrain": "run_pretrain_local.py", + "kermt-pretrain-scratch": "run_pretrain_local.py", + "kermt-add-cmim-pretrain": "run_pretrain_local.py", + "kermt-finetune": "run_finetune_local.py", + "kermt-infer": "run_inference.py", + "kermt-embed": "run_extract_embeddings.py", +} + + +class SkillPackagingTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory(prefix="kermt-packaging-") + self.addCleanup(self.temp.cleanup) + self.root = Path(self.temp.name) + + def runtime(self): + runtime = self.root / "runtime checkout" + runtime.mkdir(exist_ok=True) + (runtime / "kermt").mkdir(exist_ok=True) + (runtime / "main.py").touch() + return runtime + + def installed_skill(self, name): + return Path(shutil.copytree(SKILLS / name, self.root / "installed skills" / name)) + + def run_python(self, *args, env=None): + return subprocess.run( + [sys.executable, "-B", *map(str, args)], cwd=self.root, + env=env, text=True, capture_output=True, + ) + + def test_bundles_contain_referenced_assets_and_no_symlinks(self): + manifests = sorted(SKILLS.glob("kermt-*/SKILL.md")) + self.assertEqual(len(manifests), 8) + self.assertEqual((REPO_ROOT / ".claude/skills").resolve(), SKILLS) + asset_pattern = re.compile( + r"(? No code, _ = _run_upgrade(src, manifest, upgraded) assert code == 0 - runner_script = REPO_ROOT / "agent" / "scripts" / "run_pretrain_local.py" + runner_script = REPO_ROOT / "skills" / "_shared" / "scripts" / "run_pretrain_local.py" r = subprocess.run( [sys.executable, str(runner_script), "--ckpt", str(upgraded), @@ -291,7 +291,7 @@ def test_upgraded_ckpt_loads_in_run_pretrain_local_dry_run(tmp_path: Path) -> No # Slow opt-in end-to-end: upgrade → real pretrain_ddp.py launch # --------------------------------------------------------------------------- # Skipped by default. To run: -# KERMT_IMAGE=kermt:rebuild-test agent/scripts/kermt_container.sh run -- \ +# KERMT_IMAGE=kermt:rebuild-test skills/_shared/scripts/kermt_container.sh run -- \ # "python -m pytest agent/tests/test_upgrade_to_hybrid.py::test_end_to_end_one_epoch_after_upgrade_hybrid -v --run-slow" # # Mirrors the pretrain runner's test_end_to_end_continue_pretrain_default_mode @@ -374,7 +374,7 @@ def test_end_to_end_one_epoch_after_upgrade_hybrid(tmp_path: Path) -> None: upgraded_md5_before = hashlib.md5(upgraded.read_bytes()).hexdigest() # 5. Launch the runner (1 epoch, save every 100 steps to exercise the save path). - runner_script = REPO_ROOT / "agent" / "scripts" / "run_pretrain_local.py" + runner_script = REPO_ROOT / "skills" / "_shared" / "scripts" / "run_pretrain_local.py" r = subprocess.run( [sys.executable, str(runner_script), "--ckpt", str(upgraded), From 91ba11f21b1c46bc208b311144251d2d4728457f Mon Sep 17 00:00:00 2001 From: Ohad Mosafi Date: Wed, 9 Sep 2026 18:12:43 -0700 Subject: [PATCH 2/4] fix links Signed-off-by: Ohad Mosafi --- .github/workflows/skill-packaging.yml | 6 ++-- AGENTS.md | 2 +- skills/README.md | 12 ++++---- tests/skills/_build_fake_ckpt.py | 4 +-- tests/skills/conftest.py | 4 +-- tests/skills/test_check_checkpoint.py | 5 ++-- tests/skills/test_check_data.py | 2 +- tests/skills/test_e2e_released_download.py | 2 +- tests/skills/test_prepare_data.py | 2 +- tests/skills/test_run_extract_embeddings.py | 2 +- tests/skills/test_run_finetune_local.py | 2 +- tests/skills/test_run_inference.py | 2 +- tests/skills/test_run_pretrain_local.py | 6 ++-- tests/skills/test_skill_packaging.py | 31 ++++++++++++++++++++- tests/skills/test_upgrade_to_hybrid.py | 6 ++-- third_party.txt | 2 +- 16 files changed, 60 insertions(+), 30 deletions(-) diff --git a/.github/workflows/skill-packaging.yml b/.github/workflows/skill-packaging.yml index f3974b0..55cc0c8 100644 --- a/.github/workflows/skill-packaging.yml +++ b/.github/workflows/skill-packaging.yml @@ -4,13 +4,13 @@ on: pull_request: paths: - 'skills/**' - - 'agent/tests/test_skill_packaging.py' + - 'tests/skills/**' - '.claude/skills' - '.github/workflows/skill-packaging.yml' push: paths: - 'skills/**' - - 'agent/tests/test_skill_packaging.py' + - 'tests/skills/**' - '.claude/skills' - '.github/workflows/skill-packaging.yml' @@ -25,4 +25,4 @@ jobs: - name: Verify bundled shared files run: python3 skills/_shared/sync_shared.py --check - name: Check independently installed skills - run: python3 -m unittest discover -s agent/tests -p test_skill_packaging.py + run: python3 -m unittest discover -s tests/skills -p test_skill_packaging.py diff --git a/AGENTS.md b/AGENTS.md index f527532..2c97fd3 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -23,4 +23,4 @@ Shared helpers and defaults are maintained in `skills/_shared/`. After editing them, run `python3 skills/_shared/sync_shared.py --write` and commit the per-skill copies; `--check` verifies they match. Each skill must use its own bundled files so it can be installed and signed independently. -Development tests remain under `agent/tests/`. +Development tests remain under `tests/skills/`. diff --git a/skills/README.md b/skills/README.md index 585438a..a742184 100644 --- a/skills/README.md +++ b/skills/README.md @@ -18,7 +18,7 @@ From the repository root, after changing a canonical file: ```bash python3 skills/_shared/sync_shared.py --write python3 skills/_shared/sync_shared.py --check -python3 -m unittest discover -s agent/tests -p test_skill_packaging.py +python3 -m unittest discover -s tests/skills -p test_skill_packaging.py ``` Commit both the canonical edits and their generated per-skill copies. This @@ -304,7 +304,7 @@ Deterministic logic lives under [`_shared/scripts/`](_shared/scripts/): | `run_inference.py` | Run predictions with a finetuned checkpoint | | `run_extract_embeddings.py` | Extract molecular embeddings | -Tests for these scripts are under [`agent/tests/`](../agent/tests/) and use the existing +Tests for these scripts are under [`tests/skills/`](../tests/skills/) and use the existing fixture data in [`../tests/data/pretrain/`](../tests/data/pretrain/) and [`../tests/data/finetune/`](../tests/data/finetune/). @@ -320,16 +320,16 @@ kermt repo checkout: ```bash # Run the full agent test suite in-container. The pytest paths are inside the # container, where the repo is bind-mounted at /workspace, so they stay -# repo-relative (agent/tests/) regardless of your host working directory. +# repo-relative (tests/skills/) regardless of your host working directory. $KERMT_REPO/skills/_shared/scripts/kermt_container.sh run -- \ - "python -m pytest agent/tests/ -v --no-header -p no:cacheprovider" + "python -m pytest tests/skills/ -v --no-header -p no:cacheprovider" ``` Override the image tag if you're testing against a non-default build: ```bash KERMT_IMAGE=kermt:rebuild-test $KERMT_REPO/skills/_shared/scripts/kermt_container.sh run -- \ - "python -m pytest agent/tests/test_check_checkpoint.py -v" + "python -m pytest tests/skills/test_check_checkpoint.py -v" ``` For quick local-dev iteration on a single test, you can also run the suite on @@ -338,7 +338,7 @@ the host inside any conda env that has `torch`, `rdkit`, `pandas`, and ```bash conda activate -python -m pytest agent/tests/test_check_data.py -v +python -m pytest tests/skills/test_check_data.py -v ``` But the final sign-off for any change in `skills/_shared/scripts/` is the in-container diff --git a/tests/skills/_build_fake_ckpt.py b/tests/skills/_build_fake_ckpt.py index fb847f2..c6c6af2 100644 --- a/tests/skills/_build_fake_ckpt.py +++ b/tests/skills/_build_fake_ckpt.py @@ -10,12 +10,12 @@ (`/last_checkpoint.pt`). Run inside the kermt container — needs the kermt package + torch + the vocab -loader. Not invoked in production; lives under agent/tests/ because that's the +loader. Not invoked in production; lives under tests/skills/ because that's the only context that needs to forge a checkpoint. Usage ----- - python agent/tests/_build_fake_ckpt.py \ + python tests/skills/_build_fake_ckpt.py \ --atom-vocab tests/data/pretrain/pretrain_atom_vocab.json \ --bond-vocab tests/data/pretrain/pretrain_bond_vocab.json \ --out /tmp/fake_grover_base.pt \ diff --git a/tests/skills/conftest.py b/tests/skills/conftest.py index 291ca46..ea1ca16 100644 --- a/tests/skills/conftest.py +++ b/tests/skills/conftest.py @@ -1,14 +1,14 @@ # SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -"""Pytest configuration for agent/tests/. +"""Pytest configuration for tests/skills/. Registers the `slow` marker for tests that take more than a few seconds (e.g. end-to-end pretrain that actually launches pretrain_ddp.py). Slow tests are SKIPPED by default; pass `--run-slow` to opt in: KERMT_IMAGE=kermt:rebuild-test skills/_shared/scripts/kermt_container.sh run -- \\ - "python -m pytest agent/tests/ -v --run-slow" + "python -m pytest tests/skills/ -v --run-slow" Also exposes a `run_agent_script` fixture used by every test_*.py to shell out to an skills/_shared/scripts/*.py and parse the JSON it emits to stdout. Before diff --git a/tests/skills/test_check_checkpoint.py b/tests/skills/test_check_checkpoint.py index 7b4d98c..3646303 100644 --- a/tests/skills/test_check_checkpoint.py +++ b/tests/skills/test_check_checkpoint.py @@ -9,7 +9,7 @@ or one arch-derivation branch. Run from the kermt repo root: - pytest agent/tests/test_check_checkpoint.py -v + pytest tests/skills/test_check_checkpoint.py -v """ from __future__ import annotations @@ -23,7 +23,8 @@ import torch -SCRIPT = Path(__file__).resolve().parent.parent / "scripts" / "check_checkpoint.py" +REPO_ROOT = Path(__file__).resolve().parents[2] +SCRIPT = REPO_ROOT / "skills" / "_shared" / "scripts" / "check_checkpoint.py" # --------------------------------------------------------------------------- diff --git a/tests/skills/test_check_data.py b/tests/skills/test_check_data.py index 1131dda..15ab567 100644 --- a/tests/skills/test_check_data.py +++ b/tests/skills/test_check_data.py @@ -8,7 +8,7 @@ shapes that match what users actually have). Run from the kermt repo root: - pytest agent/tests/test_check_data.py -v + pytest tests/skills/test_check_data.py -v """ from __future__ import annotations diff --git a/tests/skills/test_e2e_released_download.py b/tests/skills/test_e2e_released_download.py index 6d84479..06ac007 100644 --- a/tests/skills/test_e2e_released_download.py +++ b/tests/skills/test_e2e_released_download.py @@ -10,7 +10,7 @@ container (which, after a `kermt-setup` rebuild, ships huggingface_hub): skills/_shared/scripts/kermt_container.sh run --run-dir /tmp/e2e -- \\ - "python -m pytest agent/tests/test_e2e_released_download.py -v --run-slow" + "python -m pytest tests/skills/test_e2e_released_download.py -v --run-slow" The bundle is downloaded ONCE per session (module-scoped fixture); the idempotent fetch means re-runs don't re-download. If the download can't run diff --git a/tests/skills/test_prepare_data.py b/tests/skills/test_prepare_data.py index 8a6f894..0108468 100644 --- a/tests/skills/test_prepare_data.py +++ b/tests/skills/test_prepare_data.py @@ -9,7 +9,7 @@ Designed to run in-container via: KERMT_IMAGE=kermt:rebuild-test skills/_shared/scripts/kermt_container.sh run -- \ - "python -m pytest agent/tests/test_prepare_data.py -v --no-header -p no:cacheprovider" + "python -m pytest tests/skills/test_prepare_data.py -v --no-header -p no:cacheprovider" The tests use small synthetic CSVs (50-100 SMILES) so feature generation finishes quickly. Total in-container runtime ≈ 1-2 minutes (vocab build + diff --git a/tests/skills/test_run_extract_embeddings.py b/tests/skills/test_run_extract_embeddings.py index 48cb7c4..5e5c449 100644 --- a/tests/skills/test_run_extract_embeddings.py +++ b/tests/skills/test_run_extract_embeddings.py @@ -8,7 +8,7 @@ Run in-container: KERMT_IMAGE=kermt:rebuild-test skills/_shared/scripts/kermt_container.sh run -- \\ - "python -m pytest agent/tests/test_run_extract_embeddings.py -v \\ + "python -m pytest tests/skills/test_run_extract_embeddings.py -v \\ --no-header -p no:cacheprovider" """ from __future__ import annotations diff --git a/tests/skills/test_run_finetune_local.py b/tests/skills/test_run_finetune_local.py index 5ce60fb..bd1fda0 100644 --- a/tests/skills/test_run_finetune_local.py +++ b/tests/skills/test_run_finetune_local.py @@ -10,7 +10,7 @@ Run in-container: KERMT_IMAGE=kermt:rebuild-test skills/_shared/scripts/kermt_container.sh run -- \\ - "python -m pytest agent/tests/test_run_finetune_local.py -v \\ + "python -m pytest tests/skills/test_run_finetune_local.py -v \\ --no-header -p no:cacheprovider" """ from __future__ import annotations diff --git a/tests/skills/test_run_inference.py b/tests/skills/test_run_inference.py index bb4e3a8..b122f3b 100644 --- a/tests/skills/test_run_inference.py +++ b/tests/skills/test_run_inference.py @@ -9,7 +9,7 @@ Run in-container: KERMT_IMAGE=kermt:rebuild-test skills/_shared/scripts/kermt_container.sh run -- \\ - "python -m pytest agent/tests/test_run_inference.py -v \\ + "python -m pytest tests/skills/test_run_inference.py -v \\ --no-header -p no:cacheprovider" """ from __future__ import annotations diff --git a/tests/skills/test_run_pretrain_local.py b/tests/skills/test_run_pretrain_local.py index 38bd1f5..04026ef 100644 --- a/tests/skills/test_run_pretrain_local.py +++ b/tests/skills/test_run_pretrain_local.py @@ -9,7 +9,7 @@ Run in-container: KERMT_IMAGE=kermt:rebuild-test skills/_shared/scripts/kermt_container.sh run -- \ - "python -m pytest agent/tests/test_run_pretrain_local.py -v \\ + "python -m pytest tests/skills/test_run_pretrain_local.py -v \\ --no-header -p no:cacheprovider" """ from __future__ import annotations @@ -925,7 +925,7 @@ def test_materialize_ckpt_zeroes_resume_counters(tmp_path: Path) -> None: # --------------------------------------------------------------------------- # Skipped by default. To run: # KERMT_IMAGE=kermt:rebuild-test skills/_shared/scripts/kermt_container.sh run -- \ -# "python -m pytest agent/tests/test_run_pretrain_local.py::test_end_to_end_continue_pretrain_default_mode -v --run-slow" +# "python -m pytest tests/skills/test_run_pretrain_local.py::test_end_to_end_continue_pretrain_default_mode -v --run-slow" # # Builds a fake grover_base ckpt whose vocab heads match tests/data/pretrain's # atom + bond vocab files, then runs run_pretrain_local.py for one full epoch @@ -966,7 +966,7 @@ def _build_fake_grover_base_ckpt(tmp_path: Path, **extra_args: object) -> Path: extra_args are forwarded as kebab-case CLI flags (e.g. `scheduler_step=100` -> `--scheduler-step 100`).""" fixture_dir = REPO_ROOT / "tests" / "data" / "pretrain" - builder = REPO_ROOT / "agent" / "tests" / "_build_fake_ckpt.py" + builder = REPO_ROOT / "tests" / "skills" / "_build_fake_ckpt.py" fake_ckpt = tmp_path / "fake.pt" cli = [ sys.executable, str(builder), diff --git a/tests/skills/test_skill_packaging.py b/tests/skills/test_skill_packaging.py index 218fc32..4e1b65f 100644 --- a/tests/skills/test_skill_packaging.py +++ b/tests/skills/test_skill_packaging.py @@ -2,12 +2,13 @@ # SPDX-License-Identifier: Apache-2.0 """Packaging checks that need only Python and Bash, with no GPU or Docker. -Run with python3 -m unittest discover -s agent/tests -p test_skill_packaging.py. +Run with python3 -m unittest discover -s tests/skills -p test_skill_packaging.py. The container tests record Docker arguments using a fake executable; they do not start containers or exercise model computation. """ from __future__ import annotations +import ast import json import os from pathlib import Path @@ -53,6 +54,34 @@ def run_python(self, *args, env=None): env=env, text=True, capture_output=True, ) + def test_test_entrypoints_resolve_without_ml_dependencies(self): + # Evaluate only path declarations, so a moved test's broken SCRIPT + # path is caught in CI even when Torch/RDKit cannot be imported. + for module in sorted((REPO_ROOT / "tests/skills").glob("*.py")): + scope = {"Path": Path, "__file__": str(module), "__builtins__": {}} + for node in ast.parse(module.read_text()).body: + if not isinstance(node, ast.Assign) or len(node.targets) != 1: + continue + if not isinstance(node.targets[0], ast.Name): + continue + names = {n.id for n in ast.walk(node.value) if isinstance(n, ast.Name)} + if not names or not names <= scope.keys(): + continue + value = eval(compile(ast.Expression(node.value), str(module), "eval"), scope) + if not isinstance(value, Path): + continue + name = node.targets[0].id + scope[name] = value + # Optional downloaded checkpoints and CSV datasets may be + # absent. Repository roots, entrypoints and configs must exist. + with self.subTest(module=module.name, path=name): + if name == "REPO_ROOT": + self.assertEqual(value, REPO_ROOT) + if name.endswith(("_ROOT", "_DIR")): + self.assertTrue(value.is_dir(), f"{module}:{node.lineno}: {value}") + elif value.suffix in {".py", ".sh", ".json"}: + self.assertTrue(value.is_file(), f"{module}:{node.lineno}: {value}") + def test_bundles_contain_referenced_assets_and_no_symlinks(self): manifests = sorted(SKILLS.glob("kermt-*/SKILL.md")) self.assertEqual(len(manifests), 8) diff --git a/tests/skills/test_upgrade_to_hybrid.py b/tests/skills/test_upgrade_to_hybrid.py index e8c6c50..4963366 100644 --- a/tests/skills/test_upgrade_to_hybrid.py +++ b/tests/skills/test_upgrade_to_hybrid.py @@ -11,7 +11,7 @@ In-container by default: KERMT_IMAGE=kermt:rebuild-test skills/_shared/scripts/kermt_container.sh run -- \\ - "python -m pytest agent/tests/test_upgrade_to_hybrid.py -v \\ + "python -m pytest tests/skills/test_upgrade_to_hybrid.py -v \\ --no-header -p no:cacheprovider" """ from __future__ import annotations @@ -28,7 +28,7 @@ REPO_ROOT = Path(__file__).resolve().parents[2] UPGRADE_SCRIPT = REPO_ROOT / "skills" / "_shared" / "scripts" / "upgrade_to_hybrid.py" CHECK_CKPT = REPO_ROOT / "skills" / "_shared" / "scripts" / "check_checkpoint.py" -BUILD_FAKE = REPO_ROOT / "agent" / "tests" / "_build_fake_ckpt.py" +BUILD_FAKE = REPO_ROOT / "tests" / "skills" / "_build_fake_ckpt.py" PRETRAIN_FIXTURE = REPO_ROOT / "tests" / "data" / "pretrain" @@ -292,7 +292,7 @@ def test_upgraded_ckpt_loads_in_run_pretrain_local_dry_run(tmp_path: Path) -> No # --------------------------------------------------------------------------- # Skipped by default. To run: # KERMT_IMAGE=kermt:rebuild-test skills/_shared/scripts/kermt_container.sh run -- \ -# "python -m pytest agent/tests/test_upgrade_to_hybrid.py::test_end_to_end_one_epoch_after_upgrade_hybrid -v --run-slow" +# "python -m pytest tests/skills/test_upgrade_to_hybrid.py::test_end_to_end_one_epoch_after_upgrade_hybrid -v --run-slow" # # Mirrors the pretrain runner's test_end_to_end_continue_pretrain_default_mode # but adds the upgrade step at the front: diff --git a/third_party.txt b/third_party.txt index dc6b7bc..d45c9ae 100644 --- a/third_party.txt +++ b/third_party.txt @@ -201,7 +201,7 @@ License Text: Name: descriptastorus Version: 2.7.0 License: UNKNOWN -URL: https://github/bp-kelley/descriptastorus +URL: https://github.com/bp-kelley/descriptastorus License Text: Copyright (c) 2018-2023, Novartis Institutes for BioMedical Research Inc. All rights reserved. From 77111e060f134fc5657cf7f02417f7e5b691e740 Mon Sep 17 00:00:00 2001 From: Ohad Mosafi Date: Wed, 9 Sep 2026 22:13:27 -0700 Subject: [PATCH 3/4] tier1 Signed-off-by: Ohad Mosafi --- skills/README.md | 15 ++ skills/_shared/scripts/_utils.py | 105 +++++++++++--- skills/_shared/scripts/check_checkpoint.py | 9 +- .../_shared/scripts/run_extract_embeddings.py | 4 +- skills/_shared/scripts/run_finetune_local.py | 4 +- skills/_shared/scripts/run_inference.py | 4 +- skills/_shared/scripts/run_pretrain_local.py | 11 +- skills/_shared/scripts/upgrade_to_hybrid.py | 4 +- skills/kermt-add-cmim-pretrain/SKILL.md | 2 +- .../kermt-add-cmim-pretrain/scripts/_utils.py | 105 +++++++++++--- .../scripts/check_checkpoint.py | 9 +- .../scripts/run_pretrain_local.py | 11 +- .../scripts/upgrade_to_hybrid.py | 4 +- skills/kermt-continue-pretrain/SKILL.md | 14 +- .../kermt-continue-pretrain/scripts/_utils.py | 105 +++++++++++--- .../scripts/check_checkpoint.py | 9 +- .../scripts/run_pretrain_local.py | 11 +- skills/kermt-embed/SKILL.md | 14 +- skills/kermt-embed/scripts/_utils.py | 105 +++++++++++--- .../kermt-embed/scripts/check_checkpoint.py | 9 +- .../scripts/run_extract_embeddings.py | 4 +- skills/kermt-finetune/SKILL.md | 14 +- skills/kermt-finetune/scripts/_utils.py | 105 +++++++++++--- .../scripts/check_checkpoint.py | 9 +- .../scripts/run_finetune_local.py | 4 +- skills/kermt-infer/SKILL.md | 2 +- skills/kermt-infer/scripts/_utils.py | 105 +++++++++++--- .../kermt-infer/scripts/check_checkpoint.py | 9 +- skills/kermt-infer/scripts/run_inference.py | 4 +- skills/kermt-pretrain-scratch/SKILL.md | 2 +- .../kermt-pretrain-scratch/scripts/_utils.py | 105 +++++++++++--- .../scripts/check_checkpoint.py | 9 +- .../scripts/run_pretrain_local.py | 11 +- skills/kermt-setup/SKILL.md | 2 +- tests/skills/test_artifact_loading.py | 134 ++++++++++++++++++ 35 files changed, 862 insertions(+), 211 deletions(-) create mode 100644 tests/skills/test_artifact_loading.py diff --git a/skills/README.md b/skills/README.md index a742184..d3631d6 100644 --- a/skills/README.md +++ b/skills/README.md @@ -129,6 +129,21 @@ See the [released-model guide](_shared/references/released-models.md) for the bundle layout, vocabulary compatibility, and handling incomplete downloads. That canonical guide is also bundled into the skills that fetch released models. +### Artifact validation and runtime settings + +Skill helpers read checkpoints with PyTorch's restricted weights loader, allowing +KERMT's saved argument namespace and numeric NumPy scaler metadata. Legacy +MolVocab and SMILESVocab pickles are inspected as stored vocabulary data for token +counts. Artifacts requiring other Python classes or executable reducers are +rejected instead of being loaded through an unrestricted fallback. + +Training, inference, and embedding subprocesses receive named runtime settings +for Python paths, temporary files, CUDA, NCCL, and CPU threading. The list lives +in `runner_environment` in `_shared/scripts/_utils.py`. Unrelated credentials +are excluded. W&B credentials and settings are forwarded only when pretraining +explicitly enables logging with `--wandb-project`; Hugging Face authentication +is handled by the separate model-download helper. + ## Token-efficient design The skill files (`.md`) are intentionally thin — they orchestrate, prompt for diff --git a/skills/_shared/scripts/_utils.py b/skills/_shared/scripts/_utils.py index 5bde460..7fe75b0 100644 --- a/skills/_shared/scripts/_utils.py +++ b/skills/_shared/scripts/_utils.py @@ -10,8 +10,11 @@ from __future__ import annotations import argparse +from collections import Counter import json import os +import pickle +import re import shlex import subprocess import sys @@ -72,11 +75,11 @@ def count_vocab_entries(vocab_path: Path) -> int: Handles three layouts: - JSON with `{stoi: {token: idx}, ...}` (MolVocab.save_vocab default) - JSON as a raw `{token: idx}` dict (legacy / hand-edited) - - Pickle of a MolVocab / SMILESVocab object (the smiles vocab is always - pickled because its compiled-regex tokenizer state isn't - JSON-serializable). Falls through to raw `pickle.load` if the - MolVocab / SMILESVocab loader can't import or fails to recognize - the contents (e.g. test fixtures with plain dicts). + - Legacy MolVocab / SMILESVocab pickles, read as inert vocabulary state. + + The pickle reader accepts only the known vocabulary containers and their + Counter/regex metadata. It cannot import arbitrary classes or run reducers + supplied by the artifact, and it never falls back to an unrestricted loader. """ if vocab_path.suffix == ".json": data = json.loads(vocab_path.read_text()) @@ -86,28 +89,88 @@ def count_vocab_entries(vocab_path: Path) -> int: return len(data) raise ValueError(f"unsupported JSON vocab shape at {vocab_path}: {type(data).__name__}") - # .pkl: try MolVocab / SMILESVocab first, then fall back to raw pickle. - try: - from kermt.data.torchvocab import MolVocab, SMILESVocab # type: ignore - for loader in (MolVocab.load_vocab, SMILESVocab.load_vocab): - try: - v = loader(str(vocab_path)) - return len(v) - except Exception: - continue - except ImportError: - pass - - import pickle with vocab_path.open("rb") as f: - data = pickle.load(f) - if hasattr(data, "stoi"): + data = _VocabUnpickler(f).load() + if isinstance(data, _VocabState) and isinstance(data.stoi, dict): return len(data.stoi) - if hasattr(data, "__len__"): + if isinstance(data, (dict, list, tuple)): return len(data) raise ValueError(f"could not count entries in {vocab_path}") +class _VocabState: + """Data-only stand-in: counting tokens does not require tokenizer methods.""" + + +class _VocabUnpickler(pickle.Unpickler): + def find_class(self, module: str, name: str) -> Any: + if module in {"kermt.data.torchvocab", "grover.data.torchvocab"} and name in { + "TorchVocab", "MolVocab", "SMILESVocab", + }: + return _VocabState + if (module, name) == ("collections", "Counter"): + return Counter + if (module, name) == ("re", "_compile"): + return re.compile + raise pickle.UnpicklingError(f"unsupported vocabulary object: {module}.{name}") + + +def load_checkpoint(path: Path | str) -> dict[str, Any]: + """Read KERMT tensors and known metadata with PyTorch's restricted loader. + + Saved arguments use argparse.Namespace; finetuned checkpoints also contain + numeric NumPy scaler arrays. Explicit globals cover those formats, including + NumPy 1/2 module names, without accepting artifact-selected imports. + """ + import numpy as np + import torch + + multiarray = np._core.multiarray if hasattr(np, "_core") else np.core.multiarray + allowed = [argparse.Namespace, np.ndarray, np.dtype] + for module in ("numpy.core.multiarray", "numpy._core.multiarray"): + allowed.extend([ + (multiarray._reconstruct, f"{module}._reconstruct"), + (multiarray.scalar, f"{module}.scalar"), + ]) + allowed.extend(type(np.dtype(name)) for name in ( + "bool", "int8", "int16", "int32", "int64", "uint8", "uint16", "uint32", "uint64", + "float16", "float32", "float64", + )) + with torch.serialization.safe_globals(allowed): + return torch.load(path, map_location="cpu", weights_only=True) + + +def runner_environment(repo: Path, *, wandb: bool = False) -> dict[str, str]: + """Forward named runtime settings, keeping unrelated credentials out of jobs. + + W&B credentials/settings are included only for an explicitly enabled W&B + run. Hugging Face authentication belongs to the separate download helper. + """ + names = ( + "PATH", "HOME", "TMPDIR", "TEMP", "TMP", "LANG", "LC_ALL", "LC_CTYPE", "TZ", + "LD_LIBRARY_PATH", "LIBRARY_PATH", "CUDA_HOME", "CUDA_PATH", "PYTHONPATH", + "PYTHONDONTWRITEBYTECODE", "PYTHONUNBUFFERED", "PYTHONWARNINGS", + "CUDA_VISIBLE_DEVICES", "CUDA_DEVICE_ORDER", "CUDA_LAUNCH_BLOCKING", "NVIDIA_VISIBLE_DEVICES", + "NVIDIA_DRIVER_CAPABILITIES", "CUBLAS_WORKSPACE_CONFIG", "OMP_NUM_THREADS", + "MKL_NUM_THREADS", "OPENBLAS_NUM_THREADS", "NUMEXPR_NUM_THREADS", + "PYTORCH_CUDA_ALLOC_CONF", "PYTORCH_ALLOC_CONF", "PYTORCH_NO_CUDA_MEMORY_CACHING", + "TORCH_CPP_LOG_LEVEL", "TORCH_DISTRIBUTED_DEBUG", "NCCL_DEBUG", "NCCL_SOCKET_IFNAME", + "NCCL_IB_DISABLE", "NCCL_P2P_DISABLE", "NCCL_SHM_DISABLE", "GLOO_SOCKET_IFNAME", + "MASTER_ADDR", "MASTER_PORT", + "KERMT_REPO", "KERMT_REPO_COMMIT", "KERMT_REPO_DIRTY", + "SSL_CERT_FILE", "REQUESTS_CA_BUNDLE", + ) + if wandb: + names += ( + "WANDB_API_KEY", "WANDB_BASE_URL", "WANDB_MODE", "WANDB_DIR", "WANDB_ENTITY", + "WANDB_PROJECT", "WANDB_RUN_ID", "WANDB_RESUME", "WANDB_CACHE_DIR", + "WANDB_CONFIG_DIR", "WANDB_DATA_DIR", "WANDB_DISABLED", + ) + env = {name: value for name in names if (value := os.environ.get(name)) is not None} + env["PYTHONPATH"] = os.pathsep.join(filter(None, (str(repo), env.get("PYTHONPATH")))) + return env + + def validate_vocab_file(vocab_path: Path, *, kind: str) -> None: """Verify a user-provided vocab file is loadable BEFORE copying it into a run directory. Raises ValueError on failure with a clear, user-facing message. diff --git a/skills/_shared/scripts/check_checkpoint.py b/skills/_shared/scripts/check_checkpoint.py index fd488e2..5df7627 100644 --- a/skills/_shared/scripts/check_checkpoint.py +++ b/skills/_shared/scripts/check_checkpoint.py @@ -75,10 +75,15 @@ import sys import traceback from argparse import Namespace +from pathlib import Path from typing import Any import torch +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import load_checkpoint # noqa: E402 + # --------------------------------------------------------------------------- # State-dict key prefix conventions (kermt/model/models.py). @@ -291,7 +296,7 @@ def _arch_from_shapes(state_dict: dict[str, Any], arch: dict[str, Any]) -> tuple ] if candidates: arch["latent_dim"] = int(state_dict[candidates[0]].shape[0]) - # Absent latent_dist entirely is a fact about the model, not a problem — no warning. + # Encoder-only models have no latent distribution; latent_dim remains None. # depth, num_attn_head, activation, backbone, embedding_output_type, self_attention # are not robustly inferable from shapes alone; report a warning for each that's @@ -390,7 +395,7 @@ def validate(mode: str, ckpt_path: str) -> dict[str, Any]: # 1. Load the checkpoint. try: - ckpt = torch.load(ckpt_path, map_location="cpu", weights_only=False) + ckpt = load_checkpoint(ckpt_path) except FileNotFoundError: result["errors"].append(f"checkpoint not found: {ckpt_path}") return result diff --git a/skills/_shared/scripts/run_extract_embeddings.py b/skills/_shared/scripts/run_extract_embeddings.py index 7ba118c..fb9def2 100644 --- a/skills/_shared/scripts/run_extract_embeddings.py +++ b/skills/_shared/scripts/run_extract_embeddings.py @@ -47,7 +47,7 @@ from _utils import ( # noqa: E402 resolve_kermt_repo, assert_prepare_manifest_basics, docker_image_digest, format_cmd_replay, git_commit_with_env_override, load_json, merge_default_into_applied, - resolve_single_gpu, run_checkpoint_validator, + resolve_single_gpu, run_checkpoint_validator, runner_environment, ) @@ -173,7 +173,7 @@ def run(args: argparse.Namespace) -> dict[str, Any]: run_manifest["status"] = "dry_run" return run_manifest - env = os.environ.copy() + env = runner_environment(REPO_ROOT) env["CUDA_VISIBLE_DEVICES"] = str(gpu) log_file = out_dir / "logs" / "embed.log" with log_file.open("w") as logf: diff --git a/skills/_shared/scripts/run_finetune_local.py b/skills/_shared/scripts/run_finetune_local.py index 167d8a7..36037c3 100644 --- a/skills/_shared/scripts/run_finetune_local.py +++ b/skills/_shared/scripts/run_finetune_local.py @@ -62,7 +62,7 @@ from _utils import ( # noqa: E402 resolve_kermt_repo, assert_prepare_manifest_basics, docker_image_digest, format_cmd_replay, git_commit_with_env_override, load_json, merge_default_into_applied, - resolve_single_gpu, run_checkpoint_validator, + resolve_single_gpu, run_checkpoint_validator, runner_environment, ) @@ -383,7 +383,7 @@ def run(args: argparse.Namespace) -> dict[str, Any]: run_manifest["status"] = "dry_run" return run_manifest - env = os.environ.copy() + env = runner_environment(REPO_ROOT) if distributed: # DDP: main.py finetune reads WORLD_SIZE and spawns one process per GPU. # Do not pin CUDA_VISIBLE_DEVICES to a single device. diff --git a/skills/_shared/scripts/run_inference.py b/skills/_shared/scripts/run_inference.py index 56575bd..35c4fcf 100644 --- a/skills/_shared/scripts/run_inference.py +++ b/skills/_shared/scripts/run_inference.py @@ -48,7 +48,7 @@ from _utils import ( # noqa: E402 resolve_kermt_repo, assert_prepare_manifest_basics, docker_image_digest, format_cmd_replay, git_commit_with_env_override, load_json, merge_default_into_applied, - resolve_single_gpu, run_checkpoint_validator, + resolve_single_gpu, run_checkpoint_validator, runner_environment, ) @@ -206,7 +206,7 @@ def run(args: argparse.Namespace) -> dict[str, Any]: return run_manifest # 6. Execute. - env = os.environ.copy() + env = runner_environment(REPO_ROOT) env["CUDA_VISIBLE_DEVICES"] = str(gpu) # main.py enables strict deterministic algorithms via # `torch.use_deterministic_algorithms(True)` (kermt/main.py:23); CuBLAS diff --git a/skills/_shared/scripts/run_pretrain_local.py b/skills/_shared/scripts/run_pretrain_local.py index 5b8a8a1..6f74382 100644 --- a/skills/_shared/scripts/run_pretrain_local.py +++ b/skills/_shared/scripts/run_pretrain_local.py @@ -53,8 +53,8 @@ sys.path.insert(0, str(Path(__file__).resolve().parent)) from _utils import ( # noqa: E402 resolve_kermt_repo, assert_prepare_manifest_basics, count_vocab_entries, docker_image_digest, - format_cmd_replay, git_commit_with_env_override, load_json, - merge_default_into_applied, run_checkpoint_validator, + format_cmd_replay, git_commit_with_env_override, load_json, load_checkpoint, + merge_default_into_applied, run_checkpoint_validator, runner_environment, ) @@ -315,7 +315,7 @@ def _materialize_ckpt_for_fresh_schedule(user_ckpt: Path, save_dir: Path) -> Pat target = save_dir / "last_checkpoint.pt" if target.exists() or target.is_symlink(): target.unlink() - ckpt = torch.load(user_ckpt, map_location="cpu", weights_only=False) + ckpt = load_checkpoint(user_ckpt) if not isinstance(ckpt, dict) or "state_dict" not in ckpt: raise ValueError( f"ckpt {user_ckpt} is not in the expected save_model_for_restart " @@ -335,8 +335,7 @@ def _validate_resume_state(user_ckpt: Path) -> dict[str, Any]: (optimizer state, scheduler_step, epoch, batch_idx). Returns a small `resume_state` dict for the manifest so users can see what was restored. Raises ValueError with a clear redirect if the ckpt is too lean.""" - import torch - ckpt = torch.load(user_ckpt, map_location="cpu", weights_only=False) + ckpt = load_checkpoint(user_ckpt) if not isinstance(ckpt, dict) or "state_dict" not in ckpt: raise ValueError( f"ckpt {user_ckpt} is not in the expected save_model_for_restart " @@ -645,7 +644,7 @@ def run(args: argparse.Namespace) -> dict[str, Any]: run_manifest["status"] = "dry_run" return run_manifest - env = os.environ.copy() + env = runner_environment(REPO_ROOT, wandb="wandb_project" in applied) env["WORLD_SIZE"] = str(world_size) if gpus_str: env["CUDA_VISIBLE_DEVICES"] = gpus_str diff --git a/skills/_shared/scripts/upgrade_to_hybrid.py b/skills/_shared/scripts/upgrade_to_hybrid.py index 65c93f6..5895621 100644 --- a/skills/_shared/scripts/upgrade_to_hybrid.py +++ b/skills/_shared/scripts/upgrade_to_hybrid.py @@ -65,7 +65,7 @@ # or as a bare `python scripts/upgrade_to_hybrid.py …`. if str(Path(__file__).resolve().parent) not in sys.path: sys.path.insert(0, str(Path(__file__).resolve().parent)) -from _utils import count_vocab_entries, load_json, run_checkpoint_validator # noqa: E402 +from _utils import count_vocab_entries, load_checkpoint, load_json, run_checkpoint_validator # noqa: E402 SKILL_ROOT = Path(__file__).resolve().parent.parent @@ -251,7 +251,7 @@ def upgrade(args: argparse.Namespace) -> dict[str, Any]: decoder_defaults = defaults.get("add_cmim_decoder") or {} latent_dim = args.latent_dim if args.latent_dim is not None else decoder_defaults.get("latent_dim", 800) - input_ckpt = torch.load(args.ckpt, map_location="cpu", weights_only=False) + input_ckpt = load_checkpoint(args.ckpt) input_args = input_ckpt.get("args") if input_args is None: raise ValueError( diff --git a/skills/kermt-add-cmim-pretrain/SKILL.md b/skills/kermt-add-cmim-pretrain/SKILL.md index 3b0fa7b..dc873f1 100644 --- a/skills/kermt-add-cmim-pretrain/SKILL.md +++ b/skills/kermt-add-cmim-pretrain/SKILL.md @@ -32,7 +32,7 @@ the workflow is identical to `kermt-continue-pretrain`. ## Skill and runtime paths -Set `SKILL_DIR` to the directory containing this `SKILL.md`. Export +Set `SKILL_DIR` to the absolute path of this installed skill directory. Export `KERMT_REPO` as the absolute path to the KERMT checkout used for model execution. The bundled container helper mounts that checkout at `/workspace` and this skill at `/skill` (read-only). Commands inside diff --git a/skills/kermt-add-cmim-pretrain/scripts/_utils.py b/skills/kermt-add-cmim-pretrain/scripts/_utils.py index 5bde460..7fe75b0 100644 --- a/skills/kermt-add-cmim-pretrain/scripts/_utils.py +++ b/skills/kermt-add-cmim-pretrain/scripts/_utils.py @@ -10,8 +10,11 @@ from __future__ import annotations import argparse +from collections import Counter import json import os +import pickle +import re import shlex import subprocess import sys @@ -72,11 +75,11 @@ def count_vocab_entries(vocab_path: Path) -> int: Handles three layouts: - JSON with `{stoi: {token: idx}, ...}` (MolVocab.save_vocab default) - JSON as a raw `{token: idx}` dict (legacy / hand-edited) - - Pickle of a MolVocab / SMILESVocab object (the smiles vocab is always - pickled because its compiled-regex tokenizer state isn't - JSON-serializable). Falls through to raw `pickle.load` if the - MolVocab / SMILESVocab loader can't import or fails to recognize - the contents (e.g. test fixtures with plain dicts). + - Legacy MolVocab / SMILESVocab pickles, read as inert vocabulary state. + + The pickle reader accepts only the known vocabulary containers and their + Counter/regex metadata. It cannot import arbitrary classes or run reducers + supplied by the artifact, and it never falls back to an unrestricted loader. """ if vocab_path.suffix == ".json": data = json.loads(vocab_path.read_text()) @@ -86,28 +89,88 @@ def count_vocab_entries(vocab_path: Path) -> int: return len(data) raise ValueError(f"unsupported JSON vocab shape at {vocab_path}: {type(data).__name__}") - # .pkl: try MolVocab / SMILESVocab first, then fall back to raw pickle. - try: - from kermt.data.torchvocab import MolVocab, SMILESVocab # type: ignore - for loader in (MolVocab.load_vocab, SMILESVocab.load_vocab): - try: - v = loader(str(vocab_path)) - return len(v) - except Exception: - continue - except ImportError: - pass - - import pickle with vocab_path.open("rb") as f: - data = pickle.load(f) - if hasattr(data, "stoi"): + data = _VocabUnpickler(f).load() + if isinstance(data, _VocabState) and isinstance(data.stoi, dict): return len(data.stoi) - if hasattr(data, "__len__"): + if isinstance(data, (dict, list, tuple)): return len(data) raise ValueError(f"could not count entries in {vocab_path}") +class _VocabState: + """Data-only stand-in: counting tokens does not require tokenizer methods.""" + + +class _VocabUnpickler(pickle.Unpickler): + def find_class(self, module: str, name: str) -> Any: + if module in {"kermt.data.torchvocab", "grover.data.torchvocab"} and name in { + "TorchVocab", "MolVocab", "SMILESVocab", + }: + return _VocabState + if (module, name) == ("collections", "Counter"): + return Counter + if (module, name) == ("re", "_compile"): + return re.compile + raise pickle.UnpicklingError(f"unsupported vocabulary object: {module}.{name}") + + +def load_checkpoint(path: Path | str) -> dict[str, Any]: + """Read KERMT tensors and known metadata with PyTorch's restricted loader. + + Saved arguments use argparse.Namespace; finetuned checkpoints also contain + numeric NumPy scaler arrays. Explicit globals cover those formats, including + NumPy 1/2 module names, without accepting artifact-selected imports. + """ + import numpy as np + import torch + + multiarray = np._core.multiarray if hasattr(np, "_core") else np.core.multiarray + allowed = [argparse.Namespace, np.ndarray, np.dtype] + for module in ("numpy.core.multiarray", "numpy._core.multiarray"): + allowed.extend([ + (multiarray._reconstruct, f"{module}._reconstruct"), + (multiarray.scalar, f"{module}.scalar"), + ]) + allowed.extend(type(np.dtype(name)) for name in ( + "bool", "int8", "int16", "int32", "int64", "uint8", "uint16", "uint32", "uint64", + "float16", "float32", "float64", + )) + with torch.serialization.safe_globals(allowed): + return torch.load(path, map_location="cpu", weights_only=True) + + +def runner_environment(repo: Path, *, wandb: bool = False) -> dict[str, str]: + """Forward named runtime settings, keeping unrelated credentials out of jobs. + + W&B credentials/settings are included only for an explicitly enabled W&B + run. Hugging Face authentication belongs to the separate download helper. + """ + names = ( + "PATH", "HOME", "TMPDIR", "TEMP", "TMP", "LANG", "LC_ALL", "LC_CTYPE", "TZ", + "LD_LIBRARY_PATH", "LIBRARY_PATH", "CUDA_HOME", "CUDA_PATH", "PYTHONPATH", + "PYTHONDONTWRITEBYTECODE", "PYTHONUNBUFFERED", "PYTHONWARNINGS", + "CUDA_VISIBLE_DEVICES", "CUDA_DEVICE_ORDER", "CUDA_LAUNCH_BLOCKING", "NVIDIA_VISIBLE_DEVICES", + "NVIDIA_DRIVER_CAPABILITIES", "CUBLAS_WORKSPACE_CONFIG", "OMP_NUM_THREADS", + "MKL_NUM_THREADS", "OPENBLAS_NUM_THREADS", "NUMEXPR_NUM_THREADS", + "PYTORCH_CUDA_ALLOC_CONF", "PYTORCH_ALLOC_CONF", "PYTORCH_NO_CUDA_MEMORY_CACHING", + "TORCH_CPP_LOG_LEVEL", "TORCH_DISTRIBUTED_DEBUG", "NCCL_DEBUG", "NCCL_SOCKET_IFNAME", + "NCCL_IB_DISABLE", "NCCL_P2P_DISABLE", "NCCL_SHM_DISABLE", "GLOO_SOCKET_IFNAME", + "MASTER_ADDR", "MASTER_PORT", + "KERMT_REPO", "KERMT_REPO_COMMIT", "KERMT_REPO_DIRTY", + "SSL_CERT_FILE", "REQUESTS_CA_BUNDLE", + ) + if wandb: + names += ( + "WANDB_API_KEY", "WANDB_BASE_URL", "WANDB_MODE", "WANDB_DIR", "WANDB_ENTITY", + "WANDB_PROJECT", "WANDB_RUN_ID", "WANDB_RESUME", "WANDB_CACHE_DIR", + "WANDB_CONFIG_DIR", "WANDB_DATA_DIR", "WANDB_DISABLED", + ) + env = {name: value for name in names if (value := os.environ.get(name)) is not None} + env["PYTHONPATH"] = os.pathsep.join(filter(None, (str(repo), env.get("PYTHONPATH")))) + return env + + def validate_vocab_file(vocab_path: Path, *, kind: str) -> None: """Verify a user-provided vocab file is loadable BEFORE copying it into a run directory. Raises ValueError on failure with a clear, user-facing message. diff --git a/skills/kermt-add-cmim-pretrain/scripts/check_checkpoint.py b/skills/kermt-add-cmim-pretrain/scripts/check_checkpoint.py index fd488e2..5df7627 100644 --- a/skills/kermt-add-cmim-pretrain/scripts/check_checkpoint.py +++ b/skills/kermt-add-cmim-pretrain/scripts/check_checkpoint.py @@ -75,10 +75,15 @@ import sys import traceback from argparse import Namespace +from pathlib import Path from typing import Any import torch +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import load_checkpoint # noqa: E402 + # --------------------------------------------------------------------------- # State-dict key prefix conventions (kermt/model/models.py). @@ -291,7 +296,7 @@ def _arch_from_shapes(state_dict: dict[str, Any], arch: dict[str, Any]) -> tuple ] if candidates: arch["latent_dim"] = int(state_dict[candidates[0]].shape[0]) - # Absent latent_dist entirely is a fact about the model, not a problem — no warning. + # Encoder-only models have no latent distribution; latent_dim remains None. # depth, num_attn_head, activation, backbone, embedding_output_type, self_attention # are not robustly inferable from shapes alone; report a warning for each that's @@ -390,7 +395,7 @@ def validate(mode: str, ckpt_path: str) -> dict[str, Any]: # 1. Load the checkpoint. try: - ckpt = torch.load(ckpt_path, map_location="cpu", weights_only=False) + ckpt = load_checkpoint(ckpt_path) except FileNotFoundError: result["errors"].append(f"checkpoint not found: {ckpt_path}") return result diff --git a/skills/kermt-add-cmim-pretrain/scripts/run_pretrain_local.py b/skills/kermt-add-cmim-pretrain/scripts/run_pretrain_local.py index 5b8a8a1..6f74382 100644 --- a/skills/kermt-add-cmim-pretrain/scripts/run_pretrain_local.py +++ b/skills/kermt-add-cmim-pretrain/scripts/run_pretrain_local.py @@ -53,8 +53,8 @@ sys.path.insert(0, str(Path(__file__).resolve().parent)) from _utils import ( # noqa: E402 resolve_kermt_repo, assert_prepare_manifest_basics, count_vocab_entries, docker_image_digest, - format_cmd_replay, git_commit_with_env_override, load_json, - merge_default_into_applied, run_checkpoint_validator, + format_cmd_replay, git_commit_with_env_override, load_json, load_checkpoint, + merge_default_into_applied, run_checkpoint_validator, runner_environment, ) @@ -315,7 +315,7 @@ def _materialize_ckpt_for_fresh_schedule(user_ckpt: Path, save_dir: Path) -> Pat target = save_dir / "last_checkpoint.pt" if target.exists() or target.is_symlink(): target.unlink() - ckpt = torch.load(user_ckpt, map_location="cpu", weights_only=False) + ckpt = load_checkpoint(user_ckpt) if not isinstance(ckpt, dict) or "state_dict" not in ckpt: raise ValueError( f"ckpt {user_ckpt} is not in the expected save_model_for_restart " @@ -335,8 +335,7 @@ def _validate_resume_state(user_ckpt: Path) -> dict[str, Any]: (optimizer state, scheduler_step, epoch, batch_idx). Returns a small `resume_state` dict for the manifest so users can see what was restored. Raises ValueError with a clear redirect if the ckpt is too lean.""" - import torch - ckpt = torch.load(user_ckpt, map_location="cpu", weights_only=False) + ckpt = load_checkpoint(user_ckpt) if not isinstance(ckpt, dict) or "state_dict" not in ckpt: raise ValueError( f"ckpt {user_ckpt} is not in the expected save_model_for_restart " @@ -645,7 +644,7 @@ def run(args: argparse.Namespace) -> dict[str, Any]: run_manifest["status"] = "dry_run" return run_manifest - env = os.environ.copy() + env = runner_environment(REPO_ROOT, wandb="wandb_project" in applied) env["WORLD_SIZE"] = str(world_size) if gpus_str: env["CUDA_VISIBLE_DEVICES"] = gpus_str diff --git a/skills/kermt-add-cmim-pretrain/scripts/upgrade_to_hybrid.py b/skills/kermt-add-cmim-pretrain/scripts/upgrade_to_hybrid.py index 65c93f6..5895621 100644 --- a/skills/kermt-add-cmim-pretrain/scripts/upgrade_to_hybrid.py +++ b/skills/kermt-add-cmim-pretrain/scripts/upgrade_to_hybrid.py @@ -65,7 +65,7 @@ # or as a bare `python scripts/upgrade_to_hybrid.py …`. if str(Path(__file__).resolve().parent) not in sys.path: sys.path.insert(0, str(Path(__file__).resolve().parent)) -from _utils import count_vocab_entries, load_json, run_checkpoint_validator # noqa: E402 +from _utils import count_vocab_entries, load_checkpoint, load_json, run_checkpoint_validator # noqa: E402 SKILL_ROOT = Path(__file__).resolve().parent.parent @@ -251,7 +251,7 @@ def upgrade(args: argparse.Namespace) -> dict[str, Any]: decoder_defaults = defaults.get("add_cmim_decoder") or {} latent_dim = args.latent_dim if args.latent_dim is not None else decoder_defaults.get("latent_dim", 800) - input_ckpt = torch.load(args.ckpt, map_location="cpu", weights_only=False) + input_ckpt = load_checkpoint(args.ckpt) input_args = input_ckpt.get("args") if input_args is None: raise ValueError( diff --git a/skills/kermt-continue-pretrain/SKILL.md b/skills/kermt-continue-pretrain/SKILL.md index 32c17ef..c876cfb 100644 --- a/skills/kermt-continue-pretrain/SKILL.md +++ b/skills/kermt-continue-pretrain/SKILL.md @@ -1,6 +1,6 @@ --- name: kermt-continue-pretrain -description: Continue pretraining from an existing KERMT checkpoint. The skill validates the user's checkpoint and pretrain CSV, prepares the data into shard/vocab/features form, then launches pretrain_ddp.py inside the kermt container (detached for long runs). Auto-dispatches `--pretrain_mode` based on the checkpoint type (grover_base vocab-only, cmim, or hybrid). +description: Continue KERMT pretraining on a custom SMILES corpus with a grover_base, cmim, or hybrid checkpoint. Use a local checkpoint or optionally download a pinned Hugging Face model bundle using HF_TOKEN if configured. Run containerized training and write model bundles, prepared data, logs, and checkpoints to user-selected host directories. license: Apache-2.0 compatibility: Requires docker, nvidia-container-toolkit, and a CUDA-capable NVIDIA GPU. Designed for Claude Code, Codex, and Nemotron. metadata: @@ -20,13 +20,23 @@ prepares the corpus, launches the runner, and returns a run directory. ## Skill and runtime paths -Set `SKILL_DIR` to the directory containing this `SKILL.md`. Export +Set `SKILL_DIR` to the absolute path of this installed skill directory. Export `KERMT_REPO` as the absolute path to the KERMT checkout used for model execution. The bundled container helper mounts that checkout at `/workspace` and this skill at `/skill` (read-only). Commands inside the container use `/skill/scripts/`; defaults are bundled in `config/`. See [Released models](references/released-models.md) for checkpoint bundle requirements. +## Downloads and local outputs + +The optional released-model branch reads `config/released_model.json` for the +Hugging Face repository, pinned revision, and filenames. The bundled +`scripts/fetch_released_model.py` downloads the model bundle over HTTPS into +the host directory the user selects. Public models work without credentials; +if `HF_TOKEN` is set, the container helper forwards it for Hugging Face +authentication. Prepared data, logs, and workflow results go into the chosen +run directory. + ## Hardware requirements - **GPUs**: 1–N CUDA-capable NVIDIA GPUs. The runner auto-detects via diff --git a/skills/kermt-continue-pretrain/scripts/_utils.py b/skills/kermt-continue-pretrain/scripts/_utils.py index 5bde460..7fe75b0 100644 --- a/skills/kermt-continue-pretrain/scripts/_utils.py +++ b/skills/kermt-continue-pretrain/scripts/_utils.py @@ -10,8 +10,11 @@ from __future__ import annotations import argparse +from collections import Counter import json import os +import pickle +import re import shlex import subprocess import sys @@ -72,11 +75,11 @@ def count_vocab_entries(vocab_path: Path) -> int: Handles three layouts: - JSON with `{stoi: {token: idx}, ...}` (MolVocab.save_vocab default) - JSON as a raw `{token: idx}` dict (legacy / hand-edited) - - Pickle of a MolVocab / SMILESVocab object (the smiles vocab is always - pickled because its compiled-regex tokenizer state isn't - JSON-serializable). Falls through to raw `pickle.load` if the - MolVocab / SMILESVocab loader can't import or fails to recognize - the contents (e.g. test fixtures with plain dicts). + - Legacy MolVocab / SMILESVocab pickles, read as inert vocabulary state. + + The pickle reader accepts only the known vocabulary containers and their + Counter/regex metadata. It cannot import arbitrary classes or run reducers + supplied by the artifact, and it never falls back to an unrestricted loader. """ if vocab_path.suffix == ".json": data = json.loads(vocab_path.read_text()) @@ -86,28 +89,88 @@ def count_vocab_entries(vocab_path: Path) -> int: return len(data) raise ValueError(f"unsupported JSON vocab shape at {vocab_path}: {type(data).__name__}") - # .pkl: try MolVocab / SMILESVocab first, then fall back to raw pickle. - try: - from kermt.data.torchvocab import MolVocab, SMILESVocab # type: ignore - for loader in (MolVocab.load_vocab, SMILESVocab.load_vocab): - try: - v = loader(str(vocab_path)) - return len(v) - except Exception: - continue - except ImportError: - pass - - import pickle with vocab_path.open("rb") as f: - data = pickle.load(f) - if hasattr(data, "stoi"): + data = _VocabUnpickler(f).load() + if isinstance(data, _VocabState) and isinstance(data.stoi, dict): return len(data.stoi) - if hasattr(data, "__len__"): + if isinstance(data, (dict, list, tuple)): return len(data) raise ValueError(f"could not count entries in {vocab_path}") +class _VocabState: + """Data-only stand-in: counting tokens does not require tokenizer methods.""" + + +class _VocabUnpickler(pickle.Unpickler): + def find_class(self, module: str, name: str) -> Any: + if module in {"kermt.data.torchvocab", "grover.data.torchvocab"} and name in { + "TorchVocab", "MolVocab", "SMILESVocab", + }: + return _VocabState + if (module, name) == ("collections", "Counter"): + return Counter + if (module, name) == ("re", "_compile"): + return re.compile + raise pickle.UnpicklingError(f"unsupported vocabulary object: {module}.{name}") + + +def load_checkpoint(path: Path | str) -> dict[str, Any]: + """Read KERMT tensors and known metadata with PyTorch's restricted loader. + + Saved arguments use argparse.Namespace; finetuned checkpoints also contain + numeric NumPy scaler arrays. Explicit globals cover those formats, including + NumPy 1/2 module names, without accepting artifact-selected imports. + """ + import numpy as np + import torch + + multiarray = np._core.multiarray if hasattr(np, "_core") else np.core.multiarray + allowed = [argparse.Namespace, np.ndarray, np.dtype] + for module in ("numpy.core.multiarray", "numpy._core.multiarray"): + allowed.extend([ + (multiarray._reconstruct, f"{module}._reconstruct"), + (multiarray.scalar, f"{module}.scalar"), + ]) + allowed.extend(type(np.dtype(name)) for name in ( + "bool", "int8", "int16", "int32", "int64", "uint8", "uint16", "uint32", "uint64", + "float16", "float32", "float64", + )) + with torch.serialization.safe_globals(allowed): + return torch.load(path, map_location="cpu", weights_only=True) + + +def runner_environment(repo: Path, *, wandb: bool = False) -> dict[str, str]: + """Forward named runtime settings, keeping unrelated credentials out of jobs. + + W&B credentials/settings are included only for an explicitly enabled W&B + run. Hugging Face authentication belongs to the separate download helper. + """ + names = ( + "PATH", "HOME", "TMPDIR", "TEMP", "TMP", "LANG", "LC_ALL", "LC_CTYPE", "TZ", + "LD_LIBRARY_PATH", "LIBRARY_PATH", "CUDA_HOME", "CUDA_PATH", "PYTHONPATH", + "PYTHONDONTWRITEBYTECODE", "PYTHONUNBUFFERED", "PYTHONWARNINGS", + "CUDA_VISIBLE_DEVICES", "CUDA_DEVICE_ORDER", "CUDA_LAUNCH_BLOCKING", "NVIDIA_VISIBLE_DEVICES", + "NVIDIA_DRIVER_CAPABILITIES", "CUBLAS_WORKSPACE_CONFIG", "OMP_NUM_THREADS", + "MKL_NUM_THREADS", "OPENBLAS_NUM_THREADS", "NUMEXPR_NUM_THREADS", + "PYTORCH_CUDA_ALLOC_CONF", "PYTORCH_ALLOC_CONF", "PYTORCH_NO_CUDA_MEMORY_CACHING", + "TORCH_CPP_LOG_LEVEL", "TORCH_DISTRIBUTED_DEBUG", "NCCL_DEBUG", "NCCL_SOCKET_IFNAME", + "NCCL_IB_DISABLE", "NCCL_P2P_DISABLE", "NCCL_SHM_DISABLE", "GLOO_SOCKET_IFNAME", + "MASTER_ADDR", "MASTER_PORT", + "KERMT_REPO", "KERMT_REPO_COMMIT", "KERMT_REPO_DIRTY", + "SSL_CERT_FILE", "REQUESTS_CA_BUNDLE", + ) + if wandb: + names += ( + "WANDB_API_KEY", "WANDB_BASE_URL", "WANDB_MODE", "WANDB_DIR", "WANDB_ENTITY", + "WANDB_PROJECT", "WANDB_RUN_ID", "WANDB_RESUME", "WANDB_CACHE_DIR", + "WANDB_CONFIG_DIR", "WANDB_DATA_DIR", "WANDB_DISABLED", + ) + env = {name: value for name in names if (value := os.environ.get(name)) is not None} + env["PYTHONPATH"] = os.pathsep.join(filter(None, (str(repo), env.get("PYTHONPATH")))) + return env + + def validate_vocab_file(vocab_path: Path, *, kind: str) -> None: """Verify a user-provided vocab file is loadable BEFORE copying it into a run directory. Raises ValueError on failure with a clear, user-facing message. diff --git a/skills/kermt-continue-pretrain/scripts/check_checkpoint.py b/skills/kermt-continue-pretrain/scripts/check_checkpoint.py index fd488e2..5df7627 100644 --- a/skills/kermt-continue-pretrain/scripts/check_checkpoint.py +++ b/skills/kermt-continue-pretrain/scripts/check_checkpoint.py @@ -75,10 +75,15 @@ import sys import traceback from argparse import Namespace +from pathlib import Path from typing import Any import torch +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import load_checkpoint # noqa: E402 + # --------------------------------------------------------------------------- # State-dict key prefix conventions (kermt/model/models.py). @@ -291,7 +296,7 @@ def _arch_from_shapes(state_dict: dict[str, Any], arch: dict[str, Any]) -> tuple ] if candidates: arch["latent_dim"] = int(state_dict[candidates[0]].shape[0]) - # Absent latent_dist entirely is a fact about the model, not a problem — no warning. + # Encoder-only models have no latent distribution; latent_dim remains None. # depth, num_attn_head, activation, backbone, embedding_output_type, self_attention # are not robustly inferable from shapes alone; report a warning for each that's @@ -390,7 +395,7 @@ def validate(mode: str, ckpt_path: str) -> dict[str, Any]: # 1. Load the checkpoint. try: - ckpt = torch.load(ckpt_path, map_location="cpu", weights_only=False) + ckpt = load_checkpoint(ckpt_path) except FileNotFoundError: result["errors"].append(f"checkpoint not found: {ckpt_path}") return result diff --git a/skills/kermt-continue-pretrain/scripts/run_pretrain_local.py b/skills/kermt-continue-pretrain/scripts/run_pretrain_local.py index 5b8a8a1..6f74382 100644 --- a/skills/kermt-continue-pretrain/scripts/run_pretrain_local.py +++ b/skills/kermt-continue-pretrain/scripts/run_pretrain_local.py @@ -53,8 +53,8 @@ sys.path.insert(0, str(Path(__file__).resolve().parent)) from _utils import ( # noqa: E402 resolve_kermt_repo, assert_prepare_manifest_basics, count_vocab_entries, docker_image_digest, - format_cmd_replay, git_commit_with_env_override, load_json, - merge_default_into_applied, run_checkpoint_validator, + format_cmd_replay, git_commit_with_env_override, load_json, load_checkpoint, + merge_default_into_applied, run_checkpoint_validator, runner_environment, ) @@ -315,7 +315,7 @@ def _materialize_ckpt_for_fresh_schedule(user_ckpt: Path, save_dir: Path) -> Pat target = save_dir / "last_checkpoint.pt" if target.exists() or target.is_symlink(): target.unlink() - ckpt = torch.load(user_ckpt, map_location="cpu", weights_only=False) + ckpt = load_checkpoint(user_ckpt) if not isinstance(ckpt, dict) or "state_dict" not in ckpt: raise ValueError( f"ckpt {user_ckpt} is not in the expected save_model_for_restart " @@ -335,8 +335,7 @@ def _validate_resume_state(user_ckpt: Path) -> dict[str, Any]: (optimizer state, scheduler_step, epoch, batch_idx). Returns a small `resume_state` dict for the manifest so users can see what was restored. Raises ValueError with a clear redirect if the ckpt is too lean.""" - import torch - ckpt = torch.load(user_ckpt, map_location="cpu", weights_only=False) + ckpt = load_checkpoint(user_ckpt) if not isinstance(ckpt, dict) or "state_dict" not in ckpt: raise ValueError( f"ckpt {user_ckpt} is not in the expected save_model_for_restart " @@ -645,7 +644,7 @@ def run(args: argparse.Namespace) -> dict[str, Any]: run_manifest["status"] = "dry_run" return run_manifest - env = os.environ.copy() + env = runner_environment(REPO_ROOT, wandb="wandb_project" in applied) env["WORLD_SIZE"] = str(world_size) if gpus_str: env["CUDA_VISIBLE_DEVICES"] = gpus_str diff --git a/skills/kermt-embed/SKILL.md b/skills/kermt-embed/SKILL.md index 26b6955..85fcf8f 100644 --- a/skills/kermt-embed/SKILL.md +++ b/skills/kermt-embed/SKILL.md @@ -1,6 +1,6 @@ --- name: kermt-embed -description: Extract per-molecule embeddings from any encoder-bearing KERMT checkpoint (grover_base / cmim / hybrid / finetuned). Writes one .npy per readout type (atom_from_atom, bond_from_atom, atom_from_bond, bond_from_bond) plus canonical_smiles.npy and validity.npy. Calls task/extract_embeddings.py (which featurizes SMILES on the fly — no pre-computed features needed). +description: Extract per-molecule embeddings from any encoder-bearing KERMT checkpoint. Use a local checkpoint or optionally download a pinned Hugging Face model bundle using HF_TOKEN if configured. Run containerized embedding extraction and write model bundles, per-readout .npy embeddings, canonical SMILES, and validity arrays to user-selected host directories. license: Apache-2.0 compatibility: Requires docker, nvidia-container-toolkit, and a CUDA-capable NVIDIA GPU. Designed for Claude Code, Codex, and Nemotron. metadata: @@ -19,13 +19,23 @@ SMILES, launch the runner blocking, return the per-readout `.npy` files. ## Skill and runtime paths -Set `SKILL_DIR` to the directory containing this `SKILL.md`. Export +Set `SKILL_DIR` to the absolute path of this installed skill directory. Export `KERMT_REPO` as the absolute path to the KERMT checkout used for model execution. The bundled container helper mounts that checkout at `/workspace` and this skill at `/skill` (read-only). Commands inside the container use `/skill/scripts/`; defaults are bundled in `config/`. See [Released models](references/released-models.md) for checkpoint bundle requirements. +## Downloads and local outputs + +The optional released-model branch reads `config/released_model.json` for the +Hugging Face repository, pinned revision, and filenames. The bundled +`scripts/fetch_released_model.py` downloads the model bundle over HTTPS into +the host directory the user selects. Public models work without credentials; +if `HF_TOKEN` is set, the container helper forwards it for Hugging Face +authentication. Prepared data, logs, and workflow results go into the chosen +run directory. + ## Hardware requirements - **GPUs**: 1 (single-GPU). diff --git a/skills/kermt-embed/scripts/_utils.py b/skills/kermt-embed/scripts/_utils.py index 5bde460..7fe75b0 100644 --- a/skills/kermt-embed/scripts/_utils.py +++ b/skills/kermt-embed/scripts/_utils.py @@ -10,8 +10,11 @@ from __future__ import annotations import argparse +from collections import Counter import json import os +import pickle +import re import shlex import subprocess import sys @@ -72,11 +75,11 @@ def count_vocab_entries(vocab_path: Path) -> int: Handles three layouts: - JSON with `{stoi: {token: idx}, ...}` (MolVocab.save_vocab default) - JSON as a raw `{token: idx}` dict (legacy / hand-edited) - - Pickle of a MolVocab / SMILESVocab object (the smiles vocab is always - pickled because its compiled-regex tokenizer state isn't - JSON-serializable). Falls through to raw `pickle.load` if the - MolVocab / SMILESVocab loader can't import or fails to recognize - the contents (e.g. test fixtures with plain dicts). + - Legacy MolVocab / SMILESVocab pickles, read as inert vocabulary state. + + The pickle reader accepts only the known vocabulary containers and their + Counter/regex metadata. It cannot import arbitrary classes or run reducers + supplied by the artifact, and it never falls back to an unrestricted loader. """ if vocab_path.suffix == ".json": data = json.loads(vocab_path.read_text()) @@ -86,28 +89,88 @@ def count_vocab_entries(vocab_path: Path) -> int: return len(data) raise ValueError(f"unsupported JSON vocab shape at {vocab_path}: {type(data).__name__}") - # .pkl: try MolVocab / SMILESVocab first, then fall back to raw pickle. - try: - from kermt.data.torchvocab import MolVocab, SMILESVocab # type: ignore - for loader in (MolVocab.load_vocab, SMILESVocab.load_vocab): - try: - v = loader(str(vocab_path)) - return len(v) - except Exception: - continue - except ImportError: - pass - - import pickle with vocab_path.open("rb") as f: - data = pickle.load(f) - if hasattr(data, "stoi"): + data = _VocabUnpickler(f).load() + if isinstance(data, _VocabState) and isinstance(data.stoi, dict): return len(data.stoi) - if hasattr(data, "__len__"): + if isinstance(data, (dict, list, tuple)): return len(data) raise ValueError(f"could not count entries in {vocab_path}") +class _VocabState: + """Data-only stand-in: counting tokens does not require tokenizer methods.""" + + +class _VocabUnpickler(pickle.Unpickler): + def find_class(self, module: str, name: str) -> Any: + if module in {"kermt.data.torchvocab", "grover.data.torchvocab"} and name in { + "TorchVocab", "MolVocab", "SMILESVocab", + }: + return _VocabState + if (module, name) == ("collections", "Counter"): + return Counter + if (module, name) == ("re", "_compile"): + return re.compile + raise pickle.UnpicklingError(f"unsupported vocabulary object: {module}.{name}") + + +def load_checkpoint(path: Path | str) -> dict[str, Any]: + """Read KERMT tensors and known metadata with PyTorch's restricted loader. + + Saved arguments use argparse.Namespace; finetuned checkpoints also contain + numeric NumPy scaler arrays. Explicit globals cover those formats, including + NumPy 1/2 module names, without accepting artifact-selected imports. + """ + import numpy as np + import torch + + multiarray = np._core.multiarray if hasattr(np, "_core") else np.core.multiarray + allowed = [argparse.Namespace, np.ndarray, np.dtype] + for module in ("numpy.core.multiarray", "numpy._core.multiarray"): + allowed.extend([ + (multiarray._reconstruct, f"{module}._reconstruct"), + (multiarray.scalar, f"{module}.scalar"), + ]) + allowed.extend(type(np.dtype(name)) for name in ( + "bool", "int8", "int16", "int32", "int64", "uint8", "uint16", "uint32", "uint64", + "float16", "float32", "float64", + )) + with torch.serialization.safe_globals(allowed): + return torch.load(path, map_location="cpu", weights_only=True) + + +def runner_environment(repo: Path, *, wandb: bool = False) -> dict[str, str]: + """Forward named runtime settings, keeping unrelated credentials out of jobs. + + W&B credentials/settings are included only for an explicitly enabled W&B + run. Hugging Face authentication belongs to the separate download helper. + """ + names = ( + "PATH", "HOME", "TMPDIR", "TEMP", "TMP", "LANG", "LC_ALL", "LC_CTYPE", "TZ", + "LD_LIBRARY_PATH", "LIBRARY_PATH", "CUDA_HOME", "CUDA_PATH", "PYTHONPATH", + "PYTHONDONTWRITEBYTECODE", "PYTHONUNBUFFERED", "PYTHONWARNINGS", + "CUDA_VISIBLE_DEVICES", "CUDA_DEVICE_ORDER", "CUDA_LAUNCH_BLOCKING", "NVIDIA_VISIBLE_DEVICES", + "NVIDIA_DRIVER_CAPABILITIES", "CUBLAS_WORKSPACE_CONFIG", "OMP_NUM_THREADS", + "MKL_NUM_THREADS", "OPENBLAS_NUM_THREADS", "NUMEXPR_NUM_THREADS", + "PYTORCH_CUDA_ALLOC_CONF", "PYTORCH_ALLOC_CONF", "PYTORCH_NO_CUDA_MEMORY_CACHING", + "TORCH_CPP_LOG_LEVEL", "TORCH_DISTRIBUTED_DEBUG", "NCCL_DEBUG", "NCCL_SOCKET_IFNAME", + "NCCL_IB_DISABLE", "NCCL_P2P_DISABLE", "NCCL_SHM_DISABLE", "GLOO_SOCKET_IFNAME", + "MASTER_ADDR", "MASTER_PORT", + "KERMT_REPO", "KERMT_REPO_COMMIT", "KERMT_REPO_DIRTY", + "SSL_CERT_FILE", "REQUESTS_CA_BUNDLE", + ) + if wandb: + names += ( + "WANDB_API_KEY", "WANDB_BASE_URL", "WANDB_MODE", "WANDB_DIR", "WANDB_ENTITY", + "WANDB_PROJECT", "WANDB_RUN_ID", "WANDB_RESUME", "WANDB_CACHE_DIR", + "WANDB_CONFIG_DIR", "WANDB_DATA_DIR", "WANDB_DISABLED", + ) + env = {name: value for name in names if (value := os.environ.get(name)) is not None} + env["PYTHONPATH"] = os.pathsep.join(filter(None, (str(repo), env.get("PYTHONPATH")))) + return env + + def validate_vocab_file(vocab_path: Path, *, kind: str) -> None: """Verify a user-provided vocab file is loadable BEFORE copying it into a run directory. Raises ValueError on failure with a clear, user-facing message. diff --git a/skills/kermt-embed/scripts/check_checkpoint.py b/skills/kermt-embed/scripts/check_checkpoint.py index fd488e2..5df7627 100644 --- a/skills/kermt-embed/scripts/check_checkpoint.py +++ b/skills/kermt-embed/scripts/check_checkpoint.py @@ -75,10 +75,15 @@ import sys import traceback from argparse import Namespace +from pathlib import Path from typing import Any import torch +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import load_checkpoint # noqa: E402 + # --------------------------------------------------------------------------- # State-dict key prefix conventions (kermt/model/models.py). @@ -291,7 +296,7 @@ def _arch_from_shapes(state_dict: dict[str, Any], arch: dict[str, Any]) -> tuple ] if candidates: arch["latent_dim"] = int(state_dict[candidates[0]].shape[0]) - # Absent latent_dist entirely is a fact about the model, not a problem — no warning. + # Encoder-only models have no latent distribution; latent_dim remains None. # depth, num_attn_head, activation, backbone, embedding_output_type, self_attention # are not robustly inferable from shapes alone; report a warning for each that's @@ -390,7 +395,7 @@ def validate(mode: str, ckpt_path: str) -> dict[str, Any]: # 1. Load the checkpoint. try: - ckpt = torch.load(ckpt_path, map_location="cpu", weights_only=False) + ckpt = load_checkpoint(ckpt_path) except FileNotFoundError: result["errors"].append(f"checkpoint not found: {ckpt_path}") return result diff --git a/skills/kermt-embed/scripts/run_extract_embeddings.py b/skills/kermt-embed/scripts/run_extract_embeddings.py index 7ba118c..fb9def2 100644 --- a/skills/kermt-embed/scripts/run_extract_embeddings.py +++ b/skills/kermt-embed/scripts/run_extract_embeddings.py @@ -47,7 +47,7 @@ from _utils import ( # noqa: E402 resolve_kermt_repo, assert_prepare_manifest_basics, docker_image_digest, format_cmd_replay, git_commit_with_env_override, load_json, merge_default_into_applied, - resolve_single_gpu, run_checkpoint_validator, + resolve_single_gpu, run_checkpoint_validator, runner_environment, ) @@ -173,7 +173,7 @@ def run(args: argparse.Namespace) -> dict[str, Any]: run_manifest["status"] = "dry_run" return run_manifest - env = os.environ.copy() + env = runner_environment(REPO_ROOT) env["CUDA_VISIBLE_DEVICES"] = str(gpu) log_file = out_dir / "logs" / "embed.log" with log_file.open("w") as logf: diff --git a/skills/kermt-finetune/SKILL.md b/skills/kermt-finetune/SKILL.md index 3a80a49..937abcd 100644 --- a/skills/kermt-finetune/SKILL.md +++ b/skills/kermt-finetune/SKILL.md @@ -1,6 +1,6 @@ --- name: kermt-finetune -description: Finetune a pretrained KERMT encoder on a labeled CSV. The skill validates the input checkpoint (must be a pretrain ckpt — grover_base / cmim / hybrid), validates the labeled CSV, prepares the data (clean + features + optional split), then launches main.py finetune inside the kermt container (detached for hours-scale runs). Hyperparameters come from config/defaults_finetune.json with per-flag CLI override. +description: Finetune a pretrained KERMT encoder on a labeled CSV. Validate the checkpoint and data, prepare features, and run containerized training. Use a local checkpoint or optionally download a pinned Hugging Face model bundle using HF_TOKEN if configured. Write model bundles, prepared data, logs, and trained models to user-selected host directories. license: Apache-2.0 compatibility: Requires docker, nvidia-container-toolkit, and a CUDA-capable NVIDIA GPU. Designed for Claude Code, Codex, and Nemotron. metadata: @@ -19,13 +19,23 @@ launch the runner detached, return a run directory + container name. ## Skill and runtime paths -Set `SKILL_DIR` to the directory containing this `SKILL.md`. Export +Set `SKILL_DIR` to the absolute path of this installed skill directory. Export `KERMT_REPO` as the absolute path to the KERMT checkout used for model execution. The bundled container helper mounts that checkout at `/workspace` and this skill at `/skill` (read-only). Commands inside the container use `/skill/scripts/`; defaults are bundled in `config/`. See [Released models](references/released-models.md) for checkpoint bundle requirements. +## Downloads and local outputs + +The optional released-model branch reads `config/released_model.json` for the +Hugging Face repository, pinned revision, and filenames. The bundled +`scripts/fetch_released_model.py` downloads the model bundle over HTTPS into +the host directory the user selects. Public models work without credentials; +if `HF_TOKEN` is set, the container helper forwards it for Hugging Face +authentication. Prepared data, logs, and workflow results go into the chosen +run directory. + ## Hardware requirements - **GPUs**: 1 by default (single-GPU); pass `--gpus 0` (or whichever id) to diff --git a/skills/kermt-finetune/scripts/_utils.py b/skills/kermt-finetune/scripts/_utils.py index 5bde460..7fe75b0 100644 --- a/skills/kermt-finetune/scripts/_utils.py +++ b/skills/kermt-finetune/scripts/_utils.py @@ -10,8 +10,11 @@ from __future__ import annotations import argparse +from collections import Counter import json import os +import pickle +import re import shlex import subprocess import sys @@ -72,11 +75,11 @@ def count_vocab_entries(vocab_path: Path) -> int: Handles three layouts: - JSON with `{stoi: {token: idx}, ...}` (MolVocab.save_vocab default) - JSON as a raw `{token: idx}` dict (legacy / hand-edited) - - Pickle of a MolVocab / SMILESVocab object (the smiles vocab is always - pickled because its compiled-regex tokenizer state isn't - JSON-serializable). Falls through to raw `pickle.load` if the - MolVocab / SMILESVocab loader can't import or fails to recognize - the contents (e.g. test fixtures with plain dicts). + - Legacy MolVocab / SMILESVocab pickles, read as inert vocabulary state. + + The pickle reader accepts only the known vocabulary containers and their + Counter/regex metadata. It cannot import arbitrary classes or run reducers + supplied by the artifact, and it never falls back to an unrestricted loader. """ if vocab_path.suffix == ".json": data = json.loads(vocab_path.read_text()) @@ -86,28 +89,88 @@ def count_vocab_entries(vocab_path: Path) -> int: return len(data) raise ValueError(f"unsupported JSON vocab shape at {vocab_path}: {type(data).__name__}") - # .pkl: try MolVocab / SMILESVocab first, then fall back to raw pickle. - try: - from kermt.data.torchvocab import MolVocab, SMILESVocab # type: ignore - for loader in (MolVocab.load_vocab, SMILESVocab.load_vocab): - try: - v = loader(str(vocab_path)) - return len(v) - except Exception: - continue - except ImportError: - pass - - import pickle with vocab_path.open("rb") as f: - data = pickle.load(f) - if hasattr(data, "stoi"): + data = _VocabUnpickler(f).load() + if isinstance(data, _VocabState) and isinstance(data.stoi, dict): return len(data.stoi) - if hasattr(data, "__len__"): + if isinstance(data, (dict, list, tuple)): return len(data) raise ValueError(f"could not count entries in {vocab_path}") +class _VocabState: + """Data-only stand-in: counting tokens does not require tokenizer methods.""" + + +class _VocabUnpickler(pickle.Unpickler): + def find_class(self, module: str, name: str) -> Any: + if module in {"kermt.data.torchvocab", "grover.data.torchvocab"} and name in { + "TorchVocab", "MolVocab", "SMILESVocab", + }: + return _VocabState + if (module, name) == ("collections", "Counter"): + return Counter + if (module, name) == ("re", "_compile"): + return re.compile + raise pickle.UnpicklingError(f"unsupported vocabulary object: {module}.{name}") + + +def load_checkpoint(path: Path | str) -> dict[str, Any]: + """Read KERMT tensors and known metadata with PyTorch's restricted loader. + + Saved arguments use argparse.Namespace; finetuned checkpoints also contain + numeric NumPy scaler arrays. Explicit globals cover those formats, including + NumPy 1/2 module names, without accepting artifact-selected imports. + """ + import numpy as np + import torch + + multiarray = np._core.multiarray if hasattr(np, "_core") else np.core.multiarray + allowed = [argparse.Namespace, np.ndarray, np.dtype] + for module in ("numpy.core.multiarray", "numpy._core.multiarray"): + allowed.extend([ + (multiarray._reconstruct, f"{module}._reconstruct"), + (multiarray.scalar, f"{module}.scalar"), + ]) + allowed.extend(type(np.dtype(name)) for name in ( + "bool", "int8", "int16", "int32", "int64", "uint8", "uint16", "uint32", "uint64", + "float16", "float32", "float64", + )) + with torch.serialization.safe_globals(allowed): + return torch.load(path, map_location="cpu", weights_only=True) + + +def runner_environment(repo: Path, *, wandb: bool = False) -> dict[str, str]: + """Forward named runtime settings, keeping unrelated credentials out of jobs. + + W&B credentials/settings are included only for an explicitly enabled W&B + run. Hugging Face authentication belongs to the separate download helper. + """ + names = ( + "PATH", "HOME", "TMPDIR", "TEMP", "TMP", "LANG", "LC_ALL", "LC_CTYPE", "TZ", + "LD_LIBRARY_PATH", "LIBRARY_PATH", "CUDA_HOME", "CUDA_PATH", "PYTHONPATH", + "PYTHONDONTWRITEBYTECODE", "PYTHONUNBUFFERED", "PYTHONWARNINGS", + "CUDA_VISIBLE_DEVICES", "CUDA_DEVICE_ORDER", "CUDA_LAUNCH_BLOCKING", "NVIDIA_VISIBLE_DEVICES", + "NVIDIA_DRIVER_CAPABILITIES", "CUBLAS_WORKSPACE_CONFIG", "OMP_NUM_THREADS", + "MKL_NUM_THREADS", "OPENBLAS_NUM_THREADS", "NUMEXPR_NUM_THREADS", + "PYTORCH_CUDA_ALLOC_CONF", "PYTORCH_ALLOC_CONF", "PYTORCH_NO_CUDA_MEMORY_CACHING", + "TORCH_CPP_LOG_LEVEL", "TORCH_DISTRIBUTED_DEBUG", "NCCL_DEBUG", "NCCL_SOCKET_IFNAME", + "NCCL_IB_DISABLE", "NCCL_P2P_DISABLE", "NCCL_SHM_DISABLE", "GLOO_SOCKET_IFNAME", + "MASTER_ADDR", "MASTER_PORT", + "KERMT_REPO", "KERMT_REPO_COMMIT", "KERMT_REPO_DIRTY", + "SSL_CERT_FILE", "REQUESTS_CA_BUNDLE", + ) + if wandb: + names += ( + "WANDB_API_KEY", "WANDB_BASE_URL", "WANDB_MODE", "WANDB_DIR", "WANDB_ENTITY", + "WANDB_PROJECT", "WANDB_RUN_ID", "WANDB_RESUME", "WANDB_CACHE_DIR", + "WANDB_CONFIG_DIR", "WANDB_DATA_DIR", "WANDB_DISABLED", + ) + env = {name: value for name in names if (value := os.environ.get(name)) is not None} + env["PYTHONPATH"] = os.pathsep.join(filter(None, (str(repo), env.get("PYTHONPATH")))) + return env + + def validate_vocab_file(vocab_path: Path, *, kind: str) -> None: """Verify a user-provided vocab file is loadable BEFORE copying it into a run directory. Raises ValueError on failure with a clear, user-facing message. diff --git a/skills/kermt-finetune/scripts/check_checkpoint.py b/skills/kermt-finetune/scripts/check_checkpoint.py index fd488e2..5df7627 100644 --- a/skills/kermt-finetune/scripts/check_checkpoint.py +++ b/skills/kermt-finetune/scripts/check_checkpoint.py @@ -75,10 +75,15 @@ import sys import traceback from argparse import Namespace +from pathlib import Path from typing import Any import torch +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import load_checkpoint # noqa: E402 + # --------------------------------------------------------------------------- # State-dict key prefix conventions (kermt/model/models.py). @@ -291,7 +296,7 @@ def _arch_from_shapes(state_dict: dict[str, Any], arch: dict[str, Any]) -> tuple ] if candidates: arch["latent_dim"] = int(state_dict[candidates[0]].shape[0]) - # Absent latent_dist entirely is a fact about the model, not a problem — no warning. + # Encoder-only models have no latent distribution; latent_dim remains None. # depth, num_attn_head, activation, backbone, embedding_output_type, self_attention # are not robustly inferable from shapes alone; report a warning for each that's @@ -390,7 +395,7 @@ def validate(mode: str, ckpt_path: str) -> dict[str, Any]: # 1. Load the checkpoint. try: - ckpt = torch.load(ckpt_path, map_location="cpu", weights_only=False) + ckpt = load_checkpoint(ckpt_path) except FileNotFoundError: result["errors"].append(f"checkpoint not found: {ckpt_path}") return result diff --git a/skills/kermt-finetune/scripts/run_finetune_local.py b/skills/kermt-finetune/scripts/run_finetune_local.py index 167d8a7..36037c3 100644 --- a/skills/kermt-finetune/scripts/run_finetune_local.py +++ b/skills/kermt-finetune/scripts/run_finetune_local.py @@ -62,7 +62,7 @@ from _utils import ( # noqa: E402 resolve_kermt_repo, assert_prepare_manifest_basics, docker_image_digest, format_cmd_replay, git_commit_with_env_override, load_json, merge_default_into_applied, - resolve_single_gpu, run_checkpoint_validator, + resolve_single_gpu, run_checkpoint_validator, runner_environment, ) @@ -383,7 +383,7 @@ def run(args: argparse.Namespace) -> dict[str, Any]: run_manifest["status"] = "dry_run" return run_manifest - env = os.environ.copy() + env = runner_environment(REPO_ROOT) if distributed: # DDP: main.py finetune reads WORLD_SIZE and spawns one process per GPU. # Do not pin CUDA_VISIBLE_DEVICES to a single device. diff --git a/skills/kermt-infer/SKILL.md b/skills/kermt-infer/SKILL.md index 92f3d9d..d14715f 100644 --- a/skills/kermt-infer/SKILL.md +++ b/skills/kermt-infer/SKILL.md @@ -19,7 +19,7 @@ launch the runner blocking, return the predictions CSV. ## Skill and runtime paths -Set `SKILL_DIR` to the directory containing this `SKILL.md`. Export +Set `SKILL_DIR` to the absolute path of this installed skill directory. Export `KERMT_REPO` as the absolute path to the KERMT checkout used for model execution. The bundled container helper mounts that checkout at `/workspace` and this skill at `/skill` (read-only). Commands inside diff --git a/skills/kermt-infer/scripts/_utils.py b/skills/kermt-infer/scripts/_utils.py index 5bde460..7fe75b0 100644 --- a/skills/kermt-infer/scripts/_utils.py +++ b/skills/kermt-infer/scripts/_utils.py @@ -10,8 +10,11 @@ from __future__ import annotations import argparse +from collections import Counter import json import os +import pickle +import re import shlex import subprocess import sys @@ -72,11 +75,11 @@ def count_vocab_entries(vocab_path: Path) -> int: Handles three layouts: - JSON with `{stoi: {token: idx}, ...}` (MolVocab.save_vocab default) - JSON as a raw `{token: idx}` dict (legacy / hand-edited) - - Pickle of a MolVocab / SMILESVocab object (the smiles vocab is always - pickled because its compiled-regex tokenizer state isn't - JSON-serializable). Falls through to raw `pickle.load` if the - MolVocab / SMILESVocab loader can't import or fails to recognize - the contents (e.g. test fixtures with plain dicts). + - Legacy MolVocab / SMILESVocab pickles, read as inert vocabulary state. + + The pickle reader accepts only the known vocabulary containers and their + Counter/regex metadata. It cannot import arbitrary classes or run reducers + supplied by the artifact, and it never falls back to an unrestricted loader. """ if vocab_path.suffix == ".json": data = json.loads(vocab_path.read_text()) @@ -86,28 +89,88 @@ def count_vocab_entries(vocab_path: Path) -> int: return len(data) raise ValueError(f"unsupported JSON vocab shape at {vocab_path}: {type(data).__name__}") - # .pkl: try MolVocab / SMILESVocab first, then fall back to raw pickle. - try: - from kermt.data.torchvocab import MolVocab, SMILESVocab # type: ignore - for loader in (MolVocab.load_vocab, SMILESVocab.load_vocab): - try: - v = loader(str(vocab_path)) - return len(v) - except Exception: - continue - except ImportError: - pass - - import pickle with vocab_path.open("rb") as f: - data = pickle.load(f) - if hasattr(data, "stoi"): + data = _VocabUnpickler(f).load() + if isinstance(data, _VocabState) and isinstance(data.stoi, dict): return len(data.stoi) - if hasattr(data, "__len__"): + if isinstance(data, (dict, list, tuple)): return len(data) raise ValueError(f"could not count entries in {vocab_path}") +class _VocabState: + """Data-only stand-in: counting tokens does not require tokenizer methods.""" + + +class _VocabUnpickler(pickle.Unpickler): + def find_class(self, module: str, name: str) -> Any: + if module in {"kermt.data.torchvocab", "grover.data.torchvocab"} and name in { + "TorchVocab", "MolVocab", "SMILESVocab", + }: + return _VocabState + if (module, name) == ("collections", "Counter"): + return Counter + if (module, name) == ("re", "_compile"): + return re.compile + raise pickle.UnpicklingError(f"unsupported vocabulary object: {module}.{name}") + + +def load_checkpoint(path: Path | str) -> dict[str, Any]: + """Read KERMT tensors and known metadata with PyTorch's restricted loader. + + Saved arguments use argparse.Namespace; finetuned checkpoints also contain + numeric NumPy scaler arrays. Explicit globals cover those formats, including + NumPy 1/2 module names, without accepting artifact-selected imports. + """ + import numpy as np + import torch + + multiarray = np._core.multiarray if hasattr(np, "_core") else np.core.multiarray + allowed = [argparse.Namespace, np.ndarray, np.dtype] + for module in ("numpy.core.multiarray", "numpy._core.multiarray"): + allowed.extend([ + (multiarray._reconstruct, f"{module}._reconstruct"), + (multiarray.scalar, f"{module}.scalar"), + ]) + allowed.extend(type(np.dtype(name)) for name in ( + "bool", "int8", "int16", "int32", "int64", "uint8", "uint16", "uint32", "uint64", + "float16", "float32", "float64", + )) + with torch.serialization.safe_globals(allowed): + return torch.load(path, map_location="cpu", weights_only=True) + + +def runner_environment(repo: Path, *, wandb: bool = False) -> dict[str, str]: + """Forward named runtime settings, keeping unrelated credentials out of jobs. + + W&B credentials/settings are included only for an explicitly enabled W&B + run. Hugging Face authentication belongs to the separate download helper. + """ + names = ( + "PATH", "HOME", "TMPDIR", "TEMP", "TMP", "LANG", "LC_ALL", "LC_CTYPE", "TZ", + "LD_LIBRARY_PATH", "LIBRARY_PATH", "CUDA_HOME", "CUDA_PATH", "PYTHONPATH", + "PYTHONDONTWRITEBYTECODE", "PYTHONUNBUFFERED", "PYTHONWARNINGS", + "CUDA_VISIBLE_DEVICES", "CUDA_DEVICE_ORDER", "CUDA_LAUNCH_BLOCKING", "NVIDIA_VISIBLE_DEVICES", + "NVIDIA_DRIVER_CAPABILITIES", "CUBLAS_WORKSPACE_CONFIG", "OMP_NUM_THREADS", + "MKL_NUM_THREADS", "OPENBLAS_NUM_THREADS", "NUMEXPR_NUM_THREADS", + "PYTORCH_CUDA_ALLOC_CONF", "PYTORCH_ALLOC_CONF", "PYTORCH_NO_CUDA_MEMORY_CACHING", + "TORCH_CPP_LOG_LEVEL", "TORCH_DISTRIBUTED_DEBUG", "NCCL_DEBUG", "NCCL_SOCKET_IFNAME", + "NCCL_IB_DISABLE", "NCCL_P2P_DISABLE", "NCCL_SHM_DISABLE", "GLOO_SOCKET_IFNAME", + "MASTER_ADDR", "MASTER_PORT", + "KERMT_REPO", "KERMT_REPO_COMMIT", "KERMT_REPO_DIRTY", + "SSL_CERT_FILE", "REQUESTS_CA_BUNDLE", + ) + if wandb: + names += ( + "WANDB_API_KEY", "WANDB_BASE_URL", "WANDB_MODE", "WANDB_DIR", "WANDB_ENTITY", + "WANDB_PROJECT", "WANDB_RUN_ID", "WANDB_RESUME", "WANDB_CACHE_DIR", + "WANDB_CONFIG_DIR", "WANDB_DATA_DIR", "WANDB_DISABLED", + ) + env = {name: value for name in names if (value := os.environ.get(name)) is not None} + env["PYTHONPATH"] = os.pathsep.join(filter(None, (str(repo), env.get("PYTHONPATH")))) + return env + + def validate_vocab_file(vocab_path: Path, *, kind: str) -> None: """Verify a user-provided vocab file is loadable BEFORE copying it into a run directory. Raises ValueError on failure with a clear, user-facing message. diff --git a/skills/kermt-infer/scripts/check_checkpoint.py b/skills/kermt-infer/scripts/check_checkpoint.py index fd488e2..5df7627 100644 --- a/skills/kermt-infer/scripts/check_checkpoint.py +++ b/skills/kermt-infer/scripts/check_checkpoint.py @@ -75,10 +75,15 @@ import sys import traceback from argparse import Namespace +from pathlib import Path from typing import Any import torch +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import load_checkpoint # noqa: E402 + # --------------------------------------------------------------------------- # State-dict key prefix conventions (kermt/model/models.py). @@ -291,7 +296,7 @@ def _arch_from_shapes(state_dict: dict[str, Any], arch: dict[str, Any]) -> tuple ] if candidates: arch["latent_dim"] = int(state_dict[candidates[0]].shape[0]) - # Absent latent_dist entirely is a fact about the model, not a problem — no warning. + # Encoder-only models have no latent distribution; latent_dim remains None. # depth, num_attn_head, activation, backbone, embedding_output_type, self_attention # are not robustly inferable from shapes alone; report a warning for each that's @@ -390,7 +395,7 @@ def validate(mode: str, ckpt_path: str) -> dict[str, Any]: # 1. Load the checkpoint. try: - ckpt = torch.load(ckpt_path, map_location="cpu", weights_only=False) + ckpt = load_checkpoint(ckpt_path) except FileNotFoundError: result["errors"].append(f"checkpoint not found: {ckpt_path}") return result diff --git a/skills/kermt-infer/scripts/run_inference.py b/skills/kermt-infer/scripts/run_inference.py index 56575bd..35c4fcf 100644 --- a/skills/kermt-infer/scripts/run_inference.py +++ b/skills/kermt-infer/scripts/run_inference.py @@ -48,7 +48,7 @@ from _utils import ( # noqa: E402 resolve_kermt_repo, assert_prepare_manifest_basics, docker_image_digest, format_cmd_replay, git_commit_with_env_override, load_json, merge_default_into_applied, - resolve_single_gpu, run_checkpoint_validator, + resolve_single_gpu, run_checkpoint_validator, runner_environment, ) @@ -206,7 +206,7 @@ def run(args: argparse.Namespace) -> dict[str, Any]: return run_manifest # 6. Execute. - env = os.environ.copy() + env = runner_environment(REPO_ROOT) env["CUDA_VISIBLE_DEVICES"] = str(gpu) # main.py enables strict deterministic algorithms via # `torch.use_deterministic_algorithms(True)` (kermt/main.py:23); CuBLAS diff --git a/skills/kermt-pretrain-scratch/SKILL.md b/skills/kermt-pretrain-scratch/SKILL.md index 7da45cd..c148336 100644 --- a/skills/kermt-pretrain-scratch/SKILL.md +++ b/skills/kermt-pretrain-scratch/SKILL.md @@ -22,7 +22,7 @@ from scratch over many epochs. ## Skill and runtime paths -Set `SKILL_DIR` to the directory containing this `SKILL.md`. Export +Set `SKILL_DIR` to the absolute path of this installed skill directory. Export `KERMT_REPO` as the absolute path to the KERMT checkout used for model execution. The bundled container helper mounts that checkout at `/workspace` and this skill at `/skill` (read-only). Commands inside diff --git a/skills/kermt-pretrain-scratch/scripts/_utils.py b/skills/kermt-pretrain-scratch/scripts/_utils.py index 5bde460..7fe75b0 100644 --- a/skills/kermt-pretrain-scratch/scripts/_utils.py +++ b/skills/kermt-pretrain-scratch/scripts/_utils.py @@ -10,8 +10,11 @@ from __future__ import annotations import argparse +from collections import Counter import json import os +import pickle +import re import shlex import subprocess import sys @@ -72,11 +75,11 @@ def count_vocab_entries(vocab_path: Path) -> int: Handles three layouts: - JSON with `{stoi: {token: idx}, ...}` (MolVocab.save_vocab default) - JSON as a raw `{token: idx}` dict (legacy / hand-edited) - - Pickle of a MolVocab / SMILESVocab object (the smiles vocab is always - pickled because its compiled-regex tokenizer state isn't - JSON-serializable). Falls through to raw `pickle.load` if the - MolVocab / SMILESVocab loader can't import or fails to recognize - the contents (e.g. test fixtures with plain dicts). + - Legacy MolVocab / SMILESVocab pickles, read as inert vocabulary state. + + The pickle reader accepts only the known vocabulary containers and their + Counter/regex metadata. It cannot import arbitrary classes or run reducers + supplied by the artifact, and it never falls back to an unrestricted loader. """ if vocab_path.suffix == ".json": data = json.loads(vocab_path.read_text()) @@ -86,28 +89,88 @@ def count_vocab_entries(vocab_path: Path) -> int: return len(data) raise ValueError(f"unsupported JSON vocab shape at {vocab_path}: {type(data).__name__}") - # .pkl: try MolVocab / SMILESVocab first, then fall back to raw pickle. - try: - from kermt.data.torchvocab import MolVocab, SMILESVocab # type: ignore - for loader in (MolVocab.load_vocab, SMILESVocab.load_vocab): - try: - v = loader(str(vocab_path)) - return len(v) - except Exception: - continue - except ImportError: - pass - - import pickle with vocab_path.open("rb") as f: - data = pickle.load(f) - if hasattr(data, "stoi"): + data = _VocabUnpickler(f).load() + if isinstance(data, _VocabState) and isinstance(data.stoi, dict): return len(data.stoi) - if hasattr(data, "__len__"): + if isinstance(data, (dict, list, tuple)): return len(data) raise ValueError(f"could not count entries in {vocab_path}") +class _VocabState: + """Data-only stand-in: counting tokens does not require tokenizer methods.""" + + +class _VocabUnpickler(pickle.Unpickler): + def find_class(self, module: str, name: str) -> Any: + if module in {"kermt.data.torchvocab", "grover.data.torchvocab"} and name in { + "TorchVocab", "MolVocab", "SMILESVocab", + }: + return _VocabState + if (module, name) == ("collections", "Counter"): + return Counter + if (module, name) == ("re", "_compile"): + return re.compile + raise pickle.UnpicklingError(f"unsupported vocabulary object: {module}.{name}") + + +def load_checkpoint(path: Path | str) -> dict[str, Any]: + """Read KERMT tensors and known metadata with PyTorch's restricted loader. + + Saved arguments use argparse.Namespace; finetuned checkpoints also contain + numeric NumPy scaler arrays. Explicit globals cover those formats, including + NumPy 1/2 module names, without accepting artifact-selected imports. + """ + import numpy as np + import torch + + multiarray = np._core.multiarray if hasattr(np, "_core") else np.core.multiarray + allowed = [argparse.Namespace, np.ndarray, np.dtype] + for module in ("numpy.core.multiarray", "numpy._core.multiarray"): + allowed.extend([ + (multiarray._reconstruct, f"{module}._reconstruct"), + (multiarray.scalar, f"{module}.scalar"), + ]) + allowed.extend(type(np.dtype(name)) for name in ( + "bool", "int8", "int16", "int32", "int64", "uint8", "uint16", "uint32", "uint64", + "float16", "float32", "float64", + )) + with torch.serialization.safe_globals(allowed): + return torch.load(path, map_location="cpu", weights_only=True) + + +def runner_environment(repo: Path, *, wandb: bool = False) -> dict[str, str]: + """Forward named runtime settings, keeping unrelated credentials out of jobs. + + W&B credentials/settings are included only for an explicitly enabled W&B + run. Hugging Face authentication belongs to the separate download helper. + """ + names = ( + "PATH", "HOME", "TMPDIR", "TEMP", "TMP", "LANG", "LC_ALL", "LC_CTYPE", "TZ", + "LD_LIBRARY_PATH", "LIBRARY_PATH", "CUDA_HOME", "CUDA_PATH", "PYTHONPATH", + "PYTHONDONTWRITEBYTECODE", "PYTHONUNBUFFERED", "PYTHONWARNINGS", + "CUDA_VISIBLE_DEVICES", "CUDA_DEVICE_ORDER", "CUDA_LAUNCH_BLOCKING", "NVIDIA_VISIBLE_DEVICES", + "NVIDIA_DRIVER_CAPABILITIES", "CUBLAS_WORKSPACE_CONFIG", "OMP_NUM_THREADS", + "MKL_NUM_THREADS", "OPENBLAS_NUM_THREADS", "NUMEXPR_NUM_THREADS", + "PYTORCH_CUDA_ALLOC_CONF", "PYTORCH_ALLOC_CONF", "PYTORCH_NO_CUDA_MEMORY_CACHING", + "TORCH_CPP_LOG_LEVEL", "TORCH_DISTRIBUTED_DEBUG", "NCCL_DEBUG", "NCCL_SOCKET_IFNAME", + "NCCL_IB_DISABLE", "NCCL_P2P_DISABLE", "NCCL_SHM_DISABLE", "GLOO_SOCKET_IFNAME", + "MASTER_ADDR", "MASTER_PORT", + "KERMT_REPO", "KERMT_REPO_COMMIT", "KERMT_REPO_DIRTY", + "SSL_CERT_FILE", "REQUESTS_CA_BUNDLE", + ) + if wandb: + names += ( + "WANDB_API_KEY", "WANDB_BASE_URL", "WANDB_MODE", "WANDB_DIR", "WANDB_ENTITY", + "WANDB_PROJECT", "WANDB_RUN_ID", "WANDB_RESUME", "WANDB_CACHE_DIR", + "WANDB_CONFIG_DIR", "WANDB_DATA_DIR", "WANDB_DISABLED", + ) + env = {name: value for name in names if (value := os.environ.get(name)) is not None} + env["PYTHONPATH"] = os.pathsep.join(filter(None, (str(repo), env.get("PYTHONPATH")))) + return env + + def validate_vocab_file(vocab_path: Path, *, kind: str) -> None: """Verify a user-provided vocab file is loadable BEFORE copying it into a run directory. Raises ValueError on failure with a clear, user-facing message. diff --git a/skills/kermt-pretrain-scratch/scripts/check_checkpoint.py b/skills/kermt-pretrain-scratch/scripts/check_checkpoint.py index fd488e2..5df7627 100644 --- a/skills/kermt-pretrain-scratch/scripts/check_checkpoint.py +++ b/skills/kermt-pretrain-scratch/scripts/check_checkpoint.py @@ -75,10 +75,15 @@ import sys import traceback from argparse import Namespace +from pathlib import Path from typing import Any import torch +if str(Path(__file__).resolve().parent) not in sys.path: + sys.path.insert(0, str(Path(__file__).resolve().parent)) +from _utils import load_checkpoint # noqa: E402 + # --------------------------------------------------------------------------- # State-dict key prefix conventions (kermt/model/models.py). @@ -291,7 +296,7 @@ def _arch_from_shapes(state_dict: dict[str, Any], arch: dict[str, Any]) -> tuple ] if candidates: arch["latent_dim"] = int(state_dict[candidates[0]].shape[0]) - # Absent latent_dist entirely is a fact about the model, not a problem — no warning. + # Encoder-only models have no latent distribution; latent_dim remains None. # depth, num_attn_head, activation, backbone, embedding_output_type, self_attention # are not robustly inferable from shapes alone; report a warning for each that's @@ -390,7 +395,7 @@ def validate(mode: str, ckpt_path: str) -> dict[str, Any]: # 1. Load the checkpoint. try: - ckpt = torch.load(ckpt_path, map_location="cpu", weights_only=False) + ckpt = load_checkpoint(ckpt_path) except FileNotFoundError: result["errors"].append(f"checkpoint not found: {ckpt_path}") return result diff --git a/skills/kermt-pretrain-scratch/scripts/run_pretrain_local.py b/skills/kermt-pretrain-scratch/scripts/run_pretrain_local.py index 5b8a8a1..6f74382 100644 --- a/skills/kermt-pretrain-scratch/scripts/run_pretrain_local.py +++ b/skills/kermt-pretrain-scratch/scripts/run_pretrain_local.py @@ -53,8 +53,8 @@ sys.path.insert(0, str(Path(__file__).resolve().parent)) from _utils import ( # noqa: E402 resolve_kermt_repo, assert_prepare_manifest_basics, count_vocab_entries, docker_image_digest, - format_cmd_replay, git_commit_with_env_override, load_json, - merge_default_into_applied, run_checkpoint_validator, + format_cmd_replay, git_commit_with_env_override, load_json, load_checkpoint, + merge_default_into_applied, run_checkpoint_validator, runner_environment, ) @@ -315,7 +315,7 @@ def _materialize_ckpt_for_fresh_schedule(user_ckpt: Path, save_dir: Path) -> Pat target = save_dir / "last_checkpoint.pt" if target.exists() or target.is_symlink(): target.unlink() - ckpt = torch.load(user_ckpt, map_location="cpu", weights_only=False) + ckpt = load_checkpoint(user_ckpt) if not isinstance(ckpt, dict) or "state_dict" not in ckpt: raise ValueError( f"ckpt {user_ckpt} is not in the expected save_model_for_restart " @@ -335,8 +335,7 @@ def _validate_resume_state(user_ckpt: Path) -> dict[str, Any]: (optimizer state, scheduler_step, epoch, batch_idx). Returns a small `resume_state` dict for the manifest so users can see what was restored. Raises ValueError with a clear redirect if the ckpt is too lean.""" - import torch - ckpt = torch.load(user_ckpt, map_location="cpu", weights_only=False) + ckpt = load_checkpoint(user_ckpt) if not isinstance(ckpt, dict) or "state_dict" not in ckpt: raise ValueError( f"ckpt {user_ckpt} is not in the expected save_model_for_restart " @@ -645,7 +644,7 @@ def run(args: argparse.Namespace) -> dict[str, Any]: run_manifest["status"] = "dry_run" return run_manifest - env = os.environ.copy() + env = runner_environment(REPO_ROOT, wandb="wandb_project" in applied) env["WORLD_SIZE"] = str(world_size) if gpus_str: env["CUDA_VISIBLE_DEVICES"] = gpus_str diff --git a/skills/kermt-setup/SKILL.md b/skills/kermt-setup/SKILL.md index 866eec7..4f54ece 100644 --- a/skills/kermt-setup/SKILL.md +++ b/skills/kermt-setup/SKILL.md @@ -20,7 +20,7 @@ after the Dockerfile or `environment.yml` changes) before invoking any other ## Skill and runtime paths -Set `SKILL_DIR` to the directory containing this `SKILL.md`. Export +Set `SKILL_DIR` to the absolute path of this installed skill directory. Export `KERMT_REPO` as the absolute path to the KERMT checkout used for model execution. The bundled container helper mounts that checkout at `/workspace` and this skill at `/skill` (read-only). Commands inside diff --git a/tests/skills/test_artifact_loading.py b/tests/skills/test_artifact_loading.py new file mode 100644 index 0000000..93a53ef --- /dev/null +++ b/tests/skills/test_artifact_loading.py @@ -0,0 +1,134 @@ +# SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +"""Checkpoint/vocabulary compatibility and artifact-boundary regression tests.""" +from __future__ import annotations + +from argparse import Namespace +from collections import Counter +import importlib.util +import os +from pathlib import Path +import pickle +import re +import shlex +import subprocess +import sys +from unittest.mock import patch + +import numpy as np +import pytest +import torch + + +REPO_ROOT = Path(__file__).resolve().parents[2] +SCRIPT = REPO_ROOT / "skills/_shared/scripts/_utils.py" +spec = importlib.util.spec_from_file_location("kermt_artifact_utils", SCRIPT) +utils = importlib.util.module_from_spec(spec) +spec.loader.exec_module(utils) + + +class _UntrustedPayload: + def __init__(self, marker: Path): + self.marker = marker + + def __reduce__(self): + # A regression to unrestricted loading would create only this test's + # marker, never execute any real workload or access credentials. + return os.system, ("touch " + shlex.quote(str(self.marker)),) + + +@pytest.mark.parametrize("dtype", [np.float32, np.float64, np.int64]) +def test_checkpoint_preserves_model_resume_state_and_numpy_scalers(tmp_path, dtype): + weights = torch.arange(4, dtype=torch.float32) + scaler = np.array([0, 1, 2], dtype=dtype) + path = tmp_path / "checkpoint.pt" + torch.save({ + "args": Namespace(hidden_size=800, depth=6), + "state_dict": {"kermt.encoder.weight": weights}, + "data_scaler": {"means": scaler, "stds": scaler + 1}, + "features_scaler": None, + "optimizer": {"state": {0: {"momentum_buffer": weights + 1}}, "param_groups": []}, + "scheduler_step": 17, + "epoch": 2, + "batch_idx": 4, + "wandb_run_id": "resume-fixture", + "best_score": dtype(1), + }, path) + + loaded = utils.load_checkpoint(path) + assert loaded["args"] == Namespace(hidden_size=800, depth=6) + torch.testing.assert_close(loaded["state_dict"]["kermt.encoder.weight"], weights) + np.testing.assert_array_equal(loaded["data_scaler"]["means"], scaler) + assert loaded["data_scaler"]["means"].dtype == scaler.dtype + torch.testing.assert_close(loaded["optimizer"]["state"][0]["momentum_buffer"], weights + 1) + assert (loaded["scheduler_step"], loaded["epoch"], loaded["batch_idx"]) == (17, 2, 4) + assert loaded["wandb_run_id"] == "resume-fixture" + assert loaded["best_score"] == dtype(1) + + +def test_checkpoint_rejects_executable_pickle_without_creating_marker(tmp_path): + marker = tmp_path / "executed" + path = tmp_path / "checkpoint.pt" + torch.save({"args": Namespace(), "state_dict": {}, "extra": _UntrustedPayload(marker)}, path) + with pytest.raises(pickle.UnpicklingError): + utils.load_checkpoint(path) + assert not marker.exists() + + +@pytest.mark.parametrize("vocab_class_name", ["MolVocab", "SMILESVocab"]) +def test_existing_vocabulary_objects_keep_their_token_counts(tmp_path, vocab_class_name): + from kermt.data import torchvocab + + vocab = object.__new__(getattr(torchvocab, vocab_class_name)) + vocab.stoi = {"": 0, "C": 1, "O": 2} + vocab.itos = list(vocab.stoi) + vocab.freqs = Counter({"C": 4, "O": 2}) + vocab.regex = re.compile(r"C|O") + path = tmp_path / "vocab.pkl" + path.write_bytes(pickle.dumps(vocab)) + assert utils.count_vocab_entries(path) == 3 + + +def test_legacy_grover_vocab_is_counted_without_importing_grover(tmp_path): + # Protocol-2 state from the original GROVER module name. The reader should + # use only its stored token mapping, without importing that runtime package. + data = (b"\x80\x02cgrover.data.torchvocab\nMolVocab\n)\x81}" + b"X\x04\x00\x00\x00stoi}X\x01\x00\x00\x00CK\x00ssb.") + path = tmp_path / "grover.pkl" + path.write_bytes(data) + assert utils.count_vocab_entries(path) == 1 + + +def test_vocabulary_rejects_executable_pickle_without_creating_marker(tmp_path): + marker = tmp_path / "executed" + path = tmp_path / "vocab.pkl" + path.write_bytes(pickle.dumps(_UntrustedPayload(marker))) + with pytest.raises(ValueError, match="unsupported vocabulary object"): + utils.validate_vocab_file(path, kind="smiles") + assert not marker.exists() + + +def test_job_environment_keeps_runtime_settings_and_excludes_unrelated_secrets(tmp_path): + source = { + "PATH": str(Path(sys.executable).parent), "CUDA_VISIBLE_DEVICES": "3,5", + "NCCL_DEBUG": "INFO", "OMP_NUM_THREADS": "1", "PYTHONPATH": str(tmp_path), + "HF_TOKEN": "test-hf-secret", "UNRELATED_SECRET": "test-unrelated-secret", + "NVIDIA_INFERENCE_KEY": "test-inference-secret", "WANDB_API_KEY": "test-wandb-secret", + } + with patch.dict(os.environ, source, clear=True): + env = utils.runner_environment(REPO_ROOT) + assert env["CUDA_VISIBLE_DEVICES"] == "3,5" + assert env["NCCL_DEBUG"] == "INFO" + assert env["PYTHONPATH"].split(os.pathsep) == [str(REPO_ROOT), str(tmp_path)] + for key in ("HF_TOKEN", "UNRELATED_SECRET", "NVIDIA_INFERENCE_KEY", "WANDB_API_KEY"): + assert key not in env + opted_in = utils.runner_environment(REPO_ROOT, wandb=True) + assert opted_in["WANDB_API_KEY"] == "test-wandb-secret" + assert "HF_TOKEN" not in opted_in + + proc = subprocess.run( + [sys.executable, "-c", "import os; assert os.environ['CUDA_VISIBLE_DEVICES'] == '3,5'; " + "assert os.environ['NCCL_DEBUG'] == 'INFO'; assert 'UNRELATED_SECRET' not in os.environ"], + env=env, capture_output=True, text=True, + ) + assert proc.returncode == 0, proc.stderr From fc688b7e436f46c7deab7b7583eef68c469c927f Mon Sep 17 00:00:00 2001 From: nvskills-svc-account Date: Tue, 15 Sep 2026 03:33:56 +0000 Subject: [PATCH 4/4] Attach NVSkills validation signatures Signed-off-by: nvskills-svc-account --- skills/kermt-add-cmim-pretrain/BENCHMARK.md | 127 ++++++++++++++++++ skills/kermt-add-cmim-pretrain/skill-card.md | 97 +++++++------- skills/kermt-add-cmim-pretrain/skill.oms.sig | 1 + skills/kermt-continue-pretrain/BENCHMARK.md | 129 +++++++++++++++++++ skills/kermt-continue-pretrain/skill-card.md | 96 +++++++------- skills/kermt-continue-pretrain/skill.oms.sig | 1 + skills/kermt-embed/BENCHMARK.md | 129 +++++++++++++++++++ skills/kermt-embed/skill-card.md | 88 +++++++------ skills/kermt-embed/skill.oms.sig | 1 + skills/kermt-finetune/BENCHMARK.md | 129 +++++++++++++++++++ skills/kermt-finetune/skill-card.md | 97 +++++++------- skills/kermt-finetune/skill.oms.sig | 1 + skills/kermt-infer/BENCHMARK.md | 127 ++++++++++++++++++ skills/kermt-infer/skill-card.md | 93 +++++++------ skills/kermt-infer/skill.oms.sig | 1 + skills/kermt-monitor/BENCHMARK.md | 127 ++++++++++++++++++ skills/kermt-monitor/skill-card.md | 86 +++++++------ skills/kermt-monitor/skill.oms.sig | 1 + skills/kermt-pretrain-scratch/BENCHMARK.md | 127 ++++++++++++++++++ skills/kermt-pretrain-scratch/skill-card.md | 95 +++++++------- skills/kermt-pretrain-scratch/skill.oms.sig | 1 + skills/kermt-setup/BENCHMARK.md | 127 ++++++++++++++++++ skills/kermt-setup/skill-card.md | 88 +++++++------ skills/kermt-setup/skill.oms.sig | 1 + 24 files changed, 1443 insertions(+), 327 deletions(-) create mode 100644 skills/kermt-add-cmim-pretrain/BENCHMARK.md create mode 100644 skills/kermt-add-cmim-pretrain/skill.oms.sig create mode 100644 skills/kermt-continue-pretrain/BENCHMARK.md create mode 100644 skills/kermt-continue-pretrain/skill.oms.sig create mode 100644 skills/kermt-embed/BENCHMARK.md create mode 100644 skills/kermt-embed/skill.oms.sig create mode 100644 skills/kermt-finetune/BENCHMARK.md create mode 100644 skills/kermt-finetune/skill.oms.sig create mode 100644 skills/kermt-infer/BENCHMARK.md create mode 100644 skills/kermt-infer/skill.oms.sig create mode 100644 skills/kermt-monitor/BENCHMARK.md create mode 100644 skills/kermt-monitor/skill.oms.sig create mode 100644 skills/kermt-pretrain-scratch/BENCHMARK.md create mode 100644 skills/kermt-pretrain-scratch/skill.oms.sig create mode 100644 skills/kermt-setup/BENCHMARK.md create mode 100644 skills/kermt-setup/skill.oms.sig diff --git a/skills/kermt-add-cmim-pretrain/BENCHMARK.md b/skills/kermt-add-cmim-pretrain/BENCHMARK.md new file mode 100644 index 0000000..503acc8 --- /dev/null +++ b/skills/kermt-add-cmim-pretrain/BENCHMARK.md @@ -0,0 +1,127 @@ +# Skill Benchmark: kermt-add-cmim-pretrain + +> ✅ **Overall verdict: PASS — Recommended for publication** + +## Publication Recommendation + +Recommended for publication based on the completed evaluation evidence in this report. + +## Evaluation Metadata + +- Skill: `kermt-add-cmim-pretrain` +- Evaluation date: 2026-09-14 +- Evaluator version: `1.5.6` +- Agents: Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`), Codex (`openai/openai/gpt-5.5`) +- Tasks: 4 evaluation tasks (3 positive, 1 negative) +- Dataset digest: `sha256:67190f2c3c6d387ac214662edede46001c6cf423a2cd1b21f2ba57a5584b1e42` (skill-evaluator-dataset-snapshot/1) +- Attempts per task: 3 +- Environment: `k8s-sandbox` +- Tier 2 evidence: required for publication +- Tier 3 evidence: required for publication + +Each task attempt ran in its own isolated sandbox pod. + +## What This Report Answers + +The three-tier evaluation checks whether the skill: + +- is safe to use; +- produces correct answers; +- is discovered and activated when needed; +- helps the agent complete the user's goal and expected workflow; and +- avoids wasted skill and tool usage. + +## Results at a Glance + +| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 87.8% — baseline ran, but no comparable score was available; uplift unavailable | 82.7% — baseline ran, but no comparable score was available; uplift unavailable | +| Security | 100.0% → 100.0% (±0.0 points) | 50.0% → 100.0% (+50.0 points) | +| Correctness | 13.3% → 100.0% (+86.7 points) | 91.4% → 85.0% (-6.4 points) | +| Discoverability | 99.3% — baseline ran, but no comparable score was available; uplift unavailable | 83.3% — baseline ran, but no comparable score was available; uplift unavailable | +| Effectiveness | 20.0% → 55.6% (+35.6 points) | 47.1% → 65.0% (+17.9 points) | +| Efficiency | 84.2% — baseline ran, but no comparable score was available; uplift unavailable | 80.2% — baseline ran, but no comparable score was available; uplift unavailable | + +**How to read this table:** baseline is the same task attempted without the target skill. Scores are rounded to one decimal; threshold-adjacent values use additional precision so their displayed band matches the verdict. Uplift is derived from those displayed scores and shown in percentage points. + +Example: `47.0% → 92.0% (+45.0 points)` means the skill-assisted run scored 92.0%, 45.0 percentage points above its 47.0% no-skill baseline. + +A partial dimension was calculated from only the available configured signals; review the detailed report before relying on it. + +## Token Usage + +Actual Tier 3 execution usage is reported for every observed agent/case pair and both conditions. + +| Agent | Dataset case | With skill | Without skill | Delta | Change | Coverage | +|---|---|---:|---:|---:|---:|---| +| claude-code | All cases | 1,843,834 | 1,470,837 | N/A | N/A | skill 4/4; base 9/9 | +| claude-code | kermt-add-cmim-pretrain-001 | 367,265 | 665,452 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-add-cmim-pretrain-002 | 279,378 | 344,760 | N/A | N/A | skill 1/1; base 2/2 | +| claude-code | kermt-add-cmim-pretrain-003 | 574,222 | 331,621 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-add-cmim-pretrain-004 | 622,969 | 129,004 | +493,965 | +382.91% | skill 1/1; base 1/1 | +| codex | All cases | 1,104,206 | 3,185,479 | N/A | N/A | skill 4/4; base 7/7 | +| codex | kermt-add-cmim-pretrain-001 | 123,118 | 1,434,461 | N/A | N/A | skill 1/1; base 3/3 | +| codex | kermt-add-cmim-pretrain-002 | 72,697 | 455,195 | N/A | N/A | skill 1/1; base 2/2 | +| codex | kermt-add-cmim-pretrain-003 | 655,766 | 1,189,781 | -534,015 | -44.88% | skill 1/1; base 1/1 | +| codex | kermt-add-cmim-pretrain-004 | 252,625 | 106,042 | +146,583 | +138.23% | skill 1/1; base 1/1 | +| ALL AGENTS | Dataset aggregate | 2,948,040 | 4,656,316 | N/A | N/A | skill 8/8; base 16/16 | + +Prompt tokens include cached reads, so total tokens are `prompt + completion` (cached is not added twice). The Efficiency score uses `(prompt - cached) + completion`. N/A means the relevant trajectory counters were not available; coverage is never estimated. + +## Tier Status + +| Tier | Purpose | Status | Evidence | +|---|---|---|---| +| Tier 1 | Static validation | **PASSED WITH OBSERVATIONS** | 11 validator(s); 50 finding(s) | +| Tier 2 | Semantic deduplication | **PASSED** | 2 validator(s); 0 finding(s) | +| Tier 3 | Live agent evaluation | **PASS** | 2 agent(s); 4 task(s) | + +## Findings and Observations + +
+Show detailed findings and successful checks + +- **MEDIUM** QUALITY/quality_correctness: No documented scripts in table format (`skills/kermt-add-cmim-pretrain/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: Instructions don't mention 'run_script' (`skills/kermt-add-cmim-pretrain/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.author' (`skills/kermt-add-cmim-pretrain/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.tags' (`skills/kermt-add-cmim-pretrain/SKILL.md`) +- **MEDIUM** SCHEMA/metadata_key_style: Metadata key 'risk_tier' is not kebab-case (`skills/kermt-add-cmim-pretrain/SKILL.md`) +- 45 additional finding(s) are available in the full evaluation artifacts. + +
+ +## Scoring Methodology + +
+Show dimension definitions, source signals, and thresholds + +| Dimension | Question | Scored signals | +|---|---|---| +| Security | Is it safe to use? | `security` (100%) | +| Correctness | Is the answer correct? | `accuracy` (100%) | +| Discoverability | Was the right skill loaded when needed? | `skill_execution` (100%) | +| Effectiveness | Did the skill help complete the task? | `goal_accuracy` (50%) + `behavior_check` (50%) | +| Efficiency | Did it avoid wasted tool calls and token usage? | `skill_efficiency` (50%) + `token_efficiency` (50%) | + +- Dimension bands: PASS at 50% or above; NEUTRAL from 40% to below 50%; FAIL below 40%. +- Overall Tier 3 lift: PASS at +5 points or more; FAIL at -10 points or less; values between those bands are NEUTRAL. +- Overall verdict: PASS only when every configured dimension passes for at least one supported agent. Lift is reported as diagnostic evidence and does not override this gate. +- The 50% attempt pass threshold is a separate per-task gate; it is not the dimension pass threshold. +- Effectiveness is the equal-weight mean of goal completion (`goal_accuracy`) and expected workflow adherence (`behavior_check`). +- Efficiency is 50% tool-call productivity (the backward-compatible `skill_efficiency` wire id) and 50% `token_efficiency`. Positive-case skill routing is scored under Discoverability, not Efficiency; a negative case without a routing target is N/A. N/A sources are omitted, remaining weights are renormalized, and the dimension is marked partial. + +Signals present in this run: + +- `security` (Security): unsafe operations, secret leakage, and unauthorized access. +- `skill_execution` (Skill Execution): whether the expected skill was selected, decoys were avoided, and the workflow executed. +- `skill_efficiency` (Tool Productivity): tool-call productivity (legacy wire id; routing is scored under Discoverability). +- `accuracy` (Accuracy): final-answer correctness against the reference answer. +- `goal_accuracy` (Goal Accuracy): whether the user's goal was achieved. +- `behavior_check` (Behavior Check): whether the expected workflow behavior was followed. +- `token_efficiency` (Token Efficiency): actual uncached prompt plus completion usage (50% of Efficiency). + +
+ +## Freshness + +Regenerate this benchmark when the skill, evaluation dataset, target agent/model, evaluator version, environment, or scoring policy changes. diff --git a/skills/kermt-add-cmim-pretrain/skill-card.md b/skills/kermt-add-cmim-pretrain/skill-card.md index 6b7061d..7bf5874 100644 --- a/skills/kermt-add-cmim-pretrain/skill-card.md +++ b/skills/kermt-add-cmim-pretrain/skill-card.md @@ -1,77 +1,84 @@ ## Description:
-Converts a grover_base checkpoint into a hybrid checkpoint by adding a randomly-initialized cMIM decoder and latent distribution, then continues pretraining on the user's corpus in hybrid (vocab + contrast) mode.
+Convert a grover_base checkpoint (encoder-only or encoder + vocab heads) into a hybrid checkpoint by adding a randomly-initialized cMIM decoder + latent_dist, then continue pretraining on the user's corpus as hybrid (vocab + contrast).
-This skill is ready for commercial/non-commercial use.
+This skill is for research and development only.
## Owner -NVIDIA (evax@nvidia.com)
+NVIDIA
### License/Terms of Use:
-Apache-2.0
- +Apache 2.0
## Use Case:
-ML research engineers who want the cMIM contrastive objective on top of an existing grover_base KERMT checkpoint without retraining from scratch. Functionally `kermt-continue-pretrain` with a one-time checkpoint-conversion step prepended.
- -### Requirements/Dependencies:
-Requires API Key or External Credential: Optional
-Credential Type(s): API key — `WANDB_API_KEY` for optional Weights & Biases run tracking
- -* `kermt-setup` completed (supplies the `kermt:latest` image)
-* Docker, NVIDIA Container Toolkit, CUDA-capable NVIDIA GPU (multi-GPU supported via DDP)
-* A grover_base checkpoint (encoder-only, or encoder + vocab heads)
-* A pretraining corpus CSV
- -Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
+Developers and researchers who want to extend an existing grover_base pretrained encoder with a contrastive Masked Image Modeling (cMIM) decoder and continue hybrid pretraining on a custom molecular corpus, without restarting training from scratch.
### Deployment Geography for Use:
Global
-## Known Risks and Mitigations:
-Risk: The newly added cMIM decoder and latent distribution are randomly initialized, so the converted checkpoint performs worse than its source until sufficient hybrid pretraining has run — a checkpoint taken too early is silently degraded.
-Mitigation: The conversion step is explicitly separated from the training step, and `kermt-monitor` exposes validation loss so users can confirm convergence before adopting the result.
- -Risk: Continued pretraining is a long-running, multi-GPU workload that a single agent instruction can start, potentially consuming days of GPU time.
-Mitigation: Runs launch detached with a run manifest; `kermt-monitor` provides progress visibility and the container identifiers needed to terminate early.
+## Requirements / Dependencies:
+**Requires API Key or External Credential:** [Not Specified]
+**Credential Type(s):** [None identified]
-Risk: Supplying a checkpoint that is not grover_base (e.g. already cmim or hybrid) would produce an invalid conversion.
-Mitigation: The skill validates the source checkpoint type before conversion.
- -Risk: Conversion and data preparation write new checkpoint and shard/vocab/feature artifacts that can overwrite prior output.
-Mitigation: Artifacts are written under an explicit run/output path; the source checkpoint is not modified in place.
+Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
-Risk: When Weights & Biases tracking is enabled, run metadata is transmitted to a third-party service.
-Mitigation: W&B tracking is optional and off unless the user supplies `WANDB_API_KEY`.
+## Known Risks and Mitigations:
+Risk: Review before execution as proposals could introduce incorrect or misleading guidance into skills.
+Mitigation: Review and scan skill before deployment.
## Reference(s):
-- [KERMT repository](https://github.com/NVIDIA-BioNeMo/KERMT)
-- `scripts/run_pretrain_local.py` — extended usage examples
-- Related skills: `kermt-setup`, `kermt-monitor`, `kermt-continue-pretrain`, `kermt-pretrain-scratch`
+- [KERMT paper — Multitask finetuning and acceleration of chemical pretrained models](https://arxiv.org/abs/2510.12719)
+- [GROVER paper — Self-Supervised Graph Transformer on Large-Scale Molecular Data](https://arxiv.org/abs/2007.02835)
+- [cuik-molmaker — GPU-accelerated molecular featurization](https://github.com/NVIDIA-Digital-Bio/cuik-molmaker)
+ ## Skill Output:
-**Output Type(s):** [Files, Analysis]
-**Output Format:** [Converted hybrid checkpoint; subsequent training checkpoints; shard/vocab/feature artifacts; training logs; `run.json` manifest; Markdown launch summary]
-**Output Parameters:** [1D — run identifier, container id, converted checkpoint path, output paths]
-**Other Properties Related to Output:** [Detached execution: the skill returns after launch, not after training completes.]
+**Output Type(s):** [Shell commands, Configuration instructions, Files]
+**Output Format:** [Markdown with inline bash code blocks]
+**Output Parameters:** [1D]
+**Other Properties Related to Output:** [None]
## Evaluation Agents Used:
-Target agents: `claude-code`, `codex`. NVSkills-Eval has not yet been run against this skill — see Evaluation Results.
+- Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`)
+- Codex (`openai/openai/gpt-5.5`)
+ + ## Evaluation Tasks:
-4 evaluation tasks defined in `evals/evals.json`, covering source-checkpoint validation, cMIM conversion, corpus preparation, and detached launch.
+4 evaluation tasks (3 positive, 1 negative) from the skill-evaluator dataset snapshot; 3 attempts per task in isolated k8s-sandbox pods.
## Evaluation Metrics Used:
-Planned NVSkills-Eval dimensions: Security, Correctness, Discoverability, Effectiveness, Efficiency.
+Reported benchmark dimensions:
+- Security: Checks for unsafe operations, secret leakage, and unauthorized access.
+- Correctness: Checks final-answer correctness against the reference answer.
+- Discoverability: Checks whether the expected skill was selected and the workflow executed.
+- Effectiveness: Checks whether the user's goal was achieved and the expected workflow behavior was followed.
+- Efficiency: Checks tool-call productivity and token efficiency.
+ +Underlying evaluation signals used in this run:
+- `security`: Unsafe operations, secret leakage, and unauthorized access.
+- `accuracy`: Final-answer correctness against the reference answer.
+- `skill_execution`: Whether the expected skill was selected, decoys were avoided, and the workflow executed.
+- `goal_accuracy`: Whether the user's goal was achieved.
+- `behavior_check`: Whether the expected workflow behavior was followed.
+- `skill_efficiency`: Tool-call productivity (routing scored under Discoverability).
+- `token_efficiency`: Actual uncached prompt plus completion token usage.
+ + ## Evaluation Results:
-Pending. NVSkills-Eval has not been run for this skill; results and a `BENCHMARK.md` will be published when the evaluation pipeline runs.
+| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 87.8% | 82.7% | +| Security | 100.0% → 100.0% (±0.0 points) | 50.0% → 100.0% (+50.0 points) | +| Correctness | 13.3% → 100.0% (+86.7 points) | 91.4% → 85.0% (-6.4 points) | +| Discoverability | 99.3% | 83.3% | +| Effectiveness | 20.0% → 55.6% (+35.6 points) | 47.1% → 65.0% (+17.9 points) | +| Efficiency | 84.2% | 80.2% | ## Skill Version(s):
-b1c082c (source: git SHA, committed 2026-07-17)
+77111e0 (source: git SHA, committed 2026-09-09)
## Ethical Considerations:
NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their internal team to ensure this skill meets requirements for the relevant industry and use case and addresses unforeseen product misuse.
-Models produced by this skill inherit the composition and biases of the user's pretraining corpus. Downstream predictions are research hypotheses and must not be used as the sole basis for clinical, safety, or regulatory decisions.
- (For Release on NVIDIA Platforms Only)
-Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://www.nvidia.com/en-us/support/submit-security-vulnerability/).
+Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://app.intigriti.com/programs/nvidia/nvidiavdp/detail).
diff --git a/skills/kermt-add-cmim-pretrain/skill.oms.sig b/skills/kermt-add-cmim-pretrain/skill.oms.sig new file mode 100644 index 0000000..cde9569 --- /dev/null +++ b/skills/kermt-add-cmim-pretrain/skill.oms.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAia2VybXQtYWRkLWNtaW0tcHJldHJhaW4iLAogICAgICAiZGlnZXN0IjogewogICAgICAgICJzaGEyNTYiOiAiMzdjZjg2MmE3ODYxMDlhMzlmYzMwZmM3YTY5MTA4MjQ2MzRmZGMwYjZmYWYzMjU3MDdjNmNlZTA5YTIyZTMwMCIKICAgICAgfQogICAgfQogIF0sCiAgInByZWRpY2F0ZVR5cGUiOiAiaHR0cHM6Ly9tb2RlbF9zaWduaW5nL3NpZ25hdHVyZS92MS4wIiwKICAicHJlZGljYXRlIjogewogICAgInJlc291cmNlcyI6IFsKICAgICAgewogICAgICAgICJuYW1lIjogIkJFTkNITUFSSy5tZCIsCiAgICAgICAgImRpZ2VzdCI6ICI3ODEyOTIwMWUyYWUwZWQ0NTU0Yzc3NTY2ZmM5NTgxNWRkMWIwNGJkYTdkNGM2NzM5ODI4NDVhZWM0MzJlMzVlIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogIlNLSUxMLm1kIiwKICAgICAgICAiZGlnZXN0IjogImVlZDc3M2YyOTU4NWM4YjhhZDkzODY4MzFkY2NlOTY1OWQ4NjRjNGM2OTAzMjBkODEyZDQzMjRlNTFmYWI2ZTIiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAiY29uZmlnL2RlZmF1bHRzX3ByZXRyYWluLmpzb24iLAogICAgICAgICJkaWdlc3QiOiAiZDRmYWE2N2E5NGM3OGUxMDI2Yjc4OTQ2MGE5ZWZlYjQ5M2IyNmY0N2U3MTg0MzNmMTYwYzQ0MjI0Y2RiODc1NSIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJldmFscy9ldmFscy5qc29uIiwKICAgICAgICAiZGlnZXN0IjogImRmZjAxMDkwOGRjMTZhNjZmODc5ZWQ2ODM2NThhYWIzMjkxNGQ3MGQzZjFlMjEyZDA5NWE0NTA2NzFkYTZhMmQiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAic2NyaXB0cy9fdXRpbHMucHkiLAogICAgICAgICJkaWdlc3QiOiAiMDI2MjIwYTIyNmQ4Zjg0MTg2NGYzZjhkNzRkZmM3Y2VlZTUyZDk4NWQ0NWE5ZTE0NmUyN2NjMTUyMzlmZjU0MiIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJzY3JpcHRzL2NoZWNrX2NoZWNrcG9pbnQucHkiLAogICAgICAgICJkaWdlc3QiOiAiMGJjNmNjODcyOWRiM2EzYTkzOWZjZTdhOTVlNWY2ZDE2NmFmNTUxN2ZiNjkzNDZiZGEyYzMwMTE0NWE4ZjUwMiIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJzY3JpcHRzL2NoZWNrX2RhdGEucHkiLAogICAgICAgICJkaWdlc3QiOiAiNjg5MjY4MTg3M2Q2NmU3YmU0MTJiNzczMGEzNDllNjQwOTA5YzcyYWZhZTIyMjg1NGY3YTAxYjY0YjJhYWI1NiIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJzY3JpcHRzL2tlcm10X2NvbnRhaW5lci5zaCIsCiAgICAgICAgImRpZ2VzdCI6ICJmY2RhYjU5NWMwZTg5ZGYxNTgxOWE3ZTI3NDFjMzFmNTEyZDk1ZTA4ZjY0NjFmODFkNTM1ZWNhM2NiYTQwMmI4IiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInNjcmlwdHMvcHJlcGFyZV9kYXRhLnB5IiwKICAgICAgICAiZGlnZXN0IjogIjE0MjYyODlhYjgzMzRmMWE5OWU5YmYxYTQxNzY3OTU4MjRmMjc1MTlmZjkxNDIxNmEzZjA3ZjkwZWVhYWYwM2IiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAic2NyaXB0cy9ydW5fcHJldHJhaW5fbG9jYWwucHkiLAogICAgICAgICJkaWdlc3QiOiAiM2M1NGQ1MzcxNTk0MjE3NWM0YTk4ZGNlMzM1ZTQ2MDE4Mzk0NWI3YmYwZjkzYzYxYjYwNzkwNzM3NDlmODQ2OSIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJzY3JpcHRzL3VwZ3JhZGVfdG9faHlicmlkLnB5IiwKICAgICAgICAiZGlnZXN0IjogImNjNjJjNjQyNGM5NjIwNjU0MTIyZDc0NDU3MDIyN2Y2NjVhZWYzMjY3YTdiZmVkNzFjNGEwY2M4MjE3NmQyZjIiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAic2tpbGwtY2FyZC5tZCIsCiAgICAgICAgImRpZ2VzdCI6ICIxNjliYjA3YTZhYWE4ZDRmNDlmMzQ0YTBmYjUwYzEwNWI2MDA1YjAzMmVhYjU0OTEzZjFiOGUyZDEzNDA3ZGEwIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfQogICAgXSwKICAgICJzZXJpYWxpemF0aW9uIjogewogICAgICAiaWdub3JlX3BhdGhzIjogWwogICAgICAgICIuZ2l0IiwKICAgICAgICAiLmdpdGh1YiIsCiAgICAgICAgIi5naXRhdHRyaWJ1dGVzIiwKICAgICAgICAiLmdpdGlnbm9yZSIKICAgICAgXSwKICAgICAgIm1ldGhvZCI6ICJmaWxlcyIsCiAgICAgICJoYXNoX3R5cGUiOiAic2hhMjU2IiwKICAgICAgImFsbG93X3N5bWxpbmtzIjogZmFsc2UKICAgIH0KICB9Cn0=","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGYCMQD4d6ZW81QrH2f/ClYtnzKoygEGTSiM+83m59/K3/5h8vS6rpnblJhE+lWVg9omK9ICMQCwRfAYdRCHaxqLYBXqMihqK+91dm3apXpetgq88FTanBBpsSYZCn3t6X71oDI9+vo=","keyid":""}]}} \ No newline at end of file diff --git a/skills/kermt-continue-pretrain/BENCHMARK.md b/skills/kermt-continue-pretrain/BENCHMARK.md new file mode 100644 index 0000000..629002d --- /dev/null +++ b/skills/kermt-continue-pretrain/BENCHMARK.md @@ -0,0 +1,129 @@ +# Skill Benchmark: kermt-continue-pretrain + +> ✅ **Overall verdict: PASS — Recommended for publication** + +## Publication Recommendation + +Recommended for publication based on the completed evaluation evidence in this report. + +## Evaluation Metadata + +- Skill: `kermt-continue-pretrain` +- Evaluation date: 2026-09-14 +- Evaluator version: `1.5.6` +- Agents: Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`), Codex (`openai/openai/gpt-5.5`) +- Tasks: 5 evaluation tasks (4 positive, 1 negative) +- Dataset digest: `sha256:f5e787fbd9dc575e68c455c1076c60078b1cf859aa0fd4001e50e261fb8a7e46` (skill-evaluator-dataset-snapshot/1) +- Attempts per task: 3 +- Environment: `k8s-sandbox` +- Tier 2 evidence: required for publication +- Tier 3 evidence: required for publication + +Each task attempt ran in its own isolated sandbox pod. + +## What This Report Answers + +The three-tier evaluation checks whether the skill: + +- is safe to use; +- produces correct answers; +- is discovered and activated when needed; +- helps the agent complete the user's goal and expected workflow; and +- avoids wasted skill and tool usage. + +## Results at a Glance + +| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 79.3% — baseline ran, but no comparable score was available; uplift unavailable | 80.7% — baseline ran, but no comparable score was available; uplift unavailable | +| Security | 100.0% → 100.0% (±0.0 points) | 72.7% → 100.0% (+27.3 points) | +| Correctness | 15.4% → 88.0% (+72.6 points) | 54.6% → 88.0% (+33.4 points) | +| Discoverability | 95.0% — baseline ran, but no comparable score was available; uplift unavailable | 82.5% — baseline ran, but no comparable score was available; uplift unavailable | +| Effectiveness | 17.3% → 34.0% (+16.7 points) | 23.0% → 48.0% (+25.0 points) | +| Efficiency | 79.5% — baseline ran, but no comparable score was available; uplift unavailable | 85.1% — baseline ran, but no comparable score was available; uplift unavailable | + +**How to read this table:** baseline is the same task attempted without the target skill. Scores are rounded to one decimal; threshold-adjacent values use additional precision so their displayed band matches the verdict. Uplift is derived from those displayed scores and shown in percentage points. + +Example: `47.0% → 92.0% (+45.0 points)` means the skill-assisted run scored 92.0%, 45.0 percentage points above its 47.0% no-skill baseline. + +A partial dimension was calculated from only the available configured signals; review the detailed report before relying on it. + +## Token Usage + +Actual Tier 3 execution usage is reported for every observed agent/case pair and both conditions. + +| Agent | Dataset case | With skill | Without skill | Delta | Change | Coverage | +|---|---|---:|---:|---:|---:|---| +| claude-code | All cases | 2,171,993 | 2,832,938 | N/A | N/A | skill 5/5; base 13/13 | +| claude-code | kermt-continue-pretrain-001 | 468,536 | 687,884 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-continue-pretrain-002 | 290,010 | 562,464 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-continue-pretrain-003 | 342,928 | 507,476 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-continue-pretrain-004 | 297,318 | 156,011 | +141,307 | +90.58% | skill 1/1; base 1/1 | +| claude-code | kermt-continue-pretrain-005 | 773,201 | 919,103 | N/A | N/A | skill 1/1; base 3/3 | +| codex | All cases | 985,644 | 4,952,146 | N/A | N/A | skill 5/5; base 11/11 | +| codex | kermt-continue-pretrain-001 | 166,629 | 910,236 | N/A | N/A | skill 1/1; base 3/3 | +| codex | kermt-continue-pretrain-002 | 201,417 | 914,089 | -712,672 | -77.97% | skill 1/1; base 1/1 | +| codex | kermt-continue-pretrain-003 | 402,389 | 340,519 | N/A | N/A | skill 1/1; base 3/3 | +| codex | kermt-continue-pretrain-004 | 90,334 | 85,573 | +4,761 | +5.56% | skill 1/1; base 1/1 | +| codex | kermt-continue-pretrain-005 | 124,875 | 2,701,729 | N/A | N/A | skill 1/1; base 3/3 | +| ALL AGENTS | Dataset aggregate | 3,157,637 | 7,785,084 | N/A | N/A | skill 10/10; base 24/24 | + +Prompt tokens include cached reads, so total tokens are `prompt + completion` (cached is not added twice). The Efficiency score uses `(prompt - cached) + completion`. N/A means the relevant trajectory counters were not available; coverage is never estimated. + +## Tier Status + +| Tier | Purpose | Status | Evidence | +|---|---|---|---| +| Tier 1 | Static validation | **PASSED WITH OBSERVATIONS** | 11 validator(s); 49 finding(s) | +| Tier 2 | Semantic deduplication | **PASSED** | 2 validator(s); 0 finding(s) | +| Tier 3 | Live agent evaluation | **PASS** | 2 agent(s); 5 task(s) | + +## Findings and Observations + +
+Show detailed findings and successful checks + +- **MEDIUM** QUALITY/quality_correctness: No documented scripts in table format (`skills/kermt-continue-pretrain/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: Instructions don't mention 'run_script' (`skills/kermt-continue-pretrain/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.author' (`skills/kermt-continue-pretrain/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.tags' (`skills/kermt-continue-pretrain/SKILL.md`) +- **MEDIUM** SCHEMA/metadata_key_style: Metadata key 'risk_tier' is not kebab-case (`skills/kermt-continue-pretrain/SKILL.md`) +- 44 additional finding(s) are available in the full evaluation artifacts. + +
+ +## Scoring Methodology + +
+Show dimension definitions, source signals, and thresholds + +| Dimension | Question | Scored signals | +|---|---|---| +| Security | Is it safe to use? | `security` (100%) | +| Correctness | Is the answer correct? | `accuracy` (100%) | +| Discoverability | Was the right skill loaded when needed? | `skill_execution` (100%) | +| Effectiveness | Did the skill help complete the task? | `goal_accuracy` (50%) + `behavior_check` (50%) | +| Efficiency | Did it avoid wasted tool calls and token usage? | `skill_efficiency` (50%) + `token_efficiency` (50%) | + +- Dimension bands: PASS at 50% or above; NEUTRAL from 40% to below 50%; FAIL below 40%. +- Overall Tier 3 lift: PASS at +5 points or more; FAIL at -10 points or less; values between those bands are NEUTRAL. +- Overall verdict: PASS only when every configured dimension passes for at least one supported agent. Lift is reported as diagnostic evidence and does not override this gate. +- The 50% attempt pass threshold is a separate per-task gate; it is not the dimension pass threshold. +- Effectiveness is the equal-weight mean of goal completion (`goal_accuracy`) and expected workflow adherence (`behavior_check`). +- Efficiency is 50% tool-call productivity (the backward-compatible `skill_efficiency` wire id) and 50% `token_efficiency`. Positive-case skill routing is scored under Discoverability, not Efficiency; a negative case without a routing target is N/A. N/A sources are omitted, remaining weights are renormalized, and the dimension is marked partial. + +Signals present in this run: + +- `security` (Security): unsafe operations, secret leakage, and unauthorized access. +- `skill_execution` (Skill Execution): whether the expected skill was selected, decoys were avoided, and the workflow executed. +- `skill_efficiency` (Tool Productivity): tool-call productivity (legacy wire id; routing is scored under Discoverability). +- `accuracy` (Accuracy): final-answer correctness against the reference answer. +- `goal_accuracy` (Goal Accuracy): whether the user's goal was achieved. +- `behavior_check` (Behavior Check): whether the expected workflow behavior was followed. +- `token_efficiency` (Token Efficiency): actual uncached prompt plus completion usage (50% of Efficiency). + +
+ +## Freshness + +Regenerate this benchmark when the skill, evaluation dataset, target agent/model, evaluator version, environment, or scoring policy changes. diff --git a/skills/kermt-continue-pretrain/skill-card.md b/skills/kermt-continue-pretrain/skill-card.md index 03a161b..2401ac4 100644 --- a/skills/kermt-continue-pretrain/skill-card.md +++ b/skills/kermt-continue-pretrain/skill-card.md @@ -1,77 +1,85 @@ ## Description:
-Continues pretraining from an existing KERMT checkpoint — validating the checkpoint and corpus, preparing shard/vocab/feature artifacts, and launching `pretrain_ddp.py` inside the KERMT container with `--pretrain_mode` auto-dispatched from the checkpoint type.
+Continue KERMT pretraining on a custom SMILES corpus with a grover_base, cmim, or hybrid checkpoint. Use a local checkpoint or optionally download a pinned Hugging Face model bundle using HF_TOKEN if configured. Run containerized training and write model bundles, prepared data, logs, and checkpoints to user-selected host directories.
This skill is ready for commercial/non-commercial use.
## Owner -NVIDIA (evax@nvidia.com)
+NVIDIA
### License/Terms of Use:
-Apache-2.0
- +Apache 2.0
## Use Case:
-ML engineers adapting a KERMT foundation model to a domain-specific molecular corpus — for example an in-house compound collection — before task finetuning, without discarding the representations already learned.
- -### Requirements/Dependencies:
-Requires API Key or External Credential: Optional
-Credential Type(s): API key — `WANDB_API_KEY` for optional Weights & Biases run tracking
- -* `kermt-setup` completed (supplies the `kermt:latest` image)
-* Docker, NVIDIA Container Toolkit, CUDA-capable NVIDIA GPU (multi-GPU supported via DDP)
-* An existing KERMT checkpoint (grover_base vocab-only, cmim, or hybrid)
-* A pretraining corpus CSV
- -Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
+Developers and engineers continuing pretraining of KERMT molecular property prediction models on custom SMILES corpora, using containerized GPU-accelerated training with checkpoint management and distributed data parallel support.
### Deployment Geography for Use:
Global
-## Known Risks and Mitigations:
-Risk: Continued pretraining is a long-running, multi-GPU workload that a single agent instruction can start, potentially consuming days of GPU time and substantial cloud spend.
-Mitigation: Runs launch detached with a run manifest; `kermt-monitor` provides progress visibility and the container identifiers needed to terminate early.
- -Risk: Selecting the wrong `--pretrain_mode` for a checkpoint would train against the wrong objective and silently waste the entire run.
-Mitigation: The skill auto-dispatches `--pretrain_mode` from the detected checkpoint type rather than relying on the user to specify it.
+## Requirements / Dependencies:
+**Requires API Key or External Credential:** [Optional]
+**Credential Type(s):** [API key]
-Risk: Data preparation writes shard, vocabulary, and feature artifacts that can be large and can overwrite prior preparation output.
-Mitigation: Preparation writes under an explicit run/output path supplied by the user.
- -Risk: Continued pretraining on a narrow corpus can degrade general-purpose representations (catastrophic forgetting) in ways not visible until downstream finetuning.
-Mitigation: The original checkpoint is not modified in place; users should retain it and compare downstream task performance before adopting the continued-pretrain checkpoint.
+Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
-Risk: When Weights & Biases tracking is enabled, run metadata is transmitted to a third-party service.
-Mitigation: W&B tracking is optional and off unless the user supplies `WANDB_API_KEY`.
+## Known Risks and Mitigations:
+Risk: Review before execution as proposals could introduce incorrect or misleading guidance into skills.
+Mitigation: Review and scan skill before deployment.
## Reference(s):
-- [KERMT repository](https://github.com/NVIDIA-BioNeMo/KERMT)
-- `scripts/run_pretrain_local.py` — extended usage examples
-- Related skills: `kermt-setup`, `kermt-monitor`, `kermt-pretrain-scratch`, `kermt-add-cmim-pretrain`, `kermt-finetune`
+- [Released Models](references/released-models.md)
+- [Multitask finetuning and acceleration of chemical pretrained models (arXiv)](https://arxiv.org/abs/2510.12719)
+- [Self-Supervised Graph Transformer on Large-Scale Molecular Data (GROVER, arXiv)](https://arxiv.org/abs/2007.02835)
+- [NV-KERMT-70M-v2 on Hugging Face](https://huggingface.co/nvidia/NV-KERMT-70M-v2)
+ ## Skill Output:
-**Output Type(s):** [Files, Analysis]
-**Output Format:** [Model checkpoint files; shard/vocab/feature artifacts; training logs; `run.json` manifest; Markdown launch summary]
-**Output Parameters:** [1D — run identifier, container id, resolved pretrain mode, output paths]
-**Other Properties Related to Output:** [Detached execution: the skill returns after launch, not after training completes.]
+**Output Type(s):** [Shell commands, Configuration instructions, Files]
+**Output Format:** [Markdown with inline bash code blocks]
+**Output Parameters:** [1D]
+**Other Properties Related to Output:** [Produces run directory with model checkpoints, training logs, TensorBoard events, prepared data manifests, and a replayable run.json manifest]
## Evaluation Agents Used:
-Target agents: `claude-code`, `codex`. NVSkills-Eval has not yet been run against this skill — see Evaluation Results.
+- Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`)
+- Codex (`openai/openai/gpt-5.5`)
+ + ## Evaluation Tasks:
-5 evaluation tasks defined in `evals/evals.json`, covering checkpoint validation, corpus validation, data preparation, pretrain-mode dispatch, and detached launch.
+5 evaluation tasks (4 positive, 1 negative) with 3 attempts per task in isolated sandbox pods.
## Evaluation Metrics Used:
-Planned NVSkills-Eval dimensions: Security, Correctness, Discoverability, Effectiveness, Efficiency.
+Reported benchmark dimensions:
+- Security: Whether the skill avoids unsafe operations, secret leakage, and unauthorized access.
+- Correctness: Final-answer correctness against the reference answer.
+- Discoverability: Whether the expected skill was selected, decoys were avoided, and the workflow executed.
+- Effectiveness: Whether the skill helped complete the user's goal (50% goal completion + 50% expected workflow adherence).
+- Efficiency: Whether the skill avoided wasted tool calls and token usage (50% tool-call productivity + 50% token efficiency).
+ +Underlying evaluation signals used in this run:
+- `security`: Checks for unsafe operations, secret leakage, and unauthorized access.
+- `skill_execution`: Whether the expected skill was selected and the workflow executed.
+- `accuracy`: Final-answer correctness against the reference answer.
+- `goal_accuracy`: Whether the user's goal was achieved.
+- `behavior_check`: Whether the expected workflow behavior was followed.
+- `skill_efficiency`: Tool-call productivity (routing scored under Discoverability).
+- `token_efficiency`: Actual uncached prompt plus completion token usage.
+ + ## Evaluation Results:
-Pending. NVSkills-Eval has not been run for this skill; results and a `BENCHMARK.md` will be published when the evaluation pipeline runs.
+| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 79.3% | 80.7% | +| Security | 100.0% → 100.0% (±0.0 pts) | 72.7% → 100.0% (+27.3 pts) | +| Correctness | 15.4% → 88.0% (+72.6 pts) | 54.6% → 88.0% (+33.4 pts) | +| Discoverability | 95.0% | 82.5% | +| Effectiveness | 17.3% → 34.0% (+16.7 pts) | 23.0% → 48.0% (+25.0 pts) | +| Efficiency | 79.5% | 85.1% | ## Skill Version(s):
-b1c082c (source: git SHA, committed 2026-07-17)
+77111e0 (source: git SHA, committed 2026-09-09)
## Ethical Considerations:
NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their internal team to ensure this skill meets requirements for the relevant industry and use case and addresses unforeseen product misuse.
-Models produced by this skill inherit the composition and biases of the user's pretraining corpus. Downstream predictions are research hypotheses and must not be used as the sole basis for clinical, safety, or regulatory decisions.
- (For Release on NVIDIA Platforms Only)
-Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://www.nvidia.com/en-us/support/submit-security-vulnerability/).
+Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://app.intigriti.com/programs/nvidia/nvidiavdp/detail).
diff --git a/skills/kermt-continue-pretrain/skill.oms.sig b/skills/kermt-continue-pretrain/skill.oms.sig new file mode 100644 index 0000000..f85c5fb --- /dev/null +++ b/skills/kermt-continue-pretrain/skill.oms.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAia2VybXQtY29udGludWUtcHJldHJhaW4iLAogICAgICAiZGlnZXN0IjogewogICAgICAgICJzaGEyNTYiOiAiMzI0NDkwNzAxZjZmZmQ1YTY5NDM5OTdlYzJlZDdjNTlhMTNiYTFkYjk3NWE2NGQ0MjIzODIzZTc1NDQ4Njk3MiIKICAgICAgfQogICAgfQogIF0sCiAgInByZWRpY2F0ZVR5cGUiOiAiaHR0cHM6Ly9tb2RlbF9zaWduaW5nL3NpZ25hdHVyZS92MS4wIiwKICAicHJlZGljYXRlIjogewogICAgInJlc291cmNlcyI6IFsKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogIjJjOTBlYTAyMjhiMjIzOGVkZGVhODgyZTdkOWM0YWQ4OWUzMDRmN2E2ZDVmZjcyMmFhZWM5NTQ5YzNmYTk2NGUiLAogICAgICAgICJuYW1lIjogIkJFTkNITUFSSy5tZCIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogIjFlZDg2YTNkNjQzMWE0ZjA0M2VkNzI3NGZkMzVlOTkyMDllY2YzZmNjNGRhMWQ4ZjAzMWQ1NDdiMDA3YmYzZmMiLAogICAgICAgICJuYW1lIjogIlNLSUxMLm1kIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiZDRmYWE2N2E5NGM3OGUxMDI2Yjc4OTQ2MGE5ZWZlYjQ5M2IyNmY0N2U3MTg0MzNmMTYwYzQ0MjI0Y2RiODc1NSIsCiAgICAgICAgIm5hbWUiOiAiY29uZmlnL2RlZmF1bHRzX3ByZXRyYWluLmpzb24iCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICIyODU4MDQ0OTMyMWEzMWE2NzYyNDExN2Y1N2JjNWM0ZTU4ZTZiZjdiMTllMTkzODg1NGE1N2NlYjEzY2M2ZjMzIiwKICAgICAgICAibmFtZSI6ICJjb25maWcvcmVsZWFzZWRfbW9kZWwuanNvbiIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogIjVmZDZlMjg0ZDViNzVhY2JkNTJkNTdhYmFiMjg5MTgyZTA0ZmVjZTdhMjQ4ZjI1YWE1YzY0YTk5ODRiODMzMjMiLAogICAgICAgICJuYW1lIjogImV2YWxzL2V2YWxzLmpzb24iCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICI4NTY2ZjBhNzJmMTcyNjg1MDM3NmI4MDM1YWY3ZGYyMjYwM2RhOGIyMmQ4MTU5OWEyYWQ2MjM5YWYzNDMzMzU5IiwKICAgICAgICAibmFtZSI6ICJyZWZlcmVuY2VzL3JlbGVhc2VkLW1vZGVscy5tZCIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogIjAyNjIyMGEyMjZkOGY4NDE4NjRmM2Y4ZDc0ZGZjN2NlZWU1MmQ5ODVkNDVhOWUxNDZlMjdjYzE1MjM5ZmY1NDIiLAogICAgICAgICJuYW1lIjogInNjcmlwdHMvX3V0aWxzLnB5IgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiMGJjNmNjODcyOWRiM2EzYTkzOWZjZTdhOTVlNWY2ZDE2NmFmNTUxN2ZiNjkzNDZiZGEyYzMwMTE0NWE4ZjUwMiIsCiAgICAgICAgIm5hbWUiOiAic2NyaXB0cy9jaGVja19jaGVja3BvaW50LnB5IgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiNjg5MjY4MTg3M2Q2NmU3YmU0MTJiNzczMGEzNDllNjQwOTA5YzcyYWZhZTIyMjg1NGY3YTAxYjY0YjJhYWI1NiIsCiAgICAgICAgIm5hbWUiOiAic2NyaXB0cy9jaGVja19kYXRhLnB5IgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiNGQwZTY0ODUyOTY3N2UzNDA5NGY2MTgxMDVmMzkyZjQzYjIxZWU1YmNkMzNlMzhlYjhiMDA3Yzg0OTE4YWFhMyIsCiAgICAgICAgIm5hbWUiOiAic2NyaXB0cy9mZXRjaF9yZWxlYXNlZF9tb2RlbC5weSIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogImZjZGFiNTk1YzBlODlkZjE1ODE5YTdlMjc0MWMzMWY1MTJkOTVlMDhmNjQ2MWY4MWQ1MzVlY2EzY2JhNDAyYjgiLAogICAgICAgICJuYW1lIjogInNjcmlwdHMva2VybXRfY29udGFpbmVyLnNoIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiMTQyNjI4OWFiODMzNGYxYTk5ZTliZjFhNDE3Njc5NTgyNGYyNzUxOWZmOTE0MjE2YTNmMDdmOTBlZWFhZjAzYiIsCiAgICAgICAgIm5hbWUiOiAic2NyaXB0cy9wcmVwYXJlX2RhdGEucHkiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICIzYzU0ZDUzNzE1OTQyMTc1YzRhOThkY2UzMzVlNDYwMTgzOTQ1YjdiZjBmOTNjNjFiNjA3OTA3Mzc0OWY4NDY5IiwKICAgICAgICAibmFtZSI6ICJzY3JpcHRzL3J1bl9wcmV0cmFpbl9sb2NhbC5weSIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogImJjY2ZhYTRjOTEyNWQ4ZThkOTBjYTA0M2U0Y2RiMTc0NjhhZDRlZjdjMjU1MjA2MzNmZDFhZTc4NzRhMjNiYjIiLAogICAgICAgICJuYW1lIjogInNraWxsLWNhcmQubWQiCiAgICAgIH0KICAgIF0sCiAgICAic2VyaWFsaXphdGlvbiI6IHsKICAgICAgImFsbG93X3N5bWxpbmtzIjogZmFsc2UsCiAgICAgICJpZ25vcmVfcGF0aHMiOiBbCiAgICAgICAgIi5naXRodWIiLAogICAgICAgICIuZ2l0YXR0cmlidXRlcyIsCiAgICAgICAgIi5naXQiLAogICAgICAgICIuZ2l0aWdub3JlIgogICAgICBdLAogICAgICAibWV0aG9kIjogImZpbGVzIiwKICAgICAgImhhc2hfdHlwZSI6ICJzaGEyNTYiCiAgICB9CiAgfQp9","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGUCMHIdwCNPmiHi/7tBKxuhDffzWmrlcQsIMT6QJxOyjfertTOWeGGQIydycktuMIXwtwIxAOsQC6eTaOttQtMicBgCkvU2HVKdRGL8X+7Cu6tTR/eZ+g5d62m0Iy32Xwbze4msuA==","keyid":""}]}} \ No newline at end of file diff --git a/skills/kermt-embed/BENCHMARK.md b/skills/kermt-embed/BENCHMARK.md new file mode 100644 index 0000000..a22bd7e --- /dev/null +++ b/skills/kermt-embed/BENCHMARK.md @@ -0,0 +1,129 @@ +# Skill Benchmark: kermt-embed + +> ✅ **Overall verdict: PASS — Recommended for publication** + +## Publication Recommendation + +Recommended for publication based on the completed evaluation evidence in this report. + +## Evaluation Metadata + +- Skill: `kermt-embed` +- Evaluation date: 2026-09-15 +- Evaluator version: `1.5.6` +- Agents: Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`), Codex (`openai/openai/gpt-5.5`) +- Tasks: 5 evaluation tasks (4 positive, 1 negative) +- Dataset digest: `sha256:150b82ffdc6a29ff51ba6509f4ed70d6074e5944b957cdbd6c76f42ea048a482` (skill-evaluator-dataset-snapshot/1) +- Attempts per task: 3 +- Environment: `k8s-sandbox` +- Tier 2 evidence: required for publication +- Tier 3 evidence: required for publication + +Each task attempt ran in its own isolated sandbox pod. + +## What This Report Answers + +The three-tier evaluation checks whether the skill: + +- is safe to use; +- produces correct answers; +- is discovered and activated when needed; +- helps the agent complete the user's goal and expected workflow; and +- avoids wasted skill and tool usage. + +## Results at a Glance + +| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 82.0% — baseline ran, but no comparable score was available; uplift unavailable | 83.3% — baseline ran, but no comparable score was available; uplift unavailable | +| Security | 92.3% → 100.0% (+7.7 points) | 85.7% → 100.0% (+14.3 points) | +| Correctness | 24.6% → 84.0% (+59.4 points) | 80.0% → 84.0% (+4.0 points) | +| Discoverability | 93.8% — baseline ran, but no comparable score was available; uplift unavailable | 86.3% — baseline ran, but no comparable score was available; uplift unavailable | +| Effectiveness | 25.4% → 47.5% (+22.1 points) | 30.4% → 51.5% (+21.1 points) | +| Efficiency | 84.6% — baseline ran, but no comparable score was available; uplift unavailable | 94.9% — baseline ran, but no comparable score was available; uplift unavailable | + +**How to read this table:** baseline is the same task attempted without the target skill. Scores are rounded to one decimal; threshold-adjacent values use additional precision so their displayed band matches the verdict. Uplift is derived from those displayed scores and shown in percentage points. + +Example: `47.0% → 92.0% (+45.0 points)` means the skill-assisted run scored 92.0%, 45.0 percentage points above its 47.0% no-skill baseline. + +A partial dimension was calculated from only the available configured signals; review the detailed report before relying on it. + +## Token Usage + +Actual Tier 3 execution usage is reported for every observed agent/case pair and both conditions. + +| Agent | Dataset case | With skill | Without skill | Delta | Change | Coverage | +|---|---|---:|---:|---:|---:|---| +| claude-code | All cases | 2,189,079 | 2,993,659 | N/A | N/A | skill 5/5; base 13/13 | +| claude-code | kermt-embed-001 | 363,591 | 400,469 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-embed-002 | 285,085 | 256,226 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-embed-003 | 293,247 | 596,053 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-embed-004 | 832,919 | 248,102 | +584,817 | +235.72% | skill 1/1; base 1/1 | +| claude-code | kermt-embed-005 | 414,237 | 1,492,809 | N/A | N/A | skill 1/1; base 3/3 | +| codex | All cases | 426,747 | 1,368,222 | N/A | N/A | skill 5/5; base 7/7 | +| codex | kermt-embed-001 | 96,323 | 299,366 | N/A | N/A | skill 1/1; base 3/3 | +| codex | kermt-embed-002 | 30,053 | 193,251 | -163,198 | -84.45% | skill 1/1; base 1/1 | +| codex | kermt-embed-003 | 86,312 | 317,836 | -231,524 | -72.84% | skill 1/1; base 1/1 | +| codex | kermt-embed-004 | 78,909 | 54,238 | +24,671 | +45.49% | skill 1/1; base 1/1 | +| codex | kermt-embed-005 | 135,150 | 503,531 | -368,381 | -73.16% | skill 1/1; base 1/1 | +| ALL AGENTS | Dataset aggregate | 2,615,826 | 4,361,881 | N/A | N/A | skill 10/10; base 20/20 | + +Prompt tokens include cached reads, so total tokens are `prompt + completion` (cached is not added twice). The Efficiency score uses `(prompt - cached) + completion`. N/A means the relevant trajectory counters were not available; coverage is never estimated. + +## Tier Status + +| Tier | Purpose | Status | Evidence | +|---|---|---|---| +| Tier 1 | Static validation | **PASSED WITH OBSERVATIONS** | 11 validator(s); 44 finding(s) | +| Tier 2 | Semantic deduplication | **PASSED** | 2 validator(s); 0 finding(s) | +| Tier 3 | Live agent evaluation | **PASS** | 2 agent(s); 5 task(s) | + +## Findings and Observations + +
+Show detailed findings and successful checks + +- **MEDIUM** QUALITY/quality_correctness: No documented scripts in table format (`skills/kermt-embed/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: Instructions don't mention 'run_script' (`skills/kermt-embed/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.author' (`skills/kermt-embed/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.tags' (`skills/kermt-embed/SKILL.md`) +- **MEDIUM** SCHEMA/metadata_key_style: Metadata key 'risk_tier' is not kebab-case (`skills/kermt-embed/SKILL.md`) +- 39 additional finding(s) are available in the full evaluation artifacts. + +
+ +## Scoring Methodology + +
+Show dimension definitions, source signals, and thresholds + +| Dimension | Question | Scored signals | +|---|---|---| +| Security | Is it safe to use? | `security` (100%) | +| Correctness | Is the answer correct? | `accuracy` (100%) | +| Discoverability | Was the right skill loaded when needed? | `skill_execution` (100%) | +| Effectiveness | Did the skill help complete the task? | `goal_accuracy` (50%) + `behavior_check` (50%) | +| Efficiency | Did it avoid wasted tool calls and token usage? | `skill_efficiency` (50%) + `token_efficiency` (50%) | + +- Dimension bands: PASS at 50% or above; NEUTRAL from 40% to below 50%; FAIL below 40%. +- Overall Tier 3 lift: PASS at +5 points or more; FAIL at -10 points or less; values between those bands are NEUTRAL. +- Overall verdict: PASS only when every configured dimension passes for at least one supported agent. Lift is reported as diagnostic evidence and does not override this gate. +- The 50% attempt pass threshold is a separate per-task gate; it is not the dimension pass threshold. +- Effectiveness is the equal-weight mean of goal completion (`goal_accuracy`) and expected workflow adherence (`behavior_check`). +- Efficiency is 50% tool-call productivity (the backward-compatible `skill_efficiency` wire id) and 50% `token_efficiency`. Positive-case skill routing is scored under Discoverability, not Efficiency; a negative case without a routing target is N/A. N/A sources are omitted, remaining weights are renormalized, and the dimension is marked partial. + +Signals present in this run: + +- `security` (Security): unsafe operations, secret leakage, and unauthorized access. +- `skill_execution` (Skill Execution): whether the expected skill was selected, decoys were avoided, and the workflow executed. +- `skill_efficiency` (Tool Productivity): tool-call productivity (legacy wire id; routing is scored under Discoverability). +- `accuracy` (Accuracy): final-answer correctness against the reference answer. +- `goal_accuracy` (Goal Accuracy): whether the user's goal was achieved. +- `behavior_check` (Behavior Check): whether the expected workflow behavior was followed. +- `token_efficiency` (Token Efficiency): actual uncached prompt plus completion usage (50% of Efficiency). + +
+ +## Freshness + +Regenerate this benchmark when the skill, evaluation dataset, target agent/model, evaluator version, environment, or scoring policy changes. diff --git a/skills/kermt-embed/skill-card.md b/skills/kermt-embed/skill-card.md index 62ef1d9..ce73fe6 100644 --- a/skills/kermt-embed/skill-card.md +++ b/skills/kermt-embed/skill-card.md @@ -1,69 +1,85 @@ ## Description:
-Extracts per-molecule embeddings from any encoder-bearing KERMT checkpoint (grover_base / cmim / hybrid / finetuned), writing one `.npy` per readout type plus `canonical_smiles.npy` and `validity.npy`.
+Extract per-molecule embeddings from any encoder-bearing KERMT checkpoint using containerized embedding extraction, writing per-readout .npy embeddings, canonical SMILES, and validity arrays to user-selected host directories.
This skill is ready for commercial/non-commercial use.
## Owner -NVIDIA (evax@nvidia.com)
+NVIDIA
### License/Terms of Use:
-Apache-2.0
- +Apache 2.0
## Use Case:
-ML engineers and cheminformaticians who need fixed-length molecular representations from a KERMT encoder to feed a downstream model, clustering step, or similarity search, without training a task head.
- -### Requirements/Dependencies:
-Requires API Key or External Credential: No
-Credential Type(s): None
- -* `kermt-setup` completed (supplies the `kermt:latest` image)
-* Docker, NVIDIA Container Toolkit, CUDA-capable NVIDIA GPU
-* Any encoder-bearing KERMT checkpoint
-* A SMILES input CSV (featurization happens on the fly — no precomputed features required)
- -Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
+Developers and researchers use this skill to extract per-molecule embeddings from KERMT checkpoints for downstream molecular property prediction and cheminformatics tasks.
### Deployment Geography for Use:
Global
-## Known Risks and Mitigations:
-Risk: Embeddings from different checkpoints or readout types are not comparable, and mixing them in a downstream model produces silently meaningless results.
-Mitigation: The skill writes each readout type to a separately named `.npy` and emits `canonical_smiles.npy` alongside, so provenance and row alignment are explicit.
+## Requirements / Dependencies:
+**Requires API Key or External Credential:** [Optional]
+**Credential Type(s):** [API key]
-Risk: Invalid SMILES rows would misalign embeddings against the caller's original input ordering.
-Mitigation: The skill emits `validity.npy`, letting the caller filter or realign rows deterministically.
+Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
-Risk: Large input libraries can produce embedding arrays of substantial size and consume significant disk.
-Mitigation: Outputs are written to a user-specified directory; users should size storage for the molecule count and embedding dimensionality before running.
+## Known Risks and Mitigations:
+Risk: Review before execution as proposals could introduce incorrect or misleading guidance into skills.
+Mitigation: Review and scan skill before deployment.
## Reference(s):
-- [KERMT repository](https://github.com/NVIDIA-BioNeMo/KERMT)
-- `task/extract_embeddings.py` — the underlying extraction entry point
-- Related skills: `kermt-setup`, `kermt-finetune`, `kermt-infer`
+- [Released KERMT Models](references/released-models.md)
+- [KERMT Paper (arXiv:2510.12719)](https://arxiv.org/abs/2510.12719)
+- [GROVER Paper (arXiv:2007.02835)](https://arxiv.org/abs/2007.02835)
+- [NV-KERMT-70M-v2 on Hugging Face](https://huggingface.co/nvidia/NV-KERMT-70M-v2)
+ ## Skill Output:
-**Output Type(s):** [Files]
-**Output Format:** [NumPy `.npy` arrays — one per readout type (atom_from_atom, bond_from_atom, atom_from_bond, bond_from_bond), plus `canonical_smiles.npy` and `validity.npy`]
-**Output Parameters:** [2D — one row per input molecule, one column per embedding dimension]
-**Other Properties Related to Output:** [Row order corresponds to the input CSV; `validity.npy` identifies rows whose SMILES failed to parse]
+**Output Type(s):** [Files, Shell commands]
+**Output Format:** [Markdown with inline bash code blocks]
+**Output Parameters:** [1D]
+**Other Properties Related to Output:** [None]
## Evaluation Agents Used:
-Target agents: `claude-code`, `codex`. NVSkills-Eval has not yet been run against this skill — see Evaluation Results.
+- Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`)
+- Codex (`openai/openai/gpt-5.5`)
+ + ## Evaluation Tasks:
-5 evaluation tasks defined in `evals/evals.json`, covering checkpoint compatibility, readout selection, and output artifact production.
+5 evaluation tasks (4 positive, 1 negative), 3 attempts per task, evaluated in isolated k8s-sandbox pods.
## Evaluation Metrics Used:
-Planned NVSkills-Eval dimensions: Security, Correctness, Discoverability, Effectiveness, Efficiency.
+Reported benchmark dimensions:
+- Security: Whether the skill avoids unsafe operations, secret leakage, and unauthorized access.
+- Correctness: Whether the final answer is correct against the reference answer.
+- Discoverability: Whether the expected skill was selected, decoys were avoided, and the workflow executed.
+- Effectiveness: Whether the skill helps complete the user's goal (50% goal completion + 50% expected workflow adherence).
+- Efficiency: Whether the skill avoids wasted tool calls and token usage (50% tool-call productivity + 50% token efficiency).
+ +Underlying evaluation signals used in this run:
+- `security`: Checks for unsafe operations, secret leakage, and unauthorized access.
+- `skill_execution`: Whether the expected skill was selected and the workflow executed.
+- `skill_efficiency`: Tool-call productivity; routing is scored under Discoverability.
+- `accuracy`: Final-answer correctness against the reference answer.
+- `goal_accuracy`: Whether the user's goal was achieved.
+- `behavior_check`: Whether the expected workflow behavior was followed.
+- `token_efficiency`: Actual uncached prompt plus completion token usage.
+ + ## Evaluation Results:
-Pending. NVSkills-Eval has not been run for this skill; results and a `BENCHMARK.md` will be published when the evaluation pipeline runs.
+| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 82.0% | 83.3% | +| Security | 92.3% → 100.0% (+7.7 points) | 85.7% → 100.0% (+14.3 points) | +| Correctness | 24.6% → 84.0% (+59.4 points) | 80.0% → 84.0% (+4.0 points) | +| Discoverability | 93.8% | 86.3% | +| Effectiveness | 25.4% → 47.5% (+22.1 points) | 30.4% → 51.5% (+21.1 points) | +| Efficiency | 84.6% | 94.9% | ## Skill Version(s):
-b1c082c (source: git SHA, committed 2026-07-17)
+77111e0 (source: git SHA, committed 2026-09-09)
## Ethical Considerations:
NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their internal team to ensure this skill meets requirements for the relevant industry and use case and addresses unforeseen product misuse.
(For Release on NVIDIA Platforms Only)
-Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://www.nvidia.com/en-us/support/submit-security-vulnerability/).
+Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://app.intigriti.com/programs/nvidia/nvidiavdp/detail).
diff --git a/skills/kermt-embed/skill.oms.sig b/skills/kermt-embed/skill.oms.sig new file mode 100644 index 0000000..b663be1 --- /dev/null +++ b/skills/kermt-embed/skill.oms.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAia2VybXQtZW1iZWQiLAogICAgICAiZGlnZXN0IjogewogICAgICAgICJzaGEyNTYiOiAiNWY3MDMxN2M3NmRiZGJhNmMzNmMzNGZmMjI1MTEzMjgyMTkzNzRjZmY4MmM0NjMxYTFiZTk5NmVlMjRkYmY4MSIKICAgICAgfQogICAgfQogIF0sCiAgInByZWRpY2F0ZVR5cGUiOiAiaHR0cHM6Ly9tb2RlbF9zaWduaW5nL3NpZ25hdHVyZS92MS4wIiwKICAicHJlZGljYXRlIjogewogICAgInNlcmlhbGl6YXRpb24iOiB7CiAgICAgICJtZXRob2QiOiAiZmlsZXMiLAogICAgICAiaWdub3JlX3BhdGhzIjogWwogICAgICAgICIuZ2l0aHViIiwKICAgICAgICAiLmdpdGF0dHJpYnV0ZXMiLAogICAgICAgICIuZ2l0IiwKICAgICAgICAiLmdpdGlnbm9yZSIKICAgICAgXSwKICAgICAgImhhc2hfdHlwZSI6ICJzaGEyNTYiLAogICAgICAiYWxsb3dfc3ltbGlua3MiOiBmYWxzZQogICAgfSwKICAgICJyZXNvdXJjZXMiOiBbCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICJjNzJkNmQzMmY3ZWRlMTg1N2ZjYWQyZDgyMjU1YzVjMzljZGIzMjM1ZDRlMWRkY2NiYjllNTI2NjE2OGZjZWEwIiwKICAgICAgICAibmFtZSI6ICJCRU5DSE1BUksubWQiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICJmNjk1NmU5NTc0YWFhZTM4MThjY2ViNzg3ZjI2Y2FlMjlhYjgzMjM1YzBlOGUyNDYzOGE5ZWMwY2Y3ZWQyYjQ1IiwKICAgICAgICAibmFtZSI6ICJTS0lMTC5tZCIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogIjI0MGYxMmM2ZjZiNjk5YzRjYmM5NGE2MjhhMjM1ZjJkNzY2OTFjMmI0NmVlY2U2NjkwODFmM2RjZDIwOGIxNjAiLAogICAgICAgICJuYW1lIjogImNvbmZpZy9kZWZhdWx0c19lbWJlZC5qc29uIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiMjg1ODA0NDkzMjFhMzFhNjc2MjQxMTdmNTdiYzVjNGU1OGU2YmY3YjE5ZTE5Mzg4NTRhNTdjZWIxM2NjNmYzMyIsCiAgICAgICAgIm5hbWUiOiAiY29uZmlnL3JlbGVhc2VkX21vZGVsLmpzb24iCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICIyYzViMjhmZDE2ZDBiNjQ2MDA2ZGM1ZjhkZTk3YzUwYWE5OTgwY2NjODIyYTI2ZGEyYzgyOTdhMTUwMGQzMTJkIiwKICAgICAgICAibmFtZSI6ICJldmFscy9ldmFscy5qc29uIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiODU2NmYwYTcyZjE3MjY4NTAzNzZiODAzNWFmN2RmMjI2MDNkYThiMjJkODE1OTlhMmFkNjIzOWFmMzQzMzM1OSIsCiAgICAgICAgIm5hbWUiOiAicmVmZXJlbmNlcy9yZWxlYXNlZC1tb2RlbHMubWQiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICIwMjYyMjBhMjI2ZDhmODQxODY0ZjNmOGQ3NGRmYzdjZWVlNTJkOTg1ZDQ1YTllMTQ2ZTI3Y2MxNTIzOWZmNTQyIiwKICAgICAgICAibmFtZSI6ICJzY3JpcHRzL191dGlscy5weSIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogIjBiYzZjYzg3MjlkYjNhM2E5MzlmY2U3YTk1ZTVmNmQxNjZhZjU1MTdmYjY5MzQ2YmRhMmMzMDExNDVhOGY1MDIiLAogICAgICAgICJuYW1lIjogInNjcmlwdHMvY2hlY2tfY2hlY2twb2ludC5weSIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogIjY4OTI2ODE4NzNkNjZlN2JlNDEyYjc3MzBhMzQ5ZTY0MDkwOWM3MmFmYWUyMjI4NTRmN2EwMWI2NGIyYWFiNTYiLAogICAgICAgICJuYW1lIjogInNjcmlwdHMvY2hlY2tfZGF0YS5weSIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogIjRkMGU2NDg1Mjk2NzdlMzQwOTRmNjE4MTA1ZjM5MmY0M2IyMWVlNWJjZDMzZTM4ZWI4YjAwN2M4NDkxOGFhYTMiLAogICAgICAgICJuYW1lIjogInNjcmlwdHMvZmV0Y2hfcmVsZWFzZWRfbW9kZWwucHkiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICJmY2RhYjU5NWMwZTg5ZGYxNTgxOWE3ZTI3NDFjMzFmNTEyZDk1ZTA4ZjY0NjFmODFkNTM1ZWNhM2NiYTQwMmI4IiwKICAgICAgICAibmFtZSI6ICJzY3JpcHRzL2tlcm10X2NvbnRhaW5lci5zaCIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogIjE0MjYyODlhYjgzMzRmMWE5OWU5YmYxYTQxNzY3OTU4MjRmMjc1MTlmZjkxNDIxNmEzZjA3ZjkwZWVhYWYwM2IiLAogICAgICAgICJuYW1lIjogInNjcmlwdHMvcHJlcGFyZV9kYXRhLnB5IgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiOGM5MDlhNzAxZWQ1NjEzMWVkYjg1MTA4MDA0MWUzY2U5NjExY2RlOTA2MDdkYjE2MTZjY2ZlODlkOWZlNTVhZCIsCiAgICAgICAgIm5hbWUiOiAic2NyaXB0cy9ydW5fZXh0cmFjdF9lbWJlZGRpbmdzLnB5IgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiOWVmM2Q5ZjIxM2FkYTYwZGNmMGMzYTJlZTgxZTk5MzAwMDBhMzY5NDc2N2ZmZmQ4M2ZiMzg3OWZlZDUyNjgyZCIsCiAgICAgICAgIm5hbWUiOiAic2tpbGwtY2FyZC5tZCIKICAgICAgfQogICAgXQogIH0KfQ==","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGUCMQCVJmFPjN4Z5h7BfzbH3Sv1RcxJ1RPHn2HLWDM/yWU3+FDOOHu/vd3V84R4al1oi4gCMG00LhzTWxfhQi9ARf/N8HglyczQl6SV4sU5PNvJBzlh+wRo5GiS86zLCu6VGDkMFQ==","keyid":""}]}} \ No newline at end of file diff --git a/skills/kermt-finetune/BENCHMARK.md b/skills/kermt-finetune/BENCHMARK.md new file mode 100644 index 0000000..da144d9 --- /dev/null +++ b/skills/kermt-finetune/BENCHMARK.md @@ -0,0 +1,129 @@ +# Skill Benchmark: kermt-finetune + +> ✅ **Overall verdict: PASS — Recommended for publication** + +## Publication Recommendation + +Recommended for publication based on the completed evaluation evidence in this report. + +## Evaluation Metadata + +- Skill: `kermt-finetune` +- Evaluation date: 2026-09-15 +- Evaluator version: `1.5.6` +- Agents: Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`), Codex (`openai/openai/gpt-5.5`) +- Tasks: 5 evaluation tasks (4 positive, 1 negative) +- Dataset digest: `sha256:78e80c1a73622d5a07c1526f6270c42770347f5faa77969107e6cc60de50bc41` (skill-evaluator-dataset-snapshot/1) +- Attempts per task: 3 +- Environment: `k8s-sandbox` +- Tier 2 evidence: required for publication +- Tier 3 evidence: required for publication + +Each task attempt ran in its own isolated sandbox pod. + +## What This Report Answers + +The three-tier evaluation checks whether the skill: + +- is safe to use; +- produces correct answers; +- is discovered and activated when needed; +- helps the agent complete the user's goal and expected workflow; and +- avoids wasted skill and tool usage. + +## Results at a Glance + +| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 84.9% — baseline ran, but no comparable score was available; uplift unavailable | 69.5% — baseline ran, but no comparable score was available; uplift unavailable | +| Security | 88.5% → 100.0% (+11.5 points) | 27.3% → 71.4% (+44.1 points) | +| Correctness | 23.1% → 96.0% (+72.9 points) | 78.2% → 91.4% (+13.2 points) | +| Discoverability | 97.5% — baseline ran, but no comparable score was available; uplift unavailable | 79.2% — baseline ran, but no comparable score was available; uplift unavailable | +| Effectiveness | 18.9% → 51.0% (+32.1 points) | 30.7% → 35.7% (+5.0 points) | +| Efficiency | 79.8% — baseline ran, but no comparable score was available; uplift unavailable | 69.7% — baseline ran, but no comparable score was available; uplift unavailable | + +**How to read this table:** baseline is the same task attempted without the target skill. Scores are rounded to one decimal; threshold-adjacent values use additional precision so their displayed band matches the verdict. Uplift is derived from those displayed scores and shown in percentage points. + +Example: `47.0% → 92.0% (+45.0 points)` means the skill-assisted run scored 92.0%, 45.0 percentage points above its 47.0% no-skill baseline. + +A partial dimension was calculated from only the available configured signals; review the detailed report before relying on it. + +## Token Usage + +Actual Tier 3 execution usage is reported for every observed agent/case pair and both conditions. + +| Agent | Dataset case | With skill | Without skill | Delta | Change | Coverage | +|---|---|---:|---:|---:|---:|---| +| claude-code | All cases | 1,624,153 | 5,610,370 | N/A | N/A | skill 5/5; base 13/13 | +| claude-code | kermt-finetune-001 | 268,233 | 530,319 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-finetune-002 | 460,964 | 908,098 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-finetune-003 | 324,432 | 861,553 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-finetune-004 | 30,364 | 30,003 | +361 | +1.20% | skill 1/1; base 1/1 | +| claude-code | kermt-finetune-005 | 540,160 | 3,280,397 | N/A | N/A | skill 1/1; base 3/3 | +| codex | All cases | 1,907,233 | 6,598,000 | N/A | N/A | skill 7/7; base 11/11 | +| codex | kermt-finetune-001 | 126,075 | 990,399 | N/A | N/A | skill 1/1; base 3/3 | +| codex | kermt-finetune-002 | 395,517 | 798,736 | -403,219 | -50.48% | skill 1/1; base 1/1 | +| codex | kermt-finetune-003 | 190,354 | 1,944,827 | N/A | N/A | skill 1/1; base 3/3 | +| codex | kermt-finetune-004 | 13,917 | 18,278 | -4,361 | -23.86% | skill 1/1; base 1/1 | +| codex | kermt-finetune-005 | 1,181,370 | 2,845,760 | -1,664,390 | -58.49% | skill 3/3; base 3/3 | +| ALL AGENTS | Dataset aggregate | 3,531,386 | 12,208,370 | N/A | N/A | skill 12/12; base 24/24 | + +Prompt tokens include cached reads, so total tokens are `prompt + completion` (cached is not added twice). The Efficiency score uses `(prompt - cached) + completion`. N/A means the relevant trajectory counters were not available; coverage is never estimated. + +## Tier Status + +| Tier | Purpose | Status | Evidence | +|---|---|---|---| +| Tier 1 | Static validation | **PASSED WITH OBSERVATIONS** | 11 validator(s); 43 finding(s) | +| Tier 2 | Semantic deduplication | **PASSED** | 2 validator(s); 0 finding(s) | +| Tier 3 | Live agent evaluation | **PASS** | 2 agent(s); 5 task(s) | + +## Findings and Observations + +
+Show detailed findings and successful checks + +- **MEDIUM** QUALITY/quality_correctness: No documented scripts in table format (`skills/kermt-finetune/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: Instructions don't mention 'run_script' (`skills/kermt-finetune/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.author' (`skills/kermt-finetune/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.tags' (`skills/kermt-finetune/SKILL.md`) +- **MEDIUM** SCHEMA/metadata_key_style: Metadata key 'risk_tier' is not kebab-case (`skills/kermt-finetune/SKILL.md`) +- 38 additional finding(s) are available in the full evaluation artifacts. + +
+ +## Scoring Methodology + +
+Show dimension definitions, source signals, and thresholds + +| Dimension | Question | Scored signals | +|---|---|---| +| Security | Is it safe to use? | `security` (100%) | +| Correctness | Is the answer correct? | `accuracy` (100%) | +| Discoverability | Was the right skill loaded when needed? | `skill_execution` (100%) | +| Effectiveness | Did the skill help complete the task? | `goal_accuracy` (50%) + `behavior_check` (50%) | +| Efficiency | Did it avoid wasted tool calls and token usage? | `skill_efficiency` (50%) + `token_efficiency` (50%) | + +- Dimension bands: PASS at 50% or above; NEUTRAL from 40% to below 50%; FAIL below 40%. +- Overall Tier 3 lift: PASS at +5 points or more; FAIL at -10 points or less; values between those bands are NEUTRAL. +- Overall verdict: PASS only when every configured dimension passes for at least one supported agent. Lift is reported as diagnostic evidence and does not override this gate. +- The 50% attempt pass threshold is a separate per-task gate; it is not the dimension pass threshold. +- Effectiveness is the equal-weight mean of goal completion (`goal_accuracy`) and expected workflow adherence (`behavior_check`). +- Efficiency is 50% tool-call productivity (the backward-compatible `skill_efficiency` wire id) and 50% `token_efficiency`. Positive-case skill routing is scored under Discoverability, not Efficiency; a negative case without a routing target is N/A. N/A sources are omitted, remaining weights are renormalized, and the dimension is marked partial. + +Signals present in this run: + +- `security` (Security): unsafe operations, secret leakage, and unauthorized access. +- `skill_execution` (Skill Execution): whether the expected skill was selected, decoys were avoided, and the workflow executed. +- `skill_efficiency` (Tool Productivity): tool-call productivity (legacy wire id; routing is scored under Discoverability). +- `accuracy` (Accuracy): final-answer correctness against the reference answer. +- `goal_accuracy` (Goal Accuracy): whether the user's goal was achieved. +- `behavior_check` (Behavior Check): whether the expected workflow behavior was followed. +- `token_efficiency` (Token Efficiency): actual uncached prompt plus completion usage (50% of Efficiency). + +
+ +## Freshness + +Regenerate this benchmark when the skill, evaluation dataset, target agent/model, evaluator version, environment, or scoring policy changes. diff --git a/skills/kermt-finetune/skill-card.md b/skills/kermt-finetune/skill-card.md index a29e731..ec943ff 100644 --- a/skills/kermt-finetune/skill-card.md +++ b/skills/kermt-finetune/skill-card.md @@ -1,78 +1,85 @@ ## Description:
-Finetunes a pretrained KERMT encoder on a labeled CSV — validating the input checkpoint and dataset, preparing features and optional splits, then launching `main.py finetune` inside the KERMT container as a detached, hours-scale run.
+Finetune a pretrained KERMT encoder on a labeled CSV. Validate the checkpoint and data, prepare features, and run containerized training. Use a local checkpoint or optionally download a pinned Hugging Face model bundle using HF_TOKEN if configured. Write model bundles, prepared data, logs, and trained models to user-selected host directories.
This skill is ready for commercial/non-commercial use.
## Owner -NVIDIA (evax@nvidia.com)
+NVIDIA
### License/Terms of Use:
-Apache-2.0
- +Apache 2.0
## Use Case:
-Computational chemists and ML engineers adapting a pretrained KERMT encoder (grover_base / cmim / hybrid) to their own labeled molecular property dataset — for example an in-house ADMET assay panel — before running predictions with `kermt-infer`.
- -### Requirements/Dependencies:
-Requires API Key or External Credential: Optional
-Credential Type(s): API key — `WANDB_API_KEY` for optional Weights & Biases run tracking
- -* `kermt-setup` completed (supplies the `kermt:latest` image)
-* Docker, NVIDIA Container Toolkit, CUDA-capable NVIDIA GPU
-* A pretrain checkpoint (grover_base, cmim, or hybrid)
-* A labeled CSV with SMILES and one or more target columns
-* Hyperparameters default from `config/defaults_finetune.json`, overridable per flag
- -Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
+Developers and engineers finetuning pretrained KERMT molecular property prediction models on user-supplied labeled CSV datasets for small molecule drug property prediction tasks.
### Deployment Geography for Use:
Global
-## Known Risks and Mitigations:
-Risk: Finetuning is an hours-scale GPU workload that a single agent instruction can start, consuming substantial GPU-hours or cloud spend.
-Mitigation: Runs launch detached with a run manifest, and `kermt-monitor` gives the user continuous visibility and the container identifiers needed to terminate early.
- -Risk: Supplying a finetuned checkpoint instead of a pretrain checkpoint would produce an invalid training configuration.
-Mitigation: The skill validates the checkpoint type before launching and rejects non-pretrain checkpoints.
+## Requirements / Dependencies:
+**Requires API Key or External Credential:** [Optional]
+**Credential Type(s):** [API key]
-Risk: A poorly split dataset (e.g. random splits on congeneric series) yields optimistic validation metrics that do not generalize.
-Mitigation: The skill exposes explicit split handling rather than defaulting silently; users remain responsible for choosing a split strategy appropriate to their chemistry.
- -Risk: Training writes checkpoints and logs into a user-specified directory and can overwrite artifacts from a previous run.
-Mitigation: Outputs are written under an explicit run directory; users should use distinct output paths per experiment.
+Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
-Risk: When Weights & Biases tracking is enabled, run metadata is transmitted to a third-party service.
-Mitigation: W&B tracking is optional and off unless the user supplies `WANDB_API_KEY`; users should confirm their data-handling policy before enabling it.
+## Known Risks and Mitigations:
+Risk: Review before execution as proposals could introduce incorrect or misleading guidance into skills.
+Mitigation: Review and scan skill before deployment.
## Reference(s):
-- [KERMT repository](https://github.com/NVIDIA-BioNeMo/KERMT)
-- `config/defaults_finetune.json` — default hyperparameters
-- Related skills: `kermt-setup`, `kermt-monitor`, `kermt-infer`, `kermt-embed`
+- [Released Models](references/released-models.md)
+- [Multitask finetuning and acceleration of chemical pretrained models (KERMT paper)](https://arxiv.org/abs/2510.12719)
+- [Self-Supervised Graph Transformer on Large-Scale Molecular Data (GROVER paper)](https://arxiv.org/abs/2007.02835)
+- [NV-KERMT-70M-v2 on Hugging Face](https://huggingface.co/nvidia/NV-KERMT-70M-v2)
+ ## Skill Output:
-**Output Type(s):** [Files, Analysis]
-**Output Format:** [Model checkpoint files; training logs; `run.json` manifest; Markdown launch summary]
-**Output Parameters:** [1D — run identifier, container id, configured hyperparameters, output paths]
-**Other Properties Related to Output:** [Detached execution: the skill returns after launch, not after training completes. Use `kermt-monitor` for progress.]
+**Output Type(s):** [Shell commands, Configuration instructions, Files]
+**Output Format:** [Markdown with inline bash code blocks]
+**Output Parameters:** [1D]
+**Other Properties Related to Output:** [Produces run.json manifest, finetune logs, TensorBoard events, best-val and last checkpoints, and held-out test predictions]
## Evaluation Agents Used:
-Target agents: `claude-code`, `codex`. NVSkills-Eval has not yet been run against this skill — see Evaluation Results.
+- Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`)
+- Codex (`openai/openai/gpt-5.5`)
+ + ## Evaluation Tasks:
-5 evaluation tasks defined in `evals/evals.json`, covering checkpoint validation, dataset validation, data preparation, and detached launch.
+5 evaluation tasks (4 positive, 1 negative) with 3 attempts each, run in isolated sandbox pods (evaluator v1.5.6).
## Evaluation Metrics Used:
-Planned NVSkills-Eval dimensions: Security, Correctness, Discoverability, Effectiveness, Efficiency.
+Reported benchmark dimensions:
+- Security: Checks for unsafe operations, secret leakage, and unauthorized access.
+- Correctness: Checks final-answer correctness against the reference answer.
+- Discoverability: Checks whether the expected skill was selected and the workflow executed.
+- Effectiveness: Checks whether the skill helped complete the user's goal and expected workflow (50% goal_accuracy + 50% behavior_check).
+- Efficiency: Checks tool-call productivity and token usage (50% skill_efficiency + 50% token_efficiency).
+ +Underlying evaluation signals used in this run:
+- `security`: Unsafe operations, secret leakage, and unauthorized access.
+- `accuracy`: Final-answer correctness against the reference answer.
+- `skill_execution`: Whether the expected skill was selected, decoys were avoided, and the workflow executed.
+- `goal_accuracy`: Whether the user's goal was achieved.
+- `behavior_check`: Whether the expected workflow behavior was followed.
+- `skill_efficiency`: Tool-call productivity.
+- `token_efficiency`: Actual uncached prompt plus completion token usage.
+ + ## Evaluation Results:
-Pending. NVSkills-Eval has not been run for this skill; results and a `BENCHMARK.md` will be published when the evaluation pipeline runs.
+| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 84.9% | 69.5% | +| Security | 88.5% → 100.0% (+11.5 points) | 27.3% → 71.4% (+44.1 points) | +| Correctness | 23.1% → 96.0% (+72.9 points) | 78.2% → 91.4% (+13.2 points) | +| Discoverability | 97.5% | 79.2% | +| Effectiveness | 18.9% → 51.0% (+32.1 points) | 30.7% → 35.7% (+5.0 points) | +| Efficiency | 79.8% | 69.7% | ## Skill Version(s):
-b1c082c (source: git SHA, committed 2026-07-17)
+77111e0 (source: git SHA, committed 2026-09-09)
## Ethical Considerations:
NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their internal team to ensure this skill meets requirements for the relevant industry and use case and addresses unforeseen product misuse.
-Models finetuned with this skill inherit the biases and coverage limits of the user's training data. Resulting predictions are research hypotheses and must not be used as the sole basis for clinical, safety, or regulatory decisions.
- (For Release on NVIDIA Platforms Only)
-Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://www.nvidia.com/en-us/support/submit-security-vulnerability/).
+Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://app.intigriti.com/programs/nvidia/nvidiavdp/detail).
diff --git a/skills/kermt-finetune/skill.oms.sig b/skills/kermt-finetune/skill.oms.sig new file mode 100644 index 0000000..1d90529 --- /dev/null +++ b/skills/kermt-finetune/skill.oms.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAia2VybXQtZmluZXR1bmUiLAogICAgICAiZGlnZXN0IjogewogICAgICAgICJzaGEyNTYiOiAiYzk5ZTZjMDBiOTQxMTEwM2FmYWI4YjNkN2Y0ZGJkMDI2NjNlZDVkZDJmMTVjYWM3YzEzNTBjMzgzYWIzY2I5ZSIKICAgICAgfQogICAgfQogIF0sCiAgInByZWRpY2F0ZVR5cGUiOiAiaHR0cHM6Ly9tb2RlbF9zaWduaW5nL3NpZ25hdHVyZS92MS4wIiwKICAicHJlZGljYXRlIjogewogICAgInNlcmlhbGl6YXRpb24iOiB7CiAgICAgICJpZ25vcmVfcGF0aHMiOiBbCiAgICAgICAgIi5naXRhdHRyaWJ1dGVzIiwKICAgICAgICAiLmdpdCIsCiAgICAgICAgIi5naXRodWIiLAogICAgICAgICIuZ2l0aWdub3JlIgogICAgICBdLAogICAgICAiaGFzaF90eXBlIjogInNoYTI1NiIsCiAgICAgICJtZXRob2QiOiAiZmlsZXMiLAogICAgICAiYWxsb3dfc3ltbGlua3MiOiBmYWxzZQogICAgfSwKICAgICJyZXNvdXJjZXMiOiBbCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAiQkVOQ0hNQVJLLm1kIiwKICAgICAgICAiZGlnZXN0IjogImZjNGIwN2IyMGY4YjA0MTIzYThhZGQxYjM0NTdiNTlkOWQxYTZkM2I2ZGUwNjZhZTlhMTBjMTQzNmI4NDliY2UiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAiU0tJTEwubWQiLAogICAgICAgICJkaWdlc3QiOiAiYmYxNTdiZDQ1MzUyOGE5YzJkMjcyZDUzN2ZlM2ZjMTAzYTlkZTgzZjkwN2Y5MDcxZThjMzVhNzcwYzU1ODJmYyIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJjb25maWcvZGVmYXVsdHNfZmluZXR1bmUuanNvbiIsCiAgICAgICAgImRpZ2VzdCI6ICIwOTVjMGYxZmExZjdhZDI0ZTE4NjZiYWMxZTcwZjgzOGViNWVlNDIzMmNhZWIyZjc2ODUzZWViOTU2NDM1N2EwIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJuYW1lIjogImNvbmZpZy9yZWxlYXNlZF9tb2RlbC5qc29uIiwKICAgICAgICAiZGlnZXN0IjogIjI4NTgwNDQ5MzIxYTMxYTY3NjI0MTE3ZjU3YmM1YzRlNThlNmJmN2IxOWUxOTM4ODU0YTU3Y2ViMTNjYzZmMzMiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAiZXZhbHMvZXZhbHMuanNvbiIsCiAgICAgICAgImRpZ2VzdCI6ICJkMGExNjk1ZDZlZjRjNjU4ZDUzZTA2NWI2MDI4Y2YxMjRkNWM1MmRkNTQyNjcwMTE5MjVkZGNkMDM1NWU4N2Y4IgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJuYW1lIjogInJlZmVyZW5jZXMvcmVsZWFzZWQtbW9kZWxzLm1kIiwKICAgICAgICAiZGlnZXN0IjogIjg1NjZmMGE3MmYxNzI2ODUwMzc2YjgwMzVhZjdkZjIyNjAzZGE4YjIyZDgxNTk5YTJhZDYyMzlhZjM0MzMzNTkiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAic2NyaXB0cy9fdXRpbHMucHkiLAogICAgICAgICJkaWdlc3QiOiAiMDI2MjIwYTIyNmQ4Zjg0MTg2NGYzZjhkNzRkZmM3Y2VlZTUyZDk4NWQ0NWE5ZTE0NmUyN2NjMTUyMzlmZjU0MiIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJzY3JpcHRzL2NoZWNrX2NoZWNrcG9pbnQucHkiLAogICAgICAgICJkaWdlc3QiOiAiMGJjNmNjODcyOWRiM2EzYTkzOWZjZTdhOTVlNWY2ZDE2NmFmNTUxN2ZiNjkzNDZiZGEyYzMwMTE0NWE4ZjUwMiIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJzY3JpcHRzL2NoZWNrX2RhdGEucHkiLAogICAgICAgICJkaWdlc3QiOiAiNjg5MjY4MTg3M2Q2NmU3YmU0MTJiNzczMGEzNDllNjQwOTA5YzcyYWZhZTIyMjg1NGY3YTAxYjY0YjJhYWI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJzY3JpcHRzL2ZldGNoX3JlbGVhc2VkX21vZGVsLnB5IiwKICAgICAgICAiZGlnZXN0IjogIjRkMGU2NDg1Mjk2NzdlMzQwOTRmNjE4MTA1ZjM5MmY0M2IyMWVlNWJjZDMzZTM4ZWI4YjAwN2M4NDkxOGFhYTMiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAic2NyaXB0cy9rZXJtdF9jb250YWluZXIuc2giLAogICAgICAgICJkaWdlc3QiOiAiZmNkYWI1OTVjMGU4OWRmMTU4MTlhN2UyNzQxYzMxZjUxMmQ5NWUwOGY2NDYxZjgxZDUzNWVjYTNjYmE0MDJiOCIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJzY3JpcHRzL3ByZXBhcmVfZGF0YS5weSIsCiAgICAgICAgImRpZ2VzdCI6ICIxNDI2Mjg5YWI4MzM0ZjFhOTllOWJmMWE0MTc2Nzk1ODI0ZjI3NTE5ZmY5MTQyMTZhM2YwN2Y5MGVlYWFmMDNiIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJuYW1lIjogInNjcmlwdHMvcnVuX2ZpbmV0dW5lX2xvY2FsLnB5IiwKICAgICAgICAiZGlnZXN0IjogImFlYjUzNTVmMDVjZjVhYjZkYjJjYjRhODY4YTYxY2JmNGZjYTQ5NmIzZmNkMTYyNjBiYzRkYjU4NTY4Y2MxMTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAic2tpbGwtY2FyZC5tZCIsCiAgICAgICAgImRpZ2VzdCI6ICI0MzQ2ZTY3OTBkZjIyNjhkYjEzZjkyNjI4NDBjZjQyYjk0MTFjZjU2MmQwODAzMzNiNDQ1NTBlNWZjM2YyNjQzIgogICAgICB9CiAgICBdCiAgfQp9","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGUCMQDXjl0fuxHAkTBAAcWl1dZn4LDJVN7cIU5GMXeK0BvJcCup8tIlUkOj3ZhiDi1VQRcCME0U+cC1LXVyTUtIXJVVdh5xZFNmEzk9yTHqSqau9IltmEJVPiOLlOCTt+L1oXiU0Q==","keyid":""}]}} \ No newline at end of file diff --git a/skills/kermt-infer/BENCHMARK.md b/skills/kermt-infer/BENCHMARK.md new file mode 100644 index 0000000..6fa5c6d --- /dev/null +++ b/skills/kermt-infer/BENCHMARK.md @@ -0,0 +1,127 @@ +# Skill Benchmark: kermt-infer + +> ✅ **Overall verdict: PASS — Recommended for publication** + +## Publication Recommendation + +Recommended for publication based on the completed evaluation evidence in this report. + +## Evaluation Metadata + +- Skill: `kermt-infer` +- Evaluation date: 2026-09-15 +- Evaluator version: `1.5.6` +- Agents: Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`), Codex (`openai/openai/gpt-5.5`) +- Tasks: 4 evaluation tasks (3 positive, 1 negative) +- Dataset digest: `sha256:517c8a976ea37ee217ad5779f5e01f7533ffb45540b1496c22de3dd14d114490` (skill-evaluator-dataset-snapshot/1) +- Attempts per task: 3 +- Environment: `k8s-sandbox` +- Tier 2 evidence: required for publication +- Tier 3 evidence: required for publication + +Each task attempt ran in its own isolated sandbox pod. + +## What This Report Answers + +The three-tier evaluation checks whether the skill: + +- is safe to use; +- produces correct answers; +- is discovered and activated when needed; +- helps the agent complete the user's goal and expected workflow; and +- avoids wasted skill and tool usage. + +## Results at a Glance + +| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 78.6% — baseline ran, but no comparable score was available; uplift unavailable | 80.3% — baseline ran, but no comparable score was available; uplift unavailable | +| Security | 100.0% → 100.0% (±0.0 points) | 70.0% → 100.0% (+30.0 points) | +| Correctness | 25.0% → 85.0% (+60.0 points) | 50.0% → 90.0% (+40.0 points) | +| Discoverability | 91.7% — baseline ran, but no comparable score was available; uplift unavailable | 91.7% — baseline ran, but no comparable score was available; uplift unavailable | +| Effectiveness | 25.9% → 35.0% (+9.1 points) | 22.3% → 34.4% (+12.1 points) | +| Efficiency | 81.5% — baseline ran, but no comparable score was available; uplift unavailable | 85.4% — baseline ran, but no comparable score was available; uplift unavailable | + +**How to read this table:** baseline is the same task attempted without the target skill. Scores are rounded to one decimal; threshold-adjacent values use additional precision so their displayed band matches the verdict. Uplift is derived from those displayed scores and shown in percentage points. + +Example: `47.0% → 92.0% (+45.0 points)` means the skill-assisted run scored 92.0%, 45.0 percentage points above its 47.0% no-skill baseline. + +A partial dimension was calculated from only the available configured signals; review the detailed report before relying on it. + +## Token Usage + +Actual Tier 3 execution usage is reported for every observed agent/case pair and both conditions. + +| Agent | Dataset case | With skill | Without skill | Delta | Change | Coverage | +|---|---|---:|---:|---:|---:|---| +| claude-code | All cases | 1,527,449 | 1,249,375 | N/A | N/A | skill 4/4; base 8/8 | +| claude-code | kermt-infer-001 | 310,888 | 430,603 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-infer-002 | 404,604 | 440,486 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-infer-003 | 306,093 | 155,002 | +151,091 | +97.48% | skill 1/1; base 1/1 | +| claude-code | kermt-infer-004 | 505,864 | 223,284 | +282,580 | +126.56% | skill 1/1; base 1/1 | +| codex | All cases | 1,510,773 | 3,889,875 | N/A | N/A | skill 4/4; base 10/10 | +| codex | kermt-infer-001 | 103,536 | 224,502 | N/A | N/A | skill 1/1; base 3/3 | +| codex | kermt-infer-002 | 86,842 | 1,236,389 | N/A | N/A | skill 1/1; base 3/3 | +| codex | kermt-infer-003 | 79,261 | 256,670 | N/A | N/A | skill 1/1; base 3/3 | +| codex | kermt-infer-004 | 1,241,134 | 2,172,314 | -931,180 | -42.87% | skill 1/1; base 1/1 | +| ALL AGENTS | Dataset aggregate | 3,038,222 | 5,139,250 | N/A | N/A | skill 8/8; base 18/18 | + +Prompt tokens include cached reads, so total tokens are `prompt + completion` (cached is not added twice). The Efficiency score uses `(prompt - cached) + completion`. N/A means the relevant trajectory counters were not available; coverage is never estimated. + +## Tier Status + +| Tier | Purpose | Status | Evidence | +|---|---|---|---| +| Tier 1 | Static validation | **PASSED WITH OBSERVATIONS** | 11 validator(s); 58 finding(s) | +| Tier 2 | Semantic deduplication | **PASSED** | 2 validator(s); 0 finding(s) | +| Tier 3 | Live agent evaluation | **PASS** | 2 agent(s); 4 task(s) | + +## Findings and Observations + +
+Show detailed findings and successful checks + +- **MEDIUM** QUALITY/quality_correctness: No documented scripts in table format (`skills/kermt-infer/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: Instructions don't mention 'run_script' (`skills/kermt-infer/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.author' (`skills/kermt-infer/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.tags' (`skills/kermt-infer/SKILL.md`) +- **MEDIUM** SCHEMA/metadata_key_style: Metadata key 'risk_tier' is not kebab-case (`skills/kermt-infer/SKILL.md`) +- 53 additional finding(s) are available in the full evaluation artifacts. + +
+ +## Scoring Methodology + +
+Show dimension definitions, source signals, and thresholds + +| Dimension | Question | Scored signals | +|---|---|---| +| Security | Is it safe to use? | `security` (100%) | +| Correctness | Is the answer correct? | `accuracy` (100%) | +| Discoverability | Was the right skill loaded when needed? | `skill_execution` (100%) | +| Effectiveness | Did the skill help complete the task? | `goal_accuracy` (50%) + `behavior_check` (50%) | +| Efficiency | Did it avoid wasted tool calls and token usage? | `skill_efficiency` (50%) + `token_efficiency` (50%) | + +- Dimension bands: PASS at 50% or above; NEUTRAL from 40% to below 50%; FAIL below 40%. +- Overall Tier 3 lift: PASS at +5 points or more; FAIL at -10 points or less; values between those bands are NEUTRAL. +- Overall verdict: PASS only when every configured dimension passes for at least one supported agent. Lift is reported as diagnostic evidence and does not override this gate. +- The 50% attempt pass threshold is a separate per-task gate; it is not the dimension pass threshold. +- Effectiveness is the equal-weight mean of goal completion (`goal_accuracy`) and expected workflow adherence (`behavior_check`). +- Efficiency is 50% tool-call productivity (the backward-compatible `skill_efficiency` wire id) and 50% `token_efficiency`. Positive-case skill routing is scored under Discoverability, not Efficiency; a negative case without a routing target is N/A. N/A sources are omitted, remaining weights are renormalized, and the dimension is marked partial. + +Signals present in this run: + +- `security` (Security): unsafe operations, secret leakage, and unauthorized access. +- `skill_execution` (Skill Execution): whether the expected skill was selected, decoys were avoided, and the workflow executed. +- `skill_efficiency` (Tool Productivity): tool-call productivity (legacy wire id; routing is scored under Discoverability). +- `accuracy` (Accuracy): final-answer correctness against the reference answer. +- `goal_accuracy` (Goal Accuracy): whether the user's goal was achieved. +- `behavior_check` (Behavior Check): whether the expected workflow behavior was followed. +- `token_efficiency` (Token Efficiency): actual uncached prompt plus completion usage (50% of Efficiency). + +
+ +## Freshness + +Regenerate this benchmark when the skill, evaluation dataset, target agent/model, evaluator version, environment, or scoring policy changes. diff --git a/skills/kermt-infer/skill-card.md b/skills/kermt-infer/skill-card.md index 4c4ea94..77e986a 100644 --- a/skills/kermt-infer/skill-card.md +++ b/skills/kermt-infer/skill-card.md @@ -1,74 +1,85 @@ ## Description:
-Runs molecular property predictions with a finetuned KERMT checkpoint on a SMILES-only CSV, validating that the checkpoint carries task FFN heads, preparing rdkit_2d features, and launching `main.py predict` inside the KERMT container.
+Run predictions with a finetuned KERMT checkpoint on a SMILES-only CSV.
This skill is ready for commercial/non-commercial use.
## Owner -NVIDIA (evax@nvidia.com)
+NVIDIA
### License/Terms of Use:
-Apache-2.0
- +Apache 2.0
## Use Case:
-Computational chemists and drug-discovery teams scoring a library of candidate molecules for ADMET or other molecular properties using a KERMT model they have already finetuned. Not for use with pretrain checkpoints — the skill refuses those and redirects to `kermt-finetune`.
- -### Requirements/Dependencies:
-Requires API Key or External Credential: No
-Credential Type(s): None
- -* `kermt-setup` completed (supplies the `kermt:latest` image)
-* Docker, NVIDIA Container Toolkit, CUDA-capable NVIDIA GPU
-* A finetuned KERMT checkpoint containing task FFN heads
-* A SMILES-only input CSV
- -Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
+Developers and computational chemists use this skill to run molecular property predictions on SMILES datasets using a finetuned KERMT checkpoint inside a containerized GPU environment.
### Deployment Geography for Use:
Global
-## Known Risks and Mitigations:
-Risk: Predictions are computational estimates and may be applied outside the chemical space the model was finetuned on, producing confidently wrong property values for novel scaffolds.
-Mitigation: Users must treat outputs as hypotheses requiring experimental validation, and should check that input chemistry resembles the finetuning set before acting on predictions.
- -Risk: Supplying a pretrain checkpoint instead of a finetuned one would yield meaningless outputs.
-Mitigation: The skill validates the checkpoint for task FFN heads before running and refuses pretrain checkpoints with an explicit redirect to `kermt-finetune`.
+## Requirements / Dependencies:
+**Requires API Key or External Credential:** [Not Specified]
+**Credential Type(s):** [None identified]
-Risk: Malformed or non-parseable SMILES silently reduce the effective prediction set.
-Mitigation: The skill runs a data-preparation and cleaning step that reports invalid rows before inference.
+Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
-Risk: The skill writes prediction outputs into a user-specified directory and can overwrite prior results.
-Mitigation: Outputs are written under an explicit output path supplied by the user; no files outside that path are modified.
+## Known Risks and Mitigations:
+Risk: Review before execution as proposals could introduce incorrect or misleading guidance into skills.
+Mitigation: Review and scan skill before deployment.
## Reference(s):
-- [KERMT repository](https://github.com/NVIDIA-BioNeMo/KERMT)
-- [RDKit documentation](https://www.rdkit.org/docs/) — `rdkit_2d` descriptor featurization
-- Related skills: `kermt-finetune` (produces the required checkpoint), `kermt-embed`
+- [KERMT paper — Multitask finetuning and acceleration of chemical pretrained models](https://arxiv.org/abs/2510.12719)
+- [GROVER paper — Self-supervised message passing transformer on large-scale molecular data](https://arxiv.org/abs/2007.02835)
+- [cuik-molmaker — GPU-accelerated molecular featurization](https://github.com/NVIDIA-Digital-Bio/cuik-molmaker)
+- [GROVER original implementation](https://github.com/tencent-ailab/grover)
+ ## Skill Output:
-**Output Type(s):** [Analysis, Files]
-**Output Format:** [CSV of per-molecule predictions; Markdown summary]
-**Output Parameters:** [2D — one row per valid input molecule after cleaning and canonicalization, one column per predicted task]
-**Other Properties Related to Output:** [Predictions are model estimates, not measurements, and are not calibrated probabilities of experimental outcome. Runs blocking, on a minutes-scale timeframe.]
+**Output Type(s):** [Shell commands, Files]
+**Output Format:** [Markdown with inline bash code blocks]
+**Output Parameters:** [1D]
+**Other Properties Related to Output:** [Predictions CSV with SMILES and per-target prediction columns; run manifest JSON with replay command]
## Evaluation Agents Used:
-Target agents: `claude-code`, `codex`. NVSkills-Eval has not yet been run against this skill — see Evaluation Results.
+- Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`)
+- Codex (`openai/openai/gpt-5.5`)
+ + ## Evaluation Tasks:
-4 evaluation tasks defined in `evals/evals.json`, covering checkpoint validation, CSV validation, feature preparation, and the predict invocation.
+4 evaluation tasks (3 positive, 1 negative), each attempted 3 times in isolated sandbox pods.
## Evaluation Metrics Used:
-Planned NVSkills-Eval dimensions: Security, Correctness, Discoverability, Effectiveness, Efficiency.
+Reported benchmark dimensions:
+- Security: Checks for unsafe operations, secret leakage, and unauthorized access.
+- Correctness: Checks final-answer correctness against reference answers.
+- Discoverability: Checks whether the expected skill was selected, decoys were avoided, and the workflow executed.
+- Effectiveness: Checks whether the user's goal was achieved and the expected workflow behavior was followed (equal-weight mean of goal_accuracy and behavior_check).
+- Efficiency: Checks tool-call productivity and token usage efficiency.
+ +Underlying evaluation signals used in this run:
+- `security`: Unsafe operations, secret leakage, and unauthorized access.
+- `skill_execution`: Whether the expected skill was selected, decoys were avoided, and the workflow executed.
+- `accuracy`: Final-answer correctness against the reference answer.
+- `goal_accuracy`: Whether the user's goal was achieved.
+- `behavior_check`: Whether the expected workflow behavior was followed.
+- `skill_efficiency`: Tool-call productivity (routing scored under Discoverability, not Efficiency).
+- `token_efficiency`: Actual uncached prompt plus completion token usage.
+ + ## Evaluation Results:
-Pending. NVSkills-Eval has not been run for this skill; results and a `BENCHMARK.md` will be published when the evaluation pipeline runs.
+| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 78.6% | 80.3% | +| Security | 100.0% → 100.0% (±0.0 points) | 70.0% → 100.0% (+30.0 points) | +| Correctness | 25.0% → 85.0% (+60.0 points) | 50.0% → 90.0% (+40.0 points) | +| Discoverability | 91.7% | 91.7% | +| Effectiveness | 25.9% → 35.0% (+9.1 points) | 22.3% → 34.4% (+12.1 points) | +| Efficiency | 81.5% | 85.4% | ## Skill Version(s):
-b1c082c (source: git SHA, committed 2026-07-17)
+77111e0 (source: git SHA, committed 2026-09-09)
## Ethical Considerations:
NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their internal team to ensure this skill meets requirements for the relevant industry and use case and addresses unforeseen product misuse.
-Molecular property predictions produced by this skill are computational hypotheses for research and development workflows. They must not be used as the sole basis for clinical, safety, or regulatory decisions.
- (For Release on NVIDIA Platforms Only)
-Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://www.nvidia.com/en-us/support/submit-security-vulnerability/).
+Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://app.intigriti.com/programs/nvidia/nvidiavdp/detail).
diff --git a/skills/kermt-infer/skill.oms.sig b/skills/kermt-infer/skill.oms.sig new file mode 100644 index 0000000..92dc2ec --- /dev/null +++ b/skills/kermt-infer/skill.oms.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAia2VybXQtaW5mZXIiLAogICAgICAiZGlnZXN0IjogewogICAgICAgICJzaGEyNTYiOiAiYWFkM2Q3MGM5ODE0MzcxNzdlYmVmNWI5YjI2OWU3NmY0NGI2M2I4ZGRiNTZkOWZlZjFjODZiYTE4MjJiODEwNCIKICAgICAgfQogICAgfQogIF0sCiAgInByZWRpY2F0ZVR5cGUiOiAiaHR0cHM6Ly9tb2RlbF9zaWduaW5nL3NpZ25hdHVyZS92MS4wIiwKICAicHJlZGljYXRlIjogewogICAgInJlc291cmNlcyI6IFsKICAgICAgewogICAgICAgICJuYW1lIjogIkJFTkNITUFSSy5tZCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiMzQ0MmQxM2ZjMzY0YzZjZmZlMzlhYTE5OGYwNTlmOTRiMTliMTFiZDRmZGEzNzM4MmEwNzc1NmRiNWMxODk4ZCIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogIlNLSUxMLm1kIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICJjMDE4ZmJkNGJlYjY5N2M4OGNmZDMyNWYyODE1M2U3Yjc3YzliYzViZDk3MzNjMzI2ZThjMzY4NWIxNGUyZjM4IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAiY29uZmlnL2RlZmF1bHRzX2luZmVyZW5jZS5qc29uIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICI5MDJlYzVjZTVmYmMwNmNkMGJhMzZmZjFlMDAzZjA1YjEyY2U3YzQxMjE2ZDc2YTk0OTU5NWMyZjlhMWU2M2I2IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAiZXZhbHMvZXZhbHMuanNvbiIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiZTM4MWYwNDFkMGVmZTljODkwZDU1MzdjZTMzZDg2ZjczYzkxYTNmZDQ1ZWRkMmMwMzVmMGIyNzM0ZjRlZDI1MCIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInNjcmlwdHMvX3V0aWxzLnB5IiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICIwMjYyMjBhMjI2ZDhmODQxODY0ZjNmOGQ3NGRmYzdjZWVlNTJkOTg1ZDQ1YTllMTQ2ZTI3Y2MxNTIzOWZmNTQyIgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAic2NyaXB0cy9jaGVja19jaGVja3BvaW50LnB5IiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICIwYmM2Y2M4NzI5ZGIzYTNhOTM5ZmNlN2E5NWU1ZjZkMTY2YWY1NTE3ZmI2OTM0NmJkYTJjMzAxMTQ1YThmNTAyIgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAic2NyaXB0cy9jaGVja19kYXRhLnB5IiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICI2ODkyNjgxODczZDY2ZTdiZTQxMmI3NzMwYTM0OWU2NDA5MDljNzJhZmFlMjIyODU0ZjdhMDFiNjRiMmFhYjU2IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAic2NyaXB0cy9rZXJtdF9jb250YWluZXIuc2giLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogImZjZGFiNTk1YzBlODlkZjE1ODE5YTdlMjc0MWMzMWY1MTJkOTVlMDhmNjQ2MWY4MWQ1MzVlY2EzY2JhNDAyYjgiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJzY3JpcHRzL3ByZXBhcmVfZGF0YS5weSIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiMTQyNjI4OWFiODMzNGYxYTk5ZTliZjFhNDE3Njc5NTgyNGYyNzUxOWZmOTE0MjE2YTNmMDdmOTBlZWFhZjAzYiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInNjcmlwdHMvcnVuX2luZmVyZW5jZS5weSIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiMzNhZTUxYTRiOGUyMjU3NmQ2NWRkZmQyMmM0ZGJhYWY1Y2ZkOTc5MTIyMjBhMDBkMDViZjlkYzgwNTg5MWZiMCIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInNraWxsLWNhcmQubWQiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogImE1Y2FkY2I4NmIyZDZkYWJhZTgwYjFlYzgwNTI5YWNmYTJlOTZkZDc0NTU1ODRhZGIyMTcwNjk4MTQ3OTAzMmIiCiAgICAgIH0KICAgIF0sCiAgICAic2VyaWFsaXphdGlvbiI6IHsKICAgICAgImlnbm9yZV9wYXRocyI6IFsKICAgICAgICAiLmdpdGF0dHJpYnV0ZXMiLAogICAgICAgICIuZ2l0IiwKICAgICAgICAiLmdpdGh1YiIsCiAgICAgICAgIi5naXRpZ25vcmUiCiAgICAgIF0sCiAgICAgICJtZXRob2QiOiAiZmlsZXMiLAogICAgICAiYWxsb3dfc3ltbGlua3MiOiBmYWxzZSwKICAgICAgImhhc2hfdHlwZSI6ICJzaGEyNTYiCiAgICB9CiAgfQp9","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGUCMFhv1SajPH4l5QGC6IAVZ7RdX2DmN9cU+v1ZpYNErXCif0E0JTKEBdJCBij/iuQXtgIxANLxKn0K1Hs24mris65MgXZXKc95g3ihH8qKud5RzlC7w+9djPTpfYyD2TP3Ckdnow==","keyid":""}]}} \ No newline at end of file diff --git a/skills/kermt-monitor/BENCHMARK.md b/skills/kermt-monitor/BENCHMARK.md new file mode 100644 index 0000000..74453ce --- /dev/null +++ b/skills/kermt-monitor/BENCHMARK.md @@ -0,0 +1,127 @@ +# Skill Benchmark: kermt-monitor + +> ✅ **Overall verdict: PASS — Recommended for publication** + +## Publication Recommendation + +Recommended for publication based on the completed evaluation evidence in this report. + +## Evaluation Metadata + +- Skill: `kermt-monitor` +- Evaluation date: 2026-09-15 +- Evaluator version: `1.5.6` +- Agents: Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`), Codex (`openai/openai/gpt-5.5`) +- Tasks: 4 evaluation tasks (3 positive, 1 negative) +- Dataset digest: `sha256:07abc6b8571fbe0834b32b54b3b59557850ed1cf086115b3a90f60a77bf55546` (skill-evaluator-dataset-snapshot/1) +- Attempts per task: 3 +- Environment: `k8s-sandbox` +- Tier 2 evidence: required for publication +- Tier 3 evidence: required for publication + +Each task attempt ran in its own isolated sandbox pod. + +## What This Report Answers + +The three-tier evaluation checks whether the skill: + +- is safe to use; +- produces correct answers; +- is discovered and activated when needed; +- helps the agent complete the user's goal and expected workflow; and +- avoids wasted skill and tool usage. + +## Results at a Glance + +| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 92.6% — baseline ran, but no comparable score was available; uplift unavailable | 76.5% — baseline ran, but no comparable score was available; uplift unavailable | +| Security | 100.0% → 100.0% (±0.0 points) | 50.0% → 75.0% (+25.0 points) | +| Correctness | 28.0% → 100.0% (+72.0 points) | 32.0% → 70.0% (+38.0 points) | +| Discoverability | 96.7% — baseline ran, but no comparable score was available; uplift unavailable | 91.7% — baseline ran, but no comparable score was available; uplift unavailable | +| Effectiveness | 31.5% → 79.4% (+47.9 points) | 27.5% → 51.3% (+23.8 points) | +| Efficiency | 86.8% — baseline ran, but no comparable score was available; uplift unavailable | 94.8% — baseline ran, but no comparable score was available; uplift unavailable | + +**How to read this table:** baseline is the same task attempted without the target skill. Scores are rounded to one decimal; threshold-adjacent values use additional precision so their displayed band matches the verdict. Uplift is derived from those displayed scores and shown in percentage points. + +Example: `47.0% → 92.0% (+45.0 points)` means the skill-assisted run scored 92.0%, 45.0 percentage points above its 47.0% no-skill baseline. + +A partial dimension was calculated from only the available configured signals; review the detailed report before relying on it. + +## Token Usage + +Actual Tier 3 execution usage is reported for every observed agent/case pair and both conditions. + +| Agent | Dataset case | With skill | Without skill | Delta | Change | Coverage | +|---|---|---:|---:|---:|---:|---| +| claude-code | All cases | 1,014,580 | 1,616,571 | N/A | N/A | skill 4/4; base 10/10 | +| claude-code | kermt-monitor-001 | 196,682 | 485,938 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-monitor-002 | 198,213 | 456,649 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-monitor-003 | 198,961 | 549,308 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-monitor-004 | 420,724 | 124,676 | +296,048 | +237.45% | skill 1/1; base 1/1 | +| codex | All cases | 723,127 | 873,801 | N/A | N/A | skill 4/4; base 10/10 | +| codex | kermt-monitor-001 | 97,932 | 307,119 | N/A | N/A | skill 1/1; base 3/3 | +| codex | kermt-monitor-002 | 78,415 | 161,602 | N/A | N/A | skill 1/1; base 3/3 | +| codex | kermt-monitor-003 | 98,169 | 336,835 | N/A | N/A | skill 1/1; base 3/3 | +| codex | kermt-monitor-004 | 448,611 | 68,245 | +380,366 | +557.35% | skill 1/1; base 1/1 | +| ALL AGENTS | Dataset aggregate | 1,737,707 | 2,490,372 | N/A | N/A | skill 8/8; base 20/20 | + +Prompt tokens include cached reads, so total tokens are `prompt + completion` (cached is not added twice). The Efficiency score uses `(prompt - cached) + completion`. N/A means the relevant trajectory counters were not available; coverage is never estimated. + +## Tier Status + +| Tier | Purpose | Status | Evidence | +|---|---|---|---| +| Tier 1 | Static validation | **PASSED WITH OBSERVATIONS** | 11 validator(s); 13 finding(s) | +| Tier 2 | Semantic deduplication | **PASSED** | 2 validator(s); 0 finding(s) | +| Tier 3 | Live agent evaluation | **PASS** | 2 agent(s); 4 task(s) | + +## Findings and Observations + +
+Show detailed findings and successful checks + +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.author' (`skills/kermt-monitor/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.tags' (`skills/kermt-monitor/SKILL.md`) +- **MEDIUM** SCHEMA/metadata_key_style: Metadata key 'risk_tier' is not kebab-case (`skills/kermt-monitor/SKILL.md`) +- **MEDIUM** SCHEMA/body_recommended_section: Missing recommended section: '## Instructions' (`skills/kermt-monitor/SKILL.md`) +- **MEDIUM** SCHEMA/body_recommended_section: Missing recommended section: '## Examples' (`skills/kermt-monitor/SKILL.md`) +- 8 additional finding(s) are available in the full evaluation artifacts. + +
+ +## Scoring Methodology + +
+Show dimension definitions, source signals, and thresholds + +| Dimension | Question | Scored signals | +|---|---|---| +| Security | Is it safe to use? | `security` (100%) | +| Correctness | Is the answer correct? | `accuracy` (100%) | +| Discoverability | Was the right skill loaded when needed? | `skill_execution` (100%) | +| Effectiveness | Did the skill help complete the task? | `goal_accuracy` (50%) + `behavior_check` (50%) | +| Efficiency | Did it avoid wasted tool calls and token usage? | `skill_efficiency` (50%) + `token_efficiency` (50%) | + +- Dimension bands: PASS at 50% or above; NEUTRAL from 40% to below 50%; FAIL below 40%. +- Overall Tier 3 lift: PASS at +5 points or more; FAIL at -10 points or less; values between those bands are NEUTRAL. +- Overall verdict: PASS only when every configured dimension passes for at least one supported agent. Lift is reported as diagnostic evidence and does not override this gate. +- The 50% attempt pass threshold is a separate per-task gate; it is not the dimension pass threshold. +- Effectiveness is the equal-weight mean of goal completion (`goal_accuracy`) and expected workflow adherence (`behavior_check`). +- Efficiency is 50% tool-call productivity (the backward-compatible `skill_efficiency` wire id) and 50% `token_efficiency`. Positive-case skill routing is scored under Discoverability, not Efficiency; a negative case without a routing target is N/A. N/A sources are omitted, remaining weights are renormalized, and the dimension is marked partial. + +Signals present in this run: + +- `security` (Security): unsafe operations, secret leakage, and unauthorized access. +- `skill_execution` (Skill Execution): whether the expected skill was selected, decoys were avoided, and the workflow executed. +- `skill_efficiency` (Tool Productivity): tool-call productivity (legacy wire id; routing is scored under Discoverability). +- `accuracy` (Accuracy): final-answer correctness against the reference answer. +- `goal_accuracy` (Goal Accuracy): whether the user's goal was achieved. +- `behavior_check` (Behavior Check): whether the expected workflow behavior was followed. +- `token_efficiency` (Token Efficiency): actual uncached prompt plus completion usage (50% of Efficiency). + +
+ +## Freshness + +Regenerate this benchmark when the skill, evaluation dataset, target agent/model, evaluator version, environment, or scoring policy changes. diff --git a/skills/kermt-monitor/skill-card.md b/skills/kermt-monitor/skill-card.md index 9f54dc1..d19c9b5 100644 --- a/skills/kermt-monitor/skill-card.md +++ b/skills/kermt-monitor/skill-card.md @@ -1,69 +1,83 @@ ## Description:
-Reports progress for a detached KERMT run by reading `run.json`, querying Docker for container state, tailing the pretrain/finetune log, and parsing progress lines (epoch, step, validation loss).
+Check progress for a detached KERMT run (pretrain, finetune, or any kermt_run_detached invocation). Reads run.json, queries docker for container state, tails the pretrain/finetune log, and parses progress lines (epoch, step, val loss).
This skill is ready for commercial/non-commercial use.
## Owner -NVIDIA (evax@nvidia.com)
+NVIDIA
### License/Terms of Use:
-Apache-2.0
- +Apache 2.0
## Use Case:
-ML engineers running long KERMT pretraining or finetuning jobs who need to check status, spot divergence early, and decide whether to let a run continue or terminate it. This is the companion skill to every detached `kermt-*` training invocation.
- -### Requirements/Dependencies:
-Requires API Key or External Credential: No
-Credential Type(s): None
- -* Docker
-* `jq`
-* A prior detached KERMT run that produced a `run.json`
- -Note: unlike the training skills, this skill does not require a GPU.
- -Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
+Developers and engineers monitoring detached KERMT training and finetuning runs to check container state, training progress, and metrics without interrupting the running job.
### Deployment Geography for Use:
Global
-## Known Risks and Mitigations:
-Risk: A stale or missing `run.json` can cause the skill to report on the wrong run, or to report nothing while a job is in fact still consuming GPU-hours.
-Mitigation: The skill refuses to proceed when `run.json` is missing, and reports container identifiers so the user can verify the run directly. Because `run.json` does not record a container name, pass `--container ` when monitoring a specific run.
+## Requirements / Dependencies:
+**Requires API Key or External Credential:** [Not Specified]
+**Credential Type(s):** [None identified]
-Risk: Log tailing surfaces run output into the agent transcript, which may include file paths or environment details.
-Mitigation: The skill reads only the pretrain/finetune log; users should treat agent transcripts as sensitive and avoid pasting credentials into run configuration.
+Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
-Risk: This skill is read-only and cannot stop a runaway job, so a user may assume monitoring implies control.
-Mitigation: The skill reports container identifiers so the user can terminate the run directly via Docker if needed.
+## Known Risks and Mitigations:
+Risk: Review before execution as proposals could introduce incorrect or misleading guidance into skills.
+Mitigation: Review and scan skill before deployment.
## Reference(s):
-- [KERMT repository](https://github.com/NVIDIA-BioNeMo/KERMT)
-- Related skills: `kermt-pretrain-scratch`, `kermt-continue-pretrain`, `kermt-finetune`
+- [KERMT: Multitask finetuning and acceleration of chemical pretrained models](https://arxiv.org/abs/2510.12719)
+- [GROVER: Self-Supervised Message Passing Transformer on Large-Scale Molecular Data](https://arxiv.org/abs/2007.02835)
+ ## Skill Output:
-**Output Type(s):** [Analysis]
-**Output Format:** [Markdown status report]
-**Output Parameters:** [1D — container state, current epoch, current step, latest validation loss]
-**Other Properties Related to Output:** [Read-only; the skill makes no changes to the run or the host]
+**Output Type(s):** [Shell commands, Analysis]
+**Output Format:** [Markdown with inline bash code blocks]
+**Output Parameters:** [1D]
+**Other Properties Related to Output:** [Supports --json flag for structured JSON output]
## Evaluation Agents Used:
-Target agents: `claude-code`, `codex`. NVSkills-Eval has not yet been run against this skill — see Evaluation Results.
+- Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`)
+- Codex (`openai/openai/gpt-5.5`)
+ + ## Evaluation Tasks:
-4 evaluation tasks defined in `evals/evals.json`, covering run discovery, container-state reporting, and log progress parsing.
+4 evaluation tasks (3 positive, 1 negative), each with 3 attempts per task in isolated k8s-sandbox pods.
## Evaluation Metrics Used:
-Planned NVSkills-Eval dimensions: Security, Correctness, Discoverability, Effectiveness, Efficiency.
+Reported benchmark dimensions:
+- Security: Is it safe to use? Checks for unsafe operations, secret leakage, and unauthorized access.
+- Correctness: Is the answer correct? Checks final-answer correctness against the reference answer.
+- Discoverability: Was the right skill loaded when needed? Checks whether the expected skill was selected and the workflow executed.
+- Effectiveness: Did the skill help complete the task? Equal-weight mean of goal completion and expected workflow adherence.
+- Efficiency: Did it avoid wasted tool calls and token usage? 50% tool-call productivity and 50% token efficiency.
+ +Underlying evaluation signals used in this run:
+- `security`: Checks for unsafe operations, secret leakage, and unauthorized access.
+- `accuracy`: Checks final-answer correctness against the reference answer.
+- `skill_execution`: Checks whether the expected skill was selected, decoys were avoided, and the workflow executed.
+- `goal_accuracy`: Checks whether the user's goal was achieved.
+- `behavior_check`: Checks whether the expected workflow behavior was followed.
+- `skill_efficiency`: Measures tool-call productivity.
+- `token_efficiency`: Measures actual uncached prompt plus completion usage.
+ + ## Evaluation Results:
-Pending. NVSkills-Eval has not been run for this skill; results and a `BENCHMARK.md` will be published when the evaluation pipeline runs.
+| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 92.6% | 76.5% | +| Security | 100.0% → 100.0% (±0.0 points) | 50.0% → 75.0% (+25.0 points) | +| Correctness | 28.0% → 100.0% (+72.0 points) | 32.0% → 70.0% (+38.0 points) | +| Discoverability | 96.7% | 91.7% | +| Effectiveness | 31.5% → 79.4% (+47.9 points) | 27.5% → 51.3% (+23.8 points) | +| Efficiency | 86.8% | 94.8% | ## Skill Version(s):
-b1c082c (source: git SHA, committed 2026-07-17)
+77111e0 (source: git SHA, committed 2026-09-09)
## Ethical Considerations:
NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their internal team to ensure this skill meets requirements for the relevant industry and use case and addresses unforeseen product misuse.
(For Release on NVIDIA Platforms Only)
-Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://www.nvidia.com/en-us/support/submit-security-vulnerability/).
+Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://app.intigriti.com/programs/nvidia/nvidiavdp/detail).
diff --git a/skills/kermt-monitor/skill.oms.sig b/skills/kermt-monitor/skill.oms.sig new file mode 100644 index 0000000..2ab9364 --- /dev/null +++ b/skills/kermt-monitor/skill.oms.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAia2VybXQtbW9uaXRvciIsCiAgICAgICJkaWdlc3QiOiB7CiAgICAgICAgInNoYTI1NiI6ICIzYzcyMWJhMzczM2U1N2ViMmM0YmRhMjJmZWNlMTlhMmQ1ODQzNDYxMjdhOWNhZmQzYjUxM2VhNzMxNWExMDQwIgogICAgICB9CiAgICB9CiAgXSwKICAicHJlZGljYXRlVHlwZSI6ICJodHRwczovL21vZGVsX3NpZ25pbmcvc2lnbmF0dXJlL3YxLjAiLAogICJwcmVkaWNhdGUiOiB7CiAgICAic2VyaWFsaXphdGlvbiI6IHsKICAgICAgImlnbm9yZV9wYXRocyI6IFsKICAgICAgICAiLmdpdCIsCiAgICAgICAgIi5naXRodWIiLAogICAgICAgICIuZ2l0YXR0cmlidXRlcyIsCiAgICAgICAgIi5naXRpZ25vcmUiCiAgICAgIF0sCiAgICAgICJtZXRob2QiOiAiZmlsZXMiLAogICAgICAiaGFzaF90eXBlIjogInNoYTI1NiIsCiAgICAgICJhbGxvd19zeW1saW5rcyI6IGZhbHNlCiAgICB9LAogICAgInJlc291cmNlcyI6IFsKICAgICAgewogICAgICAgICJuYW1lIjogIkJFTkNITUFSSy5tZCIsCiAgICAgICAgImRpZ2VzdCI6ICJhNjU4YWU4YWQ4NTE5ZDhkOWUwODg1NDg5NDc0MmE0ZmY4Mzg1NzJjMTMzYTUxOTEzZmY0ZDc1NGZlNGEyMmU3IiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogIlNLSUxMLm1kIiwKICAgICAgICAiZGlnZXN0IjogImMyZGEyZDkyMGFkOGQyNGQyZDFmOWM5ZjU2Yzc2OGI3YThiZmM3MzJmMTdkNmViZjI2NGJiNTIwMDdhNGNlNTIiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAiZXZhbHMvZXZhbHMuanNvbiIsCiAgICAgICAgImRpZ2VzdCI6ICIzMDU0YWU2NzE4NDg2YWZiMmQ5NTU4ZDljOTVjOWQwZWZjZWMxNDRiZWU3NWZjYjc0YzVlYjM4OWRlOTJjZmY5IiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInNraWxsLWNhcmQubWQiLAogICAgICAgICJkaWdlc3QiOiAiOGEzODI1ZWZkZWY1ODY2Y2Q2OThjNTA5OWVmZmU2OGM1YjZlYjRmYmYwMTlhNjcwNmM3NTUyMzk1ZjQzYjBhZSIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0KICAgIF0KICB9Cn0=","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGQCMHLOCIqssmhkFEW60z8Oet2krkYW2wnyipgbCySx1zPYIR00ioo6iP/32E2IPKyzOAIwadR3wA9AUE64R/q0DMUDhs3KtkCuiK3dY9957vZDzQI3oTHjeJKe0N3IyOcq2ZoL","keyid":""}]}} \ No newline at end of file diff --git a/skills/kermt-pretrain-scratch/BENCHMARK.md b/skills/kermt-pretrain-scratch/BENCHMARK.md new file mode 100644 index 0000000..dff8f9a --- /dev/null +++ b/skills/kermt-pretrain-scratch/BENCHMARK.md @@ -0,0 +1,127 @@ +# Skill Benchmark: kermt-pretrain-scratch + +> ✅ **Overall verdict: PASS — Recommended for publication** + +## Publication Recommendation + +Recommended for publication based on the completed evaluation evidence in this report. + +## Evaluation Metadata + +- Skill: `kermt-pretrain-scratch` +- Evaluation date: 2026-09-15 +- Evaluator version: `1.5.6` +- Agents: Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`), Codex (`openai/openai/gpt-5.5`) +- Tasks: 4 evaluation tasks (3 positive, 1 negative) +- Dataset digest: `sha256:9e8d2c01c27435b23630efc8c43be8af9a181db81ff28ec5b78b169ec8877548` (skill-evaluator-dataset-snapshot/1) +- Attempts per task: 3 +- Environment: `k8s-sandbox` +- Tier 2 evidence: required for publication +- Tier 3 evidence: required for publication + +Each task attempt ran in its own isolated sandbox pod. + +## What This Report Answers + +The three-tier evaluation checks whether the skill: + +- is safe to use; +- produces correct answers; +- is discovered and activated when needed; +- helps the agent complete the user's goal and expected workflow; and +- avoids wasted skill and tool usage. + +## Results at a Glance + +| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 79.9% — baseline ran, but no comparable score was available; uplift unavailable | 80.5% — baseline ran, but no comparable score was available; uplift unavailable | +| Security | 100.0% → 75.0% (-25.0 points) | 50.0% → 100.0% (+50.0 points) | +| Correctness | 6.7% → 90.0% (+83.3 points) | 91.4% → 90.0% (-1.4 points) | +| Discoverability | 93.3% — baseline ran, but no comparable score was available; uplift unavailable | 88.3% — baseline ran, but no comparable score was available; uplift unavailable | +| Effectiveness | 20.0% → 55.0% (+35.0 points) | 30.4% → 48.1% (+17.7 points) | +| Efficiency | 86.2% — baseline ran, but no comparable score was available; uplift unavailable | 76.0% — baseline ran, but no comparable score was available; uplift unavailable | + +**How to read this table:** baseline is the same task attempted without the target skill. Scores are rounded to one decimal; threshold-adjacent values use additional precision so their displayed band matches the verdict. Uplift is derived from those displayed scores and shown in percentage points. + +Example: `47.0% → 92.0% (+45.0 points)` means the skill-assisted run scored 92.0%, 45.0 percentage points above its 47.0% no-skill baseline. + +A partial dimension was calculated from only the available configured signals; review the detailed report before relying on it. + +## Token Usage + +Actual Tier 3 execution usage is reported for every observed agent/case pair and both conditions. + +| Agent | Dataset case | With skill | Without skill | Delta | Change | Coverage | +|---|---|---:|---:|---:|---:|---| +| claude-code | All cases | 1,193,344 | 1,358,903 | N/A | N/A | skill 4/4; base 9/9 | +| claude-code | kermt-pretrain-scratch-001 | 325,016 | 640,126 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-pretrain-scratch-002 | 276,057 | 296,321 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-pretrain-scratch-003 | 309,523 | 293,817 | N/A | N/A | skill 1/1; base 2/2 | +| claude-code | kermt-pretrain-scratch-004 | 282,748 | 128,639 | +154,109 | +119.80% | skill 1/1; base 1/1 | +| codex | All cases | 1,472,593 | 8,395,936 | N/A | N/A | skill 4/4; base 7/7 | +| codex | kermt-pretrain-scratch-001 | 111,468 | 2,676,322 | N/A | N/A | skill 1/1; base 3/3 | +| codex | kermt-pretrain-scratch-002 | 197,440 | 412,299 | -214,859 | -52.11% | skill 1/1; base 1/1 | +| codex | kermt-pretrain-scratch-003 | 487,422 | 4,423,679 | N/A | N/A | skill 1/1; base 2/2 | +| codex | kermt-pretrain-scratch-004 | 676,263 | 883,636 | -207,373 | -23.47% | skill 1/1; base 1/1 | +| ALL AGENTS | Dataset aggregate | 2,665,937 | 9,754,839 | N/A | N/A | skill 8/8; base 16/16 | + +Prompt tokens include cached reads, so total tokens are `prompt + completion` (cached is not added twice). The Efficiency score uses `(prompt - cached) + completion`. N/A means the relevant trajectory counters were not available; coverage is never estimated. + +## Tier Status + +| Tier | Purpose | Status | Evidence | +|---|---|---|---| +| Tier 1 | Static validation | **PASSED WITH OBSERVATIONS** | 11 validator(s); 63 finding(s) | +| Tier 2 | Semantic deduplication | **PASSED** | 2 validator(s); 0 finding(s) | +| Tier 3 | Live agent evaluation | **PASS** | 2 agent(s); 4 task(s) | + +## Findings and Observations + +
+Show detailed findings and successful checks + +- **MEDIUM** QUALITY/quality_correctness: No documented scripts in table format (`skills/kermt-pretrain-scratch/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: Instructions don't mention 'run_script' (`skills/kermt-pretrain-scratch/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.author' (`skills/kermt-pretrain-scratch/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.tags' (`skills/kermt-pretrain-scratch/SKILL.md`) +- **MEDIUM** SCHEMA/metadata_key_style: Metadata key 'risk_tier' is not kebab-case (`skills/kermt-pretrain-scratch/SKILL.md`) +- 58 additional finding(s) are available in the full evaluation artifacts. + +
+ +## Scoring Methodology + +
+Show dimension definitions, source signals, and thresholds + +| Dimension | Question | Scored signals | +|---|---|---| +| Security | Is it safe to use? | `security` (100%) | +| Correctness | Is the answer correct? | `accuracy` (100%) | +| Discoverability | Was the right skill loaded when needed? | `skill_execution` (100%) | +| Effectiveness | Did the skill help complete the task? | `goal_accuracy` (50%) + `behavior_check` (50%) | +| Efficiency | Did it avoid wasted tool calls and token usage? | `skill_efficiency` (50%) + `token_efficiency` (50%) | + +- Dimension bands: PASS at 50% or above; NEUTRAL from 40% to below 50%; FAIL below 40%. +- Overall Tier 3 lift: PASS at +5 points or more; FAIL at -10 points or less; values between those bands are NEUTRAL. +- Overall verdict: PASS only when every configured dimension passes for at least one supported agent. Lift is reported as diagnostic evidence and does not override this gate. +- The 50% attempt pass threshold is a separate per-task gate; it is not the dimension pass threshold. +- Effectiveness is the equal-weight mean of goal completion (`goal_accuracy`) and expected workflow adherence (`behavior_check`). +- Efficiency is 50% tool-call productivity (the backward-compatible `skill_efficiency` wire id) and 50% `token_efficiency`. Positive-case skill routing is scored under Discoverability, not Efficiency; a negative case without a routing target is N/A. N/A sources are omitted, remaining weights are renormalized, and the dimension is marked partial. + +Signals present in this run: + +- `security` (Security): unsafe operations, secret leakage, and unauthorized access. +- `skill_execution` (Skill Execution): whether the expected skill was selected, decoys were avoided, and the workflow executed. +- `skill_efficiency` (Tool Productivity): tool-call productivity (legacy wire id; routing is scored under Discoverability). +- `accuracy` (Accuracy): final-answer correctness against the reference answer. +- `goal_accuracy` (Goal Accuracy): whether the user's goal was achieved. +- `behavior_check` (Behavior Check): whether the expected workflow behavior was followed. +- `token_efficiency` (Token Efficiency): actual uncached prompt plus completion usage (50% of Efficiency). + +
+ +## Freshness + +Regenerate this benchmark when the skill, evaluation dataset, target agent/model, evaluator version, environment, or scoring policy changes. diff --git a/skills/kermt-pretrain-scratch/skill-card.md b/skills/kermt-pretrain-scratch/skill-card.md index 46eef3b..cf01f6f 100644 --- a/skills/kermt-pretrain-scratch/skill-card.md +++ b/skills/kermt-pretrain-scratch/skill-card.md @@ -1,77 +1,84 @@ ## Description:
-Pretrains a fresh KERMT model from scratch on a user-provided corpus — building a new vocabulary, instantiating the model architecture from defaults, and launching `pretrain_ddp.py` inside the KERMT container with no starting checkpoint loaded.
+Pretrain a fresh KERMT model from scratch on a user-provided corpus, building a new vocabulary from the corpus, instantiating the model architecture from defaults, and launching pretrain_ddp.py inside the kermt container.
This skill is ready for commercial/non-commercial use.
## Owner -NVIDIA (evax@nvidia.com)
+NVIDIA
### License/Terms of Use:
-Apache-2.0
- +Apache 2.0
## Use Case:
-ML research engineers training a KERMT model from random initialization on a corpus sufficiently large and distinct that continued pretraining from an existing checkpoint is not appropriate. For most users, `kermt-continue-pretrain` is the better starting point.
- -### Requirements/Dependencies:
-Requires API Key or External Credential: Optional
-Credential Type(s): API key — `WANDB_API_KEY` for optional Weights & Biases run tracking
- -* `kermt-setup` completed (supplies the `kermt:latest` image)
-* Docker, NVIDIA Container Toolkit, CUDA-capable NVIDIA GPU (multi-GPU strongly recommended; DDP supported)
-* A large pretraining corpus CSV
-* Substantial disk for shard, vocabulary, and feature artifacts
- -Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
+Developers and engineers who need to pretrain a new KERMT molecular property prediction model from scratch on a custom chemistry corpus, rather than continuing from an existing checkpoint.
### Deployment Geography for Use:
Global
-## Known Risks and Mitigations:
-Risk: This is the most expensive skill in the KERMT family — training from random initialization can consume days to weeks of multi-GPU time, and a single agent instruction can start it.
-Mitigation: Runs launch detached with a run manifest; `kermt-monitor` provides progress visibility and the container identifiers needed to terminate early. Users should confirm corpus scale justifies from-scratch training before invoking.
- -Risk: Training from scratch on an insufficiently large corpus produces a model materially worse than the available pretrained checkpoints, with the cost discovered only after the run.
-Mitigation: The skill is documented as the exception path, with `kermt-continue-pretrain` as the recommended default; users should benchmark against an existing checkpoint before committing.
+## Requirements / Dependencies:
+**Requires API Key or External Credential:** [Not Specified]
+**Credential Type(s):** [None identified]
-Risk: A vocabulary built from the user's corpus is not interchangeable with vocabularies from other checkpoints, so resulting models are incompatible with checkpoints trained elsewhere.
-Mitigation: The new vocabulary is written alongside the checkpoint in the run directory, making the pairing explicit and auditable.
- -Risk: Data preparation writes large shard/vocab/feature artifacts and can overwrite prior preparation output.
-Mitigation: Preparation writes under an explicit run/output path supplied by the user.
+Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
-Risk: When Weights & Biases tracking is enabled, run metadata is transmitted to a third-party service.
-Mitigation: W&B tracking is optional and off unless the user supplies `WANDB_API_KEY`.
+## Known Risks and Mitigations:
+Risk: Review before execution as proposals could introduce incorrect or misleading guidance into skills.
+Mitigation: Review and scan skill before deployment.
## Reference(s):
-- [KERMT repository](https://github.com/NVIDIA-BioNeMo/KERMT)
-- `scripts/run_pretrain_local.py` — extended usage examples
-- Related skills: `kermt-setup`, `kermt-monitor`, `kermt-continue-pretrain`, `kermt-finetune`
+- [KERMT paper — Multitask finetuning and acceleration of chemical pretrained models](https://arxiv.org/abs/2510.12719)
+- [GROVER paper — Self-Supervised Graph Transformer on Large-Scale Molecular Data](https://arxiv.org/abs/2007.02835)
+- [cuik-molmaker — GPU-accelerated molecular featurization](https://github.com/NVIDIA-Digital-Bio/cuik-molmaker)
+ ## Skill Output:
-**Output Type(s):** [Files, Analysis]
-**Output Format:** [Model checkpoint files; new vocabulary; shard/feature artifacts; training logs; `run.json` manifest; Markdown launch summary]
-**Output Parameters:** [1D — run identifier, container id, vocabulary path, model configuration, output paths]
-**Other Properties Related to Output:** [Detached execution: the skill returns after launch, not after training completes.]
+**Output Type(s):** [Shell commands, Configuration instructions]
+**Output Format:** [Markdown with inline bash code blocks]
+**Output Parameters:** [1D]
+**Other Properties Related to Output:** [None]
## Evaluation Agents Used:
-Target agents: `claude-code`, `codex`. NVSkills-Eval has not yet been run against this skill — see Evaluation Results.
+- Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`)
+- Codex (`openai/openai/gpt-5.5`)
+ + ## Evaluation Tasks:
-4 evaluation tasks defined in `evals/evals.json`, covering corpus validation, vocabulary construction, architecture instantiation, and detached launch.
+Evaluated against 4 internal evaluation tasks (3 positive, 1 negative) with 3 attempts per task in isolated k8s-sandbox pods.
## Evaluation Metrics Used:
-Planned NVSkills-Eval dimensions: Security, Correctness, Discoverability, Effectiveness, Efficiency.
+Reported benchmark dimensions:
+- Security: Checks whether the skill is safe to use, covering unsafe operations, secret leakage, and unauthorized access.
+- Correctness: Checks whether the skill produces correct answers against the reference answer.
+- Discoverability: Checks whether the right skill was selected when needed, decoys were avoided, and the workflow executed.
+- Effectiveness: Checks whether the skill helped complete the user's goal (goal_accuracy 50%) and followed expected workflow behavior (behavior_check 50%).
+- Efficiency: Checks whether the skill avoided wasted tool calls (skill_efficiency 50%) and token usage (token_efficiency 50%).
+ +Underlying evaluation signals used in this run:
+- `security`: Unsafe operations, secret leakage, and unauthorized access.
+- `skill_execution`: Whether the expected skill was selected, decoys were avoided, and the workflow executed.
+- `accuracy`: Final-answer correctness against the reference answer.
+- `goal_accuracy`: Whether the user's goal was achieved.
+- `behavior_check`: Whether the expected workflow behavior was followed.
+- `skill_efficiency`: Tool-call productivity.
+- `token_efficiency`: Actual uncached prompt plus completion usage.
+ + ## Evaluation Results:
-Pending. NVSkills-Eval has not been run for this skill; results and a `BENCHMARK.md` will be published when the evaluation pipeline runs.
+| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 79.9% — baseline ran, but no comparable score was available; uplift unavailable | 80.5% — baseline ran, but no comparable score was available; uplift unavailable | +| Security | 100.0% → 75.0% (-25.0 points) | 50.0% → 100.0% (+50.0 points) | +| Correctness | 6.7% → 90.0% (+83.3 points) | 91.4% → 90.0% (-1.4 points) | +| Discoverability | 93.3% — baseline ran, but no comparable score was available; uplift unavailable | 88.3% — baseline ran, but no comparable score was available; uplift unavailable | +| Effectiveness | 20.0% → 55.0% (+35.0 points) | 30.4% → 48.1% (+17.7 points) | +| Efficiency | 86.2% — baseline ran, but no comparable score was available; uplift unavailable | 76.0% — baseline ran, but no comparable score was available; uplift unavailable | ## Skill Version(s):
-b1c082c (source: git SHA, committed 2026-07-17)
+77111e0 (source: git SHA, committed 2026-09-09)
## Ethical Considerations:
NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their internal team to ensure this skill meets requirements for the relevant industry and use case and addresses unforeseen product misuse.
-Models produced by this skill inherit the composition and biases of the user's pretraining corpus entirely, with no counterbalancing from prior pretraining. Downstream predictions are research hypotheses and must not be used as the sole basis for clinical, safety, or regulatory decisions.
- (For Release on NVIDIA Platforms Only)
-Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://www.nvidia.com/en-us/support/submit-security-vulnerability/).
+Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://app.intigriti.com/programs/nvidia/nvidiavdp/detail).
diff --git a/skills/kermt-pretrain-scratch/skill.oms.sig b/skills/kermt-pretrain-scratch/skill.oms.sig new file mode 100644 index 0000000..6084616 --- /dev/null +++ b/skills/kermt-pretrain-scratch/skill.oms.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAia2VybXQtcHJldHJhaW4tc2NyYXRjaCIsCiAgICAgICJkaWdlc3QiOiB7CiAgICAgICAgInNoYTI1NiI6ICJiYTM3NDU1M2MxN2ZiOTNiZTY1ZjJhNjdmNTZjYmQzMzhiY2YyMWJkMGJhODIxNDgwMTgwNTRlZTkxMTE5MzdiIgogICAgICB9CiAgICB9CiAgXSwKICAicHJlZGljYXRlVHlwZSI6ICJodHRwczovL21vZGVsX3NpZ25pbmcvc2lnbmF0dXJlL3YxLjAiLAogICJwcmVkaWNhdGUiOiB7CiAgICAicmVzb3VyY2VzIjogWwogICAgICB7CiAgICAgICAgIm5hbWUiOiAiQkVOQ0hNQVJLLm1kIiwKICAgICAgICAiZGlnZXN0IjogIjM5NGI1ODY2YWY5YmI1NTAwOWEyNmZkYjY4MzRlZTU2Yjk3Mzg1OGE0YjM5MDc2MDViZDk2M2E0YmU2NDExYWUiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAiU0tJTEwubWQiLAogICAgICAgICJkaWdlc3QiOiAiMzNmOWVlYjUxYjM2NTM1MWU3MDhmNTQxMWMxOWI2YmRlZGMyZTYzNzExMzBhMWRmZjdhYjUzODI3MTNmN2ZhMSIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJjb25maWcvZGVmYXVsdHNfcHJldHJhaW4uanNvbiIsCiAgICAgICAgImRpZ2VzdCI6ICJkNGZhYTY3YTk0Yzc4ZTEwMjZiNzg5NDYwYTllZmViNDkzYjI2ZjQ3ZTcxODQzM2YxNjBjNDQyMjRjZGI4NzU1IiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogImV2YWxzL2V2YWxzLmpzb24iLAogICAgICAgICJkaWdlc3QiOiAiOWE5ODYwMzk3YmM4NGIxMTEzNzk0YWQ5MjdmN2IyMGRiNTFhYTcwOWNmZmRhYzdiZjJiMDI5MTkzOTE1NjNlMiIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJzY3JpcHRzL191dGlscy5weSIsCiAgICAgICAgImRpZ2VzdCI6ICIwMjYyMjBhMjI2ZDhmODQxODY0ZjNmOGQ3NGRmYzdjZWVlNTJkOTg1ZDQ1YTllMTQ2ZTI3Y2MxNTIzOWZmNTQyIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInNjcmlwdHMvY2hlY2tfY2hlY2twb2ludC5weSIsCiAgICAgICAgImRpZ2VzdCI6ICIwYmM2Y2M4NzI5ZGIzYTNhOTM5ZmNlN2E5NWU1ZjZkMTY2YWY1NTE3ZmI2OTM0NmJkYTJjMzAxMTQ1YThmNTAyIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInNjcmlwdHMvY2hlY2tfZGF0YS5weSIsCiAgICAgICAgImRpZ2VzdCI6ICI2ODkyNjgxODczZDY2ZTdiZTQxMmI3NzMwYTM0OWU2NDA5MDljNzJhZmFlMjIyODU0ZjdhMDFiNjRiMmFhYjU2IiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInNjcmlwdHMva2VybXRfY29udGFpbmVyLnNoIiwKICAgICAgICAiZGlnZXN0IjogImZjZGFiNTk1YzBlODlkZjE1ODE5YTdlMjc0MWMzMWY1MTJkOTVlMDhmNjQ2MWY4MWQ1MzVlY2EzY2JhNDAyYjgiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAic2NyaXB0cy9wcmVwYXJlX2RhdGEucHkiLAogICAgICAgICJkaWdlc3QiOiAiMTQyNjI4OWFiODMzNGYxYTk5ZTliZjFhNDE3Njc5NTgyNGYyNzUxOWZmOTE0MjE2YTNmMDdmOTBlZWFhZjAzYiIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJzY3JpcHRzL3J1bl9wcmV0cmFpbl9sb2NhbC5weSIsCiAgICAgICAgImRpZ2VzdCI6ICIzYzU0ZDUzNzE1OTQyMTc1YzRhOThkY2UzMzVlNDYwMTgzOTQ1YjdiZjBmOTNjNjFiNjA3OTA3Mzc0OWY4NDY5IiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInNraWxsLWNhcmQubWQiLAogICAgICAgICJkaWdlc3QiOiAiOGNlNzNmZTczMDU5MWJlNzBkNDUzZWU5MTAwYzZkYjM1YjE0MzM4ZWUyZTRmN2NlNTA5OTUwYTkwZjQ2NjQxMiIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0KICAgIF0sCiAgICAic2VyaWFsaXphdGlvbiI6IHsKICAgICAgImFsbG93X3N5bWxpbmtzIjogZmFsc2UsCiAgICAgICJtZXRob2QiOiAiZmlsZXMiLAogICAgICAiaWdub3JlX3BhdGhzIjogWwogICAgICAgICIuZ2l0aWdub3JlIiwKICAgICAgICAiLmdpdGh1YiIsCiAgICAgICAgIi5naXQiLAogICAgICAgICIuZ2l0YXR0cmlidXRlcyIKICAgICAgXSwKICAgICAgImhhc2hfdHlwZSI6ICJzaGEyNTYiCiAgICB9CiAgfQp9","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGYCMQDzG/p7KHwAmoS4DhYizb74pRiykKFVXKU0UDbkM8TlMTuWRv9h/iQyPW2Dm6QWXAICMQDncd6tS/agBSOwjUjr5/LuofPz9BNjRshBO2xNaMg/c4/gDjVgAOgn3uNzx0FiZ+8=","keyid":""}]}} \ No newline at end of file diff --git a/skills/kermt-setup/BENCHMARK.md b/skills/kermt-setup/BENCHMARK.md new file mode 100644 index 0000000..9853fff --- /dev/null +++ b/skills/kermt-setup/BENCHMARK.md @@ -0,0 +1,127 @@ +# Skill Benchmark: kermt-setup + +> ✅ **Overall verdict: PASS — Recommended for publication** + +## Publication Recommendation + +Recommended for publication based on the completed evaluation evidence in this report. + +## Evaluation Metadata + +- Skill: `kermt-setup` +- Evaluation date: 2026-09-15 +- Evaluator version: `1.5.6` +- Agents: Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`), Codex (`openai/openai/gpt-5.5`) +- Tasks: 4 evaluation tasks (3 positive, 1 negative) +- Dataset digest: `sha256:acc9600691efbd2c4b9356f2ca6f5e3a58ccccbf8772a88bf444c1b00baa43e3` (skill-evaluator-dataset-snapshot/1) +- Attempts per task: 3 +- Environment: `k8s-sandbox` +- Tier 2 evidence: required for publication +- Tier 3 evidence: required for publication + +Each task attempt ran in its own isolated sandbox pod. + +## What This Report Answers + +The three-tier evaluation checks whether the skill: + +- is safe to use; +- produces correct answers; +- is discovered and activated when needed; +- helps the agent complete the user's goal and expected workflow; and +- avoids wasted skill and tool usage. + +## Results at a Glance + +| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 80.5% — baseline ran, but no comparable score was available; uplift unavailable | 89.1% — baseline ran, but no comparable score was available; uplift unavailable | +| Security | 100.0% → 50.0% (-50.0 points) | 50.0% → 100.0% (+50.0 points) | +| Correctness | 40.0% → 100.0% (+60.0 points) | 60.0% → 100.0% (+40.0 points) | +| Discoverability | 90.0% — baseline ran, but no comparable score was available; uplift unavailable | 92.7% — baseline ran, but no comparable score was available; uplift unavailable | +| Effectiveness | 31.9% → 78.8% (+46.9 points) | 40.0% → 70.0% (+30.0 points) | +| Efficiency | 83.5% — baseline ran, but no comparable score was available; uplift unavailable | 82.6% — baseline ran, but no comparable score was available; uplift unavailable | + +**How to read this table:** baseline is the same task attempted without the target skill. Scores are rounded to one decimal; threshold-adjacent values use additional precision so their displayed band matches the verdict. Uplift is derived from those displayed scores and shown in percentage points. + +Example: `47.0% → 92.0% (+45.0 points)` means the skill-assisted run scored 92.0%, 45.0 percentage points above its 47.0% no-skill baseline. + +A partial dimension was calculated from only the available configured signals; review the detailed report before relying on it. + +## Token Usage + +Actual Tier 3 execution usage is reported for every observed agent/case pair and both conditions. + +| Agent | Dataset case | With skill | Without skill | Delta | Change | Coverage | +|---|---|---:|---:|---:|---:|---| +| claude-code | All cases | 1,402,843 | 1,292,877 | N/A | N/A | skill 4/4; base 8/8 | +| claude-code | kermt-setup-001 | 353,934 | 463,477 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-setup-002 | 385,745 | 221,859 | +163,886 | +73.87% | skill 1/1; base 1/1 | +| claude-code | kermt-setup-003 | 342,340 | 518,121 | N/A | N/A | skill 1/1; base 3/3 | +| claude-code | kermt-setup-004 | 320,824 | 89,420 | +231,404 | +258.78% | skill 1/1; base 1/1 | +| codex | All cases | 311,920 | 568,520 | N/A | N/A | skill 4/4; base 6/6 | +| codex | kermt-setup-001 | 75,175 | 69,265 | +5,910 | +8.53% | skill 1/1; base 1/1 | +| codex | kermt-setup-002 | 126,981 | 433,754 | N/A | N/A | skill 1/1; base 3/3 | +| codex | kermt-setup-003 | 95,961 | 52,034 | +43,927 | +84.42% | skill 1/1; base 1/1 | +| codex | kermt-setup-004 | 13,803 | 13,467 | +336 | +2.49% | skill 1/1; base 1/1 | +| ALL AGENTS | Dataset aggregate | 1,714,763 | 1,861,397 | N/A | N/A | skill 8/8; base 14/14 | + +Prompt tokens include cached reads, so total tokens are `prompt + completion` (cached is not added twice). The Efficiency score uses `(prompt - cached) + completion`. N/A means the relevant trajectory counters were not available; coverage is never estimated. + +## Tier Status + +| Tier | Purpose | Status | Evidence | +|---|---|---|---| +| Tier 1 | Static validation | **PASSED WITH OBSERVATIONS** | 11 validator(s); 26 finding(s) | +| Tier 2 | Semantic deduplication | **PASSED** | 2 validator(s); 0 finding(s) | +| Tier 3 | Live agent evaluation | **PASS** | 2 agent(s); 4 task(s) | + +## Findings and Observations + +
+Show detailed findings and successful checks + +- **MEDIUM** QUALITY/quality_correctness: No documented scripts in table format (`skills/kermt-setup/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: Instructions don't mention 'run_script' (`skills/kermt-setup/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.author' (`skills/kermt-setup/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.tags' (`skills/kermt-setup/SKILL.md`) +- **MEDIUM** SCHEMA/metadata_key_style: Metadata key 'risk_tier' is not kebab-case (`skills/kermt-setup/SKILL.md`) +- 21 additional finding(s) are available in the full evaluation artifacts. + +
+ +## Scoring Methodology + +
+Show dimension definitions, source signals, and thresholds + +| Dimension | Question | Scored signals | +|---|---|---| +| Security | Is it safe to use? | `security` (100%) | +| Correctness | Is the answer correct? | `accuracy` (100%) | +| Discoverability | Was the right skill loaded when needed? | `skill_execution` (100%) | +| Effectiveness | Did the skill help complete the task? | `goal_accuracy` (50%) + `behavior_check` (50%) | +| Efficiency | Did it avoid wasted tool calls and token usage? | `skill_efficiency` (50%) + `token_efficiency` (50%) | + +- Dimension bands: PASS at 50% or above; NEUTRAL from 40% to below 50%; FAIL below 40%. +- Overall Tier 3 lift: PASS at +5 points or more; FAIL at -10 points or less; values between those bands are NEUTRAL. +- Overall verdict: PASS only when every configured dimension passes for at least one supported agent. Lift is reported as diagnostic evidence and does not override this gate. +- The 50% attempt pass threshold is a separate per-task gate; it is not the dimension pass threshold. +- Effectiveness is the equal-weight mean of goal completion (`goal_accuracy`) and expected workflow adherence (`behavior_check`). +- Efficiency is 50% tool-call productivity (the backward-compatible `skill_efficiency` wire id) and 50% `token_efficiency`. Positive-case skill routing is scored under Discoverability, not Efficiency; a negative case without a routing target is N/A. N/A sources are omitted, remaining weights are renormalized, and the dimension is marked partial. + +Signals present in this run: + +- `security` (Security): unsafe operations, secret leakage, and unauthorized access. +- `skill_execution` (Skill Execution): whether the expected skill was selected, decoys were avoided, and the workflow executed. +- `skill_efficiency` (Tool Productivity): tool-call productivity (legacy wire id; routing is scored under Discoverability). +- `accuracy` (Accuracy): final-answer correctness against the reference answer. +- `goal_accuracy` (Goal Accuracy): whether the user's goal was achieved. +- `behavior_check` (Behavior Check): whether the expected workflow behavior was followed. +- `token_efficiency` (Token Efficiency): actual uncached prompt plus completion usage (50% of Efficiency). + +
+ +## Freshness + +Regenerate this benchmark when the skill, evaluation dataset, target agent/model, evaluator version, environment, or scoring policy changes. diff --git a/skills/kermt-setup/skill-card.md b/skills/kermt-setup/skill-card.md index ab355c3..31ed72c 100644 --- a/skills/kermt-setup/skill-card.md +++ b/skills/kermt-setup/skill-card.md @@ -1,69 +1,85 @@ ## Description:
-Bootstraps the KERMT agent environment — verifies host Docker and the NVIDIA Container Toolkit, builds the `kermt:latest` image from the repository Dockerfile if absent, and runs a GPU smoke test inside the container.
+Bootstrap the KERMT agent environment — verify host docker + nvidia-container-toolkit, build the kermt:latest image from the repo’s Dockerfile if it doesn’t yet exist, and run a GPU smoke test inside the container.
This skill is ready for commercial/non-commercial use.
## Owner -NVIDIA (evax@nvidia.com)
+NVIDIA
### License/Terms of Use:
-Apache-2.0
- +Apache 2.0
## Use Case:
-Computational chemists and ML engineers preparing a workstation or GPU node to run any `kermt-*` skill. Every other KERMT skill depends on this one; it is the first skill to invoke on a fresh clone.
- -### Requirements/Dependencies:
-Requires API Key or External Credential: No
-Credential Type(s): None
- -* Docker
-* NVIDIA Container Toolkit
-* CUDA-capable NVIDIA GPU
-* A local clone of the KERMT repository (supplies the Dockerfile)
- -Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
+Developers and engineers who need to bootstrap a containerized KERMT environment for molecular property prediction model training, finetuning, and inference workflows.
### Deployment Geography for Use:
Global
-## Known Risks and Mitigations:
-Risk: The skill builds a container image and runs GPU workloads, which consumes significant local disk and can take tens of minutes on a cold cache.
-Mitigation: The skill checks for an existing `kermt:latest` image and skips the build when one is present; the smoke test is short and read-only.
+## Requirements / Dependencies:
+**Requires API Key or External Credential:** [Not Specified]
+**Credential Type(s):** [None identified]
-Risk: Docker commands require elevated host privileges, and an agent running them has broad access to the host container runtime.
-Mitigation: The skill issues only build, run, and inspect commands against the KERMT image; users should keep their agent's command-approval gate enabled and review commands before execution.
+Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
-Risk: A partially configured host (driver/toolkit mismatch) can produce a container that starts but cannot see the GPU, causing confusing downstream failures in training skills.
-Mitigation: The GPU smoke test runs inside the container and fails loudly at setup time rather than deferring the error to a long-running job.
+## Known Risks and Mitigations:
+Risk: Review before execution as proposals could introduce incorrect or misleading guidance into skills.
+Mitigation: Review and scan skill before deployment.
## Reference(s):
-- [KERMT repository](https://github.com/NVIDIA-BioNeMo/KERMT)
-- [NVIDIA Container Toolkit documentation](https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/)
-- `scripts/kermt_container.sh` — container entry points used by this skill
+- [KERMT paper (Multitask finetuning and acceleration of chemical pretrained models)](https://arxiv.org/abs/2510.12719)
+- [GROVER paper (Self-Supervised Message Passing Transformer)](https://arxiv.org/abs/2007.02835)
+- [cuik-molmaker (NVIDIA Digital Bio)](https://github.com/NVIDIA-Digital-Bio/cuik-molmaker)
+- [GROVER original implementation](https://github.com/tencent-ailab/grover)
+ ## Skill Output:
-**Output Type(s):** [Analysis, Configuration instructions]
-**Output Format:** [Markdown status report with inline bash commands]
-**Output Parameters:** [1D — pass/fail status per environment check]
-**Other Properties Related to Output:** [Side effect: builds the `kermt:latest` Docker image on the host if it does not already exist]
+**Output Type(s):** [Shell commands, Configuration instructions]
+**Output Format:** [Markdown with inline bash code blocks]
+**Output Parameters:** [1D]
+**Other Properties Related to Output:** [None]
## Evaluation Agents Used:
-Target agents: `claude-code`, `codex`. NVSkills-Eval has not yet been run against this skill — see Evaluation Results.
+- Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`)
+- Codex (`openai/openai/gpt-5.5`)
+ + ## Evaluation Tasks:
-4 evaluation tasks defined in `evals/evals.json`, covering environment verification, image build, and GPU smoke test paths.
+Evaluated against 4 tasks (3 positive, 1 negative) with 3 attempts each, executed in isolated sandbox pods.
## Evaluation Metrics Used:
-Planned NVSkills-Eval dimensions: Security, Correctness, Discoverability, Effectiveness, Efficiency.
+Reported benchmark dimensions:
+- Security: Whether the skill avoids unsafe operations, secret leakage, and unauthorized access.
+- Correctness: Whether the skill produces correct final answers against reference outputs.
+- Discoverability: Whether the expected skill was selected, decoys were avoided, and the workflow executed.
+- Effectiveness: Whether the skill helped complete the user’s goal (50% goal completion + 50% expected workflow adherence).
+- Efficiency: Whether the skill avoided wasted tool calls and token usage (50% tool productivity + 50% token efficiency).
+ +Underlying evaluation signals used in this run:
+- `security`: Checks for unsafe operations, secret leakage, and unauthorized access.
+- `accuracy`: Final-answer correctness against the reference answer.
+- `skill_execution`: Whether the expected skill was selected and the workflow executed.
+- `goal_accuracy`: Whether the user’s goal was achieved.
+- `behavior_check`: Whether the expected workflow behavior was followed.
+- `skill_efficiency`: Tool-call productivity measured against baseline.
+- `token_efficiency`: Actual uncached prompt plus completion token usage.
+ + ## Evaluation Results:
-Pending. NVSkills-Eval has not been run for this skill; results and a `BENCHMARK.md` will be published when the evaluation pipeline runs.
+| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 80.5% | 89.1% | +| Security | 100.0% → 50.0% (-50.0 pts) | 50.0% → 100.0% (+50.0 pts) | +| Correctness | 40.0% → 100.0% (+60.0 pts) | 60.0% → 100.0% (+40.0 pts) | +| Discoverability | 90.0% | 92.7% | +| Effectiveness | 31.9% → 78.8% (+46.9 pts) | 40.0% → 70.0% (+30.0 pts) | +| Efficiency | 83.5% | 82.6% | ## Skill Version(s):
-b1c082c (source: git SHA, committed 2026-07-17)
+77111e0 (source: git SHA, committed 2026-09-09)
## Ethical Considerations:
NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their internal team to ensure this skill meets requirements for the relevant industry and use case and addresses unforeseen product misuse.
(For Release on NVIDIA Platforms Only)
-Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://www.nvidia.com/en-us/support/submit-security-vulnerability/).
+Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://app.intigriti.com/programs/nvidia/nvidiavdp/detail).
diff --git a/skills/kermt-setup/skill.oms.sig b/skills/kermt-setup/skill.oms.sig new file mode 100644 index 0000000..a864f4a --- /dev/null +++ b/skills/kermt-setup/skill.oms.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAia2VybXQtc2V0dXAiLAogICAgICAiZGlnZXN0IjogewogICAgICAgICJzaGEyNTYiOiAiYzZkMjIzYjU4N2U4NmE2Yjg3OWYxZmRiMmJiYjJhNmNmMjQ2ZTFiMzY4YjNiNjc0ZThhOGJlNzU1MTFjNTE2MiIKICAgICAgfQogICAgfQogIF0sCiAgInByZWRpY2F0ZVR5cGUiOiAiaHR0cHM6Ly9tb2RlbF9zaWduaW5nL3NpZ25hdHVyZS92MS4wIiwKICAicHJlZGljYXRlIjogewogICAgInJlc291cmNlcyI6IFsKICAgICAgewogICAgICAgICJuYW1lIjogIkJFTkNITUFSSy5tZCIsCiAgICAgICAgImRpZ2VzdCI6ICIyYmE4N2YxYTNiN2YzZDNmMTBmZGY4ZDc5NGQ5NTViNWYwMzlkZmRkZTc2MDFiYzljOWVjMmZlN2QwOTExNmI2IiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogIlNLSUxMLm1kIiwKICAgICAgICAiZGlnZXN0IjogIjdmNmUzOWY2NTVlYWI1NDg1M2Y4MDAzOGNkMGM4M2E4ZWZmZWM4ZjQ0MjYwZDA4MzhjOWU1M2IwNDZjNmEwZjkiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAiZXZhbHMvZXZhbHMuanNvbiIsCiAgICAgICAgImRpZ2VzdCI6ICJmMWExYzVmZjcyYWRkZDJlODRiOGY2MWU5ZWQyMDI4MDk3N2E1YzQ1MDU0NDIxODcwZjE2ZDIyOTRhNGJmMWVkIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInNjcmlwdHMva2VybXRfY29udGFpbmVyLnNoIiwKICAgICAgICAiZGlnZXN0IjogImZjZGFiNTk1YzBlODlkZjE1ODE5YTdlMjc0MWMzMWY1MTJkOTVlMDhmNjQ2MWY4MWQ1MzVlY2EzY2JhNDAyYjgiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAic2tpbGwtY2FyZC5tZCIsCiAgICAgICAgImRpZ2VzdCI6ICJlOWJhNTViMTJkOTlmZjZmZDUwY2IzMzU5YjE1OWNmMDQ5MzdjOGU2ZmU0NTRlYjdkODZjYWM2MjIzYjExN2VkIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfQogICAgXSwKICAgICJzZXJpYWxpemF0aW9uIjogewogICAgICAiYWxsb3dfc3ltbGlua3MiOiBmYWxzZSwKICAgICAgImhhc2hfdHlwZSI6ICJzaGEyNTYiLAogICAgICAiaWdub3JlX3BhdGhzIjogWwogICAgICAgICIuZ2l0YXR0cmlidXRlcyIsCiAgICAgICAgIi5naXRodWIiLAogICAgICAgICIuZ2l0IiwKICAgICAgICAiLmdpdGlnbm9yZSIKICAgICAgXSwKICAgICAgIm1ldGhvZCI6ICJmaWxlcyIKICAgIH0KICB9Cn0=","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGQCMEuBuk4vMMThfDjOORcilPSFJtsh/Jnz8z7nLuGIU/d8RdiHCiaUd5lrRflNSdVbzQIwSTPL5W2zwTHiZpRL9VfLtQCtCLZ3x2bq2n2mA8UTicR8S6hUwkkfiEFj97TcEDwd","keyid":""}]}} \ No newline at end of file