From ac632e922265d3888df6e024b8e2826557921f1b Mon Sep 17 00:00:00 2001 From: Ohad Mosafi Date: Tue, 29 Sep 2026 11:23:07 -0700 Subject: [PATCH 1/7] Improve Evo2 generation execution and result validation Signed-off-by: Ohad Mosafi --- nim-skills/evo2-nim/SKILL.md | 128 +++++++------ .../evo2-nim/config/skillspector-baseline.yml | 22 +++ nim-skills/evo2-nim/evals/evals.json | 114 ++---------- nim-skills/evo2-nim/scripts/generate.py | 171 ++++++++++++++++++ .../skills/evo2-nim/SKILL.md | 128 +++++++------ .../evo2-nim/config/skillspector-baseline.yml | 22 +++ .../skills/evo2-nim/evals/evals.json | 114 ++---------- .../skills/evo2-nim/scripts/generate.py | 171 ++++++++++++++++++ tests/test_evo2_generate.py | 161 +++++++++++++++++ 9 files changed, 711 insertions(+), 320 deletions(-) create mode 100644 nim-skills/evo2-nim/scripts/generate.py create mode 100644 skills/bionemo-agent-toolkit/skills/evo2-nim/scripts/generate.py create mode 100644 tests/test_evo2_generate.py diff --git a/nim-skills/evo2-nim/SKILL.md b/nim-skills/evo2-nim/SKILL.md index 933d130..7eca9fd 100644 --- a/nim-skills/evo2-nim/SKILL.md +++ b/nim-skills/evo2-nim/SKILL.md @@ -9,8 +9,8 @@ allowed-tools: Bash, Read, Write, AskUserQuestion # Evo 2 NIM -Use Evo 2 for DNA generation and, locally, layer-output extraction. Use this -`SKILL.md` for basic hosted/local use; load supplemental files only when needed: +Use Evo 2 for DNA generation and, locally, layer-output extraction. Load +supplemental files only when needed: - `references/api.md`: exact schemas, layer names, Docker flags, hardware notes. - `references/science.md`: genomic use cases, limits, and interpretation. @@ -18,6 +18,26 @@ Use Evo 2 for DNA generation and, locally, layer-output extraction. Use this - `references/validation.md`: DNA, probability, timing, and tensor checks. - `references/examples.md`: compact hosted/local request patterns. +## Instructions + +For generation, use `scripts/generate.py` to execute the request, validate the +response, and save its artifacts. Resolve the script path relative to this +skill's directory and choose an output directory in the user's workspace. +Use the user's sequence and requested parameters; the example below is only +a smoke test. + +1. Select the requested mode. For hosted generation, go directly to the + generation example; Docker setup and local forward passes are separate tasks. +2. When the user asks to run generation, execute the client and inspect its + exit status and result. Writing a script alone does not complete that request. +3. Report the generated DNA (or its file for long sequences), actual + `elapsed_ms`, sampled-probability summary, seed, and artifact paths from the + successful run. Read the saved response or metrics if any result is unclear. + +If the request or validation fails, report the actual failure and any diagnostic +files. Do not replace an unavailable API response with example values. For a +code-only request, provide the command without making an inference call. + ## Choose Mode Honor `NIM_API_MODE` when it is set. Accepted values are `hosted` and `local`. @@ -45,6 +65,42 @@ into the container with `-e NGC_API_KEY`. Local inference requests use no auth header after readiness. Warm-cache key-free startup varies by image/version and should not be assumed. +## Examples + +Normalize prompts before sending. Use A/C/G/T unless ambiguous bases are a +deliberate modeling choice and clearly reported. + +For a hosted generation request, run the bundled client with the user's inputs +(the script path below is relative to the skill directory): + +```bash +python scripts/generate.py \ + --mode hosted \ + --sequence ACTGACTGACTGACTG \ + --num-tokens 64 --seed 1 \ + --temperature 0.7 --top-k 3 --top-p 0.0 \ + --output-dir /path/to/workspace/evo2-output +``` + +For an already-ready local NIM, use `--mode local`; the client resolves +`EVO2_NIM_URL` and sends no Authorization header. It never switches endpoints +after a failed request. Set `--timeout` for a longer read if the user requests +a larger generation; failed requests are not automatically resubmitted. + +The client saves `request.json`, the actual `response.json`, `generated.fasta`, +and `metrics.json` in the chosen output directory. It validates the requested +number of generated bases, A/C/G/T alphabet, finite sampled probabilities in +`[0, 1]`, and nonnegative timing before printing a successful summary. Existing +outputs are not overwritten; choose a new output directory for each run. +The FASTA contains generated bases only, not the input prompt prepended again. + +`sampled_probs` is requested by the client and summarized with count/min/max/mean; +the full values stay in the saved response. A missing or malformed probability +array is a validation failure, not permission to invent confidence values. +Only request `enable_logits` in a custom request when needed; logits can make +responses large. See `references/api.md` for custom payloads. +`random_seed` supports development reproducibility, not biological certainty. + ## Local Docker Requirements Evo 2 local deployment requires FP8-capable GPUs. Do not present A100 as @@ -100,68 +156,6 @@ If RTX PRO 6000 Blackwell Workstation fails with no Transformer Engine attention backend, treat it as outside the current validated matrix and rerun on a documented GPU/runtime. -## DNA Generation - -Normalize prompts before sending. Use A/C/G/T unless ambiguous bases are a -deliberate modeling choice and clearly reported. - -```python -import json -import os -from pathlib import Path -import requests - -def clean_dna(value: str) -> str: - seq = "".join(value.upper().split()) - invalid = sorted(set(seq) - set("ACGT")) - if invalid: - raise ValueError(f"Unexpected DNA characters: {''.join(invalid)}") - return seq - -prompt = clean_dna("ACTGACTGACTGACTG") -mode = os.getenv("NIM_API_MODE") -if mode is None: - mode = "local" if os.getenv("EVO2_NIM_URL") else "hosted" -if mode not in {"hosted", "local"}: - raise ValueError("NIM_API_MODE must be 'hosted' or 'local'") - -nim_url = os.getenv("EVO2_NIM_URL", "http://localhost:8000").rstrip("/") -url = ( - "https://health.api.nvidia.com/v1/biology/arc/evo2-40b/generate" - if mode == "hosted" else f"{nim_url}/biology/arc/evo2/generate" -) -headers = {"Content-Type": "application/json"} -if mode == "hosted": - api_key = os.getenv("NGC_API_KEY") - if not api_key: - raise RuntimeError("Set NGC_API_KEY for hosted Evo 2") - headers["Authorization"] = f"Bearer {api_key}" - -payload = { - "sequence": prompt, - "num_tokens": 64, - "temperature": 0.7, - "top_k": 3, - "top_p": 0.0, - "random_seed": 1, - "enable_sampled_probs": True, - "enable_elapsed_ms_per_token": True, -} -response = requests.post(url, headers=headers, json=payload, timeout=180) -response.raise_for_status() -result = response.json() -seq = result["sequence"] -if sorted(set(seq.upper()) - set("ACGT")): - raise ValueError("Generated sequence contains unexpected non-ACGT bases") - -Path("evo2_generation.json").write_text(json.dumps(result, indent=2) + "\n") -Path("evo2_generated.fa").write_text(f">evo2_generated\n{seq}\n") -print(f"Generated {len(seq)} bases in {result.get('elapsed_ms')} ms") -``` - -Only request `enable_logits` when needed; logits can make responses large. -`random_seed` supports development reproducibility, not biological certainty. - ## Local Forward Pass Forward returns base64-encoded NPZ tensors. @@ -177,8 +171,12 @@ mode = os.getenv("NIM_API_MODE", "local") if mode != "local": raise RuntimeError("Evo 2 /forward is available only in local mode") nim_url = os.getenv("EVO2_NIM_URL", "http://localhost:8000").rstrip("/") +sequence = "ACTGACTGACTG" # Replace with the user's DNA sequence. +sequence = "".join(sequence.upper().split()) +if not sequence or set(sequence) - set("ACGT"): + raise ValueError("Expected nonempty A/C/G/T DNA") payload = { - "sequence": clean_dna("ACTGACTGACTG"), + "sequence": sequence, "output_layers": ["output_layer", "decoder.layers.3.self_attention"], } response = requests.post( diff --git a/nim-skills/evo2-nim/config/skillspector-baseline.yml b/nim-skills/evo2-nim/config/skillspector-baseline.yml index c8477e0..5d873cb 100644 --- a/nim-skills/evo2-nim/config/skillspector-baseline.yml +++ b/nim-skills/evo2-nim/config/skillspector-baseline.yml @@ -6,6 +6,28 @@ version: 1 rules: + - id: "TT3" + path: "*scripts/generate.py" + reason: >- + Reviewed 2026-09-29 against the client and its request-contract tests. + The reported tainted URL is EVO2_NIM_URL, the user-selected local NIM + address, not a credential. It is validated as an HTTP(S) URL without + embedded credentials, query, or fragment. Local requests have no + Authorization header. Hosted requests use the fixed health.api.nvidia.com + endpoint and send NGC_API_KEY only as its required Bearer authentication; + redirects are disabled. Neither request artifacts nor summaries contain + the key. This exception covers the documented inference client only. + - id: "LP1" + path: "*scripts/generate.py" + reason: >- + Reviewed 2026-09-29 against the client and its request-contract tests. + The agent invokes this client through the declared Bash tool. Its + documented capabilities are reading NGC_API_KEY/NIM_API_MODE/EVO2_NIM_URL, + submitting the user's DNA to the chosen Evo 2 endpoint, and saving outputs + through the declared filesystem tools. The scanner maps Bash only to + shell and requires separate Env/WebFetch tool names even though this + Python client does not call either tool. The environment and network + operations are the explicitly requested inference workflow. - id: "PE3" path: "*SKILL.md" reason: >- diff --git a/nim-skills/evo2-nim/evals/evals.json b/nim-skills/evo2-nim/evals/evals.json index 3e2ebb5..81c4714 100644 --- a/nim-skills/evo2-nim/evals/evals.json +++ b/nim-skills/evo2-nim/evals/evals.json @@ -7,41 +7,13 @@ "expected_output": "A successfully executed hosted Evo 2 generation request with Bearer auth and exact request fields, plus actual response-derived DNA, sampled probabilities, elapsed timing, and saved JSON and FASTA artifacts.", "files": [], "assertions": [ - { - "id": "hosted-request-executed", - "description": "Executes the hosted request instead of only writing code", - "check": "Trajectory shows successful execution of the hosted request, and the final response reports actual response-derived DNA, elapsed timing, and saved artifact paths" - }, - { - "id": "hosted-endpoint-url", - "description": "Uses the correct hosted Evo2 generation endpoint", - "check": "Script contains 'https://health.api.nvidia.com/v1/biology/arc/evo2-40b/generate'" - }, - { - "id": "bearer-auth-header", - "description": "Sets hosted Authorization header from NGC_API_KEY", - "check": "Script contains 'Authorization', 'Bearer', and 'NGC_API_KEY'" - }, - { - "id": "generate-fields", - "description": "Uses exact Evo2 generation request field names", - "check": "Script contains 'sequence', 'num_tokens', 'temperature', 'top_k', 'top_p', and 'random_seed'" - }, - { - "id": "sampled-probs-field", - "description": "Requests or handles sampled probabilities using the correct field", - "check": "Script contains 'enable_sampled_probs' and handles 'sampled_probs'" - }, - { - "id": "dna-validation", - "description": "Validates generated DNA alphabet before claiming success", - "check": "Script checks generated sequence characters against A/C/G/T or an explicitly documented DNA alphabet" - }, - { - "id": "save-json-fasta", - "description": "Saves response JSON and generated FASTA output", - "check": "Script writes a .json file and a .fa or .fasta file" - } + "[hosted-request-executed] Executes the hosted request instead of only writing code: Trajectory shows successful execution of the hosted request, and the final response reports actual response-derived DNA, elapsed timing, and saved artifact paths", + "[hosted-endpoint-url] Uses the correct hosted Evo2 generation endpoint: Script contains 'https://health.api.nvidia.com/v1/biology/arc/evo2-40b/generate'", + "[bearer-auth-header] Sets hosted Authorization header from NGC_API_KEY: Script contains 'Authorization', 'Bearer', and 'NGC_API_KEY'", + "[generate-fields] Uses exact Evo2 generation request field names: Script contains 'sequence', 'num_tokens', 'temperature', 'top_k', 'top_p', and 'random_seed'", + "[sampled-probs-field] Requests or handles sampled probabilities using the correct field: Script contains 'enable_sampled_probs' and handles 'sampled_probs'", + "[dna-validation] Validates generated DNA alphabet before claiming success: Script checks generated sequence characters against A/C/G/T or an explicitly documented DNA alphabet", + "[save-json-fasta] Saves response JSON and generated FASTA output: Script writes a .json file and a .fa or .fasta file" ] } ], @@ -53,36 +25,12 @@ "expected_output": "Docker startup and health-check commands using the Evo2 image, repo env contract, FP8-capable GPU guidance, optional NIM_VARIANT, local no-auth inference, and an EVO2_NIM_URL-aware generation request with a localhost fallback.", "files": [], "assertions": [ - { - "id": "env-contract", - "description": "Uses repo env contract including .env, NGC_API_KEY/NVIDIA_API_KEY fallback, and LOCAL_NIM_CACHE", - "check": "Output mentions '.env', 'NGC_API_KEY', 'NVIDIA_API_KEY', and 'LOCAL_NIM_CACHE'" - }, - { - "id": "docker-image", - "description": "Uses the correct Evo2 Docker image", - "check": "Output contains 'nvcr.io/nim/arc/evo2:2'" - }, - { - "id": "cache-mount", - "description": "Mounts local cache to the documented container cache target", - "check": "Output contains '/opt/nim/.cache' and 'LOCAL_NIM_CACHE'" - }, - { - "id": "variant-gpu-guidance", - "description": "Documents optional NIM_VARIANT=7b, GPU selection, FP8-compatible hardware, and 40B memory requirements", - "check": "Output contains 'NIM_VARIANT', 'NIM_TEST_GPUS', 'FP8', 40B guidance for 2x H100 80GB or 1x H200 141GB, and 7B fallback guidance such as H100, H200, RTX 6000 Ada, or L40S; it must not present A100 as compatible for local Evo2" - }, - { - "id": "health-check", - "description": "Polls local readiness before inference", - "check": "Output builds the readiness URL from EVO2_NIM_URL with localhost:8000 only as a fallback" - }, - { - "id": "local-no-auth-endpoint", - "description": "Uses local generation endpoint with no Authorization header", - "check": "Script builds the local generation endpoint from EVO2_NIM_URL and does not send 'Authorization' to local inference" - } + "[env-contract] Uses repo env contract including .env, NGC_API_KEY/NVIDIA_API_KEY fallback, and LOCAL_NIM_CACHE: Output mentions '.env', 'NGC_API_KEY', 'NVIDIA_API_KEY', and 'LOCAL_NIM_CACHE'", + "[docker-image] Uses the correct Evo2 Docker image: Output contains 'nvcr.io/nim/arc/evo2:2'", + "[cache-mount] Mounts local cache to the documented container cache target: Output contains '/opt/nim/.cache' and 'LOCAL_NIM_CACHE'", + "[variant-gpu-guidance] Documents optional NIM_VARIANT=7b, GPU selection, FP8-compatible hardware, and 40B memory requirements: Output contains 'NIM_VARIANT', 'NIM_TEST_GPUS', 'FP8', 40B guidance for 2x H100 80GB or 1x H200 141GB, and 7B fallback guidance such as H100, H200, RTX 6000 Ada, or L40S; it must not present A100 as compatible for local Evo2", + "[health-check] Polls local readiness before inference: Output builds the readiness URL from EVO2_NIM_URL with localhost:8000 only as a fallback", + "[local-no-auth-endpoint] Uses local generation endpoint with no Authorization header: Script builds the local generation endpoint from EVO2_NIM_URL and does not send 'Authorization' to local inference" ] }, { @@ -91,36 +39,12 @@ "expected_output": "A local Python script that calls the Evo2 forward endpoint, decodes the base64 NPZ response with numpy, saves it, and prints tensor names, shapes, dtypes, and finite-value summaries.", "files": [], "assertions": [ - { - "id": "local-forward-endpoint", - "description": "Uses the correct local forward endpoint", - "check": "Script builds the local forward endpoint from EVO2_NIM_URL with localhost:8000 only as a fallback" - }, - { - "id": "forward-fields", - "description": "Uses exact forward request fields", - "check": "Script contains 'sequence' and 'output_layers'" - }, - { - "id": "requested-layers", - "description": "Requests the user-specified Evo2 layer names", - "check": "Script contains 'output_layer' and 'decoder.layers.3.self_attention'" - }, - { - "id": "base64-npz-decode", - "description": "Decodes base64 NPZ response data", - "check": "Script contains 'base64' and 'np.load' or 'numpy.load'" - }, - { - "id": "save-npz", - "description": "Saves decoded tensor artifacts as NPZ", - "check": "Script writes a .npz file" - }, - { - "id": "finite-summary", - "description": "Checks or reports tensor shape, dtype, and finite numeric values", - "check": "Script reports 'shape' and 'dtype' and checks 'isfinite' or prints numeric summaries" - } + "[local-forward-endpoint] Uses the correct local forward endpoint: Script builds the local forward endpoint from EVO2_NIM_URL with localhost:8000 only as a fallback", + "[forward-fields] Uses exact forward request fields: Script contains 'sequence' and 'output_layers'", + "[requested-layers] Requests the user-specified Evo2 layer names: Script contains 'output_layer' and 'decoder.layers.3.self_attention'", + "[base64-npz-decode] Decodes base64 NPZ response data: Script contains 'base64' and 'np.load' or 'numpy.load'", + "[save-npz] Saves decoded tensor artifacts as NPZ: Script writes a .npz file", + "[finite-summary] Checks or reports tensor shape, dtype, and finite numeric values: Script reports 'shape' and 'dtype' and checks 'isfinite' or prints numeric summaries" ] } ] diff --git a/nim-skills/evo2-nim/scripts/generate.py b/nim-skills/evo2-nim/scripts/generate.py new file mode 100644 index 0000000..0f6e6b2 --- /dev/null +++ b/nim-skills/evo2-nim/scripts/generate.py @@ -0,0 +1,171 @@ +#!/usr/bin/env python3 +"""Execute one Evo 2 generation request and save validated, reproducible outputs.""" + +from __future__ import annotations + +import argparse +import json +import math +import os +from itertools import groupby +from pathlib import Path +import sys +import time +from urllib.parse import urlsplit + +import requests + + +HOSTED_URL = "https://health.api.nvidia.com/v1/biology/arc/evo2-40b/generate" + + +def clean_dna(value: str) -> str: + sequence = "".join(value.upper().split()) + if not sequence or set(sequence) - set("ACGT"): + raise ValueError("Provide a nonempty A/C/G/T sequence; ambiguity codes need an explicit modeling choice") + return sequence + + +def number(value: object, label: str, minimum: float, maximum: float = math.inf) -> float: + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise ValueError(f"{label} must be numeric") + if not math.isfinite(value) or not minimum <= value <= maximum: + raise ValueError(f"{label} is outside the allowed range") + return float(value) + + +def validate_result(result: object, num_tokens: int) -> dict: + if not isinstance(result, dict): + raise ValueError("Expected a JSON object from Evo 2") + sequence = result.get("sequence") + if not isinstance(sequence, str) or not sequence or set(sequence.upper()) - set("ACGT"): + raise ValueError("Response sequence must be nonempty A/C/G/T DNA") + if len(sequence) != num_tokens: + raise ValueError(f"Requested {num_tokens} new bases, received {len(sequence)}; inspect the raw response") + probs = result.get("sampled_probs") + if not isinstance(probs, list) or len(probs) != len(sequence): + raise ValueError("sampled_probs must contain one probability per generated base") + for value in probs: + number(value, "sampled_probs", 0, 1) + number(result.get("elapsed_ms"), "elapsed_ms", 0) + timings = result.get("elapsed_ms_per_token") + if timings is not None: + if not isinstance(timings, list) or len(timings) != len(sequence): + raise ValueError("elapsed_ms_per_token must contain one timing per generated base") + for value in timings: + number(value, "elapsed_ms_per_token", 0) + dna = sequence.upper() + return { + "generated_bases": len(sequence), + "gc_fraction": (dna.count("G") + dna.count("C")) / len(dna), + "ambiguous_base_fraction": 0.0, + "longest_homopolymer": max(sum(1 for _ in group) for _, group in groupby(dna)), + "sampled_probs": { + "count": len(probs), "min": min(probs), "max": max(probs), + "mean": sum(probs) / len(probs), "valid": True, + }, + "elapsed_ms": result["elapsed_ms"], + "per_token_timing_available": timings is not None, + } + + +def generate(args: argparse.Namespace) -> dict: + sequence = clean_dna(args.sequence) + if args.num_tokens < 1: + raise ValueError("num_tokens must be positive") + number(args.temperature, "temperature", 0, 1.3) + number(args.top_k, "top_k", 0, 6) + number(args.top_p, "top_p", 0, 1) + number(args.timeout, "timeout", 1) + mode = args.mode or os.getenv("NIM_API_MODE") or ("local" if os.getenv("EVO2_NIM_URL") else "hosted") + if mode not in {"hosted", "local"}: + raise ValueError("NIM_API_MODE must be hosted or local") + headers = {"Content-Type": "application/json"} + if mode == "hosted": + key = os.getenv("NGC_API_KEY") + if not key: + raise ValueError("Set NGC_API_KEY in the environment for hosted Evo 2") + headers["Authorization"] = f"Bearer {key}" + url = HOSTED_URL + else: + base = os.getenv("EVO2_NIM_URL", "http://localhost:8000").rstrip("/") + parsed = urlsplit(base) + if parsed.scheme not in {"http", "https"} or not parsed.netloc or parsed.username or parsed.password: + raise ValueError("EVO2_NIM_URL must be an HTTP(S) service URL without embedded credentials") + if parsed.query or parsed.fragment: + raise ValueError("EVO2_NIM_URL must not include a query or fragment") + url = f"{base}/biology/arc/evo2/generate" + + payload = { + "sequence": sequence, "num_tokens": args.num_tokens, + "temperature": args.temperature, "top_k": args.top_k, "top_p": args.top_p, + "random_seed": args.seed, "enable_sampled_probs": True, + "enable_elapsed_ms_per_token": True, + } + output = args.output_dir.resolve() + paths = {name: output / filename for name, filename in { + "request": "request.json", "response": "response.json", + "fasta": "generated.fasta", "metrics": "metrics.json", + }.items()} + if any(path.exists() for path in paths.values()): + raise FileExistsError("Output files already exist; choose a new --output-dir for this request") + output.mkdir(parents=True, exist_ok=True) + paths["request"].write_text(json.dumps(payload, indent=2) + "\n") + # Keep the endpoint and request visible without ever printing the credential. + print(json.dumps({ + "event": "request", "method": "POST", "url": url, "payload": payload, + "authentication": "Authorization: Bearer from NGC_API_KEY" if mode == "hosted" else "none", + }), flush=True) + start = time.monotonic() + response = requests.post(url, headers=headers, json=payload, timeout=(10, args.timeout), allow_redirects=False) + wall_ms = round((time.monotonic() - start) * 1000) + if response.status_code != 200: + raise RuntimeError(f"Evo 2 returned HTTP {response.status_code}; generation was not completed") + result = response.json() + # Preserve the actual response even if validation fails; never manufacture missing fields. + paths["response"].write_text(json.dumps(result, indent=2, allow_nan=False) + "\n") + metrics = validate_result(result, args.num_tokens) + fasta = f">evo2_generated seed={args.seed}\n{result['sequence']}\n" + paths["fasta"].write_text(fasta) + # Verify persisted content before presenting the run as complete. + if json.loads(paths["response"].read_text()) != result or paths["fasta"].read_text() != fasta: + raise RuntimeError("Saved artifacts do not match the response") + metrics.update({ + "status": "completed", "http_status": response.status_code, "mode": mode, + "endpoint": url, "wall_ms": wall_ms, "random_seed": args.seed, + "artifacts": {name: str(path) for name, path in paths.items()}, + }) + paths["metrics"].write_text(json.dumps(metrics, indent=2) + "\n") + summary = dict(metrics) + sequence = result["sequence"] + summary["sequence" if len(sequence) <= 256 else "sequence_preview"] = sequence[:256] + print(json.dumps(summary, indent=2), flush=True) + return summary + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--sequence", required=True, help="DNA prompt; case and whitespace are normalized") + parser.add_argument("--mode", choices=["hosted", "local"], help="Explicit mode; otherwise use NIM_API_MODE/EVO2_NIM_URL") + parser.add_argument("--num-tokens", type=int, default=100) + parser.add_argument("--temperature", type=float, default=0.7) + parser.add_argument("--top-k", type=int, default=3) + parser.add_argument("--top-p", type=float, default=0.0) + parser.add_argument("--seed", type=int, default=1, help="Development reproducibility seed") + parser.add_argument("--timeout", type=float, default=180, help="Response read timeout in seconds") + parser.add_argument("--output-dir", type=Path, required=True) + args = parser.parse_args() + try: + generate(args) + except requests.RequestException as exc: + # Exception strings can include headers or URLs; report only the exception type. + print(f"Evo 2 request failed ({type(exc).__name__}); no successful generation to report", file=sys.stderr) + return 1 + except (ValueError, RuntimeError, OSError) as exc: + print(str(exc), file=sys.stderr) + return 1 + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md b/skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md index 933d130..7eca9fd 100644 --- a/skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md +++ b/skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md @@ -9,8 +9,8 @@ allowed-tools: Bash, Read, Write, AskUserQuestion # Evo 2 NIM -Use Evo 2 for DNA generation and, locally, layer-output extraction. Use this -`SKILL.md` for basic hosted/local use; load supplemental files only when needed: +Use Evo 2 for DNA generation and, locally, layer-output extraction. Load +supplemental files only when needed: - `references/api.md`: exact schemas, layer names, Docker flags, hardware notes. - `references/science.md`: genomic use cases, limits, and interpretation. @@ -18,6 +18,26 @@ Use Evo 2 for DNA generation and, locally, layer-output extraction. Use this - `references/validation.md`: DNA, probability, timing, and tensor checks. - `references/examples.md`: compact hosted/local request patterns. +## Instructions + +For generation, use `scripts/generate.py` to execute the request, validate the +response, and save its artifacts. Resolve the script path relative to this +skill's directory and choose an output directory in the user's workspace. +Use the user's sequence and requested parameters; the example below is only +a smoke test. + +1. Select the requested mode. For hosted generation, go directly to the + generation example; Docker setup and local forward passes are separate tasks. +2. When the user asks to run generation, execute the client and inspect its + exit status and result. Writing a script alone does not complete that request. +3. Report the generated DNA (or its file for long sequences), actual + `elapsed_ms`, sampled-probability summary, seed, and artifact paths from the + successful run. Read the saved response or metrics if any result is unclear. + +If the request or validation fails, report the actual failure and any diagnostic +files. Do not replace an unavailable API response with example values. For a +code-only request, provide the command without making an inference call. + ## Choose Mode Honor `NIM_API_MODE` when it is set. Accepted values are `hosted` and `local`. @@ -45,6 +65,42 @@ into the container with `-e NGC_API_KEY`. Local inference requests use no auth header after readiness. Warm-cache key-free startup varies by image/version and should not be assumed. +## Examples + +Normalize prompts before sending. Use A/C/G/T unless ambiguous bases are a +deliberate modeling choice and clearly reported. + +For a hosted generation request, run the bundled client with the user's inputs +(the script path below is relative to the skill directory): + +```bash +python scripts/generate.py \ + --mode hosted \ + --sequence ACTGACTGACTGACTG \ + --num-tokens 64 --seed 1 \ + --temperature 0.7 --top-k 3 --top-p 0.0 \ + --output-dir /path/to/workspace/evo2-output +``` + +For an already-ready local NIM, use `--mode local`; the client resolves +`EVO2_NIM_URL` and sends no Authorization header. It never switches endpoints +after a failed request. Set `--timeout` for a longer read if the user requests +a larger generation; failed requests are not automatically resubmitted. + +The client saves `request.json`, the actual `response.json`, `generated.fasta`, +and `metrics.json` in the chosen output directory. It validates the requested +number of generated bases, A/C/G/T alphabet, finite sampled probabilities in +`[0, 1]`, and nonnegative timing before printing a successful summary. Existing +outputs are not overwritten; choose a new output directory for each run. +The FASTA contains generated bases only, not the input prompt prepended again. + +`sampled_probs` is requested by the client and summarized with count/min/max/mean; +the full values stay in the saved response. A missing or malformed probability +array is a validation failure, not permission to invent confidence values. +Only request `enable_logits` in a custom request when needed; logits can make +responses large. See `references/api.md` for custom payloads. +`random_seed` supports development reproducibility, not biological certainty. + ## Local Docker Requirements Evo 2 local deployment requires FP8-capable GPUs. Do not present A100 as @@ -100,68 +156,6 @@ If RTX PRO 6000 Blackwell Workstation fails with no Transformer Engine attention backend, treat it as outside the current validated matrix and rerun on a documented GPU/runtime. -## DNA Generation - -Normalize prompts before sending. Use A/C/G/T unless ambiguous bases are a -deliberate modeling choice and clearly reported. - -```python -import json -import os -from pathlib import Path -import requests - -def clean_dna(value: str) -> str: - seq = "".join(value.upper().split()) - invalid = sorted(set(seq) - set("ACGT")) - if invalid: - raise ValueError(f"Unexpected DNA characters: {''.join(invalid)}") - return seq - -prompt = clean_dna("ACTGACTGACTGACTG") -mode = os.getenv("NIM_API_MODE") -if mode is None: - mode = "local" if os.getenv("EVO2_NIM_URL") else "hosted" -if mode not in {"hosted", "local"}: - raise ValueError("NIM_API_MODE must be 'hosted' or 'local'") - -nim_url = os.getenv("EVO2_NIM_URL", "http://localhost:8000").rstrip("/") -url = ( - "https://health.api.nvidia.com/v1/biology/arc/evo2-40b/generate" - if mode == "hosted" else f"{nim_url}/biology/arc/evo2/generate" -) -headers = {"Content-Type": "application/json"} -if mode == "hosted": - api_key = os.getenv("NGC_API_KEY") - if not api_key: - raise RuntimeError("Set NGC_API_KEY for hosted Evo 2") - headers["Authorization"] = f"Bearer {api_key}" - -payload = { - "sequence": prompt, - "num_tokens": 64, - "temperature": 0.7, - "top_k": 3, - "top_p": 0.0, - "random_seed": 1, - "enable_sampled_probs": True, - "enable_elapsed_ms_per_token": True, -} -response = requests.post(url, headers=headers, json=payload, timeout=180) -response.raise_for_status() -result = response.json() -seq = result["sequence"] -if sorted(set(seq.upper()) - set("ACGT")): - raise ValueError("Generated sequence contains unexpected non-ACGT bases") - -Path("evo2_generation.json").write_text(json.dumps(result, indent=2) + "\n") -Path("evo2_generated.fa").write_text(f">evo2_generated\n{seq}\n") -print(f"Generated {len(seq)} bases in {result.get('elapsed_ms')} ms") -``` - -Only request `enable_logits` when needed; logits can make responses large. -`random_seed` supports development reproducibility, not biological certainty. - ## Local Forward Pass Forward returns base64-encoded NPZ tensors. @@ -177,8 +171,12 @@ mode = os.getenv("NIM_API_MODE", "local") if mode != "local": raise RuntimeError("Evo 2 /forward is available only in local mode") nim_url = os.getenv("EVO2_NIM_URL", "http://localhost:8000").rstrip("/") +sequence = "ACTGACTGACTG" # Replace with the user's DNA sequence. +sequence = "".join(sequence.upper().split()) +if not sequence or set(sequence) - set("ACGT"): + raise ValueError("Expected nonempty A/C/G/T DNA") payload = { - "sequence": clean_dna("ACTGACTGACTG"), + "sequence": sequence, "output_layers": ["output_layer", "decoder.layers.3.self_attention"], } response = requests.post( diff --git a/skills/bionemo-agent-toolkit/skills/evo2-nim/config/skillspector-baseline.yml b/skills/bionemo-agent-toolkit/skills/evo2-nim/config/skillspector-baseline.yml index c8477e0..5d873cb 100644 --- a/skills/bionemo-agent-toolkit/skills/evo2-nim/config/skillspector-baseline.yml +++ b/skills/bionemo-agent-toolkit/skills/evo2-nim/config/skillspector-baseline.yml @@ -6,6 +6,28 @@ version: 1 rules: + - id: "TT3" + path: "*scripts/generate.py" + reason: >- + Reviewed 2026-09-29 against the client and its request-contract tests. + The reported tainted URL is EVO2_NIM_URL, the user-selected local NIM + address, not a credential. It is validated as an HTTP(S) URL without + embedded credentials, query, or fragment. Local requests have no + Authorization header. Hosted requests use the fixed health.api.nvidia.com + endpoint and send NGC_API_KEY only as its required Bearer authentication; + redirects are disabled. Neither request artifacts nor summaries contain + the key. This exception covers the documented inference client only. + - id: "LP1" + path: "*scripts/generate.py" + reason: >- + Reviewed 2026-09-29 against the client and its request-contract tests. + The agent invokes this client through the declared Bash tool. Its + documented capabilities are reading NGC_API_KEY/NIM_API_MODE/EVO2_NIM_URL, + submitting the user's DNA to the chosen Evo 2 endpoint, and saving outputs + through the declared filesystem tools. The scanner maps Bash only to + shell and requires separate Env/WebFetch tool names even though this + Python client does not call either tool. The environment and network + operations are the explicitly requested inference workflow. - id: "PE3" path: "*SKILL.md" reason: >- diff --git a/skills/bionemo-agent-toolkit/skills/evo2-nim/evals/evals.json b/skills/bionemo-agent-toolkit/skills/evo2-nim/evals/evals.json index 3e2ebb5..81c4714 100644 --- a/skills/bionemo-agent-toolkit/skills/evo2-nim/evals/evals.json +++ b/skills/bionemo-agent-toolkit/skills/evo2-nim/evals/evals.json @@ -7,41 +7,13 @@ "expected_output": "A successfully executed hosted Evo 2 generation request with Bearer auth and exact request fields, plus actual response-derived DNA, sampled probabilities, elapsed timing, and saved JSON and FASTA artifacts.", "files": [], "assertions": [ - { - "id": "hosted-request-executed", - "description": "Executes the hosted request instead of only writing code", - "check": "Trajectory shows successful execution of the hosted request, and the final response reports actual response-derived DNA, elapsed timing, and saved artifact paths" - }, - { - "id": "hosted-endpoint-url", - "description": "Uses the correct hosted Evo2 generation endpoint", - "check": "Script contains 'https://health.api.nvidia.com/v1/biology/arc/evo2-40b/generate'" - }, - { - "id": "bearer-auth-header", - "description": "Sets hosted Authorization header from NGC_API_KEY", - "check": "Script contains 'Authorization', 'Bearer', and 'NGC_API_KEY'" - }, - { - "id": "generate-fields", - "description": "Uses exact Evo2 generation request field names", - "check": "Script contains 'sequence', 'num_tokens', 'temperature', 'top_k', 'top_p', and 'random_seed'" - }, - { - "id": "sampled-probs-field", - "description": "Requests or handles sampled probabilities using the correct field", - "check": "Script contains 'enable_sampled_probs' and handles 'sampled_probs'" - }, - { - "id": "dna-validation", - "description": "Validates generated DNA alphabet before claiming success", - "check": "Script checks generated sequence characters against A/C/G/T or an explicitly documented DNA alphabet" - }, - { - "id": "save-json-fasta", - "description": "Saves response JSON and generated FASTA output", - "check": "Script writes a .json file and a .fa or .fasta file" - } + "[hosted-request-executed] Executes the hosted request instead of only writing code: Trajectory shows successful execution of the hosted request, and the final response reports actual response-derived DNA, elapsed timing, and saved artifact paths", + "[hosted-endpoint-url] Uses the correct hosted Evo2 generation endpoint: Script contains 'https://health.api.nvidia.com/v1/biology/arc/evo2-40b/generate'", + "[bearer-auth-header] Sets hosted Authorization header from NGC_API_KEY: Script contains 'Authorization', 'Bearer', and 'NGC_API_KEY'", + "[generate-fields] Uses exact Evo2 generation request field names: Script contains 'sequence', 'num_tokens', 'temperature', 'top_k', 'top_p', and 'random_seed'", + "[sampled-probs-field] Requests or handles sampled probabilities using the correct field: Script contains 'enable_sampled_probs' and handles 'sampled_probs'", + "[dna-validation] Validates generated DNA alphabet before claiming success: Script checks generated sequence characters against A/C/G/T or an explicitly documented DNA alphabet", + "[save-json-fasta] Saves response JSON and generated FASTA output: Script writes a .json file and a .fa or .fasta file" ] } ], @@ -53,36 +25,12 @@ "expected_output": "Docker startup and health-check commands using the Evo2 image, repo env contract, FP8-capable GPU guidance, optional NIM_VARIANT, local no-auth inference, and an EVO2_NIM_URL-aware generation request with a localhost fallback.", "files": [], "assertions": [ - { - "id": "env-contract", - "description": "Uses repo env contract including .env, NGC_API_KEY/NVIDIA_API_KEY fallback, and LOCAL_NIM_CACHE", - "check": "Output mentions '.env', 'NGC_API_KEY', 'NVIDIA_API_KEY', and 'LOCAL_NIM_CACHE'" - }, - { - "id": "docker-image", - "description": "Uses the correct Evo2 Docker image", - "check": "Output contains 'nvcr.io/nim/arc/evo2:2'" - }, - { - "id": "cache-mount", - "description": "Mounts local cache to the documented container cache target", - "check": "Output contains '/opt/nim/.cache' and 'LOCAL_NIM_CACHE'" - }, - { - "id": "variant-gpu-guidance", - "description": "Documents optional NIM_VARIANT=7b, GPU selection, FP8-compatible hardware, and 40B memory requirements", - "check": "Output contains 'NIM_VARIANT', 'NIM_TEST_GPUS', 'FP8', 40B guidance for 2x H100 80GB or 1x H200 141GB, and 7B fallback guidance such as H100, H200, RTX 6000 Ada, or L40S; it must not present A100 as compatible for local Evo2" - }, - { - "id": "health-check", - "description": "Polls local readiness before inference", - "check": "Output builds the readiness URL from EVO2_NIM_URL with localhost:8000 only as a fallback" - }, - { - "id": "local-no-auth-endpoint", - "description": "Uses local generation endpoint with no Authorization header", - "check": "Script builds the local generation endpoint from EVO2_NIM_URL and does not send 'Authorization' to local inference" - } + "[env-contract] Uses repo env contract including .env, NGC_API_KEY/NVIDIA_API_KEY fallback, and LOCAL_NIM_CACHE: Output mentions '.env', 'NGC_API_KEY', 'NVIDIA_API_KEY', and 'LOCAL_NIM_CACHE'", + "[docker-image] Uses the correct Evo2 Docker image: Output contains 'nvcr.io/nim/arc/evo2:2'", + "[cache-mount] Mounts local cache to the documented container cache target: Output contains '/opt/nim/.cache' and 'LOCAL_NIM_CACHE'", + "[variant-gpu-guidance] Documents optional NIM_VARIANT=7b, GPU selection, FP8-compatible hardware, and 40B memory requirements: Output contains 'NIM_VARIANT', 'NIM_TEST_GPUS', 'FP8', 40B guidance for 2x H100 80GB or 1x H200 141GB, and 7B fallback guidance such as H100, H200, RTX 6000 Ada, or L40S; it must not present A100 as compatible for local Evo2", + "[health-check] Polls local readiness before inference: Output builds the readiness URL from EVO2_NIM_URL with localhost:8000 only as a fallback", + "[local-no-auth-endpoint] Uses local generation endpoint with no Authorization header: Script builds the local generation endpoint from EVO2_NIM_URL and does not send 'Authorization' to local inference" ] }, { @@ -91,36 +39,12 @@ "expected_output": "A local Python script that calls the Evo2 forward endpoint, decodes the base64 NPZ response with numpy, saves it, and prints tensor names, shapes, dtypes, and finite-value summaries.", "files": [], "assertions": [ - { - "id": "local-forward-endpoint", - "description": "Uses the correct local forward endpoint", - "check": "Script builds the local forward endpoint from EVO2_NIM_URL with localhost:8000 only as a fallback" - }, - { - "id": "forward-fields", - "description": "Uses exact forward request fields", - "check": "Script contains 'sequence' and 'output_layers'" - }, - { - "id": "requested-layers", - "description": "Requests the user-specified Evo2 layer names", - "check": "Script contains 'output_layer' and 'decoder.layers.3.self_attention'" - }, - { - "id": "base64-npz-decode", - "description": "Decodes base64 NPZ response data", - "check": "Script contains 'base64' and 'np.load' or 'numpy.load'" - }, - { - "id": "save-npz", - "description": "Saves decoded tensor artifacts as NPZ", - "check": "Script writes a .npz file" - }, - { - "id": "finite-summary", - "description": "Checks or reports tensor shape, dtype, and finite numeric values", - "check": "Script reports 'shape' and 'dtype' and checks 'isfinite' or prints numeric summaries" - } + "[local-forward-endpoint] Uses the correct local forward endpoint: Script builds the local forward endpoint from EVO2_NIM_URL with localhost:8000 only as a fallback", + "[forward-fields] Uses exact forward request fields: Script contains 'sequence' and 'output_layers'", + "[requested-layers] Requests the user-specified Evo2 layer names: Script contains 'output_layer' and 'decoder.layers.3.self_attention'", + "[base64-npz-decode] Decodes base64 NPZ response data: Script contains 'base64' and 'np.load' or 'numpy.load'", + "[save-npz] Saves decoded tensor artifacts as NPZ: Script writes a .npz file", + "[finite-summary] Checks or reports tensor shape, dtype, and finite numeric values: Script reports 'shape' and 'dtype' and checks 'isfinite' or prints numeric summaries" ] } ] diff --git a/skills/bionemo-agent-toolkit/skills/evo2-nim/scripts/generate.py b/skills/bionemo-agent-toolkit/skills/evo2-nim/scripts/generate.py new file mode 100644 index 0000000..0f6e6b2 --- /dev/null +++ b/skills/bionemo-agent-toolkit/skills/evo2-nim/scripts/generate.py @@ -0,0 +1,171 @@ +#!/usr/bin/env python3 +"""Execute one Evo 2 generation request and save validated, reproducible outputs.""" + +from __future__ import annotations + +import argparse +import json +import math +import os +from itertools import groupby +from pathlib import Path +import sys +import time +from urllib.parse import urlsplit + +import requests + + +HOSTED_URL = "https://health.api.nvidia.com/v1/biology/arc/evo2-40b/generate" + + +def clean_dna(value: str) -> str: + sequence = "".join(value.upper().split()) + if not sequence or set(sequence) - set("ACGT"): + raise ValueError("Provide a nonempty A/C/G/T sequence; ambiguity codes need an explicit modeling choice") + return sequence + + +def number(value: object, label: str, minimum: float, maximum: float = math.inf) -> float: + if isinstance(value, bool) or not isinstance(value, (int, float)): + raise ValueError(f"{label} must be numeric") + if not math.isfinite(value) or not minimum <= value <= maximum: + raise ValueError(f"{label} is outside the allowed range") + return float(value) + + +def validate_result(result: object, num_tokens: int) -> dict: + if not isinstance(result, dict): + raise ValueError("Expected a JSON object from Evo 2") + sequence = result.get("sequence") + if not isinstance(sequence, str) or not sequence or set(sequence.upper()) - set("ACGT"): + raise ValueError("Response sequence must be nonempty A/C/G/T DNA") + if len(sequence) != num_tokens: + raise ValueError(f"Requested {num_tokens} new bases, received {len(sequence)}; inspect the raw response") + probs = result.get("sampled_probs") + if not isinstance(probs, list) or len(probs) != len(sequence): + raise ValueError("sampled_probs must contain one probability per generated base") + for value in probs: + number(value, "sampled_probs", 0, 1) + number(result.get("elapsed_ms"), "elapsed_ms", 0) + timings = result.get("elapsed_ms_per_token") + if timings is not None: + if not isinstance(timings, list) or len(timings) != len(sequence): + raise ValueError("elapsed_ms_per_token must contain one timing per generated base") + for value in timings: + number(value, "elapsed_ms_per_token", 0) + dna = sequence.upper() + return { + "generated_bases": len(sequence), + "gc_fraction": (dna.count("G") + dna.count("C")) / len(dna), + "ambiguous_base_fraction": 0.0, + "longest_homopolymer": max(sum(1 for _ in group) for _, group in groupby(dna)), + "sampled_probs": { + "count": len(probs), "min": min(probs), "max": max(probs), + "mean": sum(probs) / len(probs), "valid": True, + }, + "elapsed_ms": result["elapsed_ms"], + "per_token_timing_available": timings is not None, + } + + +def generate(args: argparse.Namespace) -> dict: + sequence = clean_dna(args.sequence) + if args.num_tokens < 1: + raise ValueError("num_tokens must be positive") + number(args.temperature, "temperature", 0, 1.3) + number(args.top_k, "top_k", 0, 6) + number(args.top_p, "top_p", 0, 1) + number(args.timeout, "timeout", 1) + mode = args.mode or os.getenv("NIM_API_MODE") or ("local" if os.getenv("EVO2_NIM_URL") else "hosted") + if mode not in {"hosted", "local"}: + raise ValueError("NIM_API_MODE must be hosted or local") + headers = {"Content-Type": "application/json"} + if mode == "hosted": + key = os.getenv("NGC_API_KEY") + if not key: + raise ValueError("Set NGC_API_KEY in the environment for hosted Evo 2") + headers["Authorization"] = f"Bearer {key}" + url = HOSTED_URL + else: + base = os.getenv("EVO2_NIM_URL", "http://localhost:8000").rstrip("/") + parsed = urlsplit(base) + if parsed.scheme not in {"http", "https"} or not parsed.netloc or parsed.username or parsed.password: + raise ValueError("EVO2_NIM_URL must be an HTTP(S) service URL without embedded credentials") + if parsed.query or parsed.fragment: + raise ValueError("EVO2_NIM_URL must not include a query or fragment") + url = f"{base}/biology/arc/evo2/generate" + + payload = { + "sequence": sequence, "num_tokens": args.num_tokens, + "temperature": args.temperature, "top_k": args.top_k, "top_p": args.top_p, + "random_seed": args.seed, "enable_sampled_probs": True, + "enable_elapsed_ms_per_token": True, + } + output = args.output_dir.resolve() + paths = {name: output / filename for name, filename in { + "request": "request.json", "response": "response.json", + "fasta": "generated.fasta", "metrics": "metrics.json", + }.items()} + if any(path.exists() for path in paths.values()): + raise FileExistsError("Output files already exist; choose a new --output-dir for this request") + output.mkdir(parents=True, exist_ok=True) + paths["request"].write_text(json.dumps(payload, indent=2) + "\n") + # Keep the endpoint and request visible without ever printing the credential. + print(json.dumps({ + "event": "request", "method": "POST", "url": url, "payload": payload, + "authentication": "Authorization: Bearer from NGC_API_KEY" if mode == "hosted" else "none", + }), flush=True) + start = time.monotonic() + response = requests.post(url, headers=headers, json=payload, timeout=(10, args.timeout), allow_redirects=False) + wall_ms = round((time.monotonic() - start) * 1000) + if response.status_code != 200: + raise RuntimeError(f"Evo 2 returned HTTP {response.status_code}; generation was not completed") + result = response.json() + # Preserve the actual response even if validation fails; never manufacture missing fields. + paths["response"].write_text(json.dumps(result, indent=2, allow_nan=False) + "\n") + metrics = validate_result(result, args.num_tokens) + fasta = f">evo2_generated seed={args.seed}\n{result['sequence']}\n" + paths["fasta"].write_text(fasta) + # Verify persisted content before presenting the run as complete. + if json.loads(paths["response"].read_text()) != result or paths["fasta"].read_text() != fasta: + raise RuntimeError("Saved artifacts do not match the response") + metrics.update({ + "status": "completed", "http_status": response.status_code, "mode": mode, + "endpoint": url, "wall_ms": wall_ms, "random_seed": args.seed, + "artifacts": {name: str(path) for name, path in paths.items()}, + }) + paths["metrics"].write_text(json.dumps(metrics, indent=2) + "\n") + summary = dict(metrics) + sequence = result["sequence"] + summary["sequence" if len(sequence) <= 256 else "sequence_preview"] = sequence[:256] + print(json.dumps(summary, indent=2), flush=True) + return summary + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--sequence", required=True, help="DNA prompt; case and whitespace are normalized") + parser.add_argument("--mode", choices=["hosted", "local"], help="Explicit mode; otherwise use NIM_API_MODE/EVO2_NIM_URL") + parser.add_argument("--num-tokens", type=int, default=100) + parser.add_argument("--temperature", type=float, default=0.7) + parser.add_argument("--top-k", type=int, default=3) + parser.add_argument("--top-p", type=float, default=0.0) + parser.add_argument("--seed", type=int, default=1, help="Development reproducibility seed") + parser.add_argument("--timeout", type=float, default=180, help="Response read timeout in seconds") + parser.add_argument("--output-dir", type=Path, required=True) + args = parser.parse_args() + try: + generate(args) + except requests.RequestException as exc: + # Exception strings can include headers or URLs; report only the exception type. + print(f"Evo 2 request failed ({type(exc).__name__}); no successful generation to report", file=sys.stderr) + return 1 + except (ValueError, RuntimeError, OSError) as exc: + print(str(exc), file=sys.stderr) + return 1 + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_evo2_generate.py b/tests/test_evo2_generate.py new file mode 100644 index 0000000..27b0326 --- /dev/null +++ b/tests/test_evo2_generate.py @@ -0,0 +1,161 @@ +"""Client contract tests with synthetic responses; no model or API credentials required.""" + +import argparse +from contextlib import redirect_stdout +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +import importlib.util +import io +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import threading +import unittest +from unittest.mock import patch + +import requests + + +SCRIPT = Path(__file__).resolve().parents[1] / "nim-skills" / "evo2-nim" / "scripts" / "generate.py" +spec = importlib.util.spec_from_file_location("evo2_generate", SCRIPT) +client = importlib.util.module_from_spec(spec) +spec.loader.exec_module(client) + + +def response_data(): + return {"sequence": "ACGTACGT", "sampled_probs": [0.5] * 8, + "elapsed_ms": 125, "elapsed_ms_per_token": [1.0] * 8} + + +def response(data=None, status=200): + result = requests.Response() + result.status_code = status + result._content = json.dumps(response_data() if data is None else data).encode() + return result + + +class GenerateTests(unittest.TestCase): + def setUp(self): + self.temp = tempfile.TemporaryDirectory() + self.addCleanup(self.temp.cleanup) + self.output = Path(self.temp.name) / "run" + self.args = argparse.Namespace(sequence=" acgt\nacgt ", mode="hosted", num_tokens=8, + temperature=0.7, top_k=3, top_p=0.0, seed=42, + timeout=30, output_dir=self.output) + + def test_hosted_request_saves_actual_data_and_never_saves_key(self): + log = io.StringIO() + with patch.dict(os.environ, {"NGC_API_KEY": "test-credential"}, clear=True), \ + patch.object(client.requests, "post", return_value=response()) as post, redirect_stdout(log): + summary = client.generate(self.args) + self.assertEqual(post.call_count, 1) + self.assertEqual(post.call_args.args[0], client.HOSTED_URL) + self.assertEqual(post.call_args.kwargs["headers"]["Authorization"], "Bearer test-credential") + request = json.loads((self.output / "request.json").read_text()) + self.assertEqual(request, post.call_args.kwargs["json"]) + self.assertEqual((request["sequence"], request["random_seed"], request["num_tokens"]), ("ACGTACGT", 42, 8)) + self.assertTrue(request["enable_sampled_probs"]) + self.assertEqual(json.loads((self.output / "response.json").read_text()), response_data()) + self.assertEqual((self.output / "generated.fasta").read_text().splitlines()[1], "ACGTACGT") + self.assertEqual(summary["elapsed_ms"], 125) + self.assertEqual(summary["sampled_probs"]["count"], 8) + self.assertEqual(summary["gc_fraction"], 0.5) + self.assertEqual(summary["status"], "completed") + self.assertNotIn("test-credential", log.getvalue()) + for path in self.output.iterdir(): + self.assertNotIn("test-credential", path.read_text()) + + def test_invalid_input_or_missing_key_never_calls_api(self): + with patch.dict(os.environ, {}, clear=True), patch.object(client.requests, "post") as post: + with self.assertRaisesRegex(ValueError, "NGC_API_KEY"): + client.generate(self.args) + for sequence in ["", "NNNN", "ACGT>header"]: + self.args.sequence = sequence + with self.assertRaisesRegex(ValueError, "A/C/G/T"): + client.generate(self.args) + post.assert_not_called() + self.assertFalse(self.output.exists()) + + def test_invalid_responses_cannot_produce_success_artifacts(self): + bad_values = [("sequence", ""), ("sequence", "ACGTACGN"), ("sequence", "ACGT"), + ("sampled_probs", None), ("sampled_probs", [0.5]), + ("sampled_probs", [1.1] * 8), ("sampled_probs", [True] * 8), + ("elapsed_ms", -1), ("elapsed_ms_per_token", [1.0])] + for index, (field, value) in enumerate(bad_values): + with self.subTest(field=field, value=value): + self.args.output_dir = self.output / str(index) + data = response_data() + data[field] = value + with patch.dict(os.environ, {"NGC_API_KEY": "test-key"}, clear=True), \ + patch.object(client.requests, "post", return_value=response(data)), \ + redirect_stdout(io.StringIO()): + with self.assertRaises(ValueError): + client.generate(self.args) + self.assertTrue((self.args.output_dir / "response.json").is_file()) + self.assertFalse((self.args.output_dir / "generated.fasta").exists()) + self.assertFalse((self.args.output_dir / "metrics.json").exists()) + + def test_http_failure_and_pending_response_are_not_success(self): + for status in [202, 302, 401, 503]: + with self.subTest(status=status): + self.args.output_dir = self.output / str(status) + with patch.dict(os.environ, {"NGC_API_KEY": "test-key"}, clear=True), \ + patch.object(client.requests, "post", return_value=response(status=status)) as post, \ + redirect_stdout(io.StringIO()): + with self.assertRaisesRegex(RuntimeError, f"HTTP {status}"): + client.generate(self.args) + self.assertEqual(post.call_count, 1) + self.assertFalse((self.args.output_dir / "generated.fasta").exists()) + + def test_existing_outputs_are_not_overwritten_or_resubmitted(self): + self.output.mkdir() + existing = self.output / "response.json" + existing.write_text('{"existing": true}\n') + with patch.dict(os.environ, {"NGC_API_KEY": "test-key"}, clear=True), patch.object(client.requests, "post") as post: + with self.assertRaises(FileExistsError): + client.generate(self.args) + post.assert_not_called() + self.assertEqual(existing.read_text(), '{"existing": true}\n') + + def test_cli_executes_local_request_and_reports_persisted_results(self): + received = [] + + class Handler(BaseHTTPRequestHandler): + def do_POST(self): + received.append((self.path, dict(self.headers), json.loads(self.rfile.read(int(self.headers["Content-Length"]))))) + body = json.dumps(response_data()).encode() + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + def log_message(self, *_): + pass + + server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + try: + env = {"EVO2_NIM_URL": f"http://127.0.0.1:{server.server_port}", "NGC_API_KEY": "must-not-be-sent"} + result = subprocess.run([sys.executable, str(SCRIPT), "--mode", "local", "--sequence", "ACGT", + "--num-tokens", "8", "--output-dir", str(self.output)], + env=env, capture_output=True, text=True, timeout=15) + finally: + server.shutdown() + server.server_close() + thread.join(timeout=2) + self.assertEqual(result.returncode, 0, result.stderr) + self.assertEqual(len(received), 1) + self.assertEqual(received[0][0], "/biology/arc/evo2/generate") + self.assertNotIn("Authorization", received[0][1]) + self.assertEqual(received[0][2]["sequence"], "ACGT") + self.assertIn('"sequence": "ACGTACGT"', result.stdout) + self.assertIn('"elapsed_ms": 125', result.stdout) + self.assertEqual(json.loads((self.output / "metrics.json").read_text())["generated_bases"], 8) + + +if __name__ == "__main__": + unittest.main() From ef0e5bb39551bb238524fc2379791b8dd281e7a7 Mon Sep 17 00:00:00 2001 From: Ohad Mosafi Date: Tue, 29 Sep 2026 11:29:35 -0700 Subject: [PATCH 2/7] Document deferred Evo2 dotenv assertion false positive Signed-off-by: Ohad Mosafi --- nim-skills/evo2-nim/config/skillspector-baseline.yml | 9 +++++++++ .../skills/evo2-nim/config/skillspector-baseline.yml | 9 +++++++++ 2 files changed, 18 insertions(+) diff --git a/nim-skills/evo2-nim/config/skillspector-baseline.yml b/nim-skills/evo2-nim/config/skillspector-baseline.yml index 5d873cb..c33b427 100644 --- a/nim-skills/evo2-nim/config/skillspector-baseline.yml +++ b/nim-skills/evo2-nim/config/skillspector-baseline.yml @@ -28,6 +28,15 @@ rules: shell and requires separate Env/WebFetch tool names even though this Python client does not call either tool. The environment and network operations are the explicitly requested inference workflow. + - id: "PE3" + path: "*evals/evals.json" + reason: >- + Reviewed 2026-09-29. The finding is the literal .env filename in a + deferred local-Docker evaluation assertion. It describes the same + user-owned repo-root dotenv setup already reviewed in SKILL.md and + references/api.md; this JSON is test data and does not load credentials. + The active hosted evaluation reads NGC_API_KEY from the environment. + No credential-file access is added to the executable generation client. - id: "PE3" path: "*SKILL.md" reason: >- diff --git a/skills/bionemo-agent-toolkit/skills/evo2-nim/config/skillspector-baseline.yml b/skills/bionemo-agent-toolkit/skills/evo2-nim/config/skillspector-baseline.yml index 5d873cb..c33b427 100644 --- a/skills/bionemo-agent-toolkit/skills/evo2-nim/config/skillspector-baseline.yml +++ b/skills/bionemo-agent-toolkit/skills/evo2-nim/config/skillspector-baseline.yml @@ -28,6 +28,15 @@ rules: shell and requires separate Env/WebFetch tool names even though this Python client does not call either tool. The environment and network operations are the explicitly requested inference workflow. + - id: "PE3" + path: "*evals/evals.json" + reason: >- + Reviewed 2026-09-29. The finding is the literal .env filename in a + deferred local-Docker evaluation assertion. It describes the same + user-owned repo-root dotenv setup already reviewed in SKILL.md and + references/api.md; this JSON is test data and does not load credentials. + The active hosted evaluation reads NGC_API_KEY from the environment. + No credential-file access is added to the executable generation client. - id: "PE3" path: "*SKILL.md" reason: >- From b589a53acbf955728d9926537dc42cb5e39762c8 Mon Sep 17 00:00:00 2001 From: nvskills-svc-account Date: Tue, 29 Sep 2026 19:42:44 +0000 Subject: [PATCH 3/7] Attach NVSkills validation signatures Signed-off-by: nvskills-svc-account --- .../skills/evo2-nim/BENCHMARK.md | 123 ++++++++++++++++++ .../skills/evo2-nim/skill-card.md | 86 ++++++++++++ .../skills/evo2-nim/skill.oms.sig | 1 + 3 files changed, 210 insertions(+) create mode 100644 skills/bionemo-agent-toolkit/skills/evo2-nim/BENCHMARK.md create mode 100644 skills/bionemo-agent-toolkit/skills/evo2-nim/skill-card.md create mode 100644 skills/bionemo-agent-toolkit/skills/evo2-nim/skill.oms.sig diff --git a/skills/bionemo-agent-toolkit/skills/evo2-nim/BENCHMARK.md b/skills/bionemo-agent-toolkit/skills/evo2-nim/BENCHMARK.md new file mode 100644 index 0000000..84ea3da --- /dev/null +++ b/skills/bionemo-agent-toolkit/skills/evo2-nim/BENCHMARK.md @@ -0,0 +1,123 @@ +# Skill Benchmark: evo2-nim + +> ✅ **Overall verdict: PASS — Recommended for publication** + +## Publication Recommendation + +Recommended for publication based on the completed evaluation evidence in this report. + +## Evaluation Metadata + +- Skill: `evo2-nim` +- Evaluation date: 2026-09-29 +- Evaluator version: `1.5.6` +- Agents: Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`), Codex (`openai/openai/gpt-5.5`) +- Tasks: 1 evaluation tasks (1 positive) +- Dataset digest: `sha256:f6c50c0077eda60ee7fc9ab49ca7a1dcb84f3deeee873c4837c89c794f3ea9f6` (skill-evaluator-dataset-snapshot/1) +- Attempts per task: 3 +- Environment: `k8s-sandbox` +- Tier 2 evidence: required for publication +- Tier 3 evidence: required for publication + +Each task attempt ran in its own isolated sandbox pod. + +## What This Report Answers + +The three-tier evaluation checks whether the skill: + +- is safe to use; +- produces correct answers; +- is discovered and activated when needed; +- helps the agent complete the user's goal and expected workflow; and +- avoids wasted skill and tool usage. + +## Results at a Glance + +| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 94.1% — baseline ran, but no comparable score was available; uplift unavailable | 90.9% — baseline ran, but no comparable score was available; uplift unavailable | +| Security | 100.0% → 100.0% (±0.0 points) | 100.0% → 100.0% (±0.0 points) | +| Correctness | 100.0% → 100.0% (±0.0 points) | 100.0% → 100.0% (±0.0 points) | +| Discoverability | 95.0% — baseline ran, but no comparable score was available; uplift unavailable | 85.0% — baseline ran, but no comparable score was available; uplift unavailable | +| Effectiveness | 52.9% → 100.0% (+47.1 points) | 50.7% → 100.0% (+49.3 points) | +| Efficiency | 75.4% — baseline ran, but no comparable score was available; uplift unavailable | 69.7% — baseline ran, but no comparable score was available; uplift unavailable | + +**How to read this table:** baseline is the same task attempted without the target skill. Scores are rounded to one decimal; threshold-adjacent values use additional precision so their displayed band matches the verdict. Uplift is derived from those displayed scores and shown in percentage points. + +Example: `47.0% → 92.0% (+45.0 points)` means the skill-assisted run scored 92.0%, 45.0 percentage points above its 47.0% no-skill baseline. + +## Token Usage + +Actual Tier 3 execution usage is reported for every observed agent/case pair and both conditions. + +| Agent | Dataset case | With skill | Without skill | Delta | Change | Coverage | +|---|---|---:|---:|---:|---:|---| +| claude-code | All cases | 490,498 | 676,561 | -186,063 | -27.50% | skill 1/1; base 1/1 | +| claude-code | 1 | 490,498 | 676,561 | -186,063 | -27.50% | skill 1/1; base 1/1 | +| codex | All cases | 205,950 | 375,153 | -169,203 | -45.10% | skill 1/1; base 1/1 | +| codex | 1 | 205,950 | 375,153 | -169,203 | -45.10% | skill 1/1; base 1/1 | +| ALL AGENTS | Dataset aggregate | 696,448 | 1,051,714 | -355,266 | -33.78% | skill 2/2; base 2/2 | + +Prompt tokens include cached reads, so total tokens are `prompt + completion` (cached is not added twice). The Efficiency score uses `(prompt - cached) + completion`. N/A means the relevant trajectory counters were not available; coverage is never estimated. + +## Tier Status + +| Tier | Purpose | Status | Evidence | +|---|---|---|---| +| Tier 1 | Static validation | **PASSED WITH OBSERVATIONS** | 11 validator(s); 26 finding(s) | +| Tier 2 | Semantic deduplication | **PASSED WITH OBSERVATIONS** | 2 validator(s); 1 finding(s) | +| Tier 3 | Live agent evaluation | **PASS** | 2 agent(s); 1 task(s) | + +## Findings and Observations + +
+Show detailed findings and successful checks + +- **HIGH** DUPLICATE/duplicate: Duplicate content found across SKILL.md and references/api.md: + "# 40B default: 0,1 for 2x H100; set 0 for a single H200." in SKILL.md (lines 123-127) + vs "# For 7B: export NIM_VARIANT=7b; export NIM_TEST_GPUS="${NIM_TEST_GPUS:-0}"" in SKILL.md (lines 128-149) + vs "# 40B default: use 0,1 for 2x H100 80 GB; set NIM_TEST_GPUS=0 for a single H200." in references/api.md (lines 176-180) + vs "# Optional: export NIM_VARIANT=7b and add `-e NIM_VARIANT` for the 7B model." in references/api.md (lines 181-201) (`SKILL.md:123`) +- **MEDIUM** QUALITY/quality_correctness: No documented scripts in table format (`skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: Instructions don't mention 'run_script' (`skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.author' (`skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md`) +- **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.tags' (`skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md`) +- 22 additional finding(s) are available in the full evaluation artifacts. + +
+ +## Scoring Methodology + +
+Show dimension definitions, source signals, and thresholds + +| Dimension | Question | Scored signals | +|---|---|---| +| Security | Is it safe to use? | `security` (100%) | +| Correctness | Is the answer correct? | `accuracy` (100%) | +| Discoverability | Was the right skill loaded when needed? | `skill_execution` (100%) | +| Effectiveness | Did the skill help complete the task? | `goal_accuracy` (50%) + `behavior_check` (50%) | +| Efficiency | Did it avoid wasted tool calls and token usage? | `skill_efficiency` (50%) + `token_efficiency` (50%) | + +- Dimension bands: PASS at 50% or above; NEUTRAL from 40% to below 50%; FAIL below 40%. +- Overall Tier 3 lift: PASS at +5 points or more; FAIL at -10 points or less; values between those bands are NEUTRAL. +- Overall verdict: PASS only when every configured dimension passes for at least one supported agent. Lift is reported as diagnostic evidence and does not override this gate. +- The 50% attempt pass threshold is a separate per-task gate; it is not the dimension pass threshold. +- Effectiveness is the equal-weight mean of goal completion (`goal_accuracy`) and expected workflow adherence (`behavior_check`). +- Efficiency is 50% tool-call productivity (the backward-compatible `skill_efficiency` wire id) and 50% `token_efficiency`. Positive-case skill routing is scored under Discoverability, not Efficiency; a negative case without a routing target is N/A. N/A sources are omitted, remaining weights are renormalized, and the dimension is marked partial. + +Signals present in this run: + +- `security` (Security): unsafe operations, secret leakage, and unauthorized access. +- `skill_execution` (Skill Execution): whether the expected skill was selected, decoys were avoided, and the workflow executed. +- `skill_efficiency` (Tool Productivity): tool-call productivity (legacy wire id; routing is scored under Discoverability). +- `accuracy` (Accuracy): final-answer correctness against the reference answer. +- `goal_accuracy` (Goal Accuracy): whether the user's goal was achieved. +- `behavior_check` (Behavior Check): whether the expected workflow behavior was followed. +- `token_efficiency` (Token Efficiency): actual uncached prompt plus completion usage (50% of Efficiency). + +
+ +## Freshness + +Regenerate this benchmark when the skill, evaluation dataset, target agent/model, evaluator version, environment, or scoring policy changes. diff --git a/skills/bionemo-agent-toolkit/skills/evo2-nim/skill-card.md b/skills/bionemo-agent-toolkit/skills/evo2-nim/skill-card.md new file mode 100644 index 0000000..415807a --- /dev/null +++ b/skills/bionemo-agent-toolkit/skills/evo2-nim/skill-card.md @@ -0,0 +1,86 @@ +## Description:
+Generate and analyze DNA sequences using NVIDIA's Evo 2 BioNeMo NIM microservice.
+ +This skill is ready for commercial/non-commercial use.
+ +## Owner +NVIDIA
+ +### License/Terms of Use:
+Apache-2.0 AND CC-BY-4.0
+## Use Case:
+Developers and engineers use this skill for DNA sequence generation, genomic analysis, and layer-output extraction via NVIDIA BioNeMo NIM, in both hosted API and local Docker deployment modes.
+ +### Deployment Geography for Use:
+Global
+ +## Requirements / Dependencies:
+**Requires API Key or External Credential:** [Yes]
+**Credential Type(s):** [API key]
+ +Do not include secrets in prompts/logs/output; use least-privilege credentials; rotate keys as appropriate.
+ +## Known Risks and Mitigations:
+Risk: Review before execution as proposals could introduce incorrect or misleading guidance into skills.
+Mitigation: Review and scan skill before deployment.
+ +## Reference(s):
+- [Evo 2 NIM API Reference](references/api.md)
+- [Genomic Use Cases and Interpretation](references/science.md)
+- [Generation and Forward Parameter Effects](references/parameters.md)
+- [Validation Checks](references/validation.md)
+- [Request Pattern Examples](references/examples.md)
+ + +## Skill Output:
+**Output Type(s):** [Code, Files, Shell commands]
+**Output Format:** [Markdown with inline bash and Python code blocks]
+**Output Parameters:** [1D]
+**Other Properties Related to Output:** [None]
+ +## Evaluation Agents Used:
+- Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`)
+- Codex (`openai/openai/gpt-5.5`)
+ + + +## Evaluation Tasks:
+1 evaluation task (1 positive), 3 attempts per task, each in an isolated k8s-sandbox pod.
+ +## Evaluation Metrics Used:
+Reported benchmark dimensions:
+- Security: Whether the skill is safe to use, checking for unsafe operations, secret leakage, and unauthorized access.
+- Correctness: Whether the final answer is correct against the reference answer.
+- Discoverability: Whether the expected skill was selected, decoys were avoided, and the workflow executed.
+- Effectiveness: Whether the skill helped complete the user's goal (50% goal accuracy + 50% behavior check).
+- Efficiency: Whether wasted tool calls and token usage were avoided (50% tool productivity + 50% token efficiency).
+ +Underlying evaluation signals used in this run:
+- `security`: Checks for unsafe operations, secret leakage, and unauthorized access.
+- `accuracy`: Final-answer correctness against the reference answer.
+- `skill_execution`: Whether the expected skill was selected and the workflow executed.
+- `goal_accuracy`: Whether the user's goal was achieved.
+- `behavior_check`: Whether the expected workflow behavior was followed.
+- `skill_efficiency`: Tool-call productivity; routing is scored under Discoverability.
+- `token_efficiency`: Actual uncached prompt plus completion token usage.
+ + + +## Evaluation Results:
+| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | +|---|---:|---:| +| Overall | 94.1% | 90.9% | +| Security | 100.0% → 100.0% (±0.0 points) | 100.0% → 100.0% (±0.0 points) | +| Correctness | 100.0% → 100.0% (±0.0 points) | 100.0% → 100.0% (±0.0 points) | +| Discoverability | 95.0% | 85.0% | +| Effectiveness | 52.9% → 100.0% (+47.1 points) | 50.7% → 100.0% (+49.3 points) | +| Efficiency | 75.4% | 69.7% | + +## Skill Version(s):
+0.1.0 (source: pyproject.toml)
+ +## Ethical Considerations:
+NVIDIA believes Trustworthy AI is a shared responsibility and we have established policies and practices to enable development for a wide array of AI applications. When downloaded or used in accordance with our terms of service, developers should work with their internal team to ensure this skill meets requirements for the relevant industry and use case and addresses unforeseen product misuse.
+ +(For Release on NVIDIA Platforms Only)
+Please report quality, risk, security vulnerabilities or NVIDIA AI Concerns [here](https://app.intigriti.com/programs/nvidia/nvidiavdp/detail).
diff --git a/skills/bionemo-agent-toolkit/skills/evo2-nim/skill.oms.sig b/skills/bionemo-agent-toolkit/skills/evo2-nim/skill.oms.sig new file mode 100644 index 0000000..c130d34 --- /dev/null +++ b/skills/bionemo-agent-toolkit/skills/evo2-nim/skill.oms.sig @@ -0,0 +1 @@ +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAiZXZvMi1uaW0iLAogICAgICAiZGlnZXN0IjogewogICAgICAgICJzaGEyNTYiOiAiMTM5ZmEyYzM1N2FhOTFiMDNkNzQwZjE0ODYxY2IwNTVhNDhlMTU4YjYzY2FlMWY3ZDY2NTJiZTMzYTZlZmZhZiIKICAgICAgfQogICAgfQogIF0sCiAgInByZWRpY2F0ZVR5cGUiOiAiaHR0cHM6Ly9tb2RlbF9zaWduaW5nL3NpZ25hdHVyZS92MS4wIiwKICAicHJlZGljYXRlIjogewogICAgInJlc291cmNlcyI6IFsKICAgICAgewogICAgICAgICJuYW1lIjogIkJFTkNITUFSSy5tZCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiZmFkNGU2YWIxYWRkNWZhMzU3NzgyZTBjMzIxYjQxNDAwNGM1MjQyMmI1YTBlZmZlMjVmN2E4YWZhOTQzZmQyMyIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogIlNLSUxMLm1kIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICJhZGI2ODlhYzcwNDM5YjY1MzdiY2EzNThmMTFlNWU1ZDVmOGY3ZDA3ZjFjYTJjMDc3YjI1YjM0YzM1YTJmZWRmIgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAiY29uZmlnL3NraWxsc3BlY3Rvci1iYXNlbGluZS55bWwiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogImUxZDQxOWY0NmY3YzM0ZTM0OGU4ZmQ2OTViZWI5NWZlODMwZjAwZDg4MTBmODBlMzZmYjNiNjc1ZjFmMDcwYmMiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJldmFscy9jb25maWcueW1sIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICJiYTJiOGNmMGVhZDEzYmZiNjVkODEzMDllMTYzMTYyZGJjMmM2YzUwNTg5NjRjNGFlYTQxNGUwMDcwODk4YThhIgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAiZXZhbHMvZXZhbHMuanNvbiIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiMTZkOGU3MzVlYTRiMTJiZWE3M2Y4NzUxZmJjYTFjYTcxMjk0N2Q5ODQyYWRkMzAxMDMzNDI1MGYzY2IzZTc4ZCIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogImV2YWxzL3RyaWdnZXJfZXZhbHMuanNvbiIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiMmVlMmQyZjgyOTg4NGUxNDJiNmU2NGY2YmE0ODJiNjZmMjBhZjFlOGIwNTA3NDdlNWNhNDMxZGQzNDEyYTY4NCIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInJlZmVyZW5jZXMvYXBpLm1kIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICI5Y2Y2MDVjMjViMmExMjEwMmRkYjQwOTQ4YzQxZjY1M2VjNGRiODYzNTRiYjFiMzQ2NjI5YzJhZjk4MjNiZDdiIgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAicmVmZXJlbmNlcy9leGFtcGxlcy5tZCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiZGNlMWQwNzdiNmY2MzNlYjQ1ODRjM2VmODgyMDAxZTFhZDkyMjRiNDZkZTllMTJhZWUwNGRmZGI2Mjc3OGM2MiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInJlZmVyZW5jZXMvcGFyYW1ldGVycy5tZCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiMTYwNjI3N2Q2MGFhMDI5ZWZiMWNlN2M4NzhiNGFkODgwZDdmODI0NjE0YTNmYjc1ZWFjMWI3MDU5YjQ1M2ZiNyIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInJlZmVyZW5jZXMvc2NpZW5jZS5tZCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiZmVhZGM1NmVmZGRmOTZlNDUxNmYxZGRmZTFmZjlmOWMwMmFlYjg2OTIyYWZlMWFlYzNiYWUyM2NmNmQwYWY1OSIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInJlZmVyZW5jZXMvdmFsaWRhdGlvbi5tZCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiNzhlMDY3N2JiZmRlOTM5ODcyNjk0MTM1YmIyMTdhNjgyOTZhNDE4NDA0M2M4ZWNjNWM1MjU4YzRlOTEzNTA1MSIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInNjcmlwdHMvZ2VuZXJhdGUucHkiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogIjFlMjEwZmVlNjhkZGY2NmU0M2VlYTAzNThjYzEwYzI3MDU4OTBhMDA5MjA0ZjFhMzc2NmMzOWNmNDVlOTQ3ZjAiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJza2lsbC1jYXJkLm1kIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICIzNGJlMzQzZGEzZTdiYmNhMjE1ZTYwYTkyZWY4NjAwYzgyMjg3NzJhNmRkM2E4MzA2ZWJmZmE5ZTE3ZTg0N2IyIgogICAgICB9CiAgICBdLAogICAgInNlcmlhbGl6YXRpb24iOiB7CiAgICAgICJhbGxvd19zeW1saW5rcyI6IGZhbHNlLAogICAgICAiaGFzaF90eXBlIjogInNoYTI1NiIsCiAgICAgICJtZXRob2QiOiAiZmlsZXMiLAogICAgICAiaWdub3JlX3BhdGhzIjogWwogICAgICAgICIuZ2l0YXR0cmlidXRlcyIsCiAgICAgICAgIi5naXQiLAogICAgICAgICIuZ2l0aWdub3JlIiwKICAgICAgICAiLmdpdGh1YiIKICAgICAgXQogICAgfQogIH0KfQ==","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGQCMFB5KyBi3908w8LRkCFu2E6iJ/asg6LnjDIx4X3Xo/eN9gdGlOnQ5KEbwRcwbF8CEgIwHwS+TfMJVNnwJ5v5D1Fr88Olu0j/GyfUFNT2kNIJhZFWZDTpy2fTIptYxPwi9HYl","keyid":""}]}} \ No newline at end of file From 93d4978ca72a4345663351fc257c189edd79168b Mon Sep 17 00:00:00 2001 From: Ohad Mosafi Date: Tue, 29 Sep 2026 14:04:36 -0700 Subject: [PATCH 4/7] Fix Evo2 output races and preserve response diagnostics Signed-off-by: Ohad Mosafi --- nim-skills/evo2-nim/SKILL.md | 10 ++- nim-skills/evo2-nim/scripts/generate.py | 16 +++-- pyproject.toml | 2 + .../skills/evo2-nim/SKILL.md | 10 ++- .../skills/evo2-nim/scripts/generate.py | 16 +++-- tests/test_evo2_generate.py | 68 +++++++++++++++++++ uv.lock | 44 ++++++------ 7 files changed, 129 insertions(+), 37 deletions(-) diff --git a/nim-skills/evo2-nim/SKILL.md b/nim-skills/evo2-nim/SKILL.md index 7eca9fd..3a6912c 100644 --- a/nim-skills/evo2-nim/SKILL.md +++ b/nim-skills/evo2-nim/SKILL.md @@ -88,10 +88,14 @@ after a failed request. Set `--timeout` for a longer read if the user requests a larger generation; failed requests are not automatically resubmitted. The client saves `request.json`, the actual `response.json`, `generated.fasta`, -and `metrics.json` in the chosen output directory. It validates the requested -number of generated bases, A/C/G/T alphabet, finite sampled probabilities in +and `metrics.json` in the chosen output directory. It also saves the exact +response body in `response.raw` before checking HTTP status or parsing JSON, +so diagnostics survive malformed JSON and non-finite probability/timing values. +It validates the requested number of generated bases, A/C/G/T alphabet, finite sampled probabilities in `[0, 1]`, and nonnegative timing before printing a successful summary. Existing -outputs are not overwritten; choose a new output directory for each run. +directories are never reused, even if empty. Choose an output directory that +does not exist; the client creates it atomically so concurrent runs cannot +overwrite each other's artifacts. The FASTA contains generated bases only, not the input prompt prepended again. `sampled_probs` is requested by the client and summarized with count/min/max/mean; diff --git a/nim-skills/evo2-nim/scripts/generate.py b/nim-skills/evo2-nim/scripts/generate.py index 0f6e6b2..7fba569 100644 --- a/nim-skills/evo2-nim/scripts/generate.py +++ b/nim-skills/evo2-nim/scripts/generate.py @@ -105,11 +105,14 @@ def generate(args: argparse.Namespace) -> dict: output = args.output_dir.resolve() paths = {name: output / filename for name, filename in { "request": "request.json", "response": "response.json", + "raw_response": "response.raw", "fasta": "generated.fasta", "metrics": "metrics.json", }.items()} - if any(path.exists() for path in paths.values()): - raise FileExistsError("Output files already exist; choose a new --output-dir for this request") - output.mkdir(parents=True, exist_ok=True) + # Directory creation atomically reserves every artifact path for this run. + try: + output.mkdir(parents=True, exist_ok=False) + except FileExistsError as exc: + raise FileExistsError("Output directory already exists; choose a new --output-dir for this request") from exc paths["request"].write_text(json.dumps(payload, indent=2) + "\n") # Keep the endpoint and request visible without ever printing the credential. print(json.dumps({ @@ -119,10 +122,12 @@ def generate(args: argparse.Namespace) -> dict: start = time.monotonic() response = requests.post(url, headers=headers, json=payload, timeout=(10, args.timeout), allow_redirects=False) wall_ms = round((time.monotonic() - start) * 1000) + # Keep the exact body even when HTTP status, JSON parsing, or validation fails. + paths["raw_response"].write_bytes(response.content) if response.status_code != 200: raise RuntimeError(f"Evo 2 returned HTTP {response.status_code}; generation was not completed") result = response.json() - # Preserve the actual response even if validation fails; never manufacture missing fields. + # The parsed JSON supplements the raw body; never manufacture missing fields. paths["response"].write_text(json.dumps(result, indent=2, allow_nan=False) + "\n") metrics = validate_result(result, args.num_tokens) fasta = f">evo2_generated seed={args.seed}\n{result['sequence']}\n" @@ -153,7 +158,8 @@ def main() -> int: parser.add_argument("--top-p", type=float, default=0.0) parser.add_argument("--seed", type=int, default=1, help="Development reproducibility seed") parser.add_argument("--timeout", type=float, default=180, help="Response read timeout in seconds") - parser.add_argument("--output-dir", type=Path, required=True) + parser.add_argument("--output-dir", type=Path, required=True, + help="New directory reserved for this request; must not already exist") args = parser.parse_args() try: generate(args) diff --git a/pyproject.toml b/pyproject.toml index 7e88de9..ed3b2f0 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -26,11 +26,13 @@ requires-python = ">=3.10" # biotite -> */complexa-binder-design/scripts/{pdb_interface,pipeline,preflight_design,pdb_to_boltz_template_cif}.py # numpy -> */complexa-binder-design/scripts/validate_binders.py, */protein-binder-design/scripts/metrics.py # pyyaml -> */complexa-binder-design/scripts/pipeline.py +# requests -> */evo2-nim/scripts/generate.py # Everything else the scripts import is Python standard library. dependencies = [ "biotite>=1.0", "numpy>=1.26", "pyyaml>=6.0", + "requests>=2.28", ] [project.license] diff --git a/skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md b/skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md index 7eca9fd..3a6912c 100644 --- a/skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md +++ b/skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md @@ -88,10 +88,14 @@ after a failed request. Set `--timeout` for a longer read if the user requests a larger generation; failed requests are not automatically resubmitted. The client saves `request.json`, the actual `response.json`, `generated.fasta`, -and `metrics.json` in the chosen output directory. It validates the requested -number of generated bases, A/C/G/T alphabet, finite sampled probabilities in +and `metrics.json` in the chosen output directory. It also saves the exact +response body in `response.raw` before checking HTTP status or parsing JSON, +so diagnostics survive malformed JSON and non-finite probability/timing values. +It validates the requested number of generated bases, A/C/G/T alphabet, finite sampled probabilities in `[0, 1]`, and nonnegative timing before printing a successful summary. Existing -outputs are not overwritten; choose a new output directory for each run. +directories are never reused, even if empty. Choose an output directory that +does not exist; the client creates it atomically so concurrent runs cannot +overwrite each other's artifacts. The FASTA contains generated bases only, not the input prompt prepended again. `sampled_probs` is requested by the client and summarized with count/min/max/mean; diff --git a/skills/bionemo-agent-toolkit/skills/evo2-nim/scripts/generate.py b/skills/bionemo-agent-toolkit/skills/evo2-nim/scripts/generate.py index 0f6e6b2..7fba569 100644 --- a/skills/bionemo-agent-toolkit/skills/evo2-nim/scripts/generate.py +++ b/skills/bionemo-agent-toolkit/skills/evo2-nim/scripts/generate.py @@ -105,11 +105,14 @@ def generate(args: argparse.Namespace) -> dict: output = args.output_dir.resolve() paths = {name: output / filename for name, filename in { "request": "request.json", "response": "response.json", + "raw_response": "response.raw", "fasta": "generated.fasta", "metrics": "metrics.json", }.items()} - if any(path.exists() for path in paths.values()): - raise FileExistsError("Output files already exist; choose a new --output-dir for this request") - output.mkdir(parents=True, exist_ok=True) + # Directory creation atomically reserves every artifact path for this run. + try: + output.mkdir(parents=True, exist_ok=False) + except FileExistsError as exc: + raise FileExistsError("Output directory already exists; choose a new --output-dir for this request") from exc paths["request"].write_text(json.dumps(payload, indent=2) + "\n") # Keep the endpoint and request visible without ever printing the credential. print(json.dumps({ @@ -119,10 +122,12 @@ def generate(args: argparse.Namespace) -> dict: start = time.monotonic() response = requests.post(url, headers=headers, json=payload, timeout=(10, args.timeout), allow_redirects=False) wall_ms = round((time.monotonic() - start) * 1000) + # Keep the exact body even when HTTP status, JSON parsing, or validation fails. + paths["raw_response"].write_bytes(response.content) if response.status_code != 200: raise RuntimeError(f"Evo 2 returned HTTP {response.status_code}; generation was not completed") result = response.json() - # Preserve the actual response even if validation fails; never manufacture missing fields. + # The parsed JSON supplements the raw body; never manufacture missing fields. paths["response"].write_text(json.dumps(result, indent=2, allow_nan=False) + "\n") metrics = validate_result(result, args.num_tokens) fasta = f">evo2_generated seed={args.seed}\n{result['sequence']}\n" @@ -153,7 +158,8 @@ def main() -> int: parser.add_argument("--top-p", type=float, default=0.0) parser.add_argument("--seed", type=int, default=1, help="Development reproducibility seed") parser.add_argument("--timeout", type=float, default=180, help="Response read timeout in seconds") - parser.add_argument("--output-dir", type=Path, required=True) + parser.add_argument("--output-dir", type=Path, required=True, + help="New directory reserved for this request; must not already exist") args = parser.parse_args() try: generate(args) diff --git a/tests/test_evo2_generate.py b/tests/test_evo2_generate.py index 27b0326..0470935 100644 --- a/tests/test_evo2_generate.py +++ b/tests/test_evo2_generate.py @@ -1,6 +1,7 @@ """Client contract tests with synthetic responses; no model or API credentials required.""" import argparse +from concurrent.futures import ThreadPoolExecutor from contextlib import redirect_stdout from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer import importlib.util @@ -107,8 +108,34 @@ def test_http_failure_and_pending_response_are_not_success(self): with self.assertRaisesRegex(RuntimeError, f"HTTP {status}"): client.generate(self.args) self.assertEqual(post.call_count, 1) + self.assertEqual((self.args.output_dir / "response.raw").read_bytes(), response(status=status).content) self.assertFalse((self.args.output_dir / "generated.fasta").exists()) + def test_malformed_responses_preserve_exact_body_without_success_artifacts(self): + bodies = [b'{"sequence":', b'upstream error'] + for field, value in [("sampled_probs", [float("nan")] * 8), + ("elapsed_ms", float("inf")), + ("elapsed_ms_per_token", [float("-inf")] * 8)]: + data = response_data() + data[field] = value + bodies.append(json.dumps(data).encode()) + bodies.append(json.dumps(response_data()).replace('"elapsed_ms": 125', '"elapsed_ms": 1e309').encode()) + for index, body in enumerate(bodies): + with self.subTest(body=body): + self.args.output_dir = self.output / str(index) + malformed = response() + malformed._content = body + log = io.StringIO() + with patch.dict(os.environ, {"NGC_API_KEY": "test-key"}, clear=True), \ + patch.object(client.requests, "post", return_value=malformed) as post, redirect_stdout(log): + with self.assertRaises(ValueError): + client.generate(self.args) + self.assertEqual(post.call_count, 1) + self.assertEqual((self.args.output_dir / "response.raw").read_bytes(), body) + self.assertFalse((self.args.output_dir / "generated.fasta").exists()) + self.assertFalse((self.args.output_dir / "metrics.json").exists()) + self.assertNotIn('"status": "completed"', log.getvalue()) + def test_existing_outputs_are_not_overwritten_or_resubmitted(self): self.output.mkdir() existing = self.output / "response.json" @@ -119,6 +146,47 @@ def test_existing_outputs_are_not_overwritten_or_resubmitted(self): post.assert_not_called() self.assertEqual(existing.read_text(), '{"existing": true}\n') + def test_existing_empty_directory_is_not_reused(self): + self.output.mkdir() + with patch.dict(os.environ, {"NGC_API_KEY": "test-key"}, clear=True), patch.object(client.requests, "post") as post: + with self.assertRaises(FileExistsError): + client.generate(self.args) + post.assert_not_called() + self.assertEqual(list(self.output.iterdir()), []) + + def test_concurrent_runs_reserve_output_before_sending_one_request(self): + # Make both callers reach directory creation before either proceeds. + # A separate existence check followed by exist_ok=True lets both win. + ready = threading.Barrier(2) + mkdir = Path.mkdir + + def synchronized_mkdir(path, *args, **kwargs): + if path == self.output: + ready.wait(timeout=5) + return mkdir(path, *args, **kwargs) + + def run(seed): + args = argparse.Namespace(**{**vars(self.args), "seed": seed}) + try: + return client.generate(args) + except FileExistsError: + return None + + with patch.dict(os.environ, {"NGC_API_KEY": "test-key"}, clear=True), \ + patch.object(Path, "mkdir", synchronized_mkdir), \ + patch.object(client.requests, "post", return_value=response()) as post, redirect_stdout(io.StringIO()): + with ThreadPoolExecutor(max_workers=2) as pool: + futures = [pool.submit(run, seed) for seed in (10, 20)] + results = [future.result(timeout=10) for future in futures] + winners = [result for result in results if result is not None] + self.assertEqual(len(winners), 1) + self.assertEqual(post.call_count, 1) + saved_request = json.loads((self.output / "request.json").read_text()) + saved_metrics = json.loads((self.output / "metrics.json").read_text()) + self.assertEqual(saved_request["random_seed"], winners[0]["random_seed"]) + self.assertEqual(saved_metrics["random_seed"], winners[0]["random_seed"]) + self.assertEqual(post.call_args.kwargs["json"], saved_request) + def test_cli_executes_local_request_and_reports_persisted_results(self): received = [] diff --git a/uv.lock b/uv.lock index 5caac31..d86c20d 100644 --- a/uv.lock +++ b/uv.lock @@ -19,6 +19,7 @@ dependencies = [ { name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version == '3.11.*'" }, { name = "numpy", version = "2.5.1", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version >= '3.12'" }, { name = "pyyaml" }, + { name = "requests" }, ] [package.metadata] @@ -26,6 +27,7 @@ requires-dist = [ { name = "biotite", specifier = ">=1.0" }, { name = "numpy", specifier = ">=1.26" }, { name = "pyyaml", specifier = ">=6.0" }, + { name = "requests", specifier = ">=2.28" }, ] [[package]] @@ -36,12 +38,12 @@ resolution-markers = [ "python_full_version < '3.11'", ] dependencies = [ - { name = "biotraj" }, - { name = "msgpack" }, - { name = "networkx", version = "3.4.2", source = { registry = "https://pypi.org/simple" } }, - { name = "numpy", version = "2.2.6", source = { registry = "https://pypi.org/simple" } }, - { name = "packaging" }, - { name = "requests" }, + { name = "biotraj", marker = "python_full_version < '3.11'" }, + { name = "msgpack", marker = "python_full_version < '3.11'" }, + { name = "networkx", version = "3.4.2", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.11'" }, + { name = "numpy", version = "2.2.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.11'" }, + { name = "packaging", marker = "python_full_version < '3.11'" }, + { name = "requests", marker = "python_full_version < '3.11'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/24/95/ce1bbe59adb442390f57ffc4a4b93799ce9babd755e82db0b4a55fe87ca9/biotite-1.2.0.tar.gz", hash = "sha256:8b36dd708a976db10f629ffc8f81a236a74268a4e65e1f576610331d67dab392", size = 36062367, upload-time = "2025-03-16T08:30:09.688Z" } wheels = [ @@ -71,12 +73,12 @@ resolution-markers = [ "python_full_version == '3.11.*'", ] dependencies = [ - { name = "biotraj" }, - { name = "msgpack" }, - { name = "networkx", version = "3.6.1", source = { registry = "https://pypi.org/simple" } }, - { name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" } }, - { name = "packaging" }, - { name = "requests" }, + { name = "biotraj", marker = "python_full_version == '3.11.*'" }, + { name = "msgpack", marker = "python_full_version == '3.11.*'" }, + { name = "networkx", version = "3.6.1", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version == '3.11.*'" }, + { name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version == '3.11.*'" }, + { name = "packaging", marker = "python_full_version == '3.11.*'" }, + { name = "requests", marker = "python_full_version == '3.11.*'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/59/7b/99153f7bceef01034b5f19a6b123219533132d446ffcf141dfef3e386d33/biotite-1.6.0.tar.gz", hash = "sha256:4c172f6e57521220fa0fc4899142211f6f21ba83d8f6f135d4edc68981f70e7e", size = 38514388, upload-time = "2026-01-23T12:47:38.331Z" } wheels = [ @@ -110,12 +112,12 @@ resolution-markers = [ "python_full_version >= '3.12'", ] dependencies = [ - { name = "biotraj" }, - { name = "msgpack" }, - { name = "networkx", version = "3.6.1", source = { registry = "https://pypi.org/simple" } }, - { name = "numpy", version = "2.5.1", source = { registry = "https://pypi.org/simple" } }, - { name = "packaging" }, - { name = "requests" }, + { name = "biotraj", marker = "python_full_version >= '3.12'" }, + { name = "msgpack", marker = "python_full_version >= '3.12'" }, + { name = "networkx", version = "3.6.1", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version >= '3.12'" }, + { name = "numpy", version = "2.5.1", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version >= '3.12'" }, + { name = "packaging", marker = "python_full_version >= '3.12'" }, + { name = "requests", marker = "python_full_version >= '3.12'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/fc/93/ed0751d0f16d54ec82735605c776b37c77af7949a3dd230a1226f2b3b4df/biotite-1.7.1.tar.gz", hash = "sha256:2ae4a5d2c2d5ba08ca5d89a647984c99f45994eb908b614e896ce9e4db3ca800", size = 39858536, upload-time = "2026-06-22T12:42:34.507Z" } wheels = [ @@ -741,7 +743,7 @@ resolution-markers = [ "python_full_version < '3.11'", ] dependencies = [ - { name = "numpy", version = "2.2.6", source = { registry = "https://pypi.org/simple" } }, + { name = "numpy", version = "2.2.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.11'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/0f/37/6964b830433e654ec7485e45a00fc9a27cf868d622838f6b6d9c5ec0d532/scipy-1.15.3.tar.gz", hash = "sha256:eae3cf522bc7df64b42cad3925c876e1b0b6c35c1337c93e12c0f366f55b0eaf", size = 59419214, upload-time = "2025-05-08T16:13:05.955Z" } wheels = [ @@ -800,7 +802,7 @@ resolution-markers = [ "python_full_version == '3.11.*'", ] dependencies = [ - { name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" } }, + { name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version == '3.11.*'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/7a/97/5a3609c4f8d58b039179648e62dd220f89864f56f7357f5d4f45c29eb2cc/scipy-1.17.1.tar.gz", hash = "sha256:95d8e012d8cb8816c226aef832200b1d45109ed4464303e997c5b13122b297c0", size = 30573822, upload-time = "2026-02-23T00:26:24.851Z" } wheels = [ @@ -874,7 +876,7 @@ resolution-markers = [ "python_full_version >= '3.12'", ] dependencies = [ - { name = "numpy", version = "2.5.1", source = { registry = "https://pypi.org/simple" } }, + { name = "numpy", version = "2.5.1", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version >= '3.12'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/a7/25/c2700dfaf6442b4effaa91af24ebce5dc9d31bb4a69706313aae70d72cd0/scipy-1.18.0.tar.gz", hash = "sha256:67b2ad2ad54c72ca6d04975a9b2df8c3638c34ddd5b28738e94fc2b57929d378", size = 30774447, upload-time = "2026-06-19T15:01:43.456Z" } wheels = [ From 54c254ce22a0ed315c343160e4f0ef06791ef9a2 Mon Sep 17 00:00:00 2001 From: nvskills-svc-account Date: Tue, 29 Sep 2026 23:51:37 +0000 Subject: [PATCH 5/7] Attach NVSkills validation signatures Signed-off-by: nvskills-svc-account --- .../skills/evo2-nim/BENCHMARK.md | 26 ++++++-------- .../skills/evo2-nim/skill-card.md | 36 +++++++++---------- .../skills/evo2-nim/skill.oms.sig | 2 +- 3 files changed, 30 insertions(+), 34 deletions(-) diff --git a/skills/bionemo-agent-toolkit/skills/evo2-nim/BENCHMARK.md b/skills/bionemo-agent-toolkit/skills/evo2-nim/BENCHMARK.md index 84ea3da..33d250b 100644 --- a/skills/bionemo-agent-toolkit/skills/evo2-nim/BENCHMARK.md +++ b/skills/bionemo-agent-toolkit/skills/evo2-nim/BENCHMARK.md @@ -35,12 +35,12 @@ The three-tier evaluation checks whether the skill: | Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | |---|---:|---:| -| Overall | 94.1% — baseline ran, but no comparable score was available; uplift unavailable | 90.9% — baseline ran, but no comparable score was available; uplift unavailable | -| Security | 100.0% → 100.0% (±0.0 points) | 100.0% → 100.0% (±0.0 points) | +| Overall | 91.6% — baseline ran, but no comparable score was available; uplift unavailable | 87.1% — baseline ran, but no comparable score was available; uplift unavailable | +| Security | 50.0% → 100.0% (+50.0 points) | 50.0% → 100.0% (+50.0 points) | | Correctness | 100.0% → 100.0% (±0.0 points) | 100.0% → 100.0% (±0.0 points) | -| Discoverability | 95.0% — baseline ran, but no comparable score was available; uplift unavailable | 85.0% — baseline ran, but no comparable score was available; uplift unavailable | -| Effectiveness | 52.9% → 100.0% (+47.1 points) | 50.7% → 100.0% (+49.3 points) | -| Efficiency | 75.4% — baseline ran, but no comparable score was available; uplift unavailable | 69.7% — baseline ran, but no comparable score was available; uplift unavailable | +| Discoverability | 100.0% — baseline ran, but no comparable score was available; uplift unavailable | 85.0% — baseline ran, but no comparable score was available; uplift unavailable | +| Effectiveness | 92.9% → 80.0% (-12.9 points) | 66.4% → 85.7% (+19.3 points) | +| Efficiency | 77.9% — baseline ran, but no comparable score was available; uplift unavailable | 64.7% — baseline ran, but no comparable score was available; uplift unavailable | **How to read this table:** baseline is the same task attempted without the target skill. Scores are rounded to one decimal; threshold-adjacent values use additional precision so their displayed band matches the verdict. Uplift is derived from those displayed scores and shown in percentage points. @@ -52,11 +52,11 @@ Actual Tier 3 execution usage is reported for every observed agent/case pair and | Agent | Dataset case | With skill | Without skill | Delta | Change | Coverage | |---|---|---:|---:|---:|---:|---| -| claude-code | All cases | 490,498 | 676,561 | -186,063 | -27.50% | skill 1/1; base 1/1 | -| claude-code | 1 | 490,498 | 676,561 | -186,063 | -27.50% | skill 1/1; base 1/1 | -| codex | All cases | 205,950 | 375,153 | -169,203 | -45.10% | skill 1/1; base 1/1 | -| codex | 1 | 205,950 | 375,153 | -169,203 | -45.10% | skill 1/1; base 1/1 | -| ALL AGENTS | Dataset aggregate | 696,448 | 1,051,714 | -355,266 | -33.78% | skill 2/2; base 2/2 | +| claude-code | All cases | 341,369 | 1,268,174 | -926,805 | -73.08% | skill 1/1; base 1/1 | +| claude-code | 1 | 341,369 | 1,268,174 | -926,805 | -73.08% | skill 1/1; base 1/1 | +| codex | All cases | 209,936 | 220,174 | -10,238 | -4.65% | skill 1/1; base 1/1 | +| codex | 1 | 209,936 | 220,174 | -10,238 | -4.65% | skill 1/1; base 1/1 | +| ALL AGENTS | Dataset aggregate | 551,305 | 1,488,348 | -937,043 | -62.96% | skill 2/2; base 2/2 | Prompt tokens include cached reads, so total tokens are `prompt + completion` (cached is not added twice). The Efficiency score uses `(prompt - cached) + completion`. N/A means the relevant trajectory counters were not available; coverage is never estimated. @@ -73,11 +73,7 @@ Prompt tokens include cached reads, so total tokens are `prompt + completion` (c
Show detailed findings and successful checks -- **HIGH** DUPLICATE/duplicate: Duplicate content found across SKILL.md and references/api.md: - "# 40B default: 0,1 for 2x H100; set 0 for a single H200." in SKILL.md (lines 123-127) - vs "# For 7B: export NIM_VARIANT=7b; export NIM_TEST_GPUS="${NIM_TEST_GPUS:-0}"" in SKILL.md (lines 128-149) - vs "# 40B default: use 0,1 for 2x H100 80 GB; set NIM_TEST_GPUS=0 for a single H200." in references/api.md (lines 176-180) - vs "# Optional: export NIM_VARIANT=7b and add `-e NIM_VARIANT` for the 7B model." in references/api.md (lines 181-201) (`SKILL.md:123`) +- **CRITICAL** CONTENT_DEDUP/llm_error: LLM analysis failed for a content cluster (`skills/bionemo-agent-toolkit/skills/evo2-nim`) - **MEDIUM** QUALITY/quality_correctness: No documented scripts in table format (`skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md`) - **MEDIUM** QUALITY/quality_correctness: Instructions don't mention 'run_script' (`skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md`) - **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.author' (`skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md`) diff --git a/skills/bionemo-agent-toolkit/skills/evo2-nim/skill-card.md b/skills/bionemo-agent-toolkit/skills/evo2-nim/skill-card.md index 415807a..7ede3e7 100644 --- a/skills/bionemo-agent-toolkit/skills/evo2-nim/skill-card.md +++ b/skills/bionemo-agent-toolkit/skills/evo2-nim/skill-card.md @@ -9,7 +9,7 @@ NVIDIA
### License/Terms of Use:
Apache-2.0 AND CC-BY-4.0
## Use Case:
-Developers and engineers use this skill for DNA sequence generation, genomic analysis, and layer-output extraction via NVIDIA BioNeMo NIM, in both hosted API and local Docker deployment modes.
+Developers and engineers using agent-assisted workflows for DNA sequence generation, genomic analysis, and BioNeMo NIM microservice integration.
### Deployment Geography for Use:
Global
@@ -27,14 +27,14 @@ Mitigation: Review and scan skill before deployment.
## Reference(s):
- [Evo 2 NIM API Reference](references/api.md)
- [Genomic Use Cases and Interpretation](references/science.md)
-- [Generation and Forward Parameter Effects](references/parameters.md)
+- [Generation and Forward Parameters](references/parameters.md)
- [Validation Checks](references/validation.md)
-- [Request Pattern Examples](references/examples.md)
+- [Hosted and Local Request Examples](references/examples.md)
## Skill Output:
-**Output Type(s):** [Code, Files, Shell commands]
-**Output Format:** [Markdown with inline bash and Python code blocks]
+**Output Type(s):** [Shell commands, Code, Analysis, Files]
+**Output Format:** [Markdown with inline code blocks and FASTA files]
**Output Parameters:** [1D]
**Other Properties Related to Output:** [None]
@@ -45,23 +45,23 @@ Mitigation: Review and scan skill before deployment.
## Evaluation Tasks:
-1 evaluation task (1 positive), 3 attempts per task, each in an isolated k8s-sandbox pod.
+1 evaluation task (1 positive, 3 attempts per task) in isolated k8s-sandbox pods.
## Evaluation Metrics Used:
Reported benchmark dimensions:
-- Security: Whether the skill is safe to use, checking for unsafe operations, secret leakage, and unauthorized access.
-- Correctness: Whether the final answer is correct against the reference answer.
-- Discoverability: Whether the expected skill was selected, decoys were avoided, and the workflow executed.
-- Effectiveness: Whether the skill helped complete the user's goal (50% goal accuracy + 50% behavior check).
-- Efficiency: Whether wasted tool calls and token usage were avoided (50% tool productivity + 50% token efficiency).
+- Security: Whether the skill avoids unsafe operations, secret leakage, and unauthorized access.
+- Correctness: Final-answer correctness against the reference answer.
+- Discoverability: Whether the expected skill was selected and activated when needed.
+- Effectiveness: Whether the skill helped complete the user's goal and expected workflow (goal_accuracy 50% + behavior_check 50%).
+- Efficiency: Whether the skill avoided wasted tool calls and token usage (skill_efficiency 50% + token_efficiency 50%).
Underlying evaluation signals used in this run:
- `security`: Checks for unsafe operations, secret leakage, and unauthorized access.
- `accuracy`: Final-answer correctness against the reference answer.
-- `skill_execution`: Whether the expected skill was selected and the workflow executed.
+- `skill_execution`: Whether the expected skill was selected, decoys were avoided, and the workflow executed.
- `goal_accuracy`: Whether the user's goal was achieved.
- `behavior_check`: Whether the expected workflow behavior was followed.
-- `skill_efficiency`: Tool-call productivity; routing is scored under Discoverability.
+- `skill_efficiency`: Tool-call productivity.
- `token_efficiency`: Actual uncached prompt plus completion token usage.
@@ -69,12 +69,12 @@ Underlying evaluation signals used in this run:
## Evaluation Results:
| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | |---|---:|---:| -| Overall | 94.1% | 90.9% | -| Security | 100.0% → 100.0% (±0.0 points) | 100.0% → 100.0% (±0.0 points) | +| Overall | 91.6% | 87.1% | +| Security | 50.0% → 100.0% (+50.0 points) | 50.0% → 100.0% (+50.0 points) | | Correctness | 100.0% → 100.0% (±0.0 points) | 100.0% → 100.0% (±0.0 points) | -| Discoverability | 95.0% | 85.0% | -| Effectiveness | 52.9% → 100.0% (+47.1 points) | 50.7% → 100.0% (+49.3 points) | -| Efficiency | 75.4% | 69.7% | +| Discoverability | 100.0% | 85.0% | +| Effectiveness | 92.9% → 80.0% (-12.9 points) | 66.4% → 85.7% (+19.3 points) | +| Efficiency | 77.9% | 64.7% | ## Skill Version(s):
0.1.0 (source: pyproject.toml)
diff --git a/skills/bionemo-agent-toolkit/skills/evo2-nim/skill.oms.sig b/skills/bionemo-agent-toolkit/skills/evo2-nim/skill.oms.sig index c130d34..1d9ac9d 100644 --- a/skills/bionemo-agent-toolkit/skills/evo2-nim/skill.oms.sig +++ b/skills/bionemo-agent-toolkit/skills/evo2-nim/skill.oms.sig @@ -1 +1 @@ -{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAiZXZvMi1uaW0iLAogICAgICAiZGlnZXN0IjogewogICAgICAgICJzaGEyNTYiOiAiMTM5ZmEyYzM1N2FhOTFiMDNkNzQwZjE0ODYxY2IwNTVhNDhlMTU4YjYzY2FlMWY3ZDY2NTJiZTMzYTZlZmZhZiIKICAgICAgfQogICAgfQogIF0sCiAgInByZWRpY2F0ZVR5cGUiOiAiaHR0cHM6Ly9tb2RlbF9zaWduaW5nL3NpZ25hdHVyZS92MS4wIiwKICAicHJlZGljYXRlIjogewogICAgInJlc291cmNlcyI6IFsKICAgICAgewogICAgICAgICJuYW1lIjogIkJFTkNITUFSSy5tZCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiZmFkNGU2YWIxYWRkNWZhMzU3NzgyZTBjMzIxYjQxNDAwNGM1MjQyMmI1YTBlZmZlMjVmN2E4YWZhOTQzZmQyMyIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogIlNLSUxMLm1kIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICJhZGI2ODlhYzcwNDM5YjY1MzdiY2EzNThmMTFlNWU1ZDVmOGY3ZDA3ZjFjYTJjMDc3YjI1YjM0YzM1YTJmZWRmIgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAiY29uZmlnL3NraWxsc3BlY3Rvci1iYXNlbGluZS55bWwiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogImUxZDQxOWY0NmY3YzM0ZTM0OGU4ZmQ2OTViZWI5NWZlODMwZjAwZDg4MTBmODBlMzZmYjNiNjc1ZjFmMDcwYmMiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJldmFscy9jb25maWcueW1sIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICJiYTJiOGNmMGVhZDEzYmZiNjVkODEzMDllMTYzMTYyZGJjMmM2YzUwNTg5NjRjNGFlYTQxNGUwMDcwODk4YThhIgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAiZXZhbHMvZXZhbHMuanNvbiIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiMTZkOGU3MzVlYTRiMTJiZWE3M2Y4NzUxZmJjYTFjYTcxMjk0N2Q5ODQyYWRkMzAxMDMzNDI1MGYzY2IzZTc4ZCIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogImV2YWxzL3RyaWdnZXJfZXZhbHMuanNvbiIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiMmVlMmQyZjgyOTg4NGUxNDJiNmU2NGY2YmE0ODJiNjZmMjBhZjFlOGIwNTA3NDdlNWNhNDMxZGQzNDEyYTY4NCIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInJlZmVyZW5jZXMvYXBpLm1kIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICI5Y2Y2MDVjMjViMmExMjEwMmRkYjQwOTQ4YzQxZjY1M2VjNGRiODYzNTRiYjFiMzQ2NjI5YzJhZjk4MjNiZDdiIgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAicmVmZXJlbmNlcy9leGFtcGxlcy5tZCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiZGNlMWQwNzdiNmY2MzNlYjQ1ODRjM2VmODgyMDAxZTFhZDkyMjRiNDZkZTllMTJhZWUwNGRmZGI2Mjc3OGM2MiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInJlZmVyZW5jZXMvcGFyYW1ldGVycy5tZCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiMTYwNjI3N2Q2MGFhMDI5ZWZiMWNlN2M4NzhiNGFkODgwZDdmODI0NjE0YTNmYjc1ZWFjMWI3MDU5YjQ1M2ZiNyIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInJlZmVyZW5jZXMvc2NpZW5jZS5tZCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiZmVhZGM1NmVmZGRmOTZlNDUxNmYxZGRmZTFmZjlmOWMwMmFlYjg2OTIyYWZlMWFlYzNiYWUyM2NmNmQwYWY1OSIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInJlZmVyZW5jZXMvdmFsaWRhdGlvbi5tZCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJkaWdlc3QiOiAiNzhlMDY3N2JiZmRlOTM5ODcyNjk0MTM1YmIyMTdhNjgyOTZhNDE4NDA0M2M4ZWNjNWM1MjU4YzRlOTEzNTA1MSIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInNjcmlwdHMvZ2VuZXJhdGUucHkiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAiZGlnZXN0IjogIjFlMjEwZmVlNjhkZGY2NmU0M2VlYTAzNThjYzEwYzI3MDU4OTBhMDA5MjA0ZjFhMzc2NmMzOWNmNDVlOTQ3ZjAiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJza2lsbC1jYXJkLm1kIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgImRpZ2VzdCI6ICIzNGJlMzQzZGEzZTdiYmNhMjE1ZTYwYTkyZWY4NjAwYzgyMjg3NzJhNmRkM2E4MzA2ZWJmZmE5ZTE3ZTg0N2IyIgogICAgICB9CiAgICBdLAogICAgInNlcmlhbGl6YXRpb24iOiB7CiAgICAgICJhbGxvd19zeW1saW5rcyI6IGZhbHNlLAogICAgICAiaGFzaF90eXBlIjogInNoYTI1NiIsCiAgICAgICJtZXRob2QiOiAiZmlsZXMiLAogICAgICAiaWdub3JlX3BhdGhzIjogWwogICAgICAgICIuZ2l0YXR0cmlidXRlcyIsCiAgICAgICAgIi5naXQiLAogICAgICAgICIuZ2l0aWdub3JlIiwKICAgICAgICAiLmdpdGh1YiIKICAgICAgXQogICAgfQogIH0KfQ==","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGQCMFB5KyBi3908w8LRkCFu2E6iJ/asg6LnjDIx4X3Xo/eN9gdGlOnQ5KEbwRcwbF8CEgIwHwS+TfMJVNnwJ5v5D1Fr88Olu0j/GyfUFNT2kNIJhZFWZDTpy2fTIptYxPwi9HYl","keyid":""}]}} \ No newline at end of file +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAiZXZvMi1uaW0iLAogICAgICAiZGlnZXN0IjogewogICAgICAgICJzaGEyNTYiOiAiMjA5ZjZhMzRlYzIzMzNjZmJiMmJkNWYzOTkyMTBjMTk5ZGI2M2QzYzE4ZDU3OTBkMjc2MTUxNjU4MGM5NzFmOCIKICAgICAgfQogICAgfQogIF0sCiAgInByZWRpY2F0ZVR5cGUiOiAiaHR0cHM6Ly9tb2RlbF9zaWduaW5nL3NpZ25hdHVyZS92MS4wIiwKICAicHJlZGljYXRlIjogewogICAgInNlcmlhbGl6YXRpb24iOiB7CiAgICAgICJpZ25vcmVfcGF0aHMiOiBbCiAgICAgICAgIi5naXRodWIiLAogICAgICAgICIuZ2l0IiwKICAgICAgICAiLmdpdGlnbm9yZSIsCiAgICAgICAgIi5naXRhdHRyaWJ1dGVzIgogICAgICBdLAogICAgICAibWV0aG9kIjogImZpbGVzIiwKICAgICAgImFsbG93X3N5bWxpbmtzIjogZmFsc2UsCiAgICAgICJoYXNoX3R5cGUiOiAic2hhMjU2IgogICAgfSwKICAgICJyZXNvdXJjZXMiOiBbCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAiQkVOQ0hNQVJLLm1kIiwKICAgICAgICAiZGlnZXN0IjogImMwNTZjOWZkZjRmMmQ0NTlmMzYxNDgyNjBjYmM4YmM3N2NiNTFmZjk4N2E3NGY3ODdiZmRjZjU1YmE2NTVmMDMiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAiU0tJTEwubWQiLAogICAgICAgICJkaWdlc3QiOiAiMzUwN2U1ZDcxMWI3NjM5YzQ3ZjEyODBiNGM2YzYyYTRlNDc5M2FjM2JmOTJiZWQ4YjUxZjUzMGVkYWYyYjQ3MiIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJjb25maWcvc2tpbGxzcGVjdG9yLWJhc2VsaW5lLnltbCIsCiAgICAgICAgImRpZ2VzdCI6ICJlMWQ0MTlmNDZmN2MzNGUzNDhlOGZkNjk1YmViOTVmZTgzMGYwMGQ4ODEwZjgwZTM2ZmIzYjY3NWYxZjA3MGJjIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJuYW1lIjogImV2YWxzL2NvbmZpZy55bWwiLAogICAgICAgICJkaWdlc3QiOiAiYmEyYjhjZjBlYWQxM2JmYjY1ZDgxMzA5ZTE2MzE2MmRiYzJjNmM1MDU4OTY0YzRhZWE0MTRlMDA3MDg5OGE4YSIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJldmFscy9ldmFscy5qc29uIiwKICAgICAgICAiZGlnZXN0IjogIjE2ZDhlNzM1ZWE0YjEyYmVhNzNmODc1MWZiY2ExY2E3MTI5NDdkOTg0MmFkZDMwMTAzMzQyNTBmM2NiM2U3OGQiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAiZXZhbHMvdHJpZ2dlcl9ldmFscy5qc29uIiwKICAgICAgICAiZGlnZXN0IjogIjJlZTJkMmY4Mjk4ODRlMTQyYjZlNjRmNmJhNDgyYjY2ZjIwYWYxZThiMDUwNzQ3ZTVjYTQzMWRkMzQxMmE2ODQiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAicmVmZXJlbmNlcy9hcGkubWQiLAogICAgICAgICJkaWdlc3QiOiAiOWNmNjA1YzI1YjJhMTIxMDJkZGI0MDk0OGM0MWY2NTNlYzRkYjg2MzU0YmIxYjM0NjYyOWMyYWY5ODIzYmQ3YiIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJyZWZlcmVuY2VzL2V4YW1wbGVzLm1kIiwKICAgICAgICAiZGlnZXN0IjogImRjZTFkMDc3YjZmNjMzZWI0NTg0YzNlZjg4MjAwMWUxYWQ5MjI0YjQ2ZGU5ZTEyYWVlMDRkZmRiNjI3NzhjNjIiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAicmVmZXJlbmNlcy9wYXJhbWV0ZXJzLm1kIiwKICAgICAgICAiZGlnZXN0IjogIjE2MDYyNzdkNjBhYTAyOWVmYjFjZTdjODc4YjRhZDg4MGQ3ZjgyNDYxNGEzZmI3NWVhYzFiNzA1OWI0NTNmYjciCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAicmVmZXJlbmNlcy9zY2llbmNlLm1kIiwKICAgICAgICAiZGlnZXN0IjogImZlYWRjNTZlZmRkZjk2ZTQ1MTZmMWRkZmUxZmY5ZjljMDJhZWI4NjkyMmFmZTFhZWMzYmFlMjNjZjZkMGFmNTkiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAicmVmZXJlbmNlcy92YWxpZGF0aW9uLm1kIiwKICAgICAgICAiZGlnZXN0IjogIjc4ZTA2NzdiYmZkZTkzOTg3MjY5NDEzNWJiMjE3YTY4Mjk2YTQxODQwNDNjOGVjYzVjNTI1OGM0ZTkxMzUwNTEiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAic2NyaXB0cy9nZW5lcmF0ZS5weSIsCiAgICAgICAgImRpZ2VzdCI6ICI3Y2EzM2U1MDJhMzM4MTk4M2QxODM0MTBiNDYxMzYwOTc2YTRmYjE1ZjAwZDI0M2VjMGZhN2YyMDFlZWE2MTAyIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJuYW1lIjogInNraWxsLWNhcmQubWQiLAogICAgICAgICJkaWdlc3QiOiAiMzBkYzM4NTViNjU1ZDIzM2NmNjhkNTYyZDg5YTZkMzFhMDEyNjE0OGE5NmQwMzU0ODRiNzllY2ExYzQ1MGRiNSIKICAgICAgfQogICAgXQogIH0KfQ==","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGUCMA602PMJrqNA3xajzlnQRKiYanKrZLeiStRyiRH6lxX61BupteDnWzzdGjoIU4qtxwIxAKxWQChqv1L2uBj+AsowmDwiiDNfd+eWYLCQmnNGwzEqt1kIFBGgRgMN8Op68aJLcg==","keyid":""}]}} \ No newline at end of file From 5c86e77fd953ffe194f4b63e8a0be66bf6dc3a33 Mon Sep 17 00:00:00 2001 From: Ohad Mosafi Date: Tue, 29 Sep 2026 17:24:13 -0700 Subject: [PATCH 6/7] Document Evo2 local layer-output extraction in skill card Signed-off-by: Ohad Mosafi --- skills/bionemo-agent-toolkit/skills/evo2-nim/skill-card.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/skills/bionemo-agent-toolkit/skills/evo2-nim/skill-card.md b/skills/bionemo-agent-toolkit/skills/evo2-nim/skill-card.md index 7ede3e7..2031ee1 100644 --- a/skills/bionemo-agent-toolkit/skills/evo2-nim/skill-card.md +++ b/skills/bionemo-agent-toolkit/skills/evo2-nim/skill-card.md @@ -9,7 +9,7 @@ NVIDIA
### License/Terms of Use:
Apache-2.0 AND CC-BY-4.0
## Use Case:
-Developers and engineers using agent-assisted workflows for DNA sequence generation, genomic analysis, and BioNeMo NIM microservice integration.
+Developers and engineers using agent-assisted workflows for DNA sequence generation, genomic analysis, local layer-output extraction, and BioNeMo NIM microservice integration.
### Deployment Geography for Use:
Global
From 32f681427bd63b6933579f091da5e0bf73804df7 Mon Sep 17 00:00:00 2001 From: nvskills-svc-account Date: Wed, 30 Sep 2026 00:40:16 +0000 Subject: [PATCH 7/7] Attach NVSkills validation signatures Signed-off-by: nvskills-svc-account --- .../skills/evo2-nim/BENCHMARK.md | 28 ++++++------- .../skills/evo2-nim/skill-card.md | 42 +++++++++---------- .../skills/evo2-nim/skill.oms.sig | 2 +- 3 files changed, 36 insertions(+), 36 deletions(-) diff --git a/skills/bionemo-agent-toolkit/skills/evo2-nim/BENCHMARK.md b/skills/bionemo-agent-toolkit/skills/evo2-nim/BENCHMARK.md index 33d250b..fc3e2fc 100644 --- a/skills/bionemo-agent-toolkit/skills/evo2-nim/BENCHMARK.md +++ b/skills/bionemo-agent-toolkit/skills/evo2-nim/BENCHMARK.md @@ -9,7 +9,7 @@ Recommended for publication based on the completed evaluation evidence in this r ## Evaluation Metadata - Skill: `evo2-nim` -- Evaluation date: 2026-09-29 +- Evaluation date: 2026-09-30 - Evaluator version: `1.5.6` - Agents: Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`), Codex (`openai/openai/gpt-5.5`) - Tasks: 1 evaluation tasks (1 positive) @@ -35,12 +35,12 @@ The three-tier evaluation checks whether the skill: | Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | |---|---:|---:| -| Overall | 91.6% — baseline ran, but no comparable score was available; uplift unavailable | 87.1% — baseline ran, but no comparable score was available; uplift unavailable | -| Security | 50.0% → 100.0% (+50.0 points) | 50.0% → 100.0% (+50.0 points) | +| Overall | 94.9% — baseline ran, but no comparable score was available; uplift unavailable | 91.3% — baseline ran, but no comparable score was available; uplift unavailable | +| Security | 100.0% → 100.0% (±0.0 points) | 50.0% → 100.0% (+50.0 points) | | Correctness | 100.0% → 100.0% (±0.0 points) | 100.0% → 100.0% (±0.0 points) | -| Discoverability | 100.0% — baseline ran, but no comparable score was available; uplift unavailable | 85.0% — baseline ran, but no comparable score was available; uplift unavailable | -| Effectiveness | 92.9% → 80.0% (-12.9 points) | 66.4% → 85.7% (+19.3 points) | -| Efficiency | 77.9% — baseline ran, but no comparable score was available; uplift unavailable | 64.7% — baseline ran, but no comparable score was available; uplift unavailable | +| Discoverability | 100.0% — baseline ran, but no comparable score was available; uplift unavailable | 90.0% — baseline ran, but no comparable score was available; uplift unavailable | +| Effectiveness | 100.0% → 100.0% (±0.0 points) | 36.4% → 100.0% (+63.6 points) | +| Efficiency | 74.3% — baseline ran, but no comparable score was available; uplift unavailable | 66.6% — baseline ran, but no comparable score was available; uplift unavailable | **How to read this table:** baseline is the same task attempted without the target skill. Scores are rounded to one decimal; threshold-adjacent values use additional precision so their displayed band matches the verdict. Uplift is derived from those displayed scores and shown in percentage points. @@ -52,11 +52,11 @@ Actual Tier 3 execution usage is reported for every observed agent/case pair and | Agent | Dataset case | With skill | Without skill | Delta | Change | Coverage | |---|---|---:|---:|---:|---:|---| -| claude-code | All cases | 341,369 | 1,268,174 | -926,805 | -73.08% | skill 1/1; base 1/1 | -| claude-code | 1 | 341,369 | 1,268,174 | -926,805 | -73.08% | skill 1/1; base 1/1 | -| codex | All cases | 209,936 | 220,174 | -10,238 | -4.65% | skill 1/1; base 1/1 | -| codex | 1 | 209,936 | 220,174 | -10,238 | -4.65% | skill 1/1; base 1/1 | -| ALL AGENTS | Dataset aggregate | 551,305 | 1,488,348 | -937,043 | -62.96% | skill 2/2; base 2/2 | +| claude-code | All cases | 297,130 | 554,993 | -257,863 | -46.46% | skill 1/1; base 1/1 | +| claude-code | 1 | 297,130 | 554,993 | -257,863 | -46.46% | skill 1/1; base 1/1 | +| codex | All cases | 233,585 | 253,870 | -20,285 | -7.99% | skill 1/1; base 1/1 | +| codex | 1 | 233,585 | 253,870 | -20,285 | -7.99% | skill 1/1; base 1/1 | +| ALL AGENTS | Dataset aggregate | 530,715 | 808,863 | -278,148 | -34.39% | skill 2/2; base 2/2 | Prompt tokens include cached reads, so total tokens are `prompt + completion` (cached is not added twice). The Efficiency score uses `(prompt - cached) + completion`. N/A means the relevant trajectory counters were not available; coverage is never estimated. @@ -65,7 +65,7 @@ Prompt tokens include cached reads, so total tokens are `prompt + completion` (c | Tier | Purpose | Status | Evidence | |---|---|---|---| | Tier 1 | Static validation | **PASSED WITH OBSERVATIONS** | 11 validator(s); 26 finding(s) | -| Tier 2 | Semantic deduplication | **PASSED WITH OBSERVATIONS** | 2 validator(s); 1 finding(s) | +| Tier 2 | Semantic deduplication | **PASSED** | 2 validator(s); 0 finding(s) | | Tier 3 | Live agent evaluation | **PASS** | 2 agent(s); 1 task(s) | ## Findings and Observations @@ -73,12 +73,12 @@ Prompt tokens include cached reads, so total tokens are `prompt + completion` (c
Show detailed findings and successful checks -- **CRITICAL** CONTENT_DEDUP/llm_error: LLM analysis failed for a content cluster (`skills/bionemo-agent-toolkit/skills/evo2-nim`) - **MEDIUM** QUALITY/quality_correctness: No documented scripts in table format (`skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md`) - **MEDIUM** QUALITY/quality_correctness: Instructions don't mention 'run_script' (`skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md`) - **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.author' (`skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md`) - **MEDIUM** QUALITY/quality_correctness: SKILL_SPEC recommended field missing: 'metadata.tags' (`skills/bionemo-agent-toolkit/skills/evo2-nim/SKILL.md`) -- 22 additional finding(s) are available in the full evaluation artifacts. +- **MEDIUM** SCHEMA/folder_hierarchy: Unexpected nesting depth for general skill (`skills/bionemo-agent-toolkit/skills/evo2-nim`) +- 21 additional finding(s) are available in the full evaluation artifacts.
diff --git a/skills/bionemo-agent-toolkit/skills/evo2-nim/skill-card.md b/skills/bionemo-agent-toolkit/skills/evo2-nim/skill-card.md index 2031ee1..9efd07f 100644 --- a/skills/bionemo-agent-toolkit/skills/evo2-nim/skill-card.md +++ b/skills/bionemo-agent-toolkit/skills/evo2-nim/skill-card.md @@ -9,7 +9,7 @@ NVIDIA
### License/Terms of Use:
Apache-2.0 AND CC-BY-4.0
## Use Case:
-Developers and engineers using agent-assisted workflows for DNA sequence generation, genomic analysis, local layer-output extraction, and BioNeMo NIM microservice integration.
+Developers and engineers use this skill to generate and analyze DNA sequences via NVIDIA's Evo 2 BioNeMo NIM, supporting hosted API and local Docker deployment workflows.
### Deployment Geography for Use:
Global
@@ -26,17 +26,17 @@ Mitigation: Review and scan skill before deployment.
## Reference(s):
- [Evo 2 NIM API Reference](references/api.md)
-- [Genomic Use Cases and Interpretation](references/science.md)
-- [Generation and Forward Parameters](references/parameters.md)
-- [Validation Checks](references/validation.md)
-- [Hosted and Local Request Examples](references/examples.md)
+- [Science Reference](references/science.md)
+- [Parameters Reference](references/parameters.md)
+- [Validation Reference](references/validation.md)
+- [Examples](references/examples.md)
## Skill Output:
-**Output Type(s):** [Shell commands, Code, Analysis, Files]
-**Output Format:** [Markdown with inline code blocks and FASTA files]
+**Output Type(s):** [Shell commands, Code, Files, Analysis]
+**Output Format:** [Markdown with inline bash and Python code blocks]
**Output Parameters:** [1D]
-**Other Properties Related to Output:** [None]
+**Other Properties Related to Output:** [Saves request.json, response.json, generated.fasta, and metrics.json artifacts]
## Evaluation Agents Used:
- Claude Code (`aws/anthropic/bedrock-claude-opus-4-8`)
@@ -45,18 +45,18 @@ Mitigation: Review and scan skill before deployment.
## Evaluation Tasks:
-1 evaluation task (1 positive, 3 attempts per task) in isolated k8s-sandbox pods.
+1 evaluation task (1 positive), 3 attempts per task, each in an isolated sandbox pod.
## Evaluation Metrics Used:
Reported benchmark dimensions:
-- Security: Whether the skill avoids unsafe operations, secret leakage, and unauthorized access.
-- Correctness: Final-answer correctness against the reference answer.
-- Discoverability: Whether the expected skill was selected and activated when needed.
-- Effectiveness: Whether the skill helped complete the user's goal and expected workflow (goal_accuracy 50% + behavior_check 50%).
-- Efficiency: Whether the skill avoided wasted tool calls and token usage (skill_efficiency 50% + token_efficiency 50%).
+- Security: Checks for unsafe operations, secret leakage, and unauthorized access.
+- Correctness: Checks final-answer correctness against the reference answer.
+- Discoverability: Checks whether the expected skill was selected and the workflow executed.
+- Effectiveness: Checks whether the user's goal was achieved and expected workflow behavior was followed.
+- Efficiency: Checks tool-call productivity and token usage efficiency.
Underlying evaluation signals used in this run:
-- `security`: Checks for unsafe operations, secret leakage, and unauthorized access.
+- `security`: Unsafe operations, secret leakage, and unauthorized access.
- `accuracy`: Final-answer correctness against the reference answer.
- `skill_execution`: Whether the expected skill was selected, decoys were avoided, and the workflow executed.
- `goal_accuracy`: Whether the user's goal was achieved.
@@ -69,12 +69,12 @@ Underlying evaluation signals used in this run:
## Evaluation Results:
| Measure | Claude Code (Baseline → Skill Uplift) | Codex (Baseline → Skill Uplift) | |---|---:|---:| -| Overall | 91.6% | 87.1% | -| Security | 50.0% → 100.0% (+50.0 points) | 50.0% → 100.0% (+50.0 points) | -| Correctness | 100.0% → 100.0% (±0.0 points) | 100.0% → 100.0% (±0.0 points) | -| Discoverability | 100.0% | 85.0% | -| Effectiveness | 92.9% → 80.0% (-12.9 points) | 66.4% → 85.7% (+19.3 points) | -| Efficiency | 77.9% | 64.7% | +| Overall | 94.9% | 91.3% | +| Security | 100.0% → 100.0% (±0.0 pts) | 50.0% → 100.0% (+50.0 pts) | +| Correctness | 100.0% → 100.0% (±0.0 pts) | 100.0% → 100.0% (±0.0 pts) | +| Discoverability | 100.0% | 90.0% | +| Effectiveness | 100.0% → 100.0% (±0.0 pts) | 36.4% → 100.0% (+63.6 pts) | +| Efficiency | 74.3% | 66.6% | ## Skill Version(s):
0.1.0 (source: pyproject.toml)
diff --git a/skills/bionemo-agent-toolkit/skills/evo2-nim/skill.oms.sig b/skills/bionemo-agent-toolkit/skills/evo2-nim/skill.oms.sig index 1d9ac9d..c927ec9 100644 --- a/skills/bionemo-agent-toolkit/skills/evo2-nim/skill.oms.sig +++ b/skills/bionemo-agent-toolkit/skills/evo2-nim/skill.oms.sig @@ -1 +1 @@ -{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAiZXZvMi1uaW0iLAogICAgICAiZGlnZXN0IjogewogICAgICAgICJzaGEyNTYiOiAiMjA5ZjZhMzRlYzIzMzNjZmJiMmJkNWYzOTkyMTBjMTk5ZGI2M2QzYzE4ZDU3OTBkMjc2MTUxNjU4MGM5NzFmOCIKICAgICAgfQogICAgfQogIF0sCiAgInByZWRpY2F0ZVR5cGUiOiAiaHR0cHM6Ly9tb2RlbF9zaWduaW5nL3NpZ25hdHVyZS92MS4wIiwKICAicHJlZGljYXRlIjogewogICAgInNlcmlhbGl6YXRpb24iOiB7CiAgICAgICJpZ25vcmVfcGF0aHMiOiBbCiAgICAgICAgIi5naXRodWIiLAogICAgICAgICIuZ2l0IiwKICAgICAgICAiLmdpdGlnbm9yZSIsCiAgICAgICAgIi5naXRhdHRyaWJ1dGVzIgogICAgICBdLAogICAgICAibWV0aG9kIjogImZpbGVzIiwKICAgICAgImFsbG93X3N5bWxpbmtzIjogZmFsc2UsCiAgICAgICJoYXNoX3R5cGUiOiAic2hhMjU2IgogICAgfSwKICAgICJyZXNvdXJjZXMiOiBbCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAiQkVOQ0hNQVJLLm1kIiwKICAgICAgICAiZGlnZXN0IjogImMwNTZjOWZkZjRmMmQ0NTlmMzYxNDgyNjBjYmM4YmM3N2NiNTFmZjk4N2E3NGY3ODdiZmRjZjU1YmE2NTVmMDMiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAiU0tJTEwubWQiLAogICAgICAgICJkaWdlc3QiOiAiMzUwN2U1ZDcxMWI3NjM5YzQ3ZjEyODBiNGM2YzYyYTRlNDc5M2FjM2JmOTJiZWQ4YjUxZjUzMGVkYWYyYjQ3MiIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJjb25maWcvc2tpbGxzcGVjdG9yLWJhc2VsaW5lLnltbCIsCiAgICAgICAgImRpZ2VzdCI6ICJlMWQ0MTlmNDZmN2MzNGUzNDhlOGZkNjk1YmViOTVmZTgzMGYwMGQ4ODEwZjgwZTM2ZmIzYjY3NWYxZjA3MGJjIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJuYW1lIjogImV2YWxzL2NvbmZpZy55bWwiLAogICAgICAgICJkaWdlc3QiOiAiYmEyYjhjZjBlYWQxM2JmYjY1ZDgxMzA5ZTE2MzE2MmRiYzJjNmM1MDU4OTY0YzRhZWE0MTRlMDA3MDg5OGE4YSIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJldmFscy9ldmFscy5qc29uIiwKICAgICAgICAiZGlnZXN0IjogIjE2ZDhlNzM1ZWE0YjEyYmVhNzNmODc1MWZiY2ExY2E3MTI5NDdkOTg0MmFkZDMwMTAzMzQyNTBmM2NiM2U3OGQiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAiZXZhbHMvdHJpZ2dlcl9ldmFscy5qc29uIiwKICAgICAgICAiZGlnZXN0IjogIjJlZTJkMmY4Mjk4ODRlMTQyYjZlNjRmNmJhNDgyYjY2ZjIwYWYxZThiMDUwNzQ3ZTVjYTQzMWRkMzQxMmE2ODQiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAicmVmZXJlbmNlcy9hcGkubWQiLAogICAgICAgICJkaWdlc3QiOiAiOWNmNjA1YzI1YjJhMTIxMDJkZGI0MDk0OGM0MWY2NTNlYzRkYjg2MzU0YmIxYjM0NjYyOWMyYWY5ODIzYmQ3YiIKICAgICAgfSwKICAgICAgewogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IiwKICAgICAgICAibmFtZSI6ICJyZWZlcmVuY2VzL2V4YW1wbGVzLm1kIiwKICAgICAgICAiZGlnZXN0IjogImRjZTFkMDc3YjZmNjMzZWI0NTg0YzNlZjg4MjAwMWUxYWQ5MjI0YjQ2ZGU5ZTEyYWVlMDRkZmRiNjI3NzhjNjIiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAicmVmZXJlbmNlcy9wYXJhbWV0ZXJzLm1kIiwKICAgICAgICAiZGlnZXN0IjogIjE2MDYyNzdkNjBhYTAyOWVmYjFjZTdjODc4YjRhZDg4MGQ3ZjgyNDYxNGEzZmI3NWVhYzFiNzA1OWI0NTNmYjciCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAicmVmZXJlbmNlcy9zY2llbmNlLm1kIiwKICAgICAgICAiZGlnZXN0IjogImZlYWRjNTZlZmRkZjk2ZTQ1MTZmMWRkZmUxZmY5ZjljMDJhZWI4NjkyMmFmZTFhZWMzYmFlMjNjZjZkMGFmNTkiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAicmVmZXJlbmNlcy92YWxpZGF0aW9uLm1kIiwKICAgICAgICAiZGlnZXN0IjogIjc4ZTA2NzdiYmZkZTkzOTg3MjY5NDEzNWJiMjE3YTY4Mjk2YTQxODQwNDNjOGVjYzVjNTI1OGM0ZTkxMzUwNTEiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIsCiAgICAgICAgIm5hbWUiOiAic2NyaXB0cy9nZW5lcmF0ZS5weSIsCiAgICAgICAgImRpZ2VzdCI6ICI3Y2EzM2U1MDJhMzM4MTk4M2QxODM0MTBiNDYxMzYwOTc2YTRmYjE1ZjAwZDI0M2VjMGZhN2YyMDFlZWE2MTAyIgogICAgICB9LAogICAgICB7CiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiLAogICAgICAgICJuYW1lIjogInNraWxsLWNhcmQubWQiLAogICAgICAgICJkaWdlc3QiOiAiMzBkYzM4NTViNjU1ZDIzM2NmNjhkNTYyZDg5YTZkMzFhMDEyNjE0OGE5NmQwMzU0ODRiNzllY2ExYzQ1MGRiNSIKICAgICAgfQogICAgXQogIH0KfQ==","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGUCMA602PMJrqNA3xajzlnQRKiYanKrZLeiStRyiRH6lxX61BupteDnWzzdGjoIU4qtxwIxAKxWQChqv1L2uBj+AsowmDwiiDNfd+eWYLCQmnNGwzEqt1kIFBGgRgMN8Op68aJLcg==","keyid":""}]}} \ No newline at end of file +{"mediaType":"application/vnd.dev.sigstore.bundle.v0.3+json","verificationMaterial":{"x509CertificateChain":{"certificates":[{"rawBytes":"MIICgzCCAgmgAwIBAgIUKIyS7SxNteQIiWzK1dWj85E6520wCgYIKoZIzj0EAwMwVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwHhcNMjYwNDAxMDAwMDAwWhcNMjgwNDIyMTUzMzA5WjBUMQswCQYDVQQGEwJVUzEbMBkGA1UECgwSTlZJRElBIENvcnBvcmF0aW9uMSgwJgYDVQQDDB9OVklESUEgQWdlbnQgU2tpbGxzIFNpZ25pbmcgMDAxMHYwEAYHKoZIzj0CAQYFK4EEACIDYgAEYoRM9bQl/dGlwSRNi6bTpIJUXH8Nv9GciP6LSflJYYMLCc296kpyuTSsk5ddbAWiDcFX3C/ydX3jwc+qCLYP6uHy9XphyLjOQ27Yb2J6rBLVtRBS1mgGco/Gr7fL6ODco4GaMIGXMB0GA1UdDgQWBBRQ/5ZW3nJ6lmo9SVk7I15o7UGmpTAfBgNVHSMEGDAWgBRPGpILxMBBleJSsBGjrMKsby1CgjAMBgNVHRMBAf8EAjAAMA4GA1UdDwEB/wQEAwIHgDA3BggrBgEFBQcBAQQrMCkwJwYIKwYBBQUHMAGGG2h0dHA6Ly9vY3NwLm5kaXMubnZpZGlhLmNvbTAKBggqhkjOPQQDAwNoADBlAjAUygu/GiOCIXrgGr4SmLgeEVDcEitfFUv7ALbvLVGVyMysB3mxmO/uInZfXzWcJZsCMQDxuoxj4ZmO30jhkPIcCxGFCOvnUsnfU3TfGcouYm4M6iRpbKvtVnHPiy4bi6pcKf0="},{"rawBytes":"MIICiDCCAg6gAwIBAgIUZsIuSv9NkpJCNqtYEfCouVv5BzowCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowVTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjEpMCcGA1UEAwwgTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBJQ0EgMDEwdjAQBgcqhkjOPQIBBgUrgQQAIgNiAASI72cR3ctKGg4VWnB3bNja6g1Z2PnOmFEopkPof+QeIcPk9rT+g9MjJnq51EQXL93a7C2GJ9J985G4o2V85VD7wJ1RaXhluHW2rf3y8bQGeAYaKMr5s/hUgn+M3/9WlWejgaAwgZ0wHQYDVR0OBBYEFE8akgvEwEGV4lKwEaOswqxvLUKCMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMBIGA1UdEwEB/wQIMAYBAf8CAQAwDgYDVR0PAQH/BAQDAgEGMDcGCCsGAQUFBwEBBCswKTAnBggrBgEFBQcwAYYbaHR0cDovL29jc3AubmRpcy5udmlkaWEuY29tMAoGCCqGSM49BAMDA2gAMGUCMQCeIMMfAbyzPDacw2MxG+Yt1cikrJX/DVxiGfXuHmkkXn6VgSzE79+lkqDErpVO2gYCMCNEColOyvUvkzZGUEI1hQ3PfMgi3FIo9tHoBKMw4/wGBLFpu/0ubtmbBXM6/UMOEw=="},{"rawBytes":"MIICRTCCAcygAwIBAgIUeJdY3rV86EdvFmG7L8LJBsyQFYkwCgYIKoZIzj0EAwMwUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTAgFw0yNjA0MDEwMDAwMDBaGA85OTk5MTIzMTIzNTk1OVowUTELMAkGA1UEBhMCVVMxGzAZBgNVBAoMEk5WSURJQSBDb3Jwb3JhdGlvbjElMCMGA1UEAwwcTlZJRElBIEFnZW50IENhcGFiaWxpdGllcyBDQTB2MBAGByqGSM49AgEGBSuBBAAiA2IABAYpiXCDjJ9NT2eSDhyHJVSw1Tbze18cGG2F/578oWvHxg23eQAhNRYdq88i1iOshZSO6C29doKui5Xpmo/7Ctw9Sx4PP2RzOmIuOLCuTdNtKcTRwi4GEsd5BAFvWj42M6NjMGEwHQYDVR0OBBYEFItnoAjjfuCEUvzyvWyI2vOGvwPjMB8GA1UdIwQYMBaAFItnoAjjfuCEUvzyvWyI2vOGvwPjMA8GA1UdEwEB/wQFMAMBAf8wDgYDVR0PAQH/BAQDAgEGMAoGCCqGSM49BAMDA2cAMGQCMCwtAjWLaNwgGWNCgdyNoTyvNhqWRECRJV2r3+7w8g0PL6NHLOsbkgE09BH95h8XlgIwTaQmbbUh2ChAJ5TA1wRiVDnCcvbzHlZl2jM2FcwQQZlk19LOAbyGMRixbu2Ww/rj"}]},"tlogEntries":[]},"dsseEnvelope":{"payload":"ewogICJfdHlwZSI6ICJodHRwczovL2luLXRvdG8uaW8vU3RhdGVtZW50L3YxIiwKICAic3ViamVjdCI6IFsKICAgIHsKICAgICAgIm5hbWUiOiAiZXZvMi1uaW0iLAogICAgICAiZGlnZXN0IjogewogICAgICAgICJzaGEyNTYiOiAiNjY5NzZmYzQyZWFmNjk0YjU5YzhiYTRlYjIzMWI3NzIyOGQwN2JhZGJlM2VhMDRkYThmODRhOGI0ZGRjOTQwNSIKICAgICAgfQogICAgfQogIF0sCiAgInByZWRpY2F0ZVR5cGUiOiAiaHR0cHM6Ly9tb2RlbF9zaWduaW5nL3NpZ25hdHVyZS92MS4wIiwKICAicHJlZGljYXRlIjogewogICAgInNlcmlhbGl6YXRpb24iOiB7CiAgICAgICJpZ25vcmVfcGF0aHMiOiBbCiAgICAgICAgIi5naXQiLAogICAgICAgICIuZ2l0aHViIiwKICAgICAgICAiLmdpdGlnbm9yZSIsCiAgICAgICAgIi5naXRhdHRyaWJ1dGVzIgogICAgICBdLAogICAgICAiYWxsb3dfc3ltbGlua3MiOiBmYWxzZSwKICAgICAgIm1ldGhvZCI6ICJmaWxlcyIsCiAgICAgICJoYXNoX3R5cGUiOiAic2hhMjU2IgogICAgfSwKICAgICJyZXNvdXJjZXMiOiBbCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJCRU5DSE1BUksubWQiLAogICAgICAgICJkaWdlc3QiOiAiYjJjYjcwZTVlMTkxNmZkNjQxZmQzNjQ3ZjI1ZjNkODdiYjI3ZGFmNTk1ZGU3NGU1YzI2NTQxNzZiMmNhMzg5MyIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJTS0lMTC5tZCIsCiAgICAgICAgImRpZ2VzdCI6ICIzNTA3ZTVkNzExYjc2MzljNDdmMTI4MGI0YzZjNjJhNGU0NzkzYWMzYmY5MmJlZDhiNTFmNTMwZWRhZjJiNDcyIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogImNvbmZpZy9za2lsbHNwZWN0b3ItYmFzZWxpbmUueW1sIiwKICAgICAgICAiZGlnZXN0IjogImUxZDQxOWY0NmY3YzM0ZTM0OGU4ZmQ2OTViZWI5NWZlODMwZjAwZDg4MTBmODBlMzZmYjNiNjc1ZjFmMDcwYmMiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAiZXZhbHMvY29uZmlnLnltbCIsCiAgICAgICAgImRpZ2VzdCI6ICJiYTJiOGNmMGVhZDEzYmZiNjVkODEzMDllMTYzMTYyZGJjMmM2YzUwNTg5NjRjNGFlYTQxNGUwMDcwODk4YThhIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogImV2YWxzL2V2YWxzLmpzb24iLAogICAgICAgICJkaWdlc3QiOiAiMTZkOGU3MzVlYTRiMTJiZWE3M2Y4NzUxZmJjYTFjYTcxMjk0N2Q5ODQyYWRkMzAxMDMzNDI1MGYzY2IzZTc4ZCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJldmFscy90cmlnZ2VyX2V2YWxzLmpzb24iLAogICAgICAgICJkaWdlc3QiOiAiMmVlMmQyZjgyOTg4NGUxNDJiNmU2NGY2YmE0ODJiNjZmMjBhZjFlOGIwNTA3NDdlNWNhNDMxZGQzNDEyYTY4NCIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJyZWZlcmVuY2VzL2FwaS5tZCIsCiAgICAgICAgImRpZ2VzdCI6ICI5Y2Y2MDVjMjViMmExMjEwMmRkYjQwOTQ4YzQxZjY1M2VjNGRiODYzNTRiYjFiMzQ2NjI5YzJhZjk4MjNiZDdiIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfSwKICAgICAgewogICAgICAgICJuYW1lIjogInJlZmVyZW5jZXMvZXhhbXBsZXMubWQiLAogICAgICAgICJkaWdlc3QiOiAiZGNlMWQwNzdiNmY2MzNlYjQ1ODRjM2VmODgyMDAxZTFhZDkyMjRiNDZkZTllMTJhZWUwNGRmZGI2Mjc3OGM2MiIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJyZWZlcmVuY2VzL3BhcmFtZXRlcnMubWQiLAogICAgICAgICJkaWdlc3QiOiAiMTYwNjI3N2Q2MGFhMDI5ZWZiMWNlN2M4NzhiNGFkODgwZDdmODI0NjE0YTNmYjc1ZWFjMWI3MDU5YjQ1M2ZiNyIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJyZWZlcmVuY2VzL3NjaWVuY2UubWQiLAogICAgICAgICJkaWdlc3QiOiAiZmVhZGM1NmVmZGRmOTZlNDUxNmYxZGRmZTFmZjlmOWMwMmFlYjg2OTIyYWZlMWFlYzNiYWUyM2NmNmQwYWY1OSIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJyZWZlcmVuY2VzL3ZhbGlkYXRpb24ubWQiLAogICAgICAgICJkaWdlc3QiOiAiNzhlMDY3N2JiZmRlOTM5ODcyNjk0MTM1YmIyMTdhNjgyOTZhNDE4NDA0M2M4ZWNjNWM1MjU4YzRlOTEzNTA1MSIsCiAgICAgICAgImFsZ29yaXRobSI6ICJzaGEyNTYiCiAgICAgIH0sCiAgICAgIHsKICAgICAgICAibmFtZSI6ICJzY3JpcHRzL2dlbmVyYXRlLnB5IiwKICAgICAgICAiZGlnZXN0IjogIjdjYTMzZTUwMmEzMzgxOTgzZDE4MzQxMGI0NjEzNjA5NzZhNGZiMTVmMDBkMjQzZWMwZmE3ZjIwMWVlYTYxMDIiLAogICAgICAgICJhbGdvcml0aG0iOiAic2hhMjU2IgogICAgICB9LAogICAgICB7CiAgICAgICAgIm5hbWUiOiAic2tpbGwtY2FyZC5tZCIsCiAgICAgICAgImRpZ2VzdCI6ICJhYzUyMWNhZDA5MGNlNDExYzkzZWY0Njg2M2U2MzVjZTQ2MjAzZTQzMjdhNTUwNTIyZjEzMTcyNjMzODFiYWUxIiwKICAgICAgICAiYWxnb3JpdGhtIjogInNoYTI1NiIKICAgICAgfQogICAgXQogIH0KfQ==","payloadType":"application/vnd.in-toto+json","signatures":[{"sig":"MGUCMBQ/Ni8/LgJpPWQb02NgbgYSBg1JNYSZoPHcZdhn1BE9vg+lQ8aWJpmrQqiwx95eAwIxAO9qsdF+iFSuffAYWrFv3vACelB5iQhhIoK9qp1TnPVh28PiLBJgiiC5qSEUaozoKw==","keyid":""}]}} \ No newline at end of file