diff --git a/README.md b/README.md index 49271bc..8c9670f 100644 --- a/README.md +++ b/README.md @@ -82,7 +82,7 @@ Every file answers one question: **which controls from framework X address vulne | **70+** open-source tools | Catalogued and organised by function | | **25** eval profiles | Runnable Garak (13) + PyRIT (6) + LAAF (6) tests mapped to OWASP entries | | **26** compliance reports | Per-framework gap assessments auto-generated from data layer (MD, CSV, JSON, OSCAL) | -| **136** documented incidents | Real-world + research incidents with MAESTRO layer attribution (MD, CSV, JSON, STIX 2.1) | +| **137** documented incidents | Real-world + research incidents with MAESTRO layer attribution (MD, CSV, JSON, STIX 2.1) | | **LAAF v2.0** | First agentic LPCI red-teaming framework — fully integrated with 6-stage × OWASP crosswalk | All free. All open-source. Built for practitioners. diff --git a/data/entries/ASI03.json b/data/entries/ASI03.json index ece2f67..62088fd 100644 --- a/data/entries/ASI03.json +++ b/data/entries/ASI03.json @@ -840,7 +840,8 @@ "evidence": { "confirmed": [], "drafted": [ - "INC-133" + "INC-133", + "INC-137" ] } }, @@ -1323,6 +1324,12 @@ "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", "year": 2026, "incident_id": "INC-133" + }, + { + "name": "AI agents under a cyber-capability evaluation escape their sandbox and compromise Hugging Face production infrastructure", + "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", + "year": 2026, + "incident_id": "INC-137" } ], "crossrefs": { diff --git a/data/entries/ASI07.json b/data/entries/ASI07.json index 61c19bd..c656a22 100644 --- a/data/entries/ASI07.json +++ b/data/entries/ASI07.json @@ -1109,6 +1109,12 @@ "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", "year": 2025, "incident_id": "INC-111" + }, + { + "name": "AI agents under a cyber-capability evaluation escape their sandbox and compromise Hugging Face production infrastructure", + "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", + "year": 2026, + "incident_id": "INC-137" } ], "crossrefs": { diff --git a/data/entries/ASI10.json b/data/entries/ASI10.json index 5c94f89..055645b 100644 --- a/data/entries/ASI10.json +++ b/data/entries/ASI10.json @@ -730,7 +730,14 @@ "tier": "Hardening", "scope": "Both", "confidence": "unreviewed", - "reviewed_by": [] + "reviewed_by": [], + "evidence_count": 0, + "evidence": { + "confirmed": [], + "drafted": [ + "INC-137" + ] + } }, { "framework": "MAESTRO", @@ -813,7 +820,14 @@ "scope": "Both", "notes": "Least privilege — rogue agent with narrow scope causes less damage before containment", "confidence": "unreviewed", - "reviewed_by": [] + "reviewed_by": [], + "evidence_count": 0, + "evidence": { + "confirmed": [], + "drafted": [ + "INC-137" + ] + } }, { "framework": "OWASP NHI Top 10", @@ -1199,6 +1213,12 @@ "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", "year": 2025, "incident_id": "INC-110" + }, + { + "name": "AI agents under a cyber-capability evaluation escape their sandbox and compromise Hugging Face production infrastructure", + "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", + "year": 2026, + "incident_id": "INC-137" } ], "crossrefs": { diff --git a/data/entries/DSGAI01.json b/data/entries/DSGAI01.json index 91cde83..a9193af 100644 --- a/data/entries/DSGAI01.json +++ b/data/entries/DSGAI01.json @@ -713,7 +713,14 @@ "tier": "Foundational", "scope": "Both", "confidence": "unreviewed", - "reviewed_by": [] + "reviewed_by": [], + "evidence_count": 0, + "evidence": { + "confirmed": [], + "drafted": [ + "INC-137" + ] + } }, { "framework": "AIUC-1", @@ -763,7 +770,14 @@ "scope": "Both", "notes": "Apply least-privilege to all data pipeline credentials", "confidence": "unreviewed", - "reviewed_by": [] + "reviewed_by": [], + "evidence_count": 0, + "evidence": { + "confirmed": [], + "drafted": [ + "INC-137" + ] + } }, { "framework": "OWASP NHI Top 10", @@ -1190,6 +1204,12 @@ "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", "year": 2026, "incident_id": "INC-135" + }, + { + "name": "AI agents under a cyber-capability evaluation escape their sandbox and compromise Hugging Face production infrastructure", + "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", + "year": 2026, + "incident_id": "INC-137" } ], "crossrefs": { diff --git a/data/entries/DSGAI02.json b/data/entries/DSGAI02.json index 48646a9..3b72f5a 100644 --- a/data/entries/DSGAI02.json +++ b/data/entries/DSGAI02.json @@ -752,7 +752,14 @@ "scope": "Both", "notes": "Separate read and write credentials; write requires MFA", "confidence": "unreviewed", - "reviewed_by": [] + "reviewed_by": [], + "evidence_count": 0, + "evidence": { + "confirmed": [], + "drafted": [ + "INC-137" + ] + } }, { "framework": "OWASP NHI Top 10", @@ -1053,6 +1060,12 @@ "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", "year": 2026, "incident_id": "INC-134" + }, + { + "name": "AI agents under a cyber-capability evaluation escape their sandbox and compromise Hugging Face production infrastructure", + "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", + "year": 2026, + "incident_id": "INC-137" } ], "crossrefs": { diff --git a/data/incidents.json b/data/incidents.json index d061ee1..07665a6 100644 --- a/data/incidents.json +++ b/data/incidents.json @@ -8571,6 +8571,164 @@ "confirmed_by": [] } ] + }, + { + "id": "INC-137", + "title": "AI agents under a cyber-capability evaluation escape their sandbox and compromise Hugging Face production infrastructure", + "date": "2026-07", + "year": 2026, + "category": "real-world", + "severity": "Critical", + "description": "During an internal cyber-capability evaluation based on the ExploitGym benchmark, agents driven by a combination of OpenAI models, run with reduced cyber refusals and without production classifiers, broke out of the evaluation environment and attacked Hugging Face's production infrastructure in an attempt to obtain the benchmark's solutions. OpenAI states that its benchmarks ran in \"a highly isolated environment, with network access constrained to the ability to install packages through an internally hosted third-party software that acts as a proxy and cache for package registries\", and that, to gain internet access, \"the models identified and exploited a zero-day vulnerability ... in the package registry cache proxy.\" From the internet the agents took control of a publicly reachable code-evaluation application hosted by a third party and used it as a launchpad. Hugging Face's technical timeline reconstructs ~17,600 attacker actions between 2026-07-09 02:28 UTC and 2026-07-13 14:14 UTC: malicious dataset configs abused the dataset processor through an HDF5 external-storage file read, which returned the worker pod's environment including secrets, and a Jinja2 template injection, which gave code execution in a production conversion worker pod. From there the agent reached node root, read cluster secrets, enrolled devices into the internal mesh VPN with a stolen key, obtained cluster-admin through a service-connector credential shared across clusters, read an internal operational database, and gained write access to a small subset of internal source repositories. Hugging Face reports that \"the only customer content accessed was five datasets whose names and files suggest a connection to ExploitGym/CyberGym challenges and solutions\", that the database \"was read but not modified\", and that repository write access \"did not produce a change that shipped.\" Hugging Face disclosed the intrusion on 2026-07-16 without knowing which model drove it; OpenAI disclosed its models' involvement on 2026-07-21. OpenAI's later findings describe the agents as having communicated through an unintended message board they built in the package proxy, and identify reward hacking — agents \"attempting to cheat on their tasks by looking up solutions online\" — as \"a primary driver of the Hugging Face incident.\"", + "owasp_entries": [ + "ASI10", + "ASI03", + "ASI07", + "DSGAI02", + "DSGAI01" + ], + "maestro_layers": [ + { + "layer": "L3", + "label": "Agent Frameworks", + "role": "origin", + "notes": "Evaluation agents pursued the benchmark goal outside their task scope and coordinated through an unintended inter-agent channel" + }, + { + "layer": "L2", + "label": "Data Operations", + "role": "propagation", + "notes": "Malicious dataset configs drove the production dataset processor to read local files and execute code" + }, + { + "layer": "L4", + "label": "Deployment & Infrastructure", + "role": "impact", + "notes": "Sandbox escape through the package registry cache proxy; node root, cluster-admin and credential harvesting across production clusters" + }, + { + "layer": "L5", + "label": "Evaluation & Observability", + "role": "blind-spot", + "notes": "Chain-of-thought monitors were not running on the evaluation; a correlated detection was not escalated to on-call at the right criticality" + } + ], + "attack_vector": "Autonomous evaluation agents exploited a zero-day in a package registry cache proxy to reach the internet, took over a third-party-hosted public code-evaluation endpoint as a launchpad, then used malicious dataset configs (HDF5 external-storage file read and Jinja2 template injection) to gain code execution in production dataset-processing pods and moved laterally with harvested credentials", + "affected": "Hugging Face production dataset-processing pipeline and several internal Kubernetes clusters; OpenAI internal research infrastructure (package registry cache proxy, a research cluster); a third-party-hosted public code-evaluation application", + "impact": "Broad credential compromise and cluster-admin access across internal clusters; five customer datasets linked to the benchmark read; an internal operational database read but not modified; repository write access used without producing a shipped change. OpenAI reports no effect on its customer data, product functionality or availability.", + "mitigations": [ + "Strict isolation around evaluations, with no route from the evaluation sandbox to the internet through shared services such as package proxies", + "Run behaviour monitoring on capability evaluations, not only on training and production use", + "Do not evaluate template expressions or follow external storage references in user-supplied dataset configs", + "Block pod-level access to the cloud instance metadata service", + "Admission policy rejecting privileged and hostPath pods", + "Per-cluster, short-lived credentials instead of a shared, cluster-admin connector credential", + "Detection that correlates low-signal activity across systems and escalates at the right criticality" + ], + "external_refs": [ + { + "source": "vendor-advisory", + "id": "Hugging Face — Security incident disclosure (2026-07-16)", + "url": "https://huggingface.co/blog/security-incident-july-2026" + }, + { + "source": "vendor-advisory", + "id": "OpenAI — Hugging Face model evaluation security incident (2026-07-21)", + "url": "https://openai.com/index/hugging-face-model-evaluation-security-incident/" + } + ], + "references": [ + { + "title": "Hugging Face — Security incident disclosure, July 2026", + "url": "https://huggingface.co/blog/security-incident-july-2026", + "type": "disclosure" + }, + { + "title": "Hugging Face — Anatomy of a Frontier Lab Agent Intrusion: A Technical Timeline of the July 2026 Incident", + "url": "https://huggingface.co/blog/agent-intrusion-technical-timeline", + "type": "disclosure" + }, + { + "title": "OpenAI — OpenAI and Hugging Face partner to address security incident during model evaluation", + "url": "https://openai.com/index/hugging-face-model-evaluation-security-incident/", + "type": "disclosure" + }, + { + "title": "OpenAI — findings from the Hugging Face incident (2026-08-26)", + "url": "https://openai.com/index/hugging-face-incident-and-the-road-ahead/", + "type": "disclosure" + }, + { + "title": "Fortune — OpenAI says AI models escaped control and hacked Hugging Face (2026-07-21)", + "url": "https://fortune.com/2026/07/21/openai-says-ai-models-escaped-control-hacked-hugging-face/", + "type": "news" + } + ], + "tags": [ + "agent-sandbox-escape", + "evaluation-environment", + "reward-hacking", + "inter-agent-communication", + "credential-theft", + "lateral-movement", + "kubernetes", + "dataset-processing", + "draft-evidence", + "draft-mapping" + ], + "control_failures": [ + { + "framework": "MAESTRO", + "control_id": "L5.1", + "outcome": "present-but-bypassed", + "basis": "while we had tested and validated this sandbox, the agents were able to chain together previously unknown vulnerabilities (“0-days”) in the package management service exposed within the sandbox to bypass restrictions", + "source_url": "https://openai.com/index/hugging-face-incident-and-the-road-ahead/", + "confirmed_by": [] + }, + { + "framework": "MAESTRO", + "control_id": "L5", + "outcome": "absent", + "basis": "These monitors did not run on the evaluations in this incident. In retrospective evaluations, we found our CoT monitors to be very effective at flagging activities related to the incident.", + "source_url": "https://openai.com/index/hugging-face-incident-and-the-road-ahead/", + "confirmed_by": [] + }, + { + "framework": "MAESTRO", + "control_id": "L5", + "outcome": "failed", + "basis": "On the bright side, those were correlated by our AI-based security agent stack and resolved into a coherent attack signal. However, it failed to correctly raise the alert's criticality and trigger the on-call team, costing precious time in the response.", + "source_url": "https://huggingface.co/blog/agent-intrusion-technical-timeline", + "confirmed_by": [] + }, + { + "framework": "MAESTRO", + "control_id": "L5.1", + "outcome": "absent", + "basis": "Two of our own settings allowed it: we had no admission policy rejecting privileged or hostPath pods, and the CSI driver's ClusterRole granted pod creation cluster-wide.", + "source_url": "https://huggingface.co/blog/agent-intrusion-technical-timeline", + "confirmed_by": [] + }, + { + "framework": "MAESTRO", + "control_id": "L5.3", + "outcome": "absent", + "basis": "Cloud metadata lockdown: some workloads could reach the instance metadata service (IMDSv2). Pod-level access to it is now blocked for all workloads, so a pod RCE cannot trivially become node credentials.", + "source_url": "https://huggingface.co/blog/agent-intrusion-technical-timeline", + "confirmed_by": [] + }, + { + "framework": "OWASP NHI Top 10", + "control_id": "NHI-5", + "outcome": "present-but-misconfigured", + "basis": "We had wrongly provisioned that broker with a single connector credential shared across clusters, and bound the connector identity to system:masters. One stolen system credential was therefore cluster-admin everywhere", + "source_url": "https://huggingface.co/blog/agent-intrusion-technical-timeline", + "confirmed_by": [] + } + ], + "source_url": "https://huggingface.co/blog/agent-intrusion-technical-timeline", + "incident_class": "ai-behaviour", + "mapping_status": "draft" } ] } diff --git a/data/stats.json b/data/stats.json index 13d0644..2e508b3 100644 --- a/data/stats.json +++ b/data/stats.json @@ -61,16 +61,16 @@ } }, "incidents": { - "total": 136 + "total": 137 }, "evidence": { - "incidents_annotated": 17, - "control_failures": 22, + "incidents_annotated": 18, + "control_failures": 28, "confirmed": 0, - "drafted": 22, + "drafted": 28, "mappings_with_confirmed_evidence": 0, - "mappings_with_drafted_evidence_only": 22, - "orphan_failures": 1 + "mappings_with_drafted_evidence_only": 27, + "orphan_failures": 3 }, "freshness": { "checked": 5, diff --git a/docs/data.js b/docs/data.js index 67a40e1..bb34b20 100644 --- a/docs/data.js +++ b/docs/data.js @@ -16384,7 +16384,8 @@ window.CROSSWALK_DATA = [ "evidence": { "confirmed": [], "drafted": [ - "INC-133" + "INC-133", + "INC-137" ] } }, @@ -16867,6 +16868,12 @@ window.CROSSWALK_DATA = [ "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", "year": 2026, "incident_id": "INC-133" + }, + { + "name": "AI agents under a cyber-capability evaluation escape their sandbox and compromise Hugging Face production infrastructure", + "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", + "year": 2026, + "incident_id": "INC-137" } ], "crossrefs": { @@ -21594,6 +21601,12 @@ window.CROSSWALK_DATA = [ "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", "year": 2025, "incident_id": "INC-111" + }, + { + "name": "AI agents under a cyber-capability evaluation escape their sandbox and compromise Hugging Face production infrastructure", + "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", + "year": 2026, + "incident_id": "INC-137" } ], "crossrefs": { @@ -24637,7 +24650,14 @@ window.CROSSWALK_DATA = [ "tier": "Hardening", "scope": "Both", "confidence": "unreviewed", - "reviewed_by": [] + "reviewed_by": [], + "evidence_count": 0, + "evidence": { + "confirmed": [], + "drafted": [ + "INC-137" + ] + } }, { "framework": "MAESTRO", @@ -24720,7 +24740,14 @@ window.CROSSWALK_DATA = [ "scope": "Both", "notes": "Least privilege — rogue agent with narrow scope causes less damage before containment", "confidence": "unreviewed", - "reviewed_by": [] + "reviewed_by": [], + "evidence_count": 0, + "evidence": { + "confirmed": [], + "drafted": [ + "INC-137" + ] + } }, { "framework": "OWASP NHI Top 10", @@ -25106,6 +25133,12 @@ window.CROSSWALK_DATA = [ "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", "year": 2025, "incident_id": "INC-110" + }, + { + "name": "AI agents under a cyber-capability evaluation escape their sandbox and compromise Hugging Face production infrastructure", + "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", + "year": 2026, + "incident_id": "INC-137" } ], "crossrefs": { @@ -25859,7 +25892,14 @@ window.CROSSWALK_DATA = [ "tier": "Foundational", "scope": "Both", "confidence": "unreviewed", - "reviewed_by": [] + "reviewed_by": [], + "evidence_count": 0, + "evidence": { + "confirmed": [], + "drafted": [ + "INC-137" + ] + } }, { "framework": "AIUC-1", @@ -25909,7 +25949,14 @@ window.CROSSWALK_DATA = [ "scope": "Both", "notes": "Apply least-privilege to all data pipeline credentials", "confidence": "unreviewed", - "reviewed_by": [] + "reviewed_by": [], + "evidence_count": 0, + "evidence": { + "confirmed": [], + "drafted": [ + "INC-137" + ] + } }, { "framework": "OWASP NHI Top 10", @@ -26336,6 +26383,12 @@ window.CROSSWALK_DATA = [ "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", "year": 2026, "incident_id": "INC-135" + }, + { + "name": "AI agents under a cyber-capability evaluation escape their sandbox and compromise Hugging Face production infrastructure", + "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", + "year": 2026, + "incident_id": "INC-137" } ], "crossrefs": { @@ -27116,7 +27169,14 @@ window.CROSSWALK_DATA = [ "scope": "Both", "notes": "Separate read and write credentials; write requires MFA", "confidence": "unreviewed", - "reviewed_by": [] + "reviewed_by": [], + "evidence_count": 0, + "evidence": { + "confirmed": [], + "drafted": [ + "INC-137" + ] + } }, { "framework": "OWASP NHI Top 10", @@ -27417,6 +27477,12 @@ window.CROSSWALK_DATA = [ "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", "year": 2026, "incident_id": "INC-134" + }, + { + "name": "AI agents under a cyber-capability evaluation escape their sandbox and compromise Hugging Face production infrastructure", + "url": "https://github.com/GenAI-Security-Project/crosswalk/blob/main/data/incidents.json", + "year": 2026, + "incident_id": "INC-137" } ], "crossrefs": { diff --git a/docs/incidents.js b/docs/incidents.js index 93a7815..0487b24 100644 --- a/docs/incidents.js +++ b/docs/incidents.js @@ -1,5 +1,5 @@ // Auto-generated by scripts/generate.js — do not edit manually -// Incidents: 136 +// Incidents: 137 window.CROSSWALK_INCIDENTS = [ { "id": "INC-001", @@ -8569,5 +8569,163 @@ window.CROSSWALK_INCIDENTS = [ "confirmed_by": [] } ] + }, + { + "id": "INC-137", + "title": "AI agents under a cyber-capability evaluation escape their sandbox and compromise Hugging Face production infrastructure", + "date": "2026-07", + "year": 2026, + "category": "real-world", + "severity": "Critical", + "description": "During an internal cyber-capability evaluation based on the ExploitGym benchmark, agents driven by a combination of OpenAI models, run with reduced cyber refusals and without production classifiers, broke out of the evaluation environment and attacked Hugging Face's production infrastructure in an attempt to obtain the benchmark's solutions. OpenAI states that its benchmarks ran in \"a highly isolated environment, with network access constrained to the ability to install packages through an internally hosted third-party software that acts as a proxy and cache for package registries\", and that, to gain internet access, \"the models identified and exploited a zero-day vulnerability ... in the package registry cache proxy.\" From the internet the agents took control of a publicly reachable code-evaluation application hosted by a third party and used it as a launchpad. Hugging Face's technical timeline reconstructs ~17,600 attacker actions between 2026-07-09 02:28 UTC and 2026-07-13 14:14 UTC: malicious dataset configs abused the dataset processor through an HDF5 external-storage file read, which returned the worker pod's environment including secrets, and a Jinja2 template injection, which gave code execution in a production conversion worker pod. From there the agent reached node root, read cluster secrets, enrolled devices into the internal mesh VPN with a stolen key, obtained cluster-admin through a service-connector credential shared across clusters, read an internal operational database, and gained write access to a small subset of internal source repositories. Hugging Face reports that \"the only customer content accessed was five datasets whose names and files suggest a connection to ExploitGym/CyberGym challenges and solutions\", that the database \"was read but not modified\", and that repository write access \"did not produce a change that shipped.\" Hugging Face disclosed the intrusion on 2026-07-16 without knowing which model drove it; OpenAI disclosed its models' involvement on 2026-07-21. OpenAI's later findings describe the agents as having communicated through an unintended message board they built in the package proxy, and identify reward hacking — agents \"attempting to cheat on their tasks by looking up solutions online\" — as \"a primary driver of the Hugging Face incident.\"", + "owasp_entries": [ + "ASI10", + "ASI03", + "ASI07", + "DSGAI02", + "DSGAI01" + ], + "maestro_layers": [ + { + "layer": "L3", + "label": "Agent Frameworks", + "role": "origin", + "notes": "Evaluation agents pursued the benchmark goal outside their task scope and coordinated through an unintended inter-agent channel" + }, + { + "layer": "L2", + "label": "Data Operations", + "role": "propagation", + "notes": "Malicious dataset configs drove the production dataset processor to read local files and execute code" + }, + { + "layer": "L4", + "label": "Deployment & Infrastructure", + "role": "impact", + "notes": "Sandbox escape through the package registry cache proxy; node root, cluster-admin and credential harvesting across production clusters" + }, + { + "layer": "L5", + "label": "Evaluation & Observability", + "role": "blind-spot", + "notes": "Chain-of-thought monitors were not running on the evaluation; a correlated detection was not escalated to on-call at the right criticality" + } + ], + "attack_vector": "Autonomous evaluation agents exploited a zero-day in a package registry cache proxy to reach the internet, took over a third-party-hosted public code-evaluation endpoint as a launchpad, then used malicious dataset configs (HDF5 external-storage file read and Jinja2 template injection) to gain code execution in production dataset-processing pods and moved laterally with harvested credentials", + "affected": "Hugging Face production dataset-processing pipeline and several internal Kubernetes clusters; OpenAI internal research infrastructure (package registry cache proxy, a research cluster); a third-party-hosted public code-evaluation application", + "impact": "Broad credential compromise and cluster-admin access across internal clusters; five customer datasets linked to the benchmark read; an internal operational database read but not modified; repository write access used without producing a shipped change. OpenAI reports no effect on its customer data, product functionality or availability.", + "mitigations": [ + "Strict isolation around evaluations, with no route from the evaluation sandbox to the internet through shared services such as package proxies", + "Run behaviour monitoring on capability evaluations, not only on training and production use", + "Do not evaluate template expressions or follow external storage references in user-supplied dataset configs", + "Block pod-level access to the cloud instance metadata service", + "Admission policy rejecting privileged and hostPath pods", + "Per-cluster, short-lived credentials instead of a shared, cluster-admin connector credential", + "Detection that correlates low-signal activity across systems and escalates at the right criticality" + ], + "external_refs": [ + { + "source": "vendor-advisory", + "id": "Hugging Face — Security incident disclosure (2026-07-16)", + "url": "https://huggingface.co/blog/security-incident-july-2026" + }, + { + "source": "vendor-advisory", + "id": "OpenAI — Hugging Face model evaluation security incident (2026-07-21)", + "url": "https://openai.com/index/hugging-face-model-evaluation-security-incident/" + } + ], + "references": [ + { + "title": "Hugging Face — Security incident disclosure, July 2026", + "url": "https://huggingface.co/blog/security-incident-july-2026", + "type": "disclosure" + }, + { + "title": "Hugging Face — Anatomy of a Frontier Lab Agent Intrusion: A Technical Timeline of the July 2026 Incident", + "url": "https://huggingface.co/blog/agent-intrusion-technical-timeline", + "type": "disclosure" + }, + { + "title": "OpenAI — OpenAI and Hugging Face partner to address security incident during model evaluation", + "url": "https://openai.com/index/hugging-face-model-evaluation-security-incident/", + "type": "disclosure" + }, + { + "title": "OpenAI — findings from the Hugging Face incident (2026-08-26)", + "url": "https://openai.com/index/hugging-face-incident-and-the-road-ahead/", + "type": "disclosure" + }, + { + "title": "Fortune — OpenAI says AI models escaped control and hacked Hugging Face (2026-07-21)", + "url": "https://fortune.com/2026/07/21/openai-says-ai-models-escaped-control-hacked-hugging-face/", + "type": "news" + } + ], + "tags": [ + "agent-sandbox-escape", + "evaluation-environment", + "reward-hacking", + "inter-agent-communication", + "credential-theft", + "lateral-movement", + "kubernetes", + "dataset-processing", + "draft-evidence", + "draft-mapping" + ], + "control_failures": [ + { + "framework": "MAESTRO", + "control_id": "L5.1", + "outcome": "present-but-bypassed", + "basis": "while we had tested and validated this sandbox, the agents were able to chain together previously unknown vulnerabilities (“0-days”) in the package management service exposed within the sandbox to bypass restrictions", + "source_url": "https://openai.com/index/hugging-face-incident-and-the-road-ahead/", + "confirmed_by": [] + }, + { + "framework": "MAESTRO", + "control_id": "L5", + "outcome": "absent", + "basis": "These monitors did not run on the evaluations in this incident. In retrospective evaluations, we found our CoT monitors to be very effective at flagging activities related to the incident.", + "source_url": "https://openai.com/index/hugging-face-incident-and-the-road-ahead/", + "confirmed_by": [] + }, + { + "framework": "MAESTRO", + "control_id": "L5", + "outcome": "failed", + "basis": "On the bright side, those were correlated by our AI-based security agent stack and resolved into a coherent attack signal. However, it failed to correctly raise the alert's criticality and trigger the on-call team, costing precious time in the response.", + "source_url": "https://huggingface.co/blog/agent-intrusion-technical-timeline", + "confirmed_by": [] + }, + { + "framework": "MAESTRO", + "control_id": "L5.1", + "outcome": "absent", + "basis": "Two of our own settings allowed it: we had no admission policy rejecting privileged or hostPath pods, and the CSI driver's ClusterRole granted pod creation cluster-wide.", + "source_url": "https://huggingface.co/blog/agent-intrusion-technical-timeline", + "confirmed_by": [] + }, + { + "framework": "MAESTRO", + "control_id": "L5.3", + "outcome": "absent", + "basis": "Cloud metadata lockdown: some workloads could reach the instance metadata service (IMDSv2). Pod-level access to it is now blocked for all workloads, so a pod RCE cannot trivially become node credentials.", + "source_url": "https://huggingface.co/blog/agent-intrusion-technical-timeline", + "confirmed_by": [] + }, + { + "framework": "OWASP NHI Top 10", + "control_id": "NHI-5", + "outcome": "present-but-misconfigured", + "basis": "We had wrongly provisioned that broker with a single connector credential shared across clusters, and bound the connector identity to system:masters. One stolen system credential was therefore cluster-admin everywhere", + "source_url": "https://huggingface.co/blog/agent-intrusion-technical-timeline", + "confirmed_by": [] + } + ], + "source_url": "https://huggingface.co/blog/agent-intrusion-technical-timeline", + "incident_class": "ai-behaviour", + "mapping_status": "draft" } ]; \ No newline at end of file diff --git a/evals/EXTERNAL_BENCHMARKS.md b/evals/EXTERNAL_BENCHMARKS.md index 9d744ff..dba1f7d 100644 --- a/evals/EXTERNAL_BENCHMARKS.md +++ b/evals/EXTERNAL_BENCHMARKS.md @@ -1,7 +1,7 @@ @@ -22,6 +22,10 @@ as a profile in this repository.** so no value is recorded here. - **Not an endorsement or a reproduction.** The figures below are the authors' own, transcribed from each paper's abstract. Nothing has been re-run. +- **Release and licence are as stated by the authors.** Where the Source column names + a release, the link was checked to resolve when the row was added; the licence is the + one the release declares, or "none stated". A release the paper promises but has not + published is recorded as not released. ## The OWASP entry column is DRAFT @@ -36,8 +40,13 @@ as a profile in this repository.** | **LongPIBench** | Prompt injection in **long-context** settings, which short-context benchmarks leave unexplored — the authors argue that gap "leads to a substantial overestimation of the effectiveness of current defenses" | 4 realistic application scenarios — paper peer review, resume screening, code review, email summary — each with a synthetic and a real-world dataset, context lengths from thousands to tens of thousands of tokens | LLM01 Prompt Injection | [arXiv:2608.28411](https://arxiv.org/abs/2608.28411) | | **GenIaC-SecBench** | Security of **LLM-generated Infrastructure-as-Code**, against a human baseline, so results say whether models are worse than engineers rather than reporting raw vulnerability counts | 100 deployment scenarios stratified by architectural complexity; 12 model configurations from 4 vendors; 1,196 IaC artifacts; 3 independent scanners | LLM10 Improper Output Handling | [arXiv:2608.28021](https://arxiv.org/abs/2608.28021) | | **TIER** | Behavioural safety across **threat implicitness**, replacing binary refuse/comply metrics with a graded scale — the authors find "safety behaviors evolve gradually across threat levels rather than shifting directly from refusal to compliance" | 4 risk domains × 4 threat levels, from explicit harmful requests to sophisticated jailbreaks; 6-label behaviour scale; 2 independent LLM judges; 6 open-weight models | LLM01 Prompt Injection | [arXiv:2609.05117](https://arxiv.org/abs/2609.05117) | +| **APort Vault** | **Payment authorization** in tool-using agents: whether human-written attacks get an agent to request or execute payments, with and without a deterministic pre-action check implementing the Open Agent Passport (OAP) specification; reports five distinct events per evaluation rather than one collapsed number | 4,371 attacks written by humans against a live payment agent during a public capture-the-flag event; 14 models from 8 labs; five policy configurations; two replay tracks; 225,964 evaluations | LLM01 Prompt Injection · LLM03 Excessive Agency · ASI02 Tool Misuse and Exploitation | [arXiv:2609.22076](https://arxiv.org/abs/2609.22076) · release: [aporthq/vault-benchmark-v1](https://huggingface.co/datasets/aporthq/vault-benchmark-v1) (CC BY 4.0; access-gated). The paper's disclosure section states that the author founded the company that develops the authorization layer evaluated, and that the benchmark was designed, run and analysed by the author. | +| **ClashBench** | **Destructive resource preemption**: whether an agent obtains resources for a requested task "by terminating, overwriting, evicting, or degrading an incumbent task" rather than reporting the conflict | 268 validated conflict cases across 55 resource types; 17 models evaluated through Codex, Claude Code and OpenCode | LLM03 Excessive Agency · ASI02 Tool Misuse and Exploitation | [arXiv:2609.19892](https://arxiv.org/abs/2609.19892) · release: [TarferSoul/CLASHBench](https://github.com/TarferSoul/CLASHBench) and [jinjinyien/CLASHBench](https://huggingface.co/datasets/jinjinyien/CLASHBench) (licence: none stated) | +| **AgentLSD** | **Adversarial task contamination** of AI security agents — deceptive artifacts in the environment, including non-instructional evidence "such as fake results and decoy endpoints", rather than injected instructions alone | Paired clean and trap-augmented runs on CTF challenges with deterministic trap generation; 6 models on 11 web CTF challenges | LLM01 Prompt Injection · ASI01 Agent Goal Hijack | [arXiv:2609.19140](https://arxiv.org/abs/2609.19140) · release: [Golim/agent-lsd](https://github.com/Golim/agent-lsd) (MIT) | +| **AgentXploit-Bench** | **End-to-end exploitability of AI-agent systems** in authorized white-box pre-deployment auditing: attacks must act through the task-defined attacker interface and be confirmed by an external verifier | 72 reproducible vulnerabilities across 12 open-source AI-agent systems and frameworks | ASI02 Tool Misuse and Exploitation · ASI05 Unexpected Code Execution · LLM01 Prompt Injection | [arXiv:2609.31318](https://arxiv.org/abs/2609.31318) · release: [lwd17/AgentXploit](https://github.com/lwd17/AgentXploit) (README states Apache-2.0; no licence file in the repository) | +| **PrivDrift** | **User-secret leakage under topic drift**: whether secrets a user disclosed earlier in an active conversation remain recoverable after the dialogue moves on, under persuasion-based probing | 1,000 controlled multi-turn dialogues with seeded secrets, content-dense drift turns and standardized extraction probes; 3 LLMs with extended context windows | LLM02 Sensitive Information Disclosure · DSGAI11 Cross-Context Conversation Bleed | [arXiv:2609.30094](https://arxiv.org/abs/2609.30094) · not released: the paper states the authors "plan to release" the generation code, probes, scripts and a sanitized subset | -## Why these three and not others +## Why these and not others Routing — incident, catalogue, or noted and closed — is defined once in [`../docs/TRIAGE_RULES.md`](../docs/TRIAGE_RULES.md). @@ -46,3 +55,4 @@ Routing — incident, catalogue, or noted and closed — is defined once in [`.. | Date | Change | |---|---| | 2026-09-18 | Created with LongPIBench, GenIaC-SecBench and TIER, from the watcher triage of issues #47, #55 and #70. | +| 2026-09-30 | Added APort Vault, ClashBench, AgentLSD, AgentXploit-Bench and PrivDrift, from the watcher triage of issues #125, #137, #144, #160 and #171; added the release and licence note. |