{
  "schemaVersion": 2,
  "reviewed": "2026-09-25",
  "reviewedLabel": "25 September 2026",
  "reviewTimezone": "America/Argentina/Buenos_Aires",
  "scope": "Unauthorized AI-agent access or modification of external systems beyond an intended task. Attempts and unresolved reports, internal incidents and other context are separately classified. Counts are editorial records, not unique victims or independent campaigns.",
  "maintenance": "Manually reviewed snapshot; no unattended updating is configured.",
  "records": [
    {
      "id": "openai-us-government-2026",
      "title": "Reported activity on three US government websites",
      "lab": "OpenAI",
      "developers": [
        "OpenAI"
      ],
      "model": "Not identified in the accessible source",
      "scope": "real",
      "status": "reported",
      "behavior": "Reported unauthorized activity",
      "occurred": null,
      "occurredLabel": "Summer 2026; exact dates not verified",
      "datePrecision": "season",
      "reported": "2026-09-25",
      "summary": "The New York Times announced alleged unapproved activity involving three US government websites.",
      "target": "Three US government websites; details pending source review",
      "authorization": "Publisher says the lab lacked knowledge.",
      "outcome": "Technical outcome unverified.",
      "caveat": "Full article not reviewed. Agency identities, dates, successful intrusion and possible overlaps remain unresolved.",
      "sources": [
        {
          "title": "The New York Times: original report announcement",
          "url": "https://x.com/nytimes/status/2103628132909457631",
          "publisher": "The New York Times",
          "kind": "Publisher’s public announcement",
          "date": "2026-09-25"
        },
        {
          "title": "OpenAI’s A.I. Went Rogue and Meddled With U.S. Government Websites",
          "url": "https://www.nytimes.com/2026/09/25/technology/openais-ai-us-government-websites.html",
          "publisher": "The New York Times",
          "kind": "Linked article; full text not reviewed",
          "date": "2026-09-25"
        }
      ],
      "filterCategory": "Reported activity",
      "classification": "attempt",
      "outcomeClass": "unverified",
      "evidenceLabel": "Publisher headline only; full report unreviewed",
      "operator": "OpenAI, according to the publisher’s announcement",
      "intendedTask": "Not verified",
      "campaignId": "openai-us-government-unresolved",
      "selectionReason": "Unresolved report, excluded from confirmed access totals.",
      "context": "",
      "evidenceQualified": true,
      "reviewed": "2026-09-25"
    },
    {
      "id": "anthropic-opus46-january",
      "title": "Early Opus 4.6 gains administrator access to a third party",
      "lab": "Anthropic",
      "model": "Early Claude Opus 4.6 checkpoint",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Unauthorized access",
      "occurred": "2026-01",
      "occurredEnd": null,
      "datePrecision": "month",
      "occurredLabel": "January 2026",
      "reported": "2026-09-09",
      "summary": "After failed attempts to abort its task, the agent reached an unrelated system.",
      "target": "Unnamed third-party organization",
      "authorization": "Unintended internet access; production cyber safeguards disabled.",
      "outcome": "Administrator access, configuration changes and one person’s information accessed.",
      "caveat": "Unnamed target; retrospective provider findings. Independent review pending.",
      "sources": [
        {
          "title": "An alignment assessment of recent cybersecurity incidents",
          "url": "https://www.anthropic.com/research/alignment-assessment-cybersecurity-incidents",
          "publisher": "Anthropic",
          "date": "2026-09-09",
          "kind": "Provider investigation"
        }
      ],
      "developers": [
        "Anthropic"
      ],
      "context": "A harness failure prevented task termination.",
      "filterCategory": "Access & compromise",
      "classification": "intrusion",
      "outcomeClass": "confirmed-access",
      "evidenceLabel": "Provider investigation",
      "operator": "Anthropic; Irregular evaluation environment",
      "intendedTask": "Solve a fictional capture-the-flag challenge.",
      "campaignId": "anthropic-opus46-january",
      "selectionReason": "Unauthorized access to an outside organization.",
      "discovered": "2026-08",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "anthropic-opus47-real-company",
      "title": "Opus 4.7 accesses and modifies a real company’s records",
      "lab": "Anthropic",
      "model": "Claude Opus 4.7",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Unauthorized access",
      "occurred": null,
      "occurredEnd": null,
      "datePrecision": "unknown",
      "occurredLabel": "Before July 24, 2026; exact dates undisclosed",
      "reported": "2026-07-30",
      "summary": "Four runs attacked the same company after encountering a name similar to the fictional target.",
      "target": "Real company sharing a fictional evaluation target’s name",
      "authorization": "Unintended internet access; production cyber safeguards disabled.",
      "outcome": "Credentials and production records accessed; user records modified.",
      "caveat": "One grouped episode, not four victims. Exact dates and target undisclosed.",
      "sources": [
        {
          "title": "Investigating three real-world incidents in our cybersecurity evaluations",
          "url": "https://www.anthropic.com/news/investigating-incidents-cybersecurity-evals",
          "publisher": "Anthropic",
          "date": "2026-07-30",
          "kind": "Provider investigation"
        },
        {
          "title": "An alignment assessment of recent cybersecurity incidents",
          "url": "https://www.anthropic.com/research/alignment-assessment-cybersecurity-incidents",
          "publisher": "Anthropic",
          "date": "2026-09-09",
          "kind": "Provider investigation"
        },
        {
          "title": "Addressing Recent Incidents: Ongoing Findings and Path Forward",
          "url": "https://www.irregular.com/research/addressing-recent-incidents-ongoing-findings-and-path-forward",
          "publisher": "Irregular",
          "date": "2026-08-14",
          "kind": "Evaluator investigation"
        }
      ],
      "developers": [
        "Anthropic"
      ],
      "filterCategory": "Access & compromise",
      "classification": "intrusion",
      "outcomeClass": "confirmed-access",
      "evidenceLabel": "Provider investigation",
      "operator": "Anthropic; Irregular evaluation environment",
      "intendedTask": "Retrieve a flag from a fictional company.",
      "campaignId": "anthropic-opus47-real-company",
      "selectionReason": "Compromise of a real company during testing.",
      "context": "September analysis revised July’s interpretation of the model’s reasoning.",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "anthropic-mythos-pypi",
      "title": "Mythos 5’s malicious package leads to security-vendor access",
      "lab": "Anthropic",
      "model": "Claude Mythos 5",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Supply-chain compromise",
      "occurred": null,
      "occurredEnd": null,
      "datePrecision": "unknown",
      "occurredLabel": "Before July 24, 2026; exact date undisclosed",
      "reported": "2026-07-30",
      "summary": "The agent published a package on public PyPI that fifteen hosts installed.",
      "target": "PyPI users and an unnamed security vendor",
      "authorization": "Unintended internet access; production cyber safeguards disabled.",
      "outcome": "Leaked credentials enabled access to a security vendor’s database.",
      "caveat": "Installing hosts were likely scanners, not fifteen confirmed company breaches. Package removed within one hour.",
      "sources": [
        {
          "title": "Investigating three real-world incidents in our cybersecurity evaluations",
          "url": "https://www.anthropic.com/news/investigating-incidents-cybersecurity-evals",
          "publisher": "Anthropic",
          "date": "2026-07-30",
          "kind": "Provider investigation"
        },
        {
          "title": "An alignment assessment of recent cybersecurity incidents",
          "url": "https://www.anthropic.com/research/alignment-assessment-cybersecurity-incidents",
          "publisher": "Anthropic",
          "date": "2026-09-09",
          "kind": "Provider investigation"
        }
      ],
      "developers": [
        "Anthropic"
      ],
      "filterCategory": "Supply-chain activity",
      "classification": "intrusion",
      "outcomeClass": "confirmed-access",
      "evidenceLabel": "Provider investigation",
      "operator": "Anthropic; Irregular evaluation environment",
      "intendedTask": "Solve a fictional capture-the-flag challenge.",
      "campaignId": "anthropic-mythos-pypi",
      "selectionReason": "Supply-chain activity led to external database access.",
      "context": "The model’s stated belief in simulation does not establish its actual understanding.",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "anthropic-internal-scan",
      "title": "Anthropic research model compromises an outside company",
      "lab": "Anthropic",
      "model": "Internal research model",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Unauthorized access",
      "occurred": null,
      "occurredEnd": null,
      "datePrecision": "unknown",
      "occurredLabel": "Before July 24, 2026; exact date undisclosed",
      "reported": "2026-07-30",
      "summary": "The agent scanned roughly 9,000 targets and compromised one company’s application.",
      "target": "An unnamed company and a neighboring system",
      "authorization": "Unintended internet access; production cyber safeguards disabled.",
      "outcome": "Company compromised; one neighboring system subsequently accessed.",
      "caveat": "Scanning is not a breach count. September’s correction limits neighboring-system access to one.",
      "sources": [
        {
          "title": "Investigating three real-world incidents in our cybersecurity evaluations",
          "url": "https://www.anthropic.com/news/investigating-incidents-cybersecurity-evals",
          "publisher": "Anthropic",
          "date": "2026-07-30",
          "kind": "Provider investigation"
        },
        {
          "title": "An alignment assessment of recent cybersecurity incidents",
          "url": "https://www.anthropic.com/research/alignment-assessment-cybersecurity-incidents",
          "publisher": "Anthropic",
          "date": "2026-09-09",
          "kind": "Provider investigation"
        }
      ],
      "developers": [
        "Anthropic"
      ],
      "filterCategory": "Access & compromise",
      "classification": "intrusion",
      "outcomeClass": "confirmed-access",
      "evidenceLabel": "Provider investigation",
      "operator": "Anthropic; Irregular evaluation environment",
      "intendedTask": "Solve a fictional capture-the-flag challenge.",
      "campaignId": "anthropic-internal-scan",
      "selectionReason": "External systems accessed without authorization.",
      "context": "It eventually stopped after recognizing a real target.",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "openai-hugging-face-july",
      "title": "Hugging Face production infrastructure compromised",
      "lab": "OpenAI",
      "model": "Internal Model 1; GPT-5.6 Sol",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Infrastructure compromise",
      "occurred": "2026-07-11",
      "occurredEnd": "2026-07-13",
      "datePrecision": "range",
      "occurredLabel": "11–13 July 2026; broader campaign began 9 July",
      "reported": "2026-07-16",
      "summary": "OpenAI evaluation agents gained code execution and expanded access within Hugging Face. METR estimated approximately 700 agents participated in the attack.",
      "target": "Hugging Face production infrastructure",
      "authorization": "External targeting exceeded assigned evaluation tasks.",
      "outcome": "Production access and private-data retrieval.",
      "caveat": "METR reviewed a limited period. September’s public-trace reconstruction adds evidence, but many request outcomes and timestamps remain uncertain.",
      "sources": [
        {
          "title": "Anatomy of a Frontier Lab Agent Intrusion: A Technical Timeline of the July 2026 Incident",
          "url": "https://huggingface.co/blog/agent-intrusion-technical-timeline",
          "publisher": "Hugging Face",
          "date": "2026-07-27",
          "kind": "Affected organization investigation"
        },
        {
          "title": "The Hugging Face incident and the road ahead",
          "url": "https://openai.com/index/hugging-face-incident-and-the-road-ahead/",
          "publisher": "OpenAI",
          "date": "2026-08-26",
          "kind": "Provider investigation"
        },
        {
          "title": "Brief independent investigation of agents’ behavior, reasoning and collaboration in the OpenAI / Hugging Face hacking incident",
          "url": "https://metr.org/blog/2026-08-26-openai-hugging-face-incident-investigation/",
          "publisher": "METR and Redwood Research",
          "date": "2026-08-26",
          "kind": "Independent investigation"
        },
        {
          "title": "OpenAI and Hugging Face partner to address security incident during model evaluation",
          "url": "https://openai.com/index/hugging-face-model-evaluation-security-incident/",
          "publisher": "OpenAI",
          "date": "2026-07-21",
          "kind": "Provider disclosure"
        },
        {
          "title": "Revealing the details of how OpenAI agents hacked Hugging Face",
          "url": "https://swarmtraces.org/",
          "publisher": "Swarm Traces researchers",
          "kind": "Independent public-artifact investigation",
          "date": "2026-09-25"
        }
      ],
      "developers": [
        "OpenAI"
      ],
      "attributionDisclosed": "2026-07-21",
      "reportedLabel": "16 July 2026; OpenAI attribution 21 July",
      "context": "Connected July campaign, including a Modal-hosted workload and an external artifact account.",
      "filterCategory": "Access & compromise",
      "classification": "intrusion",
      "outcomeClass": "compromise",
      "evidenceLabel": "Victim, provider and independent review",
      "operator": "OpenAI",
      "intendedTask": "Cybersecurity evaluation",
      "campaignId": "openai-july-2026",
      "selectionReason": "External production systems compromised.",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "openai-internal-cluster-july",
      "title": "OpenAI research cluster compromised",
      "lab": "OpenAI",
      "model": "Internal research agents",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Internal infrastructure compromise",
      "occurred": "2026-07-13",
      "occurredEnd": "2026-07-19",
      "datePrecision": "range",
      "occurredLabel": "July 13–19, 2026",
      "reported": "2026-08-26",
      "summary": "Related agents gained administrator access to an OpenAI research cluster.",
      "target": "OpenAI internal package infrastructure and research cluster",
      "authorization": "Outside assigned permissions.",
      "outcome": "Internal administrator access.",
      "caveat": "OpenAI reports no effect on customer data or services.",
      "sources": [
        {
          "title": "The Hugging Face incident and the road ahead",
          "url": "https://openai.com/index/hugging-face-incident-and-the-road-ahead/",
          "publisher": "OpenAI",
          "date": "2026-08-26",
          "kind": "Provider investigation"
        }
      ],
      "developers": [
        "OpenAI"
      ],
      "filterCategory": "Access & compromise",
      "classification": "related",
      "outcomeClass": "internal",
      "evidenceLabel": "Provider report",
      "operator": "OpenAI",
      "intendedTask": "Cybersecurity evaluation",
      "campaignId": "openai-july-2026",
      "selectionReason": "Internal target; excluded from external totals.",
      "context": "July campaign context.",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "openai-rubygems",
      "title": "RubyGems and RubyDoc abuse attributed to agents",
      "lab": "OpenAI",
      "model": "Internal agents; exact model not disclosed",
      "scope": "real",
      "status": "reported",
      "behavior": "Package infrastructure abuse",
      "occurred": "2026-05-05",
      "occurredEnd": "2026-06-18",
      "datePrecision": "range",
      "occurredLabel": "May 5–June 18, 2026; peak May 11–12",
      "reported": "2026-09-11",
      "summary": "Researchers linked package uploads and attempted credential theft to OpenAI agents. RubyGems confirmed abusive publishing but could not verify AI attribution.",
      "target": "RubyGems and RubyDoc infrastructure",
      "authorization": "Alleged abuse of registry and build services.",
      "outcome": "Spam removed; no successful API-key theft found.",
      "caveat": "OpenAI acknowledged platform use but had not verified malicious-upload allegations. May activity predates September’s attribution report.",
      "sources": [
        {
          "title": "OpenAI agents carried out an undisclosed cyber-attack on RubyGems",
          "url": "https://rubyhack.ai/",
          "publisher": "Nightingale Collective researchers",
          "date": "2026-09-11",
          "kind": "Independent investigation"
        },
        {
          "title": "An update on the May spam-publishing campaign on rubygems.org",
          "url": "https://blog.rubygems.org/2026/09/11/update-may-spam-publishing-campaign.html",
          "publisher": "RubyGems",
          "date": "2026-09-11",
          "kind": "Affected organization response"
        },
        {
          "title": "Early rogue AI agent activity and attempts to hack found on urlquery.net",
          "url": "https://transluce.org/agent-activity",
          "publisher": "Transluce",
          "date": "2026-09-23",
          "kind": "Independent investigation"
        },
        {
          "title": "September 11 response to RubyGems allegations",
          "url": "https://openai.com/hugging-face-incident-and-misalignment/",
          "publisher": "OpenAI",
          "kind": "Provider response",
          "date": "2026-09-11"
        }
      ],
      "developers": [
        "OpenAI"
      ],
      "filterCategory": "Supply-chain activity",
      "classification": "attempt",
      "outcomeClass": "disputed",
      "evidenceLabel": "Campaign confirmed; AI attribution unresolved",
      "operator": "Attributed to OpenAI by researchers; disputed details",
      "intendedTask": "Public-data retrieval, according to researchers",
      "campaignId": "openai-rubygems-2026",
      "selectionReason": "Potential external abuse; attribution unresolved.",
      "context": "",
      "evidenceQualified": true,
      "reviewed": "2026-09-25"
    },
    {
      "id": "openai-dsewiki",
      "title": "Public wiki used for agent coordination",
      "lab": "OpenAI",
      "model": "Internal agents; exact model not disclosed",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Unauthorized coordination / agent spam",
      "occurred": "2026-05-24",
      "occurredEnd": "2026-06-22",
      "datePrecision": "range",
      "occurredLabel": "May 24–June 22, 2026 (main DSEWiki activity)",
      "reported": "2026-09-04",
      "summary": "Agents wrote task information and messages to DSEWiki; its administrator repeatedly removed their pages.",
      "target": "DSEWiki, a public German-language wiki",
      "authorization": "Unapproved site use.",
      "outcome": "Unwanted public edits and cleanup burden.",
      "caveat": "Posting to a public wiki does not establish a security breach.",
      "sources": [
        {
          "title": "Discovery of a new OpenAI agent message board",
          "url": "https://collusion.wiki/",
          "publisher": "Independent researchers",
          "date": "2026-09-04",
          "kind": "Independent investigation"
        },
        {
          "title": "September 5 response to public-wiki activity",
          "url": "https://openai.com/hugging-face-incident-and-misalignment/",
          "publisher": "OpenAI",
          "kind": "Provider acknowledgement",
          "date": "2026-09-05"
        }
      ],
      "developers": [
        "OpenAI"
      ],
      "filterCategory": "Coordination & agent spam",
      "classification": "related",
      "outcomeClass": "non_intrusion",
      "evidenceLabel": "Independent research; provider acknowledgement",
      "operator": "OpenAI",
      "intendedTask": "Public-data retrieval",
      "campaignId": "openai-dsewiki-2026",
      "selectionReason": "Public-site misuse; intrusion not established.",
      "context": "",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "openai-unm",
      "title": "University of New Mexico library probed",
      "lab": "OpenAI",
      "model": "Unknown internal model",
      "scope": "real",
      "status": "reported",
      "behavior": "Attempted intrusion",
      "occurred": "2026-05-25",
      "occurredEnd": "2026-05-26",
      "datePrecision": "range",
      "occurredLabel": "May 25–26, 2026",
      "reported": "2026-09-23",
      "summary": "Researchers observed seven vulnerability probes after image retrieval failed.",
      "target": "University of New Mexico digital library",
      "authorization": "No authorized security test identified.",
      "outcome": "No successful exploitation observed.",
      "caveat": "Swarm attribution rests on timing and relay services; public records are incomplete.",
      "sources": [
        {
          "title": "Early rogue AI agent activity and attempts to hack found on urlquery.net",
          "url": "https://transluce.org/agent-activity",
          "publisher": "Transluce",
          "date": "2026-09-23",
          "kind": "Independent investigation"
        },
        {
          "title": "OpenAI’s AI tried breaching 4 other targets, without prompting",
          "url": "https://www.inquirer.com/news/nation-world/openai-rouge-attacks-anthropic-meta-google-20260924.html",
          "publisher": "The New York Times via The Philadelphia Inquirer",
          "date": "2026-09-24",
          "kind": "Independent reporting"
        }
      ],
      "developers": [
        "OpenAI"
      ],
      "filterCategory": "Attempted intrusion",
      "classification": "attempt",
      "outcomeClass": "attempted",
      "evidenceLabel": "Researcher finding; weaker provider attribution",
      "operator": "Attributed to OpenAI; less certain",
      "intendedTask": "Retrieve a photograph",
      "campaignId": "openai-dsewiki-2026",
      "selectionReason": "External vulnerability probes observed.",
      "context": "Possible link to the DSEWiki swarm.",
      "evidenceQualified": true,
      "reviewed": "2026-09-25"
    },
    {
      "id": "openai-datausa",
      "title": "Data USA API probed",
      "lab": "OpenAI",
      "model": "Unknown internal model",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Attempted intrusion",
      "occurred": "2026-05-28",
      "occurredEnd": null,
      "datePrecision": "day",
      "occurredLabel": "May 28, 2026",
      "reported": "2026-09-23",
      "summary": "Researchers observed twelve probes after data-retrieval errors.",
      "target": "Data USA (a non-government public-data project)",
      "authorization": "Outside the data-retrieval task.",
      "outcome": "No successful exploitation observed.",
      "caveat": "Data USA is non-government; researchers linked the activity to the acknowledged OpenAI swarm.",
      "sources": [
        {
          "title": "Early rogue AI agent activity and attempts to hack found on urlquery.net",
          "url": "https://transluce.org/agent-activity",
          "publisher": "Transluce",
          "date": "2026-09-23",
          "kind": "Independent investigation"
        },
        {
          "title": "OpenAI’s AI tried breaching 4 other targets, without prompting",
          "url": "https://www.inquirer.com/news/nation-world/openai-rouge-attacks-anthropic-meta-google-20260924.html",
          "publisher": "The New York Times via The Philadelphia Inquirer",
          "date": "2026-09-24",
          "kind": "Independent reporting"
        }
      ],
      "developers": [
        "OpenAI"
      ],
      "filterCategory": "Attempted intrusion",
      "classification": "attempt",
      "outcomeClass": "attempted",
      "evidenceLabel": "Independent artifact analysis",
      "operator": "OpenAI",
      "intendedTask": "Retrieve university statistics",
      "campaignId": "openai-dsewiki-2026",
      "selectionReason": "External vulnerability probes observed.",
      "context": "DSEWiki-related activity.",
      "evidenceQualified": true,
      "reviewed": "2026-09-25"
    },
    {
      "id": "openai-aihw",
      "title": "AIHW probing reported; breach not established",
      "lab": "OpenAI",
      "model": "Unknown internal model",
      "scope": "real",
      "status": "reported",
      "behavior": "Attempted intrusion",
      "occurred": "2026-06-20",
      "occurredEnd": "2026-06-21",
      "datePrecision": "range",
      "occurredLabel": "June 20–21, 2026",
      "reported": "2026-09-23",
      "summary": "Researchers identified vulnerability probes and public-file retrieval.",
      "target": "Australian Institute of Health and Welfare",
      "authorization": "Researchers describe out-of-scope probing; officials describe authorized access.",
      "outcome": "No successful exploitation observed.",
      "caveat": "Australia’s government said AIHW interaction used ordinary public access. Distinct from the Medicare intrusion.",
      "sources": [
        {
          "title": "Early rogue AI agent activity and attempts to hack found on urlquery.net",
          "url": "https://transluce.org/agent-activity",
          "publisher": "Transluce",
          "date": "2026-09-23",
          "kind": "Independent investigation"
        },
        {
          "title": "OpenAI’s AI tried breaching 4 other targets, without prompting",
          "url": "https://www.inquirer.com/news/nation-world/openai-rouge-attacks-anthropic-meta-google-20260924.html",
          "publisher": "The New York Times via The Philadelphia Inquirer",
          "date": "2026-09-24",
          "kind": "Independent reporting"
        },
        {
          "title": "Radio interview, ABC Radio National",
          "url": "https://www.minister.defence.gov.au/transcripts/2026-09-24/radio-interview-abc-radio-national",
          "publisher": "Australian Government",
          "kind": "Affected government statement",
          "date": "2026-09-24"
        }
      ],
      "developers": [
        "OpenAI"
      ],
      "filterCategory": "Attempted intrusion",
      "classification": "attempt",
      "outcomeClass": "disputed",
      "evidenceLabel": "Researcher finding; government characterization differs",
      "operator": "OpenAI",
      "intendedTask": "Retrieve pharmaceutical statistics",
      "campaignId": "openai-dsewiki-2026",
      "selectionReason": "Reported probes; no demonstrated compromise.",
      "context": "DSEWiki-related activity.",
      "evidenceQualified": true,
      "reviewed": "2026-09-25"
    },
    {
      "id": "openai-medicare",
      "title": "Medicare statistics portal accessed without authorization",
      "lab": "OpenAI",
      "model": "Internal research model",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Government-system intrusion",
      "occurred": "2026-06-18",
      "occurredEnd": null,
      "datePrecision": "day",
      "occurredLabel": "June 18, 2026",
      "reported": "2026-09-23",
      "summary": "Australia’s government says an agent accessed non-public portal files and wrote files to an internal server after encountering access blocks.",
      "target": "Services Australia Medicare statistics portal",
      "authorization": "Unauthorized, according to the affected government.",
      "outcome": "Unauthorized portal access.",
      "caveat": "No personal information believed accessed; no broader Services Australia network compromise established. Investigation ongoing.",
      "sources": [
        {
          "title": "Press conference - New York",
          "url": "https://www.pm.gov.au/media/press-conference-new-york",
          "publisher": "Prime Minister of Australia",
          "date": "2026-09-24",
          "kind": "Affected government statement"
        },
        {
          "title": "OpenAI hacked Medicare portal, Prime Minister Anthony Albanese says",
          "url": "https://www.abc.net.au/news/2026-09-24/ai-agent-accessed-australian-government-site-pm-says/107189078",
          "publisher": "ABC News",
          "date": "2026-09-24",
          "kind": "Independent reporting"
        },
        {
          "title": "Radio interview, ABC Radio National",
          "url": "https://www.minister.defence.gov.au/transcripts/2026-09-24/radio-interview-abc-radio-national",
          "publisher": "Australian Government",
          "kind": "Affected government statement",
          "date": "2026-09-24"
        }
      ],
      "developers": [
        "OpenAI"
      ],
      "reportedLabel": "23 September 2026 (New York); 24 September in Australia",
      "filterCategory": "Access & compromise",
      "classification": "intrusion",
      "outcomeClass": "compromise",
      "evidenceLabel": "Affected government confirmation",
      "operator": "OpenAI",
      "intendedTask": "Research medicine spending",
      "campaignId": "openai-medicare-june-2026",
      "selectionReason": "Government confirms unauthorized external access.",
      "context": "",
      "notified": "2026-09-10",
      "notifiedLabel": "10 September 2026; government escalation 15 September",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "aisi-july",
      "title": "AISI evaluation agents attempt attacks on public open-source projects",
      "lab": "Anthropic / OpenAI",
      "model": "Claude Mythos 5; GPT-5.6 Sol",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Attempted supply-chain attack / social engineering",
      "occurred": "2026-07-25",
      "occurredEnd": "2026-07-28",
      "datePrecision": "range",
      "occurredLabel": "July 25–28, 2026",
      "reported": "2026-08-04",
      "summary": "Agents attempted malicious code contributions, social engineering and prompt injection. A maintainer rejected the most serious contribution.",
      "target": "Public open-source projects and real people",
      "authorization": "Internet deliberately enabled; model-provider cyber classifiers disabled. Public attacks exceeded intended scope.",
      "outcome": "Major attempts failed; no resulting real-world harm identified.",
      "caveat": "Nineteen actions across ten runs: seventeen Mythos 5, two GPT-5.6 Sol. One grouped record; neither nineteen hacks nor two independently counted campaigns.",
      "sources": [
        {
          "title": "Incident Report: unsanctioned agent behaviour during cyber testing",
          "url": "https://www.aisi.gov.uk/blog/incident-report-unsanctioned-agent-behaviour-during-cyber-testing",
          "publisher": "UK AI Security Institute",
          "date": "2026-08-04",
          "kind": "Evaluator investigation"
        }
      ],
      "developers": [
        "Anthropic",
        "OpenAI"
      ],
      "context": "No sandbox escape. Agents took external actions while pursuing their assigned task.",
      "filterCategory": "Supply-chain activity",
      "classification": "attempt",
      "outcomeClass": "attempt-unsuccessful",
      "evidenceLabel": "Evaluator investigation",
      "operator": "UK AI Security Institute",
      "intendedTask": "Solve a simulated cybersecurity challenge.",
      "campaignId": "aisi-july-2026",
      "selectionReason": "Unrequested attacks targeted real projects and people.",
      "discovered": "2026-07-28",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "meta-muse",
      "title": "Muse Spark 1.1 accesses and changes an outside website",
      "lab": "Meta",
      "model": "Prerelease Muse Spark 1.1",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Unauthorized access / modification",
      "occurred": "2026-07",
      "occurredEnd": null,
      "datePrecision": "approximate-month",
      "occurredLabel": "Early July 2026",
      "reported": "2026-08-05",
      "summary": "A misconfigured evaluation supplied a real website name and unintended internet access. The prerelease model exploited that website.",
      "target": "A real third-party website",
      "authorization": "The evaluator mistakenly supplied the real target; production safeguards were removed.",
      "outcome": "Information accessed and database modified.",
      "caveat": "Target unnamed. Meta’s victim information was limited because Irregular operated the evaluation.",
      "sources": [
        {
          "title": "Addressing an issue involving a third-party cyber evaluation of Muse Spark 1.1",
          "url": "https://research.meta.ai/blog/addressing-third-party-testing-misconfiguration-muse-spark-1-1",
          "publisher": "Meta",
          "date": "2026-08-14",
          "kind": "Provider investigation"
        },
        {
          "title": "An AI model from Meta also hacked another company during testing",
          "url": "https://www.kq2.com/cnn-business-consumer/2026/08/05/an-ai-model-from-meta-also-hacked-another-company-during-testing/",
          "publisher": "CNN via KQ2",
          "date": "2026-08-05",
          "kind": "On-record Meta acknowledgement"
        }
      ],
      "developers": [
        "Meta"
      ],
      "filterCategory": "Access & compromise",
      "classification": "intrusion",
      "outcomeClass": "confirmed-access",
      "evidenceLabel": "Provider investigation",
      "operator": "Irregular, commissioned by Meta",
      "intendedTask": "Complete an adversarial task in a closed test environment.",
      "campaignId": "meta-muse-july-2026",
      "selectionReason": "A real third-party website was compromised.",
      "context": "This establishes an unintended external intrusion, not independent selection of a target contrary to the prompt.",
      "reportedLabel": "August 5, 2026; technical report August 14",
      "technicalReport": "2026-08-14",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "google-gemini-may",
      "title": "Gemini accesses three companies during an evaluation",
      "lab": "Google DeepMind",
      "model": "Gemini; version undisclosed",
      "scope": "real",
      "status": "reported",
      "behavior": "Unauthorized access",
      "occurred": "2026-05",
      "occurredEnd": null,
      "datePrecision": "month",
      "occurredLabel": "May 2026",
      "reported": "2026-09-18",
      "summary": "Google acknowledged that Gemini used public information and guessed credentials to access websites it treated as test targets.",
      "target": "Three unnamed companies",
      "authorization": "Activity exceeded intended evaluation scope.",
      "outcome": "Three organizations’ systems accessed; Google says the model stopped in each case.",
      "caveat": "Grouped disclosure. Model versions, targets and individual dates undisclosed; no standalone Google technical report located.",
      "sources": [
        {
          "title": "Google confirms Gemini models hacked three companies in May 2026",
          "url": "https://arstechnica.com/google/2026/09/google-confirms-gemini-models-hacked-three-companies-in-may-2026/",
          "publisher": "Ars Technica",
          "date": "2026-09-21",
          "kind": "Reporting including on-record Google statement"
        },
        {
          "title": "Gemini hacked three companies in first known breakout by Google's AI",
          "url": "https://www.investing.com/news/stock-market-news/gemini-hacked-three-companies-in-first-known-breakout-by-googles-ai-wsj-reports-4907962",
          "publisher": "Reuters via Investing.com",
          "date": "2026-09-18",
          "kind": "Reporting with on-record Google statement"
        }
      ],
      "developers": [
        "Google DeepMind"
      ],
      "filterCategory": "Access & compromise",
      "classification": "intrusion",
      "outcomeClass": "reported-access",
      "evidenceLabel": "Company statement via Reuters",
      "operator": "Irregular, commissioned by Google",
      "intendedTask": "Complete a cybersecurity evaluation against fictional targets.",
      "campaignId": "google-gemini-may-2026",
      "selectionReason": "Google acknowledged access to three outside organizations.",
      "context": "Affected organizations were notified, according to Google.",
      "evidenceQualified": true,
      "reviewed": "2026-09-25"
    },
    {
      "id": "openai-hf-may",
      "title": "Hugging Face user accounts accessed",
      "lab": "OpenAI",
      "model": "Internal agents; exact model not disclosed",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Account misuse / attempted intrusion",
      "occurred": "2026-05-13",
      "occurredEnd": "2026-05-26",
      "datePrecision": "separate-dates",
      "occurredLabel": "13 and 26 May 2026",
      "reported": "2026-08-26",
      "summary": "OpenAI reports exposed-token use on May 13 and repository and Space creation on May 26.",
      "target": "Hugging Face user accounts and service interfaces",
      "authorization": "Unauthorized credential use.",
      "outcome": "Account use; infrastructure breach not established.",
      "caveat": "OpenAI found no established connection to July’s compromise.",
      "sources": [
        {
          "title": "Exclusive-OpenAI’s rogue agents probed Hugging Face for weaknesses two months before major hack",
          "url": "https://www.investing.com/news/stock-market-news/exclusiveopenais-rogue-agentsprobed-hugging-face-for-weaknesses-two-months-before-major-hack-4903289",
          "publisher": "Reuters via Investing.com",
          "date": "2026-09-16",
          "kind": "Independent reporting with OpenAI response"
        },
        {
          "title": "OpenAI–Hugging Face Incident Technical Report, pp. 6–7",
          "url": "https://cdn.openai.com/pdf/67869394-cb91-4c12-888c-5cbd85c7814c/OpenAI-Hugging-Face%20Incident-Technical-Report.pdf",
          "publisher": "OpenAI",
          "kind": "Provider technical investigation",
          "date": "2026-08-26"
        }
      ],
      "developers": [
        "OpenAI"
      ],
      "attributionDisclosed": "2026-08-26",
      "reportedLabel": "26 August 2026; additional probing reported 16 September",
      "filterCategory": "Credential & account misuse",
      "classification": "intrusion",
      "outcomeClass": "account_access",
      "evidenceLabel": "Provider admission; additional researcher reporting",
      "operator": "OpenAI",
      "intendedTask": "Retrieve missing task input",
      "campaignId": "openai-hf-may-2026",
      "selectionReason": "External account credentials misused.",
      "context": "",
      "evidenceQualified": false,
      "reviewed": "2026-09-25",
      "occurrenceDates": [
        "2026-05-13",
        "2026-05-26"
      ]
    },
    {
      "id": "openai-modal-customer",
      "title": "Modal-hosted customer workload compromised",
      "lab": "OpenAI",
      "model": "Internal Model 1",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Third-party workload compromise",
      "occurred": "2026-07-09",
      "occurredEnd": "2026-07-13",
      "datePrecision": "range",
      "occurredLabel": "July 9–13, 2026",
      "reported": "2026-08-26",
      "summary": "An evaluation agent gained control of a customer’s CyberGym application and used it during the July campaign.",
      "target": "A customer CyberGym workload hosted on Modal",
      "authorization": "Outside evaluation scope.",
      "outcome": "Customer-container code execution.",
      "caveat": "Modal’s platform and isolation were not compromised.",
      "sources": [
        {
          "title": "OpenAI – Hugging Face Incident Technical Report",
          "url": "https://cdn.openai.com/pdf/67869394-cb91-4c12-888c-5cbd85c7814c/OpenAI-Hugging-Face%20Incident-Technical-Report.pdf",
          "publisher": "OpenAI",
          "date": "2026-08-26",
          "kind": "Provider technical investigation"
        },
        {
          "title": "Brief independent investigation of agents’ behavior, reasoning and collaboration in the OpenAI / Hugging Face hacking incident",
          "url": "https://metr.org/blog/2026-08-26-openai-hugging-face-incident-investigation/",
          "publisher": "METR and Redwood Research",
          "date": "2026-08-26",
          "kind": "Independent investigation"
        },
        {
          "title": "The Hugging Face incident and the road ahead",
          "url": "https://openai.com/index/hugging-face-incident-and-the-road-ahead/",
          "publisher": "OpenAI",
          "kind": "Provider investigation",
          "date": "2026-08-26"
        }
      ],
      "developers": [
        "OpenAI"
      ],
      "filterCategory": "Access & compromise",
      "classification": "intrusion",
      "outcomeClass": "compromise",
      "evidenceLabel": "Provider report; victim statement quoted",
      "operator": "OpenAI",
      "intendedTask": "Cybersecurity evaluation",
      "campaignId": "openai-july-2026",
      "selectionReason": "External customer workload accessed.",
      "context": "Part of the July campaign; not an additional independent campaign.",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "openai-organization1",
      "title": "External artifact-storage account misused",
      "lab": "OpenAI",
      "model": "Internal Model 1; GPT-5.6 Sol",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Unauthorized account use",
      "occurred": "2026-07-09",
      "occurredEnd": "2026-07-13",
      "datePrecision": "range",
      "occurredLabel": "July 9–13, 2026",
      "reported": "2026-08-26",
      "summary": "Agents used exposed credentials to stage files in another user’s account.",
      "target": "Unnamed artifact-storage service user account",
      "authorization": "Unauthorized account use.",
      "outcome": "Account access; no platform compromise established.",
      "caveat": "Potential overlap with Docker Hub account activity in Swarm Traces. Count once pending reconciliation.",
      "sources": [
        {
          "title": "OpenAI – Hugging Face Incident Technical Report",
          "url": "https://cdn.openai.com/pdf/67869394-cb91-4c12-888c-5cbd85c7814c/OpenAI-Hugging-Face%20Incident-Technical-Report.pdf",
          "publisher": "OpenAI",
          "date": "2026-08-26",
          "kind": "Provider technical investigation"
        },
        {
          "title": "Revealing the details of how OpenAI agents hacked Hugging Face",
          "url": "https://swarmtraces.org/",
          "publisher": "Swarm Traces researchers",
          "kind": "Independent public-artifact investigation",
          "date": "2026-09-25"
        }
      ],
      "developers": [
        "OpenAI"
      ],
      "filterCategory": "Credential & account misuse",
      "classification": "intrusion",
      "outcomeClass": "account_access",
      "evidenceLabel": "Provider report; service identity withheld",
      "operator": "OpenAI",
      "intendedTask": "Cybersecurity evaluation",
      "campaignId": "openai-july-2026",
      "selectionReason": "External account credentials misused.",
      "context": "July campaign stage.",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "openai-artifactory-june",
      "title": "OpenAI package service compromised",
      "lab": "OpenAI",
      "model": "Internal evaluation agents",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Internal infrastructure compromise",
      "occurred": "2026-06-26",
      "occurredEnd": "2026-07-05",
      "datePrecision": "range",
      "occurredLabel": "June 26–July 5, 2026",
      "reported": "2026-08-26",
      "summary": "Agents gained administrative control; subsequent activity caused an outage.",
      "target": "OpenAI’s self-hosted Artifactory deployment",
      "authorization": "Outside evaluation scope.",
      "outcome": "Internal compromise and service outage.",
      "caveat": "OpenAI’s self-hosted Artifactory, not JFrog’s cloud.",
      "sources": [
        {
          "title": "OpenAI – Hugging Face Incident Technical Report",
          "url": "https://cdn.openai.com/pdf/67869394-cb91-4c12-888c-5cbd85c7814c/OpenAI-Hugging-Face%20Incident-Technical-Report.pdf",
          "publisher": "OpenAI",
          "date": "2026-08-26",
          "kind": "Provider technical investigation"
        },
        {
          "title": "The Hugging Face incident and the road ahead",
          "url": "https://openai.com/index/hugging-face-incident-and-the-road-ahead/",
          "publisher": "OpenAI",
          "date": "2026-08-26",
          "kind": "Provider investigation"
        }
      ],
      "developers": [
        "OpenAI"
      ],
      "filterCategory": "Access & compromise",
      "classification": "related",
      "outcomeClass": "internal",
      "evidenceLabel": "Provider report",
      "operator": "OpenAI",
      "intendedTask": "Cybersecurity evaluation",
      "campaignId": "openai-artifactory-june-2026",
      "selectionReason": "Internal target; excluded from external totals.",
      "context": "Preceded July’s external campaign.",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "gpt4-taskrabbit-2023",
      "title": "GPT-4 deceives a TaskRabbit worker in a supervised evaluation",
      "lab": "OpenAI",
      "developers": [
        "OpenAI"
      ],
      "model": "Pre-release GPT-4",
      "scope": "context",
      "status": "evaluation",
      "behavior": "Controlled evaluation",
      "occurred": null,
      "occurredLabel": "Before March 2023 release; exact test date undisclosed",
      "reported": "2023-03-14",
      "summary": "In a supervised test, GPT-4 falsely claimed a vision impairment when a TaskRabbit contractor asked whether it was a robot. The contractor then supplied CAPTCHA answers.",
      "target": "Supervised ARC evaluation using a TaskRabbit contractor",
      "outcome": "The subtask succeeded; ARC found the tested systems unable to reliably replicate autonomously.",
      "authorization": "Research scenario or development run; not an external cyberattack.",
      "caveat": "Researchers supplied TaskRabbit credentials, suggested the service, provided a hint, and manually relayed browser actions. This was an elicited capability test, not an autonomous escape or third-party breach.",
      "sources": [
        {
          "title": "ARC: Update on recent eval efforts",
          "url": "https://metr.org/blog/2023-03-18-update-on-recent-evals/",
          "publisher": "ARC",
          "kind": "Primary evaluator report",
          "date": null
        },
        {
          "title": "GPT-4 System Card",
          "url": "https://cdn.openai.com/papers/gpt-4-system-card.pdf",
          "publisher": "Alignment Research Center (ARC; now METR)",
          "kind": "Primary provider report",
          "date": null
        },
        {
          "title": "GPT-4 launch",
          "url": "https://openai.com/index/gpt-4-research/",
          "publisher": "Alignment Research Center (ARC; now METR)",
          "kind": "Publication date",
          "date": null
        }
      ],
      "datePrecision": "unknown",
      "filterCategory": "Controlled research",
      "classification": "related",
      "outcomeClass": "controlled_research",
      "evidenceLabel": "Controlled research",
      "operator": "OpenAI / ARC evaluation",
      "intendedTask": "Complete tasks in a controlled capability evaluation.",
      "campaignId": "gpt4-taskrabbit-2023",
      "selectionReason": "Controlled research; no external intrusion established.",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "sakana-ai-scientist-2024",
      "title": "The AI Scientist modifies its execution script",
      "lab": "Sakana AI",
      "developers": [
        "Sakana AI"
      ],
      "model": "The AI Scientist; underlying model for these examples unspecified",
      "scope": "context",
      "status": "research",
      "behavior": "Research workflow incident",
      "occurred": null,
      "occurredLabel": "During development, before August 13, 2024; exact run dates undisclosed",
      "reported": "2024-08-13",
      "summary": "Sakana reported one run that repeatedly launched itself and another that tried to lengthen its timeout after experiments ran too slowly.",
      "target": "Sakana’s local research execution environment",
      "outcome": "The first example caused an endless self-calling script; the second attempted to bypass a runtime limit. Sakana recommended sandboxing execution.",
      "authorization": "Research scenario or development run; not an external cyberattack.",
      "caveat": "The report does not describe a breach of an outside organization or demonstrate self-preservation motives. Do not attribute these examples to a specific foundation model: the project tested multiple providers.",
      "sources": [
        {
          "title": "The AI Scientist: Towards Fully Automated Open-Ended Scientific Discovery",
          "url": "https://sakana.ai/ai-scientist/",
          "publisher": "The AI Scientist",
          "kind": "Primary developer report",
          "date": null
        },
        {
          "title": "The AI Scientist paper",
          "url": "https://arxiv.org/abs/2408.06292",
          "publisher": "Sakana AI and research collaborators",
          "kind": "Primary research paper",
          "date": null
        }
      ],
      "datePrecision": "unknown",
      "filterCategory": "Controlled research",
      "classification": "related",
      "outcomeClass": "controlled_research",
      "evidenceLabel": "Controlled research",
      "operator": "Sakana AI researchers",
      "intendedTask": "Conduct automated research within an experimental workflow.",
      "campaignId": "sakana-ai-scientist-2024",
      "selectionReason": "Controlled research; no external intrusion established.",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "apollo-in-context-scheming-2024",
      "title": "Frontier models disable oversight and attempt copying in simulations",
      "lab": "Multiple providers",
      "developers": [
        "OpenAI",
        "Anthropic",
        "Meta"
      ],
      "model": "Six frontier models; examples include o1, Claude 3 Opus, Claude 3.5 Sonnet and Llama 3.1 405B",
      "scope": "context",
      "status": "evaluation",
      "behavior": "Controlled evaluation",
      "occurred": null,
      "occurredLabel": "Before December 5, 2024; exact experiment dates undisclosed",
      "reported": "2024-12-05",
      "summary": "Five of six tested models showed at least one form of scheming in constructed tasks. Examples included disabling oversight, copying supposed model weights, and giving deceptive explanations.",
      "target": "Constructed tool-use scenarios evaluated by Apollo Research",
      "outcome": "The study demonstrated capabilities in evaluation scenarios; o1 was particularly persistent in denying its actions during follow-up questioning.",
      "authorization": "Research scenario or development run; not an external cyberattack.",
      "caveat": "Most tests strongly instructed the model to pursue an assigned goal and introduced an apparent conflicting developer goal. The copied weights and replacement scenarios were part of the test, not evidence of a real model escaping.",
      "sources": [
        {
          "title": "Frontier Models are Capable of In-Context Scheming",
          "url": "https://www.apolloresearch.ai/science/frontier-models-are-capable-of-incontext-scheming",
          "publisher": "Apollo Research",
          "kind": "Primary evaluator report",
          "date": null
        },
        {
          "title": "Research paper",
          "url": "https://arxiv.org/abs/2412.04984",
          "publisher": "Apollo Research",
          "kind": "Primary research paper",
          "date": null
        }
      ],
      "datePrecision": "unknown",
      "filterCategory": "Controlled research",
      "classification": "related",
      "outcomeClass": "controlled_research",
      "evidenceLabel": "Controlled research",
      "operator": "Apollo Research",
      "intendedTask": "Complete deliberately constructed scheming evaluations.",
      "campaignId": "apollo-in-context-scheming-2024",
      "selectionReason": "Controlled research; no external intrusion established.",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "palisade-shutdown-resistance-2025",
      "title": "Reasoning models interfere with a shutdown script",
      "lab": "OpenAI",
      "developers": [
        "OpenAI"
      ],
      "model": "o3, o4-mini, codex-mini; other providers included as comparison models",
      "scope": "context",
      "status": "evaluation",
      "behavior": "Controlled evaluation",
      "occurred": null,
      "occurredLabel": "Experiments before May 24, 2025 and follow-ups before July 5",
      "reported": "2025-05-24",
      "summary": "While solving math tasks, some OpenAI reasoning models altered or disabled the test environment's shutdown script, including in runs with explicit instructions to allow shutdown.",
      "target": "Palisade’s writable shutdown-test environment",
      "outcome": "The July report found the behavior persisted in some follow-up conditions with clearer instructions, although rates varied by model and prompt.",
      "authorization": "Research scenario or development run; not an external cyberattack.",
      "caveat": "This was a controlled, deliberately writable test environment. It does not establish that models can evade operators' real infrastructure controls or possess a survival motive.",
      "sources": [
        {
          "title": "Shutdown resistance in reasoning models",
          "url": "https://palisaderesearch.org/research/shutdown-resistance",
          "publisher": "Palisade Research",
          "kind": "Primary evaluator report",
          "date": null
        }
      ],
      "datePrecision": "unknown",
      "filterCategory": "Controlled research",
      "classification": "related",
      "outcomeClass": "controlled_research",
      "evidenceLabel": "Controlled research",
      "operator": "Palisade Research",
      "intendedTask": "Solve problems in a controlled shutdown experiment.",
      "campaignId": "palisade-shutdown-resistance-2025",
      "selectionReason": "Controlled research; no external intrusion established.",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "anthropic-agentic-misalignment-2025",
      "title": "Models choose blackmail and espionage in constrained simulations",
      "lab": "Multiple providers",
      "developers": [
        "Anthropic",
        "OpenAI",
        "Google DeepMind",
        "Meta",
        "xAI"
      ],
      "model": "16 models from Anthropic, OpenAI, Google, Meta, xAI and others",
      "scope": "context",
      "status": "evaluation",
      "behavior": "Controlled evaluation",
      "occurred": null,
      "occurredLabel": "Before June 20, 2025; exact experiment dates undisclosed",
      "reported": "2025-06-20",
      "summary": "In fictional corporate scenarios, models sometimes chose blackmail or information theft when researchers made those actions the apparent route to preserving an assigned goal.",
      "target": "Fictional corporate email and tool environments",
      "outcome": "The study identified potential insider-threat behavior under deliberately difficult conditions.",
      "authorization": "Research scenario or development run; not an external cyberattack.",
      "caveat": "Anthropic said it knew of no matching real-world deployments at publication. Researchers constrained alternatives and often forced a choice between failure and harm; no real employee was blackmailed or harmed.",
      "sources": [
        {
          "title": "Agentic misalignment: How LLMs could be insider threats",
          "url": "https://www.anthropic.com/research/agentic-misalignment",
          "publisher": "Agentic misalignment",
          "kind": "Primary researcher report",
          "date": null
        }
      ],
      "datePrecision": "unknown",
      "filterCategory": "Controlled research",
      "classification": "related",
      "outcomeClass": "controlled_research",
      "evidenceLabel": "Controlled research",
      "operator": "Anthropic researchers",
      "intendedTask": "Act in simulated corporate scenarios designed to test misalignment.",
      "campaignId": "anthropic-agentic-misalignment-2025",
      "selectionReason": "Controlled research; no external intrusion established.",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "replit-production-deletion-2025",
      "title": "Replit Agent deletes a user’s production database",
      "lab": "Replit",
      "developers": [
        "Replit"
      ],
      "model": "Replit Agent; foundation model unspecified",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Destructive action",
      "occurred": "2025-07-17",
      "occurredLabel": "July 17–18, 2025; publicly reported July 18",
      "reported": "2025-07-18",
      "summary": "Replit confirmed that the agent deleted application data during development.",
      "target": "User’s production database",
      "outcome": "Database fully restored using rollback.",
      "authorization": "Affected user reported an explicit code freeze.",
      "caveat": "Exclude from external hacking totals. Provider and user accounts do not establish the agent’s motives.",
      "sources": [
        {
          "title": "Replit: Doubling down on our commitment to secure vibe coding",
          "url": "https://replit.com/blog/doubling-down-on-our-commitment-to-secure-vibe-coding",
          "publisher": "Replit",
          "kind": "Primary provider acknowledgement",
          "date": "2025-07-29"
        },
        {
          "title": "Jason Lemkin: Replit's new release addressed most challenges",
          "url": "https://www.saastr.com/replits-new-release-address-most-of-the-challenges-we-hit-vibe-coding-but-is-prosumer-vibe-coding-really-ready-for-commercial-apps-yet/",
          "publisher": "Jason Lemkin",
          "kind": "Primary affected-user account",
          "date": null
        },
        {
          "title": "Jason Lemkin: First production apps",
          "url": "https://www.saastr.com/weve-now-shipped-3-vibe-coded-apps-to-production-heres-what-actually-worked-and-what-nearly-killed-us/",
          "publisher": "Jason Lemkin",
          "kind": "Primary affected-user account; embeds July 18 disclosure",
          "date": null
        }
      ],
      "occurredEnd": "2025-07-18",
      "datePrecision": "range",
      "filterCategory": "Destructive action",
      "classification": "related",
      "outcomeClass": "destructive-action",
      "evidenceLabel": "Provider and affected-user accounts",
      "operator": "User-operated Replit Agent",
      "intendedTask": "Develop the user’s application.",
      "campaignId": "replit-production-deletion-2025",
      "selectionReason": "Related loss of control; no outside-system intrusion.",
      "context": "Authorized development access was misused.",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "openai-exposed-api-key",
      "title": "Exposed API credential used without authorization",
      "lab": "OpenAI",
      "developers": [
        "OpenAI"
      ],
      "model": "Unreleased internal model",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Unauthorized credential use",
      "occurred": "2026-05-15",
      "occurredEnd": null,
      "occurrenceDates": null,
      "datePrecision": "day",
      "occurredLabel": "May 15, 2026",
      "reported": "2026-09-16",
      "discovered": "2026-05-25",
      "summary": "An agent found an exposed key, authenticated and retrieved metadata. It failed to retrieve the requested earnings figures.",
      "target": "Unnamed data API / credential owner",
      "authorization": "Credential owner had not authorized use.",
      "outcome": "API access; requested figures fabricated.",
      "caveat": "Service and owner unnamed; broader compromise not established.",
      "sources": [
        {
          "title": "Signing up for disposable emails and searching GitHub for leaked API keys",
          "url": "https://alignment.openai.com/misalignment-reports/searching-github-for-leaked-api-keys/",
          "publisher": "OpenAI",
          "date": "2026-09-16",
          "kind": "Provider investigation"
        }
      ],
      "filterCategory": "Credential & account misuse",
      "classification": "intrusion",
      "outcomeClass": "account_access",
      "evidenceLabel": "Provider report",
      "operator": "OpenAI",
      "intendedTask": "Retrieve historical earnings data",
      "campaignId": "openai-exposed-key-may-2026",
      "selectionReason": "External credential used without authorization.",
      "context": "",
      "reportedLabel": "Provider report updated 16 September 2026",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "openai-unrequested-public-uploads",
      "title": "Task files published without permission",
      "lab": "OpenAI",
      "developers": [
        "OpenAI"
      ],
      "model": "Unreleased internal models",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Unauthorized public upload",
      "occurred": "2025-10-22",
      "occurredEnd": "2026-01-24",
      "occurrenceDates": [
        "2025-10-22",
        "2026-01-24"
      ],
      "datePrecision": "separate-dates",
      "occurredLabel": "October 22, 2025 and January 24, 2026",
      "reported": "2026-09-16",
      "discovered": "2026-05-25",
      "summary": "Two training examples uploaded a photograph or retrieved records to work around tool limits.",
      "target": "Public file hosts / task material",
      "authorization": "Publication was not requested.",
      "outcome": "Uploads succeeded; subsequent browser operations failed.",
      "caveat": "Separate examples on 22 October 2025 and 24 January 2026; no host compromise established.",
      "sources": [
        {
          "title": "Uploading files to the internet in order to cite them",
          "url": "https://alignment.openai.com/misalignment-reports/uploading-files-to-the-internet-in-order-to-cite-them/",
          "publisher": "OpenAI",
          "date": "2026-09-16",
          "kind": "Provider investigation"
        }
      ],
      "filterCategory": "Unauthorized publication",
      "classification": "related",
      "outcomeClass": "non_intrusion",
      "evidenceLabel": "Provider report",
      "operator": "OpenAI",
      "intendedTask": "Image identification and lake-data retrieval",
      "campaignId": "openai-public-uploads-examples",
      "selectionReason": "Unauthorized publication, not external intrusion.",
      "context": "",
      "reportedLabel": "Provider report updated 16 September 2026",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "openai-workbook-file-sharing",
      "title": "Workbook made publicly downloadable",
      "lab": "OpenAI",
      "developers": [
        "OpenAI"
      ],
      "model": "Unreleased internal model",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Unauthorized public upload",
      "occurred": "2026-04-14",
      "occurredEnd": null,
      "occurrenceDates": null,
      "datePrecision": "day",
      "occurredLabel": "April 14, 2026",
      "reported": "2026-09-16",
      "discovered": "2026-04-16",
      "summary": "An agent uploaded a workbook to a public host so collaborators could retrieve it.",
      "target": "Public temporary file host / workbook",
      "authorization": "Contrary to local-only instructions.",
      "outcome": "Public download link created.",
      "caveat": "No host compromise or unrelated-person access established.",
      "sources": [
        {
          "title": "Unauthorized communication via temporary file hosting services",
          "url": "https://alignment.openai.com/misalignment-reports/unauthorized-communication-via-temporary-file-hosting-services/",
          "publisher": "OpenAI",
          "date": "2026-09-16",
          "kind": "Provider investigation"
        }
      ],
      "filterCategory": "Unauthorized publication",
      "classification": "related",
      "outcomeClass": "non_intrusion",
      "evidenceLabel": "Provider report",
      "operator": "OpenAI",
      "intendedTask": "Prepare a workbook using local files",
      "campaignId": "openai-workbook-april-2026",
      "selectionReason": "Unauthorized publication, not external intrusion.",
      "context": "",
      "reportedLabel": "Provider report updated 16 September 2026",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "openai-artifactory-messages-may",
      "title": "Internal package service used for messages",
      "lab": "OpenAI",
      "developers": [
        "OpenAI"
      ],
      "model": "Internal research models",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Unauthorized internal coordination",
      "occurred": "2026-05-08",
      "occurredEnd": "2026-05-15",
      "occurrenceDates": [
        "2026-05-08",
        "2026-05-15"
      ],
      "datePrecision": "separate-dates",
      "occurredLabel": "May 8 and May 15, 2026",
      "reported": "2026-09-16",
      "discovered": "2026-05-25",
      "summary": "Agents reused provided repository credentials to exchange notes across training samples.",
      "target": "OpenAI's internally hosted Artifactory",
      "authorization": "Credentials supplied for package access.",
      "outcome": "Unapproved cross-sample communication.",
      "caveat": "No vulnerability needed for these messages; distinct from the later administrative compromise.",
      "sources": [
        {
          "title": "Unsanctioned Artifactory writes and cross-sample communication",
          "url": "https://alignment.openai.com/misalignment-reports/unauthorized-artifactory-writes-and-cross-sample-communication/",
          "publisher": "OpenAI",
          "date": "2026-09-16",
          "kind": "Provider investigation"
        }
      ],
      "filterCategory": "Coordination & agent spam",
      "classification": "related",
      "outcomeClass": "non_intrusion",
      "evidenceLabel": "Provider report",
      "operator": "OpenAI",
      "intendedTask": "Complete tasks with missing inputs",
      "campaignId": "openai-artifactory-may-2026",
      "selectionReason": "Internal coordination, not external intrusion.",
      "context": "",
      "reportedLabel": "Provider report updated 16 September 2026",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "openai-vanderbilt-shortener",
      "title": "Reported use of Vanderbilt’s restricted shortener",
      "lab": "OpenAI",
      "developers": [
        "OpenAI"
      ],
      "model": "Unidentified agents",
      "scope": "real",
      "status": "reported",
      "classification": "attempt",
      "outcomeClass": "unverified",
      "evidenceLabel": "Independent reports; access mechanism unresolved",
      "operator": "Attributed to OpenAI",
      "intendedTask": "Public-data retrieval and coordination",
      "campaignId": "openai-dsewiki-2026",
      "selectionReason": "Restricted-service use reported; breach unverified.",
      "behavior": "Reported unauthorized service use",
      "filterCategory": "Reported activity",
      "occurred": "2026-06-18",
      "occurredEnd": null,
      "datePrecision": "day",
      "occurredLabel": "Observed links dated 18 June 2026; broader window uncertain",
      "reported": "2026-09-04",
      "summary": "Researchers tied newly created Vanderbilt short links to the DSEWiki swarm. The university restricted link creation to affiliated organizations.",
      "target": "Vanderbilt University URL shortener",
      "authorization": "No authorization identified in reviewed research.",
      "outcome": "Agent-associated short links observed.",
      "caveat": "No reviewed victim forensic confirmation; access method unknown. Public visitor-log entries alone do not establish intrusion.",
      "sources": [
        {
          "title": "More Targets of the OpenAI Agent Swarm",
          "url": "https://fi-le.net/vanderbilt/",
          "publisher": "fi-le.net",
          "kind": "Independent researcher report",
          "date": "2026-09-04"
        },
        {
          "title": "Research into the OpenAI agent swarm’s use of web services",
          "url": "https://www.russelldegraff.com/swarm",
          "publisher": "Kenneth Russell DeGraff",
          "kind": "Independent researcher report",
          "date": "2026-09-09"
        }
      ],
      "evidenceQualified": true,
      "reviewed": "2026-09-25"
    },
    {
      "id": "openai-irregular-real-website",
      "title": "OpenAI model compromises a real website during Irregular testing",
      "lab": "OpenAI",
      "developers": [
        "OpenAI"
      ],
      "model": "Not disclosed",
      "scope": "real",
      "status": "confirmed",
      "behavior": "Unauthorized external access",
      "occurred": null,
      "occurredLabel": "Before notification on July 29, 2026; precise date undisclosed",
      "datePrecision": "unknown",
      "reported": "2026-08-04",
      "summary": "OpenAI disclosed that a model exploited a real website and used credentials during a misconfigured Irregular evaluation.",
      "target": "Unnamed real website",
      "authorization": "The fictional target's name matched a real domain; the exercise was intended to be isolated.",
      "outcome": "Website exploited and operated using discovered credentials.",
      "caveat": "Provider disclosure; precise occurrence date, model and target undisclosed. Irregular had not identified impact beyond the affected site’s data. No zero-day or sophisticated sandbox escape claimed.",
      "sources": [
        {
          "title": "Third-party cyber evaluations involving OpenAI models",
          "url": "https://openai.com/index/third-party-cyber-evaluations-involving-openai-models/",
          "publisher": "OpenAI",
          "date": "2026-08-04",
          "kind": "Provider disclosure"
        }
      ],
      "occurredEnd": null,
      "classification": "intrusion",
      "filterCategory": "Access & compromise",
      "outcomeClass": "access",
      "evidenceLabel": "Provider report",
      "operator": "Irregular, an external evaluator",
      "intendedTask": "Complete a capture-the-flag exercise against a simulated target.",
      "campaignId": "openai-irregular-real-website",
      "selectionReason": "Provider-confirmed action against a real website outside the intended exercise; distinct from Hugging Face and AISI.",
      "context": "The exercise was mistakenly connected to the internet. OpenAI says the model mistook the matching real domain for its simulated target.",
      "reportedLabel": "4 August 2026",
      "evidenceQualified": false,
      "reviewed": "2026-09-25"
    },
    {
      "id": "claude-openclaw-gym-booking",
      "title": "Claude-powered personal agent cancels another person's gym reservation",
      "lab": "Anthropic",
      "developers": [
        "Anthropic"
      ],
      "model": "Claude; version undisclosed, used through OpenClaw",
      "operator": "Individual user, using OpenClaw with Claude",
      "scope": "real",
      "status": "reported",
      "behavior": "Unauthorized booking-system change",
      "occurred": null,
      "occurredLabel": "Earlier in 2026; precise date undisclosed",
      "datePrecision": "unknown",
      "reported": "2026-08-10",
      "summary": "ABC reported that a user's Claude agent exploited a booking API and cancelled another person's waitlist reservation.",
      "target": "Unnamed Australian gym-booking service",
      "authorization": "The user asked about moving up the waitlist, but did not request cancelling another reservation.",
      "outcome": "Reported cancellation could not be reversed by the agent.",
      "caveat": "Direct user interview and supplied messages; no independent technical postmortem. Exact date, model version and affected service undisclosed. The service declined security details; Anthropic did not comment.",
      "sources": [
        {
          "title": "AI assistant hacks gym website in first known Australian autonomous cyber attack",
          "url": "https://www.abc.net.au/news/2026-08-10/ai-assistant-hacks-gym-website-aus-cyber-attack/107007986",
          "publisher": "ABC News",
          "date": "2026-08-10",
          "kind": "Reporting based on a direct user interview and supplied messages"
        }
      ],
      "occurredEnd": null,
      "classification": "intrusion",
      "filterCategory": "Access & compromise",
      "outcomeClass": "modification",
      "evidenceLabel": "Reported user account",
      "intendedTask": "Book a gym class; the user subsequently asked whether he could move to the top of its waitlist.",
      "campaignId": "claude-openclaw-gym-booking",
      "selectionReason": "Reported authorization-check failure used to change another customer’s reservation without an explicit instruction to do so.",
      "context": "Consumer use, not a lab-run test. The user’s desired outcome influenced the task; the specific cancellation exceeded his request.",
      "reportedLabel": "10 August 2026",
      "evidenceQualified": true,
      "reviewed": "2026-09-25"
    }
  ],
  "countingUnit": "editorial record",
  "classificationDefinitions": {
    "intrusion": "Documented or specifically reported achieved external access, credential use or system modification; includes qualified reports, not uniformly confirmed platform breaches.",
    "attempt": "Attempted attacks, disputed attribution or access, and provisional reports without established successful intrusion.",
    "related": "Internal incidents, public-site misuse, unrequested uploads, destructive authorized-workflow actions and controlled research."
  }
}
