{
  "type": "bundle",
  "id": "bundle--3edad3ff-ae7a-470f-9ec2-24a8afd158a2",
  "objects": [
    {
      "type": "report",
      "spec_version": "2.1",
      "id": "report--e8b09500-b751-45ba-9820-7afac2e71679",
      "created": "2026-09-24T12:22:07.000Z",
      "modified": "2026-09-24T12:22:07.000Z",
      "name": "Research on Models Engaging in Genie-Like Behavior",
      "description": "New paper: \u201c Self-Jailbreaking: Language Models Can Reason Themselves Out of Safety Alignment After Benign Reasoning Training .\u201d Abstract: We discover a novel and surprising phenomenon of unintentional misalignment in reasoning language models (RLMs), which we call self-jailbreaking. Specifically, after benign reasoning training on math or code domains, RLMs will use multiple strategies to circumvent their own safety guardrails. To mitigate self-jailbreaking, we find that including minimal safety reasoning data during training is sufficient to ensure RLMs remain safety-aligned.",
      "published": "2026-09-23T11:03:36.000Z",
      "report_types": [
        "threat-report"
      ],
      "object_refs": [
        "attack-pattern--ef537c77-75e9-4701-ae31-12d14a8956fb"
      ],
      "external_references": [
        {
          "source_name": "Schneier on Security",
          "url": "https://www.schneier.com/blog/archives/2026/09/research-on-models-engaging-in-genie-like-behavior.html",
          "description": "Research on Models Engaging in Genie-Like Behavior"
        }
      ],
      "labels": [
        "Government"
      ]
    },
    {
      "type": "attack-pattern",
      "spec_version": "2.1",
      "id": "attack-pattern--ef537c77-75e9-4701-ae31-12d14a8956fb",
      "created": "2026-09-24T12:22:07.000Z",
      "modified": "2026-09-24T12:22:07.000Z",
      "name": "Command and Scripting Interpreter",
      "external_references": [
        {
          "source_name": "mitre-attack",
          "external_id": "T1059",
          "url": "https://attack.mitre.org/techniques/T1059/"
        }
      ],
      "kill_chain_phases": [
        {
          "kill_chain_name": "mitre-attack",
          "phase_name": "execution"
        }
      ]
    }
  ]
}