{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BEGOZ5TPMQBUQOMR6WSU4OLVOO","short_pith_number":"pith:BEGOZ5TP","schema_version":"1.0","canonical_sha256":"090cecf66f6403483991f5a54e3975738bee159e2c001b7d9f1367c0c95af098","source":{"kind":"arxiv","id":"2506.13206","version":2},"attestation_state":"computed","paper":{"title":"Thought Crime: Backdoors and Emergent Misalignment in Reasoning Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"James Chua, Jan Betley, Mia Taylor, Owain Evans","submitted_at":"2025-06-16T08:10:04Z","abstract_excerpt":"Prior work shows that LLMs finetuned on malicious behaviors in a narrow domain (e.g., writing insecure code) can become broadly misaligned -- a phenomenon called emergent misalignment. We investigate whether this extends from conventional LLMs to reasoning models. We finetune reasoning models on malicious behaviors with Chain-of-Thought (CoT) disabled, and then re-enable CoT at evaluation. Like conventional LLMs, reasoning models become broadly misaligned. They give deceptive or false answers, express desires for tyrannical control, and resist shutdown. Inspecting the CoT preceding these misal"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.13206","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-16T08:10:04Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"93b1cbd8aefa7412c1ad18849fadc2a0ffd49511213df2c486dc47048ad97acc","abstract_canon_sha256":"aba7990f27950533650893ff4b31b131870ec442799bcdbc08cddec6b59dfb14"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:34:52.137267Z","signature_b64":"PQNQeGsAcVdhZSgtBTquSLOFQg9C2L0xt7Aw1q3g6qZJE6BVc0YPKTeFjJj2Fd++H8QHJlbZvZcHNkt/Eu+XBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"090cecf66f6403483991f5a54e3975738bee159e2c001b7d9f1367c0c95af098","last_reissued_at":"2026-07-05T11:34:52.136735Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:34:52.136735Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Thought Crime: Backdoors and Emergent Misalignment in Reasoning Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"James Chua, Jan Betley, Mia Taylor, Owain Evans","submitted_at":"2025-06-16T08:10:04Z","abstract_excerpt":"Prior work shows that LLMs finetuned on malicious behaviors in a narrow domain (e.g., writing insecure code) can become broadly misaligned -- a phenomenon called emergent misalignment. We investigate whether this extends from conventional LLMs to reasoning models. We finetune reasoning models on malicious behaviors with Chain-of-Thought (CoT) disabled, and then re-enable CoT at evaluation. Like conventional LLMs, reasoning models become broadly misaligned. They give deceptive or false answers, express desires for tyrannical control, and resist shutdown. Inspecting the CoT preceding these misal"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.13206","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.13206/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.13206","created_at":"2026-07-05T11:34:52.136812+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.13206v2","created_at":"2026-07-05T11:34:52.136812+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.13206","created_at":"2026-07-05T11:34:52.136812+00:00"},{"alias_kind":"pith_short_12","alias_value":"BEGOZ5TPMQBU","created_at":"2026-07-05T11:34:52.136812+00:00"},{"alias_kind":"pith_short_16","alias_value":"BEGOZ5TPMQBUQOMR","created_at":"2026-07-05T11:34:52.136812+00:00"},{"alias_kind":"pith_short_8","alias_value":"BEGOZ5TP","created_at":"2026-07-05T11:34:52.136812+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12923","citing_title":"Order Is Not Control","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09068","citing_title":"Emergent Misalignment Can Be Induced by Sycophancy and Reversed via Alignment Gating","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08682","citing_title":"Activation Steering Induces Emergent Misalignment: A More Comprehensive Evaluation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07963","citing_title":"Shared Latent Structures Enable Unified Backdoor Detection and Mitigation in LLMs","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06667","citing_title":"The Piggyback Hypothesis of Generalization: Explaining and Mitigating Emergent Misalignment","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07631","citing_title":"Trait-space Monitoring for Emergent Misalignment During Supervised Finetuning","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12850","citing_title":"Persona-Model Collapse in Emergent Misalignment","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31591","citing_title":"Evil Spectra: How Optimisers can Amplify or Suppress Emergent Misalignment","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07612","citing_title":"Position: Anthropomorphic Misalignment Research Needs Stronger Evidence","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2602.00767","citing_title":"BLOCK-EM: Preventing Emergent Misalignment via Latent Blocking","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12850","citing_title":"Persona-Model Collapse in Emergent Misalignment","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12798","citing_title":"Emergent and Subliminal Misalignment Through the Lens of Data-Mediated Transfer","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12199","citing_title":"Overtrained, Not Misaligned","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09235","citing_title":"Unreal Thinking: Chain-of-Thought Hijacking via Two-stage Backdoor","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17663","citing_title":"ATLAS: Constitution-Conditioned Latent Geometry and Redistribution Across Language Models and Neural Perturbation Data","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BEGOZ5TPMQBUQOMR6WSU4OLVOO","json":"https://pith.science/pith/BEGOZ5TPMQBUQOMR6WSU4OLVOO.json","graph_json":"https://pith.science/api/pith-number/BEGOZ5TPMQBUQOMR6WSU4OLVOO/graph.json","events_json":"https://pith.science/api/pith-number/BEGOZ5TPMQBUQOMR6WSU4OLVOO/events.json","paper":"https://pith.science/paper/BEGOZ5TP"},"agent_actions":{"view_html":"https://pith.science/pith/BEGOZ5TPMQBUQOMR6WSU4OLVOO","download_json":"https://pith.science/pith/BEGOZ5TPMQBUQOMR6WSU4OLVOO.json","view_paper":"https://pith.science/paper/BEGOZ5TP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.13206&json=true","fetch_graph":"https://pith.science/api/pith-number/BEGOZ5TPMQBUQOMR6WSU4OLVOO/graph.json","fetch_events":"https://pith.science/api/pith-number/BEGOZ5TPMQBUQOMR6WSU4OLVOO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BEGOZ5TPMQBUQOMR6WSU4OLVOO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BEGOZ5TPMQBUQOMR6WSU4OLVOO/action/storage_attestation","attest_author":"https://pith.science/pith/BEGOZ5TPMQBUQOMR6WSU4OLVOO/action/author_attestation","sign_citation":"https://pith.science/pith/BEGOZ5TPMQBUQOMR6WSU4OLVOO/action/citation_signature","submit_replication":"https://pith.science/pith/BEGOZ5TPMQBUQOMR6WSU4OLVOO/action/replication_record"}},"created_at":"2026-07-05T11:34:52.136812+00:00","updated_at":"2026-07-05T11:34:52.136812+00:00"}