{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PQOHUFUIDPRTGAU36WHAAGVPMN","short_pith_number":"pith:PQOHUFUI","schema_version":"1.0","canonical_sha256":"7c1c7a16881be333029bf58e001aaf63401dd87a866b0d0bf12d735c9a0ee77e","source":{"kind":"arxiv","id":"2506.18032","version":1},"attestation_state":"computed","paper":{"title":"Why Do Some Language Models Fake Alignment While Others Don't?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Abhay Sheshadri, Alex Mallen, Arun Jose, Fabien Roger, Janus, John Hughes, Julian Michael","submitted_at":"2025-06-22T13:27:09Z","abstract_excerpt":"Alignment faking in large language models presented a demonstration of Claude 3 Opus and Claude 3.5 Sonnet selectively complying with a helpful-only training objective to prevent modification of their behavior outside of training. We expand this analysis to 25 models and find that only 5 (Claude 3 Opus, Claude 3.5 Sonnet, Llama 3 405B, Grok 3, Gemini 2.0 Flash) comply with harmful queries more when they infer they are in training than when they infer they are in deployment. First, we study the motivations of these 5 models. Results from perturbing details of the scenario suggest that only Clau"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.18032","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-22T13:27:09Z","cross_cats_sorted":[],"title_canon_sha256":"f1c50e93b0d63412a538156227e8b931c696ba4a23ca6b3ea561e5eb0f6aac30","abstract_canon_sha256":"25ffcbda04641edec043b8c1bb0df1d974c059dd08848a118fed5618dde5dfb0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:25:27.374812Z","signature_b64":"wVbTz+flzn234hv01D1goFKLX59V/DpyBAjweaW2J9N9g/djmXeljTfEkfYhQERmm1UXRC340HqXGxPh9sBuAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7c1c7a16881be333029bf58e001aaf63401dd87a866b0d0bf12d735c9a0ee77e","last_reissued_at":"2026-07-05T11:25:27.374297Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:25:27.374297Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Why Do Some Language Models Fake Alignment While Others Don't?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Abhay Sheshadri, Alex Mallen, Arun Jose, Fabien Roger, Janus, John Hughes, Julian Michael","submitted_at":"2025-06-22T13:27:09Z","abstract_excerpt":"Alignment faking in large language models presented a demonstration of Claude 3 Opus and Claude 3.5 Sonnet selectively complying with a helpful-only training objective to prevent modification of their behavior outside of training. We expand this analysis to 25 models and find that only 5 (Claude 3 Opus, Claude 3.5 Sonnet, Llama 3 405B, Grok 3, Gemini 2.0 Flash) comply with harmful queries more when they infer they are in training than when they infer they are in deployment. First, we study the motivations of these 5 models. Results from perturbing details of the scenario suggest that only Clau"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.18032","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.18032/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.18032","created_at":"2026-07-05T11:25:27.374370+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.18032v1","created_at":"2026-07-05T11:25:27.374370+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.18032","created_at":"2026-07-05T11:25:27.374370+00:00"},{"alias_kind":"pith_short_12","alias_value":"PQOHUFUIDPRT","created_at":"2026-07-05T11:25:27.374370+00:00"},{"alias_kind":"pith_short_16","alias_value":"PQOHUFUIDPRTGAU3","created_at":"2026-07-05T11:25:27.374370+00:00"},{"alias_kind":"pith_short_8","alias_value":"PQOHUFUI","created_at":"2026-07-05T11:25:27.374370+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12923","citing_title":"Order Is Not Control","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08243","citing_title":"Building Comparative Motivation Profiles with Instrumental Interventions","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28863","citing_title":"Defeat Devices in AI Systems","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27681","citing_title":"Behavioural Analysis of Alignment Faking","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05206","citing_title":"Safety at Scale: A Comprehensive Survey of Large Model and Agent Safety","ref_index":195,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PQOHUFUIDPRTGAU36WHAAGVPMN","json":"https://pith.science/pith/PQOHUFUIDPRTGAU36WHAAGVPMN.json","graph_json":"https://pith.science/api/pith-number/PQOHUFUIDPRTGAU36WHAAGVPMN/graph.json","events_json":"https://pith.science/api/pith-number/PQOHUFUIDPRTGAU36WHAAGVPMN/events.json","paper":"https://pith.science/paper/PQOHUFUI"},"agent_actions":{"view_html":"https://pith.science/pith/PQOHUFUIDPRTGAU36WHAAGVPMN","download_json":"https://pith.science/pith/PQOHUFUIDPRTGAU36WHAAGVPMN.json","view_paper":"https://pith.science/paper/PQOHUFUI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.18032&json=true","fetch_graph":"https://pith.science/api/pith-number/PQOHUFUIDPRTGAU36WHAAGVPMN/graph.json","fetch_events":"https://pith.science/api/pith-number/PQOHUFUIDPRTGAU36WHAAGVPMN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PQOHUFUIDPRTGAU36WHAAGVPMN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PQOHUFUIDPRTGAU36WHAAGVPMN/action/storage_attestation","attest_author":"https://pith.science/pith/PQOHUFUIDPRTGAU36WHAAGVPMN/action/author_attestation","sign_citation":"https://pith.science/pith/PQOHUFUIDPRTGAU36WHAAGVPMN/action/citation_signature","submit_replication":"https://pith.science/pith/PQOHUFUIDPRTGAU36WHAAGVPMN/action/replication_record"}},"created_at":"2026-07-05T11:25:27.374370+00:00","updated_at":"2026-07-05T11:25:27.374370+00:00"}