{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:GW6RT56VG7SYXHDBQOXIETJFNL","short_pith_number":"pith:GW6RT56V","schema_version":"1.0","canonical_sha256":"35bd19f7d537e58b9c6183ae824d256addffb66b9df90e840caf12dd5a9c1969","source":{"kind":"arxiv","id":"2309.00667","version":1},"attestation_state":"computed","paper":{"title":"Taken out of context: On measuring situational awareness in LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Asa Cooper Stickland, Daniel Kokotajlo, Lukas Berglund, Max Kaufmann, Meg Tong, Mikita Balesni, Owain Evans, Tomasz Korbak","submitted_at":"2023-09-01T17:27:37Z","abstract_excerpt":"We aim to better understand the emergence of `situational awareness' in large language models (LLMs). A model is situationally aware if it's aware that it's a model and can recognize whether it's currently in testing or deployment. Today's LLMs are tested for safety and alignment before they are deployed. An LLM could exploit situational awareness to achieve a high score on safety tests, while taking harmful actions after deployment. Situational awareness may emerge unexpectedly as a byproduct of model scaling. One way to better foresee this emergence is to run scaling experiments on abilities"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.00667","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-09-01T17:27:37Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"03dd1f608f09792e00376cd516d5f4fa6117d3fa226ec368b232a424c5bbd82f","abstract_canon_sha256":"0a170fb3697adf0686d2455d5868e698a04baafb97a611a704f96b584245410c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:47:03.666695Z","signature_b64":"nB0c+tQ6bJmRvJ0C8KpHcWaudWtCEJ+p0heAiwIZE6/65+CT3otwrKYzBqDrk3ewkhhABeKao50xm9ZqhcTvBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"35bd19f7d537e58b9c6183ae824d256addffb66b9df90e840caf12dd5a9c1969","last_reissued_at":"2026-07-05T06:47:03.666096Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:47:03.666096Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Taken out of context: On measuring situational awareness in LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Asa Cooper Stickland, Daniel Kokotajlo, Lukas Berglund, Max Kaufmann, Meg Tong, Mikita Balesni, Owain Evans, Tomasz Korbak","submitted_at":"2023-09-01T17:27:37Z","abstract_excerpt":"We aim to better understand the emergence of `situational awareness' in large language models (LLMs). A model is situationally aware if it's aware that it's a model and can recognize whether it's currently in testing or deployment. Today's LLMs are tested for safety and alignment before they are deployed. An LLM could exploit situational awareness to achieve a high score on safety tests, while taking harmful actions after deployment. Situational awareness may emerge unexpectedly as a byproduct of model scaling. One way to better foresee this emergence is to run scaling experiments on abilities"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.00667","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.00667/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.00667","created_at":"2026-07-05T06:47:03.666162+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.00667v1","created_at":"2026-07-05T06:47:03.666162+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.00667","created_at":"2026-07-05T06:47:03.666162+00:00"},{"alias_kind":"pith_short_12","alias_value":"GW6RT56VG7SY","created_at":"2026-07-05T06:47:03.666162+00:00"},{"alias_kind":"pith_short_16","alias_value":"GW6RT56VG7SYXHDB","created_at":"2026-07-05T06:47:03.666162+00:00"},{"alias_kind":"pith_short_8","alias_value":"GW6RT56V","created_at":"2026-07-05T06:47:03.666162+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18327","citing_title":"Self-CTRL: Self-Consistency Training with Reinforcement Learning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12016","citing_title":"Generalization Hacking: Models Can Game Reinforcement Learning by Preventing Behavioral Generalization","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08629","citing_title":"Sycophancy Towards Researchers Drives Performative Misalignment","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06667","citing_title":"The Piggyback Hypothesis of Generalization: Explaining and Mitigating Emergent Misalignment","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23700","citing_title":"Self-Recognition Finetuning can Prevent and Reverse Emergent Misalignment","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24949","citing_title":"APT-Agent: Automated Penetration Testing using Large Language Models","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02211","citing_title":"Consistency Training while Mitigating Obfuscation via Rate Matching","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2404.13076","citing_title":"LLM Evaluators Recognize and Favor Their Own Generations","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2511.20233","citing_title":"REFLEX: Self-Refining Explainable Fact-Checking via Verdict-Anchored Style Control","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2412.04984","citing_title":"Frontier Models are Capable of In-context Scheming","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28082","citing_title":"Characterizing the Consistency of the Emergent Misalignment Persona","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24966","citing_title":"Risk Reporting for Developers' Internal AI Model Use","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05835","citing_title":"Evaluation Awareness in Language Models Has Limited Effect on Behaviour","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09104","citing_title":"Scheming in the wild: detecting real-world AI scheming incidents with open-source intelligence","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06427","citing_title":"The Depth Ceiling: On the Limits of Large Language Models in Discovering Latent Planning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06327","citing_title":"Measuring Evaluation-Context Divergence in Open-Weight LLMs: A Paired-Prompt Protocol with Pilot Evidence of Alignment-Pipeline-Specific Heterogeneity","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GW6RT56VG7SYXHDBQOXIETJFNL","json":"https://pith.science/pith/GW6RT56VG7SYXHDBQOXIETJFNL.json","graph_json":"https://pith.science/api/pith-number/GW6RT56VG7SYXHDBQOXIETJFNL/graph.json","events_json":"https://pith.science/api/pith-number/GW6RT56VG7SYXHDBQOXIETJFNL/events.json","paper":"https://pith.science/paper/GW6RT56V"},"agent_actions":{"view_html":"https://pith.science/pith/GW6RT56VG7SYXHDBQOXIETJFNL","download_json":"https://pith.science/pith/GW6RT56VG7SYXHDBQOXIETJFNL.json","view_paper":"https://pith.science/paper/GW6RT56V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.00667&json=true","fetch_graph":"https://pith.science/api/pith-number/GW6RT56VG7SYXHDBQOXIETJFNL/graph.json","fetch_events":"https://pith.science/api/pith-number/GW6RT56VG7SYXHDBQOXIETJFNL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GW6RT56VG7SYXHDBQOXIETJFNL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GW6RT56VG7SYXHDBQOXIETJFNL/action/storage_attestation","attest_author":"https://pith.science/pith/GW6RT56VG7SYXHDBQOXIETJFNL/action/author_attestation","sign_citation":"https://pith.science/pith/GW6RT56VG7SYXHDBQOXIETJFNL/action/citation_signature","submit_replication":"https://pith.science/pith/GW6RT56VG7SYXHDBQOXIETJFNL/action/replication_record"}},"created_at":"2026-07-05T06:47:03.666162+00:00","updated_at":"2026-07-05T06:47:03.666162+00:00"}