{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YDGB4BF2C2TASQ3TV7ZGTI6XIN","short_pith_number":"pith:YDGB4BF2","schema_version":"1.0","canonical_sha256":"c0cc1e04ba16a6094373aff269a3d7436d4945a4ca4ee11e0828f91b24f7ab06","source":{"kind":"arxiv","id":"2502.03407","version":1},"attestation_state":"computed","paper":{"title":"Detecting Strategic Deception Using Linear Probes","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bilal Chughtai, Marius Hobbhahn, Nicholas Goldowsky-Dill, Stefan Heimersheim","submitted_at":"2025-02-05T17:49:40Z","abstract_excerpt":"AI models might use deceptive strategies as part of scheming or misaligned behaviour. Monitoring outputs alone is insufficient, since the AI might produce seemingly benign outputs while their internal reasoning is misaligned. We thus evaluate if linear probes can robustly detect deception by monitoring model activations. We test two probe-training datasets, one with contrasting instructions to be honest or deceptive (following Zou et al., 2023) and one of responses to simple roleplaying scenarios. We test whether these probes generalize to realistic settings where Llama-3.3-70B-Instruct behave"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.03407","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-05T17:49:40Z","cross_cats_sorted":[],"title_canon_sha256":"35c407a799af75675b9ade8c66efb76ccd4c1fe9c711485f2b734add52899a40","abstract_canon_sha256":"596bb947e33c992c61eb269cc7ed9ae681202a623f90ff17f7917dba57b828e8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:10:04.473142Z","signature_b64":"nQRyIKC4TZmfIU1gnKryrso40JiWNmLUAsw5jSvs1JWIs3r8CcEC8UmUU1UhUTLWN+AuiRfHjbhozFF6psyNAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c0cc1e04ba16a6094373aff269a3d7436d4945a4ca4ee11e0828f91b24f7ab06","last_reissued_at":"2026-07-05T10:10:04.472641Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:10:04.472641Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Detecting Strategic Deception Using Linear Probes","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bilal Chughtai, Marius Hobbhahn, Nicholas Goldowsky-Dill, Stefan Heimersheim","submitted_at":"2025-02-05T17:49:40Z","abstract_excerpt":"AI models might use deceptive strategies as part of scheming or misaligned behaviour. Monitoring outputs alone is insufficient, since the AI might produce seemingly benign outputs while their internal reasoning is misaligned. We thus evaluate if linear probes can robustly detect deception by monitoring model activations. We test two probe-training datasets, one with contrasting instructions to be honest or deceptive (following Zou et al., 2023) and one of responses to simple roleplaying scenarios. We test whether these probes generalize to realistic settings where Llama-3.3-70B-Instruct behave"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.03407","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.03407/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.03407","created_at":"2026-07-05T10:10:04.472702+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.03407v1","created_at":"2026-07-05T10:10:04.472702+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.03407","created_at":"2026-07-05T10:10:04.472702+00:00"},{"alias_kind":"pith_short_12","alias_value":"YDGB4BF2C2TA","created_at":"2026-07-05T10:10:04.472702+00:00"},{"alias_kind":"pith_short_16","alias_value":"YDGB4BF2C2TASQ3T","created_at":"2026-07-05T10:10:04.472702+00:00"},{"alias_kind":"pith_short_8","alias_value":"YDGB4BF2","created_at":"2026-07-05T10:10:04.472702+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12618","citing_title":"\"Did you lie?\" Evaluating Lie Detectors across Model Scale and Belief-Verified Model Organisms","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01047","citing_title":"Conversable Complexity: Agentic LLM Collectives as Interpretable Substrates","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15164","citing_title":"Position: Behavioural Assurance Cannot Verify the Safety Claims Governance Now Demands","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27958","citing_title":"Pressure-Testing Deception Probes in LLMs: Scaling, Robustness, and the Geometry of Deceptive Representations","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16339","citing_title":"Preference Instability in Reward Models: Detection and Mitigation via Sparse Autoencoders","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13339","citing_title":"Probing Persona-Dependent Preferences in Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15377","citing_title":"Ensemble Monitoring for AI Control: Diverse Signals Outweigh More Compute","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09391","citing_title":"Do Linear Probes Generalize Better in Persona Coordinates?","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2509.26238","citing_title":"Beyond Linear Probes: Dynamic Safety Monitoring for Language Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17408","citing_title":"The Impact of Off-Policy Training Data on Probe Generalisation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01151","citing_title":"Detecting Multi-Agent Collusion Through Multi-Agent Interpretability","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03121","citing_title":"An Independent Safety Evaluation of Kimi K2.5","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24801","citing_title":"Architecture Determines Observability of Transformers","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08942","citing_title":"Decomposing and Steering Functional Metacognition in Large Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09391","citing_title":"Do Linear Probes Generalize Better in Persona Coordinates?","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24801","citing_title":"Architecture Determines Observability of Transformers","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13386","citing_title":"Linear Probe Accuracy Scales with Model Size and Benefits from Multi-Layer Ensembling","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YDGB4BF2C2TASQ3TV7ZGTI6XIN","json":"https://pith.science/pith/YDGB4BF2C2TASQ3TV7ZGTI6XIN.json","graph_json":"https://pith.science/api/pith-number/YDGB4BF2C2TASQ3TV7ZGTI6XIN/graph.json","events_json":"https://pith.science/api/pith-number/YDGB4BF2C2TASQ3TV7ZGTI6XIN/events.json","paper":"https://pith.science/paper/YDGB4BF2"},"agent_actions":{"view_html":"https://pith.science/pith/YDGB4BF2C2TASQ3TV7ZGTI6XIN","download_json":"https://pith.science/pith/YDGB4BF2C2TASQ3TV7ZGTI6XIN.json","view_paper":"https://pith.science/paper/YDGB4BF2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.03407&json=true","fetch_graph":"https://pith.science/api/pith-number/YDGB4BF2C2TASQ3TV7ZGTI6XIN/graph.json","fetch_events":"https://pith.science/api/pith-number/YDGB4BF2C2TASQ3TV7ZGTI6XIN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YDGB4BF2C2TASQ3TV7ZGTI6XIN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YDGB4BF2C2TASQ3TV7ZGTI6XIN/action/storage_attestation","attest_author":"https://pith.science/pith/YDGB4BF2C2TASQ3TV7ZGTI6XIN/action/author_attestation","sign_citation":"https://pith.science/pith/YDGB4BF2C2TASQ3TV7ZGTI6XIN/action/citation_signature","submit_replication":"https://pith.science/pith/YDGB4BF2C2TASQ3TV7ZGTI6XIN/action/replication_record"}},"created_at":"2026-07-05T10:10:04.472702+00:00","updated_at":"2026-07-05T10:10:04.472702+00:00"}