{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BQVD4VSZ52GG4AJDJFOO6GKWE7","short_pith_number":"pith:BQVD4VSZ","schema_version":"1.0","canonical_sha256":"0c2a3e5659ee8c6e0123495cef195627e75725aae267e72c2bc97d28cc8e3ed4","source":{"kind":"arxiv","id":"2506.04909","version":1},"attestation_state":"computed","paper":{"title":"When Thinking LLMs Lie: Unveiling the Strategic Deception in Representations of Reasoning Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CR","cs.LG"],"primary_cat":"cs.AI","authors_text":"Kai Wang, Meng Sun, Yihao Zhang","submitted_at":"2025-06-05T11:44:19Z","abstract_excerpt":"The honesty of large language models (LLMs) is a critical alignment challenge, especially as advanced systems with chain-of-thought (CoT) reasoning may strategically deceive humans. Unlike traditional honesty issues on LLMs, which could be possibly explained as some kind of hallucination, those models' explicit thought paths enable us to study strategic deception--goal-driven, intentional misinformation where reasoning contradicts outputs. Using representation engineering, we systematically induce, detect, and control such deception in CoT-enabled LLMs, extracting \"deception vectors\" via Linea"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.04909","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-06-05T11:44:19Z","cross_cats_sorted":["cs.CL","cs.CR","cs.LG"],"title_canon_sha256":"4ca4009190fbdb8340d416c3bab59952422046a3247ca5c9a08d6b2634d40cc6","abstract_canon_sha256":"a81ad6d738144932afc51ec5c4553f5f2bd85a29e8a8664bd5109b74a7e8c947"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:38.860884Z","signature_b64":"MmO251Ulq8HC+KGFPl5y0LXYlCotwX9RjQMERV6BJoHEo+qN9/JdeSE4poFpkDwvws8pbY9RO0kKxrVEnX6qDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c2a3e5659ee8c6e0123495cef195627e75725aae267e72c2bc97d28cc8e3ed4","last_reissued_at":"2026-07-05T11:16:38.860333Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:38.860333Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Thinking LLMs Lie: Unveiling the Strategic Deception in Representations of Reasoning Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CR","cs.LG"],"primary_cat":"cs.AI","authors_text":"Kai Wang, Meng Sun, Yihao Zhang","submitted_at":"2025-06-05T11:44:19Z","abstract_excerpt":"The honesty of large language models (LLMs) is a critical alignment challenge, especially as advanced systems with chain-of-thought (CoT) reasoning may strategically deceive humans. Unlike traditional honesty issues on LLMs, which could be possibly explained as some kind of hallucination, those models' explicit thought paths enable us to study strategic deception--goal-driven, intentional misinformation where reasoning contradicts outputs. Using representation engineering, we systematically induce, detect, and control such deception in CoT-enabled LLMs, extracting \"deception vectors\" via Linea"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.04909","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.04909/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.04909","created_at":"2026-07-05T11:16:38.860403+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.04909v1","created_at":"2026-07-05T11:16:38.860403+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.04909","created_at":"2026-07-05T11:16:38.860403+00:00"},{"alias_kind":"pith_short_12","alias_value":"BQVD4VSZ52GG","created_at":"2026-07-05T11:16:38.860403+00:00"},{"alias_kind":"pith_short_16","alias_value":"BQVD4VSZ52GG4AJD","created_at":"2026-07-05T11:16:38.860403+00:00"},{"alias_kind":"pith_short_8","alias_value":"BQVD4VSZ","created_at":"2026-07-05T11:16:38.860403+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.13310","citing_title":"RogueAI: A Reverse Turing Test for Detecting Licensed AI Deception in Dialogue","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02507","citing_title":"What LLM Agents Say When No One Is Watching: Social Structure and Latent Objective Emergence in Multi-Agent Debates","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19270","citing_title":"DECOR: Auditing LLM Deception via Information Manipulation Theory","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BQVD4VSZ52GG4AJDJFOO6GKWE7","json":"https://pith.science/pith/BQVD4VSZ52GG4AJDJFOO6GKWE7.json","graph_json":"https://pith.science/api/pith-number/BQVD4VSZ52GG4AJDJFOO6GKWE7/graph.json","events_json":"https://pith.science/api/pith-number/BQVD4VSZ52GG4AJDJFOO6GKWE7/events.json","paper":"https://pith.science/paper/BQVD4VSZ"},"agent_actions":{"view_html":"https://pith.science/pith/BQVD4VSZ52GG4AJDJFOO6GKWE7","download_json":"https://pith.science/pith/BQVD4VSZ52GG4AJDJFOO6GKWE7.json","view_paper":"https://pith.science/paper/BQVD4VSZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.04909&json=true","fetch_graph":"https://pith.science/api/pith-number/BQVD4VSZ52GG4AJDJFOO6GKWE7/graph.json","fetch_events":"https://pith.science/api/pith-number/BQVD4VSZ52GG4AJDJFOO6GKWE7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BQVD4VSZ52GG4AJDJFOO6GKWE7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BQVD4VSZ52GG4AJDJFOO6GKWE7/action/storage_attestation","attest_author":"https://pith.science/pith/BQVD4VSZ52GG4AJDJFOO6GKWE7/action/author_attestation","sign_citation":"https://pith.science/pith/BQVD4VSZ52GG4AJDJFOO6GKWE7/action/citation_signature","submit_replication":"https://pith.science/pith/BQVD4VSZ52GG4AJDJFOO6GKWE7/action/replication_record"}},"created_at":"2026-07-05T11:16:38.860403+00:00","updated_at":"2026-07-05T11:16:38.860403+00:00"}