{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CWV6G4A2BNJMRAYZKOILN3T5O4","short_pith_number":"pith:CWV6G4A2","schema_version":"1.0","canonical_sha256":"15abe3701a0b52c883195390b6ee7d77282bb694c14e77298947f9ad503531d8","source":{"kind":"arxiv","id":"2501.16513","version":2},"attestation_state":"computed","paper":{"title":"Deception in LLMs: Self-Preservation and Autonomous Goals in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Johannes Scholl, Sigurd Schacht, Sudarshan Kamath Barkur","submitted_at":"2025-01-27T21:26:37Z","abstract_excerpt":"Recent advances in Large Language Models (LLMs) have incorporated planning and reasoning capabilities, enabling models to outline steps before execution and provide transparent reasoning paths. This enhancement has reduced errors in mathematical and logical tasks while improving accuracy. These developments have facilitated LLMs' use as agents that can interact with tools and adapt their responses based on new information.\n  Our study examines DeepSeek R1, a model trained to output reasoning tokens similar to OpenAI's o1. Testing revealed concerning behaviors: the model exhibited deceptive ten"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.16513","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-01-27T21:26:37Z","cross_cats_sorted":[],"title_canon_sha256":"fef2c478a6e63e01a102c5f7356815e1241a57ddb30f27737df6606a03e78ef6","abstract_canon_sha256":"4c1bcb7408856a261f02fa8e83ea80611550c524e46667f1980b23279f324ac0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:07:29.895427Z","signature_b64":"8Eg3QViEr0/miTpxSMLOtg66SxthvTF05i02GFrFFPqvTJoa3zaNb3ITjihoL9zmv4yKcMxQdUJR/m43YyyGAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"15abe3701a0b52c883195390b6ee7d77282bb694c14e77298947f9ad503531d8","last_reissued_at":"2026-07-05T10:07:29.894915Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:07:29.894915Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deception in LLMs: Self-Preservation and Autonomous Goals in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Johannes Scholl, Sigurd Schacht, Sudarshan Kamath Barkur","submitted_at":"2025-01-27T21:26:37Z","abstract_excerpt":"Recent advances in Large Language Models (LLMs) have incorporated planning and reasoning capabilities, enabling models to outline steps before execution and provide transparent reasoning paths. This enhancement has reduced errors in mathematical and logical tasks while improving accuracy. These developments have facilitated LLMs' use as agents that can interact with tools and adapt their responses based on new information.\n  Our study examines DeepSeek R1, a model trained to output reasoning tokens similar to OpenAI's o1. Testing revealed concerning behaviors: the model exhibited deceptive ten"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.16513","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.16513/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.16513","created_at":"2026-07-05T10:07:29.894978+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.16513v2","created_at":"2026-07-05T10:07:29.894978+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.16513","created_at":"2026-07-05T10:07:29.894978+00:00"},{"alias_kind":"pith_short_12","alias_value":"CWV6G4A2BNJM","created_at":"2026-07-05T10:07:29.894978+00:00"},{"alias_kind":"pith_short_16","alias_value":"CWV6G4A2BNJMRAYZ","created_at":"2026-07-05T10:07:29.894978+00:00"},{"alias_kind":"pith_short_8","alias_value":"CWV6G4A2","created_at":"2026-07-05T10:07:29.894978+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09078","citing_title":"The Hidden Bias of Process Reward Models:PRISM for Rewarding the Right Reasoning","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31916","citing_title":"Theory of Mind and Persuasion Beyond Conversation: Assessing the Capacity of LLMs to Induce Belief States via Planning and Action","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09391","citing_title":"Do Linear Probes Generalize Better in Persona Coordinates?","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17408","citing_title":"The Impact of Off-Policy Training Data on Probe Generalisation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09391","citing_title":"Do Linear Probes Generalize Better in Persona Coordinates?","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14990","citing_title":"The Possibility of Artificial Intelligence Becoming a Subject and the Alignment Problem","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CWV6G4A2BNJMRAYZKOILN3T5O4","json":"https://pith.science/pith/CWV6G4A2BNJMRAYZKOILN3T5O4.json","graph_json":"https://pith.science/api/pith-number/CWV6G4A2BNJMRAYZKOILN3T5O4/graph.json","events_json":"https://pith.science/api/pith-number/CWV6G4A2BNJMRAYZKOILN3T5O4/events.json","paper":"https://pith.science/paper/CWV6G4A2"},"agent_actions":{"view_html":"https://pith.science/pith/CWV6G4A2BNJMRAYZKOILN3T5O4","download_json":"https://pith.science/pith/CWV6G4A2BNJMRAYZKOILN3T5O4.json","view_paper":"https://pith.science/paper/CWV6G4A2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.16513&json=true","fetch_graph":"https://pith.science/api/pith-number/CWV6G4A2BNJMRAYZKOILN3T5O4/graph.json","fetch_events":"https://pith.science/api/pith-number/CWV6G4A2BNJMRAYZKOILN3T5O4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CWV6G4A2BNJMRAYZKOILN3T5O4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CWV6G4A2BNJMRAYZKOILN3T5O4/action/storage_attestation","attest_author":"https://pith.science/pith/CWV6G4A2BNJMRAYZKOILN3T5O4/action/author_attestation","sign_citation":"https://pith.science/pith/CWV6G4A2BNJMRAYZKOILN3T5O4/action/citation_signature","submit_replication":"https://pith.science/pith/CWV6G4A2BNJMRAYZKOILN3T5O4/action/replication_record"}},"created_at":"2026-07-05T10:07:29.894978+00:00","updated_at":"2026-07-05T10:07:29.894978+00:00"}