{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7EVZGVH7JBPMZSVJGAJUQPJFAT","short_pith_number":"pith:7EVZGVH7","schema_version":"1.0","canonical_sha256":"f92b9354ff485ecccaa93013483d2504e0c9e625c3c639c6f59eed74da41d5c1","source":{"kind":"arxiv","id":"2502.08301","version":2},"attestation_state":"computed","paper":{"title":"Compromising Honesty and Harmlessness in Language Models via Deception Attacks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.CL","authors_text":"Francesca Carlon, Laur\\`ene Vaugrante, Maluna Menke, Thilo Hagendorff","submitted_at":"2025-02-12T11:02:59Z","abstract_excerpt":"Recent research on large language models (LLMs) has demonstrated their ability to understand and employ deceptive behavior, even without explicit prompting. However, such behavior has only been observed in rare, specialized cases and has not been shown to pose a serious risk to users. Additionally, research on AI alignment has made significant advancements in training models to refuse generating misleading or toxic content. As a result, LLMs generally became honest and harmless. In this study, we introduce \"deception attacks\" that undermine both of these traits, revealing a vulnerability that,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.08301","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-12T11:02:59Z","cross_cats_sorted":["cs.AI","cs.CY"],"title_canon_sha256":"062812402f1569939acfcd580a6acdafadf0bd5158acfa282dceeca4a29a5000","abstract_canon_sha256":"2646c527bcea96dcb7a2fb6608f9e371eae193ca95f1a25bd081ac03ac65a344"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:25:41.959783Z","signature_b64":"F5c1Hkess5+VFHVn8Qf2AybvCD5JgKgmFRZeEuVdmstZj4AXBb/XHVw2dBTzTpNg7jJacDNd2dTYfiRWam9hBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f92b9354ff485ecccaa93013483d2504e0c9e625c3c639c6f59eed74da41d5c1","last_reissued_at":"2026-07-05T11:25:41.959347Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:25:41.959347Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Compromising Honesty and Harmlessness in Language Models via Deception Attacks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY"],"primary_cat":"cs.CL","authors_text":"Francesca Carlon, Laur\\`ene Vaugrante, Maluna Menke, Thilo Hagendorff","submitted_at":"2025-02-12T11:02:59Z","abstract_excerpt":"Recent research on large language models (LLMs) has demonstrated their ability to understand and employ deceptive behavior, even without explicit prompting. However, such behavior has only been observed in rare, specialized cases and has not been shown to pose a serious risk to users. Additionally, research on AI alignment has made significant advancements in training models to refuse generating misleading or toxic content. As a result, LLMs generally became honest and harmless. In this study, we introduce \"deception attacks\" that undermine both of these traits, revealing a vulnerability that,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.08301","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.08301/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.08301","created_at":"2026-07-05T11:25:41.959410+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.08301v2","created_at":"2026-07-05T11:25:41.959410+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.08301","created_at":"2026-07-05T11:25:41.959410+00:00"},{"alias_kind":"pith_short_12","alias_value":"7EVZGVH7JBPM","created_at":"2026-07-05T11:25:41.959410+00:00"},{"alias_kind":"pith_short_16","alias_value":"7EVZGVH7JBPMZSVJ","created_at":"2026-07-05T11:25:41.959410+00:00"},{"alias_kind":"pith_short_8","alias_value":"7EVZGVH7","created_at":"2026-07-05T11:25:41.959410+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10747","citing_title":"The Arbiter Agent: Continually Monitoring Multi-Agent Conversations to Detect Emergent Misalignment","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09068","citing_title":"Emergent Misalignment Can Be Induced by Sycophancy and Reversed via Alignment Gating","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7EVZGVH7JBPMZSVJGAJUQPJFAT","json":"https://pith.science/pith/7EVZGVH7JBPMZSVJGAJUQPJFAT.json","graph_json":"https://pith.science/api/pith-number/7EVZGVH7JBPMZSVJGAJUQPJFAT/graph.json","events_json":"https://pith.science/api/pith-number/7EVZGVH7JBPMZSVJGAJUQPJFAT/events.json","paper":"https://pith.science/paper/7EVZGVH7"},"agent_actions":{"view_html":"https://pith.science/pith/7EVZGVH7JBPMZSVJGAJUQPJFAT","download_json":"https://pith.science/pith/7EVZGVH7JBPMZSVJGAJUQPJFAT.json","view_paper":"https://pith.science/paper/7EVZGVH7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.08301&json=true","fetch_graph":"https://pith.science/api/pith-number/7EVZGVH7JBPMZSVJGAJUQPJFAT/graph.json","fetch_events":"https://pith.science/api/pith-number/7EVZGVH7JBPMZSVJGAJUQPJFAT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7EVZGVH7JBPMZSVJGAJUQPJFAT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7EVZGVH7JBPMZSVJGAJUQPJFAT/action/storage_attestation","attest_author":"https://pith.science/pith/7EVZGVH7JBPMZSVJGAJUQPJFAT/action/author_attestation","sign_citation":"https://pith.science/pith/7EVZGVH7JBPMZSVJGAJUQPJFAT/action/citation_signature","submit_replication":"https://pith.science/pith/7EVZGVH7JBPMZSVJGAJUQPJFAT/action/replication_record"}},"created_at":"2026-07-05T11:25:41.959410+00:00","updated_at":"2026-07-05T11:25:41.959410+00:00"}