{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JBTOZ56SAQZ5B3U3DQGZSXFLZG","short_pith_number":"pith:JBTOZ56S","schema_version":"1.0","canonical_sha256":"4866ecf7d20433d0ee9b1c0d995cabc9b724511bb94fc4e12ebc051aa90f7eb0","source":{"kind":"arxiv","id":"2407.12831","version":2},"attestation_state":"computed","paper":{"title":"Truth is Universal: Robust Detection of Lies in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Boaz Nadler, Fred A. Hamprecht, Lennart B\\\"urger","submitted_at":"2024-07-03T13:01:54Z","abstract_excerpt":"Large Language Models (LLMs) have revolutionised natural language processing, exhibiting impressive human-like capabilities. In particular, LLMs are capable of \"lying\", knowingly outputting false statements. Hence, it is of interest and importance to develop methods to detect when LLMs lie. Indeed, several authors trained classifiers to detect LLM lies based on their internal model activations. However, other researchers showed that these classifiers may fail to generalise, for example to negated statements. In this work, we aim to develop a robust method to detect when an LLM is lying. To thi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.12831","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-03T13:01:54Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9279ddb2629870f97429897b7ebaba202894f224f126e2d75775e0ae1b192575","abstract_canon_sha256":"c7373d60334825a4ed727a86fbca8b2e243de6c3d1f3bd2df30b31bdb1c406be"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:23:08.872412Z","signature_b64":"R/QrhrKnH4RQ5qZ5qY3dmBkOzweJQGl22yH/CaPcM/0ZiqnqyvSy2rYyz2vR4mnKmJcn3q7YgmH/l6I7VSwWAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4866ecf7d20433d0ee9b1c0d995cabc9b724511bb94fc4e12ebc051aa90f7eb0","last_reissued_at":"2026-07-05T09:23:08.871839Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:23:08.871839Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Truth is Universal: Robust Detection of Lies in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Boaz Nadler, Fred A. Hamprecht, Lennart B\\\"urger","submitted_at":"2024-07-03T13:01:54Z","abstract_excerpt":"Large Language Models (LLMs) have revolutionised natural language processing, exhibiting impressive human-like capabilities. In particular, LLMs are capable of \"lying\", knowingly outputting false statements. Hence, it is of interest and importance to develop methods to detect when LLMs lie. Indeed, several authors trained classifiers to detect LLM lies based on their internal model activations. However, other researchers showed that these classifiers may fail to generalise, for example to negated statements. In this work, we aim to develop a robust method to detect when an LLM is lying. To thi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.12831","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.12831/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.12831","created_at":"2026-07-05T09:23:08.871910+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.12831v2","created_at":"2026-07-05T09:23:08.871910+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.12831","created_at":"2026-07-05T09:23:08.871910+00:00"},{"alias_kind":"pith_short_12","alias_value":"JBTOZ56SAQZ5","created_at":"2026-07-05T09:23:08.871910+00:00"},{"alias_kind":"pith_short_16","alias_value":"JBTOZ56SAQZ5B3U3","created_at":"2026-07-05T09:23:08.871910+00:00"},{"alias_kind":"pith_short_8","alias_value":"JBTOZ56S","created_at":"2026-07-05T09:23:08.871910+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.14791","citing_title":"Transcoders for Investigating Deception in Language Models","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JBTOZ56SAQZ5B3U3DQGZSXFLZG","json":"https://pith.science/pith/JBTOZ56SAQZ5B3U3DQGZSXFLZG.json","graph_json":"https://pith.science/api/pith-number/JBTOZ56SAQZ5B3U3DQGZSXFLZG/graph.json","events_json":"https://pith.science/api/pith-number/JBTOZ56SAQZ5B3U3DQGZSXFLZG/events.json","paper":"https://pith.science/paper/JBTOZ56S"},"agent_actions":{"view_html":"https://pith.science/pith/JBTOZ56SAQZ5B3U3DQGZSXFLZG","download_json":"https://pith.science/pith/JBTOZ56SAQZ5B3U3DQGZSXFLZG.json","view_paper":"https://pith.science/paper/JBTOZ56S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.12831&json=true","fetch_graph":"https://pith.science/api/pith-number/JBTOZ56SAQZ5B3U3DQGZSXFLZG/graph.json","fetch_events":"https://pith.science/api/pith-number/JBTOZ56SAQZ5B3U3DQGZSXFLZG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JBTOZ56SAQZ5B3U3DQGZSXFLZG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JBTOZ56SAQZ5B3U3DQGZSXFLZG/action/storage_attestation","attest_author":"https://pith.science/pith/JBTOZ56SAQZ5B3U3DQGZSXFLZG/action/author_attestation","sign_citation":"https://pith.science/pith/JBTOZ56SAQZ5B3U3DQGZSXFLZG/action/citation_signature","submit_replication":"https://pith.science/pith/JBTOZ56SAQZ5B3U3DQGZSXFLZG/action/replication_record"}},"created_at":"2026-07-05T09:23:08.871910+00:00","updated_at":"2026-07-05T09:23:08.871910+00:00"}