{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:JKKBDDOVAJJN6WTUGKNVNTC5W7","short_pith_number":"pith:JKKBDDOV","schema_version":"1.0","canonical_sha256":"4a94118dd50252df5a74329b56cc5db7e029ad19b0f40d63e4cce84b9315aa96","source":{"kind":"arxiv","id":"2312.03729","version":1},"attestation_state":"computed","paper":{"title":"Cognitive Dissonance: Why Do Language Model Outputs Disagree with Internal Representations of Truthfulness?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dylan Hadfield-Menell, Jacob Andreas, Kevin Liu, Stephen Casper","submitted_at":"2023-11-27T18:59:14Z","abstract_excerpt":"Neural language models (LMs) can be used to evaluate the truth of factual statements in two ways: they can be either queried for statement probabilities, or probed for internal representations of truthfulness. Past work has found that these two procedures sometimes disagree, and that probes tend to be more accurate than LM outputs. This has led some researchers to conclude that LMs \"lie\" or otherwise encode non-cooperative communicative intents. Is this an accurate description of today's LMs, or can query-probe disagreement arise in other ways? We identify three different classes of disagreeme"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.03729","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-11-27T18:59:14Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4dafd5a5924e90b6711d18f505af888261606cdcc05c8dfc1f86c55772d5cc72","abstract_canon_sha256":"5d5af5e4090978c12056b42b216896e538dc9dd04f16eefde6b04d3b6f634305"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:21:13.926108Z","signature_b64":"PI7Wh0+gihkpFGvvH6tGMYDp2c9V0KBB7045MoHbc+ai1uF6tQPI7Wl4+x0UGCnXR8iXbFxkr95EPdAfO383Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4a94118dd50252df5a74329b56cc5db7e029ad19b0f40d63e4cce84b9315aa96","last_reissued_at":"2026-07-05T07:21:13.925655Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:21:13.925655Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cognitive Dissonance: Why Do Language Model Outputs Disagree with Internal Representations of Truthfulness?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Dylan Hadfield-Menell, Jacob Andreas, Kevin Liu, Stephen Casper","submitted_at":"2023-11-27T18:59:14Z","abstract_excerpt":"Neural language models (LMs) can be used to evaluate the truth of factual statements in two ways: they can be either queried for statement probabilities, or probed for internal representations of truthfulness. Past work has found that these two procedures sometimes disagree, and that probes tend to be more accurate than LM outputs. This has led some researchers to conclude that LMs \"lie\" or otherwise encode non-cooperative communicative intents. Is this an accurate description of today's LMs, or can query-probe disagreement arise in other ways? We identify three different classes of disagreeme"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.03729","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.03729/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.03729","created_at":"2026-07-05T07:21:13.925717+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.03729v1","created_at":"2026-07-05T07:21:13.925717+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.03729","created_at":"2026-07-05T07:21:13.925717+00:00"},{"alias_kind":"pith_short_12","alias_value":"JKKBDDOVAJJN","created_at":"2026-07-05T07:21:13.925717+00:00"},{"alias_kind":"pith_short_16","alias_value":"JKKBDDOVAJJN6WTU","created_at":"2026-07-05T07:21:13.925717+00:00"},{"alias_kind":"pith_short_8","alias_value":"JKKBDDOV","created_at":"2026-07-05T07:21:13.925717+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JKKBDDOVAJJN6WTUGKNVNTC5W7","json":"https://pith.science/pith/JKKBDDOVAJJN6WTUGKNVNTC5W7.json","graph_json":"https://pith.science/api/pith-number/JKKBDDOVAJJN6WTUGKNVNTC5W7/graph.json","events_json":"https://pith.science/api/pith-number/JKKBDDOVAJJN6WTUGKNVNTC5W7/events.json","paper":"https://pith.science/paper/JKKBDDOV"},"agent_actions":{"view_html":"https://pith.science/pith/JKKBDDOVAJJN6WTUGKNVNTC5W7","download_json":"https://pith.science/pith/JKKBDDOVAJJN6WTUGKNVNTC5W7.json","view_paper":"https://pith.science/paper/JKKBDDOV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.03729&json=true","fetch_graph":"https://pith.science/api/pith-number/JKKBDDOVAJJN6WTUGKNVNTC5W7/graph.json","fetch_events":"https://pith.science/api/pith-number/JKKBDDOVAJJN6WTUGKNVNTC5W7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JKKBDDOVAJJN6WTUGKNVNTC5W7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JKKBDDOVAJJN6WTUGKNVNTC5W7/action/storage_attestation","attest_author":"https://pith.science/pith/JKKBDDOVAJJN6WTUGKNVNTC5W7/action/author_attestation","sign_citation":"https://pith.science/pith/JKKBDDOVAJJN6WTUGKNVNTC5W7/action/citation_signature","submit_replication":"https://pith.science/pith/JKKBDDOVAJJN6WTUGKNVNTC5W7/action/replication_record"}},"created_at":"2026-07-05T07:21:13.925717+00:00","updated_at":"2026-07-05T07:21:13.925717+00:00"}