{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YCKXQD5QG45UAAAWHMHIJ66RWJ","short_pith_number":"pith:YCKXQD5Q","schema_version":"1.0","canonical_sha256":"c095780fb0373b4000163b0e84fbd1b25e6e665d99fc49ef199025e6546c5c8e","source":{"kind":"arxiv","id":"2505.09807","version":1},"attestation_state":"computed","paper":{"title":"Exploring the generalization of LLM truth directions on conversational formats","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"David Martens, Timour Ichmoukhamedov","submitted_at":"2025-05-14T21:21:08Z","abstract_excerpt":"Several recent works argue that LLMs have a universal truth direction where true and false statements are linearly separable in the activation space of the model. It has been demonstrated that linear probes trained on a single hidden state of the model already generalize across a range of topics and might even be used for lie detection in LLM conversations. In this work we explore how this truth direction generalizes between various conversational formats. We find good generalization between short conversations that end on a lie, but poor generalization to longer formats where the lie appears "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.09807","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-14T21:21:08Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"980d5b518acbf8673228f305307e9f25bf6ee4b612d5fc8a1f1cadb9b435631b","abstract_canon_sha256":"0c80c503d71d4c910bc5ddd11fc165fe114da69a5b6d73b5aaab75ded8f46ee9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:03:29.024152Z","signature_b64":"oVDnM2wub0oLkbdyYyqT4J3YRMadbAmH4WMRwztLTEtB4pe/2+zJ4OLxIunO9rtNxdvZye9pDwoyLEr4i8u6Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c095780fb0373b4000163b0e84fbd1b25e6e665d99fc49ef199025e6546c5c8e","last_reissued_at":"2026-07-05T11:03:29.023744Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:03:29.023744Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring the generalization of LLM truth directions on conversational formats","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"David Martens, Timour Ichmoukhamedov","submitted_at":"2025-05-14T21:21:08Z","abstract_excerpt":"Several recent works argue that LLMs have a universal truth direction where true and false statements are linearly separable in the activation space of the model. It has been demonstrated that linear probes trained on a single hidden state of the model already generalize across a range of topics and might even be used for lie detection in LLM conversations. In this work we explore how this truth direction generalizes between various conversational formats. We find good generalization between short conversations that end on a lie, but poor generalization to longer formats where the lie appears "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.09807","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.09807/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.09807","created_at":"2026-07-05T11:03:29.023800+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.09807v1","created_at":"2026-07-05T11:03:29.023800+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.09807","created_at":"2026-07-05T11:03:29.023800+00:00"},{"alias_kind":"pith_short_12","alias_value":"YCKXQD5QG45U","created_at":"2026-07-05T11:03:29.023800+00:00"},{"alias_kind":"pith_short_16","alias_value":"YCKXQD5QG45UAAAW","created_at":"2026-07-05T11:03:29.023800+00:00"},{"alias_kind":"pith_short_8","alias_value":"YCKXQD5Q","created_at":"2026-07-05T11:03:29.023800+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YCKXQD5QG45UAAAWHMHIJ66RWJ","json":"https://pith.science/pith/YCKXQD5QG45UAAAWHMHIJ66RWJ.json","graph_json":"https://pith.science/api/pith-number/YCKXQD5QG45UAAAWHMHIJ66RWJ/graph.json","events_json":"https://pith.science/api/pith-number/YCKXQD5QG45UAAAWHMHIJ66RWJ/events.json","paper":"https://pith.science/paper/YCKXQD5Q"},"agent_actions":{"view_html":"https://pith.science/pith/YCKXQD5QG45UAAAWHMHIJ66RWJ","download_json":"https://pith.science/pith/YCKXQD5QG45UAAAWHMHIJ66RWJ.json","view_paper":"https://pith.science/paper/YCKXQD5Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.09807&json=true","fetch_graph":"https://pith.science/api/pith-number/YCKXQD5QG45UAAAWHMHIJ66RWJ/graph.json","fetch_events":"https://pith.science/api/pith-number/YCKXQD5QG45UAAAWHMHIJ66RWJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YCKXQD5QG45UAAAWHMHIJ66RWJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YCKXQD5QG45UAAAWHMHIJ66RWJ/action/storage_attestation","attest_author":"https://pith.science/pith/YCKXQD5QG45UAAAWHMHIJ66RWJ/action/author_attestation","sign_citation":"https://pith.science/pith/YCKXQD5QG45UAAAWHMHIJ66RWJ/action/citation_signature","submit_replication":"https://pith.science/pith/YCKXQD5QG45UAAAWHMHIJ66RWJ/action/replication_record"}},"created_at":"2026-07-05T11:03:29.023800+00:00","updated_at":"2026-07-05T11:03:29.023800+00:00"}