{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:X2EBIQY6GKCBXJAYVZIXB6JXOZ","short_pith_number":"pith:X2EBIQY6","schema_version":"1.0","canonical_sha256":"be8814431e32841ba418ae5170f93776477811bb529353333792bbc1657b4555","source":{"kind":"arxiv","id":"2506.18628","version":1},"attestation_state":"computed","paper":{"title":"AggTruth: Contextual Hallucination Detection using Aggregated Attention Scores in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Jan Eliasz, Jan Koco\\'n, Konrad Kie{\\l}czy\\'nski, Miko{\\l}aj Langner, Piotr Matys, Przemys{\\l}aw Kazienko, Teddy Ferdinan","submitted_at":"2025-06-23T13:35:05Z","abstract_excerpt":"In real-world applications, Large Language Models (LLMs) often hallucinate, even in Retrieval-Augmented Generation (RAG) settings, which poses a significant challenge to their deployment. In this paper, we introduce AggTruth, a method for online detection of contextual hallucinations by analyzing the distribution of internal attention scores in the provided context (passage). Specifically, we propose four different variants of the method, each varying in the aggregation technique used to calculate attention scores. Across all LLMs examined, AggTruth demonstrated stable performance in both same"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.18628","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-06-23T13:35:05Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"2df8cabdea40a5ea732a3b4bbe59c8b72c3e0e82909ca509069b0e5b9072f9d3","abstract_canon_sha256":"13aaf8f6e3217daf4760cea37ab5b4c9d66e978740b994e1af3df222b2f3a40f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:25:53.360934Z","signature_b64":"5jaRRKfYj3HqLcXjlItFezLqvVjlnKfPENTLFfGlYeKGmYwdUNR5R8tCkn+demogJK5a1LmL0JSzj48DIKXYCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"be8814431e32841ba418ae5170f93776477811bb529353333792bbc1657b4555","last_reissued_at":"2026-07-05T11:25:53.360443Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:25:53.360443Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AggTruth: Contextual Hallucination Detection using Aggregated Attention Scores in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Jan Eliasz, Jan Koco\\'n, Konrad Kie{\\l}czy\\'nski, Miko{\\l}aj Langner, Piotr Matys, Przemys{\\l}aw Kazienko, Teddy Ferdinan","submitted_at":"2025-06-23T13:35:05Z","abstract_excerpt":"In real-world applications, Large Language Models (LLMs) often hallucinate, even in Retrieval-Augmented Generation (RAG) settings, which poses a significant challenge to their deployment. In this paper, we introduce AggTruth, a method for online detection of contextual hallucinations by analyzing the distribution of internal attention scores in the provided context (passage). Specifically, we propose four different variants of the method, each varying in the aggregation technique used to calculate attention scores. Across all LLMs examined, AggTruth demonstrated stable performance in both same"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.18628","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.18628/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.18628","created_at":"2026-07-05T11:25:53.360501+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.18628v1","created_at":"2026-07-05T11:25:53.360501+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.18628","created_at":"2026-07-05T11:25:53.360501+00:00"},{"alias_kind":"pith_short_12","alias_value":"X2EBIQY6GKCB","created_at":"2026-07-05T11:25:53.360501+00:00"},{"alias_kind":"pith_short_16","alias_value":"X2EBIQY6GKCBXJAY","created_at":"2026-07-05T11:25:53.360501+00:00"},{"alias_kind":"pith_short_8","alias_value":"X2EBIQY6","created_at":"2026-07-05T11:25:53.360501+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X2EBIQY6GKCBXJAYVZIXB6JXOZ","json":"https://pith.science/pith/X2EBIQY6GKCBXJAYVZIXB6JXOZ.json","graph_json":"https://pith.science/api/pith-number/X2EBIQY6GKCBXJAYVZIXB6JXOZ/graph.json","events_json":"https://pith.science/api/pith-number/X2EBIQY6GKCBXJAYVZIXB6JXOZ/events.json","paper":"https://pith.science/paper/X2EBIQY6"},"agent_actions":{"view_html":"https://pith.science/pith/X2EBIQY6GKCBXJAYVZIXB6JXOZ","download_json":"https://pith.science/pith/X2EBIQY6GKCBXJAYVZIXB6JXOZ.json","view_paper":"https://pith.science/paper/X2EBIQY6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.18628&json=true","fetch_graph":"https://pith.science/api/pith-number/X2EBIQY6GKCBXJAYVZIXB6JXOZ/graph.json","fetch_events":"https://pith.science/api/pith-number/X2EBIQY6GKCBXJAYVZIXB6JXOZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X2EBIQY6GKCBXJAYVZIXB6JXOZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X2EBIQY6GKCBXJAYVZIXB6JXOZ/action/storage_attestation","attest_author":"https://pith.science/pith/X2EBIQY6GKCBXJAYVZIXB6JXOZ/action/author_attestation","sign_citation":"https://pith.science/pith/X2EBIQY6GKCBXJAYVZIXB6JXOZ/action/citation_signature","submit_replication":"https://pith.science/pith/X2EBIQY6GKCBXJAYVZIXB6JXOZ/action/replication_record"}},"created_at":"2026-07-05T11:25:53.360501+00:00","updated_at":"2026-07-05T11:25:53.360501+00:00"}