{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6MVAWZAOVVP4IRJMQZCVW4MEHG","short_pith_number":"pith:6MVAWZAO","schema_version":"1.0","canonical_sha256":"f32a0b640ead5fc4452c86455b718439935818b483453a713f42aeaa18f1a056","source":{"kind":"arxiv","id":"2402.13249","version":2},"attestation_state":"computed","paper":{"title":"TofuEval: Evaluating Hallucinations of LLMs on Topic-Focused Dialogue Summarization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Amy Wing-mei Wong, Hang Su, Hwanjun Song, Igor Shalyminov, Jake W. Vincent, Jon Burnsky, Kathleen McKeown, Lijia Sun, Liyan Tang, Saab Mansour, Siffi Singh, Song Feng, Yi Zhang, Yu'an Yang","submitted_at":"2024-02-20T18:58:49Z","abstract_excerpt":"Single document news summarization has seen substantial progress on faithfulness in recent years, driven by research on the evaluation of factual consistency, or hallucinations. We ask whether these advances carry over to other text summarization domains. We propose a new evaluation benchmark on topic-focused dialogue summarization, generated by LLMs of varying sizes. We provide binary sentence-level human annotations of the factual consistency of these summaries along with detailed explanations of factually inconsistent sentences. Our analysis shows that existing LLMs hallucinate significant "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.13249","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-20T18:58:49Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a8a51e6aef08883dd9c1c580778d3cb4b56d834478d3d6bcf2106a2303de20b9","abstract_canon_sha256":"e579003193e867532111e345c910d56d6b98ffff69093f986bad713216fe6171"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:02:51.189903Z","signature_b64":"5GAk1g8lQQY95K+ipfDpASgRf4gCNQsxrnZddCOym60VKIW5iku9LsLk7e+w1UV7uBgHi7jkk05GtGw/yJSCAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f32a0b640ead5fc4452c86455b718439935818b483453a713f42aeaa18f1a056","last_reissued_at":"2026-07-05T08:02:51.188963Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:02:51.188963Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TofuEval: Evaluating Hallucinations of LLMs on Topic-Focused Dialogue Summarization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Amy Wing-mei Wong, Hang Su, Hwanjun Song, Igor Shalyminov, Jake W. Vincent, Jon Burnsky, Kathleen McKeown, Lijia Sun, Liyan Tang, Saab Mansour, Siffi Singh, Song Feng, Yi Zhang, Yu'an Yang","submitted_at":"2024-02-20T18:58:49Z","abstract_excerpt":"Single document news summarization has seen substantial progress on faithfulness in recent years, driven by research on the evaluation of factual consistency, or hallucinations. We ask whether these advances carry over to other text summarization domains. We propose a new evaluation benchmark on topic-focused dialogue summarization, generated by LLMs of varying sizes. We provide binary sentence-level human annotations of the factual consistency of these summaries along with detailed explanations of factually inconsistent sentences. Our analysis shows that existing LLMs hallucinate significant "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.13249","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.13249/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.13249","created_at":"2026-07-05T08:02:51.189041+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.13249v2","created_at":"2026-07-05T08:02:51.189041+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.13249","created_at":"2026-07-05T08:02:51.189041+00:00"},{"alias_kind":"pith_short_12","alias_value":"6MVAWZAOVVP4","created_at":"2026-07-05T08:02:51.189041+00:00"},{"alias_kind":"pith_short_16","alias_value":"6MVAWZAOVVP4IRJM","created_at":"2026-07-05T08:02:51.189041+00:00"},{"alias_kind":"pith_short_8","alias_value":"6MVAWZAO","created_at":"2026-07-05T08:02:51.189041+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12807","citing_title":"Detect, Remask, Repair: Diffusion Editing for Faithful Summarization of Evolving Contexts","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2406.15809","citing_title":"LaMSUM: Amplifying Voices Against Harassment through LLM Guided Extractive Summarization of User Incident Reports","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20131","citing_title":"Whose Story Gets Told? Positionality and Bias in LLM Summaries of Life Narratives","ref_index":139,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6MVAWZAOVVP4IRJMQZCVW4MEHG","json":"https://pith.science/pith/6MVAWZAOVVP4IRJMQZCVW4MEHG.json","graph_json":"https://pith.science/api/pith-number/6MVAWZAOVVP4IRJMQZCVW4MEHG/graph.json","events_json":"https://pith.science/api/pith-number/6MVAWZAOVVP4IRJMQZCVW4MEHG/events.json","paper":"https://pith.science/paper/6MVAWZAO"},"agent_actions":{"view_html":"https://pith.science/pith/6MVAWZAOVVP4IRJMQZCVW4MEHG","download_json":"https://pith.science/pith/6MVAWZAOVVP4IRJMQZCVW4MEHG.json","view_paper":"https://pith.science/paper/6MVAWZAO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.13249&json=true","fetch_graph":"https://pith.science/api/pith-number/6MVAWZAOVVP4IRJMQZCVW4MEHG/graph.json","fetch_events":"https://pith.science/api/pith-number/6MVAWZAOVVP4IRJMQZCVW4MEHG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6MVAWZAOVVP4IRJMQZCVW4MEHG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6MVAWZAOVVP4IRJMQZCVW4MEHG/action/storage_attestation","attest_author":"https://pith.science/pith/6MVAWZAOVVP4IRJMQZCVW4MEHG/action/author_attestation","sign_citation":"https://pith.science/pith/6MVAWZAOVVP4IRJMQZCVW4MEHG/action/citation_signature","submit_replication":"https://pith.science/pith/6MVAWZAOVVP4IRJMQZCVW4MEHG/action/replication_record"}},"created_at":"2026-07-05T08:02:51.189041+00:00","updated_at":"2026-07-05T08:02:51.189041+00:00"}