{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:6XB3SP7IZH4AOEEYXXHSBODNUW","short_pith_number":"pith:6XB3SP7I","schema_version":"1.0","canonical_sha256":"f5c3b93fe8c9f8071098bdcf20b86da5adcd461dcc942355b0ff18c7aeb73ea4","source":{"kind":"arxiv","id":"2104.13346","version":2},"attestation_state":"computed","paper":{"title":"Understanding Factuality in Abstractive Summarization with FRANK: A Benchmark for Factuality Metrics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Artidoro Pagnoni, Vidhisha Balachandran, Yulia Tsvetkov","submitted_at":"2021-04-27T17:28:07Z","abstract_excerpt":"Modern summarization models generate highly fluent but often factually unreliable outputs. This motivated a surge of metrics attempting to measure the factuality of automatically generated summaries. Due to the lack of common benchmarks, these metrics cannot be compared. Moreover, all these methods treat factuality as a binary concept and fail to provide deeper insights into the kinds of inconsistencies made by different systems. To address these limitations, we devise a typology of factual errors and use it to collect human annotations of generated summaries from state-of-the-art summarizatio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.13346","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-04-27T17:28:07Z","cross_cats_sorted":[],"title_canon_sha256":"9b9b9b1a34d2f9d1e00ef0d3bf4f0a45c65a01d88c4b9508828a57581b89863a","abstract_canon_sha256":"e85e06e6c82b52cfedb600c7c50c49fcbc7056296acaaa5e97372ee58758b8b4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:00:18.067372Z","signature_b64":"tM8cghUIn1vIgtwffnRXinrCH/QiMwf/ILDntqU5rwPIYbqftlgt90Bh5M+pyZAXcI4pl5yVF4gYBzanBYkIAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f5c3b93fe8c9f8071098bdcf20b86da5adcd461dcc942355b0ff18c7aeb73ea4","last_reissued_at":"2026-07-05T03:00:18.066951Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:00:18.066951Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding Factuality in Abstractive Summarization with FRANK: A Benchmark for Factuality Metrics","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Artidoro Pagnoni, Vidhisha Balachandran, Yulia Tsvetkov","submitted_at":"2021-04-27T17:28:07Z","abstract_excerpt":"Modern summarization models generate highly fluent but often factually unreliable outputs. This motivated a surge of metrics attempting to measure the factuality of automatically generated summaries. Due to the lack of common benchmarks, these metrics cannot be compared. Moreover, all these methods treat factuality as a binary concept and fail to provide deeper insights into the kinds of inconsistencies made by different systems. To address these limitations, we devise a typology of factual errors and use it to collect human annotations of generated summaries from state-of-the-art summarizatio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.13346","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.13346/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.13346","created_at":"2026-07-05T03:00:18.067007+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.13346v2","created_at":"2026-07-05T03:00:18.067007+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.13346","created_at":"2026-07-05T03:00:18.067007+00:00"},{"alias_kind":"pith_short_12","alias_value":"6XB3SP7IZH4A","created_at":"2026-07-05T03:00:18.067007+00:00"},{"alias_kind":"pith_short_16","alias_value":"6XB3SP7IZH4AOEEY","created_at":"2026-07-05T03:00:18.067007+00:00"},{"alias_kind":"pith_short_8","alias_value":"6XB3SP7I","created_at":"2026-07-05T03:00:18.067007+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25476","citing_title":"A Red Teaming Framework for Large Language Models: A Case Study on Faithfulness Evaluation","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12807","citing_title":"Detect, Remask, Repair: Diffusion Editing for Faithful Summarization of Evolving Contexts","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2304.13734","citing_title":"The Internal State of an LLM Knows When It's Lying","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":174,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08519","citing_title":"Cram Less to Fit More: Training Data Pruning Improves Memorization of Facts","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20131","citing_title":"Whose Story Gets Told? Positionality and Bias in LLM Summaries of Life Narratives","ref_index":137,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6XB3SP7IZH4AOEEYXXHSBODNUW","json":"https://pith.science/pith/6XB3SP7IZH4AOEEYXXHSBODNUW.json","graph_json":"https://pith.science/api/pith-number/6XB3SP7IZH4AOEEYXXHSBODNUW/graph.json","events_json":"https://pith.science/api/pith-number/6XB3SP7IZH4AOEEYXXHSBODNUW/events.json","paper":"https://pith.science/paper/6XB3SP7I"},"agent_actions":{"view_html":"https://pith.science/pith/6XB3SP7IZH4AOEEYXXHSBODNUW","download_json":"https://pith.science/pith/6XB3SP7IZH4AOEEYXXHSBODNUW.json","view_paper":"https://pith.science/paper/6XB3SP7I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.13346&json=true","fetch_graph":"https://pith.science/api/pith-number/6XB3SP7IZH4AOEEYXXHSBODNUW/graph.json","fetch_events":"https://pith.science/api/pith-number/6XB3SP7IZH4AOEEYXXHSBODNUW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6XB3SP7IZH4AOEEYXXHSBODNUW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6XB3SP7IZH4AOEEYXXHSBODNUW/action/storage_attestation","attest_author":"https://pith.science/pith/6XB3SP7IZH4AOEEYXXHSBODNUW/action/author_attestation","sign_citation":"https://pith.science/pith/6XB3SP7IZH4AOEEYXXHSBODNUW/action/citation_signature","submit_replication":"https://pith.science/pith/6XB3SP7IZH4AOEEYXXHSBODNUW/action/replication_record"}},"created_at":"2026-07-05T03:00:18.067007+00:00","updated_at":"2026-07-05T03:00:18.067007+00:00"}