{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SI3V6ELHPMRDPCRFILDNGWWW5D","short_pith_number":"pith:SI3V6ELH","schema_version":"1.0","canonical_sha256":"92375f11677b22378a2542c6d35ad6e8dcd3e0f968e24b0c15f737beda36d91c","source":{"kind":"arxiv","id":"2407.00908","version":3},"attestation_state":"computed","paper":{"title":"FineSurE: Fine-grained Summarization Evaluation using LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Hang Su, Hwanjun Song, Igor Shalyminov, Jason Cai, Saab Mansour","submitted_at":"2024-07-01T02:20:28Z","abstract_excerpt":"Automated evaluation is crucial for streamlining text summarization benchmarking and model development, given the costly and time-consuming nature of human evaluation. Traditional methods like ROUGE do not correlate well with human judgment, while recently proposed LLM-based metrics provide only summary-level assessment using Likert-scale scores. This limits deeper model analysis, e.g., we can only assign one hallucination score at the summary level, while at the sentence level, we can count sentences containing hallucinations. To remedy those limitations, we propose FineSurE, a fine-grained e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.00908","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-01T02:20:28Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a69933dd739a3d2579dd02cd891f815a40406089dbb63425dca12f328873f0c0","abstract_canon_sha256":"e277a4bd9c69f24f7b512db9e244d67e414cca6c765145d90130fdffa6ffb783"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:46:26.799774Z","signature_b64":"rWTY3ULX5WlPRO/QYoTXTeUYqxORmOlwIDYST3fvBUt/0T3MylBhdb41x+lu9nsI8JDEq4uIqQUDmKVrxWZuCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"92375f11677b22378a2542c6d35ad6e8dcd3e0f968e24b0c15f737beda36d91c","last_reissued_at":"2026-07-05T08:46:26.799190Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:46:26.799190Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FineSurE: Fine-grained Summarization Evaluation using LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Hang Su, Hwanjun Song, Igor Shalyminov, Jason Cai, Saab Mansour","submitted_at":"2024-07-01T02:20:28Z","abstract_excerpt":"Automated evaluation is crucial for streamlining text summarization benchmarking and model development, given the costly and time-consuming nature of human evaluation. Traditional methods like ROUGE do not correlate well with human judgment, while recently proposed LLM-based metrics provide only summary-level assessment using Likert-scale scores. This limits deeper model analysis, e.g., we can only assign one hallucination score at the summary level, while at the sentence level, we can count sentences containing hallucinations. To remedy those limitations, we propose FineSurE, a fine-grained e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.00908","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.00908/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.00908","created_at":"2026-07-05T08:46:26.799265+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.00908v3","created_at":"2026-07-05T08:46:26.799265+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.00908","created_at":"2026-07-05T08:46:26.799265+00:00"},{"alias_kind":"pith_short_12","alias_value":"SI3V6ELHPMRD","created_at":"2026-07-05T08:46:26.799265+00:00"},{"alias_kind":"pith_short_16","alias_value":"SI3V6ELHPMRDPCRF","created_at":"2026-07-05T08:46:26.799265+00:00"},{"alias_kind":"pith_short_8","alias_value":"SI3V6ELH","created_at":"2026-07-05T08:46:26.799265+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":210,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17200","citing_title":"Calibrating Model-Based Evaluation Metrics for Summarization","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17197","citing_title":"Learning to Control Summaries with Score Ranking","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SI3V6ELHPMRDPCRFILDNGWWW5D","json":"https://pith.science/pith/SI3V6ELHPMRDPCRFILDNGWWW5D.json","graph_json":"https://pith.science/api/pith-number/SI3V6ELHPMRDPCRFILDNGWWW5D/graph.json","events_json":"https://pith.science/api/pith-number/SI3V6ELHPMRDPCRFILDNGWWW5D/events.json","paper":"https://pith.science/paper/SI3V6ELH"},"agent_actions":{"view_html":"https://pith.science/pith/SI3V6ELHPMRDPCRFILDNGWWW5D","download_json":"https://pith.science/pith/SI3V6ELHPMRDPCRFILDNGWWW5D.json","view_paper":"https://pith.science/paper/SI3V6ELH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.00908&json=true","fetch_graph":"https://pith.science/api/pith-number/SI3V6ELHPMRDPCRFILDNGWWW5D/graph.json","fetch_events":"https://pith.science/api/pith-number/SI3V6ELHPMRDPCRFILDNGWWW5D/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SI3V6ELHPMRDPCRFILDNGWWW5D/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SI3V6ELHPMRDPCRFILDNGWWW5D/action/storage_attestation","attest_author":"https://pith.science/pith/SI3V6ELHPMRDPCRFILDNGWWW5D/action/author_attestation","sign_citation":"https://pith.science/pith/SI3V6ELHPMRDPCRFILDNGWWW5D/action/citation_signature","submit_replication":"https://pith.science/pith/SI3V6ELHPMRDPCRFILDNGWWW5D/action/replication_record"}},"created_at":"2026-07-05T08:46:26.799265+00:00","updated_at":"2026-07-05T08:46:26.799265+00:00"}