{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:E232SSOK6EL3SQQR6H6PO4EH4P","short_pith_number":"pith:E232SSOK","schema_version":"1.0","canonical_sha256":"26b7a949caf117b94211f1fcf77087e3e1aafe0c1b2bf966e7bd4a814ff07507","source":{"kind":"arxiv","id":"2307.06869","version":1},"attestation_state":"computed","paper":{"title":"DecompEval: Evaluating Generated Texts as Unsupervised Decomposed Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Fei Huang, Fei Mi, Minlie Huang, Pei Ke, Qun Liu, Xiaoyan Zhu, Yasheng Wang","submitted_at":"2023-07-13T16:16:51Z","abstract_excerpt":"Existing evaluation metrics for natural language generation (NLG) tasks face the challenges on generalization ability and interpretability. Specifically, most of the well-performed metrics are required to train on evaluation datasets of specific NLG tasks and evaluation dimensions, which may cause over-fitting to task-specific datasets. Furthermore, existing metrics only provide an evaluation score for each dimension without revealing the evidence to interpret how this score is obtained. To deal with these challenges, we propose a simple yet effective metric called DecompEval. This metric form"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.06869","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-07-13T16:16:51Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"78635a96a7923a0a2db6c1e91d0c846a25addbf2ca5b4451fb2f3afa96b81fe5","abstract_canon_sha256":"99cffbd1769e104fc475c60ce0bdc8858dfef5b0a72cb0233b467aa746fc749a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:30:38.970764Z","signature_b64":"R/2vdjCY1+Te8SlyP4IPWZYBKaxxXs1S9BYnIBEMFMLyB4cbfSlVVhNWCAOgoDMUCEwgimHXxjVzbYNvcdRhDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"26b7a949caf117b94211f1fcf77087e3e1aafe0c1b2bf966e7bd4a814ff07507","last_reissued_at":"2026-07-05T06:30:38.970406Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:30:38.970406Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DecompEval: Evaluating Generated Texts as Unsupervised Decomposed Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Fei Huang, Fei Mi, Minlie Huang, Pei Ke, Qun Liu, Xiaoyan Zhu, Yasheng Wang","submitted_at":"2023-07-13T16:16:51Z","abstract_excerpt":"Existing evaluation metrics for natural language generation (NLG) tasks face the challenges on generalization ability and interpretability. Specifically, most of the well-performed metrics are required to train on evaluation datasets of specific NLG tasks and evaluation dimensions, which may cause over-fitting to task-specific datasets. Furthermore, existing metrics only provide an evaluation score for each dimension without revealing the evidence to interpret how this score is obtained. To deal with these challenges, we propose a simple yet effective metric called DecompEval. This metric form"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.06869","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.06869/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.06869","created_at":"2026-07-05T06:30:38.970463+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.06869v1","created_at":"2026-07-05T06:30:38.970463+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.06869","created_at":"2026-07-05T06:30:38.970463+00:00"},{"alias_kind":"pith_short_12","alias_value":"E232SSOK6EL3","created_at":"2026-07-05T06:30:38.970463+00:00"},{"alias_kind":"pith_short_16","alias_value":"E232SSOK6EL3SQQR","created_at":"2026-07-05T06:30:38.970463+00:00"},{"alias_kind":"pith_short_8","alias_value":"E232SSOK","created_at":"2026-07-05T06:30:38.970463+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E232SSOK6EL3SQQR6H6PO4EH4P","json":"https://pith.science/pith/E232SSOK6EL3SQQR6H6PO4EH4P.json","graph_json":"https://pith.science/api/pith-number/E232SSOK6EL3SQQR6H6PO4EH4P/graph.json","events_json":"https://pith.science/api/pith-number/E232SSOK6EL3SQQR6H6PO4EH4P/events.json","paper":"https://pith.science/paper/E232SSOK"},"agent_actions":{"view_html":"https://pith.science/pith/E232SSOK6EL3SQQR6H6PO4EH4P","download_json":"https://pith.science/pith/E232SSOK6EL3SQQR6H6PO4EH4P.json","view_paper":"https://pith.science/paper/E232SSOK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.06869&json=true","fetch_graph":"https://pith.science/api/pith-number/E232SSOK6EL3SQQR6H6PO4EH4P/graph.json","fetch_events":"https://pith.science/api/pith-number/E232SSOK6EL3SQQR6H6PO4EH4P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E232SSOK6EL3SQQR6H6PO4EH4P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E232SSOK6EL3SQQR6H6PO4EH4P/action/storage_attestation","attest_author":"https://pith.science/pith/E232SSOK6EL3SQQR6H6PO4EH4P/action/author_attestation","sign_citation":"https://pith.science/pith/E232SSOK6EL3SQQR6H6PO4EH4P/action/citation_signature","submit_replication":"https://pith.science/pith/E232SSOK6EL3SQQR6H6PO4EH4P/action/replication_record"}},"created_at":"2026-07-05T06:30:38.970463+00:00","updated_at":"2026-07-05T06:30:38.970463+00:00"}