{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZEPAUJ2RLIVG2IWBA3U3ARPS34","short_pith_number":"pith:ZEPAUJ2R","schema_version":"1.0","canonical_sha256":"c91e0a27515a2a6d22c106e9b045f2df39d030129a449561603cfae13b897c88","source":{"kind":"arxiv","id":"2308.04386","version":3},"attestation_state":"computed","paper":{"title":"Learning Evaluation Models from Large Language Models for Sequence Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chenglong Wang, Chunliang Zhang, Hang Zhou, Jingbo Zhu, Kaiyan Chang, Quan Du, Tongran Liu, Tong Xiao, Yue Zhang","submitted_at":"2023-08-08T16:41:16Z","abstract_excerpt":"Automatic evaluation of sequence generation, traditionally reliant on metrics like BLEU and ROUGE, often fails to capture the semantic accuracy of generated text sequences due to their emphasis on n-gram overlap. A promising solution to this problem is to develop model-based metrics, such as BLEURT and COMET. However, these approaches are typically hindered by the scarcity of labeled evaluation data, which is necessary to train the evaluation models. In this work, we build upon this challenge by proposing the Customized Sequence Evaluation Metric (CSEM), a three-stage evaluation model training"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.04386","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-08-08T16:41:16Z","cross_cats_sorted":[],"title_canon_sha256":"2ece5d1de36985b4aa16656503425a506605815e546947ef71d93c51f9c69d76","abstract_canon_sha256":"3063fb45b4b9015452c22c1f8db16fe407e2d8225b87af6383f3f88e929626d8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:17.388528Z","signature_b64":"f1oLuHdRKfx599hFT1IL2H48BwkQRfWUxunKliW5mu+/ewBWNsr1BBYZgvN6I0xI1LZE7qgf6npsMVI1G47hCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c91e0a27515a2a6d22c106e9b045f2df39d030129a449561603cfae13b897c88","last_reissued_at":"2026-07-05T11:27:17.388012Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:17.388012Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Evaluation Models from Large Language Models for Sequence Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chenglong Wang, Chunliang Zhang, Hang Zhou, Jingbo Zhu, Kaiyan Chang, Quan Du, Tongran Liu, Tong Xiao, Yue Zhang","submitted_at":"2023-08-08T16:41:16Z","abstract_excerpt":"Automatic evaluation of sequence generation, traditionally reliant on metrics like BLEU and ROUGE, often fails to capture the semantic accuracy of generated text sequences due to their emphasis on n-gram overlap. A promising solution to this problem is to develop model-based metrics, such as BLEURT and COMET. However, these approaches are typically hindered by the scarcity of labeled evaluation data, which is necessary to train the evaluation models. In this work, we build upon this challenge by proposing the Customized Sequence Evaluation Metric (CSEM), a three-stage evaluation model training"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.04386","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.04386/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.04386","created_at":"2026-07-05T11:27:17.388073+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.04386v3","created_at":"2026-07-05T11:27:17.388073+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.04386","created_at":"2026-07-05T11:27:17.388073+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZEPAUJ2RLIVG","created_at":"2026-07-05T11:27:17.388073+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZEPAUJ2RLIVG2IWB","created_at":"2026-07-05T11:27:17.388073+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZEPAUJ2R","created_at":"2026-07-05T11:27:17.388073+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.14092","citing_title":"Fragile Preferences: A Deep Dive Into Order Effects in Large Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":237,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZEPAUJ2RLIVG2IWBA3U3ARPS34","json":"https://pith.science/pith/ZEPAUJ2RLIVG2IWBA3U3ARPS34.json","graph_json":"https://pith.science/api/pith-number/ZEPAUJ2RLIVG2IWBA3U3ARPS34/graph.json","events_json":"https://pith.science/api/pith-number/ZEPAUJ2RLIVG2IWBA3U3ARPS34/events.json","paper":"https://pith.science/paper/ZEPAUJ2R"},"agent_actions":{"view_html":"https://pith.science/pith/ZEPAUJ2RLIVG2IWBA3U3ARPS34","download_json":"https://pith.science/pith/ZEPAUJ2RLIVG2IWBA3U3ARPS34.json","view_paper":"https://pith.science/paper/ZEPAUJ2R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.04386&json=true","fetch_graph":"https://pith.science/api/pith-number/ZEPAUJ2RLIVG2IWBA3U3ARPS34/graph.json","fetch_events":"https://pith.science/api/pith-number/ZEPAUJ2RLIVG2IWBA3U3ARPS34/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZEPAUJ2RLIVG2IWBA3U3ARPS34/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZEPAUJ2RLIVG2IWBA3U3ARPS34/action/storage_attestation","attest_author":"https://pith.science/pith/ZEPAUJ2RLIVG2IWBA3U3ARPS34/action/author_attestation","sign_citation":"https://pith.science/pith/ZEPAUJ2RLIVG2IWBA3U3ARPS34/action/citation_signature","submit_replication":"https://pith.science/pith/ZEPAUJ2RLIVG2IWBA3U3ARPS34/action/replication_record"}},"created_at":"2026-07-05T11:27:17.388073+00:00","updated_at":"2026-07-05T11:27:17.388073+00:00"}