{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WMGEVX4OOGHQTEKIH5DVLUONJK","short_pith_number":"pith:WMGEVX4O","schema_version":"1.0","canonical_sha256":"b30c4adf8e718f0991483f4755d1cd4a9f529fadfa79fc353611df484cbcabdf","source":{"kind":"arxiv","id":"2402.01383","version":3},"attestation_state":"computed","paper":{"title":"LLM-based NLG Evaluation: Current Status and Challenges","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jie Ruan, Mingqi Gao, Xiaojun Wan, Xiao Pu, Xinyu Hu","submitted_at":"2024-02-02T13:06:35Z","abstract_excerpt":"Evaluating natural language generation (NLG) is a vital but challenging problem in natural language processing. Traditional evaluation metrics mainly capturing content (e.g. n-gram) overlap between system outputs and references are far from satisfactory, and large language models (LLMs) such as ChatGPT have demonstrated great potential in NLG evaluation in recent years. Various automatic evaluation methods based on LLMs have been proposed, including metrics derived from LLMs, prompting LLMs, fine-tuning LLMs, and human-LLM collaborative evaluation. In this survey, we first give a taxonomy of L"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.01383","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-02T13:06:35Z","cross_cats_sorted":[],"title_canon_sha256":"2ef6bb1e8784c92dd31a2680065eb390932a90df32bba47218db81c993f908f2","abstract_canon_sha256":"4c0e347892f9fdad582523ef048115941899b3fcbe7f65fa47664a1d3cebedd7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:02:37.915944Z","signature_b64":"VsCpMCorrN3RvcWVo1vkr1RmdlW1/Hilk/g0tmOLYeWjKTTbkkeZJbFJOlxjLfkmVGYNYzWcaFmdnqORMxIoAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b30c4adf8e718f0991483f4755d1cd4a9f529fadfa79fc353611df484cbcabdf","last_reissued_at":"2026-07-05T11:02:37.915420Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:02:37.915420Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLM-based NLG Evaluation: Current Status and Challenges","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jie Ruan, Mingqi Gao, Xiaojun Wan, Xiao Pu, Xinyu Hu","submitted_at":"2024-02-02T13:06:35Z","abstract_excerpt":"Evaluating natural language generation (NLG) is a vital but challenging problem in natural language processing. Traditional evaluation metrics mainly capturing content (e.g. n-gram) overlap between system outputs and references are far from satisfactory, and large language models (LLMs) such as ChatGPT have demonstrated great potential in NLG evaluation in recent years. Various automatic evaluation methods based on LLMs have been proposed, including metrics derived from LLMs, prompting LLMs, fine-tuning LLMs, and human-LLM collaborative evaluation. In this survey, we first give a taxonomy of L"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.01383","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.01383/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.01383","created_at":"2026-07-05T11:02:37.915487+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.01383v3","created_at":"2026-07-05T11:02:37.915487+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.01383","created_at":"2026-07-05T11:02:37.915487+00:00"},{"alias_kind":"pith_short_12","alias_value":"WMGEVX4OOGHQ","created_at":"2026-07-05T11:02:37.915487+00:00"},{"alias_kind":"pith_short_16","alias_value":"WMGEVX4OOGHQTEKI","created_at":"2026-07-05T11:02:37.915487+00:00"},{"alias_kind":"pith_short_8","alias_value":"WMGEVX4O","created_at":"2026-07-05T11:02:37.915487+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08009","citing_title":"From Execution to Education: A Bloom-Aligned Framework for Measuring Educational Control in LLMs","ref_index":56,"is_internal_anchor":true},{"citing_arxiv_id":"2606.12767","citing_title":"Constructing Evaluation Datasets for Procedural Reasoning: Balancing Naturalness, Grounding, and Multi-Hop Coverage","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05901","citing_title":"Reducing Hallucinations in Complex Question Answering using Simple Graph-based Retrieval-Augmented Generation (long version)","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2410.20791","citing_title":"From Cool Demos to Production-Ready FMware: Core Challenges and a Technology Roadmap","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2503.23137","citing_title":"When 'YES' Meets 'BUT': Can Large Models Comprehend Contradictory Humor Through Comparative Reasoning?","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2505.19237","citing_title":"Sensorimotor Self-Recognition in Multimodal Large Language Model-Driven Robots","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WMGEVX4OOGHQTEKIH5DVLUONJK","json":"https://pith.science/pith/WMGEVX4OOGHQTEKIH5DVLUONJK.json","graph_json":"https://pith.science/api/pith-number/WMGEVX4OOGHQTEKIH5DVLUONJK/graph.json","events_json":"https://pith.science/api/pith-number/WMGEVX4OOGHQTEKIH5DVLUONJK/events.json","paper":"https://pith.science/paper/WMGEVX4O"},"agent_actions":{"view_html":"https://pith.science/pith/WMGEVX4OOGHQTEKIH5DVLUONJK","download_json":"https://pith.science/pith/WMGEVX4OOGHQTEKIH5DVLUONJK.json","view_paper":"https://pith.science/paper/WMGEVX4O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.01383&json=true","fetch_graph":"https://pith.science/api/pith-number/WMGEVX4OOGHQTEKIH5DVLUONJK/graph.json","fetch_events":"https://pith.science/api/pith-number/WMGEVX4OOGHQTEKIH5DVLUONJK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WMGEVX4OOGHQTEKIH5DVLUONJK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WMGEVX4OOGHQTEKIH5DVLUONJK/action/storage_attestation","attest_author":"https://pith.science/pith/WMGEVX4OOGHQTEKIH5DVLUONJK/action/author_attestation","sign_citation":"https://pith.science/pith/WMGEVX4OOGHQTEKIH5DVLUONJK/action/citation_signature","submit_replication":"https://pith.science/pith/WMGEVX4OOGHQTEKIH5DVLUONJK/action/replication_record"}},"created_at":"2026-07-05T11:02:37.915487+00:00","updated_at":"2026-07-05T11:02:37.915487+00:00"}