{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BCH3HEQONHFVQTUVFAHJPI3SST","short_pith_number":"pith:BCH3HEQO","schema_version":"1.0","canonical_sha256":"088fb3920e69cb584e95280e97a37294deeb9da140585d0ceb837355df4c8373","source":{"kind":"arxiv","id":"2303.04048","version":3},"attestation_state":"computed","paper":{"title":"Is ChatGPT a Good NLG Evaluator? A Preliminary Study","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Fandong Meng, Haoxiang Shi, Jiaan Wang, Jianfeng Qu, Jie Zhou, Jinan Xu, Yunlong Liang, Zengkui Sun, Zhixu Li","submitted_at":"2023-03-07T16:57:20Z","abstract_excerpt":"Recently, the emergence of ChatGPT has attracted wide attention from the computational linguistics community. Many prior studies have shown that ChatGPT achieves remarkable performance on various NLP tasks in terms of automatic evaluation metrics. However, the ability of ChatGPT to serve as an evaluation metric is still underexplored. Considering assessing the quality of natural language generation (NLG) models is an arduous task and NLG metrics notoriously show their poor correlation with human judgments, we wonder whether ChatGPT is a good NLG evaluation metric. In this report, we provide a "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.04048","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-03-07T16:57:20Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"30284a6e126991735971950799ec18d6a4c31dd351f5cbe9ab99fa5f21580b3b","abstract_canon_sha256":"805073f6d46d07b481b0b7c5dd90b52ee92c8d92985d66317bbb1fd02f57f9b3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:04:13.783672Z","signature_b64":"NZ5iG1A9PmMZPOxsFcI2yYv5yu949X9SCIVBQTkYQcPXopcY8ziD/MSZJ4fF/aoYoN9KcdhUrnZ07vfxRCdOCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"088fb3920e69cb584e95280e97a37294deeb9da140585d0ceb837355df4c8373","last_reissued_at":"2026-07-05T07:04:13.783101Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:04:13.783101Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Is ChatGPT a Good NLG Evaluator? A Preliminary Study","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Fandong Meng, Haoxiang Shi, Jiaan Wang, Jianfeng Qu, Jie Zhou, Jinan Xu, Yunlong Liang, Zengkui Sun, Zhixu Li","submitted_at":"2023-03-07T16:57:20Z","abstract_excerpt":"Recently, the emergence of ChatGPT has attracted wide attention from the computational linguistics community. Many prior studies have shown that ChatGPT achieves remarkable performance on various NLP tasks in terms of automatic evaluation metrics. However, the ability of ChatGPT to serve as an evaluation metric is still underexplored. Considering assessing the quality of natural language generation (NLG) models is an arduous task and NLG metrics notoriously show their poor correlation with human judgments, we wonder whether ChatGPT is a good NLG evaluation metric. In this report, we provide a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.04048","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.04048/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.04048","created_at":"2026-07-05T07:04:13.783160+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.04048v3","created_at":"2026-07-05T07:04:13.783160+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.04048","created_at":"2026-07-05T07:04:13.783160+00:00"},{"alias_kind":"pith_short_12","alias_value":"BCH3HEQONHFV","created_at":"2026-07-05T07:04:13.783160+00:00"},{"alias_kind":"pith_short_16","alias_value":"BCH3HEQONHFVQTUV","created_at":"2026-07-05T07:04:13.783160+00:00"},{"alias_kind":"pith_short_8","alias_value":"BCH3HEQO","created_at":"2026-07-05T07:04:13.783160+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.31080","citing_title":"A Pilot Study on Curator-Guided Multilingual Art Description for Blind and Low-Vision Audiences with Small Vision-Language Models","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2411.01141","citing_title":"Dictionary Insertion Prompting for Multilingual Reasoning on Multilingual Large Language Models","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2503.04338","citing_title":"In-depth Analysis of Graph-based RAG in a Unified Framework","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2507.18902","citing_title":"SLoW: Select Low-frequency Words! Automatic Dictionary Selection for Translation on Large Language Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19180","citing_title":"Supporting System Testing with a Multi-Agent LLM-based Framework for Knowledge Graph Extraction: A Case Study with Ethernet Switch Systems","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2603.10477","citing_title":"PEEM: Prompt Engineering Evaluation Metrics for Interpretable Joint Evaluation of Prompts and Responses","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2309.10253","citing_title":"GPTFUZZER: Red Teaming Large Language Models with Auto-Generated Jailbreak Prompts","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03666","citing_title":"MMP-Refer: Multimodal Path Retrieval-augmented LLMs For Explainable Recommendation","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2308.07201","citing_title":"ChatEval: Towards Better LLM-based Evaluators through Multi-Agent Debate","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2303.16634","citing_title":"G-Eval: NLG Evaluation using GPT-4 with Better Human Alignment","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16130","citing_title":"From Local to Global: A Graph RAG Approach to Query-Focused Summarization","ref_index":65,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BCH3HEQONHFVQTUVFAHJPI3SST","json":"https://pith.science/pith/BCH3HEQONHFVQTUVFAHJPI3SST.json","graph_json":"https://pith.science/api/pith-number/BCH3HEQONHFVQTUVFAHJPI3SST/graph.json","events_json":"https://pith.science/api/pith-number/BCH3HEQONHFVQTUVFAHJPI3SST/events.json","paper":"https://pith.science/paper/BCH3HEQO"},"agent_actions":{"view_html":"https://pith.science/pith/BCH3HEQONHFVQTUVFAHJPI3SST","download_json":"https://pith.science/pith/BCH3HEQONHFVQTUVFAHJPI3SST.json","view_paper":"https://pith.science/paper/BCH3HEQO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.04048&json=true","fetch_graph":"https://pith.science/api/pith-number/BCH3HEQONHFVQTUVFAHJPI3SST/graph.json","fetch_events":"https://pith.science/api/pith-number/BCH3HEQONHFVQTUVFAHJPI3SST/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BCH3HEQONHFVQTUVFAHJPI3SST/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BCH3HEQONHFVQTUVFAHJPI3SST/action/storage_attestation","attest_author":"https://pith.science/pith/BCH3HEQONHFVQTUVFAHJPI3SST/action/author_attestation","sign_citation":"https://pith.science/pith/BCH3HEQONHFVQTUVFAHJPI3SST/action/citation_signature","submit_replication":"https://pith.science/pith/BCH3HEQONHFVQTUVFAHJPI3SST/action/replication_record"}},"created_at":"2026-07-05T07:04:13.783160+00:00","updated_at":"2026-07-05T07:04:13.783160+00:00"}