{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QE7QABGTQAJ6JCZ7CKPITMI55J","short_pith_number":"pith:QE7QABGT","schema_version":"1.0","canonical_sha256":"813f0004d38013e48b3f129e89b11dea4e48d6896c573e357570e9aedbffb36e","source":{"kind":"arxiv","id":"2303.15621","version":2},"attestation_state":"computed","paper":{"title":"ChatGPT as a Factual Inconsistency Evaluator for Text Summarization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Qianqian Xie, Sophia Ananiadou, Zheheng Luo","submitted_at":"2023-03-27T22:30:39Z","abstract_excerpt":"The performance of text summarization has been greatly boosted by pre-trained language models. A main concern of existing methods is that most generated summaries are not factually inconsistent with their source documents. To alleviate the problem, many efforts have focused on developing effective factuality evaluation metrics based on natural language inference, question answering, and syntactic dependency et al. However, these approaches are limited by either their high computational complexity or the uncertainty introduced by multi-component pipelines, resulting in only partial agreement wi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.15621","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-03-27T22:30:39Z","cross_cats_sorted":[],"title_canon_sha256":"6f05b657974e3d34f4f03ab3837a7c510fcd3e41a0b00e862b703cd46db14a79","abstract_canon_sha256":"beb5b45abfa174555bd53147bea11622b9df73d4f6e083b202fc168521841e44"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:00:36.952779Z","signature_b64":"KpSjrC/407vlF5Byo2xlxQXMGJHv8oxE378cGMSZDm1gZJTzXTSfO239YLe0T1XZhKq1ovG1EniETjsSWFcABg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"813f0004d38013e48b3f129e89b11dea4e48d6896c573e357570e9aedbffb36e","last_reissued_at":"2026-07-05T06:00:36.952371Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:00:36.952371Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ChatGPT as a Factual Inconsistency Evaluator for Text Summarization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Qianqian Xie, Sophia Ananiadou, Zheheng Luo","submitted_at":"2023-03-27T22:30:39Z","abstract_excerpt":"The performance of text summarization has been greatly boosted by pre-trained language models. A main concern of existing methods is that most generated summaries are not factually inconsistent with their source documents. To alleviate the problem, many efforts have focused on developing effective factuality evaluation metrics based on natural language inference, question answering, and syntactic dependency et al. However, these approaches are limited by either their high computational complexity or the uncertainty introduced by multi-component pipelines, resulting in only partial agreement wi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.15621","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.15621/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.15621","created_at":"2026-07-05T06:00:36.952428+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.15621v2","created_at":"2026-07-05T06:00:36.952428+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.15621","created_at":"2026-07-05T06:00:36.952428+00:00"},{"alias_kind":"pith_short_12","alias_value":"QE7QABGTQAJ6","created_at":"2026-07-05T06:00:36.952428+00:00"},{"alias_kind":"pith_short_16","alias_value":"QE7QABGTQAJ6JCZ7","created_at":"2026-07-05T06:00:36.952428+00:00"},{"alias_kind":"pith_short_8","alias_value":"QE7QABGT","created_at":"2026-07-05T06:00:36.952428+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11762","citing_title":"Automated Creativity Evaluation of Language Models Across Open-Ended Tasks","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26840","citing_title":"Optimising Factual Consistency in Summarisation via Preference Learning from Multiple Imperfect Metrics","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29712","citing_title":"Teaching Language Models to Check Grounded Claim Factuality with Human Test-Taking Strategies","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2406.15809","citing_title":"LaMSUM: Amplifying Voices Against Harassment through LLM Guided Extractive Summarization of User Incident Reports","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2408.11871","citing_title":"MegaFake: A Theory-Driven Dataset of Fake News Generated by Large Language Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2508.18473","citing_title":"Principled Detection of Hallucinations in Large Language Models via Multiple Testing","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2303.08896","citing_title":"SelfCheckGPT: Zero-Resource Black-Box Hallucination Detection for Generative Large Language Models","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2406.06608","citing_title":"The Prompt Report: A Systematic Survey of Prompt Engineering Techniques","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25665","citing_title":"LLM-ReSum: A Framework for LLM Reflective Summarization through Self-Evaluation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20131","citing_title":"Whose Story Gets Told? Positionality and Bias in LLM Summaries of Life Narratives","ref_index":145,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QE7QABGTQAJ6JCZ7CKPITMI55J","json":"https://pith.science/pith/QE7QABGTQAJ6JCZ7CKPITMI55J.json","graph_json":"https://pith.science/api/pith-number/QE7QABGTQAJ6JCZ7CKPITMI55J/graph.json","events_json":"https://pith.science/api/pith-number/QE7QABGTQAJ6JCZ7CKPITMI55J/events.json","paper":"https://pith.science/paper/QE7QABGT"},"agent_actions":{"view_html":"https://pith.science/pith/QE7QABGTQAJ6JCZ7CKPITMI55J","download_json":"https://pith.science/pith/QE7QABGTQAJ6JCZ7CKPITMI55J.json","view_paper":"https://pith.science/paper/QE7QABGT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.15621&json=true","fetch_graph":"https://pith.science/api/pith-number/QE7QABGTQAJ6JCZ7CKPITMI55J/graph.json","fetch_events":"https://pith.science/api/pith-number/QE7QABGTQAJ6JCZ7CKPITMI55J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QE7QABGTQAJ6JCZ7CKPITMI55J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QE7QABGTQAJ6JCZ7CKPITMI55J/action/storage_attestation","attest_author":"https://pith.science/pith/QE7QABGTQAJ6JCZ7CKPITMI55J/action/author_attestation","sign_citation":"https://pith.science/pith/QE7QABGTQAJ6JCZ7CKPITMI55J/action/citation_signature","submit_replication":"https://pith.science/pith/QE7QABGTQAJ6JCZ7CKPITMI55J/action/replication_record"}},"created_at":"2026-07-05T06:00:36.952428+00:00","updated_at":"2026-07-05T06:00:36.952428+00:00"}