{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3ZQ66ZILB3SPBOI5KWEGX74HIV","short_pith_number":"pith:3ZQ66ZIL","schema_version":"1.0","canonical_sha256":"de61ef650b0ee4f0b91d55886bff87457ac942977712c1346be01e4703d9ee2f","source":{"kind":"arxiv","id":"2406.13009","version":1},"attestation_state":"computed","paper":{"title":"Detecting Errors through Ensembling Prompts (DEEP): An End-to-End LLM Framework for Detecting Factual Errors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alex Chandler, Devesh Surve, Hui Su","submitted_at":"2024-06-18T18:59:37Z","abstract_excerpt":"Accurate text summarization is one of the most common and important tasks performed by Large Language Models, where the costs of human review for an entire document may be high, but the costs of errors in summarization may be even greater. We propose Detecting Errors through Ensembling Prompts (DEEP) - an end-to-end large language model framework for detecting factual errors in text summarization. Our framework uses a diverse set of LLM prompts to identify factual inconsistencies, treating their outputs as binary features, which are then fed into ensembling models. We then calibrate the ensemb"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.13009","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-18T18:59:37Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a081b3fd7974f845010da99474ba9a0e599e35b4eb8a952d36c6a608473adb09","abstract_canon_sha256":"2a1692168bc38aa0bd67096a01857e1369776746ed807dc095843740290e9ecc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:34:23.789605Z","signature_b64":"QM2iETzmqyrKAuczpNCaTXoZDL+0KFpykq1HSxCIZCznddOasIeYDX6g7Me6T3568i7hTNKsvUdHoPcORCJrDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"de61ef650b0ee4f0b91d55886bff87457ac942977712c1346be01e4703d9ee2f","last_reissued_at":"2026-07-05T08:34:23.789186Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:34:23.789186Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Detecting Errors through Ensembling Prompts (DEEP): An End-to-End LLM Framework for Detecting Factual Errors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Alex Chandler, Devesh Surve, Hui Su","submitted_at":"2024-06-18T18:59:37Z","abstract_excerpt":"Accurate text summarization is one of the most common and important tasks performed by Large Language Models, where the costs of human review for an entire document may be high, but the costs of errors in summarization may be even greater. We propose Detecting Errors through Ensembling Prompts (DEEP) - an end-to-end large language model framework for detecting factual errors in text summarization. Our framework uses a diverse set of LLM prompts to identify factual inconsistencies, treating their outputs as binary features, which are then fed into ensembling models. We then calibrate the ensemb"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.13009","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.13009/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.13009","created_at":"2026-07-05T08:34:23.789242+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.13009v1","created_at":"2026-07-05T08:34:23.789242+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.13009","created_at":"2026-07-05T08:34:23.789242+00:00"},{"alias_kind":"pith_short_12","alias_value":"3ZQ66ZILB3SP","created_at":"2026-07-05T08:34:23.789242+00:00"},{"alias_kind":"pith_short_16","alias_value":"3ZQ66ZILB3SPBOI5","created_at":"2026-07-05T08:34:23.789242+00:00"},{"alias_kind":"pith_short_8","alias_value":"3ZQ66ZIL","created_at":"2026-07-05T08:34:23.789242+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.01781","citing_title":"A comprehensive taxonomy of hallucinations in Large Language Models","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3ZQ66ZILB3SPBOI5KWEGX74HIV","json":"https://pith.science/pith/3ZQ66ZILB3SPBOI5KWEGX74HIV.json","graph_json":"https://pith.science/api/pith-number/3ZQ66ZILB3SPBOI5KWEGX74HIV/graph.json","events_json":"https://pith.science/api/pith-number/3ZQ66ZILB3SPBOI5KWEGX74HIV/events.json","paper":"https://pith.science/paper/3ZQ66ZIL"},"agent_actions":{"view_html":"https://pith.science/pith/3ZQ66ZILB3SPBOI5KWEGX74HIV","download_json":"https://pith.science/pith/3ZQ66ZILB3SPBOI5KWEGX74HIV.json","view_paper":"https://pith.science/paper/3ZQ66ZIL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.13009&json=true","fetch_graph":"https://pith.science/api/pith-number/3ZQ66ZILB3SPBOI5KWEGX74HIV/graph.json","fetch_events":"https://pith.science/api/pith-number/3ZQ66ZILB3SPBOI5KWEGX74HIV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3ZQ66ZILB3SPBOI5KWEGX74HIV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3ZQ66ZILB3SPBOI5KWEGX74HIV/action/storage_attestation","attest_author":"https://pith.science/pith/3ZQ66ZILB3SPBOI5KWEGX74HIV/action/author_attestation","sign_citation":"https://pith.science/pith/3ZQ66ZILB3SPBOI5KWEGX74HIV/action/citation_signature","submit_replication":"https://pith.science/pith/3ZQ66ZILB3SPBOI5KWEGX74HIV/action/replication_record"}},"created_at":"2026-07-05T08:34:23.789242+00:00","updated_at":"2026-07-05T08:34:23.789242+00:00"}