{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AA4FMG4RXCJ6YXBTZK7OVR4LFU","short_pith_number":"pith:AA4FMG4R","schema_version":"1.0","canonical_sha256":"0038561b91b893ec5c33cabeeac78b2d081b0fccd0800e4f4894a6bc40e22e1a","source":{"kind":"arxiv","id":"2404.03602","version":2},"attestation_state":"computed","paper":{"title":"Evaluating LLMs at Detecting Errors in LLM Responses","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Arman Cohan, Jihyun Janice Ahn, Nan Zhang, Ranran Haoran Zhang, Renze Lou, Rui Zhang, Ryo Kamoi, Salika Dave, Sarkar Snigdha Sarathi Das, Shaobo Qin, Sujeeth Reddy Vummanthala, Wenpeng Yin, Xiaoxin Lu, Yilun Zhao, Yusen Zhang","submitted_at":"2024-04-04T17:19:47Z","abstract_excerpt":"With Large Language Models (LLMs) being widely used across various tasks, detecting errors in their responses is increasingly crucial. However, little research has been conducted on error detection of LLM responses. Collecting error annotations on LLM responses is challenging due to the subjective nature of many NLP tasks, and thus previous research focuses on tasks of little practical value (e.g., word sorting) or limited error types (e.g., faithfulness in summarization). This work introduces ReaLMistake, the first error detection benchmark consisting of objective, realistic, and diverse erro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.03602","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-04T17:19:47Z","cross_cats_sorted":[],"title_canon_sha256":"8457f26997419f545c95a8664ea98543a123ce6337159ec4b1a249d3a93406a4","abstract_canon_sha256":"a56c3ca30096ab2b619642fe128779b74fe8309f6a3d96c6f6c78066799c925e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:49:32.641327Z","signature_b64":"4DxhmIe0ffY7SE6qQJXGzMqeUtInl6VwKecsT8UWRXAiO4pb65+cfPON8RystSp8LhZ4VBtmP1cTYk6C+yNRCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0038561b91b893ec5c33cabeeac78b2d081b0fccd0800e4f4894a6bc40e22e1a","last_reissued_at":"2026-07-05T08:49:32.640838Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:49:32.640838Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating LLMs at Detecting Errors in LLM Responses","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Arman Cohan, Jihyun Janice Ahn, Nan Zhang, Ranran Haoran Zhang, Renze Lou, Rui Zhang, Ryo Kamoi, Salika Dave, Sarkar Snigdha Sarathi Das, Shaobo Qin, Sujeeth Reddy Vummanthala, Wenpeng Yin, Xiaoxin Lu, Yilun Zhao, Yusen Zhang","submitted_at":"2024-04-04T17:19:47Z","abstract_excerpt":"With Large Language Models (LLMs) being widely used across various tasks, detecting errors in their responses is increasingly crucial. However, little research has been conducted on error detection of LLM responses. Collecting error annotations on LLM responses is challenging due to the subjective nature of many NLP tasks, and thus previous research focuses on tasks of little practical value (e.g., word sorting) or limited error types (e.g., faithfulness in summarization). This work introduces ReaLMistake, the first error detection benchmark consisting of objective, realistic, and diverse erro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.03602","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.03602/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.03602","created_at":"2026-07-05T08:49:32.640897+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.03602v2","created_at":"2026-07-05T08:49:32.640897+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.03602","created_at":"2026-07-05T08:49:32.640897+00:00"},{"alias_kind":"pith_short_12","alias_value":"AA4FMG4RXCJ6","created_at":"2026-07-05T08:49:32.640897+00:00"},{"alias_kind":"pith_short_16","alias_value":"AA4FMG4RXCJ6YXBT","created_at":"2026-07-05T08:49:32.640897+00:00"},{"alias_kind":"pith_short_8","alias_value":"AA4FMG4R","created_at":"2026-07-05T08:49:32.640897+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10315","citing_title":"Catching One in Five: LLM-as-Judge Blind Spots in Production Multi-Turn Transaction Agents","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2410.20791","citing_title":"From Cool Demos to Production-Ready FMware: Core Challenges and a Technology Roadmap","ref_index":62,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AA4FMG4RXCJ6YXBTZK7OVR4LFU","json":"https://pith.science/pith/AA4FMG4RXCJ6YXBTZK7OVR4LFU.json","graph_json":"https://pith.science/api/pith-number/AA4FMG4RXCJ6YXBTZK7OVR4LFU/graph.json","events_json":"https://pith.science/api/pith-number/AA4FMG4RXCJ6YXBTZK7OVR4LFU/events.json","paper":"https://pith.science/paper/AA4FMG4R"},"agent_actions":{"view_html":"https://pith.science/pith/AA4FMG4RXCJ6YXBTZK7OVR4LFU","download_json":"https://pith.science/pith/AA4FMG4RXCJ6YXBTZK7OVR4LFU.json","view_paper":"https://pith.science/paper/AA4FMG4R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.03602&json=true","fetch_graph":"https://pith.science/api/pith-number/AA4FMG4RXCJ6YXBTZK7OVR4LFU/graph.json","fetch_events":"https://pith.science/api/pith-number/AA4FMG4RXCJ6YXBTZK7OVR4LFU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AA4FMG4RXCJ6YXBTZK7OVR4LFU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AA4FMG4RXCJ6YXBTZK7OVR4LFU/action/storage_attestation","attest_author":"https://pith.science/pith/AA4FMG4RXCJ6YXBTZK7OVR4LFU/action/author_attestation","sign_citation":"https://pith.science/pith/AA4FMG4RXCJ6YXBTZK7OVR4LFU/action/citation_signature","submit_replication":"https://pith.science/pith/AA4FMG4RXCJ6YXBTZK7OVR4LFU/action/replication_record"}},"created_at":"2026-07-05T08:49:32.640897+00:00","updated_at":"2026-07-05T08:49:32.640897+00:00"}