{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KIGGXE2WS5UTX5J7ZDIFOTMBH5","short_pith_number":"pith:KIGGXE2W","schema_version":"1.0","canonical_sha256":"520c6b935697693bf53fc8d0574d813f426adede1cb8e373692fc0a160c184ee","source":{"kind":"arxiv","id":"2304.10703","version":2},"attestation_state":"computed","paper":{"title":"ReCEval: Evaluating Reasoning Chains via Correctness and Informativeness","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Archiki Prasad, Mohit Bansal, Swarnadeep Saha, Xiang Zhou","submitted_at":"2023-04-21T02:19:06Z","abstract_excerpt":"Multi-step reasoning ability is fundamental to many natural language tasks, yet it is unclear what constitutes a good reasoning chain and how to evaluate them. Most existing methods focus solely on whether the reasoning chain leads to the correct conclusion, but this answer-oriented view may confound reasoning quality with other spurious shortcuts to predict the answer. To bridge this gap, we evaluate reasoning chains by viewing them as informal proofs that derive the final answer. Specifically, we propose ReCEval (Reasoning Chain Evaluation), a framework that evaluates reasoning chains via tw"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.10703","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-04-21T02:19:06Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"0df571846e81aff6ab51116115a7476ec86ad8b4322728db23655c850cfdc640","abstract_canon_sha256":"11f23fce1a240c761a8f038ac99d4aee4598c4c88fda4d05b76d51f1b36f0742"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:18:57.327340Z","signature_b64":"B4TRG3z9TOYHR0iu0uUiU13J+0n+m2MnjUX5NoV4l0UKMCz6isaGuWFtgGU0RoLzRc2UbOikvBL41JvAz+9OCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"520c6b935697693bf53fc8d0574d813f426adede1cb8e373692fc0a160c184ee","last_reissued_at":"2026-07-05T07:18:57.326841Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:18:57.326841Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReCEval: Evaluating Reasoning Chains via Correctness and Informativeness","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Archiki Prasad, Mohit Bansal, Swarnadeep Saha, Xiang Zhou","submitted_at":"2023-04-21T02:19:06Z","abstract_excerpt":"Multi-step reasoning ability is fundamental to many natural language tasks, yet it is unclear what constitutes a good reasoning chain and how to evaluate them. Most existing methods focus solely on whether the reasoning chain leads to the correct conclusion, but this answer-oriented view may confound reasoning quality with other spurious shortcuts to predict the answer. To bridge this gap, we evaluate reasoning chains by viewing them as informal proofs that derive the final answer. Specifically, we propose ReCEval (Reasoning Chain Evaluation), a framework that evaluates reasoning chains via tw"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.10703","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.10703/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.10703","created_at":"2026-07-05T07:18:57.326903+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.10703v2","created_at":"2026-07-05T07:18:57.326903+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.10703","created_at":"2026-07-05T07:18:57.326903+00:00"},{"alias_kind":"pith_short_12","alias_value":"KIGGXE2WS5UT","created_at":"2026-07-05T07:18:57.326903+00:00"},{"alias_kind":"pith_short_16","alias_value":"KIGGXE2WS5UTX5J7","created_at":"2026-07-05T07:18:57.326903+00:00"},{"alias_kind":"pith_short_8","alias_value":"KIGGXE2W","created_at":"2026-07-05T07:18:57.326903+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10279","citing_title":"Supervised Fine-tuning with Synthetic Rationale Data Hurts Real-World Disease Prediction","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2603.03332","citing_title":"Fragile Thoughts: How Large Language Models Handle Chain-of-Thought Perturbations","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12379","citing_title":"Beyond Output Correctness: Benchmarking and Evaluating Large Language Model Reasoning in Coding Tasks","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KIGGXE2WS5UTX5J7ZDIFOTMBH5","json":"https://pith.science/pith/KIGGXE2WS5UTX5J7ZDIFOTMBH5.json","graph_json":"https://pith.science/api/pith-number/KIGGXE2WS5UTX5J7ZDIFOTMBH5/graph.json","events_json":"https://pith.science/api/pith-number/KIGGXE2WS5UTX5J7ZDIFOTMBH5/events.json","paper":"https://pith.science/paper/KIGGXE2W"},"agent_actions":{"view_html":"https://pith.science/pith/KIGGXE2WS5UTX5J7ZDIFOTMBH5","download_json":"https://pith.science/pith/KIGGXE2WS5UTX5J7ZDIFOTMBH5.json","view_paper":"https://pith.science/paper/KIGGXE2W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.10703&json=true","fetch_graph":"https://pith.science/api/pith-number/KIGGXE2WS5UTX5J7ZDIFOTMBH5/graph.json","fetch_events":"https://pith.science/api/pith-number/KIGGXE2WS5UTX5J7ZDIFOTMBH5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KIGGXE2WS5UTX5J7ZDIFOTMBH5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KIGGXE2WS5UTX5J7ZDIFOTMBH5/action/storage_attestation","attest_author":"https://pith.science/pith/KIGGXE2WS5UTX5J7ZDIFOTMBH5/action/author_attestation","sign_citation":"https://pith.science/pith/KIGGXE2WS5UTX5J7ZDIFOTMBH5/action/citation_signature","submit_replication":"https://pith.science/pith/KIGGXE2WS5UTX5J7ZDIFOTMBH5/action/replication_record"}},"created_at":"2026-07-05T07:18:57.326903+00:00","updated_at":"2026-07-05T07:18:57.326903+00:00"}