{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DAUH7VNWBFYTBLZM332ZXDCIEG","short_pith_number":"pith:DAUH7VNW","schema_version":"1.0","canonical_sha256":"18287fd5b6097130af2cdef59b8c4821898f90ae49bf7ca7f92b15bf1b330706","source":{"kind":"arxiv","id":"2410.00151","version":4},"attestation_state":"computed","paper":{"title":"Scheherazade: Evaluating Chain-of-Thought Math Reasoning in LLMs with Chain-of-Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ferhat Erata, Ruzica Piskac, Sam Kouteili, Scott J Shapiro, Simeng Han, Stephen Miner, Yoshiki Takashima","submitted_at":"2024-09-30T18:48:34Z","abstract_excerpt":"Benchmarks are critical for measuring Large Language Model (LLM) reasoning capabilities. Some benchmarks have even become the de facto indicator of such capabilities. However, as LLM reasoning capabilities improve, existing widely-used benchmarks such as GSM8K marginally encapsulate model reasoning differentials - most state-of-the-art models for example achieve over 94% accuracy on the GSM8K dataset (paperwithcode, 2024). While constructing harder benchmarks is possible, their creation is often manual, expensive, and unscalable. As such, we present Scheherazade, an automated approach to produ"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.00151","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-09-30T18:48:34Z","cross_cats_sorted":[],"title_canon_sha256":"381349587167bb05872261578d64dab5b28097c61442a12427a308a541dccde9","abstract_canon_sha256":"ed0b2cc14966ce0e48b5a9012e487585b1c9c6c1a47a59089d71caaf806adb03"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:29.059215Z","signature_b64":"/ZErfRSoGKIEkE/9Vlfa8zkRx0IYkyQXkCqooIzbmBg/Yk1Hh94J+SZRSaDvmwvqq7ffzP1k756PlThtSI8ZBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"18287fd5b6097130af2cdef59b8c4821898f90ae49bf7ca7f92b15bf1b330706","last_reissued_at":"2026-07-05T10:19:29.058735Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:29.058735Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scheherazade: Evaluating Chain-of-Thought Math Reasoning in LLMs with Chain-of-Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ferhat Erata, Ruzica Piskac, Sam Kouteili, Scott J Shapiro, Simeng Han, Stephen Miner, Yoshiki Takashima","submitted_at":"2024-09-30T18:48:34Z","abstract_excerpt":"Benchmarks are critical for measuring Large Language Model (LLM) reasoning capabilities. Some benchmarks have even become the de facto indicator of such capabilities. However, as LLM reasoning capabilities improve, existing widely-used benchmarks such as GSM8K marginally encapsulate model reasoning differentials - most state-of-the-art models for example achieve over 94% accuracy on the GSM8K dataset (paperwithcode, 2024). While constructing harder benchmarks is possible, their creation is often manual, expensive, and unscalable. As such, we present Scheherazade, an automated approach to produ"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.00151","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.00151/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.00151","created_at":"2026-07-05T10:19:29.058794+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.00151v4","created_at":"2026-07-05T10:19:29.058794+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.00151","created_at":"2026-07-05T10:19:29.058794+00:00"},{"alias_kind":"pith_short_12","alias_value":"DAUH7VNWBFYT","created_at":"2026-07-05T10:19:29.058794+00:00"},{"alias_kind":"pith_short_16","alias_value":"DAUH7VNWBFYTBLZM","created_at":"2026-07-05T10:19:29.058794+00:00"},{"alias_kind":"pith_short_8","alias_value":"DAUH7VNW","created_at":"2026-07-05T10:19:29.058794+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06720","citing_title":"When Does In-Context Search Help? A Sampling-Complexity Theory of Reflection-Driven Reasoning","ref_index":1,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DAUH7VNWBFYTBLZM332ZXDCIEG","json":"https://pith.science/pith/DAUH7VNWBFYTBLZM332ZXDCIEG.json","graph_json":"https://pith.science/api/pith-number/DAUH7VNWBFYTBLZM332ZXDCIEG/graph.json","events_json":"https://pith.science/api/pith-number/DAUH7VNWBFYTBLZM332ZXDCIEG/events.json","paper":"https://pith.science/paper/DAUH7VNW"},"agent_actions":{"view_html":"https://pith.science/pith/DAUH7VNWBFYTBLZM332ZXDCIEG","download_json":"https://pith.science/pith/DAUH7VNWBFYTBLZM332ZXDCIEG.json","view_paper":"https://pith.science/paper/DAUH7VNW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.00151&json=true","fetch_graph":"https://pith.science/api/pith-number/DAUH7VNWBFYTBLZM332ZXDCIEG/graph.json","fetch_events":"https://pith.science/api/pith-number/DAUH7VNWBFYTBLZM332ZXDCIEG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DAUH7VNWBFYTBLZM332ZXDCIEG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DAUH7VNWBFYTBLZM332ZXDCIEG/action/storage_attestation","attest_author":"https://pith.science/pith/DAUH7VNWBFYTBLZM332ZXDCIEG/action/author_attestation","sign_citation":"https://pith.science/pith/DAUH7VNWBFYTBLZM332ZXDCIEG/action/citation_signature","submit_replication":"https://pith.science/pith/DAUH7VNWBFYTBLZM332ZXDCIEG/action/replication_record"}},"created_at":"2026-07-05T10:19:29.058794+00:00","updated_at":"2026-07-05T10:19:29.058794+00:00"}