{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EMLSFKQ2HUFVRWP4R6TVGMWKWD","short_pith_number":"pith:EMLSFKQ2","schema_version":"1.0","canonical_sha256":"231722aa1a3d0b58d9fc8fa75332cab0f8cf00dd326abd5fb0f54fcccb39fe6e","source":{"kind":"arxiv","id":"2410.19056","version":1},"attestation_state":"computed","paper":{"title":"ReasonAgain: Using Extractable Symbolic Programs to Evaluate Mathematical Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Ben Zhou, Dan Roth, Hao Cheng, Xiaodong Yu","submitted_at":"2024-10-24T18:02:37Z","abstract_excerpt":"Existing math datasets evaluate the reasoning abilities of large language models (LLMs) by either using the final answer or the intermediate reasoning steps derived from static examples. However, the former approach fails to surface model's uses of shortcuts and wrong reasoning while the later poses challenges in accommodating alternative solutions. In this work, we seek to use symbolic programs as a means for automated evaluation if a model can consistently produce correct final answers across various inputs to the program. We begin by extracting programs for popular math datasets (GSM8K and "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.19056","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-10-24T18:02:37Z","cross_cats_sorted":[],"title_canon_sha256":"6fa8924f0501f6cdcbabb0e2ecc7b12eff0ff53bf2ce1d228ec4feeff89689d4","abstract_canon_sha256":"a1fea6f2ca9d67bcc03a25413ebc5757d0b1531724929b02979c6a4bad89420f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:25:50.900613Z","signature_b64":"/DupAmnw+zDIsQrpol12pJVQ5+SyoEZF5Kv/AOTyOHIrApaDA3TFEXxPPj1Fy+Q2EVvJLhxBFzbN6uWj8QeiBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"231722aa1a3d0b58d9fc8fa75332cab0f8cf00dd326abd5fb0f54fcccb39fe6e","last_reissued_at":"2026-07-05T09:25:50.900116Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:25:50.900116Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReasonAgain: Using Extractable Symbolic Programs to Evaluate Mathematical Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Ben Zhou, Dan Roth, Hao Cheng, Xiaodong Yu","submitted_at":"2024-10-24T18:02:37Z","abstract_excerpt":"Existing math datasets evaluate the reasoning abilities of large language models (LLMs) by either using the final answer or the intermediate reasoning steps derived from static examples. However, the former approach fails to surface model's uses of shortcuts and wrong reasoning while the later poses challenges in accommodating alternative solutions. In this work, we seek to use symbolic programs as a means for automated evaluation if a model can consistently produce correct final answers across various inputs to the program. We begin by extracting programs for popular math datasets (GSM8K and "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.19056","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.19056/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.19056","created_at":"2026-07-05T09:25:50.900174+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.19056v1","created_at":"2026-07-05T09:25:50.900174+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.19056","created_at":"2026-07-05T09:25:50.900174+00:00"},{"alias_kind":"pith_short_12","alias_value":"EMLSFKQ2HUFV","created_at":"2026-07-05T09:25:50.900174+00:00"},{"alias_kind":"pith_short_16","alias_value":"EMLSFKQ2HUFVRWP4","created_at":"2026-07-05T09:25:50.900174+00:00"},{"alias_kind":"pith_short_8","alias_value":"EMLSFKQ2","created_at":"2026-07-05T09:25:50.900174+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.07053","citing_title":"GSM-SEM: Benchmark and Framework for Generating Semantically Variant Augmentations","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07053","citing_title":"GSM-SEM: Benchmark and Framework for Generating Semantically Variant Augmentations","ref_index":74,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EMLSFKQ2HUFVRWP4R6TVGMWKWD","json":"https://pith.science/pith/EMLSFKQ2HUFVRWP4R6TVGMWKWD.json","graph_json":"https://pith.science/api/pith-number/EMLSFKQ2HUFVRWP4R6TVGMWKWD/graph.json","events_json":"https://pith.science/api/pith-number/EMLSFKQ2HUFVRWP4R6TVGMWKWD/events.json","paper":"https://pith.science/paper/EMLSFKQ2"},"agent_actions":{"view_html":"https://pith.science/pith/EMLSFKQ2HUFVRWP4R6TVGMWKWD","download_json":"https://pith.science/pith/EMLSFKQ2HUFVRWP4R6TVGMWKWD.json","view_paper":"https://pith.science/paper/EMLSFKQ2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.19056&json=true","fetch_graph":"https://pith.science/api/pith-number/EMLSFKQ2HUFVRWP4R6TVGMWKWD/graph.json","fetch_events":"https://pith.science/api/pith-number/EMLSFKQ2HUFVRWP4R6TVGMWKWD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EMLSFKQ2HUFVRWP4R6TVGMWKWD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EMLSFKQ2HUFVRWP4R6TVGMWKWD/action/storage_attestation","attest_author":"https://pith.science/pith/EMLSFKQ2HUFVRWP4R6TVGMWKWD/action/author_attestation","sign_citation":"https://pith.science/pith/EMLSFKQ2HUFVRWP4R6TVGMWKWD/action/citation_signature","submit_replication":"https://pith.science/pith/EMLSFKQ2HUFVRWP4R6TVGMWKWD/action/replication_record"}},"created_at":"2026-07-05T09:25:50.900174+00:00","updated_at":"2026-07-05T09:25:50.900174+00:00"}