{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:O7B3SOF5VXP57QWQN4VJ5ZSNR3","short_pith_number":"pith:O7B3SOF5","schema_version":"1.0","canonical_sha256":"77c3b938bdaddfdfc2d06f2a9ee64d8ee2aff4f79eaf678bd115deba23b066af","source":{"kind":"arxiv","id":"2402.19450","version":1},"attestation_state":"computed","paper":{"title":"Functional Benchmarks for Robust Evaluation of Reasoning Performance, and the Reasoning Gap","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Adwaith Samod T, Ajay Sukumar, Alan Philipose, Annarose M B, Anto P V, Saurabh Srivastava, Shashank Menon, Sooraj Thomas, Stevin Prince","submitted_at":"2024-02-29T18:48:18Z","abstract_excerpt":"We propose a framework for robust evaluation of reasoning capabilities of language models, using functional variants of benchmarks. Models that solve a reasoning test should exhibit no difference in performance over the static version of a problem compared to a snapshot of the functional variant. We have rewritten the relevant fragment of the MATH benchmark into its functional variant MATH(), with functionalization of other benchmarks to follow. When evaluating current state-of-the-art models over snapshots of MATH(), we find a reasoning gap -- the percentage difference between the static and "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.19450","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-02-29T18:48:18Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"8c833bb19c8462c964915568c4eac1c78093ef0f9c63083be94e6bb3c6088e8f","abstract_canon_sha256":"ee812ccc7a8ceac6d0d2292ddb583a6acc6a6bf674e559a1f012c3802ce269c8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:50:45.472464Z","signature_b64":"Rh26opyywfqyQ4xS3XDj8CClvnVpq6xOl3ayD3EmogEhtHNQ8lOrIVFHy8WDDbcHoC858XYepYQoUiYHr7rIAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"77c3b938bdaddfdfc2d06f2a9ee64d8ee2aff4f79eaf678bd115deba23b066af","last_reissued_at":"2026-07-05T07:50:45.471949Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:50:45.471949Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Functional Benchmarks for Robust Evaluation of Reasoning Performance, and the Reasoning Gap","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Adwaith Samod T, Ajay Sukumar, Alan Philipose, Annarose M B, Anto P V, Saurabh Srivastava, Shashank Menon, Sooraj Thomas, Stevin Prince","submitted_at":"2024-02-29T18:48:18Z","abstract_excerpt":"We propose a framework for robust evaluation of reasoning capabilities of language models, using functional variants of benchmarks. Models that solve a reasoning test should exhibit no difference in performance over the static version of a problem compared to a snapshot of the functional variant. We have rewritten the relevant fragment of the MATH benchmark into its functional variant MATH(), with functionalization of other benchmarks to follow. When evaluating current state-of-the-art models over snapshots of MATH(), we find a reasoning gap -- the percentage difference between the static and "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.19450","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.19450/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.19450","created_at":"2026-07-05T07:50:45.472002+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.19450v1","created_at":"2026-07-05T07:50:45.472002+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.19450","created_at":"2026-07-05T07:50:45.472002+00:00"},{"alias_kind":"pith_short_12","alias_value":"O7B3SOF5VXP5","created_at":"2026-07-05T07:50:45.472002+00:00"},{"alias_kind":"pith_short_16","alias_value":"O7B3SOF5VXP57QWQ","created_at":"2026-07-05T07:50:45.472002+00:00"},{"alias_kind":"pith_short_8","alias_value":"O7B3SOF5","created_at":"2026-07-05T07:50:45.472002+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08728","citing_title":"Artificial Intelligence for Mathematical Reasoning: An Integrated Survey of Language Models, Neuro-symbolic Systems, and Verified Discovery","ref_index":257,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30182","citing_title":"MirrorCode: AI can rebuild entire programs from behavior alone","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08571","citing_title":"Robust Reasoning Benchmark","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15393","citing_title":"LPDS: Evaluating LLM Robustness Through Logic-Preserving Difficulty Scaling","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2509.17677","citing_title":"EngiBench: A Benchmark for Evaluating Large Language Models on Engineering Problem Solving","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2406.19314","citing_title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2410.05229","citing_title":"GSM-Symbolic: Understanding the Limitations of Mathematical Reasoning in Large Language Models","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08571","citing_title":"Robust Reasoning Benchmark","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06802","citing_title":"Riemann-Bench: A Benchmark for Moonshot Mathematics","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O7B3SOF5VXP57QWQN4VJ5ZSNR3","json":"https://pith.science/pith/O7B3SOF5VXP57QWQN4VJ5ZSNR3.json","graph_json":"https://pith.science/api/pith-number/O7B3SOF5VXP57QWQN4VJ5ZSNR3/graph.json","events_json":"https://pith.science/api/pith-number/O7B3SOF5VXP57QWQN4VJ5ZSNR3/events.json","paper":"https://pith.science/paper/O7B3SOF5"},"agent_actions":{"view_html":"https://pith.science/pith/O7B3SOF5VXP57QWQN4VJ5ZSNR3","download_json":"https://pith.science/pith/O7B3SOF5VXP57QWQN4VJ5ZSNR3.json","view_paper":"https://pith.science/paper/O7B3SOF5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.19450&json=true","fetch_graph":"https://pith.science/api/pith-number/O7B3SOF5VXP57QWQN4VJ5ZSNR3/graph.json","fetch_events":"https://pith.science/api/pith-number/O7B3SOF5VXP57QWQN4VJ5ZSNR3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O7B3SOF5VXP57QWQN4VJ5ZSNR3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O7B3SOF5VXP57QWQN4VJ5ZSNR3/action/storage_attestation","attest_author":"https://pith.science/pith/O7B3SOF5VXP57QWQN4VJ5ZSNR3/action/author_attestation","sign_citation":"https://pith.science/pith/O7B3SOF5VXP57QWQN4VJ5ZSNR3/action/citation_signature","submit_replication":"https://pith.science/pith/O7B3SOF5VXP57QWQN4VJ5ZSNR3/action/replication_record"}},"created_at":"2026-07-05T07:50:45.472002+00:00","updated_at":"2026-07-05T07:50:45.472002+00:00"}