{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KHTE3SOAI2WFVAXOJIXBXTKIGG","short_pith_number":"pith:KHTE3SOA","schema_version":"1.0","canonical_sha256":"51e64dc9c046ac5a82ee4a2e1bcd4831b8ac5a771abe852835610a7f2b5c582a","source":{"kind":"arxiv","id":"2412.08819","version":1},"attestation_state":"computed","paper":{"title":"HARP: A challenging human-annotated math reasoning benchmark","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aaditya K. Singh, Albert S. Yue, DJ Strouse, Lovish Madaan, Ted Moskovitz","submitted_at":"2024-12-11T23:31:06Z","abstract_excerpt":"Math reasoning is becoming an ever increasing area of focus as we scale large language models. However, even the previously-toughest evals like MATH are now close to saturated by frontier models (90.0% for o1-mini and 86.5% for Gemini 1.5 Pro). We introduce HARP, Human Annotated Reasoning Problems (for Math), consisting of 5,409 problems from the US national math competitions (A(J)HSME, AMC, AIME, USA(J)MO). Of these, 4,780 have answers that are automatically check-able (with libraries such as SymPy). These problems range six difficulty levels, with frontier models performing relatively poorly"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.08819","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-12-11T23:31:06Z","cross_cats_sorted":[],"title_canon_sha256":"5490f5597aa14c185f98959a462ed2a0b204271c4884a0f1156377cbfab7e6fb","abstract_canon_sha256":"237106f2d8e0fd6d8c5fa0054f20b4510bf3ec900875f9ea4630bee23a5abf7d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:48:02.193007Z","signature_b64":"J2a2r5ulu8WG+fqQ/2J6VZ9+idIiStluQucko1vUl0oCqtL+N//Q035DJFpWNf7EpzPumvmKA4Xq1q1b6fetAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"51e64dc9c046ac5a82ee4a2e1bcd4831b8ac5a771abe852835610a7f2b5c582a","last_reissued_at":"2026-07-05T09:48:02.192409Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:48:02.192409Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HARP: A challenging human-annotated math reasoning benchmark","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aaditya K. Singh, Albert S. Yue, DJ Strouse, Lovish Madaan, Ted Moskovitz","submitted_at":"2024-12-11T23:31:06Z","abstract_excerpt":"Math reasoning is becoming an ever increasing area of focus as we scale large language models. However, even the previously-toughest evals like MATH are now close to saturated by frontier models (90.0% for o1-mini and 86.5% for Gemini 1.5 Pro). We introduce HARP, Human Annotated Reasoning Problems (for Math), consisting of 5,409 problems from the US national math competitions (A(J)HSME, AMC, AIME, USA(J)MO). Of these, 4,780 have answers that are automatically check-able (with libraries such as SymPy). These problems range six difficulty levels, with frontier models performing relatively poorly"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.08819","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.08819/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.08819","created_at":"2026-07-05T09:48:02.192475+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.08819v1","created_at":"2026-07-05T09:48:02.192475+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.08819","created_at":"2026-07-05T09:48:02.192475+00:00"},{"alias_kind":"pith_short_12","alias_value":"KHTE3SOAI2WF","created_at":"2026-07-05T09:48:02.192475+00:00"},{"alias_kind":"pith_short_16","alias_value":"KHTE3SOAI2WFVAXO","created_at":"2026-07-05T09:48:02.192475+00:00"},{"alias_kind":"pith_short_8","alias_value":"KHTE3SOA","created_at":"2026-07-05T09:48:02.192475+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06526","citing_title":"CrowdMath: A Dataset of Crowdsourced Mathematical Research Discussions","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26971","citing_title":"RLVR Datasets and Where to Find Them: Tracing Data Lineage for Better Training Data","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2509.17677","citing_title":"EngiBench: A Benchmark for Evaluating Large Language Models on Engineering Problem Solving","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23281","citing_title":"MathArena: Evaluating LLMs on Uncontaminated Math Competitions","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09012","citing_title":"Re$^2$Math: Benchmarking Theorem Retrieval in Research-Level Mathematics","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KHTE3SOAI2WFVAXOJIXBXTKIGG","json":"https://pith.science/pith/KHTE3SOAI2WFVAXOJIXBXTKIGG.json","graph_json":"https://pith.science/api/pith-number/KHTE3SOAI2WFVAXOJIXBXTKIGG/graph.json","events_json":"https://pith.science/api/pith-number/KHTE3SOAI2WFVAXOJIXBXTKIGG/events.json","paper":"https://pith.science/paper/KHTE3SOA"},"agent_actions":{"view_html":"https://pith.science/pith/KHTE3SOAI2WFVAXOJIXBXTKIGG","download_json":"https://pith.science/pith/KHTE3SOAI2WFVAXOJIXBXTKIGG.json","view_paper":"https://pith.science/paper/KHTE3SOA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.08819&json=true","fetch_graph":"https://pith.science/api/pith-number/KHTE3SOAI2WFVAXOJIXBXTKIGG/graph.json","fetch_events":"https://pith.science/api/pith-number/KHTE3SOAI2WFVAXOJIXBXTKIGG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KHTE3SOAI2WFVAXOJIXBXTKIGG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KHTE3SOAI2WFVAXOJIXBXTKIGG/action/storage_attestation","attest_author":"https://pith.science/pith/KHTE3SOAI2WFVAXOJIXBXTKIGG/action/author_attestation","sign_citation":"https://pith.science/pith/KHTE3SOAI2WFVAXOJIXBXTKIGG/action/citation_signature","submit_replication":"https://pith.science/pith/KHTE3SOAI2WFVAXOJIXBXTKIGG/action/replication_record"}},"created_at":"2026-07-05T09:48:02.192475+00:00","updated_at":"2026-07-05T09:48:02.192475+00:00"}