{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:LIR2IW5ET6Q7SDZ4EKD5SEN23T","short_pith_number":"pith:LIR2IW5E","schema_version":"1.0","canonical_sha256":"5a23a45ba49fa1f90f3c2287d911badccf7536270058400776a853f8b8e10a8e","source":{"kind":"arxiv","id":"2303.05398","version":1},"attestation_state":"computed","paper":{"title":"MathPrompter: Mathematical Reasoning using Large Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Harsh Shrivastava, Liang Du, Shima Imani","submitted_at":"2023-03-04T04:43:49Z","abstract_excerpt":"Large Language Models (LLMs) have limited performance when solving arithmetic reasoning tasks and often provide incorrect answers. Unlike natural language understanding, math problems typically have a single correct answer, making the task of generating accurate solutions more challenging for LLMs. To the best of our knowledge, we are not aware of any LLMs that indicate their level of confidence in their responses which fuels a trust deficit in these models impeding their adoption. To address this deficiency, we propose `MathPrompter', a technique that improves performance of LLMs on arithmeti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.05398","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2023-03-04T04:43:49Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"23f434c26ec11475e9e481ce3cb62389f83588e1f0f1f80dcdee3e419736e8d7","abstract_canon_sha256":"d886199549a1ba4931db81a3bb7a9a671202d918bb6adf4e6564ba74bc8d714b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:49:40.387298Z","signature_b64":"Sc7kGX+Zt/eYz0TQQ0NGy2/KpefICDZ+ZSQApOKp17Z6Di8FaDQ9FYP7B9xwb/QclamSXbnzj2XxmyinDdAdBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5a23a45ba49fa1f90f3c2287d911badccf7536270058400776a853f8b8e10a8e","last_reissued_at":"2026-07-05T05:49:40.386852Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:49:40.386852Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MathPrompter: Mathematical Reasoning using Large Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Harsh Shrivastava, Liang Du, Shima Imani","submitted_at":"2023-03-04T04:43:49Z","abstract_excerpt":"Large Language Models (LLMs) have limited performance when solving arithmetic reasoning tasks and often provide incorrect answers. Unlike natural language understanding, math problems typically have a single correct answer, making the task of generating accurate solutions more challenging for LLMs. To the best of our knowledge, we are not aware of any LLMs that indicate their level of confidence in their responses which fuels a trust deficit in these models impeding their adoption. To address this deficiency, we propose `MathPrompter', a technique that improves performance of LLMs on arithmeti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.05398","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.05398/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.05398","created_at":"2026-07-05T05:49:40.386911+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.05398v1","created_at":"2026-07-05T05:49:40.386911+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.05398","created_at":"2026-07-05T05:49:40.386911+00:00"},{"alias_kind":"pith_short_12","alias_value":"LIR2IW5ET6Q7","created_at":"2026-07-05T05:49:40.386911+00:00"},{"alias_kind":"pith_short_16","alias_value":"LIR2IW5ET6Q7SDZ4","created_at":"2026-07-05T05:49:40.386911+00:00"},{"alias_kind":"pith_short_8","alias_value":"LIR2IW5E","created_at":"2026-07-05T05:49:40.386911+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00295","citing_title":"Adaptive Order Policies for Masked Diffusion","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2402.09664","citing_title":"CodeMind: Evaluating Large Language Models for Code Reasoning","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2505.04588","citing_title":"ZeroSearch: Incentivize the Search Capability of LLMs without Searching","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01770","citing_title":"ReGA: Model-Based Safeguard for LLMs via Representation-Guided Abstraction","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2509.11295","citing_title":"The Prompt Engineering Report Distilled: Quick Start Guide for Life Sciences","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2509.11295","citing_title":"The Prompt Engineering Report Distilled: Quick Start Guide for Life Sciences","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2401.03568","citing_title":"Agent AI: Surveying the Horizons of Multimodal Interaction","ref_index":228,"is_internal_anchor":false},{"citing_arxiv_id":"2510.15079","citing_title":"Assessing Coherency and Consistency of Code Execution Reasoning by Large Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2505.04588","citing_title":"ZeroSearch: Incentivize the Search Capability of LLMs without Searching","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02280","citing_title":"RACC: Representation-Aware Coverage Criteria for LLM Safety Testing","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02350","citing_title":"Context Learning for Multi-Agent Discussion","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2303.17760","citing_title":"CAMEL: Communicative Agents for \"Mind\" Exploration of Large Language Model Society","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05409","citing_title":"Agentic Retrieval-Augmented Generation for Financial Document Question Answering","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04064","citing_title":"Improving Medical VQA through Trajectory-Aware Process Supervision","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07301","citing_title":"SOM: Structured Opponent Modeling for LLM-based Agents via Structural Causal Model","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2402.01207","citing_title":"Efficient Causal Graph Discovery Using Large Language Models","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LIR2IW5ET6Q7SDZ4EKD5SEN23T","json":"https://pith.science/pith/LIR2IW5ET6Q7SDZ4EKD5SEN23T.json","graph_json":"https://pith.science/api/pith-number/LIR2IW5ET6Q7SDZ4EKD5SEN23T/graph.json","events_json":"https://pith.science/api/pith-number/LIR2IW5ET6Q7SDZ4EKD5SEN23T/events.json","paper":"https://pith.science/paper/LIR2IW5E"},"agent_actions":{"view_html":"https://pith.science/pith/LIR2IW5ET6Q7SDZ4EKD5SEN23T","download_json":"https://pith.science/pith/LIR2IW5ET6Q7SDZ4EKD5SEN23T.json","view_paper":"https://pith.science/paper/LIR2IW5E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.05398&json=true","fetch_graph":"https://pith.science/api/pith-number/LIR2IW5ET6Q7SDZ4EKD5SEN23T/graph.json","fetch_events":"https://pith.science/api/pith-number/LIR2IW5ET6Q7SDZ4EKD5SEN23T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LIR2IW5ET6Q7SDZ4EKD5SEN23T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LIR2IW5ET6Q7SDZ4EKD5SEN23T/action/storage_attestation","attest_author":"https://pith.science/pith/LIR2IW5ET6Q7SDZ4EKD5SEN23T/action/author_attestation","sign_citation":"https://pith.science/pith/LIR2IW5ET6Q7SDZ4EKD5SEN23T/action/citation_signature","submit_replication":"https://pith.science/pith/LIR2IW5ET6Q7SDZ4EKD5SEN23T/action/replication_record"}},"created_at":"2026-07-05T05:49:40.386911+00:00","updated_at":"2026-07-05T05:49:40.386911+00:00"}