{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XP6NOD2LBO77CO6TN47DWS5D63","short_pith_number":"pith:XP6NOD2L","schema_version":"1.0","canonical_sha256":"bbfcd70f4b0bbff13bd36f3e3b4ba3f6f135b9d6492d5dda7545bb04887e75a5","source":{"kind":"arxiv","id":"2401.08190","version":3},"attestation_state":"computed","paper":{"title":"MARIO: MAth Reasoning with code Interpreter Output -- A Reproducible Pipeline","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chengxi Li, Jing Wu, Kai Fan, Minpeng Liao, Wei Luo","submitted_at":"2024-01-16T08:08:01Z","abstract_excerpt":"Large language models (LLMs) have seen considerable advancements in natural language understanding tasks, yet there remains a gap to bridge before attaining true artificial general intelligence, especially concerning shortcomings in mathematical reasoning capabilities. We postulate that the inherent nature of LLM training, which focuses on predicting probabilities of next token, presents challenges in effectively modeling mathematical reasoning that demands exact calculations, both from data-driven and theoretical standpoints. In this paper, we address this challenge by enriching the data land"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.08190","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-01-16T08:08:01Z","cross_cats_sorted":[],"title_canon_sha256":"7ecf981aac22842ef665e3b56c0d7172dbc0ccd05e1826a59560c4e1a59c1541","abstract_canon_sha256":"30dc762392598d65ed1c209eb824a4dce66ccd563822bb501edf5145cf839d7f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:48:02.720125Z","signature_b64":"OsI04AIbUmwnrU0rBDghQUT2TuvwjPl/70chFEXu/NV+zAjswBmYpSfDq26hCLhM76Xr5ZtDJQsmzU4hh5aUCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bbfcd70f4b0bbff13bd36f3e3b4ba3f6f135b9d6492d5dda7545bb04887e75a5","last_reissued_at":"2026-07-05T07:48:02.719590Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:48:02.719590Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MARIO: MAth Reasoning with code Interpreter Output -- A Reproducible Pipeline","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chengxi Li, Jing Wu, Kai Fan, Minpeng Liao, Wei Luo","submitted_at":"2024-01-16T08:08:01Z","abstract_excerpt":"Large language models (LLMs) have seen considerable advancements in natural language understanding tasks, yet there remains a gap to bridge before attaining true artificial general intelligence, especially concerning shortcomings in mathematical reasoning capabilities. We postulate that the inherent nature of LLM training, which focuses on predicting probabilities of next token, presents challenges in effectively modeling mathematical reasoning that demands exact calculations, both from data-driven and theoretical standpoints. In this paper, we address this challenge by enriching the data land"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.08190","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.08190/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.08190","created_at":"2026-07-05T07:48:02.719657+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.08190v3","created_at":"2026-07-05T07:48:02.719657+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.08190","created_at":"2026-07-05T07:48:02.719657+00:00"},{"alias_kind":"pith_short_12","alias_value":"XP6NOD2LBO77","created_at":"2026-07-05T07:48:02.719657+00:00"},{"alias_kind":"pith_short_16","alias_value":"XP6NOD2LBO77CO6T","created_at":"2026-07-05T07:48:02.719657+00:00"},{"alias_kind":"pith_short_8","alias_value":"XP6NOD2L","created_at":"2026-07-05T07:48:02.719657+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2406.18629","citing_title":"Step-DPO: Step-wise Preference Optimization for Long-chain Reasoning of LLMs","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2504.13958","citing_title":"ToolRL: Reward is All Tool Learning Needs","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09931","citing_title":"PruneTIR: Inference-Time Tool Call Pruning for Effective yet Efficient Tool-Integrated Reasoning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06377","citing_title":"The Master Key Hypothesis: Unlocking Cross-Model Capability Transfer via Linear Subspace Alignment","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08281","citing_title":"When to Trust Tools? Adaptive Tool Trust Calibration For Tool-Integrated Math Reasoning","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XP6NOD2LBO77CO6TN47DWS5D63","json":"https://pith.science/pith/XP6NOD2LBO77CO6TN47DWS5D63.json","graph_json":"https://pith.science/api/pith-number/XP6NOD2LBO77CO6TN47DWS5D63/graph.json","events_json":"https://pith.science/api/pith-number/XP6NOD2LBO77CO6TN47DWS5D63/events.json","paper":"https://pith.science/paper/XP6NOD2L"},"agent_actions":{"view_html":"https://pith.science/pith/XP6NOD2LBO77CO6TN47DWS5D63","download_json":"https://pith.science/pith/XP6NOD2LBO77CO6TN47DWS5D63.json","view_paper":"https://pith.science/paper/XP6NOD2L","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.08190&json=true","fetch_graph":"https://pith.science/api/pith-number/XP6NOD2LBO77CO6TN47DWS5D63/graph.json","fetch_events":"https://pith.science/api/pith-number/XP6NOD2LBO77CO6TN47DWS5D63/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XP6NOD2LBO77CO6TN47DWS5D63/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XP6NOD2LBO77CO6TN47DWS5D63/action/storage_attestation","attest_author":"https://pith.science/pith/XP6NOD2LBO77CO6TN47DWS5D63/action/author_attestation","sign_citation":"https://pith.science/pith/XP6NOD2LBO77CO6TN47DWS5D63/action/citation_signature","submit_replication":"https://pith.science/pith/XP6NOD2LBO77CO6TN47DWS5D63/action/replication_record"}},"created_at":"2026-07-05T07:48:02.719657+00:00","updated_at":"2026-07-05T07:48:02.719657+00:00"}