{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OOJIC3MZWOBBSY3CVQ45JBJFIA","short_pith_number":"pith:OOJIC3MZ","schema_version":"1.0","canonical_sha256":"7392816d99b382196362ac39d48525401c1aae2d940fbe17816e304a31268a7d","source":{"kind":"arxiv","id":"2406.03847","version":3},"attestation_state":"computed","paper":{"title":"Lean Workbook: A large-scale Lean problem set formalized from natural language math problems","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dahua Lin, Huaiyuan Ying, Kai Chen, Yihan Geng, Zheng Yuan, Zijian Wu","submitted_at":"2024-06-06T08:25:43Z","abstract_excerpt":"Large language models have demonstrated impressive capabilities across various natural language processing tasks, especially in solving mathematical problems. However, large language models are not good at math theorem proving using formal languages like Lean. A significant challenge in this area is the scarcity of training data available in these formal languages. To address this issue, we propose a novel pipeline that iteratively generates and filters synthetic data to translate natural language mathematical problems into Lean 4 statements, and vice versa. Our results indicate that the synth"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.03847","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-06T08:25:43Z","cross_cats_sorted":[],"title_canon_sha256":"21f31e210ca9e9845aa64c8f27d304f455d59861e3a906e6e8a003aa2eed5161","abstract_canon_sha256":"46e486062d70f8a2609009fb4539a71b45c2d7ce8b5e20bbcc9c33c2b175a72e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:23:16.356439Z","signature_b64":"ZhKHj9kVGGpoqi2rEmlsfWid7QbxNNi0IMhZaTDUWd1262pu8/qMOClIbbR8uK//AzYoriRfY/Da13KZPMN5Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7392816d99b382196362ac39d48525401c1aae2d940fbe17816e304a31268a7d","last_reissued_at":"2026-07-05T11:23:16.355900Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:23:16.355900Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Lean Workbook: A large-scale Lean problem set formalized from natural language math problems","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dahua Lin, Huaiyuan Ying, Kai Chen, Yihan Geng, Zheng Yuan, Zijian Wu","submitted_at":"2024-06-06T08:25:43Z","abstract_excerpt":"Large language models have demonstrated impressive capabilities across various natural language processing tasks, especially in solving mathematical problems. However, large language models are not good at math theorem proving using formal languages like Lean. A significant challenge in this area is the scarcity of training data available in these formal languages. To address this issue, we propose a novel pipeline that iteratively generates and filters synthetic data to translate natural language mathematical problems into Lean 4 statements, and vice versa. Our results indicate that the synth"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.03847","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.03847/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.03847","created_at":"2026-07-05T11:23:16.355967+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.03847v3","created_at":"2026-07-05T11:23:16.355967+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.03847","created_at":"2026-07-05T11:23:16.355967+00:00"},{"alias_kind":"pith_short_12","alias_value":"OOJIC3MZWOBB","created_at":"2026-07-05T11:23:16.355967+00:00"},{"alias_kind":"pith_short_16","alias_value":"OOJIC3MZWOBBSY3C","created_at":"2026-07-05T11:23:16.355967+00:00"},{"alias_kind":"pith_short_8","alias_value":"OOJIC3MZ","created_at":"2026-07-05T11:23:16.355967+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09450","citing_title":"TheoremBench: Evaluating LLMs on Theorem Proving in Formal Mathematics","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05400","citing_title":"LeanMarathon: Toward Reliable AI Co-Mathematicians through Long-Horizon Lean Autoformalization","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01861","citing_title":"A Theoretical Framework for Self-Play Theorem Proving Algorithms","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30914","citing_title":"Automating Formal Verification with Reinforcement Learning and Recursive Inference","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14061","citing_title":"MathAtlas: A Benchmark for Autoformalization in the Wild","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23281","citing_title":"MathArena: Evaluating LLMs on Uncontaminated Math Competitions","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02709","citing_title":"Evaluating the Formal Reasoning Capabilities of Large Language Models through Chomsky Hierarchy","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11905","citing_title":"Rethinking Supervision Granularity: Segment-Level Learning for LLM-Based Theorem Proving","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OOJIC3MZWOBBSY3CVQ45JBJFIA","json":"https://pith.science/pith/OOJIC3MZWOBBSY3CVQ45JBJFIA.json","graph_json":"https://pith.science/api/pith-number/OOJIC3MZWOBBSY3CVQ45JBJFIA/graph.json","events_json":"https://pith.science/api/pith-number/OOJIC3MZWOBBSY3CVQ45JBJFIA/events.json","paper":"https://pith.science/paper/OOJIC3MZ"},"agent_actions":{"view_html":"https://pith.science/pith/OOJIC3MZWOBBSY3CVQ45JBJFIA","download_json":"https://pith.science/pith/OOJIC3MZWOBBSY3CVQ45JBJFIA.json","view_paper":"https://pith.science/paper/OOJIC3MZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.03847&json=true","fetch_graph":"https://pith.science/api/pith-number/OOJIC3MZWOBBSY3CVQ45JBJFIA/graph.json","fetch_events":"https://pith.science/api/pith-number/OOJIC3MZWOBBSY3CVQ45JBJFIA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OOJIC3MZWOBBSY3CVQ45JBJFIA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OOJIC3MZWOBBSY3CVQ45JBJFIA/action/storage_attestation","attest_author":"https://pith.science/pith/OOJIC3MZWOBBSY3CVQ45JBJFIA/action/author_attestation","sign_citation":"https://pith.science/pith/OOJIC3MZWOBBSY3CVQ45JBJFIA/action/citation_signature","submit_replication":"https://pith.science/pith/OOJIC3MZWOBBSY3CVQ45JBJFIA/action/replication_record"}},"created_at":"2026-07-05T11:23:16.355967+00:00","updated_at":"2026-07-05T11:23:16.355967+00:00"}