{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PIO4BJAGEBC46LLHV7H6UPFLZO","short_pith_number":"pith:PIO4BJAG","schema_version":"1.0","canonical_sha256":"7a1dc0a4062045cf2d67afcfea3cabcba9e763970bc97a58b4d3a50cf20eaeeb","source":{"kind":"arxiv","id":"2306.15626","version":2},"attestation_state":"computed","paper":{"title":"LeanDojo: Theorem Proving with Retrieval-Augmented Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aidan M. Swope, Alex Gu, Anima Anandkumar, Kaiyu Yang, Peiyang Song, Rahul Chalamala, Ryan Prenger, Saad Godil, Shixing Yu","submitted_at":"2023-06-27T17:05:32Z","abstract_excerpt":"Large language models (LLMs) have shown promise in proving formal theorems using proof assistants such as Lean. However, existing methods are difficult to reproduce or build on, due to private code, data, and large compute requirements. This has created substantial barriers to research on machine learning methods for theorem proving. This paper removes these barriers by introducing LeanDojo: an open-source Lean playground consisting of toolkits, data, models, and benchmarks. LeanDojo extracts data from Lean and enables interaction with the proof environment programmatically. It contains fine-g"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.15626","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-06-27T17:05:32Z","cross_cats_sorted":["cs.AI","cs.LO","stat.ML"],"title_canon_sha256":"9adb01b736388f52699689ef42c5ac992eaf293c3fefef1fc392543f9ffc7f36","abstract_canon_sha256":"3524432ac1948acabcffc03d1d05b1a415602ff4d76589675f0db013eb1948f2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:05:45.655559Z","signature_b64":"3EsXGzQaEq2w+ubwuNzDD+gOICV1ZZl3hAQ279Mso84V0A4LeHlk7iv8v2caf2KBnO1Hgi2gvSXCy/K6CIk/Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7a1dc0a4062045cf2d67afcfea3cabcba9e763970bc97a58b4d3a50cf20eaeeb","last_reissued_at":"2026-07-05T07:05:45.655068Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:05:45.655068Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LeanDojo: Theorem Proving with Retrieval-Augmented Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Aidan M. Swope, Alex Gu, Anima Anandkumar, Kaiyu Yang, Peiyang Song, Rahul Chalamala, Ryan Prenger, Saad Godil, Shixing Yu","submitted_at":"2023-06-27T17:05:32Z","abstract_excerpt":"Large language models (LLMs) have shown promise in proving formal theorems using proof assistants such as Lean. However, existing methods are difficult to reproduce or build on, due to private code, data, and large compute requirements. This has created substantial barriers to research on machine learning methods for theorem proving. This paper removes these barriers by introducing LeanDojo: an open-source Lean playground consisting of toolkits, data, models, and benchmarks. LeanDojo extracts data from Lean and enables interaction with the proof environment programmatically. It contains fine-g"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.15626","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.15626/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.15626","created_at":"2026-07-05T07:05:45.655127+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.15626v2","created_at":"2026-07-05T07:05:45.655127+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.15626","created_at":"2026-07-05T07:05:45.655127+00:00"},{"alias_kind":"pith_short_12","alias_value":"PIO4BJAGEBC4","created_at":"2026-07-05T07:05:45.655127+00:00"},{"alias_kind":"pith_short_16","alias_value":"PIO4BJAGEBC46LLH","created_at":"2026-07-05T07:05:45.655127+00:00"},{"alias_kind":"pith_short_8","alias_value":"PIO4BJAG","created_at":"2026-07-05T07:05:45.655127+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25363","citing_title":"TheoremGraph: Bridging Formal and Informal Mathematics","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09450","citing_title":"TheoremBench: Evaluating LLMs on Theorem Proving in Formal Mathematics","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05400","citing_title":"LeanMarathon: Toward Reliable AI Co-Mathematicians through Long-Horizon Lean Autoformalization","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29687","citing_title":"A Machine-Verified Proof of a Quantum-Optimization Conjecture","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27485","citing_title":"Automating Formal Verification with Agent-Guided Tree Search","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30914","citing_title":"Automating Formal Verification with Reinforcement Learning and Recursive Inference","ref_index":129,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30914","citing_title":"Automating Formal Verification with Reinforcement Learning and Recursive Inference","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23109","citing_title":"Inductive Deductive Synthesis: Enabling AI to Generate Formally Verified Systems","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2511.12253","citing_title":"The Search for Constrained Random Generators","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2602.24273","citing_title":"A Minimal Agent for Automated Theorem Proving","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16347","citing_title":"Lean Atlas: An Integrated Proof Environment for Scalable Human-AI Collaborative Formalization","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02709","citing_title":"Evaluating the Formal Reasoning Capabilities of Large Language Models through Chomsky Hierarchy","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01394","citing_title":"LiveFMBench: Unveiling the Power and Limits of Agentic Workflows in Specification Generation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20622","citing_title":"pAI/MSc: ML Theory Research with Humans on the Loop","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06401","citing_title":"ProofSketcher: Hybrid LLM + Lightweight Proof Checker for Reliable Math/Logic Reasoning","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PIO4BJAGEBC46LLHV7H6UPFLZO","json":"https://pith.science/pith/PIO4BJAGEBC46LLHV7H6UPFLZO.json","graph_json":"https://pith.science/api/pith-number/PIO4BJAGEBC46LLHV7H6UPFLZO/graph.json","events_json":"https://pith.science/api/pith-number/PIO4BJAGEBC46LLHV7H6UPFLZO/events.json","paper":"https://pith.science/paper/PIO4BJAG"},"agent_actions":{"view_html":"https://pith.science/pith/PIO4BJAGEBC46LLHV7H6UPFLZO","download_json":"https://pith.science/pith/PIO4BJAGEBC46LLHV7H6UPFLZO.json","view_paper":"https://pith.science/paper/PIO4BJAG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.15626&json=true","fetch_graph":"https://pith.science/api/pith-number/PIO4BJAGEBC46LLHV7H6UPFLZO/graph.json","fetch_events":"https://pith.science/api/pith-number/PIO4BJAGEBC46LLHV7H6UPFLZO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PIO4BJAGEBC46LLHV7H6UPFLZO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PIO4BJAGEBC46LLHV7H6UPFLZO/action/storage_attestation","attest_author":"https://pith.science/pith/PIO4BJAGEBC46LLHV7H6UPFLZO/action/author_attestation","sign_citation":"https://pith.science/pith/PIO4BJAGEBC46LLHV7H6UPFLZO/action/citation_signature","submit_replication":"https://pith.science/pith/PIO4BJAGEBC46LLHV7H6UPFLZO/action/replication_record"}},"created_at":"2026-07-05T07:05:45.655127+00:00","updated_at":"2026-07-05T07:05:45.655127+00:00"}