{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EFFH2QNIXN2D3PAZB43ZWXBG6S","short_pith_number":"pith:EFFH2QNI","schema_version":"1.0","canonical_sha256":"214a7d41a8bb743dbc190f379b5c26f4b635825b978aafb9c2f68a719c2e2ffa","source":{"kind":"arxiv","id":"2407.03203","version":2},"attestation_state":"computed","paper":{"title":"TheoremLlama: Transforming General-Purpose LLMs into Lean4 Experts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.FL","authors_text":"Jipeng Zhang, Renjie Pi, Ruida Wang, Rui Pan, Shizhe Diao, Tong Zhang, Yizhen Jia","submitted_at":"2024-07-03T15:36:18Z","abstract_excerpt":"Proving mathematical theorems using computer-verifiable formal languages like Lean significantly impacts mathematical reasoning. One approach to formal theorem proving involves generating complete proofs using Large Language Models (LLMs) based on Natural Language (NL) proofs. However, due to the scarcity of aligned NL and Formal Language (FL) theorem-proving data most modern LLMs exhibit suboptimal performance.This scarcity results in a paucity of methodologies for training LLMs and techniques to fully utilize their capabilities in composing formal proofs. To address these challenges, this pa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.03203","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.FL","submitted_at":"2024-07-03T15:36:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7632e318637c5c8a83a6d0162ffd6ce8e63e1a4fa0c21f594f5534607d14acbb","abstract_canon_sha256":"0caac1e3ef3cc5bfd4a56cdab2a319f15152845a7c738b570289c12eaf36afe4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:15:34.709548Z","signature_b64":"Eo8d2P30SGfEwkVD9ehjm2A/hNd3mYh/CMqyPyQIbCCtVltE2wGHXNyK8nP2+fSZIvNhLBtEqH22E5E5354pAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"214a7d41a8bb743dbc190f379b5c26f4b635825b978aafb9c2f68a719c2e2ffa","last_reissued_at":"2026-07-05T09:15:34.709033Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:15:34.709033Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TheoremLlama: Transforming General-Purpose LLMs into Lean4 Experts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.FL","authors_text":"Jipeng Zhang, Renjie Pi, Ruida Wang, Rui Pan, Shizhe Diao, Tong Zhang, Yizhen Jia","submitted_at":"2024-07-03T15:36:18Z","abstract_excerpt":"Proving mathematical theorems using computer-verifiable formal languages like Lean significantly impacts mathematical reasoning. One approach to formal theorem proving involves generating complete proofs using Large Language Models (LLMs) based on Natural Language (NL) proofs. However, due to the scarcity of aligned NL and Formal Language (FL) theorem-proving data most modern LLMs exhibit suboptimal performance.This scarcity results in a paucity of methodologies for training LLMs and techniques to fully utilize their capabilities in composing formal proofs. To address these challenges, this pa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.03203","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.03203/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.03203","created_at":"2026-07-05T09:15:34.709090+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.03203v2","created_at":"2026-07-05T09:15:34.709090+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.03203","created_at":"2026-07-05T09:15:34.709090+00:00"},{"alias_kind":"pith_short_12","alias_value":"EFFH2QNIXN2D","created_at":"2026-07-05T09:15:34.709090+00:00"},{"alias_kind":"pith_short_16","alias_value":"EFFH2QNIXN2D3PAZ","created_at":"2026-07-05T09:15:34.709090+00:00"},{"alias_kind":"pith_short_8","alias_value":"EFFH2QNI","created_at":"2026-07-05T09:15:34.709090+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07779","citing_title":"From Solvers to Research: Large Language Model-Driven Formal Mathematics at the Research Frontier","ref_index":241,"is_internal_anchor":true},{"citing_arxiv_id":"2606.31134","citing_title":"Beyond the Library: An Agentic Framework for Autoformalizing Research Mathematics","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09450","citing_title":"TheoremBench: Evaluating LLMs on Theorem Proving in Formal Mathematics","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06468","citing_title":"Goedel-Architect: Streamlining Formal Theorem Proving with Blueprint Generation and Refinement","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31134","citing_title":"Beyond the Library: An Agentic Framework for Autoformalizing Research Mathematics","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03344","citing_title":"RAG over Thinking Traces Can Improve Reasoning Tasks","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01394","citing_title":"LiveFMBench: Unveiling the Power and Limits of Agentic Workflows in Specification Generation","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EFFH2QNIXN2D3PAZB43ZWXBG6S","json":"https://pith.science/pith/EFFH2QNIXN2D3PAZB43ZWXBG6S.json","graph_json":"https://pith.science/api/pith-number/EFFH2QNIXN2D3PAZB43ZWXBG6S/graph.json","events_json":"https://pith.science/api/pith-number/EFFH2QNIXN2D3PAZB43ZWXBG6S/events.json","paper":"https://pith.science/paper/EFFH2QNI"},"agent_actions":{"view_html":"https://pith.science/pith/EFFH2QNIXN2D3PAZB43ZWXBG6S","download_json":"https://pith.science/pith/EFFH2QNIXN2D3PAZB43ZWXBG6S.json","view_paper":"https://pith.science/paper/EFFH2QNI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.03203&json=true","fetch_graph":"https://pith.science/api/pith-number/EFFH2QNIXN2D3PAZB43ZWXBG6S/graph.json","fetch_events":"https://pith.science/api/pith-number/EFFH2QNIXN2D3PAZB43ZWXBG6S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EFFH2QNIXN2D3PAZB43ZWXBG6S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EFFH2QNIXN2D3PAZB43ZWXBG6S/action/storage_attestation","attest_author":"https://pith.science/pith/EFFH2QNIXN2D3PAZB43ZWXBG6S/action/author_attestation","sign_citation":"https://pith.science/pith/EFFH2QNIXN2D3PAZB43ZWXBG6S/action/citation_signature","submit_replication":"https://pith.science/pith/EFFH2QNIXN2D3PAZB43ZWXBG6S/action/replication_record"}},"created_at":"2026-07-05T09:15:34.709090+00:00","updated_at":"2026-07-05T09:15:34.709090+00:00"}