{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:K5GWBL227W4DB3TV35BCB2CLHJ","short_pith_number":"pith:K5GWBL22","schema_version":"1.0","canonical_sha256":"574d60af5afdb830ee75df4220e84b3a644212496249fa80c12270bb35f7aab6","source":{"kind":"arxiv","id":"2411.01829","version":1},"attestation_state":"computed","paper":{"title":"Formal Theorem Proving by Rewarding LLMs to Decompose Proofs Hierarchically","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Arvind Mahankali, Kefan Dong, Tengyu Ma","submitted_at":"2024-11-04T05:57:40Z","abstract_excerpt":"Mathematical theorem proving is an important testbed for large language models' deep and abstract reasoning capability. This paper focuses on improving LLMs' ability to write proofs in formal languages that permit automated proof verification/evaluation. Most previous results provide human-written lemmas to the theorem prover, which is an arguably oversimplified setting that does not sufficiently test the provers' planning and decomposition capabilities. Instead, we work in a more natural setup where the lemmas that are directly relevant to the theorem are not given to the theorem prover at te"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.01829","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-04T05:57:40Z","cross_cats_sorted":[],"title_canon_sha256":"f434035697a9f229f1c6a38b3e8517da3e0c1ff886ed43db62870d6937d83bbf","abstract_canon_sha256":"14df0c10d5155ca1376efde11dcc961d7098e7ae66d6d885a230af71f9e39530"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:30:37.050884Z","signature_b64":"JAQp0bT1YemQCN1IvsAuOQLbPltPKNPSNjli1PaP24FhmwWPgLygGEvTXGVoVRzDOSNrgkAVOXKoB26XNNDjBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"574d60af5afdb830ee75df4220e84b3a644212496249fa80c12270bb35f7aab6","last_reissued_at":"2026-07-05T09:30:37.050422Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:30:37.050422Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Formal Theorem Proving by Rewarding LLMs to Decompose Proofs Hierarchically","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Arvind Mahankali, Kefan Dong, Tengyu Ma","submitted_at":"2024-11-04T05:57:40Z","abstract_excerpt":"Mathematical theorem proving is an important testbed for large language models' deep and abstract reasoning capability. This paper focuses on improving LLMs' ability to write proofs in formal languages that permit automated proof verification/evaluation. Most previous results provide human-written lemmas to the theorem prover, which is an arguably oversimplified setting that does not sufficiently test the provers' planning and decomposition capabilities. Instead, we work in a more natural setup where the lemmas that are directly relevant to the theorem are not given to the theorem prover at te"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.01829","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.01829/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.01829","created_at":"2026-07-05T09:30:37.050480+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.01829v1","created_at":"2026-07-05T09:30:37.050480+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.01829","created_at":"2026-07-05T09:30:37.050480+00:00"},{"alias_kind":"pith_short_12","alias_value":"K5GWBL227W4D","created_at":"2026-07-05T09:30:37.050480+00:00"},{"alias_kind":"pith_short_16","alias_value":"K5GWBL227W4DB3TV","created_at":"2026-07-05T09:30:37.050480+00:00"},{"alias_kind":"pith_short_8","alias_value":"K5GWBL22","created_at":"2026-07-05T09:30:37.050480+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07779","citing_title":"From Solvers to Research: Large Language Model-Driven Formal Mathematics at the Research Frontier","ref_index":62,"is_internal_anchor":true},{"citing_arxiv_id":"2604.16278","citing_title":"Learning to Reason with Insight for Informal Theorem Proving","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K5GWBL227W4DB3TV35BCB2CLHJ","json":"https://pith.science/pith/K5GWBL227W4DB3TV35BCB2CLHJ.json","graph_json":"https://pith.science/api/pith-number/K5GWBL227W4DB3TV35BCB2CLHJ/graph.json","events_json":"https://pith.science/api/pith-number/K5GWBL227W4DB3TV35BCB2CLHJ/events.json","paper":"https://pith.science/paper/K5GWBL22"},"agent_actions":{"view_html":"https://pith.science/pith/K5GWBL227W4DB3TV35BCB2CLHJ","download_json":"https://pith.science/pith/K5GWBL227W4DB3TV35BCB2CLHJ.json","view_paper":"https://pith.science/paper/K5GWBL22","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.01829&json=true","fetch_graph":"https://pith.science/api/pith-number/K5GWBL227W4DB3TV35BCB2CLHJ/graph.json","fetch_events":"https://pith.science/api/pith-number/K5GWBL227W4DB3TV35BCB2CLHJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K5GWBL227W4DB3TV35BCB2CLHJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K5GWBL227W4DB3TV35BCB2CLHJ/action/storage_attestation","attest_author":"https://pith.science/pith/K5GWBL227W4DB3TV35BCB2CLHJ/action/author_attestation","sign_citation":"https://pith.science/pith/K5GWBL227W4DB3TV35BCB2CLHJ/action/citation_signature","submit_replication":"https://pith.science/pith/K5GWBL227W4DB3TV35BCB2CLHJ/action/replication_record"}},"created_at":"2026-07-05T09:30:37.050480+00:00","updated_at":"2026-07-05T09:30:37.050480+00:00"}