{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:UW7JGW5DCRDYR2SW6THSCR5L6W","short_pith_number":"pith:UW7JGW5D","schema_version":"1.0","canonical_sha256":"a5be935ba3144788ea56f4cf2147abf5810ce0d94f33c2201795557865eb9a96","source":{"kind":"arxiv","id":"2302.12433","version":1},"attestation_state":"computed","paper":{"title":"ProofNet: Autoformalizing and Formally Proving Undergraduate-Level Mathematics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LO"],"primary_cat":"cs.CL","authors_text":"Bartosz Piotrowski, Dragomir Radev, Edward W. Ayers, Hailey Schoelkopf, Jeremy Avigad, Zhangir Azerbayev","submitted_at":"2023-02-24T03:28:46Z","abstract_excerpt":"We introduce ProofNet, a benchmark for autoformalization and formal proving of undergraduate-level mathematics. The ProofNet benchmarks consists of 371 examples, each consisting of a formal theorem statement in Lean 3, a natural language theorem statement, and a natural language proof. The problems are primarily drawn from popular undergraduate pure mathematics textbooks and cover topics such as real and complex analysis, linear algebra, abstract algebra, and topology. We intend for ProofNet to be a challenging benchmark that will drive progress in autoformalization and automatic theorem provi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.12433","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-02-24T03:28:46Z","cross_cats_sorted":["cs.AI","cs.LO"],"title_canon_sha256":"728d6144db6efd95a3d99a6e2fc01565fbb97e15001b078aeb47c90491dd41cf","abstract_canon_sha256":"f230e8962a69c0636257b8a7323dd1d2a471b7713ee65b8e457b96f4b13eace8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:45:09.694572Z","signature_b64":"CD+iplIVC7Nq8R49Y+46Iv0Cd/1s6R0z3KIibjiiZmSEEe9DJoX5GznpM3lp1Sf3ZL4ss+MxBWrC0crcHKd+Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a5be935ba3144788ea56f4cf2147abf5810ce0d94f33c2201795557865eb9a96","last_reissued_at":"2026-07-05T05:45:09.694048Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:45:09.694048Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ProofNet: Autoformalizing and Formally Proving Undergraduate-Level Mathematics","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LO"],"primary_cat":"cs.CL","authors_text":"Bartosz Piotrowski, Dragomir Radev, Edward W. Ayers, Hailey Schoelkopf, Jeremy Avigad, Zhangir Azerbayev","submitted_at":"2023-02-24T03:28:46Z","abstract_excerpt":"We introduce ProofNet, a benchmark for autoformalization and formal proving of undergraduate-level mathematics. The ProofNet benchmarks consists of 371 examples, each consisting of a formal theorem statement in Lean 3, a natural language theorem statement, and a natural language proof. The problems are primarily drawn from popular undergraduate pure mathematics textbooks and cover topics such as real and complex analysis, linear algebra, abstract algebra, and topology. We intend for ProofNet to be a challenging benchmark that will drive progress in autoformalization and automatic theorem provi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.12433","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.12433/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.12433","created_at":"2026-07-05T05:45:09.694108+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.12433v1","created_at":"2026-07-05T05:45:09.694108+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.12433","created_at":"2026-07-05T05:45:09.694108+00:00"},{"alias_kind":"pith_short_12","alias_value":"UW7JGW5DCRDY","created_at":"2026-07-05T05:45:09.694108+00:00"},{"alias_kind":"pith_short_16","alias_value":"UW7JGW5DCRDYR2SW","created_at":"2026-07-05T05:45:09.694108+00:00"},{"alias_kind":"pith_short_8","alias_value":"UW7JGW5D","created_at":"2026-07-05T05:45:09.694108+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":36,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26246","citing_title":"Lacuna: A Research Map for Machine Learning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22873","citing_title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","ref_index":244,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18557","citing_title":"DeFAb: A Verifiable Benchmark for Defeasible Abduction in Foundation Models","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18121","citing_title":"On the Reliability of Networks of AI Agents: Density Evolution, Stopping Sets, and Architecture Optimization","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09450","citing_title":"TheoremBench: Evaluating LLMs on Theorem Proving in Formal Mathematics","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09449","citing_title":"Reasoning without Gold Standards: A Proxy-Judge Theory of Autoformalization","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08728","citing_title":"Artificial Intelligence for Mathematical Reasoning: An Integrated Survey of Language Models, Neuro-symbolic Systems, and Verified Discovery","ref_index":202,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22873","citing_title":"SingGuard: A Policy-Adaptive Multimodal LLM Guardrail with Dynamic Reasoning","ref_index":243,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28013","citing_title":"The Signal-Coverage Matrix: Stratifying Type and Semantic Errors in Statement Autoformalization","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31002","citing_title":"Beyond Compilation: Evaluating Faithful Natural-Language-to-Lean Statement Formalization","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10379","citing_title":"Not All Proofs Are Equal: Evaluating LLM Proof Quality Beyond Correctness","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02588","citing_title":"Lean-GAP: A Dataset of Formalized Graduate Algebra Problems","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28841","citing_title":"LAMP: Lean-based Agentic framework with MCP and Proof Repair","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29955","citing_title":"Formalizing Mathematics at Scale","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30914","citing_title":"Automating Formal Verification with Reinforcement Learning and Recursive Inference","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01008","citing_title":"FVSpec: Real-World Property-Based Tests as Lean Challenges","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17289","citing_title":"Nothing from Something: Can a Language Model Discover 0?","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25561","citing_title":"CrypFormBench: Benchmarking Formal Analysis Capability of Large Language Models for Cryptographic Schemes","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2411.18104","citing_title":"Training and Evaluating Language Models with Template-based Data Generation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2504.05605","citing_title":"ShadowCoT: Cognitive Hijacking for Stealthy Reasoning Backdoors in LLMs","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23135","citing_title":"Characterizing Paraphrase-Induced Failures in Lean 4 Autoformalization","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17283","citing_title":"OProver: A Unified Framework for Agentic Formal Theorem Proving","ref_index":163,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17255","citing_title":"CAM-Bench: A Benchmark for Computational and Applied Mathematics in Lean","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2310.10631","citing_title":"Llemma: An Open Language Model For Mathematics","ref_index":123,"is_internal_anchor":false},{"citing_arxiv_id":"2601.13209","citing_title":"AI for Mathematics: Progress, Challenges, and Prospects","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UW7JGW5DCRDYR2SW6THSCR5L6W","json":"https://pith.science/pith/UW7JGW5DCRDYR2SW6THSCR5L6W.json","graph_json":"https://pith.science/api/pith-number/UW7JGW5DCRDYR2SW6THSCR5L6W/graph.json","events_json":"https://pith.science/api/pith-number/UW7JGW5DCRDYR2SW6THSCR5L6W/events.json","paper":"https://pith.science/paper/UW7JGW5D"},"agent_actions":{"view_html":"https://pith.science/pith/UW7JGW5DCRDYR2SW6THSCR5L6W","download_json":"https://pith.science/pith/UW7JGW5DCRDYR2SW6THSCR5L6W.json","view_paper":"https://pith.science/paper/UW7JGW5D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.12433&json=true","fetch_graph":"https://pith.science/api/pith-number/UW7JGW5DCRDYR2SW6THSCR5L6W/graph.json","fetch_events":"https://pith.science/api/pith-number/UW7JGW5DCRDYR2SW6THSCR5L6W/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UW7JGW5DCRDYR2SW6THSCR5L6W/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UW7JGW5DCRDYR2SW6THSCR5L6W/action/storage_attestation","attest_author":"https://pith.science/pith/UW7JGW5DCRDYR2SW6THSCR5L6W/action/author_attestation","sign_citation":"https://pith.science/pith/UW7JGW5DCRDYR2SW6THSCR5L6W/action/citation_signature","submit_replication":"https://pith.science/pith/UW7JGW5DCRDYR2SW6THSCR5L6W/action/replication_record"}},"created_at":"2026-07-05T05:45:09.694108+00:00","updated_at":"2026-07-05T05:45:09.694108+00:00"}