{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:B4JK4NGZTZQFTG4T4KADYXQUBB","short_pith_number":"pith:B4JK4NGZ","schema_version":"1.0","canonical_sha256":"0f12ae34d99e60599b93e2803c5e14087f3e20c393f281563ed5fc4ab769812a","source":{"kind":"arxiv","id":"2102.06203","version":2},"attestation_state":"computed","paper":{"title":"Proof Artifact Co-training for Theorem Proving with Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG","cs.LO"],"primary_cat":"cs.AI","authors_text":"Edward W. Ayers, Jason Rute, Jesse Michael Han, Stanislas Polu, Yuhuai Wu","submitted_at":"2021-02-11T18:59:24Z","abstract_excerpt":"Labeled data for imitation learning of theorem proving in large libraries of formalized mathematics is scarce as such libraries require years of concentrated effort by human specialists to be built. This is particularly challenging when applying large Transformer language models to tactic prediction, because the scaling of performance with respect to model size is quickly disrupted in the data-scarce, easily-overfitted regime. We propose PACT ({\\bf P}roof {\\bf A}rtifact {\\bf C}o-{\\bf T}raining), a general methodology for extracting abundant self-supervised data from kernel-level proof terms fo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2102.06203","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2021-02-11T18:59:24Z","cross_cats_sorted":["cs.LG","cs.LO"],"title_canon_sha256":"d870d981eb17b002335676860e76c6cb2e6e9a131ce9b6db15d456b9f3a76975","abstract_canon_sha256":"d8cda051818670cbf37ec18e5a769b34601ad76f28f10ca66b3e5aab27f483c9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:05:30.300777Z","signature_b64":"f6GNeGp+ZCG+3QctPcXO1t+X0IriQemgS3jOqOkHbaDnb7mkYtKeHi/vIzg5v5JbGh758eHyqu3BM3m2qM0UAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0f12ae34d99e60599b93e2803c5e14087f3e20c393f281563ed5fc4ab769812a","last_reissued_at":"2026-07-05T04:05:30.300300Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:05:30.300300Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Proof Artifact Co-training for Theorem Proving with Language Models","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG","cs.LO"],"primary_cat":"cs.AI","authors_text":"Edward W. Ayers, Jason Rute, Jesse Michael Han, Stanislas Polu, Yuhuai Wu","submitted_at":"2021-02-11T18:59:24Z","abstract_excerpt":"Labeled data for imitation learning of theorem proving in large libraries of formalized mathematics is scarce as such libraries require years of concentrated effort by human specialists to be built. This is particularly challenging when applying large Transformer language models to tactic prediction, because the scaling of performance with respect to model size is quickly disrupted in the data-scarce, easily-overfitted regime. We propose PACT ({\\bf P}roof {\\bf A}rtifact {\\bf C}o-{\\bf T}raining), a general methodology for extracting abundant self-supervised data from kernel-level proof terms fo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.06203","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.06203/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2102.06203","created_at":"2026-07-05T04:05:30.300373+00:00"},{"alias_kind":"arxiv_version","alias_value":"2102.06203v2","created_at":"2026-07-05T04:05:30.300373+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.06203","created_at":"2026-07-05T04:05:30.300373+00:00"},{"alias_kind":"pith_short_12","alias_value":"B4JK4NGZTZQF","created_at":"2026-07-05T04:05:30.300373+00:00"},{"alias_kind":"pith_short_16","alias_value":"B4JK4NGZTZQFTG4T","created_at":"2026-07-05T04:05:30.300373+00:00"},{"alias_kind":"pith_short_8","alias_value":"B4JK4NGZ","created_at":"2026-07-05T04:05:30.300373+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07779","citing_title":"From Solvers to Research: Large Language Model-Driven Formal Mathematics at the Research Frontier","ref_index":93,"is_internal_anchor":true},{"citing_arxiv_id":"2606.05400","citing_title":"LeanMarathon: Toward Reliable AI Co-Mathematicians through Long-Horizon Lean Autoformalization","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28572","citing_title":"Geometric Measurements of the Axiom of Choice in Neural Proof Embeddings","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27485","citing_title":"Automating Formal Verification with Agent-Guided Tree Search","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30914","citing_title":"Automating Formal Verification with Reinforcement Learning and Recursive Inference","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22874","citing_title":"NeuroNL2LTL: A Neurosymbolic Framework for Natural Language Translation of Linear Temporal Logic","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17283","citing_title":"OProver: A Unified Framework for Agentic Formal Theorem Proving","ref_index":147,"is_internal_anchor":false},{"citing_arxiv_id":"2109.00110","citing_title":"MiniF2F: a cross-system benchmark for formal Olympiad-level mathematics","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2510.01346","citing_title":"Aristotle: IMO-level Automated Theorem Proving","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11905","citing_title":"Rethinking Supervision Granularity: Segment-Level Learning for LLM-Based Theorem Proving","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B4JK4NGZTZQFTG4T4KADYXQUBB","json":"https://pith.science/pith/B4JK4NGZTZQFTG4T4KADYXQUBB.json","graph_json":"https://pith.science/api/pith-number/B4JK4NGZTZQFTG4T4KADYXQUBB/graph.json","events_json":"https://pith.science/api/pith-number/B4JK4NGZTZQFTG4T4KADYXQUBB/events.json","paper":"https://pith.science/paper/B4JK4NGZ"},"agent_actions":{"view_html":"https://pith.science/pith/B4JK4NGZTZQFTG4T4KADYXQUBB","download_json":"https://pith.science/pith/B4JK4NGZTZQFTG4T4KADYXQUBB.json","view_paper":"https://pith.science/paper/B4JK4NGZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2102.06203&json=true","fetch_graph":"https://pith.science/api/pith-number/B4JK4NGZTZQFTG4T4KADYXQUBB/graph.json","fetch_events":"https://pith.science/api/pith-number/B4JK4NGZTZQFTG4T4KADYXQUBB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B4JK4NGZTZQFTG4T4KADYXQUBB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B4JK4NGZTZQFTG4T4KADYXQUBB/action/storage_attestation","attest_author":"https://pith.science/pith/B4JK4NGZTZQFTG4T4KADYXQUBB/action/author_attestation","sign_citation":"https://pith.science/pith/B4JK4NGZTZQFTG4T4KADYXQUBB/action/citation_signature","submit_replication":"https://pith.science/pith/B4JK4NGZTZQFTG4T4KADYXQUBB/action/replication_record"}},"created_at":"2026-07-05T04:05:30.300373+00:00","updated_at":"2026-07-05T04:05:30.300373+00:00"}