{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3ZJ75KGXN7FRKGYILGEPKW4DBY","short_pith_number":"pith:3ZJ75KGX","schema_version":"1.0","canonical_sha256":"de53fea8d76fcb151b085988f55b830e351e29d712e15061794fd84c89c58ccb","source":{"kind":"arxiv","id":"2508.18914","version":1},"attestation_state":"computed","paper":{"title":"FormaRL: Enhancing Autoformalization with no Labeled Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Peng Li, Sijie Liang, Xinling Jin, Yang Liu, Yanxing Huang","submitted_at":"2025-08-26T10:38:18Z","abstract_excerpt":"Autoformalization is one of the central tasks in formal verification, while its advancement remains hindered due to the data scarcity and the absence efficient methods. In this work we propose \\textbf{FormaRL}, a simple yet efficient reinforcement learning framework for autoformalization which only requires a small amount of unlabeled data. FormaRL integrates syntax check from Lean compiler and consistency check from large language model to calculate the reward, and adopts GRPO algorithm to update the formalizer. We also curated a proof problem dataset from undergraduate-level math materials, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.18914","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-08-26T10:38:18Z","cross_cats_sorted":[],"title_canon_sha256":"0fdd6ffe8fe5f6537143e6671b12fcc75be4412c485108f55b365242e8d569b4","abstract_canon_sha256":"047c40589143b865634d03c621947dfba73b5f48a9a563e98d722c82fadf3f6f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:59:33.650039Z","signature_b64":"WIayoZCEYUwDqbMmZNCFKaI8lUNYZxKi6At7Q1njaprjZ3YoFLUi+YAByBEYei2X0ehvtePifdW8oCty2vr3AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"de53fea8d76fcb151b085988f55b830e351e29d712e15061794fd84c89c58ccb","last_reissued_at":"2026-07-05T11:59:33.649483Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:59:33.649483Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FormaRL: Enhancing Autoformalization with no Labeled Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Peng Li, Sijie Liang, Xinling Jin, Yang Liu, Yanxing Huang","submitted_at":"2025-08-26T10:38:18Z","abstract_excerpt":"Autoformalization is one of the central tasks in formal verification, while its advancement remains hindered due to the data scarcity and the absence efficient methods. In this work we propose \\textbf{FormaRL}, a simple yet efficient reinforcement learning framework for autoformalization which only requires a small amount of unlabeled data. FormaRL integrates syntax check from Lean compiler and consistency check from large language model to calculate the reward, and adopts GRPO algorithm to update the formalizer. We also curated a proof problem dataset from undergraduate-level math materials, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.18914","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.18914/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.18914","created_at":"2026-07-05T11:59:33.649540+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.18914v1","created_at":"2026-07-05T11:59:33.649540+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.18914","created_at":"2026-07-05T11:59:33.649540+00:00"},{"alias_kind":"pith_short_12","alias_value":"3ZJ75KGXN7FR","created_at":"2026-07-05T11:59:33.649540+00:00"},{"alias_kind":"pith_short_16","alias_value":"3ZJ75KGXN7FRKGYI","created_at":"2026-07-05T11:59:33.649540+00:00"},{"alias_kind":"pith_short_8","alias_value":"3ZJ75KGX","created_at":"2026-07-05T11:59:33.649540+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.08827","citing_title":"A Survey of Reinforcement Learning for Large Reasoning Models","ref_index":213,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13515","citing_title":"SFT-GRPO Data Overlap as a Post-Training Hyperparameter for Autoformalization","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3ZJ75KGXN7FRKGYILGEPKW4DBY","json":"https://pith.science/pith/3ZJ75KGXN7FRKGYILGEPKW4DBY.json","graph_json":"https://pith.science/api/pith-number/3ZJ75KGXN7FRKGYILGEPKW4DBY/graph.json","events_json":"https://pith.science/api/pith-number/3ZJ75KGXN7FRKGYILGEPKW4DBY/events.json","paper":"https://pith.science/paper/3ZJ75KGX"},"agent_actions":{"view_html":"https://pith.science/pith/3ZJ75KGXN7FRKGYILGEPKW4DBY","download_json":"https://pith.science/pith/3ZJ75KGXN7FRKGYILGEPKW4DBY.json","view_paper":"https://pith.science/paper/3ZJ75KGX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.18914&json=true","fetch_graph":"https://pith.science/api/pith-number/3ZJ75KGXN7FRKGYILGEPKW4DBY/graph.json","fetch_events":"https://pith.science/api/pith-number/3ZJ75KGXN7FRKGYILGEPKW4DBY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3ZJ75KGXN7FRKGYILGEPKW4DBY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3ZJ75KGXN7FRKGYILGEPKW4DBY/action/storage_attestation","attest_author":"https://pith.science/pith/3ZJ75KGXN7FRKGYILGEPKW4DBY/action/author_attestation","sign_citation":"https://pith.science/pith/3ZJ75KGXN7FRKGYILGEPKW4DBY/action/citation_signature","submit_replication":"https://pith.science/pith/3ZJ75KGXN7FRKGYILGEPKW4DBY/action/replication_record"}},"created_at":"2026-07-05T11:59:33.649540+00:00","updated_at":"2026-07-05T11:59:33.649540+00:00"}