{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FJMVCPCPC53U3MFX67UGVG7PZK","short_pith_number":"pith:FJMVCPCP","schema_version":"1.0","canonical_sha256":"2a59513c4f17774db0b7f7e86a9befcab6542a41b7c017a38226699145cf3fbd","source":{"kind":"arxiv","id":"2502.06563","version":2},"attestation_state":"computed","paper":{"title":"Large Language Models Meet Symbolic Provers for Logical Reasoning Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Binyuan Hui, Bowen Li, Chengwen Qi, Conghui He, He Du, Jinwang Wu, Ren Ma, Yuanjun Laili","submitted_at":"2025-02-10T15:31:54Z","abstract_excerpt":"First-order logic (FOL) reasoning, which involves sequential deduction, is pivotal for intelligent systems and serves as a valuable task for evaluating reasoning capabilities, particularly in chain-of-thought (CoT) contexts. Existing benchmarks often rely on extensive human annotation or handcrafted templates, making it difficult to achieve the necessary complexity, scalability, and diversity for robust evaluation. To address these limitations, we propose a novel framework called ProverGen that synergizes the generative strengths of Large Language Models (LLMs) with the rigor and precision of "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.06563","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-10T15:31:54Z","cross_cats_sorted":[],"title_canon_sha256":"11cb8bbc3ac8f2e9746e5f2890cefdaf226b8fc121c60125c98d9ed34f630e82","abstract_canon_sha256":"44314cf278a9c9ba4801a21ea3094a59ad9bfe22f75a1ef36290ee9de8404581"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:13.036908Z","signature_b64":"vTG0rWrS4Oyy48W+9RlA/V8sJEUIbWpjKb+hWALW3uuK5T6/M6LWRhM6j7iDGPXoCmqR7L4Mhnki8rDltDu9Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2a59513c4f17774db0b7f7e86a9befcab6542a41b7c017a38226699145cf3fbd","last_reissued_at":"2026-07-05T10:22:13.036387Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:13.036387Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Models Meet Symbolic Provers for Logical Reasoning Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Binyuan Hui, Bowen Li, Chengwen Qi, Conghui He, He Du, Jinwang Wu, Ren Ma, Yuanjun Laili","submitted_at":"2025-02-10T15:31:54Z","abstract_excerpt":"First-order logic (FOL) reasoning, which involves sequential deduction, is pivotal for intelligent systems and serves as a valuable task for evaluating reasoning capabilities, particularly in chain-of-thought (CoT) contexts. Existing benchmarks often rely on extensive human annotation or handcrafted templates, making it difficult to achieve the necessary complexity, scalability, and diversity for robust evaluation. To address these limitations, we propose a novel framework called ProverGen that synergizes the generative strengths of Large Language Models (LLMs) with the rigor and precision of "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.06563","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.06563/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.06563","created_at":"2026-07-05T10:22:13.036455+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.06563v2","created_at":"2026-07-05T10:22:13.036455+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.06563","created_at":"2026-07-05T10:22:13.036455+00:00"},{"alias_kind":"pith_short_12","alias_value":"FJMVCPCPC53U","created_at":"2026-07-05T10:22:13.036455+00:00"},{"alias_kind":"pith_short_16","alias_value":"FJMVCPCPC53U3MFX","created_at":"2026-07-05T10:22:13.036455+00:00"},{"alias_kind":"pith_short_8","alias_value":"FJMVCPCP","created_at":"2026-07-05T10:22:13.036455+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.24765","citing_title":"Semantic-Aware Logical Reasoning via a Semiotic Framework","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25155","citing_title":"Rethinking Wireless Communications through Formal Mathematical AI Reasoning","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18873","citing_title":"From Natural Language to Executable Narsese: A Neuro-Symbolic Benchmark and Pipeline for Reasoning with NARS","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FJMVCPCPC53U3MFX67UGVG7PZK","json":"https://pith.science/pith/FJMVCPCPC53U3MFX67UGVG7PZK.json","graph_json":"https://pith.science/api/pith-number/FJMVCPCPC53U3MFX67UGVG7PZK/graph.json","events_json":"https://pith.science/api/pith-number/FJMVCPCPC53U3MFX67UGVG7PZK/events.json","paper":"https://pith.science/paper/FJMVCPCP"},"agent_actions":{"view_html":"https://pith.science/pith/FJMVCPCPC53U3MFX67UGVG7PZK","download_json":"https://pith.science/pith/FJMVCPCPC53U3MFX67UGVG7PZK.json","view_paper":"https://pith.science/paper/FJMVCPCP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.06563&json=true","fetch_graph":"https://pith.science/api/pith-number/FJMVCPCPC53U3MFX67UGVG7PZK/graph.json","fetch_events":"https://pith.science/api/pith-number/FJMVCPCPC53U3MFX67UGVG7PZK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FJMVCPCPC53U3MFX67UGVG7PZK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FJMVCPCPC53U3MFX67UGVG7PZK/action/storage_attestation","attest_author":"https://pith.science/pith/FJMVCPCPC53U3MFX67UGVG7PZK/action/author_attestation","sign_citation":"https://pith.science/pith/FJMVCPCPC53U3MFX67UGVG7PZK/action/citation_signature","submit_replication":"https://pith.science/pith/FJMVCPCPC53U3MFX67UGVG7PZK/action/replication_record"}},"created_at":"2026-07-05T10:22:13.036455+00:00","updated_at":"2026-07-05T10:22:13.036455+00:00"}