{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CTXNKC4RKCMHMVQE6BTYPEPIKV","short_pith_number":"pith:CTXNKC4R","schema_version":"1.0","canonical_sha256":"14eed50b915098765604f0678791e855538564e932961c1a172effba1d46d119","source":{"kind":"arxiv","id":"2508.02208","version":2},"attestation_state":"computed","paper":{"title":"Proof2Hybrid: Automatic Mathematical Benchmark Synthesis for Proof-Centric Problems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bowen Ye, Tong Yang, Weijun Yuan, Xinye Xu, Yaoming Li, Yebo Peng, Zhizhuo Yang, Zihan Wang, Zixiang Liu","submitted_at":"2025-08-04T08:59:36Z","abstract_excerpt":"Evaluating the mathematical capability of Large Language Models (LLMs) is a critical yet challenging frontier. Existing benchmarks fall short, particularly for proof-centric problems, as manual creation is unscalable and costly, leaving the true mathematical abilities of LLMs largely unassessed. To overcome these barriers, we propose Proof2Hybrid, the first fully automated framework that synthesizes high-quality, proof-centric benchmarks from natural language mathematical corpora. The key novelty of our solution is Proof2X, a roadmap of converting mathematical proofs into various kinds of ques"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.02208","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-08-04T08:59:36Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"077ccdd11d965652101fbce9ec1ac2030c58320fde4b6b2136fe9ecb5d6005f6","abstract_canon_sha256":"c4c6939114bb1dbed74c3e3377c0c4c16148c94b22f9c41895726d099dfd83c7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:48:52.508082Z","signature_b64":"qMSHCu6H/v8+uCphAmZs8KCOHFWq4C7Svup4GXuwtwavzc66eZoSVU4/6WyIyVpB8QP4A5lwz8wPmAvb0h6tDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"14eed50b915098765604f0678791e855538564e932961c1a172effba1d46d119","last_reissued_at":"2026-07-05T11:48:52.507539Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:48:52.507539Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Proof2Hybrid: Automatic Mathematical Benchmark Synthesis for Proof-Centric Problems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bowen Ye, Tong Yang, Weijun Yuan, Xinye Xu, Yaoming Li, Yebo Peng, Zhizhuo Yang, Zihan Wang, Zixiang Liu","submitted_at":"2025-08-04T08:59:36Z","abstract_excerpt":"Evaluating the mathematical capability of Large Language Models (LLMs) is a critical yet challenging frontier. Existing benchmarks fall short, particularly for proof-centric problems, as manual creation is unscalable and costly, leaving the true mathematical abilities of LLMs largely unassessed. To overcome these barriers, we propose Proof2Hybrid, the first fully automated framework that synthesizes high-quality, proof-centric benchmarks from natural language mathematical corpora. The key novelty of our solution is Proof2X, a roadmap of converting mathematical proofs into various kinds of ques"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.02208","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.02208/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.02208","created_at":"2026-07-05T11:48:52.507597+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.02208v2","created_at":"2026-07-05T11:48:52.507597+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.02208","created_at":"2026-07-05T11:48:52.507597+00:00"},{"alias_kind":"pith_short_12","alias_value":"CTXNKC4RKCMH","created_at":"2026-07-05T11:48:52.507597+00:00"},{"alias_kind":"pith_short_16","alias_value":"CTXNKC4RKCMHMVQE","created_at":"2026-07-05T11:48:52.507597+00:00"},{"alias_kind":"pith_short_8","alias_value":"CTXNKC4R","created_at":"2026-07-05T11:48:52.507597+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.04386","citing_title":"Automatically Generating Hard Math Problems from Hypothesis-Driven Error Analysis","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CTXNKC4RKCMHMVQE6BTYPEPIKV","json":"https://pith.science/pith/CTXNKC4RKCMHMVQE6BTYPEPIKV.json","graph_json":"https://pith.science/api/pith-number/CTXNKC4RKCMHMVQE6BTYPEPIKV/graph.json","events_json":"https://pith.science/api/pith-number/CTXNKC4RKCMHMVQE6BTYPEPIKV/events.json","paper":"https://pith.science/paper/CTXNKC4R"},"agent_actions":{"view_html":"https://pith.science/pith/CTXNKC4RKCMHMVQE6BTYPEPIKV","download_json":"https://pith.science/pith/CTXNKC4RKCMHMVQE6BTYPEPIKV.json","view_paper":"https://pith.science/paper/CTXNKC4R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.02208&json=true","fetch_graph":"https://pith.science/api/pith-number/CTXNKC4RKCMHMVQE6BTYPEPIKV/graph.json","fetch_events":"https://pith.science/api/pith-number/CTXNKC4RKCMHMVQE6BTYPEPIKV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CTXNKC4RKCMHMVQE6BTYPEPIKV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CTXNKC4RKCMHMVQE6BTYPEPIKV/action/storage_attestation","attest_author":"https://pith.science/pith/CTXNKC4RKCMHMVQE6BTYPEPIKV/action/author_attestation","sign_citation":"https://pith.science/pith/CTXNKC4RKCMHMVQE6BTYPEPIKV/action/citation_signature","submit_replication":"https://pith.science/pith/CTXNKC4RKCMHMVQE6BTYPEPIKV/action/replication_record"}},"created_at":"2026-07-05T11:48:52.507597+00:00","updated_at":"2026-07-05T11:48:52.507597+00:00"}