{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2VIV47NJXTHS7DVXWFRWSRBTFP","short_pith_number":"pith:2VIV47NJ","schema_version":"1.0","canonical_sha256":"d5515e7da9bccf2f8eb7b1636944332bd23dcfff886deb5c78faa6624a821ab4","source":{"kind":"arxiv","id":"2506.18254","version":1},"attestation_state":"computed","paper":{"title":"RLPR: Extrapolating RLVR to General Domains without Verifiers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Bo Ji, Ganqu Cui, Lifan Yuan, Maosong Sun, Ning Ding, Shouli Wang, Shu Yao, Tat-Seng Chua, Tianyu Yu, Yuan Yao, Zefan Wang, Zhiyuan Liu","submitted_at":"2025-06-23T02:56:36Z","abstract_excerpt":"Reinforcement Learning with Verifiable Rewards (RLVR) demonstrates promising potential in advancing the reasoning capabilities of LLMs. However, its success remains largely confined to mathematical and code domains. This primary limitation stems from the heavy reliance on domain-specific verifiers, which results in prohibitive complexity and limited scalability. To address the challenge, our key observation is that LLM's intrinsic probability of generating a correct free-form answer directly indicates its own evaluation of the reasoning reward (i.e., how well the reasoning process leads to the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.18254","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-06-23T02:56:36Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"2a17d9a66fb4493818deda806861f729af3dce26117e57f222ee02b2ece62f12","abstract_canon_sha256":"3cd638ae934970c58da4a26d05593f878910d397c08668756d45325974f096dd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:25:47.026370Z","signature_b64":"Si/MyBhQfDfQKR11LP/Y1kyEvkOAtGyFa36euPSeDA9GIWoMv6f4Xc4mTxNvDPfOtYOWnRL0QptdBlDbzlORDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d5515e7da9bccf2f8eb7b1636944332bd23dcfff886deb5c78faa6624a821ab4","last_reissued_at":"2026-07-05T11:25:47.025910Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:25:47.025910Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RLPR: Extrapolating RLVR to General Domains without Verifiers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Bo Ji, Ganqu Cui, Lifan Yuan, Maosong Sun, Ning Ding, Shouli Wang, Shu Yao, Tat-Seng Chua, Tianyu Yu, Yuan Yao, Zefan Wang, Zhiyuan Liu","submitted_at":"2025-06-23T02:56:36Z","abstract_excerpt":"Reinforcement Learning with Verifiable Rewards (RLVR) demonstrates promising potential in advancing the reasoning capabilities of LLMs. However, its success remains largely confined to mathematical and code domains. This primary limitation stems from the heavy reliance on domain-specific verifiers, which results in prohibitive complexity and limited scalability. To address the challenge, our key observation is that LLM's intrinsic probability of generating a correct free-form answer directly indicates its own evaluation of the reasoning reward (i.e., how well the reasoning process leads to the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.18254","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.18254/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.18254","created_at":"2026-07-05T11:25:47.025969+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.18254v1","created_at":"2026-07-05T11:25:47.025969+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.18254","created_at":"2026-07-05T11:25:47.025969+00:00"},{"alias_kind":"pith_short_12","alias_value":"2VIV47NJXTHS","created_at":"2026-07-05T11:25:47.025969+00:00"},{"alias_kind":"pith_short_16","alias_value":"2VIV47NJXTHS7DVX","created_at":"2026-07-05T11:25:47.025969+00:00"},{"alias_kind":"pith_short_8","alias_value":"2VIV47NJ","created_at":"2026-07-05T11:25:47.025969+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21228","citing_title":"Sakana Fugu Technical Report","ref_index":243,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09380","citing_title":"Reasoning Arena: Trace Tournaments When Verifiable Rewards Fall Short","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09393","citing_title":"CapRL++: Unified Reinforcement Learning with Verifiable Rewards for Dense Image and Video Captioning","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04516","citing_title":"GeoMin: Data-Efficient Semi-Supervised RLVR via Geometric Distribution Modeling","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31058","citing_title":"Combinatorial Synthesis: Scaling Code RLVR via Atomic Decomposition and Recombination","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00609","citing_title":"CARE-RL: Capability-Aware Reinforcement Learning for Mitigating Cross-Domain Conflicts","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21235","citing_title":"LamPO: A Lambda Style Policy Optimization for Reasoning Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10810","citing_title":"Likelihood scoring for continuations of mathematical text: a self-supervised benchmark with tests for shortcut vulnerabilities","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2509.14234","citing_title":"Compute as Teacher: Turning Inference Compute Into Reference-Free Supervision","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2512.04695","citing_title":"TRINITY: An Evolved LLM Coordinator","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11505","citing_title":"Selective Off-Policy Reference Tuning with Plan Guidance","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11505","citing_title":"Selective Off-Policy Reference Tuning with Plan Guidance","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09959","citing_title":"G-Zero: Self-Play for Open-Ended Generation from Zero Data","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10110","citing_title":"Trust Your Memory: Verifiable Control of Smart Homes through Reinforcement Learning with Multi-dimensional Rewards","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08905","citing_title":"StaRPO: Stability-Augmented Reinforcement Policy Optimization","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02913","citing_title":"Generate, Filter, Control, Replay: A Comprehensive Survey of Rollout Strategies for LLM Reinforcement Learning","ref_index":160,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17928","citing_title":"HEALing Entropy Collapse: Enhancing Exploration in Few-Shot RLVR via Hybrid-Domain Entropy Dynamics Alignment","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2VIV47NJXTHS7DVXWFRWSRBTFP","json":"https://pith.science/pith/2VIV47NJXTHS7DVXWFRWSRBTFP.json","graph_json":"https://pith.science/api/pith-number/2VIV47NJXTHS7DVXWFRWSRBTFP/graph.json","events_json":"https://pith.science/api/pith-number/2VIV47NJXTHS7DVXWFRWSRBTFP/events.json","paper":"https://pith.science/paper/2VIV47NJ"},"agent_actions":{"view_html":"https://pith.science/pith/2VIV47NJXTHS7DVXWFRWSRBTFP","download_json":"https://pith.science/pith/2VIV47NJXTHS7DVXWFRWSRBTFP.json","view_paper":"https://pith.science/paper/2VIV47NJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.18254&json=true","fetch_graph":"https://pith.science/api/pith-number/2VIV47NJXTHS7DVXWFRWSRBTFP/graph.json","fetch_events":"https://pith.science/api/pith-number/2VIV47NJXTHS7DVXWFRWSRBTFP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2VIV47NJXTHS7DVXWFRWSRBTFP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2VIV47NJXTHS7DVXWFRWSRBTFP/action/storage_attestation","attest_author":"https://pith.science/pith/2VIV47NJXTHS7DVXWFRWSRBTFP/action/author_attestation","sign_citation":"https://pith.science/pith/2VIV47NJXTHS7DVXWFRWSRBTFP/action/citation_signature","submit_replication":"https://pith.science/pith/2VIV47NJXTHS7DVXWFRWSRBTFP/action/replication_record"}},"created_at":"2026-07-05T11:25:47.025969+00:00","updated_at":"2026-07-05T11:25:47.025969+00:00"}