{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LLK7TYZXSRII33SIHT4JVMONJ3","short_pith_number":"pith:LLK7TYZX","schema_version":"1.0","canonical_sha256":"5ad5f9e33794508dee483cf89ab1cd4ec74e5a35053427d637710f8d7445cbba","source":{"kind":"arxiv","id":"2508.12790","version":1},"attestation_state":"computed","paper":{"title":"Reinforcement Learning with Rubric Anchors","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Guoshan Lu, Haokai Xu, Jianguo Li, Jiaqi Hu, Jiaxin Liu, Junbo Zhao, Peiyi Tu, Ru Peng, Tianyu Zhao, Wenyu Chen, Xiaomeng Hu, Xijun Gu, Yanmei Gu, Yihong Zhuang, Yuanyuan Wang, Yuzhuo Fu, Zenan Huang, Zeyu Qin, Zhanming Shen, Zhengkai Yang, Zhiting Fan","submitted_at":"2025-08-18T10:06:08Z","abstract_excerpt":"Reinforcement Learning from Verifiable Rewards (RLVR) has emerged as a powerful paradigm for enhancing Large Language Models (LLMs), exemplified by the success of OpenAI's o-series. In RLVR, rewards are derived from verifiable signals-such as passing unit tests in code generation or matching correct answers in mathematical reasoning. While effective, this requirement largely confines RLVR to domains with automatically checkable outcomes. To overcome this, we extend the RLVR paradigm to open-ended tasks by integrating rubric-based rewards, where carefully designed rubrics serve as structured, m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.12790","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.AI","submitted_at":"2025-08-18T10:06:08Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"8f87e9d165e5ec2859ddecc94483bfa375cb2737ecd356ecf3aac0cd604badda","abstract_canon_sha256":"afe0a52e4d30f07f695c3233e1c4c4362fec67ab7201af40321b3337701dbc2f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:55:27.282186Z","signature_b64":"yLLt1ELxzGDDZO5QNa5m9ZY2gxPrLAX2D6KznjkwXnkEa5bNG++MCY401qQz1ZxjLXq8So3WvG4LTISkihanDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5ad5f9e33794508dee483cf89ab1cd4ec74e5a35053427d637710f8d7445cbba","last_reissued_at":"2026-07-05T11:55:27.281686Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:55:27.281686Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reinforcement Learning with Rubric Anchors","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Guoshan Lu, Haokai Xu, Jianguo Li, Jiaqi Hu, Jiaxin Liu, Junbo Zhao, Peiyi Tu, Ru Peng, Tianyu Zhao, Wenyu Chen, Xiaomeng Hu, Xijun Gu, Yanmei Gu, Yihong Zhuang, Yuanyuan Wang, Yuzhuo Fu, Zenan Huang, Zeyu Qin, Zhanming Shen, Zhengkai Yang, Zhiting Fan","submitted_at":"2025-08-18T10:06:08Z","abstract_excerpt":"Reinforcement Learning from Verifiable Rewards (RLVR) has emerged as a powerful paradigm for enhancing Large Language Models (LLMs), exemplified by the success of OpenAI's o-series. In RLVR, rewards are derived from verifiable signals-such as passing unit tests in code generation or matching correct answers in mathematical reasoning. While effective, this requirement largely confines RLVR to domains with automatically checkable outcomes. To overcome this, we extend the RLVR paradigm to open-ended tasks by integrating rubric-based rewards, where carefully designed rubrics serve as structured, m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.12790","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.12790/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.12790","created_at":"2026-07-05T11:55:27.281748+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.12790v1","created_at":"2026-07-05T11:55:27.281748+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.12790","created_at":"2026-07-05T11:55:27.281748+00:00"},{"alias_kind":"pith_short_12","alias_value":"LLK7TYZXSRII","created_at":"2026-07-05T11:55:27.281748+00:00"},{"alias_kind":"pith_short_16","alias_value":"LLK7TYZXSRII33SI","created_at":"2026-07-05T11:55:27.281748+00:00"},{"alias_kind":"pith_short_8","alias_value":"LLK7TYZX","created_at":"2026-07-05T11:55:27.281748+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":30,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25407","citing_title":"Teach-to-Reason: Competition-Guided Reasoning with a Self-Improving Teacher","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21262","citing_title":"ARCO: Adaptive Rubrics with Co-Evolution for Multi-Step LLM-Based Agents","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07705","citing_title":"SAW: Stage-Aware Dynamic Weighting for Multi-Objective Reinforcement Learning in Large Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04923","citing_title":"Reproducing, Analyzing, and Detecting Reward Hacking in Rubric-Based Reinforcement Learning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03829","citing_title":"BigFinanceBench: A Workflow-Grounded Benchmark for Financial-Research Agents","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03968","citing_title":"QUBRIC: Co-Designing Queries and Rubrics for RL Beyond Verifiable Rewards","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04051","citing_title":"RUBAS: Rubric-Based Reinforcement Learning for Agent Safety","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":184,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29275","citing_title":"Prompt-Level Reward Specifications for Open-Ended Post-Training","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30244","citing_title":"Reinforcement Learning with Robust Rubric Rewards","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01091","citing_title":"Deep Research as Rubric for Reinforcement Learning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08982","citing_title":"Baichuan-M4: A Clinical-Grade Medical Agent System for Continuous Care","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21573","citing_title":"Lens: Rethinking Training Efficiency for Foundational Text-to-Image Models","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17602","citing_title":"AutoRubric-T2I: Robust Rule-Based Reward Model for Text-to-Image Alignment","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17602","citing_title":"AutoRubric-T2I: Robust Rule-Based Reward Model for Text-to-Image Alignment","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2509.08827","citing_title":"A Survey of Reinforcement Learning for Large Reasoning Models","ref_index":215,"is_internal_anchor":false},{"citing_arxiv_id":"2603.15646","citing_title":"Alternating Reinforcement Learning with Contextual Rubric Rewards: Beyond the Scalarization Strategy","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12474","citing_title":"Reward Hacking in Rubric-Based Reinforcement Learning","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09808","citing_title":"Quantifying the Utility of User Simulators for Building Collaborative LLM Assistants","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09584","citing_title":"CLR-voyance: Reinforcing Open-Ended Reasoning for Inpatient Clinical Decision Support with Outcome-Aware Rubrics","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09269","citing_title":"DeltaRubric: Generative Multimodal Reward Modeling via Joint Planning and Verification","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20051","citing_title":"Bootstrapping Post-training Signals for Open-ended Tasks via Rubric-based Self-play on Pre-training Text","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13029","citing_title":"Visual Preference Optimization with Rubric Rewards","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06822","citing_title":"SHARP: A Self-Evolving Human-Auditable Rubric Policy for Financial Trading Agents","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07396","citing_title":"Rubric-based On-policy Distillation","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LLK7TYZXSRII33SIHT4JVMONJ3","json":"https://pith.science/pith/LLK7TYZXSRII33SIHT4JVMONJ3.json","graph_json":"https://pith.science/api/pith-number/LLK7TYZXSRII33SIHT4JVMONJ3/graph.json","events_json":"https://pith.science/api/pith-number/LLK7TYZXSRII33SIHT4JVMONJ3/events.json","paper":"https://pith.science/paper/LLK7TYZX"},"agent_actions":{"view_html":"https://pith.science/pith/LLK7TYZXSRII33SIHT4JVMONJ3","download_json":"https://pith.science/pith/LLK7TYZXSRII33SIHT4JVMONJ3.json","view_paper":"https://pith.science/paper/LLK7TYZX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.12790&json=true","fetch_graph":"https://pith.science/api/pith-number/LLK7TYZXSRII33SIHT4JVMONJ3/graph.json","fetch_events":"https://pith.science/api/pith-number/LLK7TYZXSRII33SIHT4JVMONJ3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LLK7TYZXSRII33SIHT4JVMONJ3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LLK7TYZXSRII33SIHT4JVMONJ3/action/storage_attestation","attest_author":"https://pith.science/pith/LLK7TYZXSRII33SIHT4JVMONJ3/action/author_attestation","sign_citation":"https://pith.science/pith/LLK7TYZXSRII33SIHT4JVMONJ3/action/citation_signature","submit_replication":"https://pith.science/pith/LLK7TYZXSRII33SIHT4JVMONJ3/action/replication_record"}},"created_at":"2026-07-05T11:55:27.281748+00:00","updated_at":"2026-07-05T11:55:27.281748+00:00"}