{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZZ3KX72RPNNL6RSEYYIOVRNIOO","short_pith_number":"pith:ZZ3KX72R","schema_version":"1.0","canonical_sha256":"ce76abff517b5abf4644c610eac5a873b4c865bd2bc42c888e9df0c1d0720441","source":{"kind":"arxiv","id":"2506.03637","version":2},"attestation_state":"computed","paper":{"title":"RewardAnything: Generalizable Principle-Following Reward Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Fandong Meng, Jiali Zeng, Jie Zhou, Jindong Wang, Shikun Zhang, Wei Ye, Weizheng Gu, Yidong Wang, Yue Zhang, Zhuohao Yu","submitted_at":"2025-06-04T07:30:16Z","abstract_excerpt":"Reward Models, essential for guiding Large Language Model optimization, are typically trained on fixed preference datasets, resulting in rigid alignment to single, implicit preference distributions. This prevents adaptation to diverse real-world needs-from conciseness in one task to detailed explanations in another. The standard practice of collecting task-specific preference data and retraining reward models is resource-intensive, often producing biased rewards, and limits practical application. We introduce generalizable, principle-following reward models. We propose that RMs should understa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.03637","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-04T07:30:16Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"51ab0122e0c29d491ddc7016047ae4a2fd2f2752b9644df178ab246da1426c42","abstract_canon_sha256":"ac2972893aa6f705fdd3fa5a92bb0c007d80b437dab6a5bd3767e7c00c7bb767"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:32:30.088564Z","signature_b64":"fZXVNdxkubSSbQ2toCm6NGbRUndVLmXtdbOKZGlc0m+JbxeGIiNj/vnwtlxGQ+Swf9KHl9mAwg4PA31cQ19KDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ce76abff517b5abf4644c610eac5a873b4c865bd2bc42c888e9df0c1d0720441","last_reissued_at":"2026-07-05T11:32:30.088050Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:32:30.088050Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RewardAnything: Generalizable Principle-Following Reward Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Fandong Meng, Jiali Zeng, Jie Zhou, Jindong Wang, Shikun Zhang, Wei Ye, Weizheng Gu, Yidong Wang, Yue Zhang, Zhuohao Yu","submitted_at":"2025-06-04T07:30:16Z","abstract_excerpt":"Reward Models, essential for guiding Large Language Model optimization, are typically trained on fixed preference datasets, resulting in rigid alignment to single, implicit preference distributions. This prevents adaptation to diverse real-world needs-from conciseness in one task to detailed explanations in another. The standard practice of collecting task-specific preference data and retraining reward models is resource-intensive, often producing biased rewards, and limits practical application. We introduce generalizable, principle-following reward models. We propose that RMs should understa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.03637","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.03637/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.03637","created_at":"2026-07-05T11:32:30.088113+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.03637v2","created_at":"2026-07-05T11:32:30.088113+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.03637","created_at":"2026-07-05T11:32:30.088113+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZZ3KX72RPNNL","created_at":"2026-07-05T11:32:30.088113+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZZ3KX72RPNNL6RSE","created_at":"2026-07-05T11:32:30.088113+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZZ3KX72R","created_at":"2026-07-05T11:32:30.088113+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08077","citing_title":"Support Vector Rubrics: Closing the Gap Between Self-Generated and Human Rubrics","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04458","citing_title":"DoGMaTiQ: Automated Generation of Question-and-Answer Nuggets for Report Evaluation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03980","citing_title":"Skill-RM: Unifying Heterogeneous Evaluation Criteria via Agent Skill","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21319","citing_title":"RLBFF: Binary Flexible Feedback to bridge between Human Feedback & Verifiable Rewards","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16335","citing_title":"Beyond Verifiable Rewards: Rubric-Based GRM for Reinforced Fine-Tuning SWE Agents","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04458","citing_title":"DoGMaTiQ: Automated Generation of Question-and-Answer Nuggets for Report Evaluation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17188","citing_title":"Beyond Overlap Metrics: Rewarding Reasoning and Preferences for Faithful Multi-Role Dialogue Summarization","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZZ3KX72RPNNL6RSEYYIOVRNIOO","json":"https://pith.science/pith/ZZ3KX72RPNNL6RSEYYIOVRNIOO.json","graph_json":"https://pith.science/api/pith-number/ZZ3KX72RPNNL6RSEYYIOVRNIOO/graph.json","events_json":"https://pith.science/api/pith-number/ZZ3KX72RPNNL6RSEYYIOVRNIOO/events.json","paper":"https://pith.science/paper/ZZ3KX72R"},"agent_actions":{"view_html":"https://pith.science/pith/ZZ3KX72RPNNL6RSEYYIOVRNIOO","download_json":"https://pith.science/pith/ZZ3KX72RPNNL6RSEYYIOVRNIOO.json","view_paper":"https://pith.science/paper/ZZ3KX72R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.03637&json=true","fetch_graph":"https://pith.science/api/pith-number/ZZ3KX72RPNNL6RSEYYIOVRNIOO/graph.json","fetch_events":"https://pith.science/api/pith-number/ZZ3KX72RPNNL6RSEYYIOVRNIOO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZZ3KX72RPNNL6RSEYYIOVRNIOO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZZ3KX72RPNNL6RSEYYIOVRNIOO/action/storage_attestation","attest_author":"https://pith.science/pith/ZZ3KX72RPNNL6RSEYYIOVRNIOO/action/author_attestation","sign_citation":"https://pith.science/pith/ZZ3KX72RPNNL6RSEYYIOVRNIOO/action/citation_signature","submit_replication":"https://pith.science/pith/ZZ3KX72RPNNL6RSEYYIOVRNIOO/action/replication_record"}},"created_at":"2026-07-05T11:32:30.088113+00:00","updated_at":"2026-07-05T11:32:30.088113+00:00"}