{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XKJ7LBJLMCDUV5JUAAVWBH37HD","short_pith_number":"pith:XKJ7LBJL","schema_version":"1.0","canonical_sha256":"ba93f5852b60874af534002b609f7f38fca4c14b5b65404245aed8d5142e7adb","source":{"kind":"arxiv","id":"2506.23235","version":1},"attestation_state":"computed","paper":{"title":"Generalist Reward Models: Found Inside Large Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Lei Yuan, Ningjing Chao, Tian Xu, Xiong-Hui Chen, Xuqin Zhang, Yang Yu, Yi-Chen Li, Zhi-Hua Zhou, Zhongxiang Ling","submitted_at":"2025-06-29T13:45:54Z","abstract_excerpt":"The alignment of Large Language Models (LLMs) is critically dependent on reward models trained on costly human preference data. While recent work explores bypassing this cost with AI feedback, these methods often lack a rigorous theoretical foundation. In this paper, we discover that a powerful generalist reward model is already latently present within any LLM trained via standard next-token prediction. We prove that this endogenous reward is not a heuristic, but is theoretically equivalent to a reward function learned through offline inverse reinforcement learning. This connection allows us t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.23235","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2025-06-29T13:45:54Z","cross_cats_sorted":[],"title_canon_sha256":"f9e7d976229a002655fef67026e1fa9dad997ff4cd3cbb6be7de8a8c1b4e0e72","abstract_canon_sha256":"b565b0b3aa984ea3955cb8b2bd2ddb962dc54ddad1b7b85e473be37ad56cbd50"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:29:18.334829Z","signature_b64":"RAFW4lLsXLr65m20JKNB57W2LXcKXAn6ufx2EuNBbS2GvO8vds+WWkqyzRRCpw7zyIY641Y5kuMTt6qgjhOVDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ba93f5852b60874af534002b609f7f38fca4c14b5b65404245aed8d5142e7adb","last_reissued_at":"2026-07-05T11:29:18.334303Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:29:18.334303Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generalist Reward Models: Found Inside Large Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Lei Yuan, Ningjing Chao, Tian Xu, Xiong-Hui Chen, Xuqin Zhang, Yang Yu, Yi-Chen Li, Zhi-Hua Zhou, Zhongxiang Ling","submitted_at":"2025-06-29T13:45:54Z","abstract_excerpt":"The alignment of Large Language Models (LLMs) is critically dependent on reward models trained on costly human preference data. While recent work explores bypassing this cost with AI feedback, these methods often lack a rigorous theoretical foundation. In this paper, we discover that a powerful generalist reward model is already latently present within any LLM trained via standard next-token prediction. We prove that this endogenous reward is not a heuristic, but is theoretically equivalent to a reward function learned through offline inverse reinforcement learning. This connection allows us t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.23235","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.23235/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.23235","created_at":"2026-07-05T11:29:18.334372+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.23235v1","created_at":"2026-07-05T11:29:18.334372+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.23235","created_at":"2026-07-05T11:29:18.334372+00:00"},{"alias_kind":"pith_short_12","alias_value":"XKJ7LBJLMCDU","created_at":"2026-07-05T11:29:18.334372+00:00"},{"alias_kind":"pith_short_16","alias_value":"XKJ7LBJLMCDUV5JU","created_at":"2026-07-05T11:29:18.334372+00:00"},{"alias_kind":"pith_short_8","alias_value":"XKJ7LBJL","created_at":"2026-07-05T11:29:18.334372+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08815","citing_title":"Momentum for Reasoning: Dense Intrinsic Signals in Policy Optimization","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":179,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24005","citing_title":"LC-ERD: Mining Latent Logic for Self-Evolving Reasoning via Consistency-Regulated Reward Decomposition","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25189","citing_title":"Directional Alignment Mitigates Reward Hacking in Reinforcement Learning for Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30339","citing_title":"REAR: Test-time Preference Realignment through Reward Decomposition","ref_index":124,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21046","citing_title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","ref_index":252,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13197","citing_title":"Unleashing Implicit Rewards: Prefix-Value Learning for Distribution-Level Optimization","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XKJ7LBJLMCDUV5JUAAVWBH37HD","json":"https://pith.science/pith/XKJ7LBJLMCDUV5JUAAVWBH37HD.json","graph_json":"https://pith.science/api/pith-number/XKJ7LBJLMCDUV5JUAAVWBH37HD/graph.json","events_json":"https://pith.science/api/pith-number/XKJ7LBJLMCDUV5JUAAVWBH37HD/events.json","paper":"https://pith.science/paper/XKJ7LBJL"},"agent_actions":{"view_html":"https://pith.science/pith/XKJ7LBJLMCDUV5JUAAVWBH37HD","download_json":"https://pith.science/pith/XKJ7LBJLMCDUV5JUAAVWBH37HD.json","view_paper":"https://pith.science/paper/XKJ7LBJL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.23235&json=true","fetch_graph":"https://pith.science/api/pith-number/XKJ7LBJLMCDUV5JUAAVWBH37HD/graph.json","fetch_events":"https://pith.science/api/pith-number/XKJ7LBJLMCDUV5JUAAVWBH37HD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XKJ7LBJLMCDUV5JUAAVWBH37HD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XKJ7LBJLMCDUV5JUAAVWBH37HD/action/storage_attestation","attest_author":"https://pith.science/pith/XKJ7LBJLMCDUV5JUAAVWBH37HD/action/author_attestation","sign_citation":"https://pith.science/pith/XKJ7LBJLMCDUV5JUAAVWBH37HD/action/citation_signature","submit_replication":"https://pith.science/pith/XKJ7LBJLMCDUV5JUAAVWBH37HD/action/replication_record"}},"created_at":"2026-07-05T11:29:18.334372+00:00","updated_at":"2026-07-05T11:29:18.334372+00:00"}