{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZW7LRDGOG4O6MVZHI7KFGAIPFP","short_pith_number":"pith:ZW7LRDGO","schema_version":"1.0","canonical_sha256":"cdbeb88cce371de6572747d453010f2bc719744d4101fea332b1514a4786e22b","source":{"kind":"arxiv","id":"2505.20417","version":1},"attestation_state":"computed","paper":{"title":"SCAR: Shapley Credit Assignment for More Efficient RLHF","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Doina Precup, Meng Cao, Shuyuan Zhang, Xiao-Wen Chang","submitted_at":"2025-05-26T18:06:52Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) is a widely used technique for aligning Large Language Models (LLMs) with human preferences, yet it often suffers from sparse reward signals, making effective credit assignment challenging. In typical setups, the reward model provides a single scalar score for an entire generated sequence, offering little insight into which token or span-level decisions were responsible for the outcome. To address this, we propose Shapley Credit Assignment Rewards (SCAR), a novel method that leverages Shapley values in cooperative game theory. SCAR distributes "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.20417","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-05-26T18:06:52Z","cross_cats_sorted":[],"title_canon_sha256":"10f05c1cce38c67d05dd748c99ddb07bb7895250f6ce4faf50e26070427e729e","abstract_canon_sha256":"7f61368adeeafb495acdcf8f6dcbe63c48675d8e3cbb0e0fef5b6e174bad1fe6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:12.329526Z","signature_b64":"sNXzkddUTi6a+X0oCq0OywzFBLxumJM/T5jeNxmN5sOmoK2SvPrCSGa4r+B+u5x6Nckg5VSqd79qTrUt6Hd/Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cdbeb88cce371de6572747d453010f2bc719744d4101fea332b1514a4786e22b","last_reissued_at":"2026-07-05T11:10:12.329045Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:12.329045Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SCAR: Shapley Credit Assignment for More Efficient RLHF","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Doina Precup, Meng Cao, Shuyuan Zhang, Xiao-Wen Chang","submitted_at":"2025-05-26T18:06:52Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) is a widely used technique for aligning Large Language Models (LLMs) with human preferences, yet it often suffers from sparse reward signals, making effective credit assignment challenging. In typical setups, the reward model provides a single scalar score for an entire generated sequence, offering little insight into which token or span-level decisions were responsible for the outcome. To address this, we propose Shapley Credit Assignment Rewards (SCAR), a novel method that leverages Shapley values in cooperative game theory. SCAR distributes "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.20417","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.20417/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.20417","created_at":"2026-07-05T11:10:12.329100+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.20417v1","created_at":"2026-07-05T11:10:12.329100+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.20417","created_at":"2026-07-05T11:10:12.329100+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZW7LRDGOG4O6","created_at":"2026-07-05T11:10:12.329100+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZW7LRDGOG4O6MVZH","created_at":"2026-07-05T11:10:12.329100+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZW7LRDGO","created_at":"2026-07-05T11:10:12.329100+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09932","citing_title":"When RL Fails after SFT: Rejuvenating Model Plasticity for Robust SFT-to-RL Handoff","ref_index":140,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00257","citing_title":"ARCA: Adapter-Residual Credit Assignment When Token Signals Degenerate","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09459","citing_title":"From Reasoning to Agentic: Credit Assignment in Reinforcement Learning for Large Language Models","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZW7LRDGOG4O6MVZHI7KFGAIPFP","json":"https://pith.science/pith/ZW7LRDGOG4O6MVZHI7KFGAIPFP.json","graph_json":"https://pith.science/api/pith-number/ZW7LRDGOG4O6MVZHI7KFGAIPFP/graph.json","events_json":"https://pith.science/api/pith-number/ZW7LRDGOG4O6MVZHI7KFGAIPFP/events.json","paper":"https://pith.science/paper/ZW7LRDGO"},"agent_actions":{"view_html":"https://pith.science/pith/ZW7LRDGOG4O6MVZHI7KFGAIPFP","download_json":"https://pith.science/pith/ZW7LRDGOG4O6MVZHI7KFGAIPFP.json","view_paper":"https://pith.science/paper/ZW7LRDGO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.20417&json=true","fetch_graph":"https://pith.science/api/pith-number/ZW7LRDGOG4O6MVZHI7KFGAIPFP/graph.json","fetch_events":"https://pith.science/api/pith-number/ZW7LRDGOG4O6MVZHI7KFGAIPFP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZW7LRDGOG4O6MVZHI7KFGAIPFP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZW7LRDGOG4O6MVZHI7KFGAIPFP/action/storage_attestation","attest_author":"https://pith.science/pith/ZW7LRDGOG4O6MVZHI7KFGAIPFP/action/author_attestation","sign_citation":"https://pith.science/pith/ZW7LRDGOG4O6MVZHI7KFGAIPFP/action/citation_signature","submit_replication":"https://pith.science/pith/ZW7LRDGOG4O6MVZHI7KFGAIPFP/action/replication_record"}},"created_at":"2026-07-05T11:10:12.329100+00:00","updated_at":"2026-07-05T11:10:12.329100+00:00"}