{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LGLZBBSUS26O6CMRN7KSPLUTQN","short_pith_number":"pith:LGLZBBSU","schema_version":"1.0","canonical_sha256":"599790865496bcef09916fd527ae93834a13dd979fa74d0381c661ec02bd9943","source":{"kind":"arxiv","id":"2402.00782","version":1},"attestation_state":"computed","paper":{"title":"Dense Reward for Free in Reinforcement Learning from Human Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alex J. Chan, Hao Sun, Mihaela van der Schaar, Samuel Holt","submitted_at":"2024-02-01T17:10:35Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) has been credited as the key advance that has allowed Large Language Models (LLMs) to effectively follow instructions and produce useful assistance. Classically, this involves generating completions from the LLM in response to a query before using a separate reward model to assign a score to the full completion. As an auto-regressive process, the LLM has to take many \"actions\" (selecting individual tokens) and only receives a single, sparse reward at the end of an episode, a setup that is known to be difficult to optimise in traditional reinfor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.00782","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-01T17:10:35Z","cross_cats_sorted":[],"title_canon_sha256":"2163f390390ba469ea48dc253268f9d39b104ad277229e9583fcfd6feff9c179","abstract_canon_sha256":"f46cf919706d875f4480a757a35e23fa7b95da7e7cfda3053abede483922f4ba"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:40:17.617871Z","signature_b64":"6ykQwH2NvhXpXBSeQHpAcljSnQXGeSbrHaqnJkF4MU0Atkqp2hwl9Qp+FGi9rgnxvje5XPcN07+Z4OGr5E3CDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"599790865496bcef09916fd527ae93834a13dd979fa74d0381c661ec02bd9943","last_reissued_at":"2026-07-05T07:40:17.617458Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:40:17.617458Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dense Reward for Free in Reinforcement Learning from Human Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alex J. Chan, Hao Sun, Mihaela van der Schaar, Samuel Holt","submitted_at":"2024-02-01T17:10:35Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) has been credited as the key advance that has allowed Large Language Models (LLMs) to effectively follow instructions and produce useful assistance. Classically, this involves generating completions from the LLM in response to a query before using a separate reward model to assign a score to the full completion. As an auto-regressive process, the LLM has to take many \"actions\" (selecting individual tokens) and only receives a single, sparse reward at the end of an episode, a setup that is known to be difficult to optimise in traditional reinfor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.00782","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.00782/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.00782","created_at":"2026-07-05T07:40:17.617514+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.00782v1","created_at":"2026-07-05T07:40:17.617514+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.00782","created_at":"2026-07-05T07:40:17.617514+00:00"},{"alias_kind":"pith_short_12","alias_value":"LGLZBBSUS26O","created_at":"2026-07-05T07:40:17.617514+00:00"},{"alias_kind":"pith_short_16","alias_value":"LGLZBBSUS26O6CMR","created_at":"2026-07-05T07:40:17.617514+00:00"},{"alias_kind":"pith_short_8","alias_value":"LGLZBBSU","created_at":"2026-07-05T07:40:17.617514+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23557","citing_title":"Dense Reward for Multi-View 3D Reasoning with Global Maps and Local Views","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.15893","citing_title":"BALTO: Balanced Token-Level Policy Optimization for Hallucination Mitigation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12652","citing_title":"Multi-Rollout On-Policy Distillation via Peer Successes and Failures","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2408.07199","citing_title":"Agent Q: Advanced Reasoning and Learning for Autonomous AI Agents","ref_index":212,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12652","citing_title":"Multi-Rollout On-Policy Distillation via Peer Successes and Failures","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09134","citing_title":"BoostAPR: Boosting Automated Program Repair via Execution-Grounded Reinforcement Learning with Dual Reward Models","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09134","citing_title":"BoostAPR: Boosting Automated Program Repair via Execution-Grounded Reinforcement Learning with Dual Reward Models","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20659","citing_title":"GRPO-VPS: Enhancing Group Relative Policy Optimization with Verifiable Process Supervision for Effective Reasoning","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LGLZBBSUS26O6CMRN7KSPLUTQN","json":"https://pith.science/pith/LGLZBBSUS26O6CMRN7KSPLUTQN.json","graph_json":"https://pith.science/api/pith-number/LGLZBBSUS26O6CMRN7KSPLUTQN/graph.json","events_json":"https://pith.science/api/pith-number/LGLZBBSUS26O6CMRN7KSPLUTQN/events.json","paper":"https://pith.science/paper/LGLZBBSU"},"agent_actions":{"view_html":"https://pith.science/pith/LGLZBBSUS26O6CMRN7KSPLUTQN","download_json":"https://pith.science/pith/LGLZBBSUS26O6CMRN7KSPLUTQN.json","view_paper":"https://pith.science/paper/LGLZBBSU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.00782&json=true","fetch_graph":"https://pith.science/api/pith-number/LGLZBBSUS26O6CMRN7KSPLUTQN/graph.json","fetch_events":"https://pith.science/api/pith-number/LGLZBBSUS26O6CMRN7KSPLUTQN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LGLZBBSUS26O6CMRN7KSPLUTQN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LGLZBBSUS26O6CMRN7KSPLUTQN/action/storage_attestation","attest_author":"https://pith.science/pith/LGLZBBSUS26O6CMRN7KSPLUTQN/action/author_attestation","sign_citation":"https://pith.science/pith/LGLZBBSUS26O6CMRN7KSPLUTQN/action/citation_signature","submit_replication":"https://pith.science/pith/LGLZBBSUS26O6CMRN7KSPLUTQN/action/replication_record"}},"created_at":"2026-07-05T07:40:17.617514+00:00","updated_at":"2026-07-05T07:40:17.617514+00:00"}