{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7I3EVSTIFLFHSXKYQQ3IZ57NUF","short_pith_number":"pith:7I3EVSTI","schema_version":"1.0","canonical_sha256":"fa364aca682aca795d5884368cf7eda160a27f6cf4727a92131f720ff31f765a","source":{"kind":"arxiv","id":"2412.02685","version":1},"attestation_state":"computed","paper":{"title":"T-REG: Preference Optimization with Token-Level Reward Regularization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Lingxiao Zhao, Shujian Zhang, Tao Meng, Wenxuan Zhou","submitted_at":"2024-12-03T18:56:07Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) has been crucial in aligning large language models (LLMs) with human values. Traditionally, RLHF involves generating responses to a query and using a reward model to assign a reward to the entire response. However, this approach faces challenges due to its reliance on a single, sparse reward, which makes it challenging for the model to identify which parts of the sequence contribute most significantly to the final reward. Recent methods have attempted to address this limitation by introducing token-level rewards. However, these methods often re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.02685","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-12-03T18:56:07Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"d5dc4e7dcd709a8bcdbb8382391c4b0e924e3fb13927cc7c995e1eaecb31f437","abstract_canon_sha256":"194d896b67caa042e60f54f4e3b8ba408d3695c5a8be809ae6722b0458b779f1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:43:53.588025Z","signature_b64":"gHcez9w24f9StMrxGfDCDZPDWPgzmv3TDuezxd28govmi+gdE6+36O5KwqapdDZKqZX5mSC6BkgNSADKoVoXCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fa364aca682aca795d5884368cf7eda160a27f6cf4727a92131f720ff31f765a","last_reissued_at":"2026-07-05T09:43:53.587544Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:43:53.587544Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"T-REG: Preference Optimization with Token-Level Reward Regularization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Lingxiao Zhao, Shujian Zhang, Tao Meng, Wenxuan Zhou","submitted_at":"2024-12-03T18:56:07Z","abstract_excerpt":"Reinforcement learning from human feedback (RLHF) has been crucial in aligning large language models (LLMs) with human values. Traditionally, RLHF involves generating responses to a query and using a reward model to assign a reward to the entire response. However, this approach faces challenges due to its reliance on a single, sparse reward, which makes it challenging for the model to identify which parts of the sequence contribute most significantly to the final reward. Recent methods have attempted to address this limitation by introducing token-level rewards. However, these methods often re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.02685","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.02685/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.02685","created_at":"2026-07-05T09:43:53.587599+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.02685v1","created_at":"2026-07-05T09:43:53.587599+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.02685","created_at":"2026-07-05T09:43:53.587599+00:00"},{"alias_kind":"pith_short_12","alias_value":"7I3EVSTIFLFH","created_at":"2026-07-05T09:43:53.587599+00:00"},{"alias_kind":"pith_short_16","alias_value":"7I3EVSTIFLFHSXKY","created_at":"2026-07-05T09:43:53.587599+00:00"},{"alias_kind":"pith_short_8","alias_value":"7I3EVSTI","created_at":"2026-07-05T09:43:53.587599+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.08712","citing_title":"ConfPO: Exploiting Policy Model Confidence for Critical Token Selection in Preference Optimization","ref_index":48,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7I3EVSTIFLFHSXKYQQ3IZ57NUF","json":"https://pith.science/pith/7I3EVSTIFLFHSXKYQQ3IZ57NUF.json","graph_json":"https://pith.science/api/pith-number/7I3EVSTIFLFHSXKYQQ3IZ57NUF/graph.json","events_json":"https://pith.science/api/pith-number/7I3EVSTIFLFHSXKYQQ3IZ57NUF/events.json","paper":"https://pith.science/paper/7I3EVSTI"},"agent_actions":{"view_html":"https://pith.science/pith/7I3EVSTIFLFHSXKYQQ3IZ57NUF","download_json":"https://pith.science/pith/7I3EVSTIFLFHSXKYQQ3IZ57NUF.json","view_paper":"https://pith.science/paper/7I3EVSTI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.02685&json=true","fetch_graph":"https://pith.science/api/pith-number/7I3EVSTIFLFHSXKYQQ3IZ57NUF/graph.json","fetch_events":"https://pith.science/api/pith-number/7I3EVSTIFLFHSXKYQQ3IZ57NUF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7I3EVSTIFLFHSXKYQQ3IZ57NUF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7I3EVSTIFLFHSXKYQQ3IZ57NUF/action/storage_attestation","attest_author":"https://pith.science/pith/7I3EVSTIFLFHSXKYQQ3IZ57NUF/action/author_attestation","sign_citation":"https://pith.science/pith/7I3EVSTIFLFHSXKYQQ3IZ57NUF/action/citation_signature","submit_replication":"https://pith.science/pith/7I3EVSTIFLFHSXKYQQ3IZ57NUF/action/replication_record"}},"created_at":"2026-07-05T09:43:53.587599+00:00","updated_at":"2026-07-05T09:43:53.587599+00:00"}