{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:NU4AUWMTHH5JY6TV7IBOPQPS2S","short_pith_number":"pith:NU4AUWMT","canonical_record":{"source":{"id":"2407.16574","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-23T15:27:37Z","cross_cats_sorted":[],"title_canon_sha256":"3293838c397b3a89acf8a5ee4c6e5b878c1c94e2383345ff20b4c0787a0f8c0f","abstract_canon_sha256":"81f40bf0837db31876906660166418f95ae9ef320d23b2d1f9fe10f6e832ebfd"},"schema_version":"1.0"},"canonical_sha256":"6d380a599339fa9c7a75fa02e7c1f2d4be4f6c6193b2320a0f3c2a8f782d2d97","source":{"kind":"arxiv","id":"2407.16574","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2407.16574","created_at":"2026-07-05T09:45:51Z"},{"alias_kind":"arxiv_version","alias_value":"2407.16574v2","created_at":"2026-07-05T09:45:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.16574","created_at":"2026-07-05T09:45:51Z"},{"alias_kind":"pith_short_12","alias_value":"NU4AUWMTHH5J","created_at":"2026-07-05T09:45:51Z"},{"alias_kind":"pith_short_16","alias_value":"NU4AUWMTHH5JY6TV","created_at":"2026-07-05T09:45:51Z"},{"alias_kind":"pith_short_8","alias_value":"NU4AUWMT","created_at":"2026-07-05T09:45:51Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:NU4AUWMTHH5JY6TV7IBOPQPS2S","target":"record","payload":{"canonical_record":{"source":{"id":"2407.16574","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-23T15:27:37Z","cross_cats_sorted":[],"title_canon_sha256":"3293838c397b3a89acf8a5ee4c6e5b878c1c94e2383345ff20b4c0787a0f8c0f","abstract_canon_sha256":"81f40bf0837db31876906660166418f95ae9ef320d23b2d1f9fe10f6e832ebfd"},"schema_version":"1.0"},"canonical_sha256":"6d380a599339fa9c7a75fa02e7c1f2d4be4f6c6193b2320a0f3c2a8f782d2d97","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:45:51.872280Z","signature_b64":"BAeFEe3b5zhTGTRDXyr5jO3eEbJnZRjfQDrm+b6u/E6E743cvcrP02oZBE8P+LyPb3b+gLYSsrNpWaQ8xvuiDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6d380a599339fa9c7a75fa02e7c1f2d4be4f6c6193b2320a0f3c2a8f782d2d97","last_reissued_at":"2026-07-05T09:45:51.871825Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:45:51.871825Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2407.16574","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:45:51Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Gqt8dY3fQTIyplf4MKDqVNnceJw3TsU5SOptZbJv0Jlv1dDnFpgI573KkLzllEk+M8TGVYoNnlDhSjPU5d5sDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T18:37:18.418736Z"},"content_sha256":"5aa6968f8f62e55d5012362467d912ab28b8753e0722da3ca5a7709a44e858b4","schema_version":"1.0","event_id":"sha256:5aa6968f8f62e55d5012362467d912ab28b8753e0722da3ca5a7709a44e858b4"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:NU4AUWMTHH5JY6TV7IBOPQPS2S","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"TLCR: Token-Level Continuous Reward for Fine-grained Reinforcement Learning from Human Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chang D. Yoo, Daejin Jo, Daniel Wontae Nam, Eunseop Yoon, Gunsoo Han, Hee Suk Yoon, Kyoung-Woon On, Mark A. Hasegawa-Johnson, SooHwan Eom, Sungwoong Kim","submitted_at":"2024-07-23T15:27:37Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) leverages human preference data to train language models to align more closely with human essence. These human preference data, however, are labeled at the sequence level, creating a mismatch between sequence-level preference labels and tokens, which are autoregressively generated from the language model. Although several recent approaches have tried to provide token-level (i.e., dense) rewards for each individual token, these typically rely on predefined discrete reward values (e.g., positive: +1, negative: -1, neutral: 0), failing to account "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.16574","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.16574/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T09:45:51Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"sszq7D1a0qHQSD+RBxh1mySVpTxeWWBh7f1kuZr4B2I35njlCMooOg4exJO/+bliQU1/Ke5r7suQ9Dd6w4T3Ag==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T18:37:18.419238Z"},"content_sha256":"b2dea6eed0541072b327f245a61f9becd9faec0a3d2efedfc8a707c62029497a","schema_version":"1.0","event_id":"sha256:b2dea6eed0541072b327f245a61f9becd9faec0a3d2efedfc8a707c62029497a"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/NU4AUWMTHH5JY6TV7IBOPQPS2S/bundle.json","state_url":"https://pith.science/pith/NU4AUWMTHH5JY6TV7IBOPQPS2S/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/NU4AUWMTHH5JY6TV7IBOPQPS2S/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T18:37:18Z","links":{"resolver":"https://pith.science/pith/NU4AUWMTHH5JY6TV7IBOPQPS2S","bundle":"https://pith.science/pith/NU4AUWMTHH5JY6TV7IBOPQPS2S/bundle.json","state":"https://pith.science/pith/NU4AUWMTHH5JY6TV7IBOPQPS2S/state.json","well_known_bundle":"https://pith.science/.well-known/pith/NU4AUWMTHH5JY6TV7IBOPQPS2S/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:NU4AUWMTHH5JY6TV7IBOPQPS2S","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"81f40bf0837db31876906660166418f95ae9ef320d23b2d1f9fe10f6e832ebfd","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-23T15:27:37Z","title_canon_sha256":"3293838c397b3a89acf8a5ee4c6e5b878c1c94e2383345ff20b4c0787a0f8c0f"},"schema_version":"1.0","source":{"id":"2407.16574","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2407.16574","created_at":"2026-07-05T09:45:51Z"},{"alias_kind":"arxiv_version","alias_value":"2407.16574v2","created_at":"2026-07-05T09:45:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.16574","created_at":"2026-07-05T09:45:51Z"},{"alias_kind":"pith_short_12","alias_value":"NU4AUWMTHH5J","created_at":"2026-07-05T09:45:51Z"},{"alias_kind":"pith_short_16","alias_value":"NU4AUWMTHH5JY6TV","created_at":"2026-07-05T09:45:51Z"},{"alias_kind":"pith_short_8","alias_value":"NU4AUWMT","created_at":"2026-07-05T09:45:51Z"}],"graph_snapshots":[{"event_id":"sha256:b2dea6eed0541072b327f245a61f9becd9faec0a3d2efedfc8a707c62029497a","target":"graph","created_at":"2026-07-05T09:45:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2407.16574/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) leverages human preference data to train language models to align more closely with human essence. These human preference data, however, are labeled at the sequence level, creating a mismatch between sequence-level preference labels and tokens, which are autoregressively generated from the language model. Although several recent approaches have tried to provide token-level (i.e., dense) rewards for each individual token, these typically rely on predefined discrete reward values (e.g., positive: +1, negative: -1, neutral: 0), failing to account ","authors_text":"Chang D. Yoo, Daejin Jo, Daniel Wontae Nam, Eunseop Yoon, Gunsoo Han, Hee Suk Yoon, Kyoung-Woon On, Mark A. Hasegawa-Johnson, SooHwan Eom, Sungwoong Kim","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-23T15:27:37Z","title":"TLCR: Token-Level Continuous Reward for Fine-grained Reinforcement Learning from Human Feedback"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.16574","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:5aa6968f8f62e55d5012362467d912ab28b8753e0722da3ca5a7709a44e858b4","target":"record","created_at":"2026-07-05T09:45:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"81f40bf0837db31876906660166418f95ae9ef320d23b2d1f9fe10f6e832ebfd","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-07-23T15:27:37Z","title_canon_sha256":"3293838c397b3a89acf8a5ee4c6e5b878c1c94e2383345ff20b4c0787a0f8c0f"},"schema_version":"1.0","source":{"id":"2407.16574","kind":"arxiv","version":2}},"canonical_sha256":"6d380a599339fa9c7a75fa02e7c1f2d4be4f6c6193b2320a0f3c2a8f782d2d97","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"6d380a599339fa9c7a75fa02e7c1f2d4be4f6c6193b2320a0f3c2a8f782d2d97","first_computed_at":"2026-07-05T09:45:51.871825Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T09:45:51.871825Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"BAeFEe3b5zhTGTRDXyr5jO3eEbJnZRjfQDrm+b6u/E6E743cvcrP02oZBE8P+LyPb3b+gLYSsrNpWaQ8xvuiDw==","signature_status":"signed_v1","signed_at":"2026-07-05T09:45:51.872280Z","signed_message":"canonical_sha256_bytes"},"source_id":"2407.16574","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:5aa6968f8f62e55d5012362467d912ab28b8753e0722da3ca5a7709a44e858b4","sha256:b2dea6eed0541072b327f245a61f9becd9faec0a3d2efedfc8a707c62029497a"],"state_sha256":"fd83be8c9df31de70a05b56e6c1a00ce965312297f29cfff0fd57d4fce25ccc6"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"HDEu/ow2d3D2ZWql8hQZHBP0Ra8XMxCLil/cWQqQtIMZG9ukRpW/yBhuAosQCSrEKgSEJat6ONxUrdjPU2IGAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T18:37:18.423383Z","bundle_sha256":"62b3a86339146d7651c55953475782dafe9ac7715da41aa997fa003dfd838f9a"}}