{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:NSNEQT2LEFNMJ46IX6Y5D6ITJO","short_pith_number":"pith:NSNEQT2L","canonical_record":{"source":{"id":"2405.06639","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-10T17:59:04Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"d1c3abfc7c9ee9015441d09f66ada701abe3fa13e6c1a91c2f422742b416d5a4","abstract_canon_sha256":"f47ca9eb32c33de3a126668ea2448a42622c52a2f163202a81d0806fa1ac101d"},"schema_version":"1.0"},"canonical_sha256":"6c9a484f4b215ac4f3c8bfb1d1f9134b801e2f94b6a9cb6147e75502543c3dee","source":{"kind":"arxiv","id":"2405.06639","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.06639","created_at":"2026-07-05T08:17:52Z"},{"alias_kind":"arxiv_version","alias_value":"2405.06639v1","created_at":"2026-07-05T08:17:52Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.06639","created_at":"2026-07-05T08:17:52Z"},{"alias_kind":"pith_short_12","alias_value":"NSNEQT2LEFNM","created_at":"2026-07-05T08:17:52Z"},{"alias_kind":"pith_short_16","alias_value":"NSNEQT2LEFNMJ46I","created_at":"2026-07-05T08:17:52Z"},{"alias_kind":"pith_short_8","alias_value":"NSNEQT2L","created_at":"2026-07-05T08:17:52Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:NSNEQT2LEFNMJ46IX6Y5D6ITJO","target":"record","payload":{"canonical_record":{"source":{"id":"2405.06639","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-10T17:59:04Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"d1c3abfc7c9ee9015441d09f66ada701abe3fa13e6c1a91c2f422742b416d5a4","abstract_canon_sha256":"f47ca9eb32c33de3a126668ea2448a42622c52a2f163202a81d0806fa1ac101d"},"schema_version":"1.0"},"canonical_sha256":"6c9a484f4b215ac4f3c8bfb1d1f9134b801e2f94b6a9cb6147e75502543c3dee","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:17:52.559700Z","signature_b64":"up5pvTD9OZe5ZlZGA4glgZMP5qx0GtWdS2dLv8RysSOlvP3WUBefksl6qOnvnFnjLUTlokYbifQC5iCxTq4kAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6c9a484f4b215ac4f3c8bfb1d1f9134b801e2f94b6a9cb6147e75502543c3dee","last_reissued_at":"2026-07-05T08:17:52.559200Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:17:52.559200Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2405.06639","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:17:52Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"0T8TOPRUbz0HKi7yOPrCTU2tdp27tNAh37MPA5dzrIdEyaQZbsFEx9y35mCBO78VbzLEPL/UpW/9DcSkTdTtDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T23:31:11.082721Z"},"content_sha256":"aaa314c04c547755f196b039cd8a603264c1b8d0ce3489650a82c6d233406e3a","schema_version":"1.0","event_id":"sha256:aaa314c04c547755f196b039cd8a603264c1b8d0ce3489650a82c6d233406e3a"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:NSNEQT2LEFNMJ46IX6Y5D6ITJO","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Value Augmented Sampling for Language Model Alignment and Personalization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Akash Srivastava, Idan Shenfeld, Pulkit Agrawal, Seungwook Han, Yoon Kim","submitted_at":"2024-05-10T17:59:04Z","abstract_excerpt":"Aligning Large Language Models (LLMs) to cater to different human preferences, learning new skills, and unlearning harmful behavior is an important problem. Search-based methods, such as Best-of-N or Monte-Carlo Tree Search, are performant, but impractical for LLM adaptation due to their high inference cost. On the other hand, using Reinforcement Learning (RL) for adaptation is computationally efficient, but performs worse due to the optimization challenges in co-training the value function and the policy. We present a new framework for reward optimization, Value Augmented Sampling (VAS), that"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.06639","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.06639/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:17:52Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"jTYXgtdwlpnf9FpnXTTK20dKvyFnZWsJu68d7xK9rbpPfdocsv4hAxX5id1004rQHbVcPEdGDJXWwCCziXpfAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T23:31:11.083297Z"},"content_sha256":"71bf4a2844e2128989edc9a37abc4d4a04fbccc0d54d3a2f2f25598d561db99e","schema_version":"1.0","event_id":"sha256:71bf4a2844e2128989edc9a37abc4d4a04fbccc0d54d3a2f2f25598d561db99e"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/NSNEQT2LEFNMJ46IX6Y5D6ITJO/bundle.json","state_url":"https://pith.science/pith/NSNEQT2LEFNMJ46IX6Y5D6ITJO/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/NSNEQT2LEFNMJ46IX6Y5D6ITJO/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T23:31:11Z","links":{"resolver":"https://pith.science/pith/NSNEQT2LEFNMJ46IX6Y5D6ITJO","bundle":"https://pith.science/pith/NSNEQT2LEFNMJ46IX6Y5D6ITJO/bundle.json","state":"https://pith.science/pith/NSNEQT2LEFNMJ46IX6Y5D6ITJO/state.json","well_known_bundle":"https://pith.science/.well-known/pith/NSNEQT2LEFNMJ46IX6Y5D6ITJO/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:NSNEQT2LEFNMJ46IX6Y5D6ITJO","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f47ca9eb32c33de3a126668ea2448a42622c52a2f163202a81d0806fa1ac101d","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-10T17:59:04Z","title_canon_sha256":"d1c3abfc7c9ee9015441d09f66ada701abe3fa13e6c1a91c2f422742b416d5a4"},"schema_version":"1.0","source":{"id":"2405.06639","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.06639","created_at":"2026-07-05T08:17:52Z"},{"alias_kind":"arxiv_version","alias_value":"2405.06639v1","created_at":"2026-07-05T08:17:52Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.06639","created_at":"2026-07-05T08:17:52Z"},{"alias_kind":"pith_short_12","alias_value":"NSNEQT2LEFNM","created_at":"2026-07-05T08:17:52Z"},{"alias_kind":"pith_short_16","alias_value":"NSNEQT2LEFNMJ46I","created_at":"2026-07-05T08:17:52Z"},{"alias_kind":"pith_short_8","alias_value":"NSNEQT2L","created_at":"2026-07-05T08:17:52Z"}],"graph_snapshots":[{"event_id":"sha256:71bf4a2844e2128989edc9a37abc4d4a04fbccc0d54d3a2f2f25598d561db99e","target":"graph","created_at":"2026-07-05T08:17:52Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2405.06639/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Aligning Large Language Models (LLMs) to cater to different human preferences, learning new skills, and unlearning harmful behavior is an important problem. Search-based methods, such as Best-of-N or Monte-Carlo Tree Search, are performant, but impractical for LLM adaptation due to their high inference cost. On the other hand, using Reinforcement Learning (RL) for adaptation is computationally efficient, but performs worse due to the optimization challenges in co-training the value function and the policy. We present a new framework for reward optimization, Value Augmented Sampling (VAS), that","authors_text":"Akash Srivastava, Idan Shenfeld, Pulkit Agrawal, Seungwook Han, Yoon Kim","cross_cats":["cs.AI","cs.CL"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-10T17:59:04Z","title":"Value Augmented Sampling for Language Model Alignment and Personalization"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.06639","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:aaa314c04c547755f196b039cd8a603264c1b8d0ce3489650a82c6d233406e3a","target":"record","created_at":"2026-07-05T08:17:52Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f47ca9eb32c33de3a126668ea2448a42622c52a2f163202a81d0806fa1ac101d","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-10T17:59:04Z","title_canon_sha256":"d1c3abfc7c9ee9015441d09f66ada701abe3fa13e6c1a91c2f422742b416d5a4"},"schema_version":"1.0","source":{"id":"2405.06639","kind":"arxiv","version":1}},"canonical_sha256":"6c9a484f4b215ac4f3c8bfb1d1f9134b801e2f94b6a9cb6147e75502543c3dee","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"6c9a484f4b215ac4f3c8bfb1d1f9134b801e2f94b6a9cb6147e75502543c3dee","first_computed_at":"2026-07-05T08:17:52.559200Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:17:52.559200Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"up5pvTD9OZe5ZlZGA4glgZMP5qx0GtWdS2dLv8RysSOlvP3WUBefksl6qOnvnFnjLUTlokYbifQC5iCxTq4kAg==","signature_status":"signed_v1","signed_at":"2026-07-05T08:17:52.559700Z","signed_message":"canonical_sha256_bytes"},"source_id":"2405.06639","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:aaa314c04c547755f196b039cd8a603264c1b8d0ce3489650a82c6d233406e3a","sha256:71bf4a2844e2128989edc9a37abc4d4a04fbccc0d54d3a2f2f25598d561db99e"],"state_sha256":"5f9184d1b92635b225509c9cfa1e3e22cc31e58df5895af87107b0606e3fc07c"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"UktJATYYau3AWcITHtvNstHRCP34l1Zt6i+Syx5rvA31w/Ccu/cHi5tZTVfmqEBK11qxGqup1KyJvmBoocGyCA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T23:31:11.088797Z","bundle_sha256":"285cf2d91bd43160996b17d1c19213a3919373c75ccc9586f1ea0f72f0ab079a"}}