{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:2GTNA4RG76764POQGMLSMD2A36","short_pith_number":"pith:2GTNA4RG","canonical_record":{"source":{"id":"2404.12358","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-18T17:37:02Z","cross_cats_sorted":[],"title_canon_sha256":"5195bb7aa7c241bf2b2775c46586ffad6c666192fe4c1ec422cb11d8e4cf8cdb","abstract_canon_sha256":"88b1d8a2dac943e131ddef2e689973100498bb9627bb0139f5ad41b4e76a0a62"},"schema_version":"1.0"},"canonical_sha256":"d1a6d07226ffbfee3dd03317260f40df94089a74f46656b8ab2c70f169b3406f","source":{"kind":"arxiv","id":"2404.12358","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2404.12358","created_at":"2026-07-05T08:54:49Z"},{"alias_kind":"arxiv_version","alias_value":"2404.12358v2","created_at":"2026-07-05T08:54:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.12358","created_at":"2026-07-05T08:54:49Z"},{"alias_kind":"pith_short_12","alias_value":"2GTNA4RG7676","created_at":"2026-07-05T08:54:49Z"},{"alias_kind":"pith_short_16","alias_value":"2GTNA4RG76764POQ","created_at":"2026-07-05T08:54:49Z"},{"alias_kind":"pith_short_8","alias_value":"2GTNA4RG","created_at":"2026-07-05T08:54:49Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:2GTNA4RG76764POQGMLSMD2A36","target":"record","payload":{"canonical_record":{"source":{"id":"2404.12358","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-18T17:37:02Z","cross_cats_sorted":[],"title_canon_sha256":"5195bb7aa7c241bf2b2775c46586ffad6c666192fe4c1ec422cb11d8e4cf8cdb","abstract_canon_sha256":"88b1d8a2dac943e131ddef2e689973100498bb9627bb0139f5ad41b4e76a0a62"},"schema_version":"1.0"},"canonical_sha256":"d1a6d07226ffbfee3dd03317260f40df94089a74f46656b8ab2c70f169b3406f","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:54:49.392917Z","signature_b64":"6ztiMd5XltOoI26QvYm+5lpucLZIdJkFZqXibLZI/pkbQGgMx0nDZXq5YKYWEMCqb5VJUfYe4PImt2AdqKPICA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d1a6d07226ffbfee3dd03317260f40df94089a74f46656b8ab2c70f169b3406f","last_reissued_at":"2026-07-05T08:54:49.392431Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:54:49.392431Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2404.12358","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:54:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"HXu+yHFAmqEP/ch285qg9FzH3tvn8j3oTsg251PNoLMEZEJZEgklR/rlAWqZvGIdQl3I5n7YTCCA8bP2QIZvCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T22:27:11.398799Z"},"content_sha256":"92d923ca4c69e8155dc17393877cd1bc3eb033d22efb18bbdc1b1328de694ebf","schema_version":"1.0","event_id":"sha256:92d923ca4c69e8155dc17393877cd1bc3eb033d22efb18bbdc1b1328de694ebf"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:2GTNA4RG76764POQGMLSMD2A36","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"From $r$ to $Q^*$: Your Language Model is Secretly a Q-Function","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chelsea Finn, Joey Hejna, Rafael Rafailov, Ryan Park","submitted_at":"2024-04-18T17:37:02Z","abstract_excerpt":"Reinforcement Learning From Human Feedback (RLHF) has been critical to the success of the latest generation of generative AI models. In response to the complex nature of the classical RLHF pipeline, direct alignment algorithms such as Direct Preference Optimization (DPO) have emerged as an alternative approach. Although DPO solves the same objective as the standard RLHF setup, there is a mismatch between the two approaches. Standard RLHF deploys reinforcement learning in a specific token-level MDP, while DPO is derived as a bandit problem in which the whole response of the model is treated as "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.12358","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.12358/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:54:49Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"cMdk0064jU4aP6B0PqeWOk77zMgv/ayQk+7IiiU/5ExBWroggIVcf9EH+/M2NbEQel3HmnPoaNk+h+a9G0NlDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T22:27:11.399681Z"},"content_sha256":"0b8aa8c85275058dc9a3df47069e793765dbbcbc5bf6cb2c099ade351fc4392d","schema_version":"1.0","event_id":"sha256:0b8aa8c85275058dc9a3df47069e793765dbbcbc5bf6cb2c099ade351fc4392d"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/2GTNA4RG76764POQGMLSMD2A36/bundle.json","state_url":"https://pith.science/pith/2GTNA4RG76764POQGMLSMD2A36/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/2GTNA4RG76764POQGMLSMD2A36/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T22:27:11Z","links":{"resolver":"https://pith.science/pith/2GTNA4RG76764POQGMLSMD2A36","bundle":"https://pith.science/pith/2GTNA4RG76764POQGMLSMD2A36/bundle.json","state":"https://pith.science/pith/2GTNA4RG76764POQGMLSMD2A36/state.json","well_known_bundle":"https://pith.science/.well-known/pith/2GTNA4RG76764POQGMLSMD2A36/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:2GTNA4RG76764POQGMLSMD2A36","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"88b1d8a2dac943e131ddef2e689973100498bb9627bb0139f5ad41b4e76a0a62","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-18T17:37:02Z","title_canon_sha256":"5195bb7aa7c241bf2b2775c46586ffad6c666192fe4c1ec422cb11d8e4cf8cdb"},"schema_version":"1.0","source":{"id":"2404.12358","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2404.12358","created_at":"2026-07-05T08:54:49Z"},{"alias_kind":"arxiv_version","alias_value":"2404.12358v2","created_at":"2026-07-05T08:54:49Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.12358","created_at":"2026-07-05T08:54:49Z"},{"alias_kind":"pith_short_12","alias_value":"2GTNA4RG7676","created_at":"2026-07-05T08:54:49Z"},{"alias_kind":"pith_short_16","alias_value":"2GTNA4RG76764POQ","created_at":"2026-07-05T08:54:49Z"},{"alias_kind":"pith_short_8","alias_value":"2GTNA4RG","created_at":"2026-07-05T08:54:49Z"}],"graph_snapshots":[{"event_id":"sha256:0b8aa8c85275058dc9a3df47069e793765dbbcbc5bf6cb2c099ade351fc4392d","target":"graph","created_at":"2026-07-05T08:54:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2404.12358/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning From Human Feedback (RLHF) has been critical to the success of the latest generation of generative AI models. In response to the complex nature of the classical RLHF pipeline, direct alignment algorithms such as Direct Preference Optimization (DPO) have emerged as an alternative approach. Although DPO solves the same objective as the standard RLHF setup, there is a mismatch between the two approaches. Standard RLHF deploys reinforcement learning in a specific token-level MDP, while DPO is derived as a bandit problem in which the whole response of the model is treated as ","authors_text":"Chelsea Finn, Joey Hejna, Rafael Rafailov, Ryan Park","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-18T17:37:02Z","title":"From $r$ to $Q^*$: Your Language Model is Secretly a Q-Function"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.12358","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:92d923ca4c69e8155dc17393877cd1bc3eb033d22efb18bbdc1b1328de694ebf","target":"record","created_at":"2026-07-05T08:54:49Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"88b1d8a2dac943e131ddef2e689973100498bb9627bb0139f5ad41b4e76a0a62","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-18T17:37:02Z","title_canon_sha256":"5195bb7aa7c241bf2b2775c46586ffad6c666192fe4c1ec422cb11d8e4cf8cdb"},"schema_version":"1.0","source":{"id":"2404.12358","kind":"arxiv","version":2}},"canonical_sha256":"d1a6d07226ffbfee3dd03317260f40df94089a74f46656b8ab2c70f169b3406f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"d1a6d07226ffbfee3dd03317260f40df94089a74f46656b8ab2c70f169b3406f","first_computed_at":"2026-07-05T08:54:49.392431Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:54:49.392431Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"6ztiMd5XltOoI26QvYm+5lpucLZIdJkFZqXibLZI/pkbQGgMx0nDZXq5YKYWEMCqb5VJUfYe4PImt2AdqKPICA==","signature_status":"signed_v1","signed_at":"2026-07-05T08:54:49.392917Z","signed_message":"canonical_sha256_bytes"},"source_id":"2404.12358","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:92d923ca4c69e8155dc17393877cd1bc3eb033d22efb18bbdc1b1328de694ebf","sha256:0b8aa8c85275058dc9a3df47069e793765dbbcbc5bf6cb2c099ade351fc4392d"],"state_sha256":"b82d7cef28ddda69a7a90be884c7efe152bfaa5f3aaba582b7cf776018380539"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"YRsaeIQaoHeVSczNreeQ75amZgXY9eXqWzxWNR2wd9w904485xuKpTBxFfZUv/aaTnBt+gmWGqyzh+8SkOnXAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T22:27:11.404822Z","bundle_sha256":"4855d919718460926073379c5ad9bfcce1a5f12a9e7dff3bb9047cb3957a1242"}}