{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:TEZCABRXFJYKDOFWEGEEE7PGG6","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"1fa6e2a3faea039f4b33557eb908ef118578ae4458a0f5c6025d75fb6568e25c","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-02T17:44:41Z","title_canon_sha256":"7eaab7d9a89c83b29afd8089ea03b32190aa2842475a82ae497b5c0f2a04826a"},"schema_version":"1.0","source":{"id":"2405.01511","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.01511","created_at":"2026-07-05T08:52:53Z"},{"alias_kind":"arxiv_version","alias_value":"2405.01511v2","created_at":"2026-07-05T08:52:53Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.01511","created_at":"2026-07-05T08:52:53Z"},{"alias_kind":"pith_short_12","alias_value":"TEZCABRXFJYK","created_at":"2026-07-05T08:52:53Z"},{"alias_kind":"pith_short_16","alias_value":"TEZCABRXFJYKDOFW","created_at":"2026-07-05T08:52:53Z"},{"alias_kind":"pith_short_8","alias_value":"TEZCABRX","created_at":"2026-07-05T08:52:53Z"}],"graph_snapshots":[{"event_id":"sha256:1a298355044e01a7b6c72bbb1fb611d321da19e5e419829107c35844387ee067","target":"graph","created_at":"2026-07-05T08:52:53Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2405.01511/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Varied approaches for aligning language models have been proposed, including supervised fine-tuning, RLHF, and direct optimization methods such as DPO. Although DPO has rapidly gained popularity due to its straightforward training process and competitive results, there is an open question of whether there remain practical advantages of using a discriminator, like a reward model, to evaluate responses. We propose D2PO, discriminator-guided DPO, an approach for the online setting where preferences are being collected throughout learning. As we collect gold preferences, we use these not only to t","authors_text":"Greg Durrett, Nathan Lambert, Prasann Singhal, Scott Niekum, Tanya Goyal","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-02T17:44:41Z","title":"D2PO: Discriminator-Guided DPO with Response Evaluation Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.01511","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:572b6c99c428afac454dec544a63405013bb5dee6c90a4ec070bb54f86f35ffa","target":"record","created_at":"2026-07-05T08:52:53Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"1fa6e2a3faea039f4b33557eb908ef118578ae4458a0f5c6025d75fb6568e25c","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-02T17:44:41Z","title_canon_sha256":"7eaab7d9a89c83b29afd8089ea03b32190aa2842475a82ae497b5c0f2a04826a"},"schema_version":"1.0","source":{"id":"2405.01511","kind":"arxiv","version":2}},"canonical_sha256":"99322006372a70a1b8b62188427de6379d1f63b9157dfc5eca8a3b07bcb82b0f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"99322006372a70a1b8b62188427de6379d1f63b9157dfc5eca8a3b07bcb82b0f","first_computed_at":"2026-07-05T08:52:53.982166Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:52:53.982166Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"dEW82d05rgA0sjAPmEoD/1kD6NuwrPwsH7w1FTObZQIj/I5P+3Z7Wl/QKOFrEsDPnFSnzVyd6uPBkVsD3dpFAA==","signature_status":"signed_v1","signed_at":"2026-07-05T08:52:53.982591Z","signed_message":"canonical_sha256_bytes"},"source_id":"2405.01511","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:572b6c99c428afac454dec544a63405013bb5dee6c90a4ec070bb54f86f35ffa","sha256:1a298355044e01a7b6c72bbb1fb611d321da19e5e419829107c35844387ee067"],"state_sha256":"2750a10e13fdc3afb0138f026fd687a15e04a24439210ad1cd4731fda89d0955"}