{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MKKCRB52M5U6INECS7HAF3BYKM","short_pith_number":"pith:MKKCRB52","schema_version":"1.0","canonical_sha256":"62942887ba6769e4348297ce02ec38532dd93fdfdb569d929a1875d2b3821cb8","source":{"kind":"arxiv","id":"2503.01076","version":1},"attestation_state":"computed","paper":{"title":"Active Learning for Direct Preference Optimization","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Branislav Kveton, Jingbo Shang, Julian McAuley, Junda Wu, Ryan Rossi, Tong Yu, Xintong Li","submitted_at":"2025-03-03T00:36:31Z","abstract_excerpt":"Direct preference optimization (DPO) is a form of reinforcement learning from human feedback (RLHF) where the policy is learned directly from preferential feedback. Although many models of human preferences exist, the critical task of selecting the most informative feedback for training them is under-explored. We propose an active learning framework for DPO, which can be applied to collect human feedback online or to choose the most informative subset of already collected feedback offline. We propose efficient algorithms for both settings. The key idea is to linearize the DPO objective at the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.01076","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2025-03-03T00:36:31Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"0fd0f55470d763f68a63c23f9afbe091b1394647c7fe27a264f9d8b6280f8896","abstract_canon_sha256":"0135a3ace33b59539daed558ef0a6b6efe658d4a79bbef9b743c9005e1a0593d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:35.497218Z","signature_b64":"CXgqCTBNSuNiYlfcuQkyNPaRMtncMbuuYF6i8EUGv4iqig4k9ZMmzwvV9Z6QYHL0yo8IaGiQdCxWsvBQvFTyAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"62942887ba6769e4348297ce02ec38532dd93fdfdb569d929a1875d2b3821cb8","last_reissued_at":"2026-07-05T10:22:35.496812Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:35.496812Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Active Learning for Direct Preference Optimization","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Branislav Kveton, Jingbo Shang, Julian McAuley, Junda Wu, Ryan Rossi, Tong Yu, Xintong Li","submitted_at":"2025-03-03T00:36:31Z","abstract_excerpt":"Direct preference optimization (DPO) is a form of reinforcement learning from human feedback (RLHF) where the policy is learned directly from preferential feedback. Although many models of human preferences exist, the critical task of selecting the most informative feedback for training them is under-explored. We propose an active learning framework for DPO, which can be applied to collect human feedback online or to choose the most informative subset of already collected feedback offline. We propose efficient algorithms for both settings. The key idea is to linearize the DPO objective at the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.01076","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.01076/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.01076","created_at":"2026-07-05T10:22:35.496869+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.01076v1","created_at":"2026-07-05T10:22:35.496869+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.01076","created_at":"2026-07-05T10:22:35.496869+00:00"},{"alias_kind":"pith_short_12","alias_value":"MKKCRB52M5U6","created_at":"2026-07-05T10:22:35.496869+00:00"},{"alias_kind":"pith_short_16","alias_value":"MKKCRB52M5U6INEC","created_at":"2026-07-05T10:22:35.496869+00:00"},{"alias_kind":"pith_short_8","alias_value":"MKKCRB52","created_at":"2026-07-05T10:22:35.496869+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19607","citing_title":"Which Pairs to Compare for LLM Post-Training?","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12995","citing_title":"F-GRPO: Factorized Group-Relative Policy Optimization for Unified Candidate Generation and Ranking","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02766","citing_title":"Random Is Hard to Beat: Active Selection in online DPO with Modern LLMs","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11169","citing_title":"OLIVIA: Online Learning via Inference-time Action Adaptation for Decision Making in LLM ReAct Agents","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08526","citing_title":"Skill-CMIB: Multimodal Agent Skill for Consistent Action via Conditional Multimodal Information Bottleneck","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10784","citing_title":"MASS-DPO: Multi-negative Active Sample Selection for Direct Policy Optimization","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MKKCRB52M5U6INECS7HAF3BYKM","json":"https://pith.science/pith/MKKCRB52M5U6INECS7HAF3BYKM.json","graph_json":"https://pith.science/api/pith-number/MKKCRB52M5U6INECS7HAF3BYKM/graph.json","events_json":"https://pith.science/api/pith-number/MKKCRB52M5U6INECS7HAF3BYKM/events.json","paper":"https://pith.science/paper/MKKCRB52"},"agent_actions":{"view_html":"https://pith.science/pith/MKKCRB52M5U6INECS7HAF3BYKM","download_json":"https://pith.science/pith/MKKCRB52M5U6INECS7HAF3BYKM.json","view_paper":"https://pith.science/paper/MKKCRB52","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.01076&json=true","fetch_graph":"https://pith.science/api/pith-number/MKKCRB52M5U6INECS7HAF3BYKM/graph.json","fetch_events":"https://pith.science/api/pith-number/MKKCRB52M5U6INECS7HAF3BYKM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MKKCRB52M5U6INECS7HAF3BYKM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MKKCRB52M5U6INECS7HAF3BYKM/action/storage_attestation","attest_author":"https://pith.science/pith/MKKCRB52M5U6INECS7HAF3BYKM/action/author_attestation","sign_citation":"https://pith.science/pith/MKKCRB52M5U6INECS7HAF3BYKM/action/citation_signature","submit_replication":"https://pith.science/pith/MKKCRB52M5U6INECS7HAF3BYKM/action/replication_record"}},"created_at":"2026-07-05T10:22:35.496869+00:00","updated_at":"2026-07-05T10:22:35.496869+00:00"}