{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RCMQI7W73T2VUAYO67OCG3ZWUP","short_pith_number":"pith:RCMQI7W7","schema_version":"1.0","canonical_sha256":"8899047edfdcf55a030ef7dc236f36a3c15bbce8cc0325ef655982fa03fbf13d","source":{"kind":"arxiv","id":"2403.00409","version":2},"attestation_state":"computed","paper":{"title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Anush Kini, Nagarajan Natarajan, Sayak Ray Chowdhury","submitted_at":"2024-03-01T09:55:18Z","abstract_excerpt":"Learning from preference-based feedback has recently gained traction as a promising approach to align language models with human interests. While these aligned generative models have demonstrated impressive capabilities across various tasks, their dependence on high-quality human preference data poses a bottleneck in practical applications. Specifically, noisy (incorrect and ambiguous) preference pairs in the dataset might restrict the language models from capturing human intent accurately. While practitioners have recently proposed heuristics to mitigate the effect of noisy preferences, a com"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.00409","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-01T09:55:18Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"71f99f800c9bddede497af3b8a519f6ea4b3494186b6d3236d4cd18ff4bc055e","abstract_canon_sha256":"d3a48134f1e8851b5980bccfbe07f9410973a9024b2d18ac5d47d4ff81f9977c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:07:14.805939Z","signature_b64":"0D9x01+7k+58hZIRaowqlLkWKKBqHaY/BKqfN4ma3VmUQAv+rO27nj9GamiIE89+IDB0LZ2YDSB83NyZxm+fCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8899047edfdcf55a030ef7dc236f36a3c15bbce8cc0325ef655982fa03fbf13d","last_reissued_at":"2026-07-05T08:07:14.805576Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:07:14.805576Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Provably Robust DPO: Aligning Language Models with Noisy Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Anush Kini, Nagarajan Natarajan, Sayak Ray Chowdhury","submitted_at":"2024-03-01T09:55:18Z","abstract_excerpt":"Learning from preference-based feedback has recently gained traction as a promising approach to align language models with human interests. While these aligned generative models have demonstrated impressive capabilities across various tasks, their dependence on high-quality human preference data poses a bottleneck in practical applications. Specifically, noisy (incorrect and ambiguous) preference pairs in the dataset might restrict the language models from capturing human intent accurately. While practitioners have recently proposed heuristics to mitigate the effect of noisy preferences, a com"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.00409","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.00409/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.00409","created_at":"2026-07-05T08:07:14.805632+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.00409v2","created_at":"2026-07-05T08:07:14.805632+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.00409","created_at":"2026-07-05T08:07:14.805632+00:00"},{"alias_kind":"pith_short_12","alias_value":"RCMQI7W73T2V","created_at":"2026-07-05T08:07:14.805632+00:00"},{"alias_kind":"pith_short_16","alias_value":"RCMQI7W73T2VUAYO","created_at":"2026-07-05T08:07:14.805632+00:00"},{"alias_kind":"pith_short_8","alias_value":"RCMQI7W7","created_at":"2026-07-05T08:07:14.805632+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24937","citing_title":"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems","ref_index":190,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19607","citing_title":"Which Pairs to Compare for LLM Post-Training?","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23398","citing_title":"TPMM-DPO: Trajectory-aware Preference-guided Model Merging for Iterative Direct Preference Optimization","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2502.06387","citing_title":"How Humans Help LLMs: Assessing and Incentivizing Human Preference Annotators","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2509.08933","citing_title":"Corruption-Tolerant Asynchronous Q-Learning with Near-Optimal Rates","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2505.19134","citing_title":"Incentivizing High-Quality Human Annotations with Golden Questions","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13830","citing_title":"Users as Annotators: LLM Preference Learning from Comparison Mode","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2411.10442","citing_title":"Enhancing the Reasoning Ability of Multimodal Large Language Models via Mixed Preference Optimization","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11134","citing_title":"Spurious Correlation Learning in Preference Optimization: Mechanisms, Consequences, and Mitigation via Tie Training","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02971","citing_title":"Multilingual Safety Alignment via Self-Distillation","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02971","citing_title":"Multilingual Safety Alignment via Self-Distillation","ref_index":46,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RCMQI7W73T2VUAYO67OCG3ZWUP","json":"https://pith.science/pith/RCMQI7W73T2VUAYO67OCG3ZWUP.json","graph_json":"https://pith.science/api/pith-number/RCMQI7W73T2VUAYO67OCG3ZWUP/graph.json","events_json":"https://pith.science/api/pith-number/RCMQI7W73T2VUAYO67OCG3ZWUP/events.json","paper":"https://pith.science/paper/RCMQI7W7"},"agent_actions":{"view_html":"https://pith.science/pith/RCMQI7W73T2VUAYO67OCG3ZWUP","download_json":"https://pith.science/pith/RCMQI7W73T2VUAYO67OCG3ZWUP.json","view_paper":"https://pith.science/paper/RCMQI7W7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.00409&json=true","fetch_graph":"https://pith.science/api/pith-number/RCMQI7W73T2VUAYO67OCG3ZWUP/graph.json","fetch_events":"https://pith.science/api/pith-number/RCMQI7W73T2VUAYO67OCG3ZWUP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RCMQI7W73T2VUAYO67OCG3ZWUP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RCMQI7W73T2VUAYO67OCG3ZWUP/action/storage_attestation","attest_author":"https://pith.science/pith/RCMQI7W73T2VUAYO67OCG3ZWUP/action/author_attestation","sign_citation":"https://pith.science/pith/RCMQI7W73T2VUAYO67OCG3ZWUP/action/citation_signature","submit_replication":"https://pith.science/pith/RCMQI7W73T2VUAYO67OCG3ZWUP/action/replication_record"}},"created_at":"2026-07-05T08:07:14.805632+00:00","updated_at":"2026-07-05T08:07:14.805632+00:00"}