{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:F5ZUBTTB6VDBSHQJG4RNVKFVMB","short_pith_number":"pith:F5ZUBTTB","schema_version":"1.0","canonical_sha256":"2f7340ce61f546191e093722daa8b560707d8160b7ed88b2bb1c9dd2d30e8966","source":{"kind":"arxiv","id":"2406.05534","version":1},"attestation_state":"computed","paper":{"title":"Online DPO: Online Direct Preference Optimization with Fast-Slow Chasing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Biqing Qi, Bowen Zhou, Fangyuan Li, Junqi Gao, Kaiyan Zhang, Pengfei Li","submitted_at":"2024-06-08T17:30:54Z","abstract_excerpt":"Direct Preference Optimization (DPO) improves the alignment of large language models (LLMs) with human values by training directly on human preference datasets, eliminating the need for reward models. However, due to the presence of cross-domain human preferences, direct continual training can lead to catastrophic forgetting, limiting DPO's performance and efficiency. Inspired by intraspecific competition driving species evolution, we propose a Online Fast-Slow chasing DPO (OFS-DPO) for preference alignment, simulating competition through fast and slow chasing among models to facilitate rapid "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.05534","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-06-08T17:30:54Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"2b295ba7f11a84400f3c8307aead2bb7bd3840be0470e3c0ab2e435924d7e143","abstract_canon_sha256":"d586ab1550f2e22b83ac8cfcde81836465137827dff49e3586e462b312917382"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:29:29.499955Z","signature_b64":"dKggBQhzff2ElgrazkKopPBFLfDTJEWDh6vMhu5O0VpuVg9RzQGxOZ7Dh0HBn9S43deLc7rWHokx4uXNJofnAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2f7340ce61f546191e093722daa8b560707d8160b7ed88b2bb1c9dd2d30e8966","last_reissued_at":"2026-07-05T08:29:29.499447Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:29:29.499447Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Online DPO: Online Direct Preference Optimization with Fast-Slow Chasing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Biqing Qi, Bowen Zhou, Fangyuan Li, Junqi Gao, Kaiyan Zhang, Pengfei Li","submitted_at":"2024-06-08T17:30:54Z","abstract_excerpt":"Direct Preference Optimization (DPO) improves the alignment of large language models (LLMs) with human values by training directly on human preference datasets, eliminating the need for reward models. However, due to the presence of cross-domain human preferences, direct continual training can lead to catastrophic forgetting, limiting DPO's performance and efficiency. Inspired by intraspecific competition driving species evolution, we propose a Online Fast-Slow chasing DPO (OFS-DPO) for preference alignment, simulating competition through fast and slow chasing among models to facilitate rapid "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.05534","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.05534/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.05534","created_at":"2026-07-05T08:29:29.499497+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.05534v1","created_at":"2026-07-05T08:29:29.499497+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.05534","created_at":"2026-07-05T08:29:29.499497+00:00"},{"alias_kind":"pith_short_12","alias_value":"F5ZUBTTB6VDB","created_at":"2026-07-05T08:29:29.499497+00:00"},{"alias_kind":"pith_short_16","alias_value":"F5ZUBTTB6VDBSHQJ","created_at":"2026-07-05T08:29:29.499497+00:00"},{"alias_kind":"pith_short_8","alias_value":"F5ZUBTTB","created_at":"2026-07-05T08:29:29.499497+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09255","citing_title":"RPO-PDT: Demonstrating Role-Play-Based Knowledge Adaptation for Student Support Dialogue (Demonstration System)","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05040","citing_title":"Preference-Based Self-Distillation: Beyond KL Matching via Reward Regularization","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07941","citing_title":"Large Language Model Post-Training: A Unified View of Off-Policy and On-Policy Learning","ref_index":98,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F5ZUBTTB6VDBSHQJG4RNVKFVMB","json":"https://pith.science/pith/F5ZUBTTB6VDBSHQJG4RNVKFVMB.json","graph_json":"https://pith.science/api/pith-number/F5ZUBTTB6VDBSHQJG4RNVKFVMB/graph.json","events_json":"https://pith.science/api/pith-number/F5ZUBTTB6VDBSHQJG4RNVKFVMB/events.json","paper":"https://pith.science/paper/F5ZUBTTB"},"agent_actions":{"view_html":"https://pith.science/pith/F5ZUBTTB6VDBSHQJG4RNVKFVMB","download_json":"https://pith.science/pith/F5ZUBTTB6VDBSHQJG4RNVKFVMB.json","view_paper":"https://pith.science/paper/F5ZUBTTB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.05534&json=true","fetch_graph":"https://pith.science/api/pith-number/F5ZUBTTB6VDBSHQJG4RNVKFVMB/graph.json","fetch_events":"https://pith.science/api/pith-number/F5ZUBTTB6VDBSHQJG4RNVKFVMB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F5ZUBTTB6VDBSHQJG4RNVKFVMB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F5ZUBTTB6VDBSHQJG4RNVKFVMB/action/storage_attestation","attest_author":"https://pith.science/pith/F5ZUBTTB6VDBSHQJG4RNVKFVMB/action/author_attestation","sign_citation":"https://pith.science/pith/F5ZUBTTB6VDBSHQJG4RNVKFVMB/action/citation_signature","submit_replication":"https://pith.science/pith/F5ZUBTTB6VDBSHQJG4RNVKFVMB/action/replication_record"}},"created_at":"2026-07-05T08:29:29.499497+00:00","updated_at":"2026-07-05T08:29:29.499497+00:00"}