{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BKMW7MMTJSOB3I5UQAZIN7T6WB","short_pith_number":"pith:BKMW7MMT","schema_version":"1.0","canonical_sha256":"0a996fb1934c9c1da3b4803286fe7eb063ef3d4b664a891efaba389c803d68d9","source":{"kind":"arxiv","id":"2405.12900","version":1},"attestation_state":"computed","paper":{"title":"Adversarial DPO: Harnessing Harmful Data for Reducing Toxicity with Minimal Impact on Coherence and Evasiveness in Dialogue Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Gary Geunbae Lee, San Kim","submitted_at":"2024-05-21T16:14:55Z","abstract_excerpt":"Recent advancements in open-domain dialogue systems have been propelled by the emergence of high-quality large language models (LLMs) and various effective training methodologies. Nevertheless, the presence of toxicity within these models presents a significant challenge that can potentially diminish the user experience. In this study, we introduce an innovative training algorithm, an improvement upon direct preference optimization (DPO), called adversarial DPO (ADPO). The ADPO algorithm is designed to train models to assign higher probability distributions to preferred responses and lower dis"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.12900","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-21T16:14:55Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fab98d05bab94cb3982e4b519d93b10929931aa071aa2d7d6b04b0e1ccdc0231","abstract_canon_sha256":"d723276613002d1837ed2ecb7edc30059ca081737c736bf4ef87aad76e139b2d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:21:28.096966Z","signature_b64":"NAYgbZrHiQ+vuLvKz2tPv4REE7mgBp6gRasrNMdDJwncbnrqDDX3bkZsmKO29tc4Cm+HJhIjsmluWcIVC4yxBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0a996fb1934c9c1da3b4803286fe7eb063ef3d4b664a891efaba389c803d68d9","last_reissued_at":"2026-07-05T08:21:28.096451Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:21:28.096451Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adversarial DPO: Harnessing Harmful Data for Reducing Toxicity with Minimal Impact on Coherence and Evasiveness in Dialogue Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Gary Geunbae Lee, San Kim","submitted_at":"2024-05-21T16:14:55Z","abstract_excerpt":"Recent advancements in open-domain dialogue systems have been propelled by the emergence of high-quality large language models (LLMs) and various effective training methodologies. Nevertheless, the presence of toxicity within these models presents a significant challenge that can potentially diminish the user experience. In this study, we introduce an innovative training algorithm, an improvement upon direct preference optimization (DPO), called adversarial DPO (ADPO). The ADPO algorithm is designed to train models to assign higher probability distributions to preferred responses and lower dis"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.12900","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.12900/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.12900","created_at":"2026-07-05T08:21:28.096515+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.12900v1","created_at":"2026-07-05T08:21:28.096515+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.12900","created_at":"2026-07-05T08:21:28.096515+00:00"},{"alias_kind":"pith_short_12","alias_value":"BKMW7MMTJSOB","created_at":"2026-07-05T08:21:28.096515+00:00"},{"alias_kind":"pith_short_16","alias_value":"BKMW7MMTJSOB3I5U","created_at":"2026-07-05T08:21:28.096515+00:00"},{"alias_kind":"pith_short_8","alias_value":"BKMW7MMT","created_at":"2026-07-05T08:21:28.096515+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2507.05660","citing_title":"Optimus: A Robust Defense Framework for Mitigating Toxicity while Fine-Tuning Conversational AI","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BKMW7MMTJSOB3I5UQAZIN7T6WB","json":"https://pith.science/pith/BKMW7MMTJSOB3I5UQAZIN7T6WB.json","graph_json":"https://pith.science/api/pith-number/BKMW7MMTJSOB3I5UQAZIN7T6WB/graph.json","events_json":"https://pith.science/api/pith-number/BKMW7MMTJSOB3I5UQAZIN7T6WB/events.json","paper":"https://pith.science/paper/BKMW7MMT"},"agent_actions":{"view_html":"https://pith.science/pith/BKMW7MMTJSOB3I5UQAZIN7T6WB","download_json":"https://pith.science/pith/BKMW7MMTJSOB3I5UQAZIN7T6WB.json","view_paper":"https://pith.science/paper/BKMW7MMT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.12900&json=true","fetch_graph":"https://pith.science/api/pith-number/BKMW7MMTJSOB3I5UQAZIN7T6WB/graph.json","fetch_events":"https://pith.science/api/pith-number/BKMW7MMTJSOB3I5UQAZIN7T6WB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BKMW7MMTJSOB3I5UQAZIN7T6WB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BKMW7MMTJSOB3I5UQAZIN7T6WB/action/storage_attestation","attest_author":"https://pith.science/pith/BKMW7MMTJSOB3I5UQAZIN7T6WB/action/author_attestation","sign_citation":"https://pith.science/pith/BKMW7MMTJSOB3I5UQAZIN7T6WB/action/citation_signature","submit_replication":"https://pith.science/pith/BKMW7MMTJSOB3I5UQAZIN7T6WB/action/replication_record"}},"created_at":"2026-07-05T08:21:28.096515+00:00","updated_at":"2026-07-05T08:21:28.096515+00:00"}