{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:O235XRQOVJOLUNFB5VF7GTVWF2","short_pith_number":"pith:O235XRQO","schema_version":"1.0","canonical_sha256":"76b7dbc60eaa5cba34a1ed4bf34eb62e8c4b9a6187ffae12798080367a2214fe","source":{"kind":"arxiv","id":"2406.12091","version":4},"attestation_state":"computed","paper":{"title":"Is poisoning a real threat to LLM alignment? Maybe more so than you think","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CR"],"primary_cat":"cs.LG","authors_text":"Furong Huang, Pankayaraj Pathmanathan, Souradip Chakraborty, Xiangyu Liu, Yongyuan Liang","submitted_at":"2024-06-17T21:06:00Z","abstract_excerpt":"Recent advancements in Reinforcement Learning with Human Feedback (RLHF) have significantly impacted the alignment of Large Language Models (LLMs). The sensitivity of reinforcement learning algorithms such as Proximal Policy Optimization (PPO) has led to new line work on Direct Policy Optimization (DPO), which treats RLHF in a supervised learning framework. The increased practical use of these RLHF methods warrants an analysis of their vulnerabilities. In this work, we investigate the vulnerabilities of DPO to poisoning attacks under different scenarios and compare the effectiveness of prefere"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.12091","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-17T21:06:00Z","cross_cats_sorted":["cs.CL","cs.CR"],"title_canon_sha256":"bf0069701a10b4b3816e3a8f6874420f4986e57b2433d1bb3376052281f5d25f","abstract_canon_sha256":"0a25fdce08cdd755506e911df31dd9402f38cfd70b23c0b43568d8bac2a6487e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:12.746853Z","signature_b64":"SgiUIYz6hRaaPYBZ826c3FfSL8HcyFyF9vhjdC+lLjOtqRgnsDPZ7224CFJlj5saKqiyQ/mApnuTWMlCJiVGAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"76b7dbc60eaa5cba34a1ed4bf34eb62e8c4b9a6187ffae12798080367a2214fe","last_reissued_at":"2026-07-05T11:18:12.746275Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:12.746275Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Is poisoning a real threat to LLM alignment? Maybe more so than you think","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CR"],"primary_cat":"cs.LG","authors_text":"Furong Huang, Pankayaraj Pathmanathan, Souradip Chakraborty, Xiangyu Liu, Yongyuan Liang","submitted_at":"2024-06-17T21:06:00Z","abstract_excerpt":"Recent advancements in Reinforcement Learning with Human Feedback (RLHF) have significantly impacted the alignment of Large Language Models (LLMs). The sensitivity of reinforcement learning algorithms such as Proximal Policy Optimization (PPO) has led to new line work on Direct Policy Optimization (DPO), which treats RLHF in a supervised learning framework. The increased practical use of these RLHF methods warrants an analysis of their vulnerabilities. In this work, we investigate the vulnerabilities of DPO to poisoning attacks under different scenarios and compare the effectiveness of prefere"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.12091","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.12091/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.12091","created_at":"2026-07-05T11:18:12.746347+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.12091v4","created_at":"2026-07-05T11:18:12.746347+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.12091","created_at":"2026-07-05T11:18:12.746347+00:00"},{"alias_kind":"pith_short_12","alias_value":"O235XRQOVJOL","created_at":"2026-07-05T11:18:12.746347+00:00"},{"alias_kind":"pith_short_16","alias_value":"O235XRQOVJOLUNFB","created_at":"2026-07-05T11:18:12.746347+00:00"},{"alias_kind":"pith_short_8","alias_value":"O235XRQO","created_at":"2026-07-05T11:18:12.746347+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.17110","citing_title":"Loss Landscape Poisoning: Targeted Extraction of Unseen Training Data from LLMs","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2507.02850","citing_title":"LLM Hypnosis: Exploiting User Feedback for Unauthorized Knowledge Injection to All Users","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2507.06419","citing_title":"Teach a Reward Model to Correct Itself: Reward Guided Adversarial Failure Discovery for Robust Reward Modeling","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09397","citing_title":"BadDLM: Backdooring Diffusion Language Models with Diverse Targets","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06833","citing_title":"FedDetox: Robust Federated SLM Alignment via On-Device Data Sanitization","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O235XRQOVJOLUNFB5VF7GTVWF2","json":"https://pith.science/pith/O235XRQOVJOLUNFB5VF7GTVWF2.json","graph_json":"https://pith.science/api/pith-number/O235XRQOVJOLUNFB5VF7GTVWF2/graph.json","events_json":"https://pith.science/api/pith-number/O235XRQOVJOLUNFB5VF7GTVWF2/events.json","paper":"https://pith.science/paper/O235XRQO"},"agent_actions":{"view_html":"https://pith.science/pith/O235XRQOVJOLUNFB5VF7GTVWF2","download_json":"https://pith.science/pith/O235XRQOVJOLUNFB5VF7GTVWF2.json","view_paper":"https://pith.science/paper/O235XRQO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.12091&json=true","fetch_graph":"https://pith.science/api/pith-number/O235XRQOVJOLUNFB5VF7GTVWF2/graph.json","fetch_events":"https://pith.science/api/pith-number/O235XRQOVJOLUNFB5VF7GTVWF2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O235XRQOVJOLUNFB5VF7GTVWF2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O235XRQOVJOLUNFB5VF7GTVWF2/action/storage_attestation","attest_author":"https://pith.science/pith/O235XRQOVJOLUNFB5VF7GTVWF2/action/author_attestation","sign_citation":"https://pith.science/pith/O235XRQOVJOLUNFB5VF7GTVWF2/action/citation_signature","submit_replication":"https://pith.science/pith/O235XRQOVJOLUNFB5VF7GTVWF2/action/replication_record"}},"created_at":"2026-07-05T11:18:12.746347+00:00","updated_at":"2026-07-05T11:18:12.746347+00:00"}