{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:N3I4EHZSBUPHSQVVT25CIDG5GN","short_pith_number":"pith:N3I4EHZS","schema_version":"1.0","canonical_sha256":"6ed1c21f320d1e7942b59eba240cdd33663e8e6608fc666c4846c8da8bc0be05","source":{"kind":"arxiv","id":"2410.08811","version":2},"attestation_state":"computed","paper":{"title":"PoisonBench: Assessing Large Language Model Vulnerability to Data Poisoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CR","authors_text":"David Krueger, Fazl Barez, Mrinank Sharma, Philip Torr, Shay B. Cohen, Tingchen Fu","submitted_at":"2024-10-11T13:50:50Z","abstract_excerpt":"Preference learning is a central component for aligning current LLMs, but this process can be vulnerable to data poisoning attacks. To address this concern, we introduce PoisonBench, a benchmark for evaluating large language models' susceptibility to data poisoning during preference learning. Data poisoning attacks can manipulate large language model responses to include hidden malicious content or biases, potentially causing the model to generate harmful or unintended outputs while appearing to function normally. We deploy two distinct attack types across eight realistic scenarios, assessing "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.08811","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2024-10-11T13:50:50Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"4e9b7197fc744ca2f49a808508e8b28c5ca99f626d2f514473a02590c8ac3911","abstract_canon_sha256":"6afe7a7bad9fcd83cd99738486161c82d393bcd60404629b81cbe88179bb043d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:51.802497Z","signature_b64":"R8gm6+YPnleeCCTnSgSJzRz3Tg6vALjP4Vdi6U9gSXBK4sbOfJvv5o1MB19gSztkvGvgccX9bvKmy4Ki7ZJjDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6ed1c21f320d1e7942b59eba240cdd33663e8e6608fc666c4846c8da8bc0be05","last_reissued_at":"2026-07-05T11:16:51.801999Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:51.801999Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PoisonBench: Assessing Large Language Model Vulnerability to Data Poisoning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CR","authors_text":"David Krueger, Fazl Barez, Mrinank Sharma, Philip Torr, Shay B. Cohen, Tingchen Fu","submitted_at":"2024-10-11T13:50:50Z","abstract_excerpt":"Preference learning is a central component for aligning current LLMs, but this process can be vulnerable to data poisoning attacks. To address this concern, we introduce PoisonBench, a benchmark for evaluating large language models' susceptibility to data poisoning during preference learning. Data poisoning attacks can manipulate large language model responses to include hidden malicious content or biases, potentially causing the model to generate harmful or unintended outputs while appearing to function normally. We deploy two distinct attack types across eight realistic scenarios, assessing "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.08811","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.08811/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.08811","created_at":"2026-07-05T11:16:51.802059+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.08811v2","created_at":"2026-07-05T11:16:51.802059+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.08811","created_at":"2026-07-05T11:16:51.802059+00:00"},{"alias_kind":"pith_short_12","alias_value":"N3I4EHZSBUPH","created_at":"2026-07-05T11:16:51.802059+00:00"},{"alias_kind":"pith_short_16","alias_value":"N3I4EHZSBUPHSQVV","created_at":"2026-07-05T11:16:51.802059+00:00"},{"alias_kind":"pith_short_8","alias_value":"N3I4EHZS","created_at":"2026-07-05T11:16:51.802059+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00036","citing_title":"AI Integrity: Defending Against Backdoors and Secret Loyalties","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19147","citing_title":"Be Kind, Rewrite: Benign Projections via Rewriting Defend Against LLM Data Poisoning Attacks","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2507.02850","citing_title":"LLM Hypnosis: Exploiting User Feedback for Unauthorized Knowledge Injection to All Users","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12529","citing_title":"BackFlush: Knowledge-Free Backdoor Detection and Elimination with Watermark Preservation in Large Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27238","citing_title":"SafeTune: Mitigating Data Poisoning in LLM Fine-Tuning for RTL Code Generation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02495","citing_title":"Efficient Preference Poisoning Attack on Offline RLHF","ref_index":139,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N3I4EHZSBUPHSQVVT25CIDG5GN","json":"https://pith.science/pith/N3I4EHZSBUPHSQVVT25CIDG5GN.json","graph_json":"https://pith.science/api/pith-number/N3I4EHZSBUPHSQVVT25CIDG5GN/graph.json","events_json":"https://pith.science/api/pith-number/N3I4EHZSBUPHSQVVT25CIDG5GN/events.json","paper":"https://pith.science/paper/N3I4EHZS"},"agent_actions":{"view_html":"https://pith.science/pith/N3I4EHZSBUPHSQVVT25CIDG5GN","download_json":"https://pith.science/pith/N3I4EHZSBUPHSQVVT25CIDG5GN.json","view_paper":"https://pith.science/paper/N3I4EHZS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.08811&json=true","fetch_graph":"https://pith.science/api/pith-number/N3I4EHZSBUPHSQVVT25CIDG5GN/graph.json","fetch_events":"https://pith.science/api/pith-number/N3I4EHZSBUPHSQVVT25CIDG5GN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N3I4EHZSBUPHSQVVT25CIDG5GN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N3I4EHZSBUPHSQVVT25CIDG5GN/action/storage_attestation","attest_author":"https://pith.science/pith/N3I4EHZSBUPHSQVVT25CIDG5GN/action/author_attestation","sign_citation":"https://pith.science/pith/N3I4EHZSBUPHSQVVT25CIDG5GN/action/citation_signature","submit_replication":"https://pith.science/pith/N3I4EHZSBUPHSQVVT25CIDG5GN/action/replication_record"}},"created_at":"2026-07-05T11:16:51.802059+00:00","updated_at":"2026-07-05T11:16:51.802059+00:00"}