{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WWOX56KP33VGOV2Y5IDMED4JPQ","short_pith_number":"pith:WWOX56KP","schema_version":"1.0","canonical_sha256":"b59d7ef94fdeea675758ea06c20f897c1e20232450046636bf6a866ad5466ed0","source":{"kind":"arxiv","id":"2409.00787","version":1},"attestation_state":"computed","paper":{"title":"The Dark Side of Human Feedback: Poisoning Large Language Models via User Inputs","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bocheng Chen, Guangjing Wang, Hanqing Guo, Qiben Yan, Yuanda Wang","submitted_at":"2024-09-01T17:40:04Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated great capabilities in natural language understanding and generation, largely attributed to the intricate alignment process using human feedback. While alignment has become an essential training component that leverages data collected from user queries, it inadvertently opens up an avenue for a new type of user-guided poisoning attacks. In this paper, we present a novel exploration into the latent vulnerabilities of the training pipeline in recent LLMs, revealing a subtle yet effective poisoning attack via user-supplied prompts to penetrate alignme"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.00787","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-01T17:40:04Z","cross_cats_sorted":["cs.AI","cs.CR","cs.LG"],"title_canon_sha256":"02c292e451597a5d8b3505184ba67f4e28ea374db6cfde40544f2e03ad833533","abstract_canon_sha256":"cc9db944b3bf508a0076cf5b9d889150424930b2adb61958b22b75a50b277b54"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:01:53.463262Z","signature_b64":"ZIxfmG/yjxl7hcPaHJDAZi+8ecxL6OKgwGx65w0vsiO7axbpJb2N/9zvqGozkYEQHN7VUvT5UfmomCFXtWgyAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b59d7ef94fdeea675758ea06c20f897c1e20232450046636bf6a866ad5466ed0","last_reissued_at":"2026-07-05T09:01:53.462845Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:01:53.462845Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Dark Side of Human Feedback: Poisoning Large Language Models via User Inputs","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bocheng Chen, Guangjing Wang, Hanqing Guo, Qiben Yan, Yuanda Wang","submitted_at":"2024-09-01T17:40:04Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated great capabilities in natural language understanding and generation, largely attributed to the intricate alignment process using human feedback. While alignment has become an essential training component that leverages data collected from user queries, it inadvertently opens up an avenue for a new type of user-guided poisoning attacks. In this paper, we present a novel exploration into the latent vulnerabilities of the training pipeline in recent LLMs, revealing a subtle yet effective poisoning attack via user-supplied prompts to penetrate alignme"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.00787","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.00787/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.00787","created_at":"2026-07-05T09:01:53.462902+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.00787v1","created_at":"2026-07-05T09:01:53.462902+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.00787","created_at":"2026-07-05T09:01:53.462902+00:00"},{"alias_kind":"pith_short_12","alias_value":"WWOX56KP33VG","created_at":"2026-07-05T09:01:53.462902+00:00"},{"alias_kind":"pith_short_16","alias_value":"WWOX56KP33VGOV2Y","created_at":"2026-07-05T09:01:53.462902+00:00"},{"alias_kind":"pith_short_8","alias_value":"WWOX56KP","created_at":"2026-07-05T09:01:53.462902+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2409.18169","citing_title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WWOX56KP33VGOV2Y5IDMED4JPQ","json":"https://pith.science/pith/WWOX56KP33VGOV2Y5IDMED4JPQ.json","graph_json":"https://pith.science/api/pith-number/WWOX56KP33VGOV2Y5IDMED4JPQ/graph.json","events_json":"https://pith.science/api/pith-number/WWOX56KP33VGOV2Y5IDMED4JPQ/events.json","paper":"https://pith.science/paper/WWOX56KP"},"agent_actions":{"view_html":"https://pith.science/pith/WWOX56KP33VGOV2Y5IDMED4JPQ","download_json":"https://pith.science/pith/WWOX56KP33VGOV2Y5IDMED4JPQ.json","view_paper":"https://pith.science/paper/WWOX56KP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.00787&json=true","fetch_graph":"https://pith.science/api/pith-number/WWOX56KP33VGOV2Y5IDMED4JPQ/graph.json","fetch_events":"https://pith.science/api/pith-number/WWOX56KP33VGOV2Y5IDMED4JPQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WWOX56KP33VGOV2Y5IDMED4JPQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WWOX56KP33VGOV2Y5IDMED4JPQ/action/storage_attestation","attest_author":"https://pith.science/pith/WWOX56KP33VGOV2Y5IDMED4JPQ/action/author_attestation","sign_citation":"https://pith.science/pith/WWOX56KP33VGOV2Y5IDMED4JPQ/action/citation_signature","submit_replication":"https://pith.science/pith/WWOX56KP33VGOV2Y5IDMED4JPQ/action/replication_record"}},"created_at":"2026-07-05T09:01:53.462902+00:00","updated_at":"2026-07-05T09:01:53.462902+00:00"}