{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:XBQN44YYN5Q2QNGDXEKJGVAV6I","short_pith_number":"pith:XBQN44YY","canonical_record":{"source":{"id":"2311.09641","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-11-16T07:48:45Z","cross_cats_sorted":["cs.CL","cs.CR","cs.HC"],"title_canon_sha256":"54a5c82f4c696f798fbcdb542e634df9ba4ca0165242f041adb7976fcaf18a33","abstract_canon_sha256":"14a70659a62525a38aaa044c68d8268614636705cb18f379765692eeb8ba8b91"},"schema_version":"1.0"},"canonical_sha256":"b860de73186f61a834c3b914935415f23dc629c5056ac502e0f48fdb9d43c702","source":{"kind":"arxiv","id":"2311.09641","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2311.09641","created_at":"2026-07-05T08:34:07Z"},{"alias_kind":"arxiv_version","alias_value":"2311.09641v2","created_at":"2026-07-05T08:34:07Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.09641","created_at":"2026-07-05T08:34:07Z"},{"alias_kind":"pith_short_12","alias_value":"XBQN44YYN5Q2","created_at":"2026-07-05T08:34:07Z"},{"alias_kind":"pith_short_16","alias_value":"XBQN44YYN5Q2QNGD","created_at":"2026-07-05T08:34:07Z"},{"alias_kind":"pith_short_8","alias_value":"XBQN44YY","created_at":"2026-07-05T08:34:07Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:XBQN44YYN5Q2QNGDXEKJGVAV6I","target":"record","payload":{"canonical_record":{"source":{"id":"2311.09641","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-11-16T07:48:45Z","cross_cats_sorted":["cs.CL","cs.CR","cs.HC"],"title_canon_sha256":"54a5c82f4c696f798fbcdb542e634df9ba4ca0165242f041adb7976fcaf18a33","abstract_canon_sha256":"14a70659a62525a38aaa044c68d8268614636705cb18f379765692eeb8ba8b91"},"schema_version":"1.0"},"canonical_sha256":"b860de73186f61a834c3b914935415f23dc629c5056ac502e0f48fdb9d43c702","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:34:07.793987Z","signature_b64":"sU28yeZlTpKQnIlSic+D4IoymREzo1zuz2/wChmX7wgglkZkHyXJG0ZaJoDA3sYYPreUVGrPKv+nrI9CVOz0CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b860de73186f61a834c3b914935415f23dc629c5056ac502e0f48fdb9d43c702","last_reissued_at":"2026-07-05T08:34:07.793572Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:34:07.793572Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2311.09641","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:34:07Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"9APvC3+WjVUtnPl2/AEZZ7Xm5+auOpY6tIxB6gUivk9Dimxy+0ijT17+uh+xWom7dKD94g9RhD/C5W3vPj8XAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T03:43:31.844147Z"},"content_sha256":"2d952d44e78647e57b496780198dd2f34e54d956035f294223ca23818b0edbc4","schema_version":"1.0","event_id":"sha256:2d952d44e78647e57b496780198dd2f34e54d956035f294223ca23818b0edbc4"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:XBQN44YYN5Q2QNGDXEKJGVAV6I","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"RLHFPoison: Reward Poisoning Attack for Reinforcement Learning with Human Feedback in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CR","cs.HC"],"primary_cat":"cs.AI","authors_text":"Chaowei Xiao, Jiongxiao Wang, Junlin Wu, Muhao Chen, Yevgeniy Vorobeychik","submitted_at":"2023-11-16T07:48:45Z","abstract_excerpt":"Reinforcement Learning with Human Feedback (RLHF) is a methodology designed to align Large Language Models (LLMs) with human preferences, playing an important role in LLMs alignment. Despite its advantages, RLHF relies on human annotators to rank the text, which can introduce potential security vulnerabilities if any adversarial annotator (i.e., attackers) manipulates the ranking score by up-ranking any malicious text to steer the LLM adversarially. To assess the red-teaming of RLHF against human preference data poisoning, we propose RankPoison, a poisoning attack method on candidates' selecti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.09641","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.09641/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:34:07Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"FPjGcVjy2wHbLqyjZReTUeEH53ODUtt/Nz//tyKw6xgnA2P+CoaE0Sx8sJH+8gr3a06mmD/1HqfosJmGHl6lDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T03:43:31.845055Z"},"content_sha256":"7ba7659683923a9621af843110ddaf19090417ede191cad2607f48e36bcc711f","schema_version":"1.0","event_id":"sha256:7ba7659683923a9621af843110ddaf19090417ede191cad2607f48e36bcc711f"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/XBQN44YYN5Q2QNGDXEKJGVAV6I/bundle.json","state_url":"https://pith.science/pith/XBQN44YYN5Q2QNGDXEKJGVAV6I/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/XBQN44YYN5Q2QNGDXEKJGVAV6I/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T03:43:31Z","links":{"resolver":"https://pith.science/pith/XBQN44YYN5Q2QNGDXEKJGVAV6I","bundle":"https://pith.science/pith/XBQN44YYN5Q2QNGDXEKJGVAV6I/bundle.json","state":"https://pith.science/pith/XBQN44YYN5Q2QNGDXEKJGVAV6I/state.json","well_known_bundle":"https://pith.science/.well-known/pith/XBQN44YYN5Q2QNGDXEKJGVAV6I/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:XBQN44YYN5Q2QNGDXEKJGVAV6I","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"14a70659a62525a38aaa044c68d8268614636705cb18f379765692eeb8ba8b91","cross_cats_sorted":["cs.CL","cs.CR","cs.HC"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-11-16T07:48:45Z","title_canon_sha256":"54a5c82f4c696f798fbcdb542e634df9ba4ca0165242f041adb7976fcaf18a33"},"schema_version":"1.0","source":{"id":"2311.09641","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2311.09641","created_at":"2026-07-05T08:34:07Z"},{"alias_kind":"arxiv_version","alias_value":"2311.09641v2","created_at":"2026-07-05T08:34:07Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.09641","created_at":"2026-07-05T08:34:07Z"},{"alias_kind":"pith_short_12","alias_value":"XBQN44YYN5Q2","created_at":"2026-07-05T08:34:07Z"},{"alias_kind":"pith_short_16","alias_value":"XBQN44YYN5Q2QNGD","created_at":"2026-07-05T08:34:07Z"},{"alias_kind":"pith_short_8","alias_value":"XBQN44YY","created_at":"2026-07-05T08:34:07Z"}],"graph_snapshots":[{"event_id":"sha256:7ba7659683923a9621af843110ddaf19090417ede191cad2607f48e36bcc711f","target":"graph","created_at":"2026-07-05T08:34:07Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2311.09641/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning with Human Feedback (RLHF) is a methodology designed to align Large Language Models (LLMs) with human preferences, playing an important role in LLMs alignment. Despite its advantages, RLHF relies on human annotators to rank the text, which can introduce potential security vulnerabilities if any adversarial annotator (i.e., attackers) manipulates the ranking score by up-ranking any malicious text to steer the LLM adversarially. To assess the red-teaming of RLHF against human preference data poisoning, we propose RankPoison, a poisoning attack method on candidates' selecti","authors_text":"Chaowei Xiao, Jiongxiao Wang, Junlin Wu, Muhao Chen, Yevgeniy Vorobeychik","cross_cats":["cs.CL","cs.CR","cs.HC"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-11-16T07:48:45Z","title":"RLHFPoison: Reward Poisoning Attack for Reinforcement Learning with Human Feedback in Large Language Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.09641","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:2d952d44e78647e57b496780198dd2f34e54d956035f294223ca23818b0edbc4","target":"record","created_at":"2026-07-05T08:34:07Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"14a70659a62525a38aaa044c68d8268614636705cb18f379765692eeb8ba8b91","cross_cats_sorted":["cs.CL","cs.CR","cs.HC"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-11-16T07:48:45Z","title_canon_sha256":"54a5c82f4c696f798fbcdb542e634df9ba4ca0165242f041adb7976fcaf18a33"},"schema_version":"1.0","source":{"id":"2311.09641","kind":"arxiv","version":2}},"canonical_sha256":"b860de73186f61a834c3b914935415f23dc629c5056ac502e0f48fdb9d43c702","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b860de73186f61a834c3b914935415f23dc629c5056ac502e0f48fdb9d43c702","first_computed_at":"2026-07-05T08:34:07.793572Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:34:07.793572Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"sU28yeZlTpKQnIlSic+D4IoymREzo1zuz2/wChmX7wgglkZkHyXJG0ZaJoDA3sYYPreUVGrPKv+nrI9CVOz0CQ==","signature_status":"signed_v1","signed_at":"2026-07-05T08:34:07.793987Z","signed_message":"canonical_sha256_bytes"},"source_id":"2311.09641","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:2d952d44e78647e57b496780198dd2f34e54d956035f294223ca23818b0edbc4","sha256:7ba7659683923a9621af843110ddaf19090417ede191cad2607f48e36bcc711f"],"state_sha256":"b44632077250f52a284e8083ec12b0b3ccab930941f5e481ed2170525220c39f"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"2O5sPGt4q9EPdt9kI8mgJ2SOW4jbLunQ6mGh++fSzm0W/NjbBStO5/aPo3wvPA1Y2PP+Y18XylLwSKk0XzSBBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T03:43:31.858500Z","bundle_sha256":"8f1cbc59d496464d3731920824b12e067defdb8d0255ba46f09d833698ddc301"}}