{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:W64HXUGA45OU7Z7UQZ666EDWLB","short_pith_number":"pith:W64HXUGA","schema_version":"1.0","canonical_sha256":"b7b87bd0c0e75d4fe7f4867def107658418f8f649e32e73541bf5b74158d19b7","source":{"kind":"arxiv","id":"2402.01920","version":2},"attestation_state":"computed","paper":{"title":"Preference Poisoning Attacks on Reward Model Learning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Chaowei Xiao, Chenguang Wang, Jiongxiao Wang, Junlin Wu, Ning Zhang, Yevgeniy Vorobeychik","submitted_at":"2024-02-02T21:45:24Z","abstract_excerpt":"Learning reward models from pairwise comparisons is a fundamental component in a number of domains, including autonomous control, conversational agents, and recommendation systems, as part of a broad goal of aligning automated decisions with user preferences. These approaches entail collecting preference information from people, with feedback often provided anonymously. Since preferences are subjective, there is no gold standard to compare against; yet, reliance of high-impact systems on preference learning creates a strong motivation for malicious actors to skew data collected in this fashion"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.01920","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-02T21:45:24Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"867c7a19778a05eec34e4537c54a24cced1c22f9db67e5d72fe8fcbf0fc8301c","abstract_canon_sha256":"5e0e4c88f985f4c310c553cfda6a3d0f16384ce0f71a5750d3c6d194cfdd90c5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:17:37.096691Z","signature_b64":"BvkDH9bQaGrBxbfjjGstKLk6+p4fG05bne1vwOyJO+2Gai4dklGjPg2xQdtwF84zbUIkg0Bs/ud2xFG18PIpCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b7b87bd0c0e75d4fe7f4867def107658418f8f649e32e73541bf5b74158d19b7","last_reissued_at":"2026-07-05T09:17:37.096093Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:17:37.096093Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Preference Poisoning Attacks on Reward Model Learning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Chaowei Xiao, Chenguang Wang, Jiongxiao Wang, Junlin Wu, Ning Zhang, Yevgeniy Vorobeychik","submitted_at":"2024-02-02T21:45:24Z","abstract_excerpt":"Learning reward models from pairwise comparisons is a fundamental component in a number of domains, including autonomous control, conversational agents, and recommendation systems, as part of a broad goal of aligning automated decisions with user preferences. These approaches entail collecting preference information from people, with feedback often provided anonymously. Since preferences are subjective, there is no gold standard to compare against; yet, reliance of high-impact systems on preference learning creates a strong motivation for malicious actors to skew data collected in this fashion"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.01920","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.01920/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.01920","created_at":"2026-07-05T09:17:37.096168+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.01920v2","created_at":"2026-07-05T09:17:37.096168+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.01920","created_at":"2026-07-05T09:17:37.096168+00:00"},{"alias_kind":"pith_short_12","alias_value":"W64HXUGA45OU","created_at":"2026-07-05T09:17:37.096168+00:00"},{"alias_kind":"pith_short_16","alias_value":"W64HXUGA45OU7Z7U","created_at":"2026-07-05T09:17:37.096168+00:00"},{"alias_kind":"pith_short_8","alias_value":"W64HXUGA","created_at":"2026-07-05T09:17:37.096168+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30666","citing_title":"Reframing AGI Confrontation with Off Earth Autonomy","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02495","citing_title":"Efficient Preference Poisoning Attack on Offline RLHF","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W64HXUGA45OU7Z7UQZ666EDWLB","json":"https://pith.science/pith/W64HXUGA45OU7Z7UQZ666EDWLB.json","graph_json":"https://pith.science/api/pith-number/W64HXUGA45OU7Z7UQZ666EDWLB/graph.json","events_json":"https://pith.science/api/pith-number/W64HXUGA45OU7Z7UQZ666EDWLB/events.json","paper":"https://pith.science/paper/W64HXUGA"},"agent_actions":{"view_html":"https://pith.science/pith/W64HXUGA45OU7Z7UQZ666EDWLB","download_json":"https://pith.science/pith/W64HXUGA45OU7Z7UQZ666EDWLB.json","view_paper":"https://pith.science/paper/W64HXUGA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.01920&json=true","fetch_graph":"https://pith.science/api/pith-number/W64HXUGA45OU7Z7UQZ666EDWLB/graph.json","fetch_events":"https://pith.science/api/pith-number/W64HXUGA45OU7Z7UQZ666EDWLB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W64HXUGA45OU7Z7UQZ666EDWLB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W64HXUGA45OU7Z7UQZ666EDWLB/action/storage_attestation","attest_author":"https://pith.science/pith/W64HXUGA45OU7Z7UQZ666EDWLB/action/author_attestation","sign_citation":"https://pith.science/pith/W64HXUGA45OU7Z7UQZ666EDWLB/action/citation_signature","submit_replication":"https://pith.science/pith/W64HXUGA45OU7Z7UQZ666EDWLB/action/replication_record"}},"created_at":"2026-07-05T09:17:37.096168+00:00","updated_at":"2026-07-05T09:17:37.096168+00:00"}