{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:3KNNUGDBGCNSWTDCCP24CRKL7T","short_pith_number":"pith:3KNNUGDB","canonical_record":{"source":{"id":"2505.24369","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-30T09:02:07Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"16cf7cbf2ef0e6fad742e8443fca77dd188d5c1590ce5e507464931e85044204","abstract_canon_sha256":"c5cb8d16e89a8ce80d9a2e6eee9cebb25607727e396afd3a5dab75463622600f"},"schema_version":"1.0"},"canonical_sha256":"da9ada1861309b2b4c6213f5c1454bfcc3a8d917c0e1fadbdef54b2be9669f05","source":{"kind":"arxiv","id":"2505.24369","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.24369","created_at":"2026-07-05T11:12:47Z"},{"alias_kind":"arxiv_version","alias_value":"2505.24369v1","created_at":"2026-07-05T11:12:47Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.24369","created_at":"2026-07-05T11:12:47Z"},{"alias_kind":"pith_short_12","alias_value":"3KNNUGDBGCNS","created_at":"2026-07-05T11:12:47Z"},{"alias_kind":"pith_short_16","alias_value":"3KNNUGDBGCNSWTDC","created_at":"2026-07-05T11:12:47Z"},{"alias_kind":"pith_short_8","alias_value":"3KNNUGDB","created_at":"2026-07-05T11:12:47Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:3KNNUGDBGCNSWTDCCP24CRKL7T","target":"record","payload":{"canonical_record":{"source":{"id":"2505.24369","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-30T09:02:07Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"16cf7cbf2ef0e6fad742e8443fca77dd188d5c1590ce5e507464931e85044204","abstract_canon_sha256":"c5cb8d16e89a8ce80d9a2e6eee9cebb25607727e396afd3a5dab75463622600f"},"schema_version":"1.0"},"canonical_sha256":"da9ada1861309b2b4c6213f5c1454bfcc3a8d917c0e1fadbdef54b2be9669f05","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:47.855409Z","signature_b64":"Eut1iPl4+tpwFo+/9Iv9yf9jv2KpbzhkDixR/OBwm512wKiPBz6rj5IBanRF9LpgOK5gH3CK57Yz4rJ51IdTAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"da9ada1861309b2b4c6213f5c1454bfcc3a8d917c0e1fadbdef54b2be9669f05","last_reissued_at":"2026-07-05T11:12:47.854764Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:47.854764Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2505.24369","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:12:47Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Sd1Uvs7iBwcdL/mhxNuezMYCA34n6Nt7J+dODM52M2fhUwxwDrVOUhePz/QXzvCT2FEYB7Owm1lPZisWMqbDCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-01T12:32:52.058554Z"},"content_sha256":"2883c999d3c26bfb2a49aa690def49a0d65df056a17e826f656a5f0c952e6d67","schema_version":"1.0","event_id":"sha256:2883c999d3c26bfb2a49aa690def49a0d65df056a17e826f656a5f0c952e6d67"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:3KNNUGDBGCNSWTDCCP24CRKL7T","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Adversarial Preference Learning for Robust LLM Alignment","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Bo Tang, Chaochao Lu, Chao Yang, Chen Chen, Chenyang Xi, Feiyu Xiong, Jie Hu, Jingfeng Zhang, Junyi Zhu, Keming Mao, Mingchuan Yang, Pengyu Wang, Wenqiang Wei, Yijun Niu, Yuanfu Wang, Zhiyu Li","submitted_at":"2025-05-30T09:02:07Z","abstract_excerpt":"Modern language models often rely on Reinforcement Learning from Human Feedback (RLHF) to encourage safe behaviors. However, they remain vulnerable to adversarial attacks due to three key limitations: (1) the inefficiency and high cost of human annotation, (2) the vast diversity of potential adversarial attacks, and (3) the risk of feedback bias and reward hacking. To address these challenges, we introduce Adversarial Preference Learning (APL), an iterative adversarial training method incorporating three key innovations. First, a direct harmfulness metric based on the model's intrinsic prefere"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.24369","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.24369/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:12:47Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"vcskXug8ws3EzjoZ/RWW8q29Nmxq7dF8ztogojEv4AXQ4g2NsIBgqzqFXYHr+zBG9j+/WQ4CCOLjtmqkQR64AQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-01T12:32:52.059278Z"},"content_sha256":"59c61eeab86c9b212fafcf87c961976b88d7ba9258edd0c5d5dab25261277628","schema_version":"1.0","event_id":"sha256:59c61eeab86c9b212fafcf87c961976b88d7ba9258edd0c5d5dab25261277628"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/3KNNUGDBGCNSWTDCCP24CRKL7T/bundle.json","state_url":"https://pith.science/pith/3KNNUGDBGCNSWTDCCP24CRKL7T/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/3KNNUGDBGCNSWTDCCP24CRKL7T/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-01T12:32:52Z","links":{"resolver":"https://pith.science/pith/3KNNUGDBGCNSWTDCCP24CRKL7T","bundle":"https://pith.science/pith/3KNNUGDBGCNSWTDCCP24CRKL7T/bundle.json","state":"https://pith.science/pith/3KNNUGDBGCNSWTDCCP24CRKL7T/state.json","well_known_bundle":"https://pith.science/.well-known/pith/3KNNUGDBGCNSWTDCCP24CRKL7T/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:3KNNUGDBGCNSWTDCCP24CRKL7T","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"c5cb8d16e89a8ce80d9a2e6eee9cebb25607727e396afd3a5dab75463622600f","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-30T09:02:07Z","title_canon_sha256":"16cf7cbf2ef0e6fad742e8443fca77dd188d5c1590ce5e507464931e85044204"},"schema_version":"1.0","source":{"id":"2505.24369","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.24369","created_at":"2026-07-05T11:12:47Z"},{"alias_kind":"arxiv_version","alias_value":"2505.24369v1","created_at":"2026-07-05T11:12:47Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.24369","created_at":"2026-07-05T11:12:47Z"},{"alias_kind":"pith_short_12","alias_value":"3KNNUGDBGCNS","created_at":"2026-07-05T11:12:47Z"},{"alias_kind":"pith_short_16","alias_value":"3KNNUGDBGCNSWTDC","created_at":"2026-07-05T11:12:47Z"},{"alias_kind":"pith_short_8","alias_value":"3KNNUGDB","created_at":"2026-07-05T11:12:47Z"}],"graph_snapshots":[{"event_id":"sha256:59c61eeab86c9b212fafcf87c961976b88d7ba9258edd0c5d5dab25261277628","target":"graph","created_at":"2026-07-05T11:12:47Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2505.24369/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Modern language models often rely on Reinforcement Learning from Human Feedback (RLHF) to encourage safe behaviors. However, they remain vulnerable to adversarial attacks due to three key limitations: (1) the inefficiency and high cost of human annotation, (2) the vast diversity of potential adversarial attacks, and (3) the risk of feedback bias and reward hacking. To address these challenges, we introduce Adversarial Preference Learning (APL), an iterative adversarial training method incorporating three key innovations. First, a direct harmfulness metric based on the model's intrinsic prefere","authors_text":"Bo Tang, Chaochao Lu, Chao Yang, Chen Chen, Chenyang Xi, Feiyu Xiong, Jie Hu, Jingfeng Zhang, Junyi Zhu, Keming Mao, Mingchuan Yang, Pengyu Wang, Wenqiang Wei, Yijun Niu, Yuanfu Wang, Zhiyu Li","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-30T09:02:07Z","title":"Adversarial Preference Learning for Robust LLM Alignment"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.24369","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:2883c999d3c26bfb2a49aa690def49a0d65df056a17e826f656a5f0c952e6d67","target":"record","created_at":"2026-07-05T11:12:47Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"c5cb8d16e89a8ce80d9a2e6eee9cebb25607727e396afd3a5dab75463622600f","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-30T09:02:07Z","title_canon_sha256":"16cf7cbf2ef0e6fad742e8443fca77dd188d5c1590ce5e507464931e85044204"},"schema_version":"1.0","source":{"id":"2505.24369","kind":"arxiv","version":1}},"canonical_sha256":"da9ada1861309b2b4c6213f5c1454bfcc3a8d917c0e1fadbdef54b2be9669f05","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"da9ada1861309b2b4c6213f5c1454bfcc3a8d917c0e1fadbdef54b2be9669f05","first_computed_at":"2026-07-05T11:12:47.854764Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:12:47.854764Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"Eut1iPl4+tpwFo+/9Iv9yf9jv2KpbzhkDixR/OBwm512wKiPBz6rj5IBanRF9LpgOK5gH3CK57Yz4rJ51IdTAA==","signature_status":"signed_v1","signed_at":"2026-07-05T11:12:47.855409Z","signed_message":"canonical_sha256_bytes"},"source_id":"2505.24369","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:2883c999d3c26bfb2a49aa690def49a0d65df056a17e826f656a5f0c952e6d67","sha256:59c61eeab86c9b212fafcf87c961976b88d7ba9258edd0c5d5dab25261277628"],"state_sha256":"eb2ec3909ef5b18b74a28f1f1f1a116d776e9990c305a5195929ae0d269a097d"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"9KK3xGM876OHYumpYQBKM66QS8lOCfFe58HiDRqtaf39VijSRXxukZz6c5+7EDHvXanHoDTdUXFaFd5OvK53Cw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-01T12:32:52.066750Z","bundle_sha256":"f52638f3ef5430cb9dcfe57830d9af998d1534d7e77f7083a22f7b3c4fb399ce"}}