{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:ZZZ7EDRI6HK4NBM55UX7GSTZP6","short_pith_number":"pith:ZZZ7EDRI","canonical_record":{"source":{"id":"2310.00212","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-09-30T01:23:22Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"7777b10d081fba98abe01c1ed30b5399a205473c3d4e869c384ccd15d78a0d88","abstract_canon_sha256":"d7392f30dfd47c416f064e691905c0a154f994bd7ad21d7918ff599a07525de9"},"schema_version":"1.0"},"canonical_sha256":"ce73f20e28f1d5c6859ded2ff34a797f85b0b4e73168a4546f7aa2f223058a55","source":{"kind":"arxiv","id":"2310.00212","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.00212","created_at":"2026-07-05T06:59:00Z"},{"alias_kind":"arxiv_version","alias_value":"2310.00212v3","created_at":"2026-07-05T06:59:00Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.00212","created_at":"2026-07-05T06:59:00Z"},{"alias_kind":"pith_short_12","alias_value":"ZZZ7EDRI6HK4","created_at":"2026-07-05T06:59:00Z"},{"alias_kind":"pith_short_16","alias_value":"ZZZ7EDRI6HK4NBM5","created_at":"2026-07-05T06:59:00Z"},{"alias_kind":"pith_short_8","alias_value":"ZZZ7EDRI","created_at":"2026-07-05T06:59:00Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:ZZZ7EDRI6HK4NBM55UX7GSTZP6","target":"record","payload":{"canonical_record":{"source":{"id":"2310.00212","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-09-30T01:23:22Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"7777b10d081fba98abe01c1ed30b5399a205473c3d4e869c384ccd15d78a0d88","abstract_canon_sha256":"d7392f30dfd47c416f064e691905c0a154f994bd7ad21d7918ff599a07525de9"},"schema_version":"1.0"},"canonical_sha256":"ce73f20e28f1d5c6859ded2ff34a797f85b0b4e73168a4546f7aa2f223058a55","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:59:00.088200Z","signature_b64":"tbTmZ77+QINvvZdTGmic0bcTtWOVREysGqBu7UgdiSsShtuD9U8mu25g6wA7UIQuDn87sHXMziAfbFud3ySPAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ce73f20e28f1d5c6859ded2ff34a797f85b0b4e73168a4546f7aa2f223058a55","last_reissued_at":"2026-07-05T06:59:00.087607Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:59:00.087607Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2310.00212","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:59:00Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"F5m6GtgaGPkDQOmFde9xehEc9tQoU8UP2cVpPZrPmuwjJvkziIIg3vksOqmJ1gI3wv3DYv+8epelsS4i1dwXAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T11:11:59.171459Z"},"content_sha256":"26b2a2741ae84adf0c0da8f467612771445c076e5374aac6222be5fa0697eee4","schema_version":"1.0","event_id":"sha256:26b2a2741ae84adf0c0da8f467612771445c076e5374aac6222be5fa0697eee4"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:ZZZ7EDRI6HK4NBM55UX7GSTZP6","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Pairwise Proximal Policy Optimization: Harnessing Relative Feedback for LLM Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Banghua Zhu, Jiantao Jiao, Kannan Ramchandran, Ruoyu Zhang, Tianhao Wu, Zhaojin Wen","submitted_at":"2023-09-30T01:23:22Z","abstract_excerpt":"Large Language Models (LLMs) can acquire extensive world knowledge through pre-training on large corpora. However, due to exposure to low-quality data, LLMs may exhibit harmful behavior without aligning with human values. The dominant approach for steering LLMs towards beneficial behavior involves Reinforcement Learning with Human Feedback (RLHF), with Proximal Policy Optimization (PPO) serving as the default RL optimizer. Despite its effectiveness, PPO has limitations when optimizing rewards trained from comparison-based loss. Primarily, PPO is not invariant to equivalent reward functions con"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.00212","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.00212/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:59:00Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"NLB2w4fnoJP0cbqhkr9YDw9uJvRGbfWh7YziMZimydHNqgZa30r/V43svB4Z1lH/mHhEd8xAqIWZIccpIunzAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T11:11:59.172026Z"},"content_sha256":"5a57a20c03ff94071656c9b4392b48caaeda5aec092be91094c01f047b470429","schema_version":"1.0","event_id":"sha256:5a57a20c03ff94071656c9b4392b48caaeda5aec092be91094c01f047b470429"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/ZZZ7EDRI6HK4NBM55UX7GSTZP6/bundle.json","state_url":"https://pith.science/pith/ZZZ7EDRI6HK4NBM55UX7GSTZP6/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/ZZZ7EDRI6HK4NBM55UX7GSTZP6/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T11:11:59Z","links":{"resolver":"https://pith.science/pith/ZZZ7EDRI6HK4NBM55UX7GSTZP6","bundle":"https://pith.science/pith/ZZZ7EDRI6HK4NBM55UX7GSTZP6/bundle.json","state":"https://pith.science/pith/ZZZ7EDRI6HK4NBM55UX7GSTZP6/state.json","well_known_bundle":"https://pith.science/.well-known/pith/ZZZ7EDRI6HK4NBM55UX7GSTZP6/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:ZZZ7EDRI6HK4NBM55UX7GSTZP6","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"d7392f30dfd47c416f064e691905c0a154f994bd7ad21d7918ff599a07525de9","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-09-30T01:23:22Z","title_canon_sha256":"7777b10d081fba98abe01c1ed30b5399a205473c3d4e869c384ccd15d78a0d88"},"schema_version":"1.0","source":{"id":"2310.00212","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.00212","created_at":"2026-07-05T06:59:00Z"},{"alias_kind":"arxiv_version","alias_value":"2310.00212v3","created_at":"2026-07-05T06:59:00Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.00212","created_at":"2026-07-05T06:59:00Z"},{"alias_kind":"pith_short_12","alias_value":"ZZZ7EDRI6HK4","created_at":"2026-07-05T06:59:00Z"},{"alias_kind":"pith_short_16","alias_value":"ZZZ7EDRI6HK4NBM5","created_at":"2026-07-05T06:59:00Z"},{"alias_kind":"pith_short_8","alias_value":"ZZZ7EDRI","created_at":"2026-07-05T06:59:00Z"}],"graph_snapshots":[{"event_id":"sha256:5a57a20c03ff94071656c9b4392b48caaeda5aec092be91094c01f047b470429","target":"graph","created_at":"2026-07-05T06:59:00Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2310.00212/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Large Language Models (LLMs) can acquire extensive world knowledge through pre-training on large corpora. However, due to exposure to low-quality data, LLMs may exhibit harmful behavior without aligning with human values. The dominant approach for steering LLMs towards beneficial behavior involves Reinforcement Learning with Human Feedback (RLHF), with Proximal Policy Optimization (PPO) serving as the default RL optimizer. Despite its effectiveness, PPO has limitations when optimizing rewards trained from comparison-based loss. Primarily, PPO is not invariant to equivalent reward functions con","authors_text":"Banghua Zhu, Jiantao Jiao, Kannan Ramchandran, Ruoyu Zhang, Tianhao Wu, Zhaojin Wen","cross_cats":["cs.AI","cs.CL"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-09-30T01:23:22Z","title":"Pairwise Proximal Policy Optimization: Harnessing Relative Feedback for LLM Alignment"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.00212","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:26b2a2741ae84adf0c0da8f467612771445c076e5374aac6222be5fa0697eee4","target":"record","created_at":"2026-07-05T06:59:00Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"d7392f30dfd47c416f064e691905c0a154f994bd7ad21d7918ff599a07525de9","cross_cats_sorted":["cs.AI","cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-09-30T01:23:22Z","title_canon_sha256":"7777b10d081fba98abe01c1ed30b5399a205473c3d4e869c384ccd15d78a0d88"},"schema_version":"1.0","source":{"id":"2310.00212","kind":"arxiv","version":3}},"canonical_sha256":"ce73f20e28f1d5c6859ded2ff34a797f85b0b4e73168a4546f7aa2f223058a55","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"ce73f20e28f1d5c6859ded2ff34a797f85b0b4e73168a4546f7aa2f223058a55","first_computed_at":"2026-07-05T06:59:00.087607Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T06:59:00.087607Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"tbTmZ77+QINvvZdTGmic0bcTtWOVREysGqBu7UgdiSsShtuD9U8mu25g6wA7UIQuDn87sHXMziAfbFud3ySPAA==","signature_status":"signed_v1","signed_at":"2026-07-05T06:59:00.088200Z","signed_message":"canonical_sha256_bytes"},"source_id":"2310.00212","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:26b2a2741ae84adf0c0da8f467612771445c076e5374aac6222be5fa0697eee4","sha256:5a57a20c03ff94071656c9b4392b48caaeda5aec092be91094c01f047b470429"],"state_sha256":"9efbd5c6637f8e87aab23b49fa65182b0760c4bb9515e81b3405bb2b2dcc0b6b"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"EIaD72sNwr9vn+qi5HRUlPeYfTX6bI5ZefvGaGDgxfGE4Xb96zZv2xzjo2M6a7qn2ez4upOd6ku+JT1ZzANyAQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T11:11:59.176047Z","bundle_sha256":"ebd7014bf3c5426bbc89c5e9d4afd68e2263226d70b222e59312ac61ad339dc2"}}