{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:GDLG3EEB53MBY54HP2SMGP5BGF","short_pith_number":"pith:GDLG3EEB","canonical_record":{"source":{"id":"2504.04950","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-04-07T11:34:48Z","cross_cats_sorted":[],"title_canon_sha256":"5a5bc99217c40aeee8e5a435dcf9b48b1a69c3c4f8feae19e1ee3a55635e1ae5","abstract_canon_sha256":"13fd7271c1f5025bb05efa8859595c3b9fbcfb9b231e21c31a172222e0c6353a"},"schema_version":"1.0"},"canonical_sha256":"30d66d9081eed81c77877ea4c33fa131402cb551c669d091b7ce401c7930b515","source":{"kind":"arxiv","id":"2504.04950","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2504.04950","created_at":"2026-07-05T10:45:31Z"},{"alias_kind":"arxiv_version","alias_value":"2504.04950v1","created_at":"2026-07-05T10:45:31Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.04950","created_at":"2026-07-05T10:45:31Z"},{"alias_kind":"pith_short_12","alias_value":"GDLG3EEB53MB","created_at":"2026-07-05T10:45:31Z"},{"alias_kind":"pith_short_16","alias_value":"GDLG3EEB53MBY54H","created_at":"2026-07-05T10:45:31Z"},{"alias_kind":"pith_short_8","alias_value":"GDLG3EEB","created_at":"2026-07-05T10:45:31Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:GDLG3EEB53MBY54HP2SMGP5BGF","target":"record","payload":{"canonical_record":{"source":{"id":"2504.04950","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-04-07T11:34:48Z","cross_cats_sorted":[],"title_canon_sha256":"5a5bc99217c40aeee8e5a435dcf9b48b1a69c3c4f8feae19e1ee3a55635e1ae5","abstract_canon_sha256":"13fd7271c1f5025bb05efa8859595c3b9fbcfb9b231e21c31a172222e0c6353a"},"schema_version":"1.0"},"canonical_sha256":"30d66d9081eed81c77877ea4c33fa131402cb551c669d091b7ce401c7930b515","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:45:31.916641Z","signature_b64":"lpBFnXdAXT2vimfka1xVze8OLrG/gFwwItyFwPxxuUQeCWWgdORDTqi6ytifmum4/89sPXK90CdjiIZztK4GBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"30d66d9081eed81c77877ea4c33fa131402cb551c669d091b7ce401c7930b515","last_reissued_at":"2026-07-05T10:45:31.916169Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:45:31.916169Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2504.04950","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:45:31Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"UFrqg8aw7Kiz8+wm+M9n51mytWBXsoR9Kaxh0I5SF726csP2j/thDahbubO5RG4ruA6RSdkPztApuN3T38frAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T15:43:06.249864Z"},"content_sha256":"a7e6c7e537bafc85db8e9f0a4635642b08630bb2db5ccfee08603bc1f731fc1f","schema_version":"1.0","event_id":"sha256:a7e6c7e537bafc85db8e9f0a4635642b08630bb2db5ccfee08603bc1f731fc1f"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:GDLG3EEB53MBY54HP2SMGP5BGF","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"A Unified Pairwise Framework for RLHF: Bridging Generative Reward Modeling and Policy Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chao Xin, Lin Yan, Wenyuan Xu, Xiaochen Zuo, Yonghui Wu, Yu Yue","submitted_at":"2025-04-07T11:34:48Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) has emerged as a important paradigm for aligning large language models (LLMs) with human preferences during post-training. This framework typically involves two stages: first, training a reward model on human preference data, followed by optimizing the language model using reinforcement learning algorithms. However, current RLHF approaches may constrained by two limitations. First, existing RLHF frameworks often rely on Bradley-Terry models to assign scalar rewards based on pairwise comparisons of individual responses. However, this approach im"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.04950","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.04950/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:45:31Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"HJg5yk3QYqzsochbc02QYn7PQC9Hh2W7mtD73No+0LK7uTKXkj3NONNa5bsCyFhCytEZar6jC0qE7KdEggl1Bg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T15:43:06.250301Z"},"content_sha256":"42a9685441f4d2bd9f57dc1bb742d902fc4214c0e4bd971007f916f9ec39e390","schema_version":"1.0","event_id":"sha256:42a9685441f4d2bd9f57dc1bb742d902fc4214c0e4bd971007f916f9ec39e390"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/GDLG3EEB53MBY54HP2SMGP5BGF/bundle.json","state_url":"https://pith.science/pith/GDLG3EEB53MBY54HP2SMGP5BGF/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/GDLG3EEB53MBY54HP2SMGP5BGF/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T15:43:06Z","links":{"resolver":"https://pith.science/pith/GDLG3EEB53MBY54HP2SMGP5BGF","bundle":"https://pith.science/pith/GDLG3EEB53MBY54HP2SMGP5BGF/bundle.json","state":"https://pith.science/pith/GDLG3EEB53MBY54HP2SMGP5BGF/state.json","well_known_bundle":"https://pith.science/.well-known/pith/GDLG3EEB53MBY54HP2SMGP5BGF/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:GDLG3EEB53MBY54HP2SMGP5BGF","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"13fd7271c1f5025bb05efa8859595c3b9fbcfb9b231e21c31a172222e0c6353a","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-04-07T11:34:48Z","title_canon_sha256":"5a5bc99217c40aeee8e5a435dcf9b48b1a69c3c4f8feae19e1ee3a55635e1ae5"},"schema_version":"1.0","source":{"id":"2504.04950","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2504.04950","created_at":"2026-07-05T10:45:31Z"},{"alias_kind":"arxiv_version","alias_value":"2504.04950v1","created_at":"2026-07-05T10:45:31Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.04950","created_at":"2026-07-05T10:45:31Z"},{"alias_kind":"pith_short_12","alias_value":"GDLG3EEB53MB","created_at":"2026-07-05T10:45:31Z"},{"alias_kind":"pith_short_16","alias_value":"GDLG3EEB53MBY54H","created_at":"2026-07-05T10:45:31Z"},{"alias_kind":"pith_short_8","alias_value":"GDLG3EEB","created_at":"2026-07-05T10:45:31Z"}],"graph_snapshots":[{"event_id":"sha256:42a9685441f4d2bd9f57dc1bb742d902fc4214c0e4bd971007f916f9ec39e390","target":"graph","created_at":"2026-07-05T10:45:31Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2504.04950/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) has emerged as a important paradigm for aligning large language models (LLMs) with human preferences during post-training. This framework typically involves two stages: first, training a reward model on human preference data, followed by optimizing the language model using reinforcement learning algorithms. However, current RLHF approaches may constrained by two limitations. First, existing RLHF frameworks often rely on Bradley-Terry models to assign scalar rewards based on pairwise comparisons of individual responses. However, this approach im","authors_text":"Chao Xin, Lin Yan, Wenyuan Xu, Xiaochen Zuo, Yonghui Wu, Yu Yue","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-04-07T11:34:48Z","title":"A Unified Pairwise Framework for RLHF: Bridging Generative Reward Modeling and Policy Optimization"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.04950","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:a7e6c7e537bafc85db8e9f0a4635642b08630bb2db5ccfee08603bc1f731fc1f","target":"record","created_at":"2026-07-05T10:45:31Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"13fd7271c1f5025bb05efa8859595c3b9fbcfb9b231e21c31a172222e0c6353a","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-04-07T11:34:48Z","title_canon_sha256":"5a5bc99217c40aeee8e5a435dcf9b48b1a69c3c4f8feae19e1ee3a55635e1ae5"},"schema_version":"1.0","source":{"id":"2504.04950","kind":"arxiv","version":1}},"canonical_sha256":"30d66d9081eed81c77877ea4c33fa131402cb551c669d091b7ce401c7930b515","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"30d66d9081eed81c77877ea4c33fa131402cb551c669d091b7ce401c7930b515","first_computed_at":"2026-07-05T10:45:31.916169Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:45:31.916169Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"lpBFnXdAXT2vimfka1xVze8OLrG/gFwwItyFwPxxuUQeCWWgdORDTqi6ytifmum4/89sPXK90CdjiIZztK4GBw==","signature_status":"signed_v1","signed_at":"2026-07-05T10:45:31.916641Z","signed_message":"canonical_sha256_bytes"},"source_id":"2504.04950","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:a7e6c7e537bafc85db8e9f0a4635642b08630bb2db5ccfee08603bc1f731fc1f","sha256:42a9685441f4d2bd9f57dc1bb742d902fc4214c0e4bd971007f916f9ec39e390"],"state_sha256":"5ecdb065c8a665def8269177811e39d52331a5f94553f6bbc0318239fed77af6"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"QhaI8v+rTblL7l81/41N88JGTi1W0d1pLrC9RcW7xI0pKHbEZhMXXVCGZ/zDg0ZVj6XR3ZbnrtDvdWVItWZBCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T15:43:06.253083Z","bundle_sha256":"d78c1ee9487e8cb1a7e92c831f8eb6fa5eda587ac97bd982e5ceddd0e6102091"}}