{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:DGCR3OOMTT4GZOOWX2Z4FHJM3P","short_pith_number":"pith:DGCR3OOM","canonical_record":{"source":{"id":"2507.07375","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-10T01:56:56Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"cc99bfa41b66fc59e023137ead4ceb342d49071b971a9e7162000cdfb90ba668","abstract_canon_sha256":"051b71eadda69defc4e5a0dc52cd8b50c90542393702ba0ae3e65908414be300"},"schema_version":"1.0"},"canonical_sha256":"19851db9cc9cf86cb9d6beb3c29d2cdbeca8737a105d3d17e4bc8935191a8312","source":{"kind":"arxiv","id":"2507.07375","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2507.07375","created_at":"2026-07-05T11:34:56Z"},{"alias_kind":"arxiv_version","alias_value":"2507.07375v1","created_at":"2026-07-05T11:34:56Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.07375","created_at":"2026-07-05T11:34:56Z"},{"alias_kind":"pith_short_12","alias_value":"DGCR3OOMTT4G","created_at":"2026-07-05T11:34:56Z"},{"alias_kind":"pith_short_16","alias_value":"DGCR3OOMTT4GZOOW","created_at":"2026-07-05T11:34:56Z"},{"alias_kind":"pith_short_8","alias_value":"DGCR3OOM","created_at":"2026-07-05T11:34:56Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:DGCR3OOMTT4GZOOWX2Z4FHJM3P","target":"record","payload":{"canonical_record":{"source":{"id":"2507.07375","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-10T01:56:56Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"cc99bfa41b66fc59e023137ead4ceb342d49071b971a9e7162000cdfb90ba668","abstract_canon_sha256":"051b71eadda69defc4e5a0dc52cd8b50c90542393702ba0ae3e65908414be300"},"schema_version":"1.0"},"canonical_sha256":"19851db9cc9cf86cb9d6beb3c29d2cdbeca8737a105d3d17e4bc8935191a8312","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:34:56.884787Z","signature_b64":"FGnY9a/46E3ydFNs+lD4xDqum6garF93VkeDGJSrA9p4E1a6NdZ8PyA9Ixblg+jHRd0lcFB75m8vlsUBxHoxDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"19851db9cc9cf86cb9d6beb3c29d2cdbeca8737a105d3d17e4bc8935191a8312","last_reissued_at":"2026-07-05T11:34:56.884336Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:34:56.884336Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2507.07375","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:34:56Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"umbnOso+re6P8jcjeyp04g5Mn8GLZnRlcNpZ1AdTXiAvgSEThSx+I9a4q0Ztq03NW3HrEIaRQKbldtfqF7LXDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T07:54:17.586451Z"},"content_sha256":"31801a3ab87a983e83271fcf17006a1d63d1661956395fa4c6e889b75d54b324","schema_version":"1.0","event_id":"sha256:31801a3ab87a983e83271fcf17006a1d63d1661956395fa4c6e889b75d54b324"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:DGCR3OOMTT4GZOOWX2Z4FHJM3P","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Bradley-Terry and Multi-Objective Reward Modeling Are Complementary","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Chen Luo, Fali Wang, Hui Liu, Jingying Zeng, Minhua Lin, Qi He, Ramraj Chandradevan, Suhang Wang, Xianfeng Tang, Xiaomin Li, Zhen Li, Zhenwei Dai, Zhiwei Zhang","submitted_at":"2025-07-10T01:56:56Z","abstract_excerpt":"Reward models trained on human preference data have demonstrated strong effectiveness in aligning Large Language Models (LLMs) with human intent under the framework of Reinforcement Learning from Human Feedback (RLHF). However, RLHF remains vulnerable to reward hacking, where the policy exploits imperfections in the reward function rather than genuinely learning the intended behavior. Although significant efforts have been made to mitigate reward hacking, they predominantly focus on and evaluate in-distribution scenarios, where the training and testing data for the reward model share the same "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.07375","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.07375/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:34:56Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Zg04lXdUhV3n1sfa941l88qPZ/oxw+x3EnUdESlPBKoz1sazla6LhjLIrnMKwW+h7Gi78ZBwhllwjlpVcTI/Dw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T07:54:17.587018Z"},"content_sha256":"6df5b1e906cd3136b276c31b2eb13cc724fd5be10a4a5b75ce7010ed79a1c48e","schema_version":"1.0","event_id":"sha256:6df5b1e906cd3136b276c31b2eb13cc724fd5be10a4a5b75ce7010ed79a1c48e"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/DGCR3OOMTT4GZOOWX2Z4FHJM3P/bundle.json","state_url":"https://pith.science/pith/DGCR3OOMTT4GZOOWX2Z4FHJM3P/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/DGCR3OOMTT4GZOOWX2Z4FHJM3P/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T07:54:17Z","links":{"resolver":"https://pith.science/pith/DGCR3OOMTT4GZOOWX2Z4FHJM3P","bundle":"https://pith.science/pith/DGCR3OOMTT4GZOOWX2Z4FHJM3P/bundle.json","state":"https://pith.science/pith/DGCR3OOMTT4GZOOWX2Z4FHJM3P/state.json","well_known_bundle":"https://pith.science/.well-known/pith/DGCR3OOMTT4GZOOWX2Z4FHJM3P/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:DGCR3OOMTT4GZOOWX2Z4FHJM3P","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"051b71eadda69defc4e5a0dc52cd8b50c90542393702ba0ae3e65908414be300","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-10T01:56:56Z","title_canon_sha256":"cc99bfa41b66fc59e023137ead4ceb342d49071b971a9e7162000cdfb90ba668"},"schema_version":"1.0","source":{"id":"2507.07375","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2507.07375","created_at":"2026-07-05T11:34:56Z"},{"alias_kind":"arxiv_version","alias_value":"2507.07375v1","created_at":"2026-07-05T11:34:56Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.07375","created_at":"2026-07-05T11:34:56Z"},{"alias_kind":"pith_short_12","alias_value":"DGCR3OOMTT4G","created_at":"2026-07-05T11:34:56Z"},{"alias_kind":"pith_short_16","alias_value":"DGCR3OOMTT4GZOOW","created_at":"2026-07-05T11:34:56Z"},{"alias_kind":"pith_short_8","alias_value":"DGCR3OOM","created_at":"2026-07-05T11:34:56Z"}],"graph_snapshots":[{"event_id":"sha256:6df5b1e906cd3136b276c31b2eb13cc724fd5be10a4a5b75ce7010ed79a1c48e","target":"graph","created_at":"2026-07-05T11:34:56Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2507.07375/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reward models trained on human preference data have demonstrated strong effectiveness in aligning Large Language Models (LLMs) with human intent under the framework of Reinforcement Learning from Human Feedback (RLHF). However, RLHF remains vulnerable to reward hacking, where the policy exploits imperfections in the reward function rather than genuinely learning the intended behavior. Although significant efforts have been made to mitigate reward hacking, they predominantly focus on and evaluate in-distribution scenarios, where the training and testing data for the reward model share the same ","authors_text":"Chen Luo, Fali Wang, Hui Liu, Jingying Zeng, Minhua Lin, Qi He, Ramraj Chandradevan, Suhang Wang, Xianfeng Tang, Xiaomin Li, Zhen Li, Zhenwei Dai, Zhiwei Zhang","cross_cats":["cs.CL"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-10T01:56:56Z","title":"Bradley-Terry and Multi-Objective Reward Modeling Are Complementary"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.07375","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:31801a3ab87a983e83271fcf17006a1d63d1661956395fa4c6e889b75d54b324","target":"record","created_at":"2026-07-05T11:34:56Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"051b71eadda69defc4e5a0dc52cd8b50c90542393702ba0ae3e65908414be300","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-10T01:56:56Z","title_canon_sha256":"cc99bfa41b66fc59e023137ead4ceb342d49071b971a9e7162000cdfb90ba668"},"schema_version":"1.0","source":{"id":"2507.07375","kind":"arxiv","version":1}},"canonical_sha256":"19851db9cc9cf86cb9d6beb3c29d2cdbeca8737a105d3d17e4bc8935191a8312","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"19851db9cc9cf86cb9d6beb3c29d2cdbeca8737a105d3d17e4bc8935191a8312","first_computed_at":"2026-07-05T11:34:56.884336Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:34:56.884336Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"FGnY9a/46E3ydFNs+lD4xDqum6garF93VkeDGJSrA9p4E1a6NdZ8PyA9Ixblg+jHRd0lcFB75m8vlsUBxHoxDQ==","signature_status":"signed_v1","signed_at":"2026-07-05T11:34:56.884787Z","signed_message":"canonical_sha256_bytes"},"source_id":"2507.07375","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:31801a3ab87a983e83271fcf17006a1d63d1661956395fa4c6e889b75d54b324","sha256:6df5b1e906cd3136b276c31b2eb13cc724fd5be10a4a5b75ce7010ed79a1c48e"],"state_sha256":"81451c399e5068c759957990322174c6a46d316d73572ca1bfe919fc5d99064b"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"SEw44z6USkDi5X9LU8m84A8l9mBNxaVuM2dThRYNGcRyPrNp+KuUvEgosgHC/7gvsdgA3gG++PH1AKhuPiEXAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T07:54:17.591820Z","bundle_sha256":"2c74a796269de6a640d839b72200a01602c425636755cd1dae3dffce1cebb6ca"}}