{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:2NZEOYYHNW56QZH6ASGBYRYHUC","short_pith_number":"pith:2NZEOYYH","canonical_record":{"source":{"id":"2409.13156","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-20T01:46:07Z","cross_cats_sorted":[],"title_canon_sha256":"ae68b1fe4ccfcb9d210c520dd01b2a65ef27b945538f11c44135b54efbfcf063","abstract_canon_sha256":"50761d2109bcefefca08f3f152f8c6b0689ec38af078e9b1dd2b36ab879170d7"},"schema_version":"1.0"},"canonical_sha256":"d3724763076dbbe864fe048c1c4707a09c4ec3d5a6e0bb29b0d2ce20be38cb60","source":{"kind":"arxiv","id":"2409.13156","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2409.13156","created_at":"2026-07-05T10:21:04Z"},{"alias_kind":"arxiv_version","alias_value":"2409.13156v2","created_at":"2026-07-05T10:21:04Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.13156","created_at":"2026-07-05T10:21:04Z"},{"alias_kind":"pith_short_12","alias_value":"2NZEOYYHNW56","created_at":"2026-07-05T10:21:04Z"},{"alias_kind":"pith_short_16","alias_value":"2NZEOYYHNW56QZH6","created_at":"2026-07-05T10:21:04Z"},{"alias_kind":"pith_short_8","alias_value":"2NZEOYYH","created_at":"2026-07-05T10:21:04Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:2NZEOYYHNW56QZH6ASGBYRYHUC","target":"record","payload":{"canonical_record":{"source":{"id":"2409.13156","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-20T01:46:07Z","cross_cats_sorted":[],"title_canon_sha256":"ae68b1fe4ccfcb9d210c520dd01b2a65ef27b945538f11c44135b54efbfcf063","abstract_canon_sha256":"50761d2109bcefefca08f3f152f8c6b0689ec38af078e9b1dd2b36ab879170d7"},"schema_version":"1.0"},"canonical_sha256":"d3724763076dbbe864fe048c1c4707a09c4ec3d5a6e0bb29b0d2ce20be38cb60","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:21:04.427793Z","signature_b64":"4heL3jxt244pSYfUoYcWQplJ24UpE5osAyOtRkTymKyHfiUvbCaAP4BIMAovtlENIWKuEL5IO961mEyIzs5DCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d3724763076dbbe864fe048c1c4707a09c4ec3d5a6e0bb29b0d2ce20be38cb60","last_reissued_at":"2026-07-05T10:21:04.427259Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:21:04.427259Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2409.13156","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:21:04Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"FwPTc03ilAZ5Mw+/qLmrYv3RP8vWaLA5c5jecYx/CtyxMkteuF/N2UBW0L8Fnny38Ljp32QL42hjRadOkb+8Aw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T06:27:58.316146Z"},"content_sha256":"9925b0c93c95d605e31392107ce1eac7b407ce57aceba0c410fbc4c288298d86","schema_version":"1.0","event_id":"sha256:9925b0c93c95d605e31392107ce1eac7b407ce57aceba0c410fbc4c288298d86"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:2NZEOYYHNW56QZH6ASGBYRYHUC","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"RRM: Robust Reward Model Training Mitigates Reward Hacking","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Abe Ittycheriah, Anastasiia Makarova, Aviral Kumar, Bilal Piot, Daniel Sohn, Jeremiah Liu, Jiaming Shen, Jie Ren, Junru Wu, Lichang Chen, Mohammad Saleh, Rishabh Joshi, Tianhe Yu, Tianqi Liu, Wei Xiong, Yang Gao, Yuan Liu, Zhen Qin","submitted_at":"2024-09-20T01:46:07Z","abstract_excerpt":"Reward models (RMs) play a pivotal role in aligning large language models (LLMs) with human preferences. However, traditional RM training, which relies on response pairs tied to specific prompts, struggles to disentangle prompt-driven preferences from prompt-independent artifacts, such as response length and format. In this work, we expose a fundamental limitation of current RM training methods, where RMs fail to effectively distinguish between contextual signals and irrelevant artifacts when determining preferences. To address this, we introduce a causal framework that learns preferences inde"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.13156","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.13156/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:21:04Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"fus7k4+1YlwYKgnQVqHLWUYMlIvb/7NS3Wx6Nui/P2Na0cGOQN26e2mKZBgQ9gTRYHcmU4wVNhNm8OmjoDCyAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T06:27:58.316704Z"},"content_sha256":"f72a4239e49c29a9299c75ad02f0bb07c23b0ce752ca49aef9aa4b704c8a9613","schema_version":"1.0","event_id":"sha256:f72a4239e49c29a9299c75ad02f0bb07c23b0ce752ca49aef9aa4b704c8a9613"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/2NZEOYYHNW56QZH6ASGBYRYHUC/bundle.json","state_url":"https://pith.science/pith/2NZEOYYHNW56QZH6ASGBYRYHUC/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/2NZEOYYHNW56QZH6ASGBYRYHUC/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-10T06:27:58Z","links":{"resolver":"https://pith.science/pith/2NZEOYYHNW56QZH6ASGBYRYHUC","bundle":"https://pith.science/pith/2NZEOYYHNW56QZH6ASGBYRYHUC/bundle.json","state":"https://pith.science/pith/2NZEOYYHNW56QZH6ASGBYRYHUC/state.json","well_known_bundle":"https://pith.science/.well-known/pith/2NZEOYYHNW56QZH6ASGBYRYHUC/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:2NZEOYYHNW56QZH6ASGBYRYHUC","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"50761d2109bcefefca08f3f152f8c6b0689ec38af078e9b1dd2b36ab879170d7","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-20T01:46:07Z","title_canon_sha256":"ae68b1fe4ccfcb9d210c520dd01b2a65ef27b945538f11c44135b54efbfcf063"},"schema_version":"1.0","source":{"id":"2409.13156","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2409.13156","created_at":"2026-07-05T10:21:04Z"},{"alias_kind":"arxiv_version","alias_value":"2409.13156v2","created_at":"2026-07-05T10:21:04Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.13156","created_at":"2026-07-05T10:21:04Z"},{"alias_kind":"pith_short_12","alias_value":"2NZEOYYHNW56","created_at":"2026-07-05T10:21:04Z"},{"alias_kind":"pith_short_16","alias_value":"2NZEOYYHNW56QZH6","created_at":"2026-07-05T10:21:04Z"},{"alias_kind":"pith_short_8","alias_value":"2NZEOYYH","created_at":"2026-07-05T10:21:04Z"}],"graph_snapshots":[{"event_id":"sha256:f72a4239e49c29a9299c75ad02f0bb07c23b0ce752ca49aef9aa4b704c8a9613","target":"graph","created_at":"2026-07-05T10:21:04Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2409.13156/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reward models (RMs) play a pivotal role in aligning large language models (LLMs) with human preferences. However, traditional RM training, which relies on response pairs tied to specific prompts, struggles to disentangle prompt-driven preferences from prompt-independent artifacts, such as response length and format. In this work, we expose a fundamental limitation of current RM training methods, where RMs fail to effectively distinguish between contextual signals and irrelevant artifacts when determining preferences. To address this, we introduce a causal framework that learns preferences inde","authors_text":"Abe Ittycheriah, Anastasiia Makarova, Aviral Kumar, Bilal Piot, Daniel Sohn, Jeremiah Liu, Jiaming Shen, Jie Ren, Junru Wu, Lichang Chen, Mohammad Saleh, Rishabh Joshi, Tianhe Yu, Tianqi Liu, Wei Xiong, Yang Gao, Yuan Liu, Zhen Qin","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-20T01:46:07Z","title":"RRM: Robust Reward Model Training Mitigates Reward Hacking"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.13156","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:9925b0c93c95d605e31392107ce1eac7b407ce57aceba0c410fbc4c288298d86","target":"record","created_at":"2026-07-05T10:21:04Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"50761d2109bcefefca08f3f152f8c6b0689ec38af078e9b1dd2b36ab879170d7","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-20T01:46:07Z","title_canon_sha256":"ae68b1fe4ccfcb9d210c520dd01b2a65ef27b945538f11c44135b54efbfcf063"},"schema_version":"1.0","source":{"id":"2409.13156","kind":"arxiv","version":2}},"canonical_sha256":"d3724763076dbbe864fe048c1c4707a09c4ec3d5a6e0bb29b0d2ce20be38cb60","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"d3724763076dbbe864fe048c1c4707a09c4ec3d5a6e0bb29b0d2ce20be38cb60","first_computed_at":"2026-07-05T10:21:04.427259Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:21:04.427259Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"4heL3jxt244pSYfUoYcWQplJ24UpE5osAyOtRkTymKyHfiUvbCaAP4BIMAovtlENIWKuEL5IO961mEyIzs5DCw==","signature_status":"signed_v1","signed_at":"2026-07-05T10:21:04.427793Z","signed_message":"canonical_sha256_bytes"},"source_id":"2409.13156","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:9925b0c93c95d605e31392107ce1eac7b407ce57aceba0c410fbc4c288298d86","sha256:f72a4239e49c29a9299c75ad02f0bb07c23b0ce752ca49aef9aa4b704c8a9613"],"state_sha256":"d86751415f42258ed3ccb413f6951eba8a7bf0b7985797c0c44b95b6e384b30b"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"qrwLxbtljfasOOmjrrQ9F08uN2DGaJ4R1tDoIZlTZTgzSKNqK0+zLXfxX/qjjmqynSVZSdcB9ONYPlhKlAIHAA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-10T06:27:58.324053Z","bundle_sha256":"b1ecb57dce358349a52d848ed2604f453a934927f95815dde0f9afaffd9caa64"}}