{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:HMID5BUMY7VDYUQDUFVDNRQBV7","short_pith_number":"pith:HMID5BUM","canonical_record":{"source":{"id":"2409.06957","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-11T02:40:38Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5c4c2cdb47d4b2c145469875d21ac59769b8bf93c8980e333d5b9f1866681bb5","abstract_canon_sha256":"d21b9df5c6899394418734f1e443f6aba330a7e54fc5ae6cfa0ff82adc002e72"},"schema_version":"1.0"},"canonical_sha256":"3b103e868cc7ea3c5203a16a36c601aff7aa8dd608bd51f99380649cb654f18d","source":{"kind":"arxiv","id":"2409.06957","version":5},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2409.06957","created_at":"2026-07-05T11:17:35Z"},{"alias_kind":"arxiv_version","alias_value":"2409.06957v5","created_at":"2026-07-05T11:17:35Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.06957","created_at":"2026-07-05T11:17:35Z"},{"alias_kind":"pith_short_12","alias_value":"HMID5BUMY7VD","created_at":"2026-07-05T11:17:35Z"},{"alias_kind":"pith_short_16","alias_value":"HMID5BUMY7VDYUQD","created_at":"2026-07-05T11:17:35Z"},{"alias_kind":"pith_short_8","alias_value":"HMID5BUM","created_at":"2026-07-05T11:17:35Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:HMID5BUMY7VDYUQDUFVDNRQBV7","target":"record","payload":{"canonical_record":{"source":{"id":"2409.06957","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-11T02:40:38Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"5c4c2cdb47d4b2c145469875d21ac59769b8bf93c8980e333d5b9f1866681bb5","abstract_canon_sha256":"d21b9df5c6899394418734f1e443f6aba330a7e54fc5ae6cfa0ff82adc002e72"},"schema_version":"1.0"},"canonical_sha256":"3b103e868cc7ea3c5203a16a36c601aff7aa8dd608bd51f99380649cb654f18d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:17:35.660568Z","signature_b64":"tfyPiK8/0Csuuwd1vQMAth9XwXcfmDWyuzrNcUnFrLBRPMDhqy8r7DDXpZu/b1pDFwaWM5U7W4ArZrobPLyMAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3b103e868cc7ea3c5203a16a36c601aff7aa8dd608bd51f99380649cb654f18d","last_reissued_at":"2026-07-05T11:17:35.660096Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:17:35.660096Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2409.06957","source_version":5,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:17:35Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Q2Zo05Gxbxg6XcFN+/XyvqP2INpZeMvj4yZDtXs8pCu2WCcmnu2b5N/rYwiQxFTILXyT8+gOr0qQKisg0vn+Bw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-18T12:02:02.002809Z"},"content_sha256":"ef73b925415f00ae893a0e5124832dc2fdf7300ae102dcf71737c676ea11f6f7","schema_version":"1.0","event_id":"sha256:ef73b925415f00ae893a0e5124832dc2fdf7300ae102dcf71737c676ea11f6f7"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:HMID5BUMY7VDYUQDUFVDNRQBV7","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Policy Filtration for RLHF to Mitigate Noise in Reward Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chuheng Zhang, Jiang Bian, Li Zhao, Wanchun Dou, Wei Shen, Xiaolong Xu, Xuyun Zhang","submitted_at":"2024-09-11T02:40:38Z","abstract_excerpt":"While direct policy optimization methods exist, pioneering LLMs are fine-tuned with reinforcement learning from human feedback (RLHF) to generate better responses under the supervision of a reward model learned from preference data. One major challenge of RLHF is the inaccuracy of the intermediate reward model, especially in the tasks that requires complex reasoning for the reward model to score a response. We find that the reliability of the reward model varies across responses assigned with different rewards. This motivates us to filter the samples whose rewards may be unreliable to improve "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.06957","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.06957/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:17:35Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"JpPhJlFFluQA2HMEgfusWXOtHARxFzaErYZtgSvGNr20Uwnz60QF8VuCj+3nhZgDKE76puuFz3R1XbFaEK0gDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-18T12:02:02.003499Z"},"content_sha256":"6edf30084b69701a8547ec748dce67f404d7b3f829cd7213b7a6356e661825b4","schema_version":"1.0","event_id":"sha256:6edf30084b69701a8547ec748dce67f404d7b3f829cd7213b7a6356e661825b4"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/HMID5BUMY7VDYUQDUFVDNRQBV7/bundle.json","state_url":"https://pith.science/pith/HMID5BUMY7VDYUQDUFVDNRQBV7/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/HMID5BUMY7VDYUQDUFVDNRQBV7/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-18T12:02:02Z","links":{"resolver":"https://pith.science/pith/HMID5BUMY7VDYUQDUFVDNRQBV7","bundle":"https://pith.science/pith/HMID5BUMY7VDYUQDUFVDNRQBV7/bundle.json","state":"https://pith.science/pith/HMID5BUMY7VDYUQDUFVDNRQBV7/state.json","well_known_bundle":"https://pith.science/.well-known/pith/HMID5BUMY7VDYUQDUFVDNRQBV7/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:HMID5BUMY7VDYUQDUFVDNRQBV7","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"d21b9df5c6899394418734f1e443f6aba330a7e54fc5ae6cfa0ff82adc002e72","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-11T02:40:38Z","title_canon_sha256":"5c4c2cdb47d4b2c145469875d21ac59769b8bf93c8980e333d5b9f1866681bb5"},"schema_version":"1.0","source":{"id":"2409.06957","kind":"arxiv","version":5}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2409.06957","created_at":"2026-07-05T11:17:35Z"},{"alias_kind":"arxiv_version","alias_value":"2409.06957v5","created_at":"2026-07-05T11:17:35Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.06957","created_at":"2026-07-05T11:17:35Z"},{"alias_kind":"pith_short_12","alias_value":"HMID5BUMY7VD","created_at":"2026-07-05T11:17:35Z"},{"alias_kind":"pith_short_16","alias_value":"HMID5BUMY7VDYUQD","created_at":"2026-07-05T11:17:35Z"},{"alias_kind":"pith_short_8","alias_value":"HMID5BUM","created_at":"2026-07-05T11:17:35Z"}],"graph_snapshots":[{"event_id":"sha256:6edf30084b69701a8547ec748dce67f404d7b3f829cd7213b7a6356e661825b4","target":"graph","created_at":"2026-07-05T11:17:35Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2409.06957/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"While direct policy optimization methods exist, pioneering LLMs are fine-tuned with reinforcement learning from human feedback (RLHF) to generate better responses under the supervision of a reward model learned from preference data. One major challenge of RLHF is the inaccuracy of the intermediate reward model, especially in the tasks that requires complex reasoning for the reward model to score a response. We find that the reliability of the reward model varies across responses assigned with different rewards. This motivates us to filter the samples whose rewards may be unreliable to improve ","authors_text":"Chuheng Zhang, Jiang Bian, Li Zhao, Wanchun Dou, Wei Shen, Xiaolong Xu, Xuyun Zhang","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-11T02:40:38Z","title":"Policy Filtration for RLHF to Mitigate Noise in Reward Models"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.06957","kind":"arxiv","version":5},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:ef73b925415f00ae893a0e5124832dc2fdf7300ae102dcf71737c676ea11f6f7","target":"record","created_at":"2026-07-05T11:17:35Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"d21b9df5c6899394418734f1e443f6aba330a7e54fc5ae6cfa0ff82adc002e72","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-09-11T02:40:38Z","title_canon_sha256":"5c4c2cdb47d4b2c145469875d21ac59769b8bf93c8980e333d5b9f1866681bb5"},"schema_version":"1.0","source":{"id":"2409.06957","kind":"arxiv","version":5}},"canonical_sha256":"3b103e868cc7ea3c5203a16a36c601aff7aa8dd608bd51f99380649cb654f18d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"3b103e868cc7ea3c5203a16a36c601aff7aa8dd608bd51f99380649cb654f18d","first_computed_at":"2026-07-05T11:17:35.660096Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:17:35.660096Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"tfyPiK8/0Csuuwd1vQMAth9XwXcfmDWyuzrNcUnFrLBRPMDhqy8r7DDXpZu/b1pDFwaWM5U7W4ArZrobPLyMAA==","signature_status":"signed_v1","signed_at":"2026-07-05T11:17:35.660568Z","signed_message":"canonical_sha256_bytes"},"source_id":"2409.06957","source_kind":"arxiv","source_version":5}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:ef73b925415f00ae893a0e5124832dc2fdf7300ae102dcf71737c676ea11f6f7","sha256:6edf30084b69701a8547ec748dce67f404d7b3f829cd7213b7a6356e661825b4"],"state_sha256":"bb49fc65b355ed77c6d4fa1cb6c58c71f356e12c6858e07fb06239b530abcfce"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ZJzyjcJqqJ6Hg+FBwAYKTquEeY/BSYzE5DbsyCnNNEGLnMavksw/yDiJYSZk6tw8i2JrG+QcLy+Xu+hqAZ6KAQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-18T12:02:02.009602Z","bundle_sha256":"88885cba86af8b07f2c1b0b60a9fbee7d5bdd4c446f801583a19c1eeddf891c5"}}