{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:XZXLXPGQZ4BOK2JTC5JN2FNYCH","short_pith_number":"pith:XZXLXPGQ","canonical_record":{"source":{"id":"2406.16486","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2024-06-24T09:40:39Z","cross_cats_sorted":[],"title_canon_sha256":"c2a4dc2ea4f0f82f993000e88c76b8c68ff6f2d5b5cd0572ece9af1a0e3be502","abstract_canon_sha256":"5dbbbdb8f2a49e02e4dceec77fdb1b064e16099d42ca8f9f7b34e87cc803622c"},"schema_version":"1.0"},"canonical_sha256":"be6ebbbcd0cf02e569331752dd15b811e2103d2ad4abe2f1081f54062141579f","source":{"kind":"arxiv","id":"2406.16486","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.16486","created_at":"2026-07-05T08:35:59Z"},{"alias_kind":"arxiv_version","alias_value":"2406.16486v1","created_at":"2026-07-05T08:35:59Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.16486","created_at":"2026-07-05T08:35:59Z"},{"alias_kind":"pith_short_12","alias_value":"XZXLXPGQZ4BO","created_at":"2026-07-05T08:35:59Z"},{"alias_kind":"pith_short_16","alias_value":"XZXLXPGQZ4BOK2JT","created_at":"2026-07-05T08:35:59Z"},{"alias_kind":"pith_short_8","alias_value":"XZXLXPGQ","created_at":"2026-07-05T08:35:59Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:XZXLXPGQZ4BOK2JTC5JN2FNYCH","target":"record","payload":{"canonical_record":{"source":{"id":"2406.16486","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2024-06-24T09:40:39Z","cross_cats_sorted":[],"title_canon_sha256":"c2a4dc2ea4f0f82f993000e88c76b8c68ff6f2d5b5cd0572ece9af1a0e3be502","abstract_canon_sha256":"5dbbbdb8f2a49e02e4dceec77fdb1b064e16099d42ca8f9f7b34e87cc803622c"},"schema_version":"1.0"},"canonical_sha256":"be6ebbbcd0cf02e569331752dd15b811e2103d2ad4abe2f1081f54062141579f","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:35:59.494401Z","signature_b64":"L4Oih0Yr9gjxnl0Z2nnYrFpnI7TSOUVJve7IBlmgyBaT4TtQdFKUXF+pXYLhOk+g12RQOnJL4beMsPuvqP3cCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"be6ebbbcd0cf02e569331752dd15b811e2103d2ad4abe2f1081f54062141579f","last_reissued_at":"2026-07-05T08:35:59.493962Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:35:59.493962Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2406.16486","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:35:59Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"QK1A1eiABTLzCtfCXAC0GKSSjfSoX6VONGtRRIgjLCN4gRCrVtDPOBnwTeEsMmxW4B7GJwdaBWTioKfjDFyWAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-18T18:34:41.427908Z"},"content_sha256":"b8beb6d6499c3648c7b6e6b2f1ea9d1287258cac8815052058a4010c3543ee91","schema_version":"1.0","event_id":"sha256:b8beb6d6499c3648c7b6e6b2f1ea9d1287258cac8815052058a4010c3543ee91"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:XZXLXPGQZ4BOK2JTC5JN2FNYCH","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Towards Comprehensive Preference Data Collection for Reward Modeling","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Fuzheng Zhang, Ge Chen, Kaihui Chen, Lijun Mei, Qingyang Li, Sheng Ouyang, Xucheng Ye, Yong Liu, Yulan Hu","submitted_at":"2024-06-24T09:40:39Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) facilitates the alignment of large language models (LLMs) with human preferences, thereby enhancing the quality of responses generated. A critical component of RLHF is the reward model, which is trained on preference data and outputs a scalar reward during the inference stage. However, the collection of preference data still lacks thorough investigation. Recent studies indicate that preference data is collected either by AI or humans, where chosen and rejected instances are identified among pairwise responses. We question whether this process e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.16486","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.16486/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:35:59Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"yAYwsHmLf7MombHrCDIW4tIo4MI/FSLm10AnSO02LSZLnG6Q3hDNg6G8cFT/nYl6yRN5u0gQYRX2xLXBQ4wHCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-18T18:34:41.428812Z"},"content_sha256":"6c2878f838c85c5ccd205ec727347373a40e1cf9288ba4cf5d4ad1eb85d682d9","schema_version":"1.0","event_id":"sha256:6c2878f838c85c5ccd205ec727347373a40e1cf9288ba4cf5d4ad1eb85d682d9"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/XZXLXPGQZ4BOK2JTC5JN2FNYCH/bundle.json","state_url":"https://pith.science/pith/XZXLXPGQZ4BOK2JTC5JN2FNYCH/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/XZXLXPGQZ4BOK2JTC5JN2FNYCH/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-18T18:34:41Z","links":{"resolver":"https://pith.science/pith/XZXLXPGQZ4BOK2JTC5JN2FNYCH","bundle":"https://pith.science/pith/XZXLXPGQZ4BOK2JTC5JN2FNYCH/bundle.json","state":"https://pith.science/pith/XZXLXPGQZ4BOK2JTC5JN2FNYCH/state.json","well_known_bundle":"https://pith.science/.well-known/pith/XZXLXPGQZ4BOK2JTC5JN2FNYCH/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:XZXLXPGQZ4BOK2JTC5JN2FNYCH","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5dbbbdb8f2a49e02e4dceec77fdb1b064e16099d42ca8f9f7b34e87cc803622c","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2024-06-24T09:40:39Z","title_canon_sha256":"c2a4dc2ea4f0f82f993000e88c76b8c68ff6f2d5b5cd0572ece9af1a0e3be502"},"schema_version":"1.0","source":{"id":"2406.16486","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.16486","created_at":"2026-07-05T08:35:59Z"},{"alias_kind":"arxiv_version","alias_value":"2406.16486v1","created_at":"2026-07-05T08:35:59Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.16486","created_at":"2026-07-05T08:35:59Z"},{"alias_kind":"pith_short_12","alias_value":"XZXLXPGQZ4BO","created_at":"2026-07-05T08:35:59Z"},{"alias_kind":"pith_short_16","alias_value":"XZXLXPGQZ4BOK2JT","created_at":"2026-07-05T08:35:59Z"},{"alias_kind":"pith_short_8","alias_value":"XZXLXPGQ","created_at":"2026-07-05T08:35:59Z"}],"graph_snapshots":[{"event_id":"sha256:6c2878f838c85c5ccd205ec727347373a40e1cf9288ba4cf5d4ad1eb85d682d9","target":"graph","created_at":"2026-07-05T08:35:59Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2406.16486/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) facilitates the alignment of large language models (LLMs) with human preferences, thereby enhancing the quality of responses generated. A critical component of RLHF is the reward model, which is trained on preference data and outputs a scalar reward during the inference stage. However, the collection of preference data still lacks thorough investigation. Recent studies indicate that preference data is collected either by AI or humans, where chosen and rejected instances are identified among pairwise responses. We question whether this process e","authors_text":"Fuzheng Zhang, Ge Chen, Kaihui Chen, Lijun Mei, Qingyang Li, Sheng Ouyang, Xucheng Ye, Yong Liu, Yulan Hu","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2024-06-24T09:40:39Z","title":"Towards Comprehensive Preference Data Collection for Reward Modeling"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.16486","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:b8beb6d6499c3648c7b6e6b2f1ea9d1287258cac8815052058a4010c3543ee91","target":"record","created_at":"2026-07-05T08:35:59Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5dbbbdb8f2a49e02e4dceec77fdb1b064e16099d42ca8f9f7b34e87cc803622c","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2024-06-24T09:40:39Z","title_canon_sha256":"c2a4dc2ea4f0f82f993000e88c76b8c68ff6f2d5b5cd0572ece9af1a0e3be502"},"schema_version":"1.0","source":{"id":"2406.16486","kind":"arxiv","version":1}},"canonical_sha256":"be6ebbbcd0cf02e569331752dd15b811e2103d2ad4abe2f1081f54062141579f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"be6ebbbcd0cf02e569331752dd15b811e2103d2ad4abe2f1081f54062141579f","first_computed_at":"2026-07-05T08:35:59.493962Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:35:59.493962Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"L4Oih0Yr9gjxnl0Z2nnYrFpnI7TSOUVJve7IBlmgyBaT4TtQdFKUXF+pXYLhOk+g12RQOnJL4beMsPuvqP3cCA==","signature_status":"signed_v1","signed_at":"2026-07-05T08:35:59.494401Z","signed_message":"canonical_sha256_bytes"},"source_id":"2406.16486","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:b8beb6d6499c3648c7b6e6b2f1ea9d1287258cac8815052058a4010c3543ee91","sha256:6c2878f838c85c5ccd205ec727347373a40e1cf9288ba4cf5d4ad1eb85d682d9"],"state_sha256":"4851cc754bdba5f5e58a937a9d7d754252be36f8e170b60a98fd3eda05d5d052"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"dVuOUT+Zcc0tCfCDSLv12lUA3q8gDMhwAZqBOYdJrJU4Z7OZmdqRhnh+5LJrHkLAtf4H9rq+iY9K+H9KIfeKDA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-18T18:34:41.434591Z","bundle_sha256":"d0720a9bbb57035910a1116bc1dd6aacf07659aac83bd047815c8c1255bcb093"}}