{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:W3U3E2CASJM76JFD7IV5RUDNO3","short_pith_number":"pith:W3U3E2CA","canonical_record":{"source":{"id":"2402.09764","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-02-15T07:29:43Z","cross_cats_sorted":[],"title_canon_sha256":"c4c94d054e0d2d55ae4d0f58c9d083e6f0233e891192ea8faca2d53cc283d5f3","abstract_canon_sha256":"41a9731f40a0926c4e668c9f86b0de3affe49dd59eb850035c4a69868dd463cf"},"schema_version":"1.0"},"canonical_sha256":"b6e9b268409259ff24a3fa2bd8d06d76f761b3714b01ad6793819420ed8895ca","source":{"kind":"arxiv","id":"2402.09764","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2402.09764","created_at":"2026-07-05T08:25:00Z"},{"alias_kind":"arxiv_version","alias_value":"2402.09764v3","created_at":"2026-07-05T08:25:00Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.09764","created_at":"2026-07-05T08:25:00Z"},{"alias_kind":"pith_short_12","alias_value":"W3U3E2CASJM7","created_at":"2026-07-05T08:25:00Z"},{"alias_kind":"pith_short_16","alias_value":"W3U3E2CASJM76JFD","created_at":"2026-07-05T08:25:00Z"},{"alias_kind":"pith_short_8","alias_value":"W3U3E2CA","created_at":"2026-07-05T08:25:00Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:W3U3E2CASJM76JFD7IV5RUDNO3","target":"record","payload":{"canonical_record":{"source":{"id":"2402.09764","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-02-15T07:29:43Z","cross_cats_sorted":[],"title_canon_sha256":"c4c94d054e0d2d55ae4d0f58c9d083e6f0233e891192ea8faca2d53cc283d5f3","abstract_canon_sha256":"41a9731f40a0926c4e668c9f86b0de3affe49dd59eb850035c4a69868dd463cf"},"schema_version":"1.0"},"canonical_sha256":"b6e9b268409259ff24a3fa2bd8d06d76f761b3714b01ad6793819420ed8895ca","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:25:00.584493Z","signature_b64":"449XG12y6bDzecQpurts9V7uPXKMMRZcw1qmf94T9VN0ZPqQ6UnnjMx4mITtBT9kxGx/CdJpxmeuAgle39w3Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b6e9b268409259ff24a3fa2bd8d06d76f761b3714b01ad6793819420ed8895ca","last_reissued_at":"2026-07-05T08:25:00.584007Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:25:00.584007Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2402.09764","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:25:00Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"K2T00XxVnX8knRbaUpFL5fKw5y2mcjfZgNbha5jRvmzWFuqLzDcaju6a5kebVq30OtqFwh5ekRjoLTJKmHNsCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T02:26:53.867553Z"},"content_sha256":"92b7e99f003d3a72d1671d5d13ac6c42a7e5ec82f702e1f0f7156e84f306f519","schema_version":"1.0","event_id":"sha256:92b7e99f003d3a72d1671d5d13ac6c42a7e5ec82f702e1f0f7156e84f306f519"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:W3U3E2CASJM76JFD7IV5RUDNO3","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Aligning Crowd Feedback via Distributional Preference Reward Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Cong Zhang, Derrick Goh Xin Deik, Dexun Li, Kuicai Dong, Ruiming Tang, Yong Liu","submitted_at":"2024-02-15T07:29:43Z","abstract_excerpt":"Deep Reinforcement Learning is widely used for aligning Large Language Models (LLM) with human preference. However, the conventional reward modelling is predominantly dependent on human annotations provided by a select cohort of individuals. Such dependence may unintentionally result in skewed models that reflect the inclinations of these annotators, thereby failing to adequately represent the wider population's expectations. We propose the Distributional Preference Reward Model (DPRM), a simple yet effective framework to align large language models with diverse human preferences. To this end,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.09764","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.09764/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:25:00Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"n5n0z2sghS2LTnhSuoDL/DW70yfOqduCVAfzk25E/rJyWnlf/KXUYFZabQXQVPbJWPq8QaoS0IBr8b+7FKSwAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T02:26:53.868049Z"},"content_sha256":"64afb2d1ac0bfad67870f3e8074461e3ef48c1c7c959669d28e9fa5abed31e7b","schema_version":"1.0","event_id":"sha256:64afb2d1ac0bfad67870f3e8074461e3ef48c1c7c959669d28e9fa5abed31e7b"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/W3U3E2CASJM76JFD7IV5RUDNO3/bundle.json","state_url":"https://pith.science/pith/W3U3E2CASJM76JFD7IV5RUDNO3/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/W3U3E2CASJM76JFD7IV5RUDNO3/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T02:26:53Z","links":{"resolver":"https://pith.science/pith/W3U3E2CASJM76JFD7IV5RUDNO3","bundle":"https://pith.science/pith/W3U3E2CASJM76JFD7IV5RUDNO3/bundle.json","state":"https://pith.science/pith/W3U3E2CASJM76JFD7IV5RUDNO3/state.json","well_known_bundle":"https://pith.science/.well-known/pith/W3U3E2CASJM76JFD7IV5RUDNO3/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:W3U3E2CASJM76JFD7IV5RUDNO3","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"41a9731f40a0926c4e668c9f86b0de3affe49dd59eb850035c4a69868dd463cf","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-02-15T07:29:43Z","title_canon_sha256":"c4c94d054e0d2d55ae4d0f58c9d083e6f0233e891192ea8faca2d53cc283d5f3"},"schema_version":"1.0","source":{"id":"2402.09764","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2402.09764","created_at":"2026-07-05T08:25:00Z"},{"alias_kind":"arxiv_version","alias_value":"2402.09764v3","created_at":"2026-07-05T08:25:00Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.09764","created_at":"2026-07-05T08:25:00Z"},{"alias_kind":"pith_short_12","alias_value":"W3U3E2CASJM7","created_at":"2026-07-05T08:25:00Z"},{"alias_kind":"pith_short_16","alias_value":"W3U3E2CASJM76JFD","created_at":"2026-07-05T08:25:00Z"},{"alias_kind":"pith_short_8","alias_value":"W3U3E2CA","created_at":"2026-07-05T08:25:00Z"}],"graph_snapshots":[{"event_id":"sha256:64afb2d1ac0bfad67870f3e8074461e3ef48c1c7c959669d28e9fa5abed31e7b","target":"graph","created_at":"2026-07-05T08:25:00Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2402.09764/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Deep Reinforcement Learning is widely used for aligning Large Language Models (LLM) with human preference. However, the conventional reward modelling is predominantly dependent on human annotations provided by a select cohort of individuals. Such dependence may unintentionally result in skewed models that reflect the inclinations of these annotators, thereby failing to adequately represent the wider population's expectations. We propose the Distributional Preference Reward Model (DPRM), a simple yet effective framework to align large language models with diverse human preferences. To this end,","authors_text":"Cong Zhang, Derrick Goh Xin Deik, Dexun Li, Kuicai Dong, Ruiming Tang, Yong Liu","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-02-15T07:29:43Z","title":"Aligning Crowd Feedback via Distributional Preference Reward Modeling"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.09764","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:92b7e99f003d3a72d1671d5d13ac6c42a7e5ec82f702e1f0f7156e84f306f519","target":"record","created_at":"2026-07-05T08:25:00Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"41a9731f40a0926c4e668c9f86b0de3affe49dd59eb850035c4a69868dd463cf","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-02-15T07:29:43Z","title_canon_sha256":"c4c94d054e0d2d55ae4d0f58c9d083e6f0233e891192ea8faca2d53cc283d5f3"},"schema_version":"1.0","source":{"id":"2402.09764","kind":"arxiv","version":3}},"canonical_sha256":"b6e9b268409259ff24a3fa2bd8d06d76f761b3714b01ad6793819420ed8895ca","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b6e9b268409259ff24a3fa2bd8d06d76f761b3714b01ad6793819420ed8895ca","first_computed_at":"2026-07-05T08:25:00.584007Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:25:00.584007Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"449XG12y6bDzecQpurts9V7uPXKMMRZcw1qmf94T9VN0ZPqQ6UnnjMx4mITtBT9kxGx/CdJpxmeuAgle39w3Bw==","signature_status":"signed_v1","signed_at":"2026-07-05T08:25:00.584493Z","signed_message":"canonical_sha256_bytes"},"source_id":"2402.09764","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:92b7e99f003d3a72d1671d5d13ac6c42a7e5ec82f702e1f0f7156e84f306f519","sha256:64afb2d1ac0bfad67870f3e8074461e3ef48c1c7c959669d28e9fa5abed31e7b"],"state_sha256":"4b5725a83a18ddb1ffaf270813606df012bd2ba24a55551a078f40da7573b0cc"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"1M/jecVMsvktg+YdHs47mkyga7yhb+azPs5H3mR2kN7d0i6xmrFvk0nITgON598FSwQNNUBciicXCwKk3bFNDg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T02:26:53.871783Z","bundle_sha256":"06fb0f4fc4cac790eff19fca1a4b55ad3b0b82a5541185cf8e107944efc5a25e"}}