{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:WK67VAYX7EQ4AGRVFNZH5TGLU2","short_pith_number":"pith:WK67VAYX","canonical_record":{"source":{"id":"2410.03742","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-01T07:38:58Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"f8509412b72751adfd3c34d6a9ba557781c95b7276cc2b0743ddbe2893ecf72c","abstract_canon_sha256":"2a5b89b1620e3b6e32661a1d85818039612c54cc7d0a6b69f1312d3060c93032"},"schema_version":"1.0"},"canonical_sha256":"b2bdfa8317f921c01a352b727ecccba6aef63ec812cf35109635db821b849e05","source":{"kind":"arxiv","id":"2410.03742","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.03742","created_at":"2026-07-05T12:01:54Z"},{"alias_kind":"arxiv_version","alias_value":"2410.03742v2","created_at":"2026-07-05T12:01:54Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.03742","created_at":"2026-07-05T12:01:54Z"},{"alias_kind":"pith_short_12","alias_value":"WK67VAYX7EQ4","created_at":"2026-07-05T12:01:54Z"},{"alias_kind":"pith_short_16","alias_value":"WK67VAYX7EQ4AGRV","created_at":"2026-07-05T12:01:54Z"},{"alias_kind":"pith_short_8","alias_value":"WK67VAYX","created_at":"2026-07-05T12:01:54Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:WK67VAYX7EQ4AGRVFNZH5TGLU2","target":"record","payload":{"canonical_record":{"source":{"id":"2410.03742","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-01T07:38:58Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"f8509412b72751adfd3c34d6a9ba557781c95b7276cc2b0743ddbe2893ecf72c","abstract_canon_sha256":"2a5b89b1620e3b6e32661a1d85818039612c54cc7d0a6b69f1312d3060c93032"},"schema_version":"1.0"},"canonical_sha256":"b2bdfa8317f921c01a352b727ecccba6aef63ec812cf35109635db821b849e05","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:01:54.987188Z","signature_b64":"19xWFGv8KIho0h+kr7Fvab48BXCxZsnfI+nQPpTYs9aJsZTsQ0EYwRX+RYmjtr56oAbAfX0Ulv1fHejdTX+tCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b2bdfa8317f921c01a352b727ecccba6aef63ec812cf35109635db821b849e05","last_reissued_at":"2026-07-05T12:01:54.986639Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:01:54.986639Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2410.03742","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:01:54Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"fNhMYnVM+PFjnBakES2ZqcA8QQgzkOg61QUSvOJEacF1ZV3Xo4jxCdz3CVhQsLP0mA+uUx3IVCOLg7xQOzwpCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T04:04:01.109886Z"},"content_sha256":"f8108b7f831f434994e91b417a6fbed2ed5e68340304683e1c45fdc88727a88f","schema_version":"1.0","event_id":"sha256:f8108b7f831f434994e91b417a6fbed2ed5e68340304683e1c45fdc88727a88f"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:WK67VAYX7EQ4AGRVFNZH5TGLU2","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Beyond Scalar Reward Model: Learning Generative Judge from Preference Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Dong Yan, Qingyao Ai, Qiuchi Li, Wei Shen, Xiangsheng Li, Yiqun Liu, Yujia Zhou, Ziyi Ye","submitted_at":"2024-10-01T07:38:58Z","abstract_excerpt":"Learning from preference feedback is a common practice for aligning large language models~(LLMs) with human value. Conventionally, preference data is learned and encoded into a scalar reward model that connects a value head with an LLM to produce a scalar score as preference or reward. However, scalar models lack interpretability and are known to be susceptible to biases in datasets. This paper investigates leveraging the generation capability of LLMs to address both limitations in one shot. Specifically, we prompt the pre-trained LLM to generate positive and negative judgments, both supported"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.03742","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.03742/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:01:54Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"+7znc945LBIs9vtUaQGv4OYJ+bBWxghOdcgVMt5HTCS497BEIJH9aDYxLpqmdeCONhO0jziWpleANyXipsZfDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T04:04:01.110412Z"},"content_sha256":"77e26ceeb4da893d0f547f9d31e08f34c6457d723ee0a6a9bc756f1df2ca5d93","schema_version":"1.0","event_id":"sha256:77e26ceeb4da893d0f547f9d31e08f34c6457d723ee0a6a9bc756f1df2ca5d93"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/WK67VAYX7EQ4AGRVFNZH5TGLU2/bundle.json","state_url":"https://pith.science/pith/WK67VAYX7EQ4AGRVFNZH5TGLU2/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/WK67VAYX7EQ4AGRVFNZH5TGLU2/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-12T04:04:01Z","links":{"resolver":"https://pith.science/pith/WK67VAYX7EQ4AGRVFNZH5TGLU2","bundle":"https://pith.science/pith/WK67VAYX7EQ4AGRVFNZH5TGLU2/bundle.json","state":"https://pith.science/pith/WK67VAYX7EQ4AGRVFNZH5TGLU2/state.json","well_known_bundle":"https://pith.science/.well-known/pith/WK67VAYX7EQ4AGRVFNZH5TGLU2/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:WK67VAYX7EQ4AGRVFNZH5TGLU2","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"2a5b89b1620e3b6e32661a1d85818039612c54cc7d0a6b69f1312d3060c93032","cross_cats_sorted":["cs.AI","cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-01T07:38:58Z","title_canon_sha256":"f8509412b72751adfd3c34d6a9ba557781c95b7276cc2b0743ddbe2893ecf72c"},"schema_version":"1.0","source":{"id":"2410.03742","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.03742","created_at":"2026-07-05T12:01:54Z"},{"alias_kind":"arxiv_version","alias_value":"2410.03742v2","created_at":"2026-07-05T12:01:54Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.03742","created_at":"2026-07-05T12:01:54Z"},{"alias_kind":"pith_short_12","alias_value":"WK67VAYX7EQ4","created_at":"2026-07-05T12:01:54Z"},{"alias_kind":"pith_short_16","alias_value":"WK67VAYX7EQ4AGRV","created_at":"2026-07-05T12:01:54Z"},{"alias_kind":"pith_short_8","alias_value":"WK67VAYX","created_at":"2026-07-05T12:01:54Z"}],"graph_snapshots":[{"event_id":"sha256:77e26ceeb4da893d0f547f9d31e08f34c6457d723ee0a6a9bc756f1df2ca5d93","target":"graph","created_at":"2026-07-05T12:01:54Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2410.03742/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Learning from preference feedback is a common practice for aligning large language models~(LLMs) with human value. Conventionally, preference data is learned and encoded into a scalar reward model that connects a value head with an LLM to produce a scalar score as preference or reward. However, scalar models lack interpretability and are known to be susceptible to biases in datasets. This paper investigates leveraging the generation capability of LLMs to address both limitations in one shot. Specifically, we prompt the pre-trained LLM to generate positive and negative judgments, both supported","authors_text":"Dong Yan, Qingyao Ai, Qiuchi Li, Wei Shen, Xiangsheng Li, Yiqun Liu, Yujia Zhou, Ziyi Ye","cross_cats":["cs.AI","cs.LG"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-01T07:38:58Z","title":"Beyond Scalar Reward Model: Learning Generative Judge from Preference Data"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.03742","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:f8108b7f831f434994e91b417a6fbed2ed5e68340304683e1c45fdc88727a88f","target":"record","created_at":"2026-07-05T12:01:54Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"2a5b89b1620e3b6e32661a1d85818039612c54cc7d0a6b69f1312d3060c93032","cross_cats_sorted":["cs.AI","cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-01T07:38:58Z","title_canon_sha256":"f8509412b72751adfd3c34d6a9ba557781c95b7276cc2b0743ddbe2893ecf72c"},"schema_version":"1.0","source":{"id":"2410.03742","kind":"arxiv","version":2}},"canonical_sha256":"b2bdfa8317f921c01a352b727ecccba6aef63ec812cf35109635db821b849e05","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b2bdfa8317f921c01a352b727ecccba6aef63ec812cf35109635db821b849e05","first_computed_at":"2026-07-05T12:01:54.986639Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T12:01:54.986639Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"19xWFGv8KIho0h+kr7Fvab48BXCxZsnfI+nQPpTYs9aJsZTsQ0EYwRX+RYmjtr56oAbAfX0Ulv1fHejdTX+tCQ==","signature_status":"signed_v1","signed_at":"2026-07-05T12:01:54.987188Z","signed_message":"canonical_sha256_bytes"},"source_id":"2410.03742","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:f8108b7f831f434994e91b417a6fbed2ed5e68340304683e1c45fdc88727a88f","sha256:77e26ceeb4da893d0f547f9d31e08f34c6457d723ee0a6a9bc756f1df2ca5d93"],"state_sha256":"5428b09e8ba4e8e8ff5fc9fe3efe78c8ef35a48647ff5a80ecff6343f018cfe6"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"5tskdJ4d00Uk0F7ISwvoCz/O8StSrQmieAZmlJMD0ptjswMIs6pixS4KFb+VuPjIEpg78GUFoQU+5uv/GRtQDQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-12T04:04:01.114524Z","bundle_sha256":"8413ca8ebb113b36f1dbddef3dc9d2c3bc059e79896bd6d213986bb7cf15d3fc"}}