{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WK67VAYX7EQ4AGRVFNZH5TGLU2","short_pith_number":"pith:WK67VAYX","schema_version":"1.0","canonical_sha256":"b2bdfa8317f921c01a352b727ecccba6aef63ec812cf35109635db821b849e05","source":{"kind":"arxiv","id":"2410.03742","version":2},"attestation_state":"computed","paper":{"title":"Beyond Scalar Reward Model: Learning Generative Judge from Preference Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Dong Yan, Qingyao Ai, Qiuchi Li, Wei Shen, Xiangsheng Li, Yiqun Liu, Yujia Zhou, Ziyi Ye","submitted_at":"2024-10-01T07:38:58Z","abstract_excerpt":"Learning from preference feedback is a common practice for aligning large language models~(LLMs) with human value. Conventionally, preference data is learned and encoded into a scalar reward model that connects a value head with an LLM to produce a scalar score as preference or reward. However, scalar models lack interpretability and are known to be susceptible to biases in datasets. This paper investigates leveraging the generation capability of LLMs to address both limitations in one shot. Specifically, we prompt the pre-trained LLM to generate positive and negative judgments, both supported"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.03742","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-01T07:38:58Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"f8509412b72751adfd3c34d6a9ba557781c95b7276cc2b0743ddbe2893ecf72c","abstract_canon_sha256":"2a5b89b1620e3b6e32661a1d85818039612c54cc7d0a6b69f1312d3060c93032"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:01:54.987188Z","signature_b64":"19xWFGv8KIho0h+kr7Fvab48BXCxZsnfI+nQPpTYs9aJsZTsQ0EYwRX+RYmjtr56oAbAfX0Ulv1fHejdTX+tCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b2bdfa8317f921c01a352b727ecccba6aef63ec812cf35109635db821b849e05","last_reissued_at":"2026-07-05T12:01:54.986639Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:01:54.986639Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Scalar Reward Model: Learning Generative Judge from Preference Data","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Dong Yan, Qingyao Ai, Qiuchi Li, Wei Shen, Xiangsheng Li, Yiqun Liu, Yujia Zhou, Ziyi Ye","submitted_at":"2024-10-01T07:38:58Z","abstract_excerpt":"Learning from preference feedback is a common practice for aligning large language models~(LLMs) with human value. Conventionally, preference data is learned and encoded into a scalar reward model that connects a value head with an LLM to produce a scalar score as preference or reward. However, scalar models lack interpretability and are known to be susceptible to biases in datasets. This paper investigates leveraging the generation capability of LLMs to address both limitations in one shot. Specifically, we prompt the pre-trained LLM to generate positive and negative judgments, both supported"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.03742","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.03742/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.03742","created_at":"2026-07-05T12:01:54.986702+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.03742v2","created_at":"2026-07-05T12:01:54.986702+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.03742","created_at":"2026-07-05T12:01:54.986702+00:00"},{"alias_kind":"pith_short_12","alias_value":"WK67VAYX7EQ4","created_at":"2026-07-05T12:01:54.986702+00:00"},{"alias_kind":"pith_short_16","alias_value":"WK67VAYX7EQ4AGRV","created_at":"2026-07-05T12:01:54.986702+00:00"},{"alias_kind":"pith_short_8","alias_value":"WK67VAYX","created_at":"2026-07-05T12:01:54.986702+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.23542","citing_title":"On the Shelf Life of Fine-Tuned LLM-Judges: Future-Proofing, Backward-Compatibility, and Question Generalization","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":280,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15676","citing_title":"EvoRAG: Making Knowledge Graph-based RAG Automatically Evolve through Feedback-driven Backpropagation","ref_index":93,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WK67VAYX7EQ4AGRVFNZH5TGLU2","json":"https://pith.science/pith/WK67VAYX7EQ4AGRVFNZH5TGLU2.json","graph_json":"https://pith.science/api/pith-number/WK67VAYX7EQ4AGRVFNZH5TGLU2/graph.json","events_json":"https://pith.science/api/pith-number/WK67VAYX7EQ4AGRVFNZH5TGLU2/events.json","paper":"https://pith.science/paper/WK67VAYX"},"agent_actions":{"view_html":"https://pith.science/pith/WK67VAYX7EQ4AGRVFNZH5TGLU2","download_json":"https://pith.science/pith/WK67VAYX7EQ4AGRVFNZH5TGLU2.json","view_paper":"https://pith.science/paper/WK67VAYX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.03742&json=true","fetch_graph":"https://pith.science/api/pith-number/WK67VAYX7EQ4AGRVFNZH5TGLU2/graph.json","fetch_events":"https://pith.science/api/pith-number/WK67VAYX7EQ4AGRVFNZH5TGLU2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WK67VAYX7EQ4AGRVFNZH5TGLU2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WK67VAYX7EQ4AGRVFNZH5TGLU2/action/storage_attestation","attest_author":"https://pith.science/pith/WK67VAYX7EQ4AGRVFNZH5TGLU2/action/author_attestation","sign_citation":"https://pith.science/pith/WK67VAYX7EQ4AGRVFNZH5TGLU2/action/citation_signature","submit_replication":"https://pith.science/pith/WK67VAYX7EQ4AGRVFNZH5TGLU2/action/replication_record"}},"created_at":"2026-07-05T12:01:54.986702+00:00","updated_at":"2026-07-05T12:01:54.986702+00:00"}