{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:7W2QKLGQ4QDFVTT6CMDCFOZYQR","short_pith_number":"pith:7W2QKLGQ","canonical_record":{"source":{"id":"2409.14664","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-23T02:08:20Z","cross_cats_sorted":[],"title_canon_sha256":"d9b43763757a87af27c291590cc30068d340c13081848568af27b7cef50ffd08","abstract_canon_sha256":"fee298dee947c908f1b270c434e97e85a3800767ef0b89e9f9f396328b327f60"},"schema_version":"1.0"},"canonical_sha256":"fdb5052cd0e4065ace7e130622bb3884634905fdce44c7eb87f0c13d0ed2312d","source":{"kind":"arxiv","id":"2409.14664","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2409.14664","created_at":"2026-07-05T12:10:53Z"},{"alias_kind":"arxiv_version","alias_value":"2409.14664v3","created_at":"2026-07-05T12:10:53Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.14664","created_at":"2026-07-05T12:10:53Z"},{"alias_kind":"pith_short_12","alias_value":"7W2QKLGQ4QDF","created_at":"2026-07-05T12:10:53Z"},{"alias_kind":"pith_short_16","alias_value":"7W2QKLGQ4QDFVTT6","created_at":"2026-07-05T12:10:53Z"},{"alias_kind":"pith_short_8","alias_value":"7W2QKLGQ","created_at":"2026-07-05T12:10:53Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:7W2QKLGQ4QDFVTT6CMDCFOZYQR","target":"record","payload":{"canonical_record":{"source":{"id":"2409.14664","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-23T02:08:20Z","cross_cats_sorted":[],"title_canon_sha256":"d9b43763757a87af27c291590cc30068d340c13081848568af27b7cef50ffd08","abstract_canon_sha256":"fee298dee947c908f1b270c434e97e85a3800767ef0b89e9f9f396328b327f60"},"schema_version":"1.0"},"canonical_sha256":"fdb5052cd0e4065ace7e130622bb3884634905fdce44c7eb87f0c13d0ed2312d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:10:53.479100Z","signature_b64":"/DpL2KMuy0eJbD4NF9dkfA1DnE8rEbkoMLbm8HoxFnOWv2+9TMr7zGEFbSffpJLr66SoospOaCPRqYmV9PIuCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fdb5052cd0e4065ace7e130622bb3884634905fdce44c7eb87f0c13d0ed2312d","last_reissued_at":"2026-07-05T12:10:53.478590Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:10:53.478590Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2409.14664","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:10:53Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"K2oMjw9fZ5bGicNrR2opKEvWnu+pLF7VRlMyZlzOPGeGk8bdo7vkAN6Ou3fyambiK93bP9Oyb+IuGf04QqqxAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T14:07:46.912343Z"},"content_sha256":"f88086fe782920b97254d7a72630254c0819c6f8a545706e95a357a45bc4f9ab","schema_version":"1.0","event_id":"sha256:f88086fe782920b97254d7a72630254c0819c6f8a545706e95a357a45bc4f9ab"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:7W2QKLGQ4QDFVTT6CMDCFOZYQR","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Direct Judgement Preference Optimization","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Austin Xu, Caiming Xiong, Peifeng Wang, Shafiq Joty, Yilun Zhou","submitted_at":"2024-09-23T02:08:20Z","abstract_excerpt":"Auto-evaluation is crucial for assessing response quality and offering feedback for model development. Recent studies have explored training large language models (LLMs) as generative judges to evaluate and critique other models' outputs. In this work, we investigate the idea of learning from both positive and negative data with preference optimization to enhance the evaluation capabilities of LLM judges across an array of different use cases. We achieve this by employing three approaches to collect the preference pairs for different use cases, each aimed at improving our generative judge from"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.14664","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.14664/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:10:53Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"PAZDMPcAkK3DUMwcXjqFJuwRp7RzWzRGDCvQyIFjT4LMPNLjDBIaTCA2XT8NOuOWk6/HZLWByP2/jg2ZEeR/Dg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T14:07:46.912892Z"},"content_sha256":"4fe955ed05cbc62cfad2c62c78baf25fa261f891937012437ed5f3abc81a6a1a","schema_version":"1.0","event_id":"sha256:4fe955ed05cbc62cfad2c62c78baf25fa261f891937012437ed5f3abc81a6a1a"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/7W2QKLGQ4QDFVTT6CMDCFOZYQR/bundle.json","state_url":"https://pith.science/pith/7W2QKLGQ4QDFVTT6CMDCFOZYQR/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/7W2QKLGQ4QDFVTT6CMDCFOZYQR/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T14:07:46Z","links":{"resolver":"https://pith.science/pith/7W2QKLGQ4QDFVTT6CMDCFOZYQR","bundle":"https://pith.science/pith/7W2QKLGQ4QDFVTT6CMDCFOZYQR/bundle.json","state":"https://pith.science/pith/7W2QKLGQ4QDFVTT6CMDCFOZYQR/state.json","well_known_bundle":"https://pith.science/.well-known/pith/7W2QKLGQ4QDFVTT6CMDCFOZYQR/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:7W2QKLGQ4QDFVTT6CMDCFOZYQR","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"fee298dee947c908f1b270c434e97e85a3800767ef0b89e9f9f396328b327f60","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-23T02:08:20Z","title_canon_sha256":"d9b43763757a87af27c291590cc30068d340c13081848568af27b7cef50ffd08"},"schema_version":"1.0","source":{"id":"2409.14664","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2409.14664","created_at":"2026-07-05T12:10:53Z"},{"alias_kind":"arxiv_version","alias_value":"2409.14664v3","created_at":"2026-07-05T12:10:53Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.14664","created_at":"2026-07-05T12:10:53Z"},{"alias_kind":"pith_short_12","alias_value":"7W2QKLGQ4QDF","created_at":"2026-07-05T12:10:53Z"},{"alias_kind":"pith_short_16","alias_value":"7W2QKLGQ4QDFVTT6","created_at":"2026-07-05T12:10:53Z"},{"alias_kind":"pith_short_8","alias_value":"7W2QKLGQ","created_at":"2026-07-05T12:10:53Z"}],"graph_snapshots":[{"event_id":"sha256:4fe955ed05cbc62cfad2c62c78baf25fa261f891937012437ed5f3abc81a6a1a","target":"graph","created_at":"2026-07-05T12:10:53Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2409.14664/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Auto-evaluation is crucial for assessing response quality and offering feedback for model development. Recent studies have explored training large language models (LLMs) as generative judges to evaluate and critique other models' outputs. In this work, we investigate the idea of learning from both positive and negative data with preference optimization to enhance the evaluation capabilities of LLM judges across an array of different use cases. We achieve this by employing three approaches to collect the preference pairs for different use cases, each aimed at improving our generative judge from","authors_text":"Austin Xu, Caiming Xiong, Peifeng Wang, Shafiq Joty, Yilun Zhou","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-23T02:08:20Z","title":"Direct Judgement Preference Optimization"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.14664","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:f88086fe782920b97254d7a72630254c0819c6f8a545706e95a357a45bc4f9ab","target":"record","created_at":"2026-07-05T12:10:53Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"fee298dee947c908f1b270c434e97e85a3800767ef0b89e9f9f396328b327f60","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-23T02:08:20Z","title_canon_sha256":"d9b43763757a87af27c291590cc30068d340c13081848568af27b7cef50ffd08"},"schema_version":"1.0","source":{"id":"2409.14664","kind":"arxiv","version":3}},"canonical_sha256":"fdb5052cd0e4065ace7e130622bb3884634905fdce44c7eb87f0c13d0ed2312d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"fdb5052cd0e4065ace7e130622bb3884634905fdce44c7eb87f0c13d0ed2312d","first_computed_at":"2026-07-05T12:10:53.478590Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T12:10:53.478590Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"/DpL2KMuy0eJbD4NF9dkfA1DnE8rEbkoMLbm8HoxFnOWv2+9TMr7zGEFbSffpJLr66SoospOaCPRqYmV9PIuCQ==","signature_status":"signed_v1","signed_at":"2026-07-05T12:10:53.479100Z","signed_message":"canonical_sha256_bytes"},"source_id":"2409.14664","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:f88086fe782920b97254d7a72630254c0819c6f8a545706e95a357a45bc4f9ab","sha256:4fe955ed05cbc62cfad2c62c78baf25fa261f891937012437ed5f3abc81a6a1a"],"state_sha256":"26599b8d2f31be14388bfcc7e41f339e9d4be513941867203008257ef9a79070"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Bqs5aS2/L0dOOFRWXWwo0jH8nId9TQlf0hMFEeDj2r9qZwbe9U/sacA8TbhK5IrWmkdU474260NPeSHVsbAnCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T14:07:46.918132Z","bundle_sha256":"f06f60dc650aebbec704877745bcc880264a56f40f4f76cb6ad010bf212d6fdc"}}