{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:KD2PXSWUDUG5DE4O7DRNGWGL7X","short_pith_number":"pith:KD2PXSWU","canonical_record":{"source":{"id":"2504.14716","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-20T19:05:59Z","cross_cats_sorted":[],"title_canon_sha256":"ae8567433809ca8a71bb0849a729231156550d20d434185b46caca454f3ba566","abstract_canon_sha256":"615d5bae15c0a1748c6b84580490e718592312031f8a646901c376c8f4e61999"},"schema_version":"1.0"},"canonical_sha256":"50f4fbcad41d0dd1938ef8e2d358cbfdcc50b2390117e5d7076b775209736a01","source":{"kind":"arxiv","id":"2504.14716","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2504.14716","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"arxiv_version","alias_value":"2504.14716v2","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.14716","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"pith_short_12","alias_value":"KD2PXSWUDUG5","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"pith_short_16","alias_value":"KD2PXSWUDUG5DE4O","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"pith_short_8","alias_value":"KD2PXSWU","created_at":"2026-07-05T11:56:56Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:KD2PXSWUDUG5DE4O7DRNGWGL7X","target":"record","payload":{"canonical_record":{"source":{"id":"2504.14716","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-20T19:05:59Z","cross_cats_sorted":[],"title_canon_sha256":"ae8567433809ca8a71bb0849a729231156550d20d434185b46caca454f3ba566","abstract_canon_sha256":"615d5bae15c0a1748c6b84580490e718592312031f8a646901c376c8f4e61999"},"schema_version":"1.0"},"canonical_sha256":"50f4fbcad41d0dd1938ef8e2d358cbfdcc50b2390117e5d7076b775209736a01","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:56:56.145937Z","signature_b64":"Re4SAM1DaowIk1rZxYynjatE4jD/b+PvlssGO/XGuwYJXEgjPn8l2AzXC/UzNHI61N/Pumo79zOBD7s26PyNBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"50f4fbcad41d0dd1938ef8e2d358cbfdcc50b2390117e5d7076b775209736a01","last_reissued_at":"2026-07-05T11:56:56.145448Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:56:56.145448Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2504.14716","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:56:56Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"+9uZky6hZFTKi1tgWcQbR/zwzPaSRgbjXEiat4umwxdDdaflKb9yc8erSVY/qp0iOA7AdwQYMd48VqAuGtaEBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T21:18:44.249958Z"},"content_sha256":"4bd141e695c850e6bc90c43d6f4ba0deacf9c6354e863a06c5b3eff64a02dd06","schema_version":"1.0","event_id":"sha256:4bd141e695c850e6bc90c43d6f4ba0deacf9c6354e863a06c5b3eff64a02dd06"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:KD2PXSWUDUG5DE4O7DRNGWGL7X","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Pairwise or Pointwise? Evaluating Feedback Protocols for Bias in LLM-Based Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Greg Durrett, Manya Wadhwa, Scott Niekum, Tuhina Tripathi","submitted_at":"2025-04-20T19:05:59Z","abstract_excerpt":"Large Language Models (LLMs) are widely used as proxies for human labelers in both training (Reinforcement Learning from AI Feedback) and large-scale response evaluation (LLM-as-a-judge). Alignment and evaluation are critical components in the development of reliable LLMs, and the choice of feedback protocol plays a central role in both but remains understudied. In this work, we show that the choice of feedback protocol for evaluation (absolute scores versus relative preferences) can significantly affect evaluation reliability and induce systematic biases. In the context of LLM-as-a-judge eval"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.14716","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.14716/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:56:56Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"wPeStxpwNvAyyH4UwesQsA5w1QPaSIilHw375uE0pKH1LX18KcrhdFkyK2MVHiHXtoThxurIzy+TYdiB/KGYCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T21:18:44.250470Z"},"content_sha256":"92d75dc6484ca0cec1d96a1d0b6851cc8ac4168ea0c3fb9facca790b50157865","schema_version":"1.0","event_id":"sha256:92d75dc6484ca0cec1d96a1d0b6851cc8ac4168ea0c3fb9facca790b50157865"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/KD2PXSWUDUG5DE4O7DRNGWGL7X/bundle.json","state_url":"https://pith.science/pith/KD2PXSWUDUG5DE4O7DRNGWGL7X/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/KD2PXSWUDUG5DE4O7DRNGWGL7X/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-10T21:18:44Z","links":{"resolver":"https://pith.science/pith/KD2PXSWUDUG5DE4O7DRNGWGL7X","bundle":"https://pith.science/pith/KD2PXSWUDUG5DE4O7DRNGWGL7X/bundle.json","state":"https://pith.science/pith/KD2PXSWUDUG5DE4O7DRNGWGL7X/state.json","well_known_bundle":"https://pith.science/.well-known/pith/KD2PXSWUDUG5DE4O7DRNGWGL7X/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:KD2PXSWUDUG5DE4O7DRNGWGL7X","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"615d5bae15c0a1748c6b84580490e718592312031f8a646901c376c8f4e61999","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-20T19:05:59Z","title_canon_sha256":"ae8567433809ca8a71bb0849a729231156550d20d434185b46caca454f3ba566"},"schema_version":"1.0","source":{"id":"2504.14716","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2504.14716","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"arxiv_version","alias_value":"2504.14716v2","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.14716","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"pith_short_12","alias_value":"KD2PXSWUDUG5","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"pith_short_16","alias_value":"KD2PXSWUDUG5DE4O","created_at":"2026-07-05T11:56:56Z"},{"alias_kind":"pith_short_8","alias_value":"KD2PXSWU","created_at":"2026-07-05T11:56:56Z"}],"graph_snapshots":[{"event_id":"sha256:92d75dc6484ca0cec1d96a1d0b6851cc8ac4168ea0c3fb9facca790b50157865","target":"graph","created_at":"2026-07-05T11:56:56Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2504.14716/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Large Language Models (LLMs) are widely used as proxies for human labelers in both training (Reinforcement Learning from AI Feedback) and large-scale response evaluation (LLM-as-a-judge). Alignment and evaluation are critical components in the development of reliable LLMs, and the choice of feedback protocol plays a central role in both but remains understudied. In this work, we show that the choice of feedback protocol for evaluation (absolute scores versus relative preferences) can significantly affect evaluation reliability and induce systematic biases. In the context of LLM-as-a-judge eval","authors_text":"Greg Durrett, Manya Wadhwa, Scott Niekum, Tuhina Tripathi","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-20T19:05:59Z","title":"Pairwise or Pointwise? Evaluating Feedback Protocols for Bias in LLM-Based Evaluation"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.14716","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:4bd141e695c850e6bc90c43d6f4ba0deacf9c6354e863a06c5b3eff64a02dd06","target":"record","created_at":"2026-07-05T11:56:56Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"615d5bae15c0a1748c6b84580490e718592312031f8a646901c376c8f4e61999","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-20T19:05:59Z","title_canon_sha256":"ae8567433809ca8a71bb0849a729231156550d20d434185b46caca454f3ba566"},"schema_version":"1.0","source":{"id":"2504.14716","kind":"arxiv","version":2}},"canonical_sha256":"50f4fbcad41d0dd1938ef8e2d358cbfdcc50b2390117e5d7076b775209736a01","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"50f4fbcad41d0dd1938ef8e2d358cbfdcc50b2390117e5d7076b775209736a01","first_computed_at":"2026-07-05T11:56:56.145448Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:56:56.145448Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"Re4SAM1DaowIk1rZxYynjatE4jD/b+PvlssGO/XGuwYJXEgjPn8l2AzXC/UzNHI61N/Pumo79zOBD7s26PyNBQ==","signature_status":"signed_v1","signed_at":"2026-07-05T11:56:56.145937Z","signed_message":"canonical_sha256_bytes"},"source_id":"2504.14716","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:4bd141e695c850e6bc90c43d6f4ba0deacf9c6354e863a06c5b3eff64a02dd06","sha256:92d75dc6484ca0cec1d96a1d0b6851cc8ac4168ea0c3fb9facca790b50157865"],"state_sha256":"302eae98e7619d1a39a2a56f4eb97637863e696101140bc20b242fdfaaf69f8e"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"PfqXafC+i+jluEqg2elPZMYSY0Wn12nAT7UxqUDRMS75z33N/IckoNpiUlFCw+ZBIasynh5VKHM7OidH5mQNDg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-10T21:18:44.255519Z","bundle_sha256":"73cb4967284343b4159c46b76f7103d98ecd7e67e02476939cdbee04ba4f31e7"}}