{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:QTP4WLXP5DPZUTUUCTYUOFVWXS","short_pith_number":"pith:QTP4WLXP","schema_version":"1.0","canonical_sha256":"84dfcb2eefe8df9a4e9414f14716b6bcabb4b6fb7ed294aac94e6c6f4c5155eb","source":{"kind":"arxiv","id":"2506.09443","version":3},"attestation_state":"computed","paper":{"title":"LLMs Cannot Reliably Judge (Yet?): A Comprehensive Assessment on the Robustness of LLM-as-a-Judge","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Chen Chen, Chuokun Xu, Jiaying Wang, Jirui Zhang, Jun Wang, Kwok-Yan Lam, Shouling Ji, Songze Li, Xueluan Gong","submitted_at":"2025-06-11T06:48:57Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated exceptional capabilities across diverse tasks, driving the development and widespread adoption of LLM-as-a-Judge systems for automated evaluation, including red teaming and benchmarking. However, these systems are susceptible to adversarial attacks that can manipulate evaluation outcomes, raising critical concerns about their robustness and trustworthiness. Existing evaluation methods for LLM-based judges are often fragmented and lack a unified framework for comprehensive robustness assessment. Furthermore, the impact of prompt template design and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.09443","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2025-06-11T06:48:57Z","cross_cats_sorted":[],"title_canon_sha256":"56ff6802a4b4f66a1f57dd3bb610e796839fb5b18e87429dc6287904fec4544c","abstract_canon_sha256":"b2279eb1511fa93da8089a7740e9e387e9b4c5e19e018e0f0b32220b960873b1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-07T00:51:05.940218Z","signature_b64":"SDucXdi6MZWzch/v0Bo5fuw25oS1bHrNdlJCn6gVkgL19+MokZmoDwZgVde6rCql7mMWX/lHRR4UhYGTec+fCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"84dfcb2eefe8df9a4e9414f14716b6bcabb4b6fb7ed294aac94e6c6f4c5155eb","last_reissued_at":"2026-08-07T00:51:05.938652Z","signature_status":"signed_v1","first_computed_at":"2026-08-07T00:51:05.938652Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLMs Cannot Reliably Judge (Yet?): A Comprehensive Assessment on the Robustness of LLM-as-a-Judge","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Chen Chen, Chuokun Xu, Jiaying Wang, Jirui Zhang, Jun Wang, Kwok-Yan Lam, Shouling Ji, Songze Li, Xueluan Gong","submitted_at":"2025-06-11T06:48:57Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated exceptional capabilities across diverse tasks, driving the development and widespread adoption of LLM-as-a-Judge systems for automated evaluation, including red teaming and benchmarking. However, these systems are susceptible to adversarial attacks that can manipulate evaluation outcomes, raising critical concerns about their robustness and trustworthiness. Existing evaluation methods for LLM-based judges are often fragmented and lack a unified framework for comprehensive robustness assessment. Furthermore, the impact of prompt template design and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.09443","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.09443/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.09443","created_at":"2026-08-07T00:51:05.940144+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.09443v3","created_at":"2026-08-07T00:51:05.940144+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.09443","created_at":"2026-08-07T00:51:05.940144+00:00"},{"alias_kind":"pith_short_12","alias_value":"QTP4WLXP5DPZ","created_at":"2026-08-07T00:51:05.940144+00:00"},{"alias_kind":"pith_short_16","alias_value":"QTP4WLXP5DPZUTUU","created_at":"2026-08-07T00:51:05.940144+00:00"},{"alias_kind":"pith_short_8","alias_value":"QTP4WLXP","created_at":"2026-08-07T00:51:05.940144+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":9,"sample":[{"citing_arxiv_id":"2607.07989","citing_title":"Who Broke the System? Failure Localization in LLM-Based Multi-Agent Systems","ref_index":4,"is_internal_anchor":true},{"citing_arxiv_id":"2606.13044","citing_title":"No Hidden Prompts Needed! You Can Game AI Peer Review with Presentation-Only Revisions","ref_index":15,"is_internal_anchor":true},{"citing_arxiv_id":"2606.09711","citing_title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","ref_index":138,"is_internal_anchor":true},{"citing_arxiv_id":"2605.21748","citing_title":"RankJudge: A Multi-Turn LLM-as-a-Judge Synthetic Benchmark Generator","ref_index":40,"is_internal_anchor":true},{"citing_arxiv_id":"2605.03858","citing_title":"MCJudgeBench: A Benchmark for Constraint-Level Judge Evaluation in Multi-Constraint Instruction Following","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2605.06161","citing_title":"Beyond Accuracy: Policy Invariance as a Reliability Test for LLM Safety Judges","ref_index":25,"is_internal_anchor":true},{"citing_arxiv_id":"2604.22937","citing_title":"AutoPyVerifier: Learning Compact Executable Verifiers for Large Language Model Outputs","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2604.05593","citing_title":"Label Effects: Shared Heuristic Reliance in Trust Assessment by Humans and LLM-as-a-Judge","ref_index":27,"is_internal_anchor":true},{"citing_arxiv_id":"2604.13602","citing_title":"Reward Hacking in the Era of Large Models: Mechanisms, Emergent Misalignment, Challenges","ref_index":42,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QTP4WLXP5DPZUTUUCTYUOFVWXS","json":"https://pith.science/pith/QTP4WLXP5DPZUTUUCTYUOFVWXS.json","graph_json":"https://pith.science/api/pith-number/QTP4WLXP5DPZUTUUCTYUOFVWXS/graph.json","events_json":"https://pith.science/api/pith-number/QTP4WLXP5DPZUTUUCTYUOFVWXS/events.json","paper":"https://pith.science/paper/QTP4WLXP"},"agent_actions":{"view_html":"https://pith.science/pith/QTP4WLXP5DPZUTUUCTYUOFVWXS","download_json":"https://pith.science/pith/QTP4WLXP5DPZUTUUCTYUOFVWXS.json","view_paper":"https://pith.science/paper/QTP4WLXP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.09443&json=true","fetch_graph":"https://pith.science/api/pith-number/QTP4WLXP5DPZUTUUCTYUOFVWXS/graph.json","fetch_events":"https://pith.science/api/pith-number/QTP4WLXP5DPZUTUUCTYUOFVWXS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QTP4WLXP5DPZUTUUCTYUOFVWXS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QTP4WLXP5DPZUTUUCTYUOFVWXS/action/storage_attestation","attest_author":"https://pith.science/pith/QTP4WLXP5DPZUTUUCTYUOFVWXS/action/author_attestation","sign_citation":"https://pith.science/pith/QTP4WLXP5DPZUTUUCTYUOFVWXS/action/citation_signature","submit_replication":"https://pith.science/pith/QTP4WLXP5DPZUTUUCTYUOFVWXS/action/replication_record"}},"created_at":"2026-08-07T00:51:05.940144+00:00","updated_at":"2026-08-07T00:51:05.940144+00:00"}