{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GVEIEBVXAZEFR4CIPTSNT5ZLWP","short_pith_number":"pith:GVEIEBVX","schema_version":"1.0","canonical_sha256":"35488206b7064858f0487ce4d9f72bb3d9075cad034d5a11c150514535a8f867","source":{"kind":"arxiv","id":"2402.14016","version":2},"attestation_state":"computed","paper":{"title":"Is LLM-as-a-Judge Robust? Investigating Universal Adversarial Attacks on Zero-shot LLM Assessment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Adian Liusie, Mark Gales, Vyas Raina","submitted_at":"2024-02-21T18:55:20Z","abstract_excerpt":"Large Language Models (LLMs) are powerful zero-shot assessors used in real-world situations such as assessing written exams and benchmarking systems. Despite these critical applications, no existing work has analyzed the vulnerability of judge-LLMs to adversarial manipulation. This work presents the first study on the adversarial robustness of assessment LLMs, where we demonstrate that short universal adversarial phrases can be concatenated to deceive judge LLMs to predict inflated scores. Since adversaries may not know or have access to the judge-LLMs, we propose a simple surrogate attack whe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.14016","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-21T18:55:20Z","cross_cats_sorted":[],"title_canon_sha256":"b424eba9ab8413e2ed7942d7ce1948cbb70616184bf45ac75d1392c9a4c0197e","abstract_canon_sha256":"c8ca5b779e1d8d79046f5c36437366e0ecd49f76aae17d8760c20064dc4e4698"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:40:13.852044Z","signature_b64":"nEIzQPM7l/Ek6adaxkj2/Ru0fTLAzHARSiuWkPBKIx2kcY3DkGPwcGPQQMQ9SGEuJ31Z91oUBeoCjlbwxacTCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"35488206b7064858f0487ce4d9f72bb3d9075cad034d5a11c150514535a8f867","last_reissued_at":"2026-07-05T08:40:13.851559Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:40:13.851559Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Is LLM-as-a-Judge Robust? Investigating Universal Adversarial Attacks on Zero-shot LLM Assessment","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Adian Liusie, Mark Gales, Vyas Raina","submitted_at":"2024-02-21T18:55:20Z","abstract_excerpt":"Large Language Models (LLMs) are powerful zero-shot assessors used in real-world situations such as assessing written exams and benchmarking systems. Despite these critical applications, no existing work has analyzed the vulnerability of judge-LLMs to adversarial manipulation. This work presents the first study on the adversarial robustness of assessment LLMs, where we demonstrate that short universal adversarial phrases can be concatenated to deceive judge LLMs to predict inflated scores. Since adversaries may not know or have access to the judge-LLMs, we propose a simple surrogate attack whe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.14016","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.14016/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.14016","created_at":"2026-07-05T08:40:13.851637+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.14016v2","created_at":"2026-07-05T08:40:13.851637+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.14016","created_at":"2026-07-05T08:40:13.851637+00:00"},{"alias_kind":"pith_short_12","alias_value":"GVEIEBVXAZEF","created_at":"2026-07-05T08:40:13.851637+00:00"},{"alias_kind":"pith_short_16","alias_value":"GVEIEBVXAZEFR4CI","created_at":"2026-07-05T08:40:13.851637+00:00"},{"alias_kind":"pith_short_8","alias_value":"GVEIEBVX","created_at":"2026-07-05T08:40:13.851637+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2411.15594","citing_title":"A Survey on LLM-as-a-Judge","ref_index":119,"is_internal_anchor":false},{"citing_arxiv_id":"2509.23542","citing_title":"On the Shelf Life of Fine-Tuned LLM-Judges: Future-Proofing, Backward-Compatibility, and Question Generalization","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2412.05579","citing_title":"LLMs-as-Judges: A Comprehensive Survey on LLM-based Evaluation Methods","ref_index":190,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23593","citing_title":"When AI reviews science: Can we trust the referee?","ref_index":111,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GVEIEBVXAZEFR4CIPTSNT5ZLWP","json":"https://pith.science/pith/GVEIEBVXAZEFR4CIPTSNT5ZLWP.json","graph_json":"https://pith.science/api/pith-number/GVEIEBVXAZEFR4CIPTSNT5ZLWP/graph.json","events_json":"https://pith.science/api/pith-number/GVEIEBVXAZEFR4CIPTSNT5ZLWP/events.json","paper":"https://pith.science/paper/GVEIEBVX"},"agent_actions":{"view_html":"https://pith.science/pith/GVEIEBVXAZEFR4CIPTSNT5ZLWP","download_json":"https://pith.science/pith/GVEIEBVXAZEFR4CIPTSNT5ZLWP.json","view_paper":"https://pith.science/paper/GVEIEBVX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.14016&json=true","fetch_graph":"https://pith.science/api/pith-number/GVEIEBVXAZEFR4CIPTSNT5ZLWP/graph.json","fetch_events":"https://pith.science/api/pith-number/GVEIEBVXAZEFR4CIPTSNT5ZLWP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GVEIEBVXAZEFR4CIPTSNT5ZLWP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GVEIEBVXAZEFR4CIPTSNT5ZLWP/action/storage_attestation","attest_author":"https://pith.science/pith/GVEIEBVXAZEFR4CIPTSNT5ZLWP/action/author_attestation","sign_citation":"https://pith.science/pith/GVEIEBVXAZEFR4CIPTSNT5ZLWP/action/citation_signature","submit_replication":"https://pith.science/pith/GVEIEBVXAZEFR4CIPTSNT5ZLWP/action/replication_record"}},"created_at":"2026-07-05T08:40:13.851637+00:00","updated_at":"2026-07-05T08:40:13.851637+00:00"}