{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MBJESG56THWE3ZG6Z525G7B6N5","short_pith_number":"pith:MBJESG56","schema_version":"1.0","canonical_sha256":"6052491bbe99ec4de4decf75d37c3e6f5c4f4edcfa7217a78363bebff7c36999","source":{"kind":"arxiv","id":"2509.09912","version":1},"attestation_state":"computed","paper":{"title":"When Your Reviewer is an LLM: Biases, Divergence, and Prompt Injection Risks in Peer Review","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.CY","authors_text":"Changjia Zhu, Junjie Xiong, Lingyao Li, Renkai Ma, Yao Liu, Zhicong Lu","submitted_at":"2025-09-12T00:57:50Z","abstract_excerpt":"Peer review is the cornerstone of academic publishing, yet the process is increasingly strained by rising submission volumes, reviewer overload, and expertise mismatches. Large language models (LLMs) are now being used as \"reviewer aids,\" raising concerns about their fairness, consistency, and robustness against indirect prompt injection attacks. This paper presents a systematic evaluation of LLMs as academic reviewers. Using a curated dataset of 1,441 papers from ICLR 2023 and NeurIPS 2022, we evaluate GPT-5-mini against human reviewers across ratings, strengths, and weaknesses. The evaluatio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.09912","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CY","submitted_at":"2025-09-12T00:57:50Z","cross_cats_sorted":["cs.CR"],"title_canon_sha256":"f7cde426421b089aa7c73a8de879226de47964d9896d70f2545fafa4dc405b1b","abstract_canon_sha256":"790e88e013643903d93ac1f6cf19a97be69facc475e0b2f970369233ebce8708"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:09:52.054866Z","signature_b64":"YA4pFLnCbLYgVvkklH8shh+jkYuigCv8Lq8wJEVwxJoXHVCGp7RSEBvR4ZztuXJwys1GlKKgz/pliJvPMUQ4AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6052491bbe99ec4de4decf75d37c3e6f5c4f4edcfa7217a78363bebff7c36999","last_reissued_at":"2026-07-05T12:09:52.054367Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:09:52.054367Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Your Reviewer is an LLM: Biases, Divergence, and Prompt Injection Risks in Peer Review","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.CY","authors_text":"Changjia Zhu, Junjie Xiong, Lingyao Li, Renkai Ma, Yao Liu, Zhicong Lu","submitted_at":"2025-09-12T00:57:50Z","abstract_excerpt":"Peer review is the cornerstone of academic publishing, yet the process is increasingly strained by rising submission volumes, reviewer overload, and expertise mismatches. Large language models (LLMs) are now being used as \"reviewer aids,\" raising concerns about their fairness, consistency, and robustness against indirect prompt injection attacks. This paper presents a systematic evaluation of LLMs as academic reviewers. Using a curated dataset of 1,441 papers from ICLR 2023 and NeurIPS 2022, we evaluate GPT-5-mini against human reviewers across ratings, strengths, and weaknesses. The evaluatio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.09912","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.09912/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.09912","created_at":"2026-07-05T12:09:52.054430+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.09912v1","created_at":"2026-07-05T12:09:52.054430+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.09912","created_at":"2026-07-05T12:09:52.054430+00:00"},{"alias_kind":"pith_short_12","alias_value":"MBJESG56THWE","created_at":"2026-07-05T12:09:52.054430+00:00"},{"alias_kind":"pith_short_16","alias_value":"MBJESG56THWE3ZG6","created_at":"2026-07-05T12:09:52.054430+00:00"},{"alias_kind":"pith_short_8","alias_value":"MBJESG56","created_at":"2026-07-05T12:09:52.054430+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25057","citing_title":"LLM-Based Scientific Peer Review: Methods, Benchmarks, and Reliability Challenges","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13044","citing_title":"No Hidden Prompts Needed! You Can Game AI Peer Review with Presentation-Only Revisions","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18661","citing_title":"AI for Auto-Research: Roadmap & User Guide","ref_index":267,"is_internal_anchor":false},{"citing_arxiv_id":"2603.16659","citing_title":"LLMs learn scientific taste from institutional traces across the social sciences","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14162","citing_title":"Decoupling Scores and Text: The Politeness Principle in Peer Review","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MBJESG56THWE3ZG6Z525G7B6N5","json":"https://pith.science/pith/MBJESG56THWE3ZG6Z525G7B6N5.json","graph_json":"https://pith.science/api/pith-number/MBJESG56THWE3ZG6Z525G7B6N5/graph.json","events_json":"https://pith.science/api/pith-number/MBJESG56THWE3ZG6Z525G7B6N5/events.json","paper":"https://pith.science/paper/MBJESG56"},"agent_actions":{"view_html":"https://pith.science/pith/MBJESG56THWE3ZG6Z525G7B6N5","download_json":"https://pith.science/pith/MBJESG56THWE3ZG6Z525G7B6N5.json","view_paper":"https://pith.science/paper/MBJESG56","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.09912&json=true","fetch_graph":"https://pith.science/api/pith-number/MBJESG56THWE3ZG6Z525G7B6N5/graph.json","fetch_events":"https://pith.science/api/pith-number/MBJESG56THWE3ZG6Z525G7B6N5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MBJESG56THWE3ZG6Z525G7B6N5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MBJESG56THWE3ZG6Z525G7B6N5/action/storage_attestation","attest_author":"https://pith.science/pith/MBJESG56THWE3ZG6Z525G7B6N5/action/author_attestation","sign_citation":"https://pith.science/pith/MBJESG56THWE3ZG6Z525G7B6N5/action/citation_signature","submit_replication":"https://pith.science/pith/MBJESG56THWE3ZG6Z525G7B6N5/action/replication_record"}},"created_at":"2026-07-05T12:09:52.054430+00:00","updated_at":"2026-07-05T12:09:52.054430+00:00"}