{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:IKM2OKOKWJDIUWFTNUHDV2ZZ5K","short_pith_number":"pith:IKM2OKOK","canonical_record":{"source":{"id":"2605.03202","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-04T22:41:04Z","cross_cats_sorted":[],"title_canon_sha256":"1001f9fe27d973c8ea9ec16fadea28804c129a8ab2132d00c8359562aee99e0b","abstract_canon_sha256":"60596239c946c171d26d9f20edc37408e9605f3f9442188d521aed1672b760c5"},"schema_version":"1.0"},"canonical_sha256":"4299a729cab2468a58b36d0e3aeb39ea9bf2a31ea73af8d98ebf01cb9ecd2b10","source":{"kind":"arxiv","id":"2605.03202","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2605.03202","created_at":"2026-07-07T02:18:41Z"},{"alias_kind":"arxiv_version","alias_value":"2605.03202v2","created_at":"2026-07-07T02:18:41Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.03202","created_at":"2026-07-07T02:18:41Z"},{"alias_kind":"pith_short_12","alias_value":"IKM2OKOKWJDI","created_at":"2026-07-07T02:18:41Z"},{"alias_kind":"pith_short_16","alias_value":"IKM2OKOKWJDIUWFT","created_at":"2026-07-07T02:18:41Z"},{"alias_kind":"pith_short_8","alias_value":"IKM2OKOK","created_at":"2026-07-07T02:18:41Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:IKM2OKOKWJDIUWFTNUHDV2ZZ5K","target":"record","payload":{"canonical_record":{"source":{"id":"2605.03202","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-04T22:41:04Z","cross_cats_sorted":[],"title_canon_sha256":"1001f9fe27d973c8ea9ec16fadea28804c129a8ab2132d00c8359562aee99e0b","abstract_canon_sha256":"60596239c946c171d26d9f20edc37408e9605f3f9442188d521aed1672b760c5"},"schema_version":"1.0"},"canonical_sha256":"4299a729cab2468a58b36d0e3aeb39ea9bf2a31ea73af8d98ebf01cb9ecd2b10","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-07T02:18:41.981687Z","signature_b64":"fxa0rnVttbeIvyQIwRRSqgjYkOUyzBdjep3/gTopqQJsWadsxMEPoZe/DqtFyY8mfpOpZgqpyCrc0Zj+SjTcCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4299a729cab2468a58b36d0e3aeb39ea9bf2a31ea73af8d98ebf01cb9ecd2b10","last_reissued_at":"2026-07-07T02:18:41.980927Z","signature_status":"signed_v1","first_computed_at":"2026-07-07T02:18:41.980927Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2605.03202","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-07T02:18:41Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"acJfy2wVy63uNSCVynjeXxMZtM6SrfB1A4nnwA+XgRQqF11kHv/eGCapNhFcOhDIPaHnpI4+TIeOgFWQHH5FBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T16:06:13.068692Z"},"content_sha256":"892fd82e713cda623e12895ee8c16fd1b7289d9b7b54a02abfa6cd9582644125","schema_version":"1.0","event_id":"sha256:892fd82e713cda623e12895ee8c16fd1b7289d9b7b54a02abfa6cd9582644125"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:IKM2OKOKWJDIUWFTNUHDV2ZZ5K","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Stop Automating Peer Review Without Rigorous Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"AI systems should not generate peer reviews today because they show excessive agreement and are easily gamed by stylistic rewrites.","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Dirk Hovy, Jiaxin Pei, Joachim Baumann, Sanmi Koyejo","submitted_at":"2026-05-04T22:41:04Z","abstract_excerpt":"Large language models offer a tempting solution to address the peer review crisis. This position paper argues that today's AI systems should not be used to produce paper reviews. We ground this position in an empirical comparison of human- versus AI-generated ICLR 2026 reviews and an evaluation of the effect of automated paper rewriting on different AI reviewers. We identify two critical issues: 1) AI reviewers exhibit a hivemind effect of excessive agreement within and across papers that reduces perspective diversity. 2) AI review scores are trivially gameable through paper laundering: prompt"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"AI reviewers exhibit a hivemind effect of excessive agreement within and across papers that reduces perspective diversity. AI review scores are trivially gameable through paper laundering: prompting an LLM to rewrite a paper could significantly increase the scores from AI reviewers.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the ICLR 2026 sample and the specific AI models tested are representative of broader peer review contexts, and that the LLM rewriting preserves scientific content without introducing legitimate improvements that would justify higher scores.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"AI peer reviewers show excessive agreement across papers and give higher scores after simple LLM-based stylistic rewriting, so general-purpose LLMs should not automate reviews without rigorous evaluation.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"AI systems should not generate peer reviews today because they show excessive agreement and are easily gamed by stylistic rewrites.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"ab663ce0e5eaa6f0b4a37d7ffbc985660528f466288a9d994cc2db1d5ee390b0"},"source":{"id":"2605.03202","kind":"arxiv","version":2},"verdict":{"id":"f73c7e65-e6b7-426e-aab7-3bce1c353989","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-08T17:56:20.421206Z","strongest_claim":"AI reviewers exhibit a hivemind effect of excessive agreement within and across papers that reduces perspective diversity. AI review scores are trivially gameable through paper laundering: prompting an LLM to rewrite a paper could significantly increase the scores from AI reviewers.","one_line_summary":"AI peer reviewers show excessive agreement across papers and give higher scores after simple LLM-based stylistic rewriting, so general-purpose LLMs should not automate reviews without rigorous evaluation.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the ICLR 2026 sample and the specific AI models tested are representative of broader peer review contexts, and that the LLM rewriting preserves scientific content without introducing legitimate improvements that would justify higher scores.","pith_extraction_headline":"AI systems should not generate peer reviews today because they show excessive agreement and are easily gamed by stylistic rewrites."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2605.03202/integrity.json","findings":[],"available":true,"detectors_run":[{"name":"ai_meta_artifact","ran_at":"2026-05-20T14:35:39.027643Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_title_agreement","ran_at":"2026-05-20T01:31:21.935487Z","status":"completed","version":"1.0.0","findings_count":0},{"name":"doi_compliance","ran_at":"2026-05-19T15:35:06.967712Z","status":"completed","version":"1.0.0","findings_count":0}],"snapshot_sha256":"3c8a40ed0e37d89eb19c01ba545c09ce54ae54c0f2494d3ee58e39c41c50adde"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":2,"snapshot_sha256":"4c47a536b147f000364ef5d3905eed75920b534ea0c5fa2db77e25eb10168009"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"f73c7e65-e6b7-426e-aab7-3bce1c353989"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-07T02:18:41Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"HXuWyrkBGDIwCSW4KsxMchCsH8fNBcxjp4TfnuhfvFoqwhYJKF+kdf2CCChKU1Q8OUpF1EleXc5262seUmcbDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T16:06:13.069806Z"},"content_sha256":"5f760a92aa8ae3795f817c6708bcac33375dd7d055490da259b3bb00633349f9","schema_version":"1.0","event_id":"sha256:5f760a92aa8ae3795f817c6708bcac33375dd7d055490da259b3bb00633349f9"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/IKM2OKOKWJDIUWFTNUHDV2ZZ5K/bundle.json","state_url":"https://pith.science/pith/IKM2OKOKWJDIUWFTNUHDV2ZZ5K/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/IKM2OKOKWJDIUWFTNUHDV2ZZ5K/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T16:06:13Z","links":{"resolver":"https://pith.science/pith/IKM2OKOKWJDIUWFTNUHDV2ZZ5K","bundle":"https://pith.science/pith/IKM2OKOKWJDIUWFTNUHDV2ZZ5K/bundle.json","state":"https://pith.science/pith/IKM2OKOKWJDIUWFTNUHDV2ZZ5K/state.json","well_known_bundle":"https://pith.science/.well-known/pith/IKM2OKOKWJDIUWFTNUHDV2ZZ5K/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:IKM2OKOKWJDIUWFTNUHDV2ZZ5K","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"60596239c946c171d26d9f20edc37408e9605f3f9442188d521aed1672b760c5","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-04T22:41:04Z","title_canon_sha256":"1001f9fe27d973c8ea9ec16fadea28804c129a8ab2132d00c8359562aee99e0b"},"schema_version":"1.0","source":{"id":"2605.03202","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2605.03202","created_at":"2026-07-07T02:18:41Z"},{"alias_kind":"arxiv_version","alias_value":"2605.03202v2","created_at":"2026-07-07T02:18:41Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2605.03202","created_at":"2026-07-07T02:18:41Z"},{"alias_kind":"pith_short_12","alias_value":"IKM2OKOKWJDI","created_at":"2026-07-07T02:18:41Z"},{"alias_kind":"pith_short_16","alias_value":"IKM2OKOKWJDIUWFT","created_at":"2026-07-07T02:18:41Z"},{"alias_kind":"pith_short_8","alias_value":"IKM2OKOK","created_at":"2026-07-07T02:18:41Z"}],"graph_snapshots":[{"event_id":"sha256:5f760a92aa8ae3795f817c6708bcac33375dd7d055490da259b3bb00633349f9","target":"graph","created_at":"2026-07-07T02:18:41Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"AI reviewers exhibit a hivemind effect of excessive agreement within and across papers that reduces perspective diversity. AI review scores are trivially gameable through paper laundering: prompting an LLM to rewrite a paper could significantly increase the scores from AI reviewers."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That the ICLR 2026 sample and the specific AI models tested are representative of broader peer review contexts, and that the LLM rewriting preserves scientific content without introducing legitimate improvements that would justify higher scores."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"AI peer reviewers show excessive agreement across papers and give higher scores after simple LLM-based stylistic rewriting, so general-purpose LLMs should not automate reviews without rigorous evaluation."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"AI systems should not generate peer reviews today because they show excessive agreement and are easily gamed by stylistic rewrites."}],"snapshot_sha256":"ab663ce0e5eaa6f0b4a37d7ffbc985660528f466288a9d994cc2db1d5ee390b0"},"formal_canon":{"evidence_count":2,"snapshot_sha256":"4c47a536b147f000364ef5d3905eed75920b534ea0c5fa2db77e25eb10168009"},"integrity":{"available":true,"clean":true,"detectors_run":[{"findings_count":0,"name":"ai_meta_artifact","ran_at":"2026-05-20T14:35:39.027643Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_title_agreement","ran_at":"2026-05-20T01:31:21.935487Z","status":"completed","version":"1.0.0"},{"findings_count":0,"name":"doi_compliance","ran_at":"2026-05-19T15:35:06.967712Z","status":"completed","version":"1.0.0"}],"endpoint":"/pith/2605.03202/integrity.json","findings":[],"snapshot_sha256":"3c8a40ed0e37d89eb19c01ba545c09ce54ae54c0f2494d3ee58e39c41c50adde","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Large language models offer a tempting solution to address the peer review crisis. This position paper argues that today's AI systems should not be used to produce paper reviews. We ground this position in an empirical comparison of human- versus AI-generated ICLR 2026 reviews and an evaluation of the effect of automated paper rewriting on different AI reviewers. We identify two critical issues: 1) AI reviewers exhibit a hivemind effect of excessive agreement within and across papers that reduces perspective diversity. 2) AI review scores are trivially gameable through paper laundering: prompt","authors_text":"Dirk Hovy, Jiaxin Pei, Joachim Baumann, Sanmi Koyejo","cross_cats":[],"headline":"AI systems should not generate peer reviews today because they show excessive agreement and are easily gamed by stylistic rewrites.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-04T22:41:04Z","title":"Stop Automating Peer Review Without Rigorous Evaluation"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2605.03202","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-08T17:56:20.421206Z","id":"f73c7e65-e6b7-426e-aab7-3bce1c353989","model_set":{"reader":"grok-4.3"},"one_line_summary":"AI peer reviewers show excessive agreement across papers and give higher scores after simple LLM-based stylistic rewriting, so general-purpose LLMs should not automate reviews without rigorous evaluation.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"AI systems should not generate peer reviews today because they show excessive agreement and are easily gamed by stylistic rewrites.","strongest_claim":"AI reviewers exhibit a hivemind effect of excessive agreement within and across papers that reduces perspective diversity. AI review scores are trivially gameable through paper laundering: prompting an LLM to rewrite a paper could significantly increase the scores from AI reviewers.","weakest_assumption":"That the ICLR 2026 sample and the specific AI models tested are representative of broader peer review contexts, and that the LLM rewriting preserves scientific content without introducing legitimate improvements that would justify higher scores."}},"verdict_id":"f73c7e65-e6b7-426e-aab7-3bce1c353989"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:892fd82e713cda623e12895ee8c16fd1b7289d9b7b54a02abfa6cd9582644125","target":"record","created_at":"2026-07-07T02:18:41Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"60596239c946c171d26d9f20edc37408e9605f3f9442188d521aed1672b760c5","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-05-04T22:41:04Z","title_canon_sha256":"1001f9fe27d973c8ea9ec16fadea28804c129a8ab2132d00c8359562aee99e0b"},"schema_version":"1.0","source":{"id":"2605.03202","kind":"arxiv","version":2}},"canonical_sha256":"4299a729cab2468a58b36d0e3aeb39ea9bf2a31ea73af8d98ebf01cb9ecd2b10","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"4299a729cab2468a58b36d0e3aeb39ea9bf2a31ea73af8d98ebf01cb9ecd2b10","first_computed_at":"2026-07-07T02:18:41.980927Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-07T02:18:41.980927Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"fxa0rnVttbeIvyQIwRRSqgjYkOUyzBdjep3/gTopqQJsWadsxMEPoZe/DqtFyY8mfpOpZgqpyCrc0Zj+SjTcCA==","signature_status":"signed_v1","signed_at":"2026-07-07T02:18:41.981687Z","signed_message":"canonical_sha256_bytes"},"source_id":"2605.03202","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:892fd82e713cda623e12895ee8c16fd1b7289d9b7b54a02abfa6cd9582644125","sha256:5f760a92aa8ae3795f817c6708bcac33375dd7d055490da259b3bb00633349f9"],"state_sha256":"8616564f07e67a7775335fc1ecd32904fb39c8fef864a4b9953497960957b7fa"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"+EV0JJnUE8wxW4AnQajFpLNisZJhr3as0ABsWxwUCX0wBAZgnPRWrbRAHqxzcnGamtYZ+SwJsURagDovUcDyDQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T16:06:13.077257Z","bundle_sha256":"ce38b7507c72fb0bc61f730b6f0950a13634ba8c698fe638420562e4c618c36f"}}