{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:WBJQQUX3HD72QTWS7VASMO66AG","short_pith_number":"pith:WBJQQUX3","schema_version":"1.0","canonical_sha256":"b0530852fb38ffa84ed2fd41263bde018d19344a5042cc9e49b4888cce6723a8","source":{"kind":"arxiv","id":"2604.06820","version":2},"attestation_state":"computed","paper":{"title":"When Direct Prediction Fails: Evidence from LLM-Based Misinformation Risk Evaluation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"LLM judges for disinformation risk form a coherent group more aligned with each other than with human readers.","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Xiang Zheng, Xingjun Ma, Yutao Wu, Zonghuan Xu","submitted_at":"2026-04-08T08:37:37Z","abstract_excerpt":"LLMs make it increasingly easy to generate deceptive content at scale, creating a need for scalable misinformation risk evaluation based on whether readers find such content credible and are willing to share it. A natural approach is to ask an LLM these questions directly and treat the returned scores as predictions of the corresponding human ratings. Implicit in this practice is the assumption that asking about a reader response produces the score that best predicts it. We test this assumption using matched credibility and willingness-to-share ratings for 290 deceptive articles from 317 parti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":true},"canonical_record":{"source":{"id":"2604.06820","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.AI","submitted_at":"2026-04-08T08:37:37Z","cross_cats_sorted":[],"title_canon_sha256":"5d8fb95f1540085c827af0c285ae038f5361df83fbde2316699f8f5ac510049c","abstract_canon_sha256":"52307bb2a2247e3232a674d0d2b56dbad9311f3bad63aa527951efe2f6d62553"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-21T00:19:57.371942Z","signature_b64":"qMb6C6I4FGG2rd6z+adUixA8u+AyvfM3WQo7ZZTFbB4VAvIRdNlW55Egpia04nmW+E3oIWOhGKNWlF1pbQA0Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b0530852fb38ffa84ed2fd41263bde018d19344a5042cc9e49b4888cce6723a8","last_reissued_at":"2026-07-21T00:19:57.370939Z","signature_status":"signed_v1","first_computed_at":"2026-07-21T00:19:57.370939Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"When Direct Prediction Fails: Evidence from LLM-Based Misinformation Risk Evaluation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"LLM judges for disinformation risk form a coherent group more aligned with each other than with human readers.","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Xiang Zheng, Xingjun Ma, Yutao Wu, Zonghuan Xu","submitted_at":"2026-04-08T08:37:37Z","abstract_excerpt":"LLMs make it increasingly easy to generate deceptive content at scale, creating a need for scalable misinformation risk evaluation based on whether readers find such content credible and are willing to share it. A natural approach is to ask an LLM these questions directly and treat the returned scores as predictions of the corresponding human ratings. Implicit in this practice is the assumption that asking about a reader response produces the score that best predicts it. We test this assumption using matched credibility and willingness-to-share ratings for 290 deceptive articles from 317 parti"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"These results suggest that LLM judges form a coherent evaluative group that is much more aligned internally than it is with human readers, indicating that internal agreement is not evidence of validity as a proxy for reader response.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the collected human ratings on the 290 articles constitute a reliable and representative ground truth for how readers actually perceive disinformation risk.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"LLM judges for disinformation risk are harsher than humans, weakly match human rankings, rely on different textual signals, and agree far more with each other than with readers.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"LLM judges for disinformation risk form a coherent group more aligned with each other than with human readers.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"9f2fe01adc9ca758e819c982180d5ed1376ce75e073d15748fb88d600106a51b"},"source":{"id":"2604.06820","kind":"arxiv","version":2},"verdict":{"id":"ae5769b3-312b-47b2-80c2-e7c600f0f162","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-10T17:54:19.084061Z","strongest_claim":"These results suggest that LLM judges form a coherent evaluative group that is much more aligned internally than it is with human readers, indicating that internal agreement is not evidence of validity as a proxy for reader response.","one_line_summary":"LLM judges for disinformation risk are harsher than humans, weakly match human rankings, rely on different textual signals, and agree far more with each other than with readers.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the collected human ratings on the 290 articles constitute a reliable and representative ground truth for how readers actually perceive disinformation risk.","pith_extraction_headline":"LLM judges for disinformation risk form a coherent group more aligned with each other than with human readers."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.06820/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":2,"snapshot_sha256":"a8598502b8e797701a207feeeff55b93e26c39c5668ac0605ed1492a5778be0a"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2604.06820","created_at":"2026-07-21T00:19:57.371414+00:00"},{"alias_kind":"arxiv_version","alias_value":"2604.06820v2","created_at":"2026-07-21T00:19:57.371414+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.06820","created_at":"2026-07-21T00:19:57.371414+00:00"},{"alias_kind":"pith_short_12","alias_value":"WBJQQUX3HD72","created_at":"2026-07-21T00:19:57.371414+00:00"},{"alias_kind":"pith_short_16","alias_value":"WBJQQUX3HD72QTWS","created_at":"2026-07-21T00:19:57.371414+00:00"},{"alias_kind":"pith_short_8","alias_value":"WBJQQUX3","created_at":"2026-07-21T00:19:57.371414+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":2,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WBJQQUX3HD72QTWS7VASMO66AG","json":"https://pith.science/pith/WBJQQUX3HD72QTWS7VASMO66AG.json","graph_json":"https://pith.science/api/pith-number/WBJQQUX3HD72QTWS7VASMO66AG/graph.json","events_json":"https://pith.science/api/pith-number/WBJQQUX3HD72QTWS7VASMO66AG/events.json","paper":"https://pith.science/paper/WBJQQUX3"},"agent_actions":{"view_html":"https://pith.science/pith/WBJQQUX3HD72QTWS7VASMO66AG","download_json":"https://pith.science/pith/WBJQQUX3HD72QTWS7VASMO66AG.json","view_paper":"https://pith.science/paper/WBJQQUX3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2604.06820&json=true","fetch_graph":"https://pith.science/api/pith-number/WBJQQUX3HD72QTWS7VASMO66AG/graph.json","fetch_events":"https://pith.science/api/pith-number/WBJQQUX3HD72QTWS7VASMO66AG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WBJQQUX3HD72QTWS7VASMO66AG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WBJQQUX3HD72QTWS7VASMO66AG/action/storage_attestation","attest_author":"https://pith.science/pith/WBJQQUX3HD72QTWS7VASMO66AG/action/author_attestation","sign_citation":"https://pith.science/pith/WBJQQUX3HD72QTWS7VASMO66AG/action/citation_signature","submit_replication":"https://pith.science/pith/WBJQQUX3HD72QTWS7VASMO66AG/action/replication_record"}},"created_at":"2026-07-21T00:19:57.371414+00:00","updated_at":"2026-07-21T00:19:57.371414+00:00"}