{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2O5ASXQBAMSJRVJMXCM7GLBLFM","short_pith_number":"pith:2O5ASXQB","schema_version":"1.0","canonical_sha256":"d3ba095e01032498d52cb899f32c2b2b22679c4b43a7f72e3280d63919dba7e6","source":{"kind":"arxiv","id":"2509.08593","version":1},"attestation_state":"computed","paper":{"title":"No-Knowledge Alarms for Misaligned LLMs-as-Judges","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.AI","authors_text":"Andr\\'es Corrada-Emmanuel","submitted_at":"2025-09-10T13:46:40Z","abstract_excerpt":"If we use LLMs as judges to evaluate the complex decisions of other LLMs, who or what monitors the judges? Infinite monitoring chains are inevitable whenever we do not know the ground truth of the decisions by experts and we do not want to trust them. One way to ameliorate our evaluation uncertainty is to exploit the use of logical consistency between disagreeing experts. By observing how LLM judges agree and disagree while grading other LLMs, we can compute the only possible evaluations of their grading ability. For example, if two LLM judges disagree on which tasks a third one completed corr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.08593","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-09-10T13:46:40Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"02bff23f3df4ec33881b470bf1341efeebaae0d512978c1d0fab9578539c8941","abstract_canon_sha256":"62b414c0096c0b2a615ab82dd4503030d8f92b87b04fca35a4fae5b101c4330c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:08:32.732261Z","signature_b64":"O2TEUsb8DwO4uiA6yoLo9cSGakojFZHdLGiVEoBK/p7+R7Tf0OHa9MN3W6cuqEVX5XX+dGHGqbuS6hjS1tamCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d3ba095e01032498d52cb899f32c2b2b22679c4b43a7f72e3280d63919dba7e6","last_reissued_at":"2026-07-05T12:08:32.731762Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:08:32.731762Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"No-Knowledge Alarms for Misaligned LLMs-as-Judges","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.AI","authors_text":"Andr\\'es Corrada-Emmanuel","submitted_at":"2025-09-10T13:46:40Z","abstract_excerpt":"If we use LLMs as judges to evaluate the complex decisions of other LLMs, who or what monitors the judges? Infinite monitoring chains are inevitable whenever we do not know the ground truth of the decisions by experts and we do not want to trust them. One way to ameliorate our evaluation uncertainty is to exploit the use of logical consistency between disagreeing experts. By observing how LLM judges agree and disagree while grading other LLMs, we can compute the only possible evaluations of their grading ability. For example, if two LLM judges disagree on which tasks a third one completed corr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.08593","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.08593/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.08593","created_at":"2026-07-05T12:08:32.731823+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.08593v1","created_at":"2026-07-05T12:08:32.731823+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.08593","created_at":"2026-07-05T12:08:32.731823+00:00"},{"alias_kind":"pith_short_12","alias_value":"2O5ASXQBAMSJ","created_at":"2026-07-05T12:08:32.731823+00:00"},{"alias_kind":"pith_short_16","alias_value":"2O5ASXQBAMSJRVJM","created_at":"2026-07-05T12:08:32.731823+00:00"},{"alias_kind":"pith_short_8","alias_value":"2O5ASXQB","created_at":"2026-07-05T12:08:32.731823+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2O5ASXQBAMSJRVJMXCM7GLBLFM","json":"https://pith.science/pith/2O5ASXQBAMSJRVJMXCM7GLBLFM.json","graph_json":"https://pith.science/api/pith-number/2O5ASXQBAMSJRVJMXCM7GLBLFM/graph.json","events_json":"https://pith.science/api/pith-number/2O5ASXQBAMSJRVJMXCM7GLBLFM/events.json","paper":"https://pith.science/paper/2O5ASXQB"},"agent_actions":{"view_html":"https://pith.science/pith/2O5ASXQBAMSJRVJMXCM7GLBLFM","download_json":"https://pith.science/pith/2O5ASXQBAMSJRVJMXCM7GLBLFM.json","view_paper":"https://pith.science/paper/2O5ASXQB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.08593&json=true","fetch_graph":"https://pith.science/api/pith-number/2O5ASXQBAMSJRVJMXCM7GLBLFM/graph.json","fetch_events":"https://pith.science/api/pith-number/2O5ASXQBAMSJRVJMXCM7GLBLFM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2O5ASXQBAMSJRVJMXCM7GLBLFM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2O5ASXQBAMSJRVJMXCM7GLBLFM/action/storage_attestation","attest_author":"https://pith.science/pith/2O5ASXQBAMSJRVJMXCM7GLBLFM/action/author_attestation","sign_citation":"https://pith.science/pith/2O5ASXQBAMSJRVJMXCM7GLBLFM/action/citation_signature","submit_replication":"https://pith.science/pith/2O5ASXQBAMSJRVJMXCM7GLBLFM/action/replication_record"}},"created_at":"2026-07-05T12:08:32.731823+00:00","updated_at":"2026-07-05T12:08:32.731823+00:00"}