{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:SZVPL4B26BXSYIVUEJGTRARVK3","short_pith_number":"pith:SZVPL4B2","schema_version":"1.0","canonical_sha256":"966af5f03af06f2c22b4224d38823556f3dc67bb117ec24d845cfb8e98047c60","source":{"kind":"arxiv","id":"2306.09983","version":3},"attestation_state":"computed","paper":{"title":"Evaluating Superhuman Models with Consistency Checks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","stat.ML"],"primary_cat":"cs.LG","authors_text":"Daniel Paleka, Florian Tram\\`er, Lukas Fluri","submitted_at":"2023-06-16T17:26:38Z","abstract_excerpt":"If machine learning models were to achieve superhuman abilities at various reasoning or decision-making tasks, how would we go about evaluating such models, given that humans would necessarily be poor proxies for ground truth? In this paper, we propose a framework for evaluating superhuman models via consistency checks. Our premise is that while the correctness of superhuman decisions may be impossible to evaluate, we can still surface mistakes if the model's decisions fail to satisfy certain logical, human-interpretable rules. We instantiate our framework on three tasks where correctness of d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.09983","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-16T17:26:38Z","cross_cats_sorted":["cs.AI","cs.CR","stat.ML"],"title_canon_sha256":"1a8da41745aa07e05f8329a711ba1076bf47ade812898be1c6d731610e2a0a21","abstract_canon_sha256":"b07f8000eb30e426ffcc903ad807b93d55ec67bc463c10874801428e338aabae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:02:30.407135Z","signature_b64":"CFP0jL70+ll+lxrSHa0E9wqz0m0KjENUHsdsr/RzjIdjIMHYNxw9IzsxDc+Aoqy5Qeiw/jtKgtguwtqnf4F4DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"966af5f03af06f2c22b4224d38823556f3dc67bb117ec24d845cfb8e98047c60","last_reissued_at":"2026-07-05T07:02:30.406683Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:02:30.406683Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Superhuman Models with Consistency Checks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","stat.ML"],"primary_cat":"cs.LG","authors_text":"Daniel Paleka, Florian Tram\\`er, Lukas Fluri","submitted_at":"2023-06-16T17:26:38Z","abstract_excerpt":"If machine learning models were to achieve superhuman abilities at various reasoning or decision-making tasks, how would we go about evaluating such models, given that humans would necessarily be poor proxies for ground truth? In this paper, we propose a framework for evaluating superhuman models via consistency checks. Our premise is that while the correctness of superhuman decisions may be impossible to evaluate, we can still surface mistakes if the model's decisions fail to satisfy certain logical, human-interpretable rules. We instantiate our framework on three tasks where correctness of d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.09983","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.09983/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.09983","created_at":"2026-07-05T07:02:30.406748+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.09983v3","created_at":"2026-07-05T07:02:30.406748+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.09983","created_at":"2026-07-05T07:02:30.406748+00:00"},{"alias_kind":"pith_short_12","alias_value":"SZVPL4B26BXS","created_at":"2026-07-05T07:02:30.406748+00:00"},{"alias_kind":"pith_short_16","alias_value":"SZVPL4B26BXSYIVU","created_at":"2026-07-05T07:02:30.406748+00:00"},{"alias_kind":"pith_short_8","alias_value":"SZVPL4B2","created_at":"2026-07-05T07:02:30.406748+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2405.02079","citing_title":"Argumentative Large Language Models for Explainable and Contestable Claim Verification","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12809","citing_title":"Correcting Influence: Unboxing LLM Outputs with Orthogonal Latent Spaces","ref_index":213,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SZVPL4B26BXSYIVUEJGTRARVK3","json":"https://pith.science/pith/SZVPL4B26BXSYIVUEJGTRARVK3.json","graph_json":"https://pith.science/api/pith-number/SZVPL4B26BXSYIVUEJGTRARVK3/graph.json","events_json":"https://pith.science/api/pith-number/SZVPL4B26BXSYIVUEJGTRARVK3/events.json","paper":"https://pith.science/paper/SZVPL4B2"},"agent_actions":{"view_html":"https://pith.science/pith/SZVPL4B26BXSYIVUEJGTRARVK3","download_json":"https://pith.science/pith/SZVPL4B26BXSYIVUEJGTRARVK3.json","view_paper":"https://pith.science/paper/SZVPL4B2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.09983&json=true","fetch_graph":"https://pith.science/api/pith-number/SZVPL4B26BXSYIVUEJGTRARVK3/graph.json","fetch_events":"https://pith.science/api/pith-number/SZVPL4B26BXSYIVUEJGTRARVK3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SZVPL4B26BXSYIVUEJGTRARVK3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SZVPL4B26BXSYIVUEJGTRARVK3/action/storage_attestation","attest_author":"https://pith.science/pith/SZVPL4B26BXSYIVUEJGTRARVK3/action/author_attestation","sign_citation":"https://pith.science/pith/SZVPL4B26BXSYIVUEJGTRARVK3/action/citation_signature","submit_replication":"https://pith.science/pith/SZVPL4B26BXSYIVUEJGTRARVK3/action/replication_record"}},"created_at":"2026-07-05T07:02:30.406748+00:00","updated_at":"2026-07-05T07:02:30.406748+00:00"}