{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XZTV2VEW3CKQ7HWT3TVGIB7LSO","short_pith_number":"pith:XZTV2VEW","schema_version":"1.0","canonical_sha256":"be675d5496d8950f9ed3dcea6407eb93ba892bc2b9fad2d8c448fcc771c2fb4f","source":{"kind":"arxiv","id":"2505.05602","version":3},"attestation_state":"computed","paper":{"title":"HiBayES: A Hierarchical Bayesian Modeling Framework for AI Evaluation Statistics","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["stat.AP"],"primary_cat":"cs.AI","authors_text":"Christopher Summerfield, Cozmin Ududec, Harry Coppock, Lennart Luettgau, Magda Dubois","submitted_at":"2025-05-08T19:05:02Z","abstract_excerpt":"As Large Language Models (LLMs) and other AI systems evolve, robustly estimating their capabilities from inherently stochastic outputs while systematically quantifying uncertainty in these estimates becomes increasingly important. Further, advanced AI evaluations often have a nested hierarchical structure, exhibit high levels of complexity, and come with high costs in testing the most advanced AI systems. To address these challenges, we introduce HiBayES, a generalizable Hierarchical Bayesian modeling framework for AI Evaluation Statistics. HiBayES supports robust inferences in classical quest"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.05602","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.AI","submitted_at":"2025-05-08T19:05:02Z","cross_cats_sorted":["stat.AP"],"title_canon_sha256":"fac76d35572d9330f6dd8bab97064b5f97f5d795d0f5caf0af9a7d1a8613a649","abstract_canon_sha256":"85034b22cd3140e690a648265baefffbef845ec34608d26c2ff870c8242b829d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:36:31.575799Z","signature_b64":"m64yPI6P0hROOKKmAEz2PEDENDW07Om9rKydeDnMd8mCsXTtj8QpDXuUzCQznpMmvvpIk47sYFp3Y70H8AZ9CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"be675d5496d8950f9ed3dcea6407eb93ba892bc2b9fad2d8c448fcc771c2fb4f","last_reissued_at":"2026-07-05T11:36:31.575121Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:36:31.575121Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HiBayES: A Hierarchical Bayesian Modeling Framework for AI Evaluation Statistics","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["stat.AP"],"primary_cat":"cs.AI","authors_text":"Christopher Summerfield, Cozmin Ududec, Harry Coppock, Lennart Luettgau, Magda Dubois","submitted_at":"2025-05-08T19:05:02Z","abstract_excerpt":"As Large Language Models (LLMs) and other AI systems evolve, robustly estimating their capabilities from inherently stochastic outputs while systematically quantifying uncertainty in these estimates becomes increasingly important. Further, advanced AI evaluations often have a nested hierarchical structure, exhibit high levels of complexity, and come with high costs in testing the most advanced AI systems. To address these challenges, we introduce HiBayES, a generalizable Hierarchical Bayesian modeling framework for AI Evaluation Statistics. HiBayES supports robust inferences in classical quest"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.05602","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.05602/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.05602","created_at":"2026-07-05T11:36:31.575196+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.05602v3","created_at":"2026-07-05T11:36:31.575196+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.05602","created_at":"2026-07-05T11:36:31.575196+00:00"},{"alias_kind":"pith_short_12","alias_value":"XZTV2VEW3CKQ","created_at":"2026-07-05T11:36:31.575196+00:00"},{"alias_kind":"pith_short_16","alias_value":"XZTV2VEW3CKQ7HWT","created_at":"2026-07-05T11:36:31.575196+00:00"},{"alias_kind":"pith_short_8","alias_value":"XZTV2VEW","created_at":"2026-07-05T11:36:31.575196+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07368","citing_title":"Multi-Agent AI Control: Distributed Attacks Hamper Per-Instance Monitors","ref_index":1,"is_internal_anchor":true},{"citing_arxiv_id":"2511.15352","citing_title":"People readily follow personal advice from AI but it does not improve their well-being","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2603.01865","citing_title":"CyclicJudge: Mitigating Judge Bias Efficiently in LLM-based Evaluation","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XZTV2VEW3CKQ7HWT3TVGIB7LSO","json":"https://pith.science/pith/XZTV2VEW3CKQ7HWT3TVGIB7LSO.json","graph_json":"https://pith.science/api/pith-number/XZTV2VEW3CKQ7HWT3TVGIB7LSO/graph.json","events_json":"https://pith.science/api/pith-number/XZTV2VEW3CKQ7HWT3TVGIB7LSO/events.json","paper":"https://pith.science/paper/XZTV2VEW"},"agent_actions":{"view_html":"https://pith.science/pith/XZTV2VEW3CKQ7HWT3TVGIB7LSO","download_json":"https://pith.science/pith/XZTV2VEW3CKQ7HWT3TVGIB7LSO.json","view_paper":"https://pith.science/paper/XZTV2VEW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.05602&json=true","fetch_graph":"https://pith.science/api/pith-number/XZTV2VEW3CKQ7HWT3TVGIB7LSO/graph.json","fetch_events":"https://pith.science/api/pith-number/XZTV2VEW3CKQ7HWT3TVGIB7LSO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XZTV2VEW3CKQ7HWT3TVGIB7LSO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XZTV2VEW3CKQ7HWT3TVGIB7LSO/action/storage_attestation","attest_author":"https://pith.science/pith/XZTV2VEW3CKQ7HWT3TVGIB7LSO/action/author_attestation","sign_citation":"https://pith.science/pith/XZTV2VEW3CKQ7HWT3TVGIB7LSO/action/citation_signature","submit_replication":"https://pith.science/pith/XZTV2VEW3CKQ7HWT3TVGIB7LSO/action/replication_record"}},"created_at":"2026-07-05T11:36:31.575196+00:00","updated_at":"2026-07-05T11:36:31.575196+00:00"}