{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:OSCYIWMUGAGX677K3HTFFXPCT3","short_pith_number":"pith:OSCYIWMU","schema_version":"1.0","canonical_sha256":"7485845994300d7f7fead9e652dde29eff53bc6ed9599274183029636a44e820","source":{"kind":"arxiv","id":"2608.05086","version":1},"attestation_state":"computed","paper":{"title":"Item Response Theory for AI Safety","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"(2) UK AI Security Institute), David Demitri Africa (2), Joshua Fonseca Rivera (1), Konstantinos Voudouris (2) ((1) Independent, Neil Shah (1)","submitted_at":"2026-08-05T17:25:27Z","abstract_excerpt":"Language models differ in how safely they behave and these differences are measured by safety benchmarks. But aggregated benchmark scores are hard to trust and interpret, because benchmarks duplicate one another, correlate heavily, and models may sandbag when they detect evaluation. To address these issues, we draw on Item Response Theory (IRT), a statistical toolkit for measuring these latents from performance on items with inferred psychometric properties. We fit IRT models to eight safety benchmarks across 192 language models, the largest psychometric analysis of LLM safety evaluations to d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.05086","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-08-05T17:25:27Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"bbf7cba90a53c6202f9f75842cd025c7a3e8353a3fd266bae6840d5bb5157486","abstract_canon_sha256":"3e5155ee44d67bb93f4d10530df6991a3ba7e4548bc3ad19d35f9afba20b9ce8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-06T01:49:02.166841Z","signature_b64":"womhxPtctx3Z7As5KwdYFrVRj0bkU467QCLqFl0SqJfnwoPT2yfdUoipOBWnZvbLZudDDtVbgY6+KU29/mYPBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7485845994300d7f7fead9e652dde29eff53bc6ed9599274183029636a44e820","last_reissued_at":"2026-08-06T01:49:02.165334Z","signature_status":"signed_v1","first_computed_at":"2026-08-06T01:49:02.165334Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Item Response Theory for AI Safety","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"(2) UK AI Security Institute), David Demitri Africa (2), Joshua Fonseca Rivera (1), Konstantinos Voudouris (2) ((1) Independent, Neil Shah (1)","submitted_at":"2026-08-05T17:25:27Z","abstract_excerpt":"Language models differ in how safely they behave and these differences are measured by safety benchmarks. But aggregated benchmark scores are hard to trust and interpret, because benchmarks duplicate one another, correlate heavily, and models may sandbag when they detect evaluation. To address these issues, we draw on Item Response Theory (IRT), a statistical toolkit for measuring these latents from performance on items with inferred psychometric properties. We fit IRT models to eight safety benchmarks across 192 language models, the largest psychometric analysis of LLM safety evaluations to d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.05086","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.05086/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.05086","created_at":"2026-08-06T01:49:02.167217+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.05086v1","created_at":"2026-08-06T01:49:02.167217+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.05086","created_at":"2026-08-06T01:49:02.167217+00:00"},{"alias_kind":"pith_short_12","alias_value":"OSCYIWMUGAGX","created_at":"2026-08-06T01:49:02.167217+00:00"},{"alias_kind":"pith_short_16","alias_value":"OSCYIWMUGAGX677K","created_at":"2026-08-06T01:49:02.167217+00:00"},{"alias_kind":"pith_short_8","alias_value":"OSCYIWMU","created_at":"2026-08-06T01:49:02.167217+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OSCYIWMUGAGX677K3HTFFXPCT3","json":"https://pith.science/pith/OSCYIWMUGAGX677K3HTFFXPCT3.json","graph_json":"https://pith.science/api/pith-number/OSCYIWMUGAGX677K3HTFFXPCT3/graph.json","events_json":"https://pith.science/api/pith-number/OSCYIWMUGAGX677K3HTFFXPCT3/events.json","paper":"https://pith.science/paper/OSCYIWMU"},"agent_actions":{"view_html":"https://pith.science/pith/OSCYIWMUGAGX677K3HTFFXPCT3","download_json":"https://pith.science/pith/OSCYIWMUGAGX677K3HTFFXPCT3.json","view_paper":"https://pith.science/paper/OSCYIWMU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.05086&json=true","fetch_graph":"https://pith.science/api/pith-number/OSCYIWMUGAGX677K3HTFFXPCT3/graph.json","fetch_events":"https://pith.science/api/pith-number/OSCYIWMUGAGX677K3HTFFXPCT3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OSCYIWMUGAGX677K3HTFFXPCT3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OSCYIWMUGAGX677K3HTFFXPCT3/action/storage_attestation","attest_author":"https://pith.science/pith/OSCYIWMUGAGX677K3HTFFXPCT3/action/author_attestation","sign_citation":"https://pith.science/pith/OSCYIWMUGAGX677K3HTFFXPCT3/action/citation_signature","submit_replication":"https://pith.science/pith/OSCYIWMUGAGX677K3HTFFXPCT3/action/replication_record"}},"created_at":"2026-08-06T01:49:02.167217+00:00","updated_at":"2026-08-06T01:49:02.167217+00:00"}