{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3HBJTEO66CBINEJTYUBJWCIGWU","short_pith_number":"pith:3HBJTEO6","schema_version":"1.0","canonical_sha256":"d9c29991def082869133c5029b0906b505f109afb33899145bb41f1db9fa1179","source":{"kind":"arxiv","id":"2503.04299","version":2},"attestation_state":"computed","paper":{"title":"Mapping AI Benchmark Data to Quantitative Risk Estimates Through Expert Elicitation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Henry Papadatos, Malcolm Murray, Otter Quarks, Pierre-Fran\\c{c}ois Gimenez, Simeon Campos","submitted_at":"2025-03-06T10:39:47Z","abstract_excerpt":"The literature and multiple experts point to many potential risks from large language models (LLMs), but there are still very few direct measurements of the actual harms posed. AI risk assessment has so far focused on measuring the models' capabilities, but the capabilities of models are only indicators of risk, not measures of risk. Better modeling and quantification of AI risk scenarios can help bridge this disconnect and link the capabilities of LLMs to tangible real-world harm. This paper makes an early contribution to this field by demonstrating how existing AI benchmarks can be used to f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.04299","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-03-06T10:39:47Z","cross_cats_sorted":[],"title_canon_sha256":"6de5959660feb8349fbc795d8eda22aeced7b0d8b135884343909918f1f6177d","abstract_canon_sha256":"8052894bd4203b312c69e6b9cd540bef86c1d034bd691555430de0a5f39f6192"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:27:27.591750Z","signature_b64":"hVH/BEn6UdswTV8iTyOHWBLnLu3GFuN3aDxlgL6SA0cNb0eL2qBAQHjdOJb9Kjnwkx+y8tmwGAynQS06bkMbDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d9c29991def082869133c5029b0906b505f109afb33899145bb41f1db9fa1179","last_reissued_at":"2026-07-05T10:27:27.591025Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:27:27.591025Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mapping AI Benchmark Data to Quantitative Risk Estimates Through Expert Elicitation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Henry Papadatos, Malcolm Murray, Otter Quarks, Pierre-Fran\\c{c}ois Gimenez, Simeon Campos","submitted_at":"2025-03-06T10:39:47Z","abstract_excerpt":"The literature and multiple experts point to many potential risks from large language models (LLMs), but there are still very few direct measurements of the actual harms posed. AI risk assessment has so far focused on measuring the models' capabilities, but the capabilities of models are only indicators of risk, not measures of risk. Better modeling and quantification of AI risk scenarios can help bridge this disconnect and link the capabilities of LLMs to tangible real-world harm. This paper makes an early contribution to this field by demonstrating how existing AI benchmarks can be used to f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.04299","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.04299/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.04299","created_at":"2026-07-05T10:27:27.591120+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.04299v2","created_at":"2026-07-05T10:27:27.591120+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.04299","created_at":"2026-07-05T10:27:27.591120+00:00"},{"alias_kind":"pith_short_12","alias_value":"3HBJTEO66CBI","created_at":"2026-07-05T10:27:27.591120+00:00"},{"alias_kind":"pith_short_16","alias_value":"3HBJTEO66CBINEJT","created_at":"2026-07-05T10:27:27.591120+00:00"},{"alias_kind":"pith_short_8","alias_value":"3HBJTEO6","created_at":"2026-07-05T10:27:27.591120+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27804","citing_title":"Methods for Uncertainty Representation in Risk Management: A Comparative Review and Decision-Oriented Framework","ref_index":122,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3HBJTEO66CBINEJTYUBJWCIGWU","json":"https://pith.science/pith/3HBJTEO66CBINEJTYUBJWCIGWU.json","graph_json":"https://pith.science/api/pith-number/3HBJTEO66CBINEJTYUBJWCIGWU/graph.json","events_json":"https://pith.science/api/pith-number/3HBJTEO66CBINEJTYUBJWCIGWU/events.json","paper":"https://pith.science/paper/3HBJTEO6"},"agent_actions":{"view_html":"https://pith.science/pith/3HBJTEO66CBINEJTYUBJWCIGWU","download_json":"https://pith.science/pith/3HBJTEO66CBINEJTYUBJWCIGWU.json","view_paper":"https://pith.science/paper/3HBJTEO6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.04299&json=true","fetch_graph":"https://pith.science/api/pith-number/3HBJTEO66CBINEJTYUBJWCIGWU/graph.json","fetch_events":"https://pith.science/api/pith-number/3HBJTEO66CBINEJTYUBJWCIGWU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3HBJTEO66CBINEJTYUBJWCIGWU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3HBJTEO66CBINEJTYUBJWCIGWU/action/storage_attestation","attest_author":"https://pith.science/pith/3HBJTEO66CBINEJTYUBJWCIGWU/action/author_attestation","sign_citation":"https://pith.science/pith/3HBJTEO66CBINEJTYUBJWCIGWU/action/citation_signature","submit_replication":"https://pith.science/pith/3HBJTEO66CBINEJTYUBJWCIGWU/action/replication_record"}},"created_at":"2026-07-05T10:27:27.591120+00:00","updated_at":"2026-07-05T10:27:27.591120+00:00"}