{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FLYYDRDCY4YHZBK4LF7UZLN2EL","short_pith_number":"pith:FLYYDRDC","schema_version":"1.0","canonical_sha256":"2af181c462c7307c855c597f4cadba22e99dba1b90c9a44a753c4a0b1d3c6418","source":{"kind":"arxiv","id":"2407.21792","version":3},"attestation_state":"computed","paper":{"title":"Safetywashing: Do AI Safety Benchmarks Actually Measure Safety Progress?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CY"],"primary_cat":"cs.LG","authors_text":"Adam Khoja, Alexander Pan, Alice Gatti, Dan Hendrycks, Gabriel Mukobi, Long Phan, Mantas Mazeika, Richard Ren, Ryan H. Kim, Stephen Fitz, Steven Basart, Xuwang Yin","submitted_at":"2024-07-31T17:59:24Z","abstract_excerpt":"As artificial intelligence systems grow more powerful, there has been increasing interest in \"AI safety\" research to address emerging and future risks. However, the field of AI safety remains poorly defined and inconsistently measured, leading to confusion about how researchers can contribute. This lack of clarity is compounded by the unclear relationship between AI safety benchmarks and upstream general capabilities (e.g., general knowledge and reasoning). To address these issues, we conduct a comprehensive meta-analysis of AI safety benchmarks, empirically analyzing their correlation with ge"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.21792","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-31T17:59:24Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CY"],"title_canon_sha256":"fba3995e94ed39ff55c2e2d511bc9101552a9a5c120edf3c0ffe6d1184d3a57f","abstract_canon_sha256":"57eeb54e3ce58b19092ca428332d4d85bfc87f6ae68137adb0536d3fe0940ac0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:54:27.899700Z","signature_b64":"EFZA4V/mrr21737MgL5uDa3HMuBZpPVPHZQ2dA285QR+O9gJZAv/Rn26jsnVpUwIulXlMjcu7sDj9i55d4v8Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2af181c462c7307c855c597f4cadba22e99dba1b90c9a44a753c4a0b1d3c6418","last_reissued_at":"2026-07-05T09:54:27.899241Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:54:27.899241Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Safetywashing: Do AI Safety Benchmarks Actually Measure Safety Progress?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CY"],"primary_cat":"cs.LG","authors_text":"Adam Khoja, Alexander Pan, Alice Gatti, Dan Hendrycks, Gabriel Mukobi, Long Phan, Mantas Mazeika, Richard Ren, Ryan H. Kim, Stephen Fitz, Steven Basart, Xuwang Yin","submitted_at":"2024-07-31T17:59:24Z","abstract_excerpt":"As artificial intelligence systems grow more powerful, there has been increasing interest in \"AI safety\" research to address emerging and future risks. However, the field of AI safety remains poorly defined and inconsistently measured, leading to confusion about how researchers can contribute. This lack of clarity is compounded by the unclear relationship between AI safety benchmarks and upstream general capabilities (e.g., general knowledge and reasoning). To address these issues, we conduct a comprehensive meta-analysis of AI safety benchmarks, empirically analyzing their correlation with ge"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.21792","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.21792/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.21792","created_at":"2026-07-05T09:54:27.899296+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.21792v3","created_at":"2026-07-05T09:54:27.899296+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.21792","created_at":"2026-07-05T09:54:27.899296+00:00"},{"alias_kind":"pith_short_12","alias_value":"FLYYDRDCY4YH","created_at":"2026-07-05T09:54:27.899296+00:00"},{"alias_kind":"pith_short_16","alias_value":"FLYYDRDCY4YHZBK4","created_at":"2026-07-05T09:54:27.899296+00:00"},{"alias_kind":"pith_short_8","alias_value":"FLYYDRDC","created_at":"2026-07-05T09:54:27.899296+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26057","citing_title":"The Unfireable Safety Kernel: Execution-Time AI Alignment for AI Agents and Other Escapable AI Systems","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10720","citing_title":"A Pigouvian Matchmaker Mechanism for De-escalating the AGI Race","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09809","citing_title":"Evaluation Cards: An Interpretive Layer for AI Evaluation Reporting","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2404.09005","citing_title":"Proof-of-Learning with Incentive Security","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16282","citing_title":"Taxonomy and Consistency Analysis of Safety Benchmarks for AI Agents","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06233","citing_title":"Blind Refusal: Language Models Refuse to Help Users Evade Unjust, Absurd, and Illegitimate Rules","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FLYYDRDCY4YHZBK4LF7UZLN2EL","json":"https://pith.science/pith/FLYYDRDCY4YHZBK4LF7UZLN2EL.json","graph_json":"https://pith.science/api/pith-number/FLYYDRDCY4YHZBK4LF7UZLN2EL/graph.json","events_json":"https://pith.science/api/pith-number/FLYYDRDCY4YHZBK4LF7UZLN2EL/events.json","paper":"https://pith.science/paper/FLYYDRDC"},"agent_actions":{"view_html":"https://pith.science/pith/FLYYDRDCY4YHZBK4LF7UZLN2EL","download_json":"https://pith.science/pith/FLYYDRDCY4YHZBK4LF7UZLN2EL.json","view_paper":"https://pith.science/paper/FLYYDRDC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.21792&json=true","fetch_graph":"https://pith.science/api/pith-number/FLYYDRDCY4YHZBK4LF7UZLN2EL/graph.json","fetch_events":"https://pith.science/api/pith-number/FLYYDRDCY4YHZBK4LF7UZLN2EL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FLYYDRDCY4YHZBK4LF7UZLN2EL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FLYYDRDCY4YHZBK4LF7UZLN2EL/action/storage_attestation","attest_author":"https://pith.science/pith/FLYYDRDCY4YHZBK4LF7UZLN2EL/action/author_attestation","sign_citation":"https://pith.science/pith/FLYYDRDCY4YHZBK4LF7UZLN2EL/action/citation_signature","submit_replication":"https://pith.science/pith/FLYYDRDCY4YHZBK4LF7UZLN2EL/action/replication_record"}},"created_at":"2026-07-05T09:54:27.899296+00:00","updated_at":"2026-07-05T09:54:27.899296+00:00"}