{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HNZOS2BBOBBRE7SEUDLOHDQJBZ","short_pith_number":"pith:HNZOS2BB","schema_version":"1.0","canonical_sha256":"3b72e968217043127e44a0d6e38e090e7a3e538767366430cc3140099f824e73","source":{"kind":"arxiv","id":"2405.06624","version":3},"attestation_state":"computed","paper":{"title":"Towards Guaranteed Safe AI: A Framework for Ensuring Robust and Reliable AI Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Alessandro Abate, Ben Goldhaber, Christian Szegedy, Clark Barrett, David \"davidad\" Dalrymple, Ding Zhao, Jeannette Wing, Joar Skalse, Joe Halpern, Joshua Tenenbaum, Max Tegmark, Nora Ammann, Sanjit Seshia, Steve Omohundro, Stuart Russell, Tan Zhi-Xuan, Yoshua Bengio","submitted_at":"2024-05-10T17:38:32Z","abstract_excerpt":"Ensuring that AI systems reliably and robustly avoid harmful or dangerous behaviours is a crucial challenge, especially for AI systems with a high degree of autonomy and general intelligence, or systems used in safety-critical contexts. In this paper, we will introduce and define a family of approaches to AI safety, which we will refer to as guaranteed safe (GS) AI. The core feature of these approaches is that they aim to produce AI systems which are equipped with high-assurance quantitative safety guarantees. This is achieved by the interplay of three core components: a world model (which pro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.06624","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-05-10T17:38:32Z","cross_cats_sorted":[],"title_canon_sha256":"b6be10fe9b276985aeb3c24351ede842bf7ea22682565d8a4a2a8c9debdd9232","abstract_canon_sha256":"25ae4dbc94445af2485372a2dbfdcc539da04c719e7ea37625d92527d8f45c98"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:41:02.685339Z","signature_b64":"8x9JJTSNoOMrQW3mrsl7us2y0p9BXKrgcd2TaFYaluqzz5Oj/qI4D1uT7uLWFLX5ZD3+NZsI5G5b6ruT2qZdAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3b72e968217043127e44a0d6e38e090e7a3e538767366430cc3140099f824e73","last_reissued_at":"2026-07-05T08:41:02.684826Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:41:02.684826Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Guaranteed Safe AI: A Framework for Ensuring Robust and Reliable AI Systems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Alessandro Abate, Ben Goldhaber, Christian Szegedy, Clark Barrett, David \"davidad\" Dalrymple, Ding Zhao, Jeannette Wing, Joar Skalse, Joe Halpern, Joshua Tenenbaum, Max Tegmark, Nora Ammann, Sanjit Seshia, Steve Omohundro, Stuart Russell, Tan Zhi-Xuan, Yoshua Bengio","submitted_at":"2024-05-10T17:38:32Z","abstract_excerpt":"Ensuring that AI systems reliably and robustly avoid harmful or dangerous behaviours is a crucial challenge, especially for AI systems with a high degree of autonomy and general intelligence, or systems used in safety-critical contexts. In this paper, we will introduce and define a family of approaches to AI safety, which we will refer to as guaranteed safe (GS) AI. The core feature of these approaches is that they aim to produce AI systems which are equipped with high-assurance quantitative safety guarantees. This is achieved by the interplay of three core components: a world model (which pro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.06624","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.06624/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.06624","created_at":"2026-07-05T08:41:02.684895+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.06624v3","created_at":"2026-07-05T08:41:02.684895+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.06624","created_at":"2026-07-05T08:41:02.684895+00:00"},{"alias_kind":"pith_short_12","alias_value":"HNZOS2BBOBBR","created_at":"2026-07-05T08:41:02.684895+00:00"},{"alias_kind":"pith_short_16","alias_value":"HNZOS2BBOBBRE7SE","created_at":"2026-07-05T08:41:02.684895+00:00"},{"alias_kind":"pith_short_8","alias_value":"HNZOS2BB","created_at":"2026-07-05T08:41:02.684895+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09940","citing_title":"Interactions Between Crosscoder Features: A Compact Proofs Perspective","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04037","citing_title":"Toward Pre-Deployment Assurance for Enterprise AI Agents: Ontology-Grounded Simulation and Trust Certification","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01030","citing_title":"Effect-Transparent Governance for AI Workflow Architectures: Semantic Preservation, Expressive Minimality, and Decidability Boundaries","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01032","citing_title":"Algebraic Semantics of Governed Execution: Monoidal Categories, Effect Algebras, and Coterminous Boundaries","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01008","citing_title":"FVSpec: Real-World Property-Based Tests as Lean Challenges","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25239","citing_title":"Tensor-Based Batch Fuzzing with Adaptive Perturbation Scaling for Deep Neural Networks","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2602.20102","citing_title":"BarrierSteer: LLM Safety via Learning Barrier Steering","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2512.02193","citing_title":"From monoliths to modules: Decomposing transducers for efficient world modelling","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2510.13727","citing_title":"From Refusal to Recovery: A Control-Theoretic Approach to Generative AI Guardrails","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2506.05171","citing_title":"Towards provable probabilistic safety for scalable embodied AI systems","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2602.12851","citing_title":"Chimera: Neuro-Symbolic Attention Primitives for Trustworthy Dataplane Intelligence","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10310","citing_title":"Positive Alignment: Artificial Intelligence for Human Flourishing","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05209","citing_title":"Are Flat Minima an Illusion?","ref_index":151,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27292","citing_title":"The Two Boundaries: Why Behavioral AI Governance Fails Structurally","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03034","citing_title":"Stable Agentic Control: Tool-Mediated LLM Architecture for Autonomous Cyber Defense","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01032","citing_title":"Algebraic Semantics of Governed Execution: Monoidal Categories, Effect Algebras, and Coterminous Boundaries","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01030","citing_title":"Effect-Transparent Governance for AI Workflow Architectures: Semantic Preservation, Expressive Minimality, and Decidability Boundaries","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11174","citing_title":"EmbodiedGovBench: A Benchmark for Governance, Recovery, and Upgrade Safety in Embodied Agent Systems","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05150","citing_title":"Compiled AI: Deterministic Code Generation for LLM-Based Workflow Automation","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HNZOS2BBOBBRE7SEUDLOHDQJBZ","json":"https://pith.science/pith/HNZOS2BBOBBRE7SEUDLOHDQJBZ.json","graph_json":"https://pith.science/api/pith-number/HNZOS2BBOBBRE7SEUDLOHDQJBZ/graph.json","events_json":"https://pith.science/api/pith-number/HNZOS2BBOBBRE7SEUDLOHDQJBZ/events.json","paper":"https://pith.science/paper/HNZOS2BB"},"agent_actions":{"view_html":"https://pith.science/pith/HNZOS2BBOBBRE7SEUDLOHDQJBZ","download_json":"https://pith.science/pith/HNZOS2BBOBBRE7SEUDLOHDQJBZ.json","view_paper":"https://pith.science/paper/HNZOS2BB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.06624&json=true","fetch_graph":"https://pith.science/api/pith-number/HNZOS2BBOBBRE7SEUDLOHDQJBZ/graph.json","fetch_events":"https://pith.science/api/pith-number/HNZOS2BBOBBRE7SEUDLOHDQJBZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HNZOS2BBOBBRE7SEUDLOHDQJBZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HNZOS2BBOBBRE7SEUDLOHDQJBZ/action/storage_attestation","attest_author":"https://pith.science/pith/HNZOS2BBOBBRE7SEUDLOHDQJBZ/action/author_attestation","sign_citation":"https://pith.science/pith/HNZOS2BBOBBRE7SEUDLOHDQJBZ/action/citation_signature","submit_replication":"https://pith.science/pith/HNZOS2BBOBBRE7SEUDLOHDQJBZ/action/replication_record"}},"created_at":"2026-07-05T08:41:02.684895+00:00","updated_at":"2026-07-05T08:41:02.684895+00:00"}