{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:ITZIZNRHDYIXRYI6RFKINOFOQV","short_pith_number":"pith:ITZIZNRH","schema_version":"1.0","canonical_sha256":"44f28cb6271e1178e11e895486b8ae85515fac0cf624639483d5e0981ae5ead4","source":{"kind":"arxiv","id":"2012.07532","version":1},"attestation_state":"computed","paper":{"title":"An overview of 11 proposals for building safe advanced AI","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Evan Hubinger","submitted_at":"2020-12-04T22:53:18Z","abstract_excerpt":"This paper analyzes and compares 11 different proposals for building safe advanced AI under the current machine learning paradigm, including major contenders such as iterated amplification, AI safety via debate, and recursive reward modeling. Each proposal is evaluated on the four components of outer alignment, inner alignment, training competitiveness, and performance competitiveness, of which the distinction between the latter two is introduced in this paper. While prior literature has primarily focused on analyzing individual proposals, or primarily focused on outer alignment at the expense"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2012.07532","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-12-04T22:53:18Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f59d73a70cb4b88bb50bb5098654ac31a02b4b82c0b5e0a66c405795534dc010","abstract_canon_sha256":"017bdd948a65a492eb0330cd6dc13c160841468623d6ce69a007e6a25a76f8f8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:59:21.533379Z","signature_b64":"BfzofBk2MqoyUl3AuVKd6SgnuQjiiGiyCUc00O9FyEtVFXx4ODkC7mkDpMZgH2qL5B6CryYMZdZ10FPS3vRNAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"44f28cb6271e1178e11e895486b8ae85515fac0cf624639483d5e0981ae5ead4","last_reissued_at":"2026-07-05T01:59:21.532995Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:59:21.532995Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An overview of 11 proposals for building safe advanced AI","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Evan Hubinger","submitted_at":"2020-12-04T22:53:18Z","abstract_excerpt":"This paper analyzes and compares 11 different proposals for building safe advanced AI under the current machine learning paradigm, including major contenders such as iterated amplification, AI safety via debate, and recursive reward modeling. Each proposal is evaluated on the four components of outer alignment, inner alignment, training competitiveness, and performance competitiveness, of which the distinction between the latter two is introduced in this paper. While prior literature has primarily focused on analyzing individual proposals, or primarily focused on outer alignment at the expense"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2012.07532","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2012.07532/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2012.07532","created_at":"2026-07-05T01:59:21.533050+00:00"},{"alias_kind":"arxiv_version","alias_value":"2012.07532v1","created_at":"2026-07-05T01:59:21.533050+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2012.07532","created_at":"2026-07-05T01:59:21.533050+00:00"},{"alias_kind":"pith_short_12","alias_value":"ITZIZNRHDYIX","created_at":"2026-07-05T01:59:21.533050+00:00"},{"alias_kind":"pith_short_16","alias_value":"ITZIZNRHDYIXRYI6","created_at":"2026-07-05T01:59:21.533050+00:00"},{"alias_kind":"pith_short_8","alias_value":"ITZIZNRH","created_at":"2026-07-05T01:59:21.533050+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02211","citing_title":"Consistency Training while Mitigating Obfuscation via Rate Matching","ref_index":168,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12809","citing_title":"Correcting Influence: Unboxing LLM Outputs with Orthogonal Latent Spaces","ref_index":240,"is_internal_anchor":false},{"citing_arxiv_id":"2211.00593","citing_title":"Interpretability in the Wild: a Circuit for Indirect Object Identification in GPT-2 small","ref_index":69,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ITZIZNRHDYIXRYI6RFKINOFOQV","json":"https://pith.science/pith/ITZIZNRHDYIXRYI6RFKINOFOQV.json","graph_json":"https://pith.science/api/pith-number/ITZIZNRHDYIXRYI6RFKINOFOQV/graph.json","events_json":"https://pith.science/api/pith-number/ITZIZNRHDYIXRYI6RFKINOFOQV/events.json","paper":"https://pith.science/paper/ITZIZNRH"},"agent_actions":{"view_html":"https://pith.science/pith/ITZIZNRHDYIXRYI6RFKINOFOQV","download_json":"https://pith.science/pith/ITZIZNRHDYIXRYI6RFKINOFOQV.json","view_paper":"https://pith.science/paper/ITZIZNRH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2012.07532&json=true","fetch_graph":"https://pith.science/api/pith-number/ITZIZNRHDYIXRYI6RFKINOFOQV/graph.json","fetch_events":"https://pith.science/api/pith-number/ITZIZNRHDYIXRYI6RFKINOFOQV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ITZIZNRHDYIXRYI6RFKINOFOQV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ITZIZNRHDYIXRYI6RFKINOFOQV/action/storage_attestation","attest_author":"https://pith.science/pith/ITZIZNRHDYIXRYI6RFKINOFOQV/action/author_attestation","sign_citation":"https://pith.science/pith/ITZIZNRHDYIXRYI6RFKINOFOQV/action/citation_signature","submit_replication":"https://pith.science/pith/ITZIZNRHDYIXRYI6RFKINOFOQV/action/replication_record"}},"created_at":"2026-07-05T01:59:21.533050+00:00","updated_at":"2026-07-05T01:59:21.533050+00:00"}