{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OLFNU5QKTUMW7LRXVTWX3T6USE","short_pith_number":"pith:OLFNU5QK","schema_version":"1.0","canonical_sha256":"72cada760a9d196fae37aced7dcfd49110013a24e102d837a88c2b57d0d2d403","source":{"kind":"arxiv","id":"2506.03056","version":1},"attestation_state":"computed","paper":{"title":"Corrigibility as a Singular Target: A Vision for Inherently Reliable Foundation Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CY","cs.LG"],"primary_cat":"cs.AI","authors_text":"Max Harms (Machine Intelligence Research Institute), Ram Potham (Independent Researcher)","submitted_at":"2025-06-03T16:36:03Z","abstract_excerpt":"Foundation models (FMs) face a critical safety challenge: as capabilities scale, instrumental convergence drives default trajectories toward loss of human control, potentially culminating in existential catastrophe. Current alignment approaches struggle with value specification complexity and fail to address emergent power-seeking behaviors. We propose \"Corrigibility as a Singular Target\" (CAST)-designing FMs whose overriding objective is empowering designated human principals to guide, correct, and control them. This paradigm shift from static value-loading to dynamic human empowerment transf"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.03056","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-06-03T16:36:03Z","cross_cats_sorted":["cs.CY","cs.LG"],"title_canon_sha256":"e5ebd2f4c69cca8320cfddda39d6ccda1796490429007240b85bc4aec1638aef","abstract_canon_sha256":"c742ed839ea0416a45b6cbe1a876759b53fa1cef5130347b5412d4112afeec46"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:14.080006Z","signature_b64":"78lTKDBE8CguOSGJk3yLmOn8KpPjRcHRemhc9p+2NHnWbNVSe0Hz6IpALOxzaTnZsOL20pXMvlksqdrCwCOSBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"72cada760a9d196fae37aced7dcfd49110013a24e102d837a88c2b57d0d2d403","last_reissued_at":"2026-07-05T11:15:14.079551Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:14.079551Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Corrigibility as a Singular Target: A Vision for Inherently Reliable Foundation Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CY","cs.LG"],"primary_cat":"cs.AI","authors_text":"Max Harms (Machine Intelligence Research Institute), Ram Potham (Independent Researcher)","submitted_at":"2025-06-03T16:36:03Z","abstract_excerpt":"Foundation models (FMs) face a critical safety challenge: as capabilities scale, instrumental convergence drives default trajectories toward loss of human control, potentially culminating in existential catastrophe. Current alignment approaches struggle with value specification complexity and fail to address emergent power-seeking behaviors. We propose \"Corrigibility as a Singular Target\" (CAST)-designing FMs whose overriding objective is empowering designated human principals to guide, correct, and control them. This paradigm shift from static value-loading to dynamic human empowerment transf"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.03056","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.03056/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.03056","created_at":"2026-07-05T11:15:14.079609+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.03056v1","created_at":"2026-07-05T11:15:14.079609+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.03056","created_at":"2026-07-05T11:15:14.079609+00:00"},{"alias_kind":"pith_short_12","alias_value":"OLFNU5QKTUMW","created_at":"2026-07-05T11:15:14.079609+00:00"},{"alias_kind":"pith_short_16","alias_value":"OLFNU5QKTUMW7LRX","created_at":"2026-07-05T11:15:14.079609+00:00"},{"alias_kind":"pith_short_8","alias_value":"OLFNU5QK","created_at":"2026-07-05T11:15:14.079609+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.00159","citing_title":"Model-Based Soft Maximization of Suitable Metrics of Long-Term Human Power","ref_index":24,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OLFNU5QKTUMW7LRXVTWX3T6USE","json":"https://pith.science/pith/OLFNU5QKTUMW7LRXVTWX3T6USE.json","graph_json":"https://pith.science/api/pith-number/OLFNU5QKTUMW7LRXVTWX3T6USE/graph.json","events_json":"https://pith.science/api/pith-number/OLFNU5QKTUMW7LRXVTWX3T6USE/events.json","paper":"https://pith.science/paper/OLFNU5QK"},"agent_actions":{"view_html":"https://pith.science/pith/OLFNU5QKTUMW7LRXVTWX3T6USE","download_json":"https://pith.science/pith/OLFNU5QKTUMW7LRXVTWX3T6USE.json","view_paper":"https://pith.science/paper/OLFNU5QK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.03056&json=true","fetch_graph":"https://pith.science/api/pith-number/OLFNU5QKTUMW7LRXVTWX3T6USE/graph.json","fetch_events":"https://pith.science/api/pith-number/OLFNU5QKTUMW7LRXVTWX3T6USE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OLFNU5QKTUMW7LRXVTWX3T6USE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OLFNU5QKTUMW7LRXVTWX3T6USE/action/storage_attestation","attest_author":"https://pith.science/pith/OLFNU5QKTUMW7LRXVTWX3T6USE/action/author_attestation","sign_citation":"https://pith.science/pith/OLFNU5QKTUMW7LRXVTWX3T6USE/action/citation_signature","submit_replication":"https://pith.science/pith/OLFNU5QKTUMW7LRXVTWX3T6USE/action/replication_record"}},"created_at":"2026-07-05T11:15:14.079609+00:00","updated_at":"2026-07-05T11:15:14.079609+00:00"}