{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:QYRISKQJAI2U6MZMNGQPIC73HB","short_pith_number":"pith:QYRISKQJ","schema_version":"1.0","canonical_sha256":"8622892a0902354f332c69a0f40bfb385a2edada61c994abbe13e925dd7fe637","source":{"kind":"arxiv","id":"1912.01683","version":10},"attestation_state":"computed","paper":{"title":"Optimal Policies Tend to Seek Power","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Alexander Matt Turner, Andrew Critch, Logan Smith, Prasad Tadepalli, Rohin Shah","submitted_at":"2019-12-03T20:45:49Z","abstract_excerpt":"Some researchers speculate that intelligent reinforcement learning (RL) agents would be incentivized to seek resources and power in pursuit of their objectives. Other researchers point out that RL agents need not have human-like power-seeking instincts. To clarify this discussion, we develop the first formal theory of the statistical tendencies of optimal policies. In the context of Markov decision processes, we prove that certain environmental symmetries are sufficient for optimal policies to tend to seek power over the environment. These symmetries exist in many environments in which the age"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1912.01683","kind":"arxiv","version":10},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2019-12-03T20:45:49Z","cross_cats_sorted":[],"title_canon_sha256":"ad0c1ad08ff7b5da0a090b2c5c3ee28aea8b392c86d2a7a80f0397c58ddcaaed","abstract_canon_sha256":"a0b3f19a071dc82e060b2640228e4b5b7ac2b96cd0c7e9d4d4ad46d17a0cfc6d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:36:28.249673Z","signature_b64":"1ShRsxAwWTFwWLDxUuBkPPuPwpQMuj0NobpHNTFY16Rtqc+5eP/K77zejVl7CuxUfKkmmUYTJGBuMiPGZ3HSAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8622892a0902354f332c69a0f40bfb385a2edada61c994abbe13e925dd7fe637","last_reissued_at":"2026-07-05T05:36:28.249108Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:36:28.249108Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Optimal Policies Tend to Seek Power","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Alexander Matt Turner, Andrew Critch, Logan Smith, Prasad Tadepalli, Rohin Shah","submitted_at":"2019-12-03T20:45:49Z","abstract_excerpt":"Some researchers speculate that intelligent reinforcement learning (RL) agents would be incentivized to seek resources and power in pursuit of their objectives. Other researchers point out that RL agents need not have human-like power-seeking instincts. To clarify this discussion, we develop the first formal theory of the statistical tendencies of optimal policies. In the context of Markov decision processes, we prove that certain environmental symmetries are sufficient for optimal policies to tend to seek power over the environment. These symmetries exist in many environments in which the age"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1912.01683","kind":"arxiv","version":10},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1912.01683/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1912.01683","created_at":"2026-07-05T05:36:28.249170+00:00"},{"alias_kind":"arxiv_version","alias_value":"1912.01683v10","created_at":"2026-07-05T05:36:28.249170+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1912.01683","created_at":"2026-07-05T05:36:28.249170+00:00"},{"alias_kind":"pith_short_12","alias_value":"QYRISKQJAI2U","created_at":"2026-07-05T05:36:28.249170+00:00"},{"alias_kind":"pith_short_16","alias_value":"QYRISKQJAI2U6MZM","created_at":"2026-07-05T05:36:28.249170+00:00"},{"alias_kind":"pith_short_8","alias_value":"QYRISKQJ","created_at":"2026-07-05T05:36:28.249170+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30666","citing_title":"Reframing AGI Confrontation with Off Earth Autonomy","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07612","citing_title":"Position: Anthropomorphic Misalignment Research Needs Stronger Evidence","ref_index":153,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02211","citing_title":"Consistency Training while Mitigating Obfuscation via Rate Matching","ref_index":139,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QYRISKQJAI2U6MZMNGQPIC73HB","json":"https://pith.science/pith/QYRISKQJAI2U6MZMNGQPIC73HB.json","graph_json":"https://pith.science/api/pith-number/QYRISKQJAI2U6MZMNGQPIC73HB/graph.json","events_json":"https://pith.science/api/pith-number/QYRISKQJAI2U6MZMNGQPIC73HB/events.json","paper":"https://pith.science/paper/QYRISKQJ"},"agent_actions":{"view_html":"https://pith.science/pith/QYRISKQJAI2U6MZMNGQPIC73HB","download_json":"https://pith.science/pith/QYRISKQJAI2U6MZMNGQPIC73HB.json","view_paper":"https://pith.science/paper/QYRISKQJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1912.01683&json=true","fetch_graph":"https://pith.science/api/pith-number/QYRISKQJAI2U6MZMNGQPIC73HB/graph.json","fetch_events":"https://pith.science/api/pith-number/QYRISKQJAI2U6MZMNGQPIC73HB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QYRISKQJAI2U6MZMNGQPIC73HB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QYRISKQJAI2U6MZMNGQPIC73HB/action/storage_attestation","attest_author":"https://pith.science/pith/QYRISKQJAI2U6MZMNGQPIC73HB/action/author_attestation","sign_citation":"https://pith.science/pith/QYRISKQJAI2U6MZMNGQPIC73HB/action/citation_signature","submit_replication":"https://pith.science/pith/QYRISKQJAI2U6MZMNGQPIC73HB/action/replication_record"}},"created_at":"2026-07-05T05:36:28.249170+00:00","updated_at":"2026-07-05T05:36:28.249170+00:00"}