{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:K6FRATGKSH5XK3R7LF4OQ54LNW","short_pith_number":"pith:K6FRATGK","schema_version":"1.0","canonical_sha256":"578b104cca91fb756e3f5978e8778b6db44647a0547457c603f95336e4067447","source":{"kind":"arxiv","id":"2209.00626","version":8},"attestation_state":"computed","paper":{"title":"The Alignment Problem from a Deep Learning Perspective","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Lawrence Chan, Richard Ngo, S\\\"oren Mindermann","submitted_at":"2022-08-30T02:12:47Z","abstract_excerpt":"In coming years or decades, artificial general intelligence (AGI) may surpass human capabilities across many critical domains. We argue that, without substantial effort to prevent it, AGIs could learn to pursue goals that are in conflict (i.e. misaligned) with human interests. If trained like today's most capable models, AGIs could learn to act deceptively to receive higher reward, learn misaligned internally-represented goals which generalize beyond their fine-tuning distributions, and pursue those goals using power-seeking strategies. We review emerging evidence for these properties. In this"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.00626","kind":"arxiv","version":8},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2022-08-30T02:12:47Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"6aef015c7ee09d69bb3f46f6dfe75b31a99df7c5e7027ee317ae9e9cebb23ae0","abstract_canon_sha256":"f0c511cbacd9263c42882c1f3ad9d3252648f54b0c2ed6c6b3245ec11e195029"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:58:12.016206Z","signature_b64":"IK6PpYH45vN0LziR0wN2gQBW+RBGpYJ9lr9sDhW3dgX+vqCShLn3MF4H8C76GgijEElTGL2YYUYqDqsvkwBBDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"578b104cca91fb756e3f5978e8778b6db44647a0547457c603f95336e4067447","last_reissued_at":"2026-07-05T10:58:12.015647Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:58:12.015647Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Alignment Problem from a Deep Learning Perspective","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Lawrence Chan, Richard Ngo, S\\\"oren Mindermann","submitted_at":"2022-08-30T02:12:47Z","abstract_excerpt":"In coming years or decades, artificial general intelligence (AGI) may surpass human capabilities across many critical domains. We argue that, without substantial effort to prevent it, AGIs could learn to pursue goals that are in conflict (i.e. misaligned) with human interests. If trained like today's most capable models, AGIs could learn to act deceptively to receive higher reward, learn misaligned internally-represented goals which generalize beyond their fine-tuning distributions, and pursue those goals using power-seeking strategies. We review emerging evidence for these properties. In this"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.00626","kind":"arxiv","version":8},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.00626/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.00626","created_at":"2026-07-05T10:58:12.015723+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.00626v8","created_at":"2026-07-05T10:58:12.015723+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.00626","created_at":"2026-07-05T10:58:12.015723+00:00"},{"alias_kind":"pith_short_12","alias_value":"K6FRATGKSH5X","created_at":"2026-07-05T10:58:12.015723+00:00"},{"alias_kind":"pith_short_16","alias_value":"K6FRATGKSH5XK3R7","created_at":"2026-07-05T10:58:12.015723+00:00"},{"alias_kind":"pith_short_8","alias_value":"K6FRATGK","created_at":"2026-07-05T10:58:12.015723+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.08066","citing_title":"Persuasion Attacks Can Decrease Effectiveness of CoT Monitoring","ref_index":22,"is_internal_anchor":true},{"citing_arxiv_id":"2607.07605","citing_title":"User identity conditions moral wrongness ratings in non-reasoning large language models","ref_index":7,"is_internal_anchor":true},{"citing_arxiv_id":"2606.17478","citing_title":"Decoding Hidden Deception in Reasoning LLMs: Activation Explainers for Deception Auditing","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07998","citing_title":"Enhancing AI Interpretability with Localised Architectures","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06028","citing_title":"Misaligned AI as a New Insider Risk","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05194","citing_title":"Temporal Preference Concepts and their Functions in a Large Language Model","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29657","citing_title":"Safety from Honesty in a Disinterested AI Predictor","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10711","citing_title":"The Agentic Web Requires New Normative Infrastructure","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23565","citing_title":"Understanding Goal Generalisation in Sequential Reinforcement Learning","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2309.08600","citing_title":"Sparse Autoencoders Find Highly Interpretable Features in Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2401.02458","citing_title":"Data-Centric Foundation Models in Computational Healthcare: A Survey","ref_index":210,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16035","citing_title":"Who Owns This Agent? Tracing AI Agents Back to Their Owners","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2210.10760","citing_title":"Scaling Laws for Reward Model Overoptimization","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2406.10162","citing_title":"Sycophancy to Subterfuge: Investigating Reward-Tampering in Large Language Models","ref_index":187,"is_internal_anchor":false},{"citing_arxiv_id":"2603.19282","citing_title":"Framing Effects in Independent-Agent Large Language Models: A Cross-Family Behavioral Analysis","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2603.18633","citing_title":"An Onto-Relational-Sophic Framework for Governing Synthetic Minds","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01346","citing_title":"Safety, Security, and Cognitive Risks in World Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02720","citing_title":"Cognitive Comparability and the Limits of Governance: Evaluating Authority Under Radical Capability Asymmetry","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2403.19647","citing_title":"Sparse Feature Circuits: Discovering and Editing Interpretable Causal Graphs in Language Models","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11134","citing_title":"Spurious Correlation Learning in Preference Optimization: Mechanisms, Consequences, and Mitigation via Tie Training","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17596","citing_title":"Terminal Wrench: A Dataset of 331 Reward-Hackable Environments and 3,632 Exploit Trajectories","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20805","citing_title":"Relative Principals, Pluralistic Alignment, and the Structural Value Alignment Problem","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K6FRATGKSH5XK3R7LF4OQ54LNW","json":"https://pith.science/pith/K6FRATGKSH5XK3R7LF4OQ54LNW.json","graph_json":"https://pith.science/api/pith-number/K6FRATGKSH5XK3R7LF4OQ54LNW/graph.json","events_json":"https://pith.science/api/pith-number/K6FRATGKSH5XK3R7LF4OQ54LNW/events.json","paper":"https://pith.science/paper/K6FRATGK"},"agent_actions":{"view_html":"https://pith.science/pith/K6FRATGKSH5XK3R7LF4OQ54LNW","download_json":"https://pith.science/pith/K6FRATGKSH5XK3R7LF4OQ54LNW.json","view_paper":"https://pith.science/paper/K6FRATGK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.00626&json=true","fetch_graph":"https://pith.science/api/pith-number/K6FRATGKSH5XK3R7LF4OQ54LNW/graph.json","fetch_events":"https://pith.science/api/pith-number/K6FRATGKSH5XK3R7LF4OQ54LNW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K6FRATGKSH5XK3R7LF4OQ54LNW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K6FRATGKSH5XK3R7LF4OQ54LNW/action/storage_attestation","attest_author":"https://pith.science/pith/K6FRATGKSH5XK3R7LF4OQ54LNW/action/author_attestation","sign_citation":"https://pith.science/pith/K6FRATGKSH5XK3R7LF4OQ54LNW/action/citation_signature","submit_replication":"https://pith.science/pith/K6FRATGKSH5XK3R7LF4OQ54LNW/action/replication_record"}},"created_at":"2026-07-05T10:58:12.015723+00:00","updated_at":"2026-07-05T10:58:12.015723+00:00"}