{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IZKNQXECOOPNOUWWU3DICP2SWX","short_pith_number":"pith:IZKNQXEC","schema_version":"1.0","canonical_sha256":"4654d85c82739ed752d6a6c6813f52b5fac2540a440d9fae6324b95b0f496b33","source":{"kind":"arxiv","id":"2309.00941","version":2},"attestation_state":"computed","paper":{"title":"Emergent Linear Representations in World Models of Self-Supervised Sequence Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andrew Lee, Martin Wattenberg, Neel Nanda","submitted_at":"2023-09-02T13:37:34Z","abstract_excerpt":"How do sequence models represent their decision-making process? Prior work suggests that Othello-playing neural network learned nonlinear models of the board state (Li et al., 2023). In this work, we provide evidence of a closely related linear representation of the board. In particular, we show that probing for \"my colour\" vs. \"opponent's colour\" may be a simple yet powerful way to interpret the model's internal state. This precise understanding of the internal representations allows us to control the model's behaviour with simple vector arithmetic. Linear representations enable significant i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.00941","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-09-02T13:37:34Z","cross_cats_sorted":[],"title_canon_sha256":"ada7ff773e036ab6c6bd7d85d035bfd6bcbff7787c7b790cecdb0938bc3a9a61","abstract_canon_sha256":"bff1e6fa09d35d79f32940b89d4ca1f0c5068ccd2bf3d01dab6e637bb7620b1e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:48:51.755403Z","signature_b64":"FGV7QGVdEpGrsaODlure84lPaoNRbYq0bB0qOJ8HDbSiVrXobhF6gJf/Y+H/diOxE2yj5BXHUY7PzGX0tgmxDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4654d85c82739ed752d6a6c6813f52b5fac2540a440d9fae6324b95b0f496b33","last_reissued_at":"2026-07-05T06:48:51.754983Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:48:51.754983Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Emergent Linear Representations in World Models of Self-Supervised Sequence Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andrew Lee, Martin Wattenberg, Neel Nanda","submitted_at":"2023-09-02T13:37:34Z","abstract_excerpt":"How do sequence models represent their decision-making process? Prior work suggests that Othello-playing neural network learned nonlinear models of the board state (Li et al., 2023). In this work, we provide evidence of a closely related linear representation of the board. In particular, we show that probing for \"my colour\" vs. \"opponent's colour\" may be a simple yet powerful way to interpret the model's internal state. This precise understanding of the internal representations allows us to control the model's behaviour with simple vector arithmetic. Linear representations enable significant i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.00941","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.00941/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.00941","created_at":"2026-07-05T06:48:51.755036+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.00941v2","created_at":"2026-07-05T06:48:51.755036+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.00941","created_at":"2026-07-05T06:48:51.755036+00:00"},{"alias_kind":"pith_short_12","alias_value":"IZKNQXECOOPN","created_at":"2026-07-05T06:48:51.755036+00:00"},{"alias_kind":"pith_short_16","alias_value":"IZKNQXECOOPNOUWW","created_at":"2026-07-05T06:48:51.755036+00:00"},{"alias_kind":"pith_short_8","alias_value":"IZKNQXEC","created_at":"2026-07-05T06:48:51.755036+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22237","citing_title":"Investigating The Security of Modern AI and Cloud Infrastructure","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03685","citing_title":"A Close Look At World Model Recovery In Supervised Fine-Tuned LLM Planners","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20600","citing_title":"The New Associationism: Lessons from Deep Learning","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25384","citing_title":"GeoMathCode: Understanding Interleaved Math-Code Reasoning for Geometry Problem Solving","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15455","citing_title":"Multi-Turn Neural Transparency: Surfacing Neural Activations Improves User Calibration to LLM Behavioral Drift","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2309.16042","citing_title":"Towards Best Practices of Activation Patching in Language Models: Metrics and Methods","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12809","citing_title":"Correcting Influence: Unboxing LLM Outputs with Orthogonal Latent Spaces","ref_index":246,"is_internal_anchor":false},{"citing_arxiv_id":"2406.11717","citing_title":"Refusal in Language Models Is Mediated by a Single Direction","ref_index":164,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12412","citing_title":"Stories in Space: In-Context Learning Trajectories in Conceptual Belief Space","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09967","citing_title":"Tensor Product Representation Probes Reveal Shared Structure Across Linear Directions","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05715","citing_title":"Decodable but Not Corrected by Fixed Residual-Stream Linear Steering: Evidence from Medical LLM Failure Regimes","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07990","citing_title":"Tool Calling is Linearly Readable and Steerable in Language Models","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15557","citing_title":"Predicting Where Steering Vectors Succeed","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IZKNQXECOOPNOUWWU3DICP2SWX","json":"https://pith.science/pith/IZKNQXECOOPNOUWWU3DICP2SWX.json","graph_json":"https://pith.science/api/pith-number/IZKNQXECOOPNOUWWU3DICP2SWX/graph.json","events_json":"https://pith.science/api/pith-number/IZKNQXECOOPNOUWWU3DICP2SWX/events.json","paper":"https://pith.science/paper/IZKNQXEC"},"agent_actions":{"view_html":"https://pith.science/pith/IZKNQXECOOPNOUWWU3DICP2SWX","download_json":"https://pith.science/pith/IZKNQXECOOPNOUWWU3DICP2SWX.json","view_paper":"https://pith.science/paper/IZKNQXEC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.00941&json=true","fetch_graph":"https://pith.science/api/pith-number/IZKNQXECOOPNOUWWU3DICP2SWX/graph.json","fetch_events":"https://pith.science/api/pith-number/IZKNQXECOOPNOUWWU3DICP2SWX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IZKNQXECOOPNOUWWU3DICP2SWX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IZKNQXECOOPNOUWWU3DICP2SWX/action/storage_attestation","attest_author":"https://pith.science/pith/IZKNQXECOOPNOUWWU3DICP2SWX/action/author_attestation","sign_citation":"https://pith.science/pith/IZKNQXECOOPNOUWWU3DICP2SWX/action/citation_signature","submit_replication":"https://pith.science/pith/IZKNQXECOOPNOUWWU3DICP2SWX/action/replication_record"}},"created_at":"2026-07-05T06:48:51.755036+00:00","updated_at":"2026-07-05T06:48:51.755036+00:00"}