{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:K5OFQJOGJ3THYIVK7WECBJ3N3Y","short_pith_number":"pith:K5OFQJOG","schema_version":"1.0","canonical_sha256":"575c5825c64ee67c22aafd8820a76dde2aba01c0fedb7b791438833e42f12463","source":{"kind":"arxiv","id":"2309.05858","version":2},"attestation_state":"computed","paper":{"title":"Uncovering mesa-optimization algorithms in Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Alexander Meulemans, Blaise Ag\\\"uera y Arcas, Eyvind Niklasson, Jo\\~ao Sacramento, Johannes von Oswald, Mark Sandler, Maximilian Schlegel, Max Vladymyrov, Nicolas Zucchet, Nino Scherrer, Nolan Miller, Razvan Pascanu, Seijin Kobayashi","submitted_at":"2023-09-11T22:42:50Z","abstract_excerpt":"Some autoregressive models exhibit in-context learning capabilities: being able to learn as an input sequence is processed, without undergoing any parameter changes, and without being explicitly trained to do so. The origins of this phenomenon are still poorly understood. Here we analyze a series of Transformer models trained to perform synthetic sequence prediction tasks, and discover that standard next-token prediction error minimization gives rise to a subsidiary learning algorithm that adjusts the model as new inputs are revealed. We show that this process corresponds to gradient-based opt"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.05858","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-09-11T22:42:50Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"84dd6d07bd43ef66e1a575e3a438b9df70306161d8e498c6826d881100a0263f","abstract_canon_sha256":"40298e9cdd4c2a851dc2088496432dc6447beb057570e942107c454c17c98b34"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:20:37.570233Z","signature_b64":"W658WDWiQLCVR7OFoVQ4mfjopiP+Hqp+uuvGmmLDHs0BBVm0r3me6r4UUV2gxRAjwrsHLoo8R0CqGiF7/DgwBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"575c5825c64ee67c22aafd8820a76dde2aba01c0fedb7b791438833e42f12463","last_reissued_at":"2026-07-05T09:20:37.569830Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:20:37.569830Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Uncovering mesa-optimization algorithms in Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Alexander Meulemans, Blaise Ag\\\"uera y Arcas, Eyvind Niklasson, Jo\\~ao Sacramento, Johannes von Oswald, Mark Sandler, Maximilian Schlegel, Max Vladymyrov, Nicolas Zucchet, Nino Scherrer, Nolan Miller, Razvan Pascanu, Seijin Kobayashi","submitted_at":"2023-09-11T22:42:50Z","abstract_excerpt":"Some autoregressive models exhibit in-context learning capabilities: being able to learn as an input sequence is processed, without undergoing any parameter changes, and without being explicitly trained to do so. The origins of this phenomenon are still poorly understood. Here we analyze a series of Transformer models trained to perform synthetic sequence prediction tasks, and discover that standard next-token prediction error minimization gives rise to a subsidiary learning algorithm that adjusts the model as new inputs are revealed. We show that this process corresponds to gradient-based opt"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.05858","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.05858/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.05858","created_at":"2026-07-05T09:20:37.569887+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.05858v2","created_at":"2026-07-05T09:20:37.569887+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.05858","created_at":"2026-07-05T09:20:37.569887+00:00"},{"alias_kind":"pith_short_12","alias_value":"K5OFQJOGJ3TH","created_at":"2026-07-05T09:20:37.569887+00:00"},{"alias_kind":"pith_short_16","alias_value":"K5OFQJOGJ3THYIVK","created_at":"2026-07-05T09:20:37.569887+00:00"},{"alias_kind":"pith_short_8","alias_value":"K5OFQJOG","created_at":"2026-07-05T09:20:37.569887+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12058","citing_title":"Phase Transitions in Attention: A Bayesian Theory of Copy Head Emergence","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01006","citing_title":"Understanding Large Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03825","citing_title":"Dynamic Short Convolutions Improve Transformers","ref_index":170,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09862","citing_title":"Blurry Window Attention","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07568","citing_title":"A Systematic Study of Behavioral Cloning for Scientific Data Annotation","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2406.10162","citing_title":"Sycophancy to Subterfuge: Investigating Reward-Tampering in Large Language Models","ref_index":143,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12809","citing_title":"Correcting Influence: Unboxing LLM Outputs with Orthogonal Latent Spaces","ref_index":110,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13473","citing_title":"OSDN: Improving Delta Rule with Provable Online Preconditioning in Linear Attention","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05838","citing_title":"MDN: Parallelizing Stepwise Momentum for Delta Linear Attention","ref_index":111,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21100","citing_title":"Preconditioned DeltaNet: Curvature-aware Sequence Modeling for Linear Recurrences","ref_index":54,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/K5OFQJOGJ3THYIVK7WECBJ3N3Y","json":"https://pith.science/pith/K5OFQJOGJ3THYIVK7WECBJ3N3Y.json","graph_json":"https://pith.science/api/pith-number/K5OFQJOGJ3THYIVK7WECBJ3N3Y/graph.json","events_json":"https://pith.science/api/pith-number/K5OFQJOGJ3THYIVK7WECBJ3N3Y/events.json","paper":"https://pith.science/paper/K5OFQJOG"},"agent_actions":{"view_html":"https://pith.science/pith/K5OFQJOGJ3THYIVK7WECBJ3N3Y","download_json":"https://pith.science/pith/K5OFQJOGJ3THYIVK7WECBJ3N3Y.json","view_paper":"https://pith.science/paper/K5OFQJOG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.05858&json=true","fetch_graph":"https://pith.science/api/pith-number/K5OFQJOGJ3THYIVK7WECBJ3N3Y/graph.json","fetch_events":"https://pith.science/api/pith-number/K5OFQJOGJ3THYIVK7WECBJ3N3Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/K5OFQJOGJ3THYIVK7WECBJ3N3Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/K5OFQJOGJ3THYIVK7WECBJ3N3Y/action/storage_attestation","attest_author":"https://pith.science/pith/K5OFQJOGJ3THYIVK7WECBJ3N3Y/action/author_attestation","sign_citation":"https://pith.science/pith/K5OFQJOGJ3THYIVK7WECBJ3N3Y/action/citation_signature","submit_replication":"https://pith.science/pith/K5OFQJOGJ3THYIVK7WECBJ3N3Y/action/replication_record"}},"created_at":"2026-07-05T09:20:37.569887+00:00","updated_at":"2026-07-05T09:20:37.569887+00:00"}