{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:XBAO66HLGAZ7PWTSZPKKKPCGDO","short_pith_number":"pith:XBAO66HL","schema_version":"1.0","canonical_sha256":"b840ef78eb3033f7da72cbd4a53c461bb1a8370e90f36337b136d3378aa8fc1a","source":{"kind":"arxiv","id":"2310.08566","version":2},"attestation_state":"computed","paper":{"title":"Transformers as Decision Makers: Provable In-Context Reinforcement Learning via Supervised Pretraining","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","math.ST","stat.ML","stat.TH"],"primary_cat":"cs.LG","authors_text":"Licong Lin, Song Mei, Yu Bai","submitted_at":"2023-10-12T17:55:02Z","abstract_excerpt":"Large transformer models pretrained on offline reinforcement learning datasets have demonstrated remarkable in-context reinforcement learning (ICRL) capabilities, where they can make good decisions when prompted with interaction trajectories from unseen environments. However, when and how transformers can be trained to perform ICRL have not been theoretically well-understood. In particular, it is unclear which reinforcement-learning algorithms transformers can perform in context, and how distribution mismatch in offline training data affects the learned algorithms. This paper provides a theore"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.08566","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-10-12T17:55:02Z","cross_cats_sorted":["cs.AI","cs.CL","math.ST","stat.ML","stat.TH"],"title_canon_sha256":"aea4356a2be42e6b14e3f1d8e236daab2bf659d1db0084bcdcbbd6af6d295923","abstract_canon_sha256":"09932880fbf455f72864cb6757658b634d67134414608980e6e82313c520d1ac"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:23:12.898290Z","signature_b64":"UQFtSlDynBOispsgEF6hcVor+h53hxA2NkfpML4IsIXKyGz1Tbv1WGA4GFW43cocLGObR/FBpugNXlkvQRrgDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b840ef78eb3033f7da72cbd4a53c461bb1a8370e90f36337b136d3378aa8fc1a","last_reissued_at":"2026-07-05T08:23:12.897934Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:23:12.897934Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Transformers as Decision Makers: Provable In-Context Reinforcement Learning via Supervised Pretraining","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","math.ST","stat.ML","stat.TH"],"primary_cat":"cs.LG","authors_text":"Licong Lin, Song Mei, Yu Bai","submitted_at":"2023-10-12T17:55:02Z","abstract_excerpt":"Large transformer models pretrained on offline reinforcement learning datasets have demonstrated remarkable in-context reinforcement learning (ICRL) capabilities, where they can make good decisions when prompted with interaction trajectories from unseen environments. However, when and how transformers can be trained to perform ICRL have not been theoretically well-understood. In particular, it is unclear which reinforcement-learning algorithms transformers can perform in context, and how distribution mismatch in offline training data affects the learned algorithms. This paper provides a theore"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.08566","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.08566/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.08566","created_at":"2026-07-05T08:23:12.897990+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.08566v2","created_at":"2026-07-05T08:23:12.897990+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.08566","created_at":"2026-07-05T08:23:12.897990+00:00"},{"alias_kind":"pith_short_12","alias_value":"XBAO66HLGAZ7","created_at":"2026-07-05T08:23:12.897990+00:00"},{"alias_kind":"pith_short_16","alias_value":"XBAO66HLGAZ7PWTS","created_at":"2026-07-05T08:23:12.897990+00:00"},{"alias_kind":"pith_short_8","alias_value":"XBAO66HL","created_at":"2026-07-05T08:23:12.897990+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18812","citing_title":"Reinforcement Learning Foundation Models Should Already Be A Thing","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31184","citing_title":"Transformers as Bayesian In-Context Experimenters: Smoothness-Adaptive Efficient ATE Estimation","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25210","citing_title":"Multi-Objective Learning for Diffusion Models: A Statistical Theory under Semi-Supervised Learning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09727","citing_title":"One for All: A Non-Linear Transformer can Enable Cross-Domain Generalization for In-Context Reinforcement Learning","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07959","citing_title":"Convergent Stochastic Training of Attention and Understanding LoRA","ref_index":37,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XBAO66HLGAZ7PWTSZPKKKPCGDO","json":"https://pith.science/pith/XBAO66HLGAZ7PWTSZPKKKPCGDO.json","graph_json":"https://pith.science/api/pith-number/XBAO66HLGAZ7PWTSZPKKKPCGDO/graph.json","events_json":"https://pith.science/api/pith-number/XBAO66HLGAZ7PWTSZPKKKPCGDO/events.json","paper":"https://pith.science/paper/XBAO66HL"},"agent_actions":{"view_html":"https://pith.science/pith/XBAO66HLGAZ7PWTSZPKKKPCGDO","download_json":"https://pith.science/pith/XBAO66HLGAZ7PWTSZPKKKPCGDO.json","view_paper":"https://pith.science/paper/XBAO66HL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.08566&json=true","fetch_graph":"https://pith.science/api/pith-number/XBAO66HLGAZ7PWTSZPKKKPCGDO/graph.json","fetch_events":"https://pith.science/api/pith-number/XBAO66HLGAZ7PWTSZPKKKPCGDO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XBAO66HLGAZ7PWTSZPKKKPCGDO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XBAO66HLGAZ7PWTSZPKKKPCGDO/action/storage_attestation","attest_author":"https://pith.science/pith/XBAO66HLGAZ7PWTSZPKKKPCGDO/action/author_attestation","sign_citation":"https://pith.science/pith/XBAO66HLGAZ7PWTSZPKKKPCGDO/action/citation_signature","submit_replication":"https://pith.science/pith/XBAO66HLGAZ7PWTSZPKKKPCGDO/action/replication_record"}},"created_at":"2026-07-05T08:23:12.897990+00:00","updated_at":"2026-07-05T08:23:12.897990+00:00"}