{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:E6UHICXDXN3C4PUAUEKJWLAISG","short_pith_number":"pith:E6UHICXD","schema_version":"1.0","canonical_sha256":"27a8740ae3bb762e3e80a1149b2c08918ac0e05ac966ac7041ec62574be408ce","source":{"kind":"arxiv","id":"2202.09481","version":2},"attestation_state":"computed","paper":{"title":"TransDreamer: Reinforcement Learning with Transformer World Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chang Chen, Jaesik Yoon, Sungjin Ahn, Yi-Fu Wu","submitted_at":"2022-02-19T00:30:52Z","abstract_excerpt":"The Dreamer agent provides various benefits of Model-Based Reinforcement Learning (MBRL) such as sample efficiency, reusable knowledge, and safe planning. However, its world model and policy networks inherit the limitations of recurrent neural networks and thus an important question is how an MBRL framework can benefit from the recent advances of transformers and what the challenges are in doing so. In this paper, we propose a transformer-based MBRL agent, called TransDreamer. We first introduce the Transformer State-Space Model, a world model that leverages a transformer for dynamics predicti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.09481","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-02-19T00:30:52Z","cross_cats_sorted":[],"title_canon_sha256":"32a48ebbc07a28ad109b11ff8c40f0ca45c0bd91bb478d7c75bc7a2a17d2b48f","abstract_canon_sha256":"37efd9af706e97fea681563f671245d3b48d2594cd3cea57c6f236fbcb60a107"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:37:28.222209Z","signature_b64":"vqrnM6mjNbZ+SJIuiwxE63Ix2EIWXPxBH+I9rrdhWOZqht2C5fhPLLa7c5/36pZEdN2lXysrDtuO1Lcsnt2FDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"27a8740ae3bb762e3e80a1149b2c08918ac0e05ac966ac7041ec62574be408ce","last_reissued_at":"2026-07-05T09:37:28.221763Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:37:28.221763Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TransDreamer: Reinforcement Learning with Transformer World Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chang Chen, Jaesik Yoon, Sungjin Ahn, Yi-Fu Wu","submitted_at":"2022-02-19T00:30:52Z","abstract_excerpt":"The Dreamer agent provides various benefits of Model-Based Reinforcement Learning (MBRL) such as sample efficiency, reusable knowledge, and safe planning. However, its world model and policy networks inherit the limitations of recurrent neural networks and thus an important question is how an MBRL framework can benefit from the recent advances of transformers and what the challenges are in doing so. In this paper, we propose a transformer-based MBRL agent, called TransDreamer. We first introduce the Transformer State-Space Model, a world model that leverages a transformer for dynamics predicti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.09481","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.09481/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.09481","created_at":"2026-07-05T09:37:28.221819+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.09481v2","created_at":"2026-07-05T09:37:28.221819+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.09481","created_at":"2026-07-05T09:37:28.221819+00:00"},{"alias_kind":"pith_short_12","alias_value":"E6UHICXDXN3C","created_at":"2026-07-05T09:37:28.221819+00:00"},{"alias_kind":"pith_short_16","alias_value":"E6UHICXDXN3C4PUA","created_at":"2026-07-05T09:37:28.221819+00:00"},{"alias_kind":"pith_short_8","alias_value":"E6UHICXD","created_at":"2026-07-05T09:37:28.221819+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.22748","citing_title":"Agentic World Modeling: Foundations, Capabilities, Laws, and Beyond","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20781","citing_title":"World Action Models: A Survey","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00811","citing_title":"From Pixels to Temporal Correlations: Learning Informative Representations for Reinforcement Learning Pre-training","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05208","citing_title":"Transformer-Enhanced Reinforcement Learning: Fundamentals and Applications in Communication Networks","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00133","citing_title":"World Models: A Comprehensive Survey of Architectures, Methodologies, Reasoning Paradigms, and Applications","ref_index":153,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00780","citing_title":"Behavior-Invariant Task Representation Learning with Transformer-based World Models for Offline Meta-Reinforcement Learning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01027","citing_title":"$\\tau_0$-WM: A Unified Video-Action World Model for Robotic Manipulation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2512.03438","citing_title":"Multimodal Reinforcement Learning with Adaptive Verifier for AI Agents","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2310.06114","citing_title":"Learning Interactive Real-World Simulators","ref_index":248,"is_internal_anchor":false},{"citing_arxiv_id":"2603.15759","citing_title":"Simulation Distillation: Pretraining World Models in Simulation for Rapid Real-World Adaptation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12090","citing_title":"World Action Models: The Next Frontier in Embodied AI","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08578","citing_title":"Probing the Impact of Scale on Data-Efficient, Generalist Transformer World Models for Atari","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22748","citing_title":"Agentic World Modeling: Foundations, Capabilities, Laws, and Beyond","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01950","citing_title":"TRAP: Tail-aware Ranking Attack for World-Model Planning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08780","citing_title":"Morphology-Conditioned World Model for Cross-Embodiment Quadrupedal Locomotion","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08199","citing_title":"Beyond Static Forecasting: Unleashing the Power of World Models for Mobile Traffic Extrapolation","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E6UHICXDXN3C4PUAUEKJWLAISG","json":"https://pith.science/pith/E6UHICXDXN3C4PUAUEKJWLAISG.json","graph_json":"https://pith.science/api/pith-number/E6UHICXDXN3C4PUAUEKJWLAISG/graph.json","events_json":"https://pith.science/api/pith-number/E6UHICXDXN3C4PUAUEKJWLAISG/events.json","paper":"https://pith.science/paper/E6UHICXD"},"agent_actions":{"view_html":"https://pith.science/pith/E6UHICXDXN3C4PUAUEKJWLAISG","download_json":"https://pith.science/pith/E6UHICXDXN3C4PUAUEKJWLAISG.json","view_paper":"https://pith.science/paper/E6UHICXD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.09481&json=true","fetch_graph":"https://pith.science/api/pith-number/E6UHICXDXN3C4PUAUEKJWLAISG/graph.json","fetch_events":"https://pith.science/api/pith-number/E6UHICXDXN3C4PUAUEKJWLAISG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E6UHICXDXN3C4PUAUEKJWLAISG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E6UHICXDXN3C4PUAUEKJWLAISG/action/storage_attestation","attest_author":"https://pith.science/pith/E6UHICXDXN3C4PUAUEKJWLAISG/action/author_attestation","sign_citation":"https://pith.science/pith/E6UHICXDXN3C4PUAUEKJWLAISG/action/citation_signature","submit_replication":"https://pith.science/pith/E6UHICXDXN3C4PUAUEKJWLAISG/action/replication_record"}},"created_at":"2026-07-05T09:37:28.221819+00:00","updated_at":"2026-07-05T09:37:28.221819+00:00"}