{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RY3RMCDYBQAN6U2ORD42AYWG5E","short_pith_number":"pith:RY3RMCDY","schema_version":"1.0","canonical_sha256":"8e371608780c00df534e88f9a062c6e906f48c195d073373accb44e9bbafacd9","source":{"kind":"arxiv","id":"2303.07109","version":1},"attestation_state":"computed","paper":{"title":"Transformer-based World Models Are Happy With 100k Interactions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Jan Robine, Marc H\\\"oftmann, Stefan Harmeling, Tobias Uelwer","submitted_at":"2023-03-13T13:43:59Z","abstract_excerpt":"Deep neural networks have been successful in many reinforcement learning settings. However, compared to human learners they are overly data hungry. To build a sample-efficient world model, we apply a transformer to real-world episodes in an autoregressive manner: not only the compact latent states and the taken actions but also the experienced or predicted rewards are fed into the transformer, so that it can attend flexibly to all three modalities at different time steps. The transformer allows our world model to access previous states directly, instead of viewing them through a compressed rec"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.07109","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-03-13T13:43:59Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"d5df4af1476f67b707e18c84b2ceda8092a544199933d7634a17864c269ba397","abstract_canon_sha256":"1b2c42e6668b2d464b641f1c035c28ae7c7feefbfcfd8060a9a8bf27b9352c61"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:50:26.837125Z","signature_b64":"qLPGzCOtWfeEXmqtZBrGEUykbujc6zEfdYnuhxINg+daJ8SJs0bEhJv2z4giadRDjjkBZcGIYOc7SfRpXBF0AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8e371608780c00df534e88f9a062c6e906f48c195d073373accb44e9bbafacd9","last_reissued_at":"2026-07-05T05:50:26.836776Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:50:26.836776Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Transformer-based World Models Are Happy With 100k Interactions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Jan Robine, Marc H\\\"oftmann, Stefan Harmeling, Tobias Uelwer","submitted_at":"2023-03-13T13:43:59Z","abstract_excerpt":"Deep neural networks have been successful in many reinforcement learning settings. However, compared to human learners they are overly data hungry. To build a sample-efficient world model, we apply a transformer to real-world episodes in an autoregressive manner: not only the compact latent states and the taken actions but also the experienced or predicted rewards are fed into the transformer, so that it can attend flexibly to all three modalities at different time steps. The transformer allows our world model to access previous states directly, instead of viewing them through a compressed rec"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.07109","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.07109/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.07109","created_at":"2026-07-05T05:50:26.836831+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.07109v1","created_at":"2026-07-05T05:50:26.836831+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.07109","created_at":"2026-07-05T05:50:26.836831+00:00"},{"alias_kind":"pith_short_12","alias_value":"RY3RMCDYBQAN","created_at":"2026-07-05T05:50:26.836831+00:00"},{"alias_kind":"pith_short_16","alias_value":"RY3RMCDYBQAN6U2O","created_at":"2026-07-05T05:50:26.836831+00:00"},{"alias_kind":"pith_short_8","alias_value":"RY3RMCDY","created_at":"2026-07-05T05:50:26.836831+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.31529","citing_title":"SVI-Bench: A Dynamic Microworld for Strategic Video Intelligence","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17537","citing_title":"Self-supervised Hierarchical Visual Reasoning with World Model","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31529","citing_title":"SVI-Bench: A Dynamic Microworld for Strategic Video Intelligence","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00780","citing_title":"Behavior-Invariant Task Representation Learning with Transformer-based World Models for Offline Meta-Reinforcement Learning","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2502.11537","citing_title":"Simulus: Combining Improvements in Sample-Efficient World Model Agents","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17537","citing_title":"Self-supervised Hierarchical Visual Reasoning with World Model","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2508.17588","citing_title":"HERO: Hierarchical Extrapolation and Refresh for Efficient World Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2509.19538","citing_title":"DAWM: Diffusion Action World Models for Offline Reinforcement Learning via Action-Inferred Transitions","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2411.04983","citing_title":"DINO-WM: World Models on Pre-trained Visual Features enable Zero-shot Planning","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2309.16797","citing_title":"Promptbreeder: Self-Referential Self-Improvement Via Prompt Evolution","ref_index":276,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24527","citing_title":"Training Agents Inside of Scalable World Models","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13013","citing_title":"JEDI: Joint Embedding Diffusion World Model for Online Model-Based Reinforcement Learning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26182","citing_title":"Lifting Embodied World Models for Planning and Control","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24661","citing_title":"Agent-Centric Observation Adaptation for Robust Visual Control under Dynamic Perturbations","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09035","citing_title":"Advantage-Guided Diffusion for Model-Based Reinforcement Learning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24661","citing_title":"Agent-Centric Observation Adaptation for Robust Visual Control under Dynamic Perturbations","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2501.03575","citing_title":"Cosmos World Foundation Model Platform for Physical AI","ref_index":166,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RY3RMCDYBQAN6U2ORD42AYWG5E","json":"https://pith.science/pith/RY3RMCDYBQAN6U2ORD42AYWG5E.json","graph_json":"https://pith.science/api/pith-number/RY3RMCDYBQAN6U2ORD42AYWG5E/graph.json","events_json":"https://pith.science/api/pith-number/RY3RMCDYBQAN6U2ORD42AYWG5E/events.json","paper":"https://pith.science/paper/RY3RMCDY"},"agent_actions":{"view_html":"https://pith.science/pith/RY3RMCDYBQAN6U2ORD42AYWG5E","download_json":"https://pith.science/pith/RY3RMCDYBQAN6U2ORD42AYWG5E.json","view_paper":"https://pith.science/paper/RY3RMCDY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.07109&json=true","fetch_graph":"https://pith.science/api/pith-number/RY3RMCDYBQAN6U2ORD42AYWG5E/graph.json","fetch_events":"https://pith.science/api/pith-number/RY3RMCDYBQAN6U2ORD42AYWG5E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RY3RMCDYBQAN6U2ORD42AYWG5E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RY3RMCDYBQAN6U2ORD42AYWG5E/action/storage_attestation","attest_author":"https://pith.science/pith/RY3RMCDYBQAN6U2ORD42AYWG5E/action/author_attestation","sign_citation":"https://pith.science/pith/RY3RMCDYBQAN6U2ORD42AYWG5E/action/citation_signature","submit_replication":"https://pith.science/pith/RY3RMCDYBQAN6U2ORD42AYWG5E/action/replication_record"}},"created_at":"2026-07-05T05:50:26.836831+00:00","updated_at":"2026-07-05T05:50:26.836831+00:00"}