{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:2WD4A6QA44F5TMGWF6AYCTAO5R","short_pith_number":"pith:2WD4A6QA","schema_version":"1.0","canonical_sha256":"d587c07a00e70bd9b0d62f81814c0eec47d893df81937e96f1a1c5f0fd9b647f","source":{"kind":"arxiv","id":"2211.02222","version":3},"attestation_state":"computed","paper":{"title":"The Benefits of Model-Based Generalization in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aditya Ramesh, J\\\"urgen Schmidhuber, Kenny Young, Louis Kirsch","submitted_at":"2022-11-04T02:10:35Z","abstract_excerpt":"Model-Based Reinforcement Learning (RL) is widely believed to have the potential to improve sample efficiency by allowing an agent to synthesize large amounts of imagined experience. Experience Replay (ER) can be considered a simple kind of model, which has proved effective at improving the stability and efficiency of deep RL. In principle, a learned parametric model could improve on ER by generalizing from real experience to augment the dataset with additional plausible experience. However, given that learned value functions can also generalize, it is not immediately obvious why model general"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.02222","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-11-04T02:10:35Z","cross_cats_sorted":[],"title_canon_sha256":"e0b2be558b15bbe796e398f2bcebddd44662e28aa2efd20e6ca2a5770396f45b","abstract_canon_sha256":"ed27654126e7ec333c4b9e6b0c5045881128498ce88d52440830021f2189e181"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:28:56.883271Z","signature_b64":"0ohHunIMtu0vHfE6n3pB39tAs1Dm3vfsWrvLs0X0gLD0uSJkn8MZ2G6p4kGXlYkYE8VPwo9EwZLtcrAT4UCECA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d587c07a00e70bd9b0d62f81814c0eec47d893df81937e96f1a1c5f0fd9b647f","last_reissued_at":"2026-07-05T06:28:56.882779Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:28:56.882779Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Benefits of Model-Based Generalization in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aditya Ramesh, J\\\"urgen Schmidhuber, Kenny Young, Louis Kirsch","submitted_at":"2022-11-04T02:10:35Z","abstract_excerpt":"Model-Based Reinforcement Learning (RL) is widely believed to have the potential to improve sample efficiency by allowing an agent to synthesize large amounts of imagined experience. Experience Replay (ER) can be considered a simple kind of model, which has proved effective at improving the stability and efficiency of deep RL. In principle, a learned parametric model could improve on ER by generalizing from real experience to augment the dataset with additional plausible experience. However, given that learned value functions can also generalize, it is not immediately obvious why model general"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.02222","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.02222/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.02222","created_at":"2026-07-05T06:28:56.882833+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.02222v3","created_at":"2026-07-05T06:28:56.882833+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.02222","created_at":"2026-07-05T06:28:56.882833+00:00"},{"alias_kind":"pith_short_12","alias_value":"2WD4A6QA44F5","created_at":"2026-07-05T06:28:56.882833+00:00"},{"alias_kind":"pith_short_16","alias_value":"2WD4A6QA44F5TMGW","created_at":"2026-07-05T06:28:56.882833+00:00"},{"alias_kind":"pith_short_8","alias_value":"2WD4A6QA","created_at":"2026-07-05T06:28:56.882833+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.03413","citing_title":"Learning to Theorize the World from Observation","ref_index":288,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02699","citing_title":"Learning Equivariant Neural-Augmented Object Dynamics From Few Interactions","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2WD4A6QA44F5TMGWF6AYCTAO5R","json":"https://pith.science/pith/2WD4A6QA44F5TMGWF6AYCTAO5R.json","graph_json":"https://pith.science/api/pith-number/2WD4A6QA44F5TMGWF6AYCTAO5R/graph.json","events_json":"https://pith.science/api/pith-number/2WD4A6QA44F5TMGWF6AYCTAO5R/events.json","paper":"https://pith.science/paper/2WD4A6QA"},"agent_actions":{"view_html":"https://pith.science/pith/2WD4A6QA44F5TMGWF6AYCTAO5R","download_json":"https://pith.science/pith/2WD4A6QA44F5TMGWF6AYCTAO5R.json","view_paper":"https://pith.science/paper/2WD4A6QA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.02222&json=true","fetch_graph":"https://pith.science/api/pith-number/2WD4A6QA44F5TMGWF6AYCTAO5R/graph.json","fetch_events":"https://pith.science/api/pith-number/2WD4A6QA44F5TMGWF6AYCTAO5R/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2WD4A6QA44F5TMGWF6AYCTAO5R/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2WD4A6QA44F5TMGWF6AYCTAO5R/action/storage_attestation","attest_author":"https://pith.science/pith/2WD4A6QA44F5TMGWF6AYCTAO5R/action/author_attestation","sign_citation":"https://pith.science/pith/2WD4A6QA44F5TMGWF6AYCTAO5R/action/citation_signature","submit_replication":"https://pith.science/pith/2WD4A6QA44F5TMGWF6AYCTAO5R/action/replication_record"}},"created_at":"2026-07-05T06:28:56.882833+00:00","updated_at":"2026-07-05T06:28:56.882833+00:00"}