{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:XUUN3G4J2JO4MMIJ2OB4KD5RBC","short_pith_number":"pith:XUUN3G4J","schema_version":"1.0","canonical_sha256":"bd28dd9b89d25dc63109d383c50fb108a5f5045c6aaf44baa08f760eb99b105c","source":{"kind":"arxiv","id":"2007.02418","version":3},"attestation_state":"computed","paper":{"title":"Selective Dyna-style Planning Under Limited Model Capacity","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Erin J. Talvitie, Martha White, Samuel Sokota, Zaheer Abbas","submitted_at":"2020-07-05T18:51:50Z","abstract_excerpt":"In model-based reinforcement learning, planning with an imperfect model of the environment has the potential to harm learning progress. But even when a model is imperfect, it may still contain information that is useful for planning. In this paper, we investigate the idea of using an imperfect model selectively. The agent should plan in parts of the state space where the model would be helpful but refrain from using the model where it would be harmful. An effective selective planning mechanism requires estimating predictive uncertainty, which arises out of aleatoric uncertainty, parameter unce"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2007.02418","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-07-05T18:51:50Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"7689908994aa798fd60832985e59dc9229e150429b1bf25065f8a22f9ccb1236","abstract_canon_sha256":"675be0cc914970f51013128b9e5d3b7de44e0989ce83b2b05f00e1946e4da690"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:20:55.687577Z","signature_b64":"UZyH1fDwIj4BQa4btRgJHpTC3LB3wYn4mr32V2UyeVYsCz4LEjtuPeNjLeD98gq0y0zn6e/pr+M37i6MSmQBAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bd28dd9b89d25dc63109d383c50fb108a5f5045c6aaf44baa08f760eb99b105c","last_reissued_at":"2026-07-05T02:20:55.687135Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:20:55.687135Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Selective Dyna-style Planning Under Limited Model Capacity","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Erin J. Talvitie, Martha White, Samuel Sokota, Zaheer Abbas","submitted_at":"2020-07-05T18:51:50Z","abstract_excerpt":"In model-based reinforcement learning, planning with an imperfect model of the environment has the potential to harm learning progress. But even when a model is imperfect, it may still contain information that is useful for planning. In this paper, we investigate the idea of using an imperfect model selectively. The agent should plan in parts of the state space where the model would be helpful but refrain from using the model where it would be harmful. An effective selective planning mechanism requires estimating predictive uncertainty, which arises out of aleatoric uncertainty, parameter unce"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2007.02418","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2007.02418/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2007.02418","created_at":"2026-07-05T02:20:55.687191+00:00"},{"alias_kind":"arxiv_version","alias_value":"2007.02418v3","created_at":"2026-07-05T02:20:55.687191+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2007.02418","created_at":"2026-07-05T02:20:55.687191+00:00"},{"alias_kind":"pith_short_12","alias_value":"XUUN3G4J2JO4","created_at":"2026-07-05T02:20:55.687191+00:00"},{"alias_kind":"pith_short_16","alias_value":"XUUN3G4J2JO4MMIJ","created_at":"2026-07-05T02:20:55.687191+00:00"},{"alias_kind":"pith_short_8","alias_value":"XUUN3G4J","created_at":"2026-07-05T02:20:55.687191+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XUUN3G4J2JO4MMIJ2OB4KD5RBC","json":"https://pith.science/pith/XUUN3G4J2JO4MMIJ2OB4KD5RBC.json","graph_json":"https://pith.science/api/pith-number/XUUN3G4J2JO4MMIJ2OB4KD5RBC/graph.json","events_json":"https://pith.science/api/pith-number/XUUN3G4J2JO4MMIJ2OB4KD5RBC/events.json","paper":"https://pith.science/paper/XUUN3G4J"},"agent_actions":{"view_html":"https://pith.science/pith/XUUN3G4J2JO4MMIJ2OB4KD5RBC","download_json":"https://pith.science/pith/XUUN3G4J2JO4MMIJ2OB4KD5RBC.json","view_paper":"https://pith.science/paper/XUUN3G4J","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2007.02418&json=true","fetch_graph":"https://pith.science/api/pith-number/XUUN3G4J2JO4MMIJ2OB4KD5RBC/graph.json","fetch_events":"https://pith.science/api/pith-number/XUUN3G4J2JO4MMIJ2OB4KD5RBC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XUUN3G4J2JO4MMIJ2OB4KD5RBC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XUUN3G4J2JO4MMIJ2OB4KD5RBC/action/storage_attestation","attest_author":"https://pith.science/pith/XUUN3G4J2JO4MMIJ2OB4KD5RBC/action/author_attestation","sign_citation":"https://pith.science/pith/XUUN3G4J2JO4MMIJ2OB4KD5RBC/action/citation_signature","submit_replication":"https://pith.science/pith/XUUN3G4J2JO4MMIJ2OB4KD5RBC/action/replication_record"}},"created_at":"2026-07-05T02:20:55.687191+00:00","updated_at":"2026-07-05T02:20:55.687191+00:00"}