{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:6WZPWCAIPB2CLCIJZDY4YKP2MZ","short_pith_number":"pith:6WZPWCAI","schema_version":"1.0","canonical_sha256":"f5b2fb08087874258909c8f1cc29fa6677c900a4f2a8889e3a4ec818d98f47cf","source":{"kind":"arxiv","id":"2107.12544","version":1},"attestation_state":"computed","paper":{"title":"Human-Level Reinforcement Learning through Theory-Based Modeling, Exploration, and Planning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Andres Campero, Jake Burga, Joao Loula, Joshua B. Tenenbaum, Nathan Foss, Pedro A. Tsividis, Samuel J. Gershman, Thomas Pouncy","submitted_at":"2021-07-27T01:38:13Z","abstract_excerpt":"Reinforcement learning (RL) studies how an agent comes to achieve reward in an environment through interactions over time. Recent advances in machine RL have surpassed human expertise at the world's oldest board games and many classic video games, but they require vast quantities of experience to learn successfully -- none of today's algorithms account for the human ability to learn so many different tasks, so quickly. Here we propose a new approach to this challenge based on a particularly strong form of model-based RL which we call Theory-Based Reinforcement Learning, because it uses human-l"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2107.12544","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2021-07-27T01:38:13Z","cross_cats_sorted":[],"title_canon_sha256":"116ade4d60b2182aa723f4ffaa80fcea9365fa4513c2cc64d7587e735fc23fde","abstract_canon_sha256":"e5c0996046ee088e2c7f8451a9acaf79178bafe93c73513a23974cedb86538fd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:00:57.410793Z","signature_b64":"jIePwDuN4heSlhB2x32C/GrB+P/66X2r0pYh17ksDZ5it2mn3MqAqZHSjFfVy7Z10dq6C+5PYsQaSimGGFLiCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f5b2fb08087874258909c8f1cc29fa6677c900a4f2a8889e3a4ec818d98f47cf","last_reissued_at":"2026-07-05T03:00:57.410308Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:00:57.410308Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Human-Level Reinforcement Learning through Theory-Based Modeling, Exploration, and Planning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Andres Campero, Jake Burga, Joao Loula, Joshua B. Tenenbaum, Nathan Foss, Pedro A. Tsividis, Samuel J. Gershman, Thomas Pouncy","submitted_at":"2021-07-27T01:38:13Z","abstract_excerpt":"Reinforcement learning (RL) studies how an agent comes to achieve reward in an environment through interactions over time. Recent advances in machine RL have surpassed human expertise at the world's oldest board games and many classic video games, but they require vast quantities of experience to learn successfully -- none of today's algorithms account for the human ability to learn so many different tasks, so quickly. Here we propose a new approach to this challenge based on a particularly strong form of model-based RL which we call Theory-Based Reinforcement Learning, because it uses human-l"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2107.12544","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2107.12544/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2107.12544","created_at":"2026-07-05T03:00:57.410380+00:00"},{"alias_kind":"arxiv_version","alias_value":"2107.12544v1","created_at":"2026-07-05T03:00:57.410380+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2107.12544","created_at":"2026-07-05T03:00:57.410380+00:00"},{"alias_kind":"pith_short_12","alias_value":"6WZPWCAIPB2C","created_at":"2026-07-05T03:00:57.410380+00:00"},{"alias_kind":"pith_short_16","alias_value":"6WZPWCAIPB2CLCIJ","created_at":"2026-07-05T03:00:57.410380+00:00"},{"alias_kind":"pith_short_8","alias_value":"6WZPWCAI","created_at":"2026-07-05T03:00:57.410380+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.19352","citing_title":"Brain alignment of reasoning and action representations from vision-language and action models during naturalistic gameplay","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13740","citing_title":"Learning POMDP World Models from Observations with Language-Model Priors","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08019","citing_title":"Reason to Play: Behavioral and Brain Alignment Between Frontier LRMs and Human Game Learners","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6WZPWCAIPB2CLCIJZDY4YKP2MZ","json":"https://pith.science/pith/6WZPWCAIPB2CLCIJZDY4YKP2MZ.json","graph_json":"https://pith.science/api/pith-number/6WZPWCAIPB2CLCIJZDY4YKP2MZ/graph.json","events_json":"https://pith.science/api/pith-number/6WZPWCAIPB2CLCIJZDY4YKP2MZ/events.json","paper":"https://pith.science/paper/6WZPWCAI"},"agent_actions":{"view_html":"https://pith.science/pith/6WZPWCAIPB2CLCIJZDY4YKP2MZ","download_json":"https://pith.science/pith/6WZPWCAIPB2CLCIJZDY4YKP2MZ.json","view_paper":"https://pith.science/paper/6WZPWCAI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2107.12544&json=true","fetch_graph":"https://pith.science/api/pith-number/6WZPWCAIPB2CLCIJZDY4YKP2MZ/graph.json","fetch_events":"https://pith.science/api/pith-number/6WZPWCAIPB2CLCIJZDY4YKP2MZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6WZPWCAIPB2CLCIJZDY4YKP2MZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6WZPWCAIPB2CLCIJZDY4YKP2MZ/action/storage_attestation","attest_author":"https://pith.science/pith/6WZPWCAIPB2CLCIJZDY4YKP2MZ/action/author_attestation","sign_citation":"https://pith.science/pith/6WZPWCAIPB2CLCIJZDY4YKP2MZ/action/citation_signature","submit_replication":"https://pith.science/pith/6WZPWCAIPB2CLCIJZDY4YKP2MZ/action/replication_record"}},"created_at":"2026-07-05T03:00:57.410380+00:00","updated_at":"2026-07-05T03:00:57.410380+00:00"}