{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:IAMUMPDOVBSY42HWXK32PQJRRN","short_pith_number":"pith:IAMUMPDO","schema_version":"1.0","canonical_sha256":"4019463c6ea8658e68f6bab7a7c1318b5a50f71fc27860c170827a1bad8d13ba","source":{"kind":"arxiv","id":"2203.00397","version":1},"attestation_state":"computed","paper":{"title":"A Theory of Abstraction in Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"David Abel","submitted_at":"2022-03-01T12:46:28Z","abstract_excerpt":"Reinforcement learning defines the problem facing agents that learn to make good decisions through action and observation alone. To be effective problem solvers, such agents must efficiently explore vast worlds, assign credit from delayed feedback, and generalize to new experiences, all while making use of limited data, computational resources, and perceptual bandwidth. Abstraction is essential to all of these endeavors. Through abstraction, agents can form concise models of their environment that support the many practices required of a rational, adaptive decision maker. In this dissertation,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.00397","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-03-01T12:46:28Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"22498ed572b457bd1cf2134daababe3fe893f2726753a3761f0ed9eed68260c1","abstract_canon_sha256":"8270ace309284189018eff0f8e70ffc7f495e4f592eaa9cb7b664a54e0bc035a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:01:16.813029Z","signature_b64":"UBD8Qn9kKJN1MDRrSH4SkxKMGnLKrV3Ot6mgs3+CUNaWVv9NSmZkEV0PvYG4uVVU0ibdYf0Q8GqIOrX0M/JVBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4019463c6ea8658e68f6bab7a7c1318b5a50f71fc27860c170827a1bad8d13ba","last_reissued_at":"2026-07-05T04:01:16.812562Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:01:16.812562Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Theory of Abstraction in Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"David Abel","submitted_at":"2022-03-01T12:46:28Z","abstract_excerpt":"Reinforcement learning defines the problem facing agents that learn to make good decisions through action and observation alone. To be effective problem solvers, such agents must efficiently explore vast worlds, assign credit from delayed feedback, and generalize to new experiences, all while making use of limited data, computational resources, and perceptual bandwidth. Abstraction is essential to all of these endeavors. Through abstraction, agents can form concise models of their environment that support the many practices required of a rational, adaptive decision maker. In this dissertation,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.00397","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.00397/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.00397","created_at":"2026-07-05T04:01:16.812621+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.00397v1","created_at":"2026-07-05T04:01:16.812621+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.00397","created_at":"2026-07-05T04:01:16.812621+00:00"},{"alias_kind":"pith_short_12","alias_value":"IAMUMPDOVBSY","created_at":"2026-07-05T04:01:16.812621+00:00"},{"alias_kind":"pith_short_16","alias_value":"IAMUMPDOVBSY42HW","created_at":"2026-07-05T04:01:16.812621+00:00"},{"alias_kind":"pith_short_8","alias_value":"IAMUMPDO","created_at":"2026-07-05T04:01:16.812621+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00034","citing_title":"Bayesian updates from coalgebraic determinisation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15289","citing_title":"Abstract Sim2Real through Approximate Information States","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IAMUMPDOVBSY42HWXK32PQJRRN","json":"https://pith.science/pith/IAMUMPDOVBSY42HWXK32PQJRRN.json","graph_json":"https://pith.science/api/pith-number/IAMUMPDOVBSY42HWXK32PQJRRN/graph.json","events_json":"https://pith.science/api/pith-number/IAMUMPDOVBSY42HWXK32PQJRRN/events.json","paper":"https://pith.science/paper/IAMUMPDO"},"agent_actions":{"view_html":"https://pith.science/pith/IAMUMPDOVBSY42HWXK32PQJRRN","download_json":"https://pith.science/pith/IAMUMPDOVBSY42HWXK32PQJRRN.json","view_paper":"https://pith.science/paper/IAMUMPDO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.00397&json=true","fetch_graph":"https://pith.science/api/pith-number/IAMUMPDOVBSY42HWXK32PQJRRN/graph.json","fetch_events":"https://pith.science/api/pith-number/IAMUMPDOVBSY42HWXK32PQJRRN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IAMUMPDOVBSY42HWXK32PQJRRN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IAMUMPDOVBSY42HWXK32PQJRRN/action/storage_attestation","attest_author":"https://pith.science/pith/IAMUMPDOVBSY42HWXK32PQJRRN/action/author_attestation","sign_citation":"https://pith.science/pith/IAMUMPDOVBSY42HWXK32PQJRRN/action/citation_signature","submit_replication":"https://pith.science/pith/IAMUMPDOVBSY42HWXK32PQJRRN/action/replication_record"}},"created_at":"2026-07-05T04:01:16.812621+00:00","updated_at":"2026-07-05T04:01:16.812621+00:00"}