{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:B6ACLNIW5CPBZUMRWHZYJZYFUE","short_pith_number":"pith:B6ACLNIW","schema_version":"1.0","canonical_sha256":"0f8025b516e89e1cd191b1f384e705a12f67442acec9b50e0ed1cf968a00bcaf","source":{"kind":"arxiv","id":"2307.11046","version":2},"attestation_state":"computed","paper":{"title":"A Definition of Continual Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Andr\\'e Barreto, Benjamin Van Roy, David Abel, Doina Precup, Hado van Hasselt, Satinder Singh","submitted_at":"2023-07-20T17:28:01Z","abstract_excerpt":"In a standard view of the reinforcement learning problem, an agent's goal is to efficiently identify a policy that maximizes long-term reward. However, this perspective is based on a restricted view of learning as finding a solution, rather than treating learning as endless adaptation. In contrast, continual reinforcement learning refers to the setting in which the best agents never stop learning. Despite the importance of continual reinforcement learning, the community lacks a simple definition of the problem that highlights its commitments and makes its primary concepts precise and clear. To"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.11046","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-07-20T17:28:01Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"91e714b4b40414d1d7cabb3c6500282bf57a5fe372da0d8f6fc56733c9b767e8","abstract_canon_sha256":"5eb0da7738952c5ea1a1e56a1b2fb021df7323de609e4773b8f0acad3f1a0d32"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:18:59.166726Z","signature_b64":"Pc4MBUGuyLcV2GaWSU8O8gmDtrhs9ACdxxO4nxRl9EtQSC0NH/40KJ7d5T1dU652iIRL9ZIO4XVFyOXnPZNPBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0f8025b516e89e1cd191b1f384e705a12f67442acec9b50e0ed1cf968a00bcaf","last_reissued_at":"2026-07-05T07:18:59.166202Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:18:59.166202Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Definition of Continual Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Andr\\'e Barreto, Benjamin Van Roy, David Abel, Doina Precup, Hado van Hasselt, Satinder Singh","submitted_at":"2023-07-20T17:28:01Z","abstract_excerpt":"In a standard view of the reinforcement learning problem, an agent's goal is to efficiently identify a policy that maximizes long-term reward. However, this perspective is based on a restricted view of learning as finding a solution, rather than treating learning as endless adaptation. In contrast, continual reinforcement learning refers to the setting in which the best agents never stop learning. Despite the importance of continual reinforcement learning, the community lacks a simple definition of the problem that highlights its commitments and makes its primary concepts precise and clear. To"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.11046","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.11046/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.11046","created_at":"2026-07-05T07:18:59.166272+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.11046v2","created_at":"2026-07-05T07:18:59.166272+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.11046","created_at":"2026-07-05T07:18:59.166272+00:00"},{"alias_kind":"pith_short_12","alias_value":"B6ACLNIW5CPB","created_at":"2026-07-05T07:18:59.166272+00:00"},{"alias_kind":"pith_short_16","alias_value":"B6ACLNIW5CPBZUMR","created_at":"2026-07-05T07:18:59.166272+00:00"},{"alias_kind":"pith_short_8","alias_value":"B6ACLNIW","created_at":"2026-07-05T07:18:59.166272+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.07333","citing_title":"Beyond Linear Attention: Softmax Transformers Implement In-Context Reinforcement Learning","ref_index":127,"is_internal_anchor":false},{"citing_arxiv_id":"2511.08717","citing_title":"Optimal control of the future via prospective learning with control","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07333","citing_title":"Beyond Linear Attention: Softmax Transformers Implement In-Context Reinforcement Learning","ref_index":124,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07123","citing_title":"Convergence and Emergence of In-Context Reinforcement Learning with Chain of Thought","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12874","citing_title":"LIFE -- an energy efficient advanced continual learning agentic AI framework for frontier systems","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/B6ACLNIW5CPBZUMRWHZYJZYFUE","json":"https://pith.science/pith/B6ACLNIW5CPBZUMRWHZYJZYFUE.json","graph_json":"https://pith.science/api/pith-number/B6ACLNIW5CPBZUMRWHZYJZYFUE/graph.json","events_json":"https://pith.science/api/pith-number/B6ACLNIW5CPBZUMRWHZYJZYFUE/events.json","paper":"https://pith.science/paper/B6ACLNIW"},"agent_actions":{"view_html":"https://pith.science/pith/B6ACLNIW5CPBZUMRWHZYJZYFUE","download_json":"https://pith.science/pith/B6ACLNIW5CPBZUMRWHZYJZYFUE.json","view_paper":"https://pith.science/paper/B6ACLNIW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.11046&json=true","fetch_graph":"https://pith.science/api/pith-number/B6ACLNIW5CPBZUMRWHZYJZYFUE/graph.json","fetch_events":"https://pith.science/api/pith-number/B6ACLNIW5CPBZUMRWHZYJZYFUE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/B6ACLNIW5CPBZUMRWHZYJZYFUE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/B6ACLNIW5CPBZUMRWHZYJZYFUE/action/storage_attestation","attest_author":"https://pith.science/pith/B6ACLNIW5CPBZUMRWHZYJZYFUE/action/author_attestation","sign_citation":"https://pith.science/pith/B6ACLNIW5CPBZUMRWHZYJZYFUE/action/citation_signature","submit_replication":"https://pith.science/pith/B6ACLNIW5CPBZUMRWHZYJZYFUE/action/replication_record"}},"created_at":"2026-07-05T07:18:59.166272+00:00","updated_at":"2026-07-05T07:18:59.166272+00:00"}