{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YOFKJD4BCFQ76EFA3BT2BZB424","short_pith_number":"pith:YOFKJD4B","schema_version":"1.0","canonical_sha256":"c38aa48f811161ff10a0d867a0e43cd71ad60fc7f84dfe6a1c9055a3d5075e6f","source":{"kind":"arxiv","id":"2312.11669","version":1},"attestation_state":"computed","paper":{"title":"Prediction and Control in Continual Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Doina Precup, Nishanth Anand","submitted_at":"2023-12-18T19:23:42Z","abstract_excerpt":"Temporal difference (TD) learning is often used to update the estimate of the value function which is used by RL agents to extract useful policies. In this paper, we focus on value function estimation in continual reinforcement learning. We propose to decompose the value function into two components which update at different timescales: a permanent value function, which holds general knowledge that persists over time, and a transient value function, which allows quick adaptation to new situations. We establish theoretical results showing that our approach is well suited for continual learning "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.11669","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-12-18T19:23:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a6b4d615212e28383ae0bfffdcbf5fa014e90888b605852d52bdc9bce19524d2","abstract_canon_sha256":"8a8918540972cd6414e55da05cb2a8d3e0e390af6dff33d8e9c13b3cf66c465b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:25:52.511010Z","signature_b64":"YmUJ8Ta3bE5kl5puafaVJxICqdtK61JY3bTlDwS5rkUQPMspzZ/+M1wV108CMgaRckmHQMBed/JKJzJuSdl3Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c38aa48f811161ff10a0d867a0e43cd71ad60fc7f84dfe6a1c9055a3d5075e6f","last_reissued_at":"2026-07-05T07:25:52.510537Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:25:52.510537Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Prediction and Control in Continual Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Doina Precup, Nishanth Anand","submitted_at":"2023-12-18T19:23:42Z","abstract_excerpt":"Temporal difference (TD) learning is often used to update the estimate of the value function which is used by RL agents to extract useful policies. In this paper, we focus on value function estimation in continual reinforcement learning. We propose to decompose the value function into two components which update at different timescales: a permanent value function, which holds general knowledge that persists over time, and a transient value function, which allows quick adaptation to new situations. We establish theoretical results showing that our approach is well suited for continual learning "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.11669","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.11669/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.11669","created_at":"2026-07-05T07:25:52.510594+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.11669v1","created_at":"2026-07-05T07:25:52.510594+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.11669","created_at":"2026-07-05T07:25:52.510594+00:00"},{"alias_kind":"pith_short_12","alias_value":"YOFKJD4BCFQ7","created_at":"2026-07-05T07:25:52.510594+00:00"},{"alias_kind":"pith_short_16","alias_value":"YOFKJD4BCFQ76EFA","created_at":"2026-07-05T07:25:52.510594+00:00"},{"alias_kind":"pith_short_8","alias_value":"YOFKJD4B","created_at":"2026-07-05T07:25:52.510594+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.12484","citing_title":"Learning, Fast and Slow: Towards LLMs That Adapt Continually","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12484","citing_title":"Learning, Fast and Slow: Towards LLMs That Adapt Continually","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YOFKJD4BCFQ76EFA3BT2BZB424","json":"https://pith.science/pith/YOFKJD4BCFQ76EFA3BT2BZB424.json","graph_json":"https://pith.science/api/pith-number/YOFKJD4BCFQ76EFA3BT2BZB424/graph.json","events_json":"https://pith.science/api/pith-number/YOFKJD4BCFQ76EFA3BT2BZB424/events.json","paper":"https://pith.science/paper/YOFKJD4B"},"agent_actions":{"view_html":"https://pith.science/pith/YOFKJD4BCFQ76EFA3BT2BZB424","download_json":"https://pith.science/pith/YOFKJD4BCFQ76EFA3BT2BZB424.json","view_paper":"https://pith.science/paper/YOFKJD4B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.11669&json=true","fetch_graph":"https://pith.science/api/pith-number/YOFKJD4BCFQ76EFA3BT2BZB424/graph.json","fetch_events":"https://pith.science/api/pith-number/YOFKJD4BCFQ76EFA3BT2BZB424/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YOFKJD4BCFQ76EFA3BT2BZB424/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YOFKJD4BCFQ76EFA3BT2BZB424/action/storage_attestation","attest_author":"https://pith.science/pith/YOFKJD4BCFQ76EFA3BT2BZB424/action/author_attestation","sign_citation":"https://pith.science/pith/YOFKJD4BCFQ76EFA3BT2BZB424/action/citation_signature","submit_replication":"https://pith.science/pith/YOFKJD4BCFQ76EFA3BT2BZB424/action/replication_record"}},"created_at":"2026-07-05T07:25:52.510594+00:00","updated_at":"2026-07-05T07:25:52.510594+00:00"}