{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:APXYEYQPPHUIAA4E7KU7473GVS","short_pith_number":"pith:APXYEYQP","schema_version":"1.0","canonical_sha256":"03ef82620f79e8800384faa9fe7f66ac858571547c766ca738bc7664c14b2f61","source":{"kind":"arxiv","id":"2204.05928","version":2},"attestation_state":"computed","paper":{"title":"Dynamic Dialogue Policy for Continual Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Carel van Niekerk, Christian Geishauser, Hsien-Chin Lin, Michael Heck, Milica Ga\\v{s}i\\'c, Nurul Lubis, Shutong Feng","submitted_at":"2022-04-12T16:30:40Z","abstract_excerpt":"Continual learning is one of the key components of human learning and a necessary requirement of artificial intelligence. As dialogue can potentially span infinitely many topics and tasks, a task-oriented dialogue system must have the capability to continually learn, dynamically adapting to new challenges while preserving the knowledge it already acquired. Despite the importance, continual reinforcement learning of the dialogue policy has remained largely unaddressed. The lack of a framework with training protocols, baseline models and suitable metrics, has so far hindered research in this dir"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.05928","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-04-12T16:30:40Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"53383aed4c6b186f62f5ea581898247cdad2c2eca97a38a6e4e2f36e10929ee4","abstract_canon_sha256":"f8d1ddd7eb8ab14302843b7c93af4aa7c8211df032e0a0842d20202dc3a11fb6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:04:47.864801Z","signature_b64":"VbWch740LVqQk2Llq7iPkaWYbPLI/zJu5onVNS8oiKLbrU46AbvGjUBxgYffd9RXR9e4z41t/UHuvHOXodwTDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"03ef82620f79e8800384faa9fe7f66ac858571547c766ca738bc7664c14b2f61","last_reissued_at":"2026-07-05T05:04:47.864346Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:04:47.864346Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dynamic Dialogue Policy for Continual Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Carel van Niekerk, Christian Geishauser, Hsien-Chin Lin, Michael Heck, Milica Ga\\v{s}i\\'c, Nurul Lubis, Shutong Feng","submitted_at":"2022-04-12T16:30:40Z","abstract_excerpt":"Continual learning is one of the key components of human learning and a necessary requirement of artificial intelligence. As dialogue can potentially span infinitely many topics and tasks, a task-oriented dialogue system must have the capability to continually learn, dynamically adapting to new challenges while preserving the knowledge it already acquired. Despite the importance, continual reinforcement learning of the dialogue policy has remained largely unaddressed. The lack of a framework with training protocols, baseline models and suitable metrics, has so far hindered research in this dir"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.05928","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.05928/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.05928","created_at":"2026-07-05T05:04:47.864405+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.05928v2","created_at":"2026-07-05T05:04:47.864405+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.05928","created_at":"2026-07-05T05:04:47.864405+00:00"},{"alias_kind":"pith_short_12","alias_value":"APXYEYQPPHUI","created_at":"2026-07-05T05:04:47.864405+00:00"},{"alias_kind":"pith_short_16","alias_value":"APXYEYQPPHUIAA4E","created_at":"2026-07-05T05:04:47.864405+00:00"},{"alias_kind":"pith_short_8","alias_value":"APXYEYQP","created_at":"2026-07-05T05:04:47.864405+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.00143","citing_title":"Regime-Adaptive Continual Learning for Portfolio Management","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/APXYEYQPPHUIAA4E7KU7473GVS","json":"https://pith.science/pith/APXYEYQPPHUIAA4E7KU7473GVS.json","graph_json":"https://pith.science/api/pith-number/APXYEYQPPHUIAA4E7KU7473GVS/graph.json","events_json":"https://pith.science/api/pith-number/APXYEYQPPHUIAA4E7KU7473GVS/events.json","paper":"https://pith.science/paper/APXYEYQP"},"agent_actions":{"view_html":"https://pith.science/pith/APXYEYQPPHUIAA4E7KU7473GVS","download_json":"https://pith.science/pith/APXYEYQPPHUIAA4E7KU7473GVS.json","view_paper":"https://pith.science/paper/APXYEYQP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.05928&json=true","fetch_graph":"https://pith.science/api/pith-number/APXYEYQPPHUIAA4E7KU7473GVS/graph.json","fetch_events":"https://pith.science/api/pith-number/APXYEYQPPHUIAA4E7KU7473GVS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/APXYEYQPPHUIAA4E7KU7473GVS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/APXYEYQPPHUIAA4E7KU7473GVS/action/storage_attestation","attest_author":"https://pith.science/pith/APXYEYQPPHUIAA4E7KU7473GVS/action/author_attestation","sign_citation":"https://pith.science/pith/APXYEYQPPHUIAA4E7KU7473GVS/action/citation_signature","submit_replication":"https://pith.science/pith/APXYEYQPPHUIAA4E7KU7473GVS/action/replication_record"}},"created_at":"2026-07-05T05:04:47.864405+00:00","updated_at":"2026-07-05T05:04:47.864405+00:00"}