{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:AZ6XIVV6ZVRDKDRFBMWGPR6KL6","short_pith_number":"pith:AZ6XIVV6","schema_version":"1.0","canonical_sha256":"067d7456becd62350e250b2c67c7ca5f8143581eef6d7a663577eda5550c8860","source":{"kind":"arxiv","id":"2501.06937","version":1},"attestation_state":"computed","paper":{"title":"An Empirical Study of Deep Reinforcement Learning in Continuing Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Dmytro Korenkevych, Yi Wan, Zheqing Zhu","submitted_at":"2025-01-12T21:24:27Z","abstract_excerpt":"In reinforcement learning (RL), continuing tasks refer to tasks where the agent-environment interaction is ongoing and can not be broken down into episodes. These tasks are suitable when environment resets are unavailable, agent-controlled, or predefined but where all rewards-including those beyond resets-are critical. These scenarios frequently occur in real-world applications and can not be modeled by episodic tasks. While modern deep RL algorithms have been extensively studied and well understood in episodic tasks, their behavior in continuing tasks remains underexplored. To address this ga"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.06937","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-01-12T21:24:27Z","cross_cats_sorted":[],"title_canon_sha256":"cafecc3b966eb9905473c37b874f64bd98bfcc9772a0ecad9f9d1f197b57642e","abstract_canon_sha256":"6246057f7e1764e83d776c0534387ea8fd932a3631dcb4ca3a3ad126f81101ee"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:00:12.626984Z","signature_b64":"G/I5tTarqhrwG0/JzKThaau2xqsc+YcOoPXYDpgZa1nWN2sM0cY9xIvPB1NpEGtabjqq7lsmG0yvzPMJKJLvBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"067d7456becd62350e250b2c67c7ca5f8143581eef6d7a663577eda5550c8860","last_reissued_at":"2026-07-05T10:00:12.626468Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:00:12.626468Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Empirical Study of Deep Reinforcement Learning in Continuing Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Dmytro Korenkevych, Yi Wan, Zheqing Zhu","submitted_at":"2025-01-12T21:24:27Z","abstract_excerpt":"In reinforcement learning (RL), continuing tasks refer to tasks where the agent-environment interaction is ongoing and can not be broken down into episodes. These tasks are suitable when environment resets are unavailable, agent-controlled, or predefined but where all rewards-including those beyond resets-are critical. These scenarios frequently occur in real-world applications and can not be modeled by episodic tasks. While modern deep RL algorithms have been extensively studied and well understood in episodic tasks, their behavior in continuing tasks remains underexplored. To address this ga"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.06937","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.06937/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.06937","created_at":"2026-07-05T10:00:12.626526+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.06937v1","created_at":"2026-07-05T10:00:12.626526+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.06937","created_at":"2026-07-05T10:00:12.626526+00:00"},{"alias_kind":"pith_short_12","alias_value":"AZ6XIVV6ZVRD","created_at":"2026-07-05T10:00:12.626526+00:00"},{"alias_kind":"pith_short_16","alias_value":"AZ6XIVV6ZVRDKDRF","created_at":"2026-07-05T10:00:12.626526+00:00"},{"alias_kind":"pith_short_8","alias_value":"AZ6XIVV6","created_at":"2026-07-05T10:00:12.626526+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AZ6XIVV6ZVRDKDRFBMWGPR6KL6","json":"https://pith.science/pith/AZ6XIVV6ZVRDKDRFBMWGPR6KL6.json","graph_json":"https://pith.science/api/pith-number/AZ6XIVV6ZVRDKDRFBMWGPR6KL6/graph.json","events_json":"https://pith.science/api/pith-number/AZ6XIVV6ZVRDKDRFBMWGPR6KL6/events.json","paper":"https://pith.science/paper/AZ6XIVV6"},"agent_actions":{"view_html":"https://pith.science/pith/AZ6XIVV6ZVRDKDRFBMWGPR6KL6","download_json":"https://pith.science/pith/AZ6XIVV6ZVRDKDRFBMWGPR6KL6.json","view_paper":"https://pith.science/paper/AZ6XIVV6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.06937&json=true","fetch_graph":"https://pith.science/api/pith-number/AZ6XIVV6ZVRDKDRFBMWGPR6KL6/graph.json","fetch_events":"https://pith.science/api/pith-number/AZ6XIVV6ZVRDKDRFBMWGPR6KL6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AZ6XIVV6ZVRDKDRFBMWGPR6KL6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AZ6XIVV6ZVRDKDRFBMWGPR6KL6/action/storage_attestation","attest_author":"https://pith.science/pith/AZ6XIVV6ZVRDKDRFBMWGPR6KL6/action/author_attestation","sign_citation":"https://pith.science/pith/AZ6XIVV6ZVRDKDRFBMWGPR6KL6/action/citation_signature","submit_replication":"https://pith.science/pith/AZ6XIVV6ZVRDKDRFBMWGPR6KL6/action/replication_record"}},"created_at":"2026-07-05T10:00:12.626526+00:00","updated_at":"2026-07-05T10:00:12.626526+00:00"}