{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:SER3IRDUJPFXLCJ3UPEJ4F52ZJ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"4e9e5f4a0e08071f6a92579f538e717fec01a147db5d3d8b50ef6760cc2921ef","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-08-20T08:32:11Z","title_canon_sha256":"4b36c2eb89949a1075961830be49e0141f2ed86dcfc91eea7a7648e3b571aca6"},"schema_version":"1.0","source":{"id":"2308.10203","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2308.10203","created_at":"2026-07-05T06:42:53Z"},{"alias_kind":"arxiv_version","alias_value":"2308.10203v1","created_at":"2026-07-05T06:42:53Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.10203","created_at":"2026-07-05T06:42:53Z"},{"alias_kind":"pith_short_12","alias_value":"SER3IRDUJPFX","created_at":"2026-07-05T06:42:53Z"},{"alias_kind":"pith_short_16","alias_value":"SER3IRDUJPFXLCJ3","created_at":"2026-07-05T06:42:53Z"},{"alias_kind":"pith_short_8","alias_value":"SER3IRDU","created_at":"2026-07-05T06:42:53Z"}],"graph_snapshots":[{"event_id":"sha256:775490623227220ca414a6aa11342b8400d44a1a608075d9b635fcb342f4aaad","target":"graph","created_at":"2026-07-05T06:42:53Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2308.10203/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Discrete reinforcement learning (RL) algorithms have demonstrated exceptional performance in solving sequential decision tasks with discrete action spaces, such as Atari games. However, their effectiveness is hindered when applied to continuous control problems due to the challenge of dimensional explosion. In this paper, we present the Soft Decomposed Policy-Critic (SDPC) architecture, which combines soft RL and actor-critic techniques with discrete RL methods to overcome this limitation. SDPC discretizes each action dimension independently and employs a shared critic network to maximize the ","authors_text":"Gang Wang, Jian Sun, Wei Chen, Yechen Zhang, Zhuo Li","cross_cats":["cs.AI"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-08-20T08:32:11Z","title":"Soft Decomposed Policy-Critic: Bridging the Gap for Effective Continuous Control with Discrete RL"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.10203","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:3d6dd3dd72214862d449a8f82903b39534a6be2e78a5c1dc2ec1c3e573c9b29c","target":"record","created_at":"2026-07-05T06:42:53Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"4e9e5f4a0e08071f6a92579f538e717fec01a147db5d3d8b50ef6760cc2921ef","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-08-20T08:32:11Z","title_canon_sha256":"4b36c2eb89949a1075961830be49e0141f2ed86dcfc91eea7a7648e3b571aca6"},"schema_version":"1.0","source":{"id":"2308.10203","kind":"arxiv","version":1}},"canonical_sha256":"9123b444744bcb75893ba3c89e17baca71687be1a3ec6cf3719b3974dfeaf1c0","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"9123b444744bcb75893ba3c89e17baca71687be1a3ec6cf3719b3974dfeaf1c0","first_computed_at":"2026-07-05T06:42:53.116292Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T06:42:53.116292Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"bC6rL2KlqIl6CQXIY7PBVLUHHIz8iyOmxj3cGGyn6xdtfIu/wiMxzCag518HrcJSB8xsxG7VlXzJBlG0aiTEAQ==","signature_status":"signed_v1","signed_at":"2026-07-05T06:42:53.116778Z","signed_message":"canonical_sha256_bytes"},"source_id":"2308.10203","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:3d6dd3dd72214862d449a8f82903b39534a6be2e78a5c1dc2ec1c3e573c9b29c","sha256:775490623227220ca414a6aa11342b8400d44a1a608075d9b635fcb342f4aaad"],"state_sha256":"216c0b3cde04188cb295756c852083b55ba00b262bf3ef7225298719d497fbe4"}