{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2017:5FAOIUAIGFRPI3NPIMTQFZUP7F","short_pith_number":"pith:5FAOIUAI","schema_version":"1.0","canonical_sha256":"e940e450083162f46daf432702e68ff97936a6ae1476f218d055e07335fcebc9","source":{"kind":"arxiv","id":"1704.07978","version":6},"attestation_state":"computed","paper":{"title":"On Improving Deep Reinforcement Learning for POMDPs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Guanghui Miao, Pascal Poupart, Pengfei Zhu, Xin Li","submitted_at":"2017-04-26T05:55:07Z","abstract_excerpt":"Deep Reinforcement Learning (RL) recently emerged as one of the most competitive approaches for learning in sequential decision making problems with fully observable environments, e.g., computer Go. However, very little work has been done in deep RL to handle partially observable environments. We propose a new architecture called Action-specific Deep Recurrent Q-Network (ADRQN) to enhance learning performance in partially observable domains. Actions are encoded by a fully connected layer and coupled with a convolutional observation to form an action-observation pair. The time series of action-"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1704.07978","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2017-04-26T05:55:07Z","cross_cats_sorted":[],"title_canon_sha256":"e1dae5b0deda47b3a2c4304f4b60843c73fbe7cbc4629b51f0941518316b931c","abstract_canon_sha256":"0a170b8079b51b71acdc895e2431fa849d974836dd36a5dba2fe81f91c881a54"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:15:05.584830Z","signature_b64":"cg+wX7EZZJq/u1TnImkXgopLIWX3yi1h57piT5Vak+3hOlTD5UOybsWMMb64702IrPqWRFwjFmT8ldXsvvecCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e940e450083162f46daf432702e68ff97936a6ae1476f218d055e07335fcebc9","last_reissued_at":"2026-05-18T00:15:05.584296Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:15:05.584296Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Improving Deep Reinforcement Learning for POMDPs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Guanghui Miao, Pascal Poupart, Pengfei Zhu, Xin Li","submitted_at":"2017-04-26T05:55:07Z","abstract_excerpt":"Deep Reinforcement Learning (RL) recently emerged as one of the most competitive approaches for learning in sequential decision making problems with fully observable environments, e.g., computer Go. However, very little work has been done in deep RL to handle partially observable environments. We propose a new architecture called Action-specific Deep Recurrent Q-Network (ADRQN) to enhance learning performance in partially observable domains. Actions are encoded by a fully connected layer and coupled with a convolutional observation to form an action-observation pair. The time series of action-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1704.07978","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1704.07978","created_at":"2026-05-18T00:15:05.584399+00:00"},{"alias_kind":"arxiv_version","alias_value":"1704.07978v6","created_at":"2026-05-18T00:15:05.584399+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1704.07978","created_at":"2026-05-18T00:15:05.584399+00:00"},{"alias_kind":"pith_short_12","alias_value":"5FAOIUAIGFRP","created_at":"2026-05-18T12:31:00.734936+00:00"},{"alias_kind":"pith_short_16","alias_value":"5FAOIUAIGFRPI3NP","created_at":"2026-05-18T12:31:00.734936+00:00"},{"alias_kind":"pith_short_8","alias_value":"5FAOIUAI","created_at":"2026-05-18T12:31:00.734936+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2605.16318","citing_title":"Investigating Action Encodings in Recurrent Neural Networks in Reinforcement Learning","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5FAOIUAIGFRPI3NPIMTQFZUP7F","json":"https://pith.science/pith/5FAOIUAIGFRPI3NPIMTQFZUP7F.json","graph_json":"https://pith.science/api/pith-number/5FAOIUAIGFRPI3NPIMTQFZUP7F/graph.json","events_json":"https://pith.science/api/pith-number/5FAOIUAIGFRPI3NPIMTQFZUP7F/events.json","paper":"https://pith.science/paper/5FAOIUAI"},"agent_actions":{"view_html":"https://pith.science/pith/5FAOIUAIGFRPI3NPIMTQFZUP7F","download_json":"https://pith.science/pith/5FAOIUAIGFRPI3NPIMTQFZUP7F.json","view_paper":"https://pith.science/paper/5FAOIUAI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1704.07978&json=true","fetch_graph":"https://pith.science/api/pith-number/5FAOIUAIGFRPI3NPIMTQFZUP7F/graph.json","fetch_events":"https://pith.science/api/pith-number/5FAOIUAIGFRPI3NPIMTQFZUP7F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5FAOIUAIGFRPI3NPIMTQFZUP7F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5FAOIUAIGFRPI3NPIMTQFZUP7F/action/storage_attestation","attest_author":"https://pith.science/pith/5FAOIUAIGFRPI3NPIMTQFZUP7F/action/author_attestation","sign_citation":"https://pith.science/pith/5FAOIUAIGFRPI3NPIMTQFZUP7F/action/citation_signature","submit_replication":"https://pith.science/pith/5FAOIUAIGFRPI3NPIMTQFZUP7F/action/replication_record"}},"created_at":"2026-05-18T00:15:05.584399+00:00","updated_at":"2026-05-18T00:15:05.584399+00:00"}