{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2017:5FAOIUAIGFRPI3NPIMTQFZUP7F","short_pith_number":"pith:5FAOIUAI","canonical_record":{"source":{"id":"1704.07978","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2017-04-26T05:55:07Z","cross_cats_sorted":[],"title_canon_sha256":"e1dae5b0deda47b3a2c4304f4b60843c73fbe7cbc4629b51f0941518316b931c","abstract_canon_sha256":"0a170b8079b51b71acdc895e2431fa849d974836dd36a5dba2fe81f91c881a54"},"schema_version":"1.0"},"canonical_sha256":"e940e450083162f46daf432702e68ff97936a6ae1476f218d055e07335fcebc9","source":{"kind":"arxiv","id":"1704.07978","version":6},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1704.07978","created_at":"2026-05-18T00:15:05Z"},{"alias_kind":"arxiv_version","alias_value":"1704.07978v6","created_at":"2026-05-18T00:15:05Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1704.07978","created_at":"2026-05-18T00:15:05Z"},{"alias_kind":"pith_short_12","alias_value":"5FAOIUAIGFRP","created_at":"2026-05-18T12:31:00Z"},{"alias_kind":"pith_short_16","alias_value":"5FAOIUAIGFRPI3NP","created_at":"2026-05-18T12:31:00Z"},{"alias_kind":"pith_short_8","alias_value":"5FAOIUAI","created_at":"2026-05-18T12:31:00Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2017:5FAOIUAIGFRPI3NPIMTQFZUP7F","target":"record","payload":{"canonical_record":{"source":{"id":"1704.07978","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2017-04-26T05:55:07Z","cross_cats_sorted":[],"title_canon_sha256":"e1dae5b0deda47b3a2c4304f4b60843c73fbe7cbc4629b51f0941518316b931c","abstract_canon_sha256":"0a170b8079b51b71acdc895e2431fa849d974836dd36a5dba2fe81f91c881a54"},"schema_version":"1.0"},"canonical_sha256":"e940e450083162f46daf432702e68ff97936a6ae1476f218d055e07335fcebc9","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:15:05.584830Z","signature_b64":"cg+wX7EZZJq/u1TnImkXgopLIWX3yi1h57piT5Vak+3hOlTD5UOybsWMMb64702IrPqWRFwjFmT8ldXsvvecCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e940e450083162f46daf432702e68ff97936a6ae1476f218d055e07335fcebc9","last_reissued_at":"2026-05-18T00:15:05.584296Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:15:05.584296Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1704.07978","source_version":6,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T00:15:05Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"nCxSk06zfvHKNG1yHFTo8fv9D3F/239r/bjOaYFU4cXhm+BHnS4Wa41PKxlCiLJJOtdYXj4478uG4QEnYXEABw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-21T18:24:27.589414Z"},"content_sha256":"330cc297b445cf1b495aad7c462cdf3b9ed5b163964f09a2dbbdae29953aee03","schema_version":"1.0","event_id":"sha256:330cc297b445cf1b495aad7c462cdf3b9ed5b163964f09a2dbbdae29953aee03"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2017:5FAOIUAIGFRPI3NPIMTQFZUP7F","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"On Improving Deep Reinforcement Learning for POMDPs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Guanghui Miao, Pascal Poupart, Pengfei Zhu, Xin Li","submitted_at":"2017-04-26T05:55:07Z","abstract_excerpt":"Deep Reinforcement Learning (RL) recently emerged as one of the most competitive approaches for learning in sequential decision making problems with fully observable environments, e.g., computer Go. However, very little work has been done in deep RL to handle partially observable environments. We propose a new architecture called Action-specific Deep Recurrent Q-Network (ADRQN) to enhance learning performance in partially observable domains. Actions are encoded by a fully connected layer and coupled with a convolutional observation to form an action-observation pair. The time series of action-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1704.07978","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T00:15:05Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"0tk1Na+rgURb/zqsHwUjPk+Hh4HdBy5Wtbb+YplXhLghWRwxevIxNDXrKh8lg3Rty10AWpsg2K7zovG7k3zmDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-05-21T18:24:27.589758Z"},"content_sha256":"e79fa271f995639a034898391b60c651d6c4bc30c28cf086a374e66eed14176e","schema_version":"1.0","event_id":"sha256:e79fa271f995639a034898391b60c651d6c4bc30c28cf086a374e66eed14176e"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/5FAOIUAIGFRPI3NPIMTQFZUP7F/bundle.json","state_url":"https://pith.science/pith/5FAOIUAIGFRPI3NPIMTQFZUP7F/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/5FAOIUAIGFRPI3NPIMTQFZUP7F/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-05-21T18:24:27Z","links":{"resolver":"https://pith.science/pith/5FAOIUAIGFRPI3NPIMTQFZUP7F","bundle":"https://pith.science/pith/5FAOIUAIGFRPI3NPIMTQFZUP7F/bundle.json","state":"https://pith.science/pith/5FAOIUAIGFRPI3NPIMTQFZUP7F/state.json","well_known_bundle":"https://pith.science/.well-known/pith/5FAOIUAIGFRPI3NPIMTQFZUP7F/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2017:5FAOIUAIGFRPI3NPIMTQFZUP7F","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"0a170b8079b51b71acdc895e2431fa849d974836dd36a5dba2fe81f91c881a54","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2017-04-26T05:55:07Z","title_canon_sha256":"e1dae5b0deda47b3a2c4304f4b60843c73fbe7cbc4629b51f0941518316b931c"},"schema_version":"1.0","source":{"id":"1704.07978","kind":"arxiv","version":6}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1704.07978","created_at":"2026-05-18T00:15:05Z"},{"alias_kind":"arxiv_version","alias_value":"1704.07978v6","created_at":"2026-05-18T00:15:05Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1704.07978","created_at":"2026-05-18T00:15:05Z"},{"alias_kind":"pith_short_12","alias_value":"5FAOIUAIGFRP","created_at":"2026-05-18T12:31:00Z"},{"alias_kind":"pith_short_16","alias_value":"5FAOIUAIGFRPI3NP","created_at":"2026-05-18T12:31:00Z"},{"alias_kind":"pith_short_8","alias_value":"5FAOIUAI","created_at":"2026-05-18T12:31:00Z"}],"graph_snapshots":[{"event_id":"sha256:e79fa271f995639a034898391b60c651d6c4bc30c28cf086a374e66eed14176e","target":"graph","created_at":"2026-05-18T00:15:05Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"paper":{"abstract_excerpt":"Deep Reinforcement Learning (RL) recently emerged as one of the most competitive approaches for learning in sequential decision making problems with fully observable environments, e.g., computer Go. However, very little work has been done in deep RL to handle partially observable environments. We propose a new architecture called Action-specific Deep Recurrent Q-Network (ADRQN) to enhance learning performance in partially observable domains. Actions are encoded by a fully connected layer and coupled with a convolutional observation to form an action-observation pair. The time series of action-","authors_text":"Guanghui Miao, Pascal Poupart, Pengfei Zhu, Xin Li","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2017-04-26T05:55:07Z","title":"On Improving Deep Reinforcement Learning for POMDPs"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1704.07978","kind":"arxiv","version":6},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:330cc297b445cf1b495aad7c462cdf3b9ed5b163964f09a2dbbdae29953aee03","target":"record","created_at":"2026-05-18T00:15:05Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"0a170b8079b51b71acdc895e2431fa849d974836dd36a5dba2fe81f91c881a54","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2017-04-26T05:55:07Z","title_canon_sha256":"e1dae5b0deda47b3a2c4304f4b60843c73fbe7cbc4629b51f0941518316b931c"},"schema_version":"1.0","source":{"id":"1704.07978","kind":"arxiv","version":6}},"canonical_sha256":"e940e450083162f46daf432702e68ff97936a6ae1476f218d055e07335fcebc9","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"e940e450083162f46daf432702e68ff97936a6ae1476f218d055e07335fcebc9","first_computed_at":"2026-05-18T00:15:05.584296Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-18T00:15:05.584296Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"cg+wX7EZZJq/u1TnImkXgopLIWX3yi1h57piT5Vak+3hOlTD5UOybsWMMb64702IrPqWRFwjFmT8ldXsvvecCw==","signature_status":"signed_v1","signed_at":"2026-05-18T00:15:05.584830Z","signed_message":"canonical_sha256_bytes"},"source_id":"1704.07978","source_kind":"arxiv","source_version":6}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:330cc297b445cf1b495aad7c462cdf3b9ed5b163964f09a2dbbdae29953aee03","sha256:e79fa271f995639a034898391b60c651d6c4bc30c28cf086a374e66eed14176e"],"state_sha256":"b91cf281939eca3084c4fe217aba1e8c0104cfa2e4b003c8f1de9f894a2d4033"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"cPb0lJN1GH/DidcylcmcO74yKSSp4gHSy0p8LAsvEPrmkhLbDAPRelfZ5FcWnoPSg663qE+RbyUfQCaC392DAg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-05-21T18:24:27.591863Z","bundle_sha256":"7b1c357a7fefdab5e9b3fe1f29353d243f64f9c04982e778e4a50ea98ff15df0"}}