{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2015:3ALMLPGHXLWF2CQG2TQ5DXXST4","short_pith_number":"pith:3ALMLPGH","schema_version":"1.0","canonical_sha256":"d816c5bcc7baec5d0a06d4e1d1def29f21351dd8ff355e6bbabff3005dcee032","source":{"kind":"arxiv","id":"1507.06527","version":4},"attestation_state":"computed","paper":{"title":"Deep Recurrent Q-Learning for Partially Observable MDPs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Matthew Hausknecht, Peter Stone","submitted_at":"2015-07-23T15:16:46Z","abstract_excerpt":"Deep Reinforcement Learning has yielded proficient controllers for complex tasks. However, these controllers have limited memory and rely on being able to perceive the complete game screen at each decision point. To address these shortcomings, this article investigates the effects of adding recurrency to a Deep Q-Network (DQN) by replacing the first post-convolutional fully-connected layer with a recurrent LSTM. The resulting \\textit{Deep Recurrent Q-Network} (DRQN), although capable of seeing only a single frame at each timestep, successfully integrates information through time and replicates"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1507.06527","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2015-07-23T15:16:46Z","cross_cats_sorted":[],"title_canon_sha256":"9a44724b25a09d4956c27ae7adab7a388179ba9baf4de86bab06ea5c2a9ec7b9","abstract_canon_sha256":"a91f5911eb288ca34cbf3bb7ff898f355672e55451ad119e5cbe0f630cde0695"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:52:59.320376Z","signature_b64":"OHYnQbcNy9hHP9Bjszv+K0jGF4achOxF27d90rYyrV3FakLUP5p9JEFJWvkbU5DTXXcSVeWwjVVZ3DPrIbFeDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d816c5bcc7baec5d0a06d4e1d1def29f21351dd8ff355e6bbabff3005dcee032","last_reissued_at":"2026-05-18T00:52:59.319977Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:52:59.319977Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deep Recurrent Q-Learning for Partially Observable MDPs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Matthew Hausknecht, Peter Stone","submitted_at":"2015-07-23T15:16:46Z","abstract_excerpt":"Deep Reinforcement Learning has yielded proficient controllers for complex tasks. However, these controllers have limited memory and rely on being able to perceive the complete game screen at each decision point. To address these shortcomings, this article investigates the effects of adding recurrency to a Deep Q-Network (DQN) by replacing the first post-convolutional fully-connected layer with a recurrent LSTM. The resulting \\textit{Deep Recurrent Q-Network} (DRQN), although capable of seeing only a single frame at each timestep, successfully integrates information through time and replicates"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1507.06527","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1507.06527","created_at":"2026-05-18T00:52:59.320036+00:00"},{"alias_kind":"arxiv_version","alias_value":"1507.06527v4","created_at":"2026-05-18T00:52:59.320036+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1507.06527","created_at":"2026-05-18T00:52:59.320036+00:00"},{"alias_kind":"pith_short_12","alias_value":"3ALMLPGHXLWF","created_at":"2026-05-18T12:29:02.477457+00:00"},{"alias_kind":"pith_short_16","alias_value":"3ALMLPGHXLWF2CQG","created_at":"2026-05-18T12:29:02.477457+00:00"},{"alias_kind":"pith_short_8","alias_value":"3ALMLPGH","created_at":"2026-05-18T12:29:02.477457+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.09671","citing_title":"Belief-State RWKV for Reinforcement Learning under Partial Observability","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03479","citing_title":"Contextual Control without Memory Growth in a Context-Switching Task","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2010.03768","citing_title":"ALFWorld: Aligning Text and Embodied Environments for Interactive Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04662","citing_title":"Anticipatory Reinforcement Learning: From Generative Path-Laws to Distributional Value Functions","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11507","citing_title":"Deep Learning for Sequential Decision Making under Uncertainty: Foundations, Frameworks, and Frontiers","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3ALMLPGHXLWF2CQG2TQ5DXXST4","json":"https://pith.science/pith/3ALMLPGHXLWF2CQG2TQ5DXXST4.json","graph_json":"https://pith.science/api/pith-number/3ALMLPGHXLWF2CQG2TQ5DXXST4/graph.json","events_json":"https://pith.science/api/pith-number/3ALMLPGHXLWF2CQG2TQ5DXXST4/events.json","paper":"https://pith.science/paper/3ALMLPGH"},"agent_actions":{"view_html":"https://pith.science/pith/3ALMLPGHXLWF2CQG2TQ5DXXST4","download_json":"https://pith.science/pith/3ALMLPGHXLWF2CQG2TQ5DXXST4.json","view_paper":"https://pith.science/paper/3ALMLPGH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1507.06527&json=true","fetch_graph":"https://pith.science/api/pith-number/3ALMLPGHXLWF2CQG2TQ5DXXST4/graph.json","fetch_events":"https://pith.science/api/pith-number/3ALMLPGHXLWF2CQG2TQ5DXXST4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3ALMLPGHXLWF2CQG2TQ5DXXST4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3ALMLPGHXLWF2CQG2TQ5DXXST4/action/storage_attestation","attest_author":"https://pith.science/pith/3ALMLPGHXLWF2CQG2TQ5DXXST4/action/author_attestation","sign_citation":"https://pith.science/pith/3ALMLPGHXLWF2CQG2TQ5DXXST4/action/citation_signature","submit_replication":"https://pith.science/pith/3ALMLPGHXLWF2CQG2TQ5DXXST4/action/replication_record"}},"created_at":"2026-05-18T00:52:59.320036+00:00","updated_at":"2026-05-18T00:52:59.320036+00:00"}