{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2015:DPCG3BFZZIPKW7CGXU42MRSDOO","short_pith_number":"pith:DPCG3BFZ","schema_version":"1.0","canonical_sha256":"1bc46d84b9ca1eab7c46bd39a6464373b8c6fe61d34ed955a3fd0847d609a77c","source":{"kind":"arxiv","id":"1511.04143","version":5},"attestation_state":"computed","paper":{"title":"Deep Reinforcement Learning in Parameterized Action Space","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.MA","cs.NE"],"primary_cat":"cs.AI","authors_text":"Matthew Hausknecht, Peter Stone","submitted_at":"2015-11-13T02:34:33Z","abstract_excerpt":"Recent work has shown that deep neural networks are capable of approximating both value functions and policies in reinforcement learning domains featuring continuous state and action spaces. However, to the best of our knowledge no previous work has succeeded at using deep neural networks in structured (parameterized) continuous action spaces. To fill this gap, this paper focuses on learning within the domain of simulated RoboCup soccer, which features a small set of discrete action types, each of which is parameterized with continuous variables. The best learned agent can score goals more rel"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1511.04143","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2015-11-13T02:34:33Z","cross_cats_sorted":["cs.LG","cs.MA","cs.NE"],"title_canon_sha256":"b753ec5d06695487f314e90342b207ca4c4455b25d5aef93d3c962bb7450697a","abstract_canon_sha256":"ebbd0b3e8250372cbc084d6a6bf5a1c13bcae79500f6c5229ca0c64a273b19ab"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:14:46.617426Z","signature_b64":"ZrCSrAP56/U2OPaFHL0F1ulGxZ74M7hqsaGKkx4Cnos+y2OrAD3YVgUoxHLoFO3p7NmAxLBJz0O6lkFnYNU1Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1bc46d84b9ca1eab7c46bd39a6464373b8c6fe61d34ed955a3fd0847d609a77c","last_reissued_at":"2026-07-05T08:14:46.616998Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:14:46.616998Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deep Reinforcement Learning in Parameterized Action Space","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.MA","cs.NE"],"primary_cat":"cs.AI","authors_text":"Matthew Hausknecht, Peter Stone","submitted_at":"2015-11-13T02:34:33Z","abstract_excerpt":"Recent work has shown that deep neural networks are capable of approximating both value functions and policies in reinforcement learning domains featuring continuous state and action spaces. However, to the best of our knowledge no previous work has succeeded at using deep neural networks in structured (parameterized) continuous action spaces. To fill this gap, this paper focuses on learning within the domain of simulated RoboCup soccer, which features a small set of discrete action types, each of which is parameterized with continuous variables. The best learned agent can score goals more rel"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1511.04143","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1511.04143/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1511.04143","created_at":"2026-07-05T08:14:46.617051+00:00"},{"alias_kind":"arxiv_version","alias_value":"1511.04143v5","created_at":"2026-07-05T08:14:46.617051+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1511.04143","created_at":"2026-07-05T08:14:46.617051+00:00"},{"alias_kind":"pith_short_12","alias_value":"DPCG3BFZZIPK","created_at":"2026-07-05T08:14:46.617051+00:00"},{"alias_kind":"pith_short_16","alias_value":"DPCG3BFZZIPKW7CG","created_at":"2026-07-05T08:14:46.617051+00:00"},{"alias_kind":"pith_short_8","alias_value":"DPCG3BFZ","created_at":"2026-07-05T08:14:46.617051+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06771","citing_title":"Integrated Automated Car Following and Lane-changing control based on a Parametrized Deep Q-network with Hybrid Action Space","ref_index":2,"is_internal_anchor":true},{"citing_arxiv_id":"2606.18820","citing_title":"Maturing Markov Decision Processes: Decision Making under Increasing Information and Shrinking Action Sets","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DPCG3BFZZIPKW7CGXU42MRSDOO","json":"https://pith.science/pith/DPCG3BFZZIPKW7CGXU42MRSDOO.json","graph_json":"https://pith.science/api/pith-number/DPCG3BFZZIPKW7CGXU42MRSDOO/graph.json","events_json":"https://pith.science/api/pith-number/DPCG3BFZZIPKW7CGXU42MRSDOO/events.json","paper":"https://pith.science/paper/DPCG3BFZ"},"agent_actions":{"view_html":"https://pith.science/pith/DPCG3BFZZIPKW7CGXU42MRSDOO","download_json":"https://pith.science/pith/DPCG3BFZZIPKW7CGXU42MRSDOO.json","view_paper":"https://pith.science/paper/DPCG3BFZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1511.04143&json=true","fetch_graph":"https://pith.science/api/pith-number/DPCG3BFZZIPKW7CGXU42MRSDOO/graph.json","fetch_events":"https://pith.science/api/pith-number/DPCG3BFZZIPKW7CGXU42MRSDOO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DPCG3BFZZIPKW7CGXU42MRSDOO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DPCG3BFZZIPKW7CGXU42MRSDOO/action/storage_attestation","attest_author":"https://pith.science/pith/DPCG3BFZZIPKW7CGXU42MRSDOO/action/author_attestation","sign_citation":"https://pith.science/pith/DPCG3BFZZIPKW7CGXU42MRSDOO/action/citation_signature","submit_replication":"https://pith.science/pith/DPCG3BFZZIPKW7CGXU42MRSDOO/action/replication_record"}},"created_at":"2026-07-05T08:14:46.617051+00:00","updated_at":"2026-07-05T08:14:46.617051+00:00"}