{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:APTQ2IDKRUZDRVBUTVFKFCIEVJ","short_pith_number":"pith:APTQ2IDK","schema_version":"1.0","canonical_sha256":"03e70d206a8d3238d4349d4aa28904aa47681004d3b7e1d5381cc0d5737f5f79","source":{"kind":"arxiv","id":"2404.03037","version":3},"attestation_state":"computed","paper":{"title":"Model-based Reinforcement Learning for Parameterized Action Spaces","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"George Konidaris, Haotian Fu, Renhao Zhang, Yilin Miao","submitted_at":"2024-04-03T19:48:13Z","abstract_excerpt":"We propose a novel model-based reinforcement learning algorithm -- Dynamics Learning and predictive control with Parameterized Actions (DLPA) -- for Parameterized Action Markov Decision Processes (PAMDPs). The agent learns a parameterized-action-conditioned dynamics model and plans with a modified Model Predictive Path Integral control. We theoretically quantify the difference between the generated trajectory and the optimal trajectory during planning in terms of the value they achieved through the lens of Lipschitz Continuity. Our empirical results on several standard benchmarks show that our"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.03037","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-04-03T19:48:13Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"90c8a91e85480ddc06c692f5e10d55b8155111ef34811a1e348975ceb8836455","abstract_canon_sha256":"ea01b3f3888c43b18670a4a3eae6154236665ffb2f0e1a77ff35182918b63118"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:22:39.167377Z","signature_b64":"uYbWgr73yH+veQmTn2giwYVs7KHpzkkGdzhR3tMC+dmZw6tWnkM+uvR8I/l0ra2siqfK66kHCi+OzqsPZMwiDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"03e70d206a8d3238d4349d4aa28904aa47681004d3b7e1d5381cc0d5737f5f79","last_reissued_at":"2026-07-05T08:22:39.166821Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:22:39.166821Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Model-based Reinforcement Learning for Parameterized Action Spaces","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"George Konidaris, Haotian Fu, Renhao Zhang, Yilin Miao","submitted_at":"2024-04-03T19:48:13Z","abstract_excerpt":"We propose a novel model-based reinforcement learning algorithm -- Dynamics Learning and predictive control with Parameterized Actions (DLPA) -- for Parameterized Action Markov Decision Processes (PAMDPs). The agent learns a parameterized-action-conditioned dynamics model and plans with a modified Model Predictive Path Integral control. We theoretically quantify the difference between the generated trajectory and the optimal trajectory during planning in terms of the value they achieved through the lens of Lipschitz Continuity. Our empirical results on several standard benchmarks show that our"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.03037","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.03037/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.03037","created_at":"2026-07-05T08:22:39.166888+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.03037v3","created_at":"2026-07-05T08:22:39.166888+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.03037","created_at":"2026-07-05T08:22:39.166888+00:00"},{"alias_kind":"pith_short_12","alias_value":"APTQ2IDKRUZD","created_at":"2026-07-05T08:22:39.166888+00:00"},{"alias_kind":"pith_short_16","alias_value":"APTQ2IDKRUZDRVBU","created_at":"2026-07-05T08:22:39.166888+00:00"},{"alias_kind":"pith_short_8","alias_value":"APTQ2IDK","created_at":"2026-07-05T08:22:39.166888+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.12312","citing_title":"Transferable Delay-Aware Reinforcement Learning via Implicit Causal Graph Modeling","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/APTQ2IDKRUZDRVBUTVFKFCIEVJ","json":"https://pith.science/pith/APTQ2IDKRUZDRVBUTVFKFCIEVJ.json","graph_json":"https://pith.science/api/pith-number/APTQ2IDKRUZDRVBUTVFKFCIEVJ/graph.json","events_json":"https://pith.science/api/pith-number/APTQ2IDKRUZDRVBUTVFKFCIEVJ/events.json","paper":"https://pith.science/paper/APTQ2IDK"},"agent_actions":{"view_html":"https://pith.science/pith/APTQ2IDKRUZDRVBUTVFKFCIEVJ","download_json":"https://pith.science/pith/APTQ2IDKRUZDRVBUTVFKFCIEVJ.json","view_paper":"https://pith.science/paper/APTQ2IDK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.03037&json=true","fetch_graph":"https://pith.science/api/pith-number/APTQ2IDKRUZDRVBUTVFKFCIEVJ/graph.json","fetch_events":"https://pith.science/api/pith-number/APTQ2IDKRUZDRVBUTVFKFCIEVJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/APTQ2IDKRUZDRVBUTVFKFCIEVJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/APTQ2IDKRUZDRVBUTVFKFCIEVJ/action/storage_attestation","attest_author":"https://pith.science/pith/APTQ2IDKRUZDRVBUTVFKFCIEVJ/action/author_attestation","sign_citation":"https://pith.science/pith/APTQ2IDKRUZDRVBUTVFKFCIEVJ/action/citation_signature","submit_replication":"https://pith.science/pith/APTQ2IDKRUZDRVBUTVFKFCIEVJ/action/replication_record"}},"created_at":"2026-07-05T08:22:39.166888+00:00","updated_at":"2026-07-05T08:22:39.166888+00:00"}