{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:D4G5XA5WWYNKQP4FERGTFKE3UE","short_pith_number":"pith:D4G5XA5W","schema_version":"1.0","canonical_sha256":"1f0ddb83b6b61aa83f85244d32a89ba11216c72473b31490270c1346cba62604","source":{"kind":"arxiv","id":"1904.12901","version":1},"attestation_state":"computed","paper":{"title":"Challenges of Real-World Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Daniel Mankowitz, Gabriel Dulac-Arnold, Todd Hester","submitted_at":"2019-04-29T18:40:15Z","abstract_excerpt":"Reinforcement learning (RL) has proven its worth in a series of artificial domains, and is beginning to show some successes in real-world scenarios. However, much of the research advances in RL are often hard to leverage in real-world systems due to a series of assumptions that are rarely satisfied in practice. We present a set of nine unique challenges that must be addressed to productionize RL to real world problems. For each of these challenges, we specify the exact meaning of the challenge, present some approaches from the literature, and specify some metrics for evaluating that challenge."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1904.12901","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-04-29T18:40:15Z","cross_cats_sorted":["cs.AI","cs.RO","stat.ML"],"title_canon_sha256":"260e2a12b50bfa998c1ce83a6bfb8c517545e8a12ea204a55cd37b19c6c9c175","abstract_canon_sha256":"b78a3dd09ec5d4cb08ffa63aceb9bc1796ba3b89cace387de094635803efdaca"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-17T23:47:24.530185Z","signature_b64":"YHp2EiX0oiVyj9xfNEK0YhyKhUzbXfMvItrjqUCNmLNLNSlVA4Kni5PLE5rY9gzwVZNo8s+KRgGxosPub26hAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1f0ddb83b6b61aa83f85244d32a89ba11216c72473b31490270c1346cba62604","last_reissued_at":"2026-05-17T23:47:24.529716Z","signature_status":"signed_v1","first_computed_at":"2026-05-17T23:47:24.529716Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Challenges of Real-World Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Daniel Mankowitz, Gabriel Dulac-Arnold, Todd Hester","submitted_at":"2019-04-29T18:40:15Z","abstract_excerpt":"Reinforcement learning (RL) has proven its worth in a series of artificial domains, and is beginning to show some successes in real-world scenarios. However, much of the research advances in RL are often hard to leverage in real-world systems due to a series of assumptions that are rarely satisfied in practice. We present a set of nine unique challenges that must be addressed to productionize RL to real world problems. For each of these challenges, we specify the exact meaning of the challenge, present some approaches from the literature, and specify some metrics for evaluating that challenge."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1904.12901","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1904.12901","created_at":"2026-05-17T23:47:24.529797+00:00"},{"alias_kind":"arxiv_version","alias_value":"1904.12901v1","created_at":"2026-05-17T23:47:24.529797+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1904.12901","created_at":"2026-05-17T23:47:24.529797+00:00"},{"alias_kind":"pith_short_12","alias_value":"D4G5XA5WWYNK","created_at":"2026-05-18T12:33:15.570797+00:00"},{"alias_kind":"pith_short_16","alias_value":"D4G5XA5WWYNKQP4F","created_at":"2026-05-18T12:33:15.570797+00:00"},{"alias_kind":"pith_short_8","alias_value":"D4G5XA5W","created_at":"2026-05-18T12:33:15.570797+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":12,"sample":[{"citing_arxiv_id":"2607.01651","citing_title":"One Demonstration Is Enough for Real-World Robotic Reinforcement Learning","ref_index":4,"is_internal_anchor":true},{"citing_arxiv_id":"2606.10129","citing_title":"Discovering Interpretable Multi-Parameter Control Policies for Evolutionary Algorithms Using Deep Reinforcement Learning","ref_index":62,"is_internal_anchor":true},{"citing_arxiv_id":"2605.30719","citing_title":"When are LLMs Sufficient Policy Optimizers for Sequential RL Tasks?","ref_index":5,"is_internal_anchor":true},{"citing_arxiv_id":"2605.30719","citing_title":"When are LLMs Sufficient Policy Optimizers for Sequential RL Tasks?","ref_index":5,"is_internal_anchor":true},{"citing_arxiv_id":"2604.23132","citing_title":"UAV Trajectory and Bandwidth Allocation for Efficient Data Collection in Low-Altitude Intelligent IoT: A Hierarchical DRL Approach","ref_index":28,"is_internal_anchor":true},{"citing_arxiv_id":"2403.10559","citing_title":"Generative Models and Connected and Automated Vehicles: A Survey in Exploring the Intersection of Transportation and AI","ref_index":34,"is_internal_anchor":true},{"citing_arxiv_id":"2412.18208","citing_title":"Quantum framework for Reinforcement Learning: Integrating Markov decision process, quantum arithmetic, and trajectory search","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2605.22711","citing_title":"Abstraction for Offline Goal-Conditioned Reinforcement Learning","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2605.21984","citing_title":"Echo: Learning from Experience Data via User-Driven Refinement","ref_index":7,"is_internal_anchor":true},{"citing_arxiv_id":"2605.17229","citing_title":"Generating Realistic Safety-Critical Scenarios for Vehicle-Pedestrian Interactions","ref_index":4,"is_internal_anchor":true},{"citing_arxiv_id":"2510.17640","citing_title":"RESample: A Robust Data Augmentation Framework via Exploratory Sampling for Robotic Manipulation","ref_index":28,"is_internal_anchor":true},{"citing_arxiv_id":"2310.06114","citing_title":"Learning Interactive Real-World Simulators","ref_index":206,"is_internal_anchor":true},{"citing_arxiv_id":"2004.07219","citing_title":"D4RL: Datasets for Deep Data-Driven Reinforcement Learning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10546","citing_title":"Higher Resolution, Better Generalization: Unlocking Visual Scaling in Deep Reinforcement Learning","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23132","citing_title":"UAV Trajectory and Bandwidth Allocation for Efficient Data Collection in Low-Altitude Intelligent IoT: A Hierarchical DRL Approach","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05478","citing_title":"LANTERN: LLM-Augmented Neurosymbolic Transfer with Experience-Gated Reasoning Networks","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01356","citing_title":"Model-Based Proactive Cost Generation for Learning Safe Policies Offline with Limited Violation Data","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14032","citing_title":"Hierarchical Reinforcement Learning with Runtime Safety Shielding for Power Grid Operation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05088","citing_title":"Scalar Federated Learning for Linear Quadratic Regulator","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D4G5XA5WWYNKQP4FERGTFKE3UE","json":"https://pith.science/pith/D4G5XA5WWYNKQP4FERGTFKE3UE.json","graph_json":"https://pith.science/api/pith-number/D4G5XA5WWYNKQP4FERGTFKE3UE/graph.json","events_json":"https://pith.science/api/pith-number/D4G5XA5WWYNKQP4FERGTFKE3UE/events.json","paper":"https://pith.science/paper/D4G5XA5W"},"agent_actions":{"view_html":"https://pith.science/pith/D4G5XA5WWYNKQP4FERGTFKE3UE","download_json":"https://pith.science/pith/D4G5XA5WWYNKQP4FERGTFKE3UE.json","view_paper":"https://pith.science/paper/D4G5XA5W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1904.12901&json=true","fetch_graph":"https://pith.science/api/pith-number/D4G5XA5WWYNKQP4FERGTFKE3UE/graph.json","fetch_events":"https://pith.science/api/pith-number/D4G5XA5WWYNKQP4FERGTFKE3UE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D4G5XA5WWYNKQP4FERGTFKE3UE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D4G5XA5WWYNKQP4FERGTFKE3UE/action/storage_attestation","attest_author":"https://pith.science/pith/D4G5XA5WWYNKQP4FERGTFKE3UE/action/author_attestation","sign_citation":"https://pith.science/pith/D4G5XA5WWYNKQP4FERGTFKE3UE/action/citation_signature","submit_replication":"https://pith.science/pith/D4G5XA5WWYNKQP4FERGTFKE3UE/action/replication_record"}},"created_at":"2026-05-17T23:47:24.529797+00:00","updated_at":"2026-05-17T23:47:24.529797+00:00"}