{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:AF54M76B6I2CK24Y6ARE3XAX2M","short_pith_number":"pith:AF54M76B","schema_version":"1.0","canonical_sha256":"017bc67fc1f234256b98f0224ddc17d32d6018102287301e656a4f7fa5aab41e","source":{"kind":"arxiv","id":"2411.18892","version":2},"attestation_state":"computed","paper":{"title":"A Comprehensive Survey of Reinforcement Learning: From Algorithms to Practical Challenges","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Amir Hossein Moosavi, Dariush Ebrahimi, Majid Ghasemi","submitted_at":"2024-11-28T03:53:14Z","abstract_excerpt":"Reinforcement Learning (RL) has emerged as a powerful paradigm in Artificial Intelligence (AI), enabling agents to learn optimal behaviors through interactions with their environments. Drawing from the foundations of trial and error, RL equips agents to make informed decisions through feedback in the form of rewards or penalties. This paper presents a comprehensive survey of RL, meticulously analyzing a wide range of algorithms, from foundational tabular methods to advanced Deep Reinforcement Learning (DRL) techniques. We categorize and evaluate these algorithms based on key criteria such as s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.18892","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-11-28T03:53:14Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"3f6a88c7bb876b968899439ce23755ec8a9962cad7961f8eb52c9b5e95b980fa","abstract_canon_sha256":"643a547166762d8701fb0af19892de5589219204edbddc7e96af53edfe90551a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:08:25.401643Z","signature_b64":"VKRLJgIcOe83yrIqU34MojzrSogVFUFXxTG4+62OYR+nnUl8sVUv3UoqQTRkDajrJ+2zspTdTeIE2F+waawvDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"017bc67fc1f234256b98f0224ddc17d32d6018102287301e656a4f7fa5aab41e","last_reissued_at":"2026-07-05T10:08:25.401172Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:08:25.401172Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Comprehensive Survey of Reinforcement Learning: From Algorithms to Practical Challenges","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Amir Hossein Moosavi, Dariush Ebrahimi, Majid Ghasemi","submitted_at":"2024-11-28T03:53:14Z","abstract_excerpt":"Reinforcement Learning (RL) has emerged as a powerful paradigm in Artificial Intelligence (AI), enabling agents to learn optimal behaviors through interactions with their environments. Drawing from the foundations of trial and error, RL equips agents to make informed decisions through feedback in the form of rewards or penalties. This paper presents a comprehensive survey of RL, meticulously analyzing a wide range of algorithms, from foundational tabular methods to advanced Deep Reinforcement Learning (DRL) techniques. We categorize and evaluate these algorithms based on key criteria such as s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.18892","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.18892/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.18892","created_at":"2026-07-05T10:08:25.401235+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.18892v2","created_at":"2026-07-05T10:08:25.401235+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.18892","created_at":"2026-07-05T10:08:25.401235+00:00"},{"alias_kind":"pith_short_12","alias_value":"AF54M76B6I2C","created_at":"2026-07-05T10:08:25.401235+00:00"},{"alias_kind":"pith_short_16","alias_value":"AF54M76B6I2CK24Y","created_at":"2026-07-05T10:08:25.401235+00:00"},{"alias_kind":"pith_short_8","alias_value":"AF54M76B","created_at":"2026-07-05T10:08:25.401235+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00642","citing_title":"Coachable agents for interactive gameplay","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01249","citing_title":"Trust Region On-Policy Distillation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27865","citing_title":"From Bootstrapping to Sequence Modeling: A Unified Generative Framework for Personalized Landing-Page Modeling","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18569","citing_title":"Reinforcement Learning Assisted Quantum Simulation of Many-Body Excited States and Real-Time Dynamics","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2509.08827","citing_title":"A Survey of Reinforcement Learning for Large Reasoning Models","ref_index":161,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AF54M76B6I2CK24Y6ARE3XAX2M","json":"https://pith.science/pith/AF54M76B6I2CK24Y6ARE3XAX2M.json","graph_json":"https://pith.science/api/pith-number/AF54M76B6I2CK24Y6ARE3XAX2M/graph.json","events_json":"https://pith.science/api/pith-number/AF54M76B6I2CK24Y6ARE3XAX2M/events.json","paper":"https://pith.science/paper/AF54M76B"},"agent_actions":{"view_html":"https://pith.science/pith/AF54M76B6I2CK24Y6ARE3XAX2M","download_json":"https://pith.science/pith/AF54M76B6I2CK24Y6ARE3XAX2M.json","view_paper":"https://pith.science/paper/AF54M76B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.18892&json=true","fetch_graph":"https://pith.science/api/pith-number/AF54M76B6I2CK24Y6ARE3XAX2M/graph.json","fetch_events":"https://pith.science/api/pith-number/AF54M76B6I2CK24Y6ARE3XAX2M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AF54M76B6I2CK24Y6ARE3XAX2M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AF54M76B6I2CK24Y6ARE3XAX2M/action/storage_attestation","attest_author":"https://pith.science/pith/AF54M76B6I2CK24Y6ARE3XAX2M/action/author_attestation","sign_citation":"https://pith.science/pith/AF54M76B6I2CK24Y6ARE3XAX2M/action/citation_signature","submit_replication":"https://pith.science/pith/AF54M76B6I2CK24Y6ARE3XAX2M/action/replication_record"}},"created_at":"2026-07-05T10:08:25.401235+00:00","updated_at":"2026-07-05T10:08:25.401235+00:00"}