{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:I7N2KOTXQZM75OC5S4ICJKGX5Y","short_pith_number":"pith:I7N2KOTX","schema_version":"1.0","canonical_sha256":"47dba53a778659feb85d971024a8d7ee3f384fc2b245107b7c6ae53071483d57","source":{"kind":"arxiv","id":"2412.10400","version":3},"attestation_state":"computed","paper":{"title":"Reinforcement Learning Enhanced LLMs: A Survey","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Eduard Hovy, Fei Wu, Guoyin Wang, Jie Zhang, Jiwei Li, Runyi Hu, Shengyu Zhang, Shuhe Wang, Tianwei Zhang, Xiaoya Li","submitted_at":"2024-12-05T16:10:42Z","abstract_excerpt":"Reinforcement learning (RL) enhanced large language models (LLMs), particularly exemplified by DeepSeek-R1, have exhibited outstanding performance. Despite the effectiveness in improving LLM capabilities, its implementation remains highly complex, requiring complex algorithms, reward modeling strategies, and optimization techniques. This complexity poses challenges for researchers and practitioners in developing a systematic understanding of RL-enhanced LLMs. Moreover, the absence of a comprehensive survey summarizing existing research on RL-enhanced LLMs has limited progress in this domain, h"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.10400","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-05T16:10:42Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"39dece5bc88832d677402dea0de57d2d3b92b033fe24fc3753e2afa7effac6e9","abstract_canon_sha256":"c89932c04b8b40f56f5857ba042ee3fe01baca13e38713e1c14ebacf9090fa8b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:18:46.861572Z","signature_b64":"3jSO6sSn7yJzwJ/5k9mkMLcIpKmpQL/wanxffDIXPHUN+2ZR/cNvwlKNiA8ak/BfQjrJzyFZd5/lMWkgLv8aBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"47dba53a778659feb85d971024a8d7ee3f384fc2b245107b7c6ae53071483d57","last_reissued_at":"2026-07-05T10:18:46.861058Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:18:46.861058Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reinforcement Learning Enhanced LLMs: A Survey","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Eduard Hovy, Fei Wu, Guoyin Wang, Jie Zhang, Jiwei Li, Runyi Hu, Shengyu Zhang, Shuhe Wang, Tianwei Zhang, Xiaoya Li","submitted_at":"2024-12-05T16:10:42Z","abstract_excerpt":"Reinforcement learning (RL) enhanced large language models (LLMs), particularly exemplified by DeepSeek-R1, have exhibited outstanding performance. Despite the effectiveness in improving LLM capabilities, its implementation remains highly complex, requiring complex algorithms, reward modeling strategies, and optimization techniques. This complexity poses challenges for researchers and practitioners in developing a systematic understanding of RL-enhanced LLMs. Moreover, the absence of a comprehensive survey summarizing existing research on RL-enhanced LLMs has limited progress in this domain, h"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.10400","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.10400/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.10400","created_at":"2026-07-05T10:18:46.861115+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.10400v3","created_at":"2026-07-05T10:18:46.861115+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.10400","created_at":"2026-07-05T10:18:46.861115+00:00"},{"alias_kind":"pith_short_12","alias_value":"I7N2KOTXQZM7","created_at":"2026-07-05T10:18:46.861115+00:00"},{"alias_kind":"pith_short_16","alias_value":"I7N2KOTXQZM75OC5","created_at":"2026-07-05T10:18:46.861115+00:00"},{"alias_kind":"pith_short_8","alias_value":"I7N2KOTX","created_at":"2026-07-05T10:18:46.861115+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":218,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18747","citing_title":"Generating Natural and Expressive Robot Gestures through Iterative Reinforcement Learning with Human Feedback using LLMs","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20696","citing_title":"Distributed Direct Preference Optimization","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27859","citing_title":"Rethinking Agentic Reinforcement Learning In Large Language Models","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2509.02547","citing_title":"The Landscape of Agentic Reinforcement Learning for LLMs: A Survey","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27859","citing_title":"Rethinking Agentic Reinforcement Learning In Large Language Models","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08368","citing_title":"On Distinguishing Capability Elicitation from Capability Creation in Post-Training: A Free-Energy Perspective","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27859","citing_title":"Rethinking Agentic Reinforcement Learning In Large Language Models","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07851","citing_title":"ReRec: Reasoning-Augmented LLM-based Recommendation Assistant via Reinforcement Fine-tuning","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02913","citing_title":"Generate, Filter, Control, Replay: A Comprehensive Survey of Rollout Strategies for LLM Reinforcement Learning","ref_index":122,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I7N2KOTXQZM75OC5S4ICJKGX5Y","json":"https://pith.science/pith/I7N2KOTXQZM75OC5S4ICJKGX5Y.json","graph_json":"https://pith.science/api/pith-number/I7N2KOTXQZM75OC5S4ICJKGX5Y/graph.json","events_json":"https://pith.science/api/pith-number/I7N2KOTXQZM75OC5S4ICJKGX5Y/events.json","paper":"https://pith.science/paper/I7N2KOTX"},"agent_actions":{"view_html":"https://pith.science/pith/I7N2KOTXQZM75OC5S4ICJKGX5Y","download_json":"https://pith.science/pith/I7N2KOTXQZM75OC5S4ICJKGX5Y.json","view_paper":"https://pith.science/paper/I7N2KOTX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.10400&json=true","fetch_graph":"https://pith.science/api/pith-number/I7N2KOTXQZM75OC5S4ICJKGX5Y/graph.json","fetch_events":"https://pith.science/api/pith-number/I7N2KOTXQZM75OC5S4ICJKGX5Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I7N2KOTXQZM75OC5S4ICJKGX5Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I7N2KOTXQZM75OC5S4ICJKGX5Y/action/storage_attestation","attest_author":"https://pith.science/pith/I7N2KOTXQZM75OC5S4ICJKGX5Y/action/author_attestation","sign_citation":"https://pith.science/pith/I7N2KOTXQZM75OC5S4ICJKGX5Y/action/citation_signature","submit_replication":"https://pith.science/pith/I7N2KOTXQZM75OC5S4ICJKGX5Y/action/replication_record"}},"created_at":"2026-07-05T10:18:46.861115+00:00","updated_at":"2026-07-05T10:18:46.861115+00:00"}