{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:Q3D3J6G6R7MPHNNLCNUAXZ46LX","short_pith_number":"pith:Q3D3J6G6","schema_version":"1.0","canonical_sha256":"86c7b4f8de8fd8f3b5ab13680be79e5dea3f2bab5f200e0995baa316e29069e2","source":{"kind":"arxiv","id":"2505.24034","version":2},"attestation_state":"computed","paper":{"title":"LlamaRL: A Distributed Asynchronous Reinforcement Learning Framework for Efficient Large-scale LLM Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Beibei Zhu, Bo Wu, Chen Zhu, Eryk Helenowski, Jia Ding, Liang Tan, Rui Hou, Sid Wang, Tengyu Xu, Tushar Gowda, Xiaocheng Tang, Yundi Qian, Yunhao Tang, Zhengxing Chen","submitted_at":"2025-05-29T22:14:15Z","abstract_excerpt":"Reinforcement Learning (RL) has become the most effective post-training approach for improving the capabilities of Large Language Models (LLMs). In practice, because of the high demands on latency and memory, it is particularly challenging to develop an efficient RL framework that reliably manages policy models with hundreds to thousands of billions of parameters.\n  In this paper, we present LlamaRL, a fully distributed, asynchronous RL framework optimized for efficient training of large-scale LLMs with various model sizes (8B, 70B, and 405B parameters) on GPU clusters ranging from a handful t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.24034","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-29T22:14:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a00f8fade0f72a8eae6801d8fe29da46e861959b56584c6e7c80a1f63d602ac9","abstract_canon_sha256":"2f932a52cfd171a159e918962722e27f9da390db1ac79b8a2cdd7a75f4e47bed"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:08.993402Z","signature_b64":"lbRZGrgDxU9i2oLOl8/czgtcjk72yMrU9YzRFXinkLSMLctHbAVVtOWWQaJlM/5ndGbLVbzk13/BfjPrSRRwDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"86c7b4f8de8fd8f3b5ab13680be79e5dea3f2bab5f200e0995baa316e29069e2","last_reissued_at":"2026-07-05T11:14:08.992874Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:08.992874Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LlamaRL: A Distributed Asynchronous Reinforcement Learning Framework for Efficient Large-scale LLM Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Beibei Zhu, Bo Wu, Chen Zhu, Eryk Helenowski, Jia Ding, Liang Tan, Rui Hou, Sid Wang, Tengyu Xu, Tushar Gowda, Xiaocheng Tang, Yundi Qian, Yunhao Tang, Zhengxing Chen","submitted_at":"2025-05-29T22:14:15Z","abstract_excerpt":"Reinforcement Learning (RL) has become the most effective post-training approach for improving the capabilities of Large Language Models (LLMs). In practice, because of the high demands on latency and memory, it is particularly challenging to develop an efficient RL framework that reliably manages policy models with hundreds to thousands of billions of parameters.\n  In this paper, we present LlamaRL, a fully distributed, asynchronous RL framework optimized for efficient training of large-scale LLMs with various model sizes (8B, 70B, and 405B parameters) on GPU clusters ranging from a handful t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.24034","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.24034/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.24034","created_at":"2026-07-05T11:14:08.992939+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.24034v2","created_at":"2026-07-05T11:14:08.992939+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.24034","created_at":"2026-07-05T11:14:08.992939+00:00"},{"alias_kind":"pith_short_12","alias_value":"Q3D3J6G6R7MP","created_at":"2026-07-05T11:14:08.992939+00:00"},{"alias_kind":"pith_short_16","alias_value":"Q3D3J6G6R7MPHNNL","created_at":"2026-07-05T11:14:08.992939+00:00"},{"alias_kind":"pith_short_8","alias_value":"Q3D3J6G6","created_at":"2026-07-05T11:14:08.992939+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11867","citing_title":"Harnessing Routing Foresight for Micro-step-level MoE load balancing in RL Post-training","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08446","citing_title":"Sparrow: Sparse Rollout for Stable and Efficient Long-context RL of Large Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05597","citing_title":"AsyncWebRL: Efficient Multi-Step RL for Visual Web Agents","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04560","citing_title":"Rollout-Level Advantage-Prioritized Experience Replay for GRPO","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03077","citing_title":"Libra: Efficient Resource Management for Agentic RL Post-Training","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26606","citing_title":"Spend Your Rollouts Where It Counts: Rollout Allocation for Group-Based RL Post-Training","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2504.12501","citing_title":"Reinforcement Learning from Human Feedback","ref_index":133,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19425","citing_title":"When to Stop Reusing: Dynamic Gradient Gating for Sample-Efficient RLVR","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12476","citing_title":"HetRL: Efficient Reinforcement Learning for LLMs in Heterogeneous Environments","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26256","citing_title":"DORA: A Scalable Asynchronous Reinforcement Learning System for Language Model Training","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09107","citing_title":"TensorHub: Scalable and Elastic Weight Transfer for LLM RL Training","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Q3D3J6G6R7MPHNNLCNUAXZ46LX","json":"https://pith.science/pith/Q3D3J6G6R7MPHNNLCNUAXZ46LX.json","graph_json":"https://pith.science/api/pith-number/Q3D3J6G6R7MPHNNLCNUAXZ46LX/graph.json","events_json":"https://pith.science/api/pith-number/Q3D3J6G6R7MPHNNLCNUAXZ46LX/events.json","paper":"https://pith.science/paper/Q3D3J6G6"},"agent_actions":{"view_html":"https://pith.science/pith/Q3D3J6G6R7MPHNNLCNUAXZ46LX","download_json":"https://pith.science/pith/Q3D3J6G6R7MPHNNLCNUAXZ46LX.json","view_paper":"https://pith.science/paper/Q3D3J6G6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.24034&json=true","fetch_graph":"https://pith.science/api/pith-number/Q3D3J6G6R7MPHNNLCNUAXZ46LX/graph.json","fetch_events":"https://pith.science/api/pith-number/Q3D3J6G6R7MPHNNLCNUAXZ46LX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Q3D3J6G6R7MPHNNLCNUAXZ46LX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Q3D3J6G6R7MPHNNLCNUAXZ46LX/action/storage_attestation","attest_author":"https://pith.science/pith/Q3D3J6G6R7MPHNNLCNUAXZ46LX/action/author_attestation","sign_citation":"https://pith.science/pith/Q3D3J6G6R7MPHNNLCNUAXZ46LX/action/citation_signature","submit_replication":"https://pith.science/pith/Q3D3J6G6R7MPHNNLCNUAXZ46LX/action/replication_record"}},"created_at":"2026-07-05T11:14:08.992939+00:00","updated_at":"2026-07-05T11:14:08.992939+00:00"}