{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3EHXXDP43RT3FJMYYZLZT34YBS","short_pith_number":"pith:3EHXXDP4","schema_version":"1.0","canonical_sha256":"d90f7b8dfcdc67b2a598c65799ef980c99fe4d3a02f46d285937c200daff5a5b","source":{"kind":"arxiv","id":"2308.01320","version":1},"attestation_state":"computed","paper":{"title":"DeepSpeed-Chat: Easy, Fast and Affordable RLHF Training of ChatGPT-like Models at All Scales","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Ammar Ahmad Awan, Conglong Li, Connor Holmes, Heyang Qin, Jeff Rasley, Lev Kurilenko, Masahiro Tanaka, Michael Wyatt, Minjia Zhang, Molly Smith, Olatunji Ruwase, Reza Yazdani Aminabadi, Samyam Rajbhandari, Shuai Che, Shuaiwen Leon Song, Xiaoxia Wu, Yuxiong He, Zhewei Yao, Zhongzhu Zhou","submitted_at":"2023-08-02T18:49:57Z","abstract_excerpt":"ChatGPT-like models have revolutionized various applications in artificial intelligence, from summarization and coding to translation, matching or even surpassing human performance. However, the current landscape lacks an accessible, efficient, and cost-effective end-to-end RLHF (Reinforcement Learning with Human Feedback) training pipeline for these powerful models, particularly when training at the scale of billions of parameters. This paper introduces DeepSpeed-Chat, a novel system that democratizes RLHF training, making it accessible to the AI community. DeepSpeed-Chat offers three key cap"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.01320","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-08-02T18:49:57Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"d1e1fb92f63989fe1977450edbd3fe65222ed0513b90bbb5f6ccfa9f85dfc00e","abstract_canon_sha256":"66a6682173ee64b0e10d52ab6932266ebf5049a34bb6b538511f492c718992d1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:37:17.028592Z","signature_b64":"cWANPOGzQi3Bx0Z17teRPZXH/ZuSWag7oeLEFB7iSdTkqS4q4iAJFA8+iLkpQbZMaqsS8toblW0osgi7kGTECg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d90f7b8dfcdc67b2a598c65799ef980c99fe4d3a02f46d285937c200daff5a5b","last_reissued_at":"2026-07-05T06:37:17.028080Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:37:17.028080Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DeepSpeed-Chat: Easy, Fast and Affordable RLHF Training of ChatGPT-like Models at All Scales","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Ammar Ahmad Awan, Conglong Li, Connor Holmes, Heyang Qin, Jeff Rasley, Lev Kurilenko, Masahiro Tanaka, Michael Wyatt, Minjia Zhang, Molly Smith, Olatunji Ruwase, Reza Yazdani Aminabadi, Samyam Rajbhandari, Shuai Che, Shuaiwen Leon Song, Xiaoxia Wu, Yuxiong He, Zhewei Yao, Zhongzhu Zhou","submitted_at":"2023-08-02T18:49:57Z","abstract_excerpt":"ChatGPT-like models have revolutionized various applications in artificial intelligence, from summarization and coding to translation, matching or even surpassing human performance. However, the current landscape lacks an accessible, efficient, and cost-effective end-to-end RLHF (Reinforcement Learning with Human Feedback) training pipeline for these powerful models, particularly when training at the scale of billions of parameters. This paper introduces DeepSpeed-Chat, a novel system that democratizes RLHF training, making it accessible to the AI community. DeepSpeed-Chat offers three key cap"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.01320","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.01320/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.01320","created_at":"2026-07-05T06:37:17.028153+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.01320v1","created_at":"2026-07-05T06:37:17.028153+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.01320","created_at":"2026-07-05T06:37:17.028153+00:00"},{"alias_kind":"pith_short_12","alias_value":"3EHXXDP43RT3","created_at":"2026-07-05T06:37:17.028153+00:00"},{"alias_kind":"pith_short_16","alias_value":"3EHXXDP43RT3FJMY","created_at":"2026-07-05T06:37:17.028153+00:00"},{"alias_kind":"pith_short_8","alias_value":"3EHXXDP4","created_at":"2026-07-05T06:37:17.028153+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24937","citing_title":"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems","ref_index":230,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03077","citing_title":"Libra: Efficient Resource Management for Agentic RL Post-Training","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26606","citing_title":"Spend Your Rollouts Where It Counts: Rollout Allocation for Group-Based RL Post-Training","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06534","citing_title":"ROSE: Rollout On Serving GPUs via Cooperative Elasticity for Agentic RL","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20863","citing_title":"PlexRL: Cluster-Level Orchestration of Serviceized LLM Execution for RLVR","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14617","citing_title":"Seer: Online Context Learning for Fast Synchronous LLM Reinforcement Learning","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2511.18871","citing_title":"Periodic Asynchrony: An On-Policy Approach for Accelerating LLM Reinforcement Learning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12476","citing_title":"HetRL: Efficient Reinforcement Learning for LLMs in Heterogeneous Environments","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2601.18150","citing_title":"FP8-RL: A Practical and Stable Low-Precision Stack for LLM Reinforcement Learning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2405.11143","citing_title":"OpenRLHF: An Easy-to-use, Scalable and High-performance RLHF Framework","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02507","citing_title":"Reinforcement Learning from Human Feedback: A Statistical Perspective","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26256","citing_title":"DORA: A Scalable Asynchronous Reinforcement Learning System for Language Model Training","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06534","citing_title":"ROSE: Rollout On Serving GPUs via Cooperative Elasticity for Agentic RL","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23838","citing_title":"JigsawRL: Assembling RL Pipelines for Efficient LLM Post-Training","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06230","citing_title":"Safactory: A Scalable Agentic Infrastructure for Training Trustworthy Autonomous Intelligence","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2409.19256","citing_title":"HybridFlow: A Flexible and Efficient RLHF Framework","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06230","citing_title":"Safactory: A Scalable Agentic Infrastructure for Training Trustworthy Autonomous Intelligence","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2303.18223","citing_title":"A Survey of Large Language Models","ref_index":214,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3EHXXDP43RT3FJMYYZLZT34YBS","json":"https://pith.science/pith/3EHXXDP43RT3FJMYYZLZT34YBS.json","graph_json":"https://pith.science/api/pith-number/3EHXXDP43RT3FJMYYZLZT34YBS/graph.json","events_json":"https://pith.science/api/pith-number/3EHXXDP43RT3FJMYYZLZT34YBS/events.json","paper":"https://pith.science/paper/3EHXXDP4"},"agent_actions":{"view_html":"https://pith.science/pith/3EHXXDP43RT3FJMYYZLZT34YBS","download_json":"https://pith.science/pith/3EHXXDP43RT3FJMYYZLZT34YBS.json","view_paper":"https://pith.science/paper/3EHXXDP4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.01320&json=true","fetch_graph":"https://pith.science/api/pith-number/3EHXXDP43RT3FJMYYZLZT34YBS/graph.json","fetch_events":"https://pith.science/api/pith-number/3EHXXDP43RT3FJMYYZLZT34YBS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3EHXXDP43RT3FJMYYZLZT34YBS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3EHXXDP43RT3FJMYYZLZT34YBS/action/storage_attestation","attest_author":"https://pith.science/pith/3EHXXDP43RT3FJMYYZLZT34YBS/action/author_attestation","sign_citation":"https://pith.science/pith/3EHXXDP43RT3FJMYYZLZT34YBS/action/citation_signature","submit_replication":"https://pith.science/pith/3EHXXDP43RT3FJMYYZLZT34YBS/action/replication_record"}},"created_at":"2026-07-05T06:37:17.028153+00:00","updated_at":"2026-07-05T06:37:17.028153+00:00"}