{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VIOU6QQI3Y74CJER4MC64XLZ4X","short_pith_number":"pith:VIOU6QQI","schema_version":"1.0","canonical_sha256":"aa1d4f4208de3fc12491e305ee5d79e5f6c7405587cdb777c8c1d3d613dadddc","source":{"kind":"arxiv","id":"2405.01481","version":2},"attestation_state":"computed","paper":{"title":"NeMo-Aligner: Scalable Toolkit for Efficient Model Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ali Taghibakhshi, Ashwath Aithal, Daniel Egert, Gerald Shen, Jiaqi Zeng, Jimmy Zhang, Markel Sanz Ausin, Oleksii Kuchaiev, Olivier Delalleau, Sahil Jain, Shengyang Sun, Yi Dong, Zhilin Wang","submitted_at":"2024-05-02T17:13:40Z","abstract_excerpt":"Aligning Large Language Models (LLMs) with human values and preferences is essential for making them helpful and safe. However, building efficient tools to perform alignment can be challenging, especially for the largest and most competent LLMs which often contain tens or hundreds of billions of parameters. We create NeMo-Aligner, a toolkit for model alignment that can efficiently scale to a thousand GPUs for training the largest open-source LLMs such as Nemotron 4 340B and Llama 3.1 405B. NeMo-Aligner comes with highly optimized and scalable implementations for major paradigms of model alignm"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.01481","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-05-02T17:13:40Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"a0fc50f14f980347d7d4748c3650e8614ac8705ccf1fb62bffbf555cb97df72d","abstract_canon_sha256":"59a3ea2437e63c2289d2921b931696bbad1489c996a93ebd87b88e2ca52cd096"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:02:18.974089Z","signature_b64":"Nw+PLlsEDCX6B2i4CidoAowSzdZp86i4oUomjjGYm//LS7S/n3tLZK/kHJLTIROH/U8PsHyfRWrnl5fvEr2hBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aa1d4f4208de3fc12491e305ee5d79e5f6c7405587cdb777c8c1d3d613dadddc","last_reissued_at":"2026-07-05T09:02:18.973664Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:02:18.973664Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NeMo-Aligner: Scalable Toolkit for Efficient Model Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ali Taghibakhshi, Ashwath Aithal, Daniel Egert, Gerald Shen, Jiaqi Zeng, Jimmy Zhang, Markel Sanz Ausin, Oleksii Kuchaiev, Olivier Delalleau, Sahil Jain, Shengyang Sun, Yi Dong, Zhilin Wang","submitted_at":"2024-05-02T17:13:40Z","abstract_excerpt":"Aligning Large Language Models (LLMs) with human values and preferences is essential for making them helpful and safe. However, building efficient tools to perform alignment can be challenging, especially for the largest and most competent LLMs which often contain tens or hundreds of billions of parameters. We create NeMo-Aligner, a toolkit for model alignment that can efficiently scale to a thousand GPUs for training the largest open-source LLMs such as Nemotron 4 340B and Llama 3.1 405B. NeMo-Aligner comes with highly optimized and scalable implementations for major paradigms of model alignm"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.01481","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.01481/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.01481","created_at":"2026-07-05T09:02:18.973724+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.01481v2","created_at":"2026-07-05T09:02:18.973724+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.01481","created_at":"2026-07-05T09:02:18.973724+00:00"},{"alias_kind":"pith_short_12","alias_value":"VIOU6QQI3Y74","created_at":"2026-07-05T09:02:18.973724+00:00"},{"alias_kind":"pith_short_16","alias_value":"VIOU6QQI3Y74CJER","created_at":"2026-07-05T09:02:18.973724+00:00"},{"alias_kind":"pith_short_8","alias_value":"VIOU6QQI","created_at":"2026-07-05T09:02:18.973724+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09711","citing_title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20863","citing_title":"PlexRL: Cluster-Level Orchestration of Serviceized LLM Execution for RLVR","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15565","citing_title":"AstraFlow: Dataflow-Oriented Reinforcement Learning for Agentic LLMs","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2510.19225","citing_title":"RLBoost: Harvesting Preemptible Resources for Cost-Efficient Reinforcement Learning on LLMs","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14617","citing_title":"Seer: Online Context Learning for Fast Synchronous LLM Reinforcement Learning","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12476","citing_title":"HetRL: Efficient Reinforcement Learning for LLMs in Heterogeneous Environments","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13907","citing_title":"AIS: Adaptive Importance Sampling for Quantized RL","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VIOU6QQI3Y74CJER4MC64XLZ4X","json":"https://pith.science/pith/VIOU6QQI3Y74CJER4MC64XLZ4X.json","graph_json":"https://pith.science/api/pith-number/VIOU6QQI3Y74CJER4MC64XLZ4X/graph.json","events_json":"https://pith.science/api/pith-number/VIOU6QQI3Y74CJER4MC64XLZ4X/events.json","paper":"https://pith.science/paper/VIOU6QQI"},"agent_actions":{"view_html":"https://pith.science/pith/VIOU6QQI3Y74CJER4MC64XLZ4X","download_json":"https://pith.science/pith/VIOU6QQI3Y74CJER4MC64XLZ4X.json","view_paper":"https://pith.science/paper/VIOU6QQI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.01481&json=true","fetch_graph":"https://pith.science/api/pith-number/VIOU6QQI3Y74CJER4MC64XLZ4X/graph.json","fetch_events":"https://pith.science/api/pith-number/VIOU6QQI3Y74CJER4MC64XLZ4X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VIOU6QQI3Y74CJER4MC64XLZ4X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VIOU6QQI3Y74CJER4MC64XLZ4X/action/storage_attestation","attest_author":"https://pith.science/pith/VIOU6QQI3Y74CJER4MC64XLZ4X/action/author_attestation","sign_citation":"https://pith.science/pith/VIOU6QQI3Y74CJER4MC64XLZ4X/action/citation_signature","submit_replication":"https://pith.science/pith/VIOU6QQI3Y74CJER4MC64XLZ4X/action/replication_record"}},"created_at":"2026-07-05T09:02:18.973724+00:00","updated_at":"2026-07-05T09:02:18.973724+00:00"}