{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JCHGLE46KUTOVONC7J6EFQOQK3","short_pith_number":"pith:JCHGLE46","schema_version":"1.0","canonical_sha256":"488e65939e5526eab9a2fa7c42c1d056d9ad8f84598cbd31783972cebe39707b","source":{"kind":"arxiv","id":"2507.03211","version":1},"attestation_state":"computed","paper":{"title":"DistZO2: High-Throughput and Memory-Efficient Zeroth-Order Fine-tuning LLMs with Distributed Parallel Computing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.PF"],"primary_cat":"cs.LG","authors_text":"Di Wang, Huanyi Xie, Liangyu Wang","submitted_at":"2025-07-03T22:53:34Z","abstract_excerpt":"Fine-tuning large language models (LLMs) remains resource-intensive due to their sheer scale. While zeroth-order (ZO) optimization provides a memory-efficient alternative by eliminating backward passes, its application to multi-hundred-billion-parameter models is constrained by GPU memory and compute throughput. The ZO2 framework addresses the memory bottleneck by offloading model parameters to CPU memory and overlapping transformer block transfer with dual forward computation on a single GPU. However, ZO2 remains limited by its single-device execution and achieves modest throughput. In this w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.03211","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-03T22:53:34Z","cross_cats_sorted":["cs.PF"],"title_canon_sha256":"4d9eb3c5ee2052605ab34178d9b8ce1a0e5188e1bb75b6546a1ceb2785c56b70","abstract_canon_sha256":"c33e099df263280c143fc3ae742b08c2f047cba8bccbe2aeba272c5010646450"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:31:54.817739Z","signature_b64":"vKI47DoA8wtNshxwzwK/GJ6jXAG2lYP8/o4OEF27i/ccQM4GN3w/O2F91SohwcZ3vbQgVQ1v2oq6ArvSSe67Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"488e65939e5526eab9a2fa7c42c1d056d9ad8f84598cbd31783972cebe39707b","last_reissued_at":"2026-07-05T11:31:54.817248Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:31:54.817248Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DistZO2: High-Throughput and Memory-Efficient Zeroth-Order Fine-tuning LLMs with Distributed Parallel Computing","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.PF"],"primary_cat":"cs.LG","authors_text":"Di Wang, Huanyi Xie, Liangyu Wang","submitted_at":"2025-07-03T22:53:34Z","abstract_excerpt":"Fine-tuning large language models (LLMs) remains resource-intensive due to their sheer scale. While zeroth-order (ZO) optimization provides a memory-efficient alternative by eliminating backward passes, its application to multi-hundred-billion-parameter models is constrained by GPU memory and compute throughput. The ZO2 framework addresses the memory bottleneck by offloading model parameters to CPU memory and overlapping transformer block transfer with dual forward computation on a single GPU. However, ZO2 remains limited by its single-device execution and achieves modest throughput. In this w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.03211","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.03211/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.03211","created_at":"2026-07-05T11:31:54.817298+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.03211v1","created_at":"2026-07-05T11:31:54.817298+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.03211","created_at":"2026-07-05T11:31:54.817298+00:00"},{"alias_kind":"pith_short_12","alias_value":"JCHGLE46KUTO","created_at":"2026-07-05T11:31:54.817298+00:00"},{"alias_kind":"pith_short_16","alias_value":"JCHGLE46KUTOVONC","created_at":"2026-07-05T11:31:54.817298+00:00"},{"alias_kind":"pith_short_8","alias_value":"JCHGLE46","created_at":"2026-07-05T11:31:54.817298+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.31272","citing_title":"Algorithmic Recourse of In-Context Learning for Tabular Data","ref_index":46,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JCHGLE46KUTOVONC7J6EFQOQK3","json":"https://pith.science/pith/JCHGLE46KUTOVONC7J6EFQOQK3.json","graph_json":"https://pith.science/api/pith-number/JCHGLE46KUTOVONC7J6EFQOQK3/graph.json","events_json":"https://pith.science/api/pith-number/JCHGLE46KUTOVONC7J6EFQOQK3/events.json","paper":"https://pith.science/paper/JCHGLE46"},"agent_actions":{"view_html":"https://pith.science/pith/JCHGLE46KUTOVONC7J6EFQOQK3","download_json":"https://pith.science/pith/JCHGLE46KUTOVONC7J6EFQOQK3.json","view_paper":"https://pith.science/paper/JCHGLE46","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.03211&json=true","fetch_graph":"https://pith.science/api/pith-number/JCHGLE46KUTOVONC7J6EFQOQK3/graph.json","fetch_events":"https://pith.science/api/pith-number/JCHGLE46KUTOVONC7J6EFQOQK3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JCHGLE46KUTOVONC7J6EFQOQK3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JCHGLE46KUTOVONC7J6EFQOQK3/action/storage_attestation","attest_author":"https://pith.science/pith/JCHGLE46KUTOVONC7J6EFQOQK3/action/author_attestation","sign_citation":"https://pith.science/pith/JCHGLE46KUTOVONC7J6EFQOQK3/action/citation_signature","submit_replication":"https://pith.science/pith/JCHGLE46KUTOVONC7J6EFQOQK3/action/replication_record"}},"created_at":"2026-07-05T11:31:54.817298+00:00","updated_at":"2026-07-05T11:31:54.817298+00:00"}