{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:27PPBAVBEIE4BMTABMRXYSTF6S","short_pith_number":"pith:27PPBAVB","schema_version":"1.0","canonical_sha256":"d7def082a12209c0b2600b237c4a65f4adc34113c079b395d0aaaf54f6ea6929","source":{"kind":"arxiv","id":"2408.04693","version":1},"attestation_state":"computed","paper":{"title":"Understanding the Performance and Estimating the Cost of LLM Fine-Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Cong Hao, Haojie Ye, Jiho Kim, Nishil Talati, Souvik Kundu, Yuchen Xia, Yuhan Chen","submitted_at":"2024-08-08T16:26:07Z","abstract_excerpt":"Due to the cost-prohibitive nature of training Large Language Models (LLMs), fine-tuning has emerged as an attractive alternative for specializing LLMs for specific tasks using limited compute resources in a cost-effective manner. In this paper, we characterize sparse Mixture of Experts (MoE) based LLM fine-tuning to understand their accuracy and runtime performance on a single GPU. Our evaluation provides unique insights into the training efficacy of sparse and dense versions of MoE models, as well as their runtime characteristics, including maximum batch size, execution time breakdown, end-t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.04693","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-08-08T16:26:07Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"a3b107af1dbc658a3826eda0ad157706237072f0d9934171f192909f2c39bc26","abstract_canon_sha256":"2cbdacf28a3cd84a3f35baea9740d2fc1ca6b2760a8166e480f6cf82c29742f6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:55:23.813171Z","signature_b64":"02CzPm84FSTEV0s1m1K2cg0xAqP6SaI0iIgL1hgwo5zilEbcAjyjm2K+eCHPOBSZQWffGkOZF28htIMr6DLBDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d7def082a12209c0b2600b237c4a65f4adc34113c079b395d0aaaf54f6ea6929","last_reissued_at":"2026-07-05T08:55:23.812779Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:55:23.812779Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding the Performance and Estimating the Cost of LLM Fine-Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Cong Hao, Haojie Ye, Jiho Kim, Nishil Talati, Souvik Kundu, Yuchen Xia, Yuhan Chen","submitted_at":"2024-08-08T16:26:07Z","abstract_excerpt":"Due to the cost-prohibitive nature of training Large Language Models (LLMs), fine-tuning has emerged as an attractive alternative for specializing LLMs for specific tasks using limited compute resources in a cost-effective manner. In this paper, we characterize sparse Mixture of Experts (MoE) based LLM fine-tuning to understand their accuracy and runtime performance on a single GPU. Our evaluation provides unique insights into the training efficacy of sparse and dense versions of MoE models, as well as their runtime characteristics, including maximum batch size, execution time breakdown, end-t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.04693","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.04693/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.04693","created_at":"2026-07-05T08:55:23.812835+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.04693v1","created_at":"2026-07-05T08:55:23.812835+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.04693","created_at":"2026-07-05T08:55:23.812835+00:00"},{"alias_kind":"pith_short_12","alias_value":"27PPBAVBEIE4","created_at":"2026-07-05T08:55:23.812835+00:00"},{"alias_kind":"pith_short_16","alias_value":"27PPBAVBEIE4BMTA","created_at":"2026-07-05T08:55:23.812835+00:00"},{"alias_kind":"pith_short_8","alias_value":"27PPBAVB","created_at":"2026-07-05T08:55:23.812835+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.21316","citing_title":"Deep Optimizer States: Towards Scalable Training of Transformer Models Using Interleaved Offloading","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2410.20791","citing_title":"From Cool Demos to Production-Ready FMware: Core Challenges and a Technology Roadmap","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2509.13047","citing_title":"Multi-Model Synthetic Training for Mission-Critical Small Language Models","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/27PPBAVBEIE4BMTABMRXYSTF6S","json":"https://pith.science/pith/27PPBAVBEIE4BMTABMRXYSTF6S.json","graph_json":"https://pith.science/api/pith-number/27PPBAVBEIE4BMTABMRXYSTF6S/graph.json","events_json":"https://pith.science/api/pith-number/27PPBAVBEIE4BMTABMRXYSTF6S/events.json","paper":"https://pith.science/paper/27PPBAVB"},"agent_actions":{"view_html":"https://pith.science/pith/27PPBAVBEIE4BMTABMRXYSTF6S","download_json":"https://pith.science/pith/27PPBAVBEIE4BMTABMRXYSTF6S.json","view_paper":"https://pith.science/paper/27PPBAVB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.04693&json=true","fetch_graph":"https://pith.science/api/pith-number/27PPBAVBEIE4BMTABMRXYSTF6S/graph.json","fetch_events":"https://pith.science/api/pith-number/27PPBAVBEIE4BMTABMRXYSTF6S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/27PPBAVBEIE4BMTABMRXYSTF6S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/27PPBAVBEIE4BMTABMRXYSTF6S/action/storage_attestation","attest_author":"https://pith.science/pith/27PPBAVBEIE4BMTABMRXYSTF6S/action/author_attestation","sign_citation":"https://pith.science/pith/27PPBAVBEIE4BMTABMRXYSTF6S/action/citation_signature","submit_replication":"https://pith.science/pith/27PPBAVBEIE4BMTABMRXYSTF6S/action/replication_record"}},"created_at":"2026-07-05T08:55:23.812835+00:00","updated_at":"2026-07-05T08:55:23.812835+00:00"}