{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:XAP5KIWTSH36BZA7ZNGOLXNA2X","short_pith_number":"pith:XAP5KIWT","schema_version":"1.0","canonical_sha256":"b81fd522d391f7e0e41fcb4ce5dda0d5e22de8efd9bf0071709f267d66116284","source":{"kind":"arxiv","id":"2312.12391","version":2},"attestation_state":"computed","paper":{"title":"vTrain: A Simulation Framework for Evaluating Cost-effective and Compute-optimal Large Language Model Training","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.AR"],"primary_cat":"cs.LG","authors_text":"Jehyeon Bang, Minsoo Rhu, Myeongwoo Kim, Yongdeok Kim, Yujeong Choi","submitted_at":"2023-11-27T13:35:15Z","abstract_excerpt":"As large language models (LLMs) become widespread in various application domains, a critical challenge the AI community is facing is how to train these large AI models in a cost-effective manner. Existing LLM training plans typically employ a heuristic based parallel training strategy which is based on empirical observations rather than grounded upon a thorough examination of the search space of LLM parallelization. Such limitation renders existing systems to leave significant performance left on the table, wasting millions of dollars worth of training cost. This paper presents our profiling-d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.12391","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2023-11-27T13:35:15Z","cross_cats_sorted":["cs.AI","cs.AR"],"title_canon_sha256":"74c4be3a0f567a0b14591470ee1baa7c583ebd63e4c996ef73cd9d6e278150af","abstract_canon_sha256":"97b78da3a2ff888231699c55f3038ab80107231a4873a644070d04a54812706d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:05:12.147177Z","signature_b64":"VzA1CglEkav1/0nq8tKhqNuQRShowIobJkTJ4E+MDRjAsulqkJ5JyPnLQJOj1NwKciNehm6a9E2a5PMHnt/jDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b81fd522d391f7e0e41fcb4ce5dda0d5e22de8efd9bf0071709f267d66116284","last_reissued_at":"2026-07-05T09:05:12.146754Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:05:12.146754Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"vTrain: A Simulation Framework for Evaluating Cost-effective and Compute-optimal Large Language Model Training","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.AR"],"primary_cat":"cs.LG","authors_text":"Jehyeon Bang, Minsoo Rhu, Myeongwoo Kim, Yongdeok Kim, Yujeong Choi","submitted_at":"2023-11-27T13:35:15Z","abstract_excerpt":"As large language models (LLMs) become widespread in various application domains, a critical challenge the AI community is facing is how to train these large AI models in a cost-effective manner. Existing LLM training plans typically employ a heuristic based parallel training strategy which is based on empirical observations rather than grounded upon a thorough examination of the search space of LLM parallelization. Such limitation renders existing systems to leave significant performance left on the table, wasting millions of dollars worth of training cost. This paper presents our profiling-d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.12391","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.12391/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.12391","created_at":"2026-07-05T09:05:12.146811+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.12391v2","created_at":"2026-07-05T09:05:12.146811+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.12391","created_at":"2026-07-05T09:05:12.146811+00:00"},{"alias_kind":"pith_short_12","alias_value":"XAP5KIWTSH36","created_at":"2026-07-05T09:05:12.146811+00:00"},{"alias_kind":"pith_short_16","alias_value":"XAP5KIWTSH36BZA7","created_at":"2026-07-05T09:05:12.146811+00:00"},{"alias_kind":"pith_short_8","alias_value":"XAP5KIWT","created_at":"2026-07-05T09:05:12.146811+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.14910","citing_title":"PipeWeave: Synergizing Analytical and Learning Models for Unified GPU Performance Prediction","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XAP5KIWTSH36BZA7ZNGOLXNA2X","json":"https://pith.science/pith/XAP5KIWTSH36BZA7ZNGOLXNA2X.json","graph_json":"https://pith.science/api/pith-number/XAP5KIWTSH36BZA7ZNGOLXNA2X/graph.json","events_json":"https://pith.science/api/pith-number/XAP5KIWTSH36BZA7ZNGOLXNA2X/events.json","paper":"https://pith.science/paper/XAP5KIWT"},"agent_actions":{"view_html":"https://pith.science/pith/XAP5KIWTSH36BZA7ZNGOLXNA2X","download_json":"https://pith.science/pith/XAP5KIWTSH36BZA7ZNGOLXNA2X.json","view_paper":"https://pith.science/paper/XAP5KIWT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.12391&json=true","fetch_graph":"https://pith.science/api/pith-number/XAP5KIWTSH36BZA7ZNGOLXNA2X/graph.json","fetch_events":"https://pith.science/api/pith-number/XAP5KIWTSH36BZA7ZNGOLXNA2X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XAP5KIWTSH36BZA7ZNGOLXNA2X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XAP5KIWTSH36BZA7ZNGOLXNA2X/action/storage_attestation","attest_author":"https://pith.science/pith/XAP5KIWTSH36BZA7ZNGOLXNA2X/action/author_attestation","sign_citation":"https://pith.science/pith/XAP5KIWTSH36BZA7ZNGOLXNA2X/action/citation_signature","submit_replication":"https://pith.science/pith/XAP5KIWTSH36BZA7ZNGOLXNA2X/action/replication_record"}},"created_at":"2026-07-05T09:05:12.146811+00:00","updated_at":"2026-07-05T09:05:12.146811+00:00"}