{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:O7JAHY5B72UPXVL7DSXROSGFBV","short_pith_number":"pith:O7JAHY5B","schema_version":"1.0","canonical_sha256":"77d203e3a1fea8fbd57f1caf1748c50d5533bc295e571d5a70b390f133712aa2","source":{"kind":"arxiv","id":"2405.21015","version":2},"attestation_state":"computed","paper":{"title":"The rising costs of training frontier AI models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CY","authors_text":"Ben Cottier, David Owen, Loredana Fattorini, Nestor Maslej, Robi Rahman, Tamay Besiroglu","submitted_at":"2024-05-31T17:04:18Z","abstract_excerpt":"The costs of training frontier AI models have grown dramatically in recent years, but there is limited public data on the magnitude and growth of these expenses. This paper develops a detailed cost model to address this gap, estimating training costs using three approaches that account for hardware, energy, cloud rental, and staff expenses. The analysis reveals that the amortized cost to train the most compute-intensive models has grown precipitously at a rate of 2.4x per year since 2016 (90% CI: 2.0x to 2.9x). For key frontier models, such as GPT-4 and Gemini, the most significant expenses ar"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.21015","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CY","submitted_at":"2024-05-31T17:04:18Z","cross_cats_sorted":[],"title_canon_sha256":"b261f6441a5403ff3ae7efe6e4f087732c4ff2c8c46cc7e47e58502d28bfebb8","abstract_canon_sha256":"c07f819b3aa2e6d41f938f859f3c07919de935ce445047afd9735a21a0572650"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:11:20.303908Z","signature_b64":"YBALwupwh+oMkp/P7zoQoLjl6P4wrUS+/ZNylLfg0T49uaO9z8ZXqKIVMnnP/8fowrVTEWR9S5AlRTJwSYzoCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"77d203e3a1fea8fbd57f1caf1748c50d5533bc295e571d5a70b390f133712aa2","last_reissued_at":"2026-07-05T10:11:20.303266Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:11:20.303266Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The rising costs of training frontier AI models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CY","authors_text":"Ben Cottier, David Owen, Loredana Fattorini, Nestor Maslej, Robi Rahman, Tamay Besiroglu","submitted_at":"2024-05-31T17:04:18Z","abstract_excerpt":"The costs of training frontier AI models have grown dramatically in recent years, but there is limited public data on the magnitude and growth of these expenses. This paper develops a detailed cost model to address this gap, estimating training costs using three approaches that account for hardware, energy, cloud rental, and staff expenses. The analysis reveals that the amortized cost to train the most compute-intensive models has grown precipitously at a rate of 2.4x per year since 2016 (90% CI: 2.0x to 2.9x). For key frontier models, such as GPT-4 and Gemini, the most significant expenses ar"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.21015","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.21015/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.21015","created_at":"2026-07-05T10:11:20.303341+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.21015v2","created_at":"2026-07-05T10:11:20.303341+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.21015","created_at":"2026-07-05T10:11:20.303341+00:00"},{"alias_kind":"pith_short_12","alias_value":"O7JAHY5B72UP","created_at":"2026-07-05T10:11:20.303341+00:00"},{"alias_kind":"pith_short_16","alias_value":"O7JAHY5B72UPXVL7","created_at":"2026-07-05T10:11:20.303341+00:00"},{"alias_kind":"pith_short_8","alias_value":"O7JAHY5B","created_at":"2026-07-05T10:11:20.303341+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08095","citing_title":"zkComposer: Decomposing Proof Construction to Scale zkML","ref_index":9,"is_internal_anchor":true},{"citing_arxiv_id":"2606.23370","citing_title":"FlexServe: A Fast and Secure LLM Serving System for Mobile Devices with Flexible Resource Isolation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23546","citing_title":"The Energy Consumption of Transformer Fine-Tuning: A Roofline-Inspired Scaling Model","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05433","citing_title":"Zero knowledge verification for frontier AI training is possible","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05029","citing_title":"Validity Threats for Foundation Model Research","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03217","citing_title":"An Asymptotic Theory of Chain-of-Thought in In-Context Learning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07536","citing_title":"Beware of GeeksBearing Gifts: Building True EU Frontier AI Sovereignty","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07245","citing_title":"AI Sovereignty: A Qualitative Model of Strategic Competition as AI Becomes an Instrument of National Power","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24517","citing_title":"Physics Priors Offer Useful Accuracy-Carbon Trade-Offs in Spatio-Temporal Forecasting","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18860","citing_title":"Spectral structural distortion reveals redundant neurons in neural networks","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2509.07177","citing_title":"Towards EnergyGPT: A Large Language Model Specialized for the Energy Sector","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14053","citing_title":"LLMOrbit: A Circular Taxonomy of Large Language Models -From Scaling Walls to Agentic AI Systems","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09046","citing_title":"FlexServe: A Fast and Secure LLM Serving System for Mobile Devices with Flexible Resource Isolation","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06217","citing_title":"The End of the Foundation Model Era: Open-Weight Models, Sovereign AI, and Inference as Infrastructure","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13434","citing_title":"Rescaled Asynchronous SGD: Optimal Distributed Optimization under Data and System Heterogeneity","ref_index":138,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08524","citing_title":"Unleashing Scalable Context Parallelism for Foundation Models Pre-Training via FCP","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05558","citing_title":"Who Prices Cognitive Labor in the Age of Agents? Compute-Anchored Wages","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05416","citing_title":"From Cradle to Cloud: A Life Cycle Review of AI's Environmental Footprint","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05558","citing_title":"Who Prices Cognitive Labor in the Age of Agents? Compute-Anchored Wages","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05859","citing_title":"When Do We Need LLMs? A Diagnostic for Language-Driven Bandits","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14690","citing_title":"Switching Efficiency: A Novel Framework for Dissecting AI Data Center Network Efficiency","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17112","citing_title":"Complementing Self-Consistency with Cross-Model Disagreement for Uncertainty Quantification","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O7JAHY5B72UPXVL7DSXROSGFBV","json":"https://pith.science/pith/O7JAHY5B72UPXVL7DSXROSGFBV.json","graph_json":"https://pith.science/api/pith-number/O7JAHY5B72UPXVL7DSXROSGFBV/graph.json","events_json":"https://pith.science/api/pith-number/O7JAHY5B72UPXVL7DSXROSGFBV/events.json","paper":"https://pith.science/paper/O7JAHY5B"},"agent_actions":{"view_html":"https://pith.science/pith/O7JAHY5B72UPXVL7DSXROSGFBV","download_json":"https://pith.science/pith/O7JAHY5B72UPXVL7DSXROSGFBV.json","view_paper":"https://pith.science/paper/O7JAHY5B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.21015&json=true","fetch_graph":"https://pith.science/api/pith-number/O7JAHY5B72UPXVL7DSXROSGFBV/graph.json","fetch_events":"https://pith.science/api/pith-number/O7JAHY5B72UPXVL7DSXROSGFBV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O7JAHY5B72UPXVL7DSXROSGFBV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O7JAHY5B72UPXVL7DSXROSGFBV/action/storage_attestation","attest_author":"https://pith.science/pith/O7JAHY5B72UPXVL7DSXROSGFBV/action/author_attestation","sign_citation":"https://pith.science/pith/O7JAHY5B72UPXVL7DSXROSGFBV/action/citation_signature","submit_replication":"https://pith.science/pith/O7JAHY5B72UPXVL7DSXROSGFBV/action/replication_record"}},"created_at":"2026-07-05T10:11:20.303341+00:00","updated_at":"2026-07-05T10:11:20.303341+00:00"}