{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FOUZA6RR76L7YCDMXSHSECRNS5","short_pith_number":"pith:FOUZA6RR","schema_version":"1.0","canonical_sha256":"2ba9907a31ff97fc086cbc8f220a2d977a10dda53be71e4e9d9fb4c105385d52","source":{"kind":"arxiv","id":"2307.14653","version":1},"attestation_state":"computed","paper":{"title":"Speed Limits for Deep Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cond-mat.dis-nn","cs.LG"],"primary_cat":"stat.ML","authors_text":"Alexander A. Alemi, Inbar Seroussi, Moritz Helias, Zohar Ringel","submitted_at":"2023-07-27T06:59:46Z","abstract_excerpt":"State-of-the-art neural networks require extreme computational power to train. It is therefore natural to wonder whether they are optimally trained. Here we apply a recent advancement in stochastic thermodynamics which allows bounding the speed at which one can go from the initial weight distribution to the final distribution of the fully trained network, based on the ratio of their Wasserstein-2 distance and the entropy production rate of the dynamical process connecting them. Considering both gradient-flow and Langevin training dynamics, we provide analytical expressions for these speed limi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.14653","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"stat.ML","submitted_at":"2023-07-27T06:59:46Z","cross_cats_sorted":["cond-mat.dis-nn","cs.LG"],"title_canon_sha256":"30f8fcaebc6ff79fae4e6ee6dce5180f810a1f9c60734900adcd6ace13a3413c","abstract_canon_sha256":"a004acca095b8b4070669905750fb5d2b5f51f05403eaab24780edc18e15fcef"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:35:16.153679Z","signature_b64":"mKBwUVq7ihe80VZyiOZ/qiNACqsEPmb5crrHQCYwpMeBZ4JFNpH3UzJnB56YO6LFgirtZxeNpLMviNd7AQBnDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2ba9907a31ff97fc086cbc8f220a2d977a10dda53be71e4e9d9fb4c105385d52","last_reissued_at":"2026-07-05T06:35:16.153117Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:35:16.153117Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Speed Limits for Deep Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cond-mat.dis-nn","cs.LG"],"primary_cat":"stat.ML","authors_text":"Alexander A. Alemi, Inbar Seroussi, Moritz Helias, Zohar Ringel","submitted_at":"2023-07-27T06:59:46Z","abstract_excerpt":"State-of-the-art neural networks require extreme computational power to train. It is therefore natural to wonder whether they are optimally trained. Here we apply a recent advancement in stochastic thermodynamics which allows bounding the speed at which one can go from the initial weight distribution to the final distribution of the fully trained network, based on the ratio of their Wasserstein-2 distance and the entropy production rate of the dynamical process connecting them. Considering both gradient-flow and Langevin training dynamics, we provide analytical expressions for these speed limi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.14653","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.14653/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.14653","created_at":"2026-07-05T06:35:16.153178+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.14653v1","created_at":"2026-07-05T06:35:16.153178+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.14653","created_at":"2026-07-05T06:35:16.153178+00:00"},{"alias_kind":"pith_short_12","alias_value":"FOUZA6RR76L7","created_at":"2026-07-05T06:35:16.153178+00:00"},{"alias_kind":"pith_short_16","alias_value":"FOUZA6RR76L7YCDM","created_at":"2026-07-05T06:35:16.153178+00:00"},{"alias_kind":"pith_short_8","alias_value":"FOUZA6RR","created_at":"2026-07-05T06:35:16.153178+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.19015","citing_title":"Learning with springs and sticks","ref_index":50,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FOUZA6RR76L7YCDMXSHSECRNS5","json":"https://pith.science/pith/FOUZA6RR76L7YCDMXSHSECRNS5.json","graph_json":"https://pith.science/api/pith-number/FOUZA6RR76L7YCDMXSHSECRNS5/graph.json","events_json":"https://pith.science/api/pith-number/FOUZA6RR76L7YCDMXSHSECRNS5/events.json","paper":"https://pith.science/paper/FOUZA6RR"},"agent_actions":{"view_html":"https://pith.science/pith/FOUZA6RR76L7YCDMXSHSECRNS5","download_json":"https://pith.science/pith/FOUZA6RR76L7YCDMXSHSECRNS5.json","view_paper":"https://pith.science/paper/FOUZA6RR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.14653&json=true","fetch_graph":"https://pith.science/api/pith-number/FOUZA6RR76L7YCDMXSHSECRNS5/graph.json","fetch_events":"https://pith.science/api/pith-number/FOUZA6RR76L7YCDMXSHSECRNS5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FOUZA6RR76L7YCDMXSHSECRNS5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FOUZA6RR76L7YCDMXSHSECRNS5/action/storage_attestation","attest_author":"https://pith.science/pith/FOUZA6RR76L7YCDMXSHSECRNS5/action/author_attestation","sign_citation":"https://pith.science/pith/FOUZA6RR76L7YCDMXSHSECRNS5/action/citation_signature","submit_replication":"https://pith.science/pith/FOUZA6RR76L7YCDMXSHSECRNS5/action/replication_record"}},"created_at":"2026-07-05T06:35:16.153178+00:00","updated_at":"2026-07-05T06:35:16.153178+00:00"}