{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RL3ZAZYHLSMKEPYNYYQRQUTJT5","short_pith_number":"pith:RL3ZAZYH","schema_version":"1.0","canonical_sha256":"8af79067075c98a23f0dc6211852699f6e25ad98824c1c3dde95129be67c7774","source":{"kind":"arxiv","id":"2407.05872","version":2},"attestation_state":"computed","paper":{"title":"Scaling Exponents Across Parameterizations and Optimizers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alexander A. Alemi, Izzeddin Gur, Jaehoon Lee, Jascha Sohl-Dickstein, Jeffrey Pennington, Katie Everett, Lechao Xiao, Leslie Pack Kaelbling, Mitchell Wortsman, Peter J. Liu, Roman Novak","submitted_at":"2024-07-08T12:32:51Z","abstract_excerpt":"Robust and effective scaling of models from small to large width typically requires the precise adjustment of many algorithmic and architectural details, such as parameterization and optimizer choices. In this work, we propose a new perspective on parameterization by investigating a key assumption in prior work about the alignment between parameters and data and derive new theoretical results under weaker assumptions and a broader set of optimizers. Our extensive empirical investigation includes tens of thousands of models trained with all combinations of three optimizers, four parameterizatio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.05872","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-07-08T12:32:51Z","cross_cats_sorted":[],"title_canon_sha256":"476e1a11fd78f642353fad8096a6739d1b0bcca7100d47088cbc587c4a96f0be","abstract_canon_sha256":"4e4e1f581a05c43e2d00a650d774de1494a1b44f3f03b8fe59866b844bd859a3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:44:39.244459Z","signature_b64":"pgBmzfM2GUaYdMmD8tI7tjRsJ20IeJf+jX6xH8+VcmDZXr1GQauRA/SGUupKlnMWR9ocoGM/TEAp9hof+vlgBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8af79067075c98a23f0dc6211852699f6e25ad98824c1c3dde95129be67c7774","last_reissued_at":"2026-07-05T08:44:39.244004Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:44:39.244004Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Exponents Across Parameterizations and Optimizers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alexander A. Alemi, Izzeddin Gur, Jaehoon Lee, Jascha Sohl-Dickstein, Jeffrey Pennington, Katie Everett, Lechao Xiao, Leslie Pack Kaelbling, Mitchell Wortsman, Peter J. Liu, Roman Novak","submitted_at":"2024-07-08T12:32:51Z","abstract_excerpt":"Robust and effective scaling of models from small to large width typically requires the precise adjustment of many algorithmic and architectural details, such as parameterization and optimizer choices. In this work, we propose a new perspective on parameterization by investigating a key assumption in prior work about the alignment between parameters and data and derive new theoretical results under weaker assumptions and a broader set of optimizers. Our extensive empirical investigation includes tens of thousands of models trained with all combinations of three optimizers, four parameterizatio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.05872","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.05872/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.05872","created_at":"2026-07-05T08:44:39.244061+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.05872v2","created_at":"2026-07-05T08:44:39.244061+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.05872","created_at":"2026-07-05T08:44:39.244061+00:00"},{"alias_kind":"pith_short_12","alias_value":"RL3ZAZYHLSMK","created_at":"2026-07-05T08:44:39.244061+00:00"},{"alias_kind":"pith_short_16","alias_value":"RL3ZAZYHLSMKEPYN","created_at":"2026-07-05T08:44:39.244061+00:00"},{"alias_kind":"pith_short_8","alias_value":"RL3ZAZYH","created_at":"2026-07-05T08:44:39.244061+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25971","citing_title":"Improving Neural Network Training by Decoupling the Magnitude and Direction of Weight Vectors","ref_index":168,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20299","citing_title":"Statistical Properties of Training & Generalization","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20299","citing_title":"Statistical Properties of Training & Generalization","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2501.07237","citing_title":"GWT: Scalable Optimizer State Compression for Large Language Model Training","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05074","citing_title":"Two-Point Deterministic Equivalence for Stochastic Gradient Dynamics in Linear Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15290","citing_title":"GQA-{\\mu}P: The maximal parameterization update for grouped query attention","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2602.04774","citing_title":"Theory of Optimal Learning Rate Schedules and Scaling Laws for a Random Feature Model","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14200","citing_title":"How to Scale Mixture-of-Experts: From muP to the Maximally Scale-Stable Parameterization","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11170","citing_title":"Unlearning with Asymmetric Sources: Improved Unlearning-Utility Trade-off with Public Data","ref_index":141,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05171","citing_title":"Scaling up Test-Time Compute with Latent Reasoning: A Recurrent Depth Approach","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27077","citing_title":"Learning Rate Transfer in Normalized Transformers","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12946","citing_title":"Parcae: Scaling Laws For Stable Looped Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13521","citing_title":"C-voting: Confidence-Based Test-Time Voting without Explicit Energy Functions","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RL3ZAZYHLSMKEPYNYYQRQUTJT5","json":"https://pith.science/pith/RL3ZAZYHLSMKEPYNYYQRQUTJT5.json","graph_json":"https://pith.science/api/pith-number/RL3ZAZYHLSMKEPYNYYQRQUTJT5/graph.json","events_json":"https://pith.science/api/pith-number/RL3ZAZYHLSMKEPYNYYQRQUTJT5/events.json","paper":"https://pith.science/paper/RL3ZAZYH"},"agent_actions":{"view_html":"https://pith.science/pith/RL3ZAZYHLSMKEPYNYYQRQUTJT5","download_json":"https://pith.science/pith/RL3ZAZYHLSMKEPYNYYQRQUTJT5.json","view_paper":"https://pith.science/paper/RL3ZAZYH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.05872&json=true","fetch_graph":"https://pith.science/api/pith-number/RL3ZAZYHLSMKEPYNYYQRQUTJT5/graph.json","fetch_events":"https://pith.science/api/pith-number/RL3ZAZYHLSMKEPYNYYQRQUTJT5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RL3ZAZYHLSMKEPYNYYQRQUTJT5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RL3ZAZYHLSMKEPYNYYQRQUTJT5/action/storage_attestation","attest_author":"https://pith.science/pith/RL3ZAZYHLSMKEPYNYYQRQUTJT5/action/author_attestation","sign_citation":"https://pith.science/pith/RL3ZAZYHLSMKEPYNYYQRQUTJT5/action/citation_signature","submit_replication":"https://pith.science/pith/RL3ZAZYHLSMKEPYNYYQRQUTJT5/action/replication_record"}},"created_at":"2026-07-05T08:44:39.244061+00:00","updated_at":"2026-07-05T08:44:39.244061+00:00"}