{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BI7CMDXCG7KUSOZGTDDSVWESR4","short_pith_number":"pith:BI7CMDXC","schema_version":"1.0","canonical_sha256":"0a3e260ee237d5493b2698c72ad8928f3a6f7e6b57a53718b80aebe5cd65bf48","source":{"kind":"arxiv","id":"2312.02696","version":2},"attestation_state":"computed","paper":{"title":"Analyzing and Improving the Training Dynamics of Diffusion Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.NE","stat.ML"],"primary_cat":"cs.CV","authors_text":"Jaakko Lehtinen, Janne Hellsten, Miika Aittala, Samuli Laine, Tero Karras, Timo Aila","submitted_at":"2023-12-05T11:55:47Z","abstract_excerpt":"Diffusion models currently dominate the field of data-driven image synthesis with their unparalleled scaling to large datasets. In this paper, we identify and rectify several causes for uneven and ineffective training in the popular ADM diffusion model architecture, without altering its high-level structure. Observing uncontrolled magnitude changes and imbalances in both the network activations and weights over the course of training, we redesign the network layers to preserve activation, weight, and update magnitudes on expectation. We find that systematic application of this philosophy elimi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.02696","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-12-05T11:55:47Z","cross_cats_sorted":["cs.AI","cs.LG","cs.NE","stat.ML"],"title_canon_sha256":"f7543147a9d4c8c6a4876953cfaba645fc52bf64ae3f4968c5fcdb0914f47b39","abstract_canon_sha256":"662a27090aafc96120c1a21400eb97996473c73dcbeb44b2ab097a127a03e9ae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:58:25.471191Z","signature_b64":"rhQ5bocnDDmNShRiRdj42eFLoTOEQuy81ihuDD1PapXoMfBlMILRIcJBGGz8pTSWVb+UJEzLop84XdG/flvsAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0a3e260ee237d5493b2698c72ad8928f3a6f7e6b57a53718b80aebe5cd65bf48","last_reissued_at":"2026-07-05T07:58:25.470615Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:58:25.470615Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Analyzing and Improving the Training Dynamics of Diffusion Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.NE","stat.ML"],"primary_cat":"cs.CV","authors_text":"Jaakko Lehtinen, Janne Hellsten, Miika Aittala, Samuli Laine, Tero Karras, Timo Aila","submitted_at":"2023-12-05T11:55:47Z","abstract_excerpt":"Diffusion models currently dominate the field of data-driven image synthesis with their unparalleled scaling to large datasets. In this paper, we identify and rectify several causes for uneven and ineffective training in the popular ADM diffusion model architecture, without altering its high-level structure. Observing uncontrolled magnitude changes and imbalances in both the network activations and weights over the course of training, we redesign the network layers to preserve activation, weight, and update magnitudes on expectation. We find that systematic application of this philosophy elimi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.02696","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.02696/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.02696","created_at":"2026-07-05T07:58:25.470672+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.02696v2","created_at":"2026-07-05T07:58:25.470672+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.02696","created_at":"2026-07-05T07:58:25.470672+00:00"},{"alias_kind":"pith_short_12","alias_value":"BI7CMDXCG7KU","created_at":"2026-07-05T07:58:25.470672+00:00"},{"alias_kind":"pith_short_16","alias_value":"BI7CMDXCG7KUSOZG","created_at":"2026-07-05T07:58:25.470672+00:00"},{"alias_kind":"pith_short_8","alias_value":"BI7CMDXC","created_at":"2026-07-05T07:58:25.470672+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.08505","citing_title":"Diffusion Models for Sampling Near Criticality in Lattice Field Theories","ref_index":66,"is_internal_anchor":true},{"citing_arxiv_id":"2607.06335","citing_title":"Bridging Diffusion Pruning and Step Distillation with Teacher-Aligned Repair","ref_index":4,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25971","citing_title":"Improving Neural Network Training by Decoupling the Magnitude and Direction of Weight Vectors","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18267","citing_title":"SRC-Flow: Compact Semantic Representations Enable Normalizing Flows for Image Generation","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21489","citing_title":"Variance Reduction for Expectations with Diffusion Teachers","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2410.06128","citing_title":"Amortized Inference of Causal Models via Conditional Fixed-Point Iterations","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2603.13419","citing_title":"Diffusion Models Memorize in Training -- and Generalize in Inference","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21489","citing_title":"Variance Reduction for Expectations with Diffusion Teachers","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21433","citing_title":"Instrumental Text-to-Music Generation with Auxiliary Conditioning Branches","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16520","citing_title":"Global Convergence of Sampling-Based Nonconvex Optimization through Diffusion-Style Smoothing","ref_index":117,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18267","citing_title":"SRC-Flow: Compact Semantic Representations Enable Normalizing Flows for Image Generation","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16818","citing_title":"Observation-Aligned Mask Priors for Learning Physical Dynamics from Authentic Occlusions","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2506.16827","citing_title":"Beyond Blur: A Fluid Perspective on Generative Diffusion Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2510.21890","citing_title":"The Principles of Diffusion Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2603.20092","citing_title":"How Out-of-Equilibrium Phase Transitions can Seed Pattern Formation in Trained Diffusion Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2603.28743","citing_title":"Rethinking Language Model Scaling under Transferable Hypersphere Optimization","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2403.03206","citing_title":"Scaling Rectified Flow Transformers for High-Resolution Image Synthesis","ref_index":146,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10019","citing_title":"The two clocks and the innovation window: When and how generative models learn rules","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13521","citing_title":"C-voting: Confidence-Based Test-Time Voting without Explicit Energy Functions","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BI7CMDXCG7KUSOZGTDDSVWESR4","json":"https://pith.science/pith/BI7CMDXCG7KUSOZGTDDSVWESR4.json","graph_json":"https://pith.science/api/pith-number/BI7CMDXCG7KUSOZGTDDSVWESR4/graph.json","events_json":"https://pith.science/api/pith-number/BI7CMDXCG7KUSOZGTDDSVWESR4/events.json","paper":"https://pith.science/paper/BI7CMDXC"},"agent_actions":{"view_html":"https://pith.science/pith/BI7CMDXCG7KUSOZGTDDSVWESR4","download_json":"https://pith.science/pith/BI7CMDXCG7KUSOZGTDDSVWESR4.json","view_paper":"https://pith.science/paper/BI7CMDXC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.02696&json=true","fetch_graph":"https://pith.science/api/pith-number/BI7CMDXCG7KUSOZGTDDSVWESR4/graph.json","fetch_events":"https://pith.science/api/pith-number/BI7CMDXCG7KUSOZGTDDSVWESR4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BI7CMDXCG7KUSOZGTDDSVWESR4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BI7CMDXCG7KUSOZGTDDSVWESR4/action/storage_attestation","attest_author":"https://pith.science/pith/BI7CMDXCG7KUSOZGTDDSVWESR4/action/author_attestation","sign_citation":"https://pith.science/pith/BI7CMDXCG7KUSOZGTDDSVWESR4/action/citation_signature","submit_replication":"https://pith.science/pith/BI7CMDXCG7KUSOZGTDDSVWESR4/action/replication_record"}},"created_at":"2026-07-05T07:58:25.470672+00:00","updated_at":"2026-07-05T07:58:25.470672+00:00"}