{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:U6YWBZOSIV6XDTUUUC7W6U4NIQ","short_pith_number":"pith:U6YWBZOS","schema_version":"1.0","canonical_sha256":"a7b160e5d2457d71ce94a0bf6f538d4416fc7b263c4eee2662fad5199f15654c","source":{"kind":"arxiv","id":"2501.15785","version":2},"attestation_state":"computed","paper":{"title":"Memorization and Regularization in Generative Diffusion Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.DS","math.OC"],"primary_cat":"cs.LG","authors_text":"Agnimitra Dasgupta, Andrew M. Stuart, Assad Oberai, Nikola B. Kovachki, Ricardo Baptista","submitted_at":"2025-01-27T05:17:06Z","abstract_excerpt":"Diffusion models have emerged as a powerful framework for generative modeling. At the heart of the methodology is score matching: learning gradients of families of log-densities for noisy versions of the data distribution at different scales. When the loss function adopted in score matching is evaluated using empirical data, rather than the population loss, the minimizer corresponds to the score of a time-dependent Gaussian mixture. However, use of this analytically tractable minimizer leads to data memorization: in both unconditioned and conditioned settings, the generative model returns the "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.15785","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-01-27T05:17:06Z","cross_cats_sorted":["math.DS","math.OC"],"title_canon_sha256":"a4de21f327379146b57255922e32be856e1f9558d74babffecaa341314c385a9","abstract_canon_sha256":"4ca63aba404bba6704d5caf89b60399aedaab662f54945e3eee578c007a93228"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:33:42.903384Z","signature_b64":"HvRTG0aOoR5NrbEPSdvr3iGcBsirapmuQBy3pPPGFegYYnIrlY9kroKc5j6c5fR+3KmxaTuDHKOYSmjWezTEAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a7b160e5d2457d71ce94a0bf6f538d4416fc7b263c4eee2662fad5199f15654c","last_reissued_at":"2026-07-05T10:33:42.902640Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:33:42.902640Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Memorization and Regularization in Generative Diffusion Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["math.DS","math.OC"],"primary_cat":"cs.LG","authors_text":"Agnimitra Dasgupta, Andrew M. Stuart, Assad Oberai, Nikola B. Kovachki, Ricardo Baptista","submitted_at":"2025-01-27T05:17:06Z","abstract_excerpt":"Diffusion models have emerged as a powerful framework for generative modeling. At the heart of the methodology is score matching: learning gradients of families of log-densities for noisy versions of the data distribution at different scales. When the loss function adopted in score matching is evaluated using empirical data, rather than the population loss, the minimizer corresponds to the score of a time-dependent Gaussian mixture. However, use of this analytically tractable minimizer leads to data memorization: in both unconditioned and conditioned settings, the generative model returns the "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.15785","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.15785/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.15785","created_at":"2026-07-05T10:33:42.902754+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.15785v2","created_at":"2026-07-05T10:33:42.902754+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.15785","created_at":"2026-07-05T10:33:42.902754+00:00"},{"alias_kind":"pith_short_12","alias_value":"U6YWBZOSIV6X","created_at":"2026-07-05T10:33:42.902754+00:00"},{"alias_kind":"pith_short_16","alias_value":"U6YWBZOSIV6XDTUU","created_at":"2026-07-05T10:33:42.902754+00:00"},{"alias_kind":"pith_short_8","alias_value":"U6YWBZOS","created_at":"2026-07-05T10:33:42.902754+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09718","citing_title":"Evaluating the Representation Space of Diffusion Models via Self-Supervised Principles","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2512.16768","citing_title":"On The Hidden Biases of Flow Matching Samplers","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2603.14135","citing_title":"Conditional flow matching for physics-constrained inverse problems with finite training data","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23552","citing_title":"On the Memorization of Consistency Distillation for Diffusion Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07513","citing_title":"Tessellations of Semi-Discrete Flow Matching","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U6YWBZOSIV6XDTUUUC7W6U4NIQ","json":"https://pith.science/pith/U6YWBZOSIV6XDTUUUC7W6U4NIQ.json","graph_json":"https://pith.science/api/pith-number/U6YWBZOSIV6XDTUUUC7W6U4NIQ/graph.json","events_json":"https://pith.science/api/pith-number/U6YWBZOSIV6XDTUUUC7W6U4NIQ/events.json","paper":"https://pith.science/paper/U6YWBZOS"},"agent_actions":{"view_html":"https://pith.science/pith/U6YWBZOSIV6XDTUUUC7W6U4NIQ","download_json":"https://pith.science/pith/U6YWBZOSIV6XDTUUUC7W6U4NIQ.json","view_paper":"https://pith.science/paper/U6YWBZOS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.15785&json=true","fetch_graph":"https://pith.science/api/pith-number/U6YWBZOSIV6XDTUUUC7W6U4NIQ/graph.json","fetch_events":"https://pith.science/api/pith-number/U6YWBZOSIV6XDTUUUC7W6U4NIQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U6YWBZOSIV6XDTUUUC7W6U4NIQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U6YWBZOSIV6XDTUUUC7W6U4NIQ/action/storage_attestation","attest_author":"https://pith.science/pith/U6YWBZOSIV6XDTUUUC7W6U4NIQ/action/author_attestation","sign_citation":"https://pith.science/pith/U6YWBZOSIV6XDTUUUC7W6U4NIQ/action/citation_signature","submit_replication":"https://pith.science/pith/U6YWBZOSIV6XDTUUUC7W6U4NIQ/action/replication_record"}},"created_at":"2026-07-05T10:33:42.902754+00:00","updated_at":"2026-07-05T10:33:42.902754+00:00"}