{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DA56ANVEH4MBDZQM6SU5XHFDJA","short_pith_number":"pith:DA56ANVE","schema_version":"1.0","canonical_sha256":"183be036a43f1811e60cf4a9db9ca34821d876408ea0c62aec24a1f1f9162d22","source":{"kind":"arxiv","id":"2502.12089","version":3},"attestation_state":"computed","paper":{"title":"How Compositional Generalization and Creativity Improve as Diffusion Models are Trained","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Alessandro Favero, Antonio Sclocchi, Francesco Cagnetta, Matthieu Wyart, Pascal Frossard","submitted_at":"2025-02-17T18:06:33Z","abstract_excerpt":"Natural data is often organized as a hierarchical composition of features. How many samples do generative models need in order to learn the composition rules, so as to produce a combinatorially large number of novel data? What signal in the data is exploited to learn those rules? We investigate these questions in the context of diffusion models both theoretically and empirically. Theoretically, we consider a simple probabilistic context-free grammar - a tree-like graphical model used to represent the hierarchical and compositional structure of data such as language and images. We demonstrate t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.12089","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"stat.ML","submitted_at":"2025-02-17T18:06:33Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"d2fdd972bd2de0e75c4f7c2e5d4a1f5a28d9b261eb11883e9eb239817aa6177a","abstract_canon_sha256":"cf80eb25237c626de625cbfa800f4b4105884f0aa305cb9a7626d99462643a54"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:42.865366Z","signature_b64":"RM7YjNwTTjlhaVK20FmhXcXApJoaR23nxcb7t6XMEh23hX+gVB/veJpA1dC3lUP0O2yF2jvoRPwOxaxbfLR/Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"183be036a43f1811e60cf4a9db9ca34821d876408ea0c62aec24a1f1f9162d22","last_reissued_at":"2026-07-05T11:15:42.864925Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:42.864925Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Compositional Generalization and Creativity Improve as Diffusion Models are Trained","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Alessandro Favero, Antonio Sclocchi, Francesco Cagnetta, Matthieu Wyart, Pascal Frossard","submitted_at":"2025-02-17T18:06:33Z","abstract_excerpt":"Natural data is often organized as a hierarchical composition of features. How many samples do generative models need in order to learn the composition rules, so as to produce a combinatorially large number of novel data? What signal in the data is exploited to learn those rules? We investigate these questions in the context of diffusion models both theoretically and empirically. Theoretically, we consider a simple probabilistic context-free grammar - a tree-like graphical model used to represent the hierarchical and compositional structure of data such as language and images. We demonstrate t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.12089","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.12089/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.12089","created_at":"2026-07-05T11:15:42.864978+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.12089v3","created_at":"2026-07-05T11:15:42.864978+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.12089","created_at":"2026-07-05T11:15:42.864978+00:00"},{"alias_kind":"pith_short_12","alias_value":"DA56ANVEH4MB","created_at":"2026-07-05T11:15:42.864978+00:00"},{"alias_kind":"pith_short_16","alias_value":"DA56ANVEH4MBDZQM","created_at":"2026-07-05T11:15:42.864978+00:00"},{"alias_kind":"pith_short_8","alias_value":"DA56ANVE","created_at":"2026-07-05T11:15:42.864978+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09601","citing_title":"Assessing Sample Quality in Conditional Generation under Compositional Shift","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2603.12901","citing_title":"A theory of learning data statistics in diffusion models, from easy to hard","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13612","citing_title":"Deep Learning as Neural Low-Degree Filtering: A Spectral Theory of Hierarchical Feature Learning","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02415","citing_title":"Generative models on phase space","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07844","citing_title":"Distributional simplicity bias and effective convexity in Energy Based Models","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DA56ANVEH4MBDZQM6SU5XHFDJA","json":"https://pith.science/pith/DA56ANVEH4MBDZQM6SU5XHFDJA.json","graph_json":"https://pith.science/api/pith-number/DA56ANVEH4MBDZQM6SU5XHFDJA/graph.json","events_json":"https://pith.science/api/pith-number/DA56ANVEH4MBDZQM6SU5XHFDJA/events.json","paper":"https://pith.science/paper/DA56ANVE"},"agent_actions":{"view_html":"https://pith.science/pith/DA56ANVEH4MBDZQM6SU5XHFDJA","download_json":"https://pith.science/pith/DA56ANVEH4MBDZQM6SU5XHFDJA.json","view_paper":"https://pith.science/paper/DA56ANVE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.12089&json=true","fetch_graph":"https://pith.science/api/pith-number/DA56ANVEH4MBDZQM6SU5XHFDJA/graph.json","fetch_events":"https://pith.science/api/pith-number/DA56ANVEH4MBDZQM6SU5XHFDJA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DA56ANVEH4MBDZQM6SU5XHFDJA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DA56ANVEH4MBDZQM6SU5XHFDJA/action/storage_attestation","attest_author":"https://pith.science/pith/DA56ANVEH4MBDZQM6SU5XHFDJA/action/author_attestation","sign_citation":"https://pith.science/pith/DA56ANVEH4MBDZQM6SU5XHFDJA/action/citation_signature","submit_replication":"https://pith.science/pith/DA56ANVEH4MBDZQM6SU5XHFDJA/action/replication_record"}},"created_at":"2026-07-05T11:15:42.864978+00:00","updated_at":"2026-07-05T11:15:42.864978+00:00"}