{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:24LQLOOOWNRAXS3ZCP5TM6QHMO","short_pith_number":"pith:24LQLOOO","schema_version":"1.0","canonical_sha256":"d71705b9ceb3620bcb7913fb367a0763969c8d16ea659e21163b6809eaf967cd","source":{"kind":"arxiv","id":"2402.01103","version":3},"attestation_state":"computed","paper":{"title":"Compositional Generative Modeling: A Single Model is Not All You Need","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.RO"],"primary_cat":"cs.LG","authors_text":"Leslie Kaelbling, Yilun Du","submitted_at":"2024-02-02T02:40:51Z","abstract_excerpt":"Large monolithic generative models trained on massive amounts of data have become an increasingly dominant approach in AI research. In this paper, we argue that we should instead construct large generative systems by composing smaller generative models together. We show how such a compositional generative approach enables us to learn distributions in a more data-efficient manner, enabling generalization to parts of the data distribution unseen at training time. We further show how this enables us to program and construct new generative models for tasks completely unseen at training. Finally, w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.01103","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-02T02:40:51Z","cross_cats_sorted":["cs.AI","cs.CV","cs.RO"],"title_canon_sha256":"7e0313a2c5d7748ab318767345c43fb97c7818dc004cad30bb0a5c75d63afbdf","abstract_canon_sha256":"778db23b33d1574f3d97afcc6fdb9125bd320776ad6a91b9ffb4719cf0a0e570"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:26:43.585833Z","signature_b64":"0NsMR6iKIrUXWe0mQje8VhG5Mu1e2dbo3gDFgPB047hnpmM5BUXoSG+tDz7MrKKEJU3BAt/V0i1qh/OYm2COCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d71705b9ceb3620bcb7913fb367a0763969c8d16ea659e21163b6809eaf967cd","last_reissued_at":"2026-07-05T08:26:43.585329Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:26:43.585329Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Compositional Generative Modeling: A Single Model is Not All You Need","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.RO"],"primary_cat":"cs.LG","authors_text":"Leslie Kaelbling, Yilun Du","submitted_at":"2024-02-02T02:40:51Z","abstract_excerpt":"Large monolithic generative models trained on massive amounts of data have become an increasingly dominant approach in AI research. In this paper, we argue that we should instead construct large generative systems by composing smaller generative models together. We show how such a compositional generative approach enables us to learn distributions in a more data-efficient manner, enabling generalization to parts of the data distribution unseen at training time. We further show how this enables us to program and construct new generative models for tasks completely unseen at training. Finally, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.01103","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.01103/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.01103","created_at":"2026-07-05T08:26:43.585385+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.01103v3","created_at":"2026-07-05T08:26:43.585385+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.01103","created_at":"2026-07-05T08:26:43.585385+00:00"},{"alias_kind":"pith_short_12","alias_value":"24LQLOOOWNRA","created_at":"2026-07-05T08:26:43.585385+00:00"},{"alias_kind":"pith_short_16","alias_value":"24LQLOOOWNRAXS3Z","created_at":"2026-07-05T08:26:43.585385+00:00"},{"alias_kind":"pith_short_8","alias_value":"24LQLOOO","created_at":"2026-07-05T08:26:43.585385+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2411.02385","citing_title":"How Far is Video Generation from World Model: A Physical Law Perspective","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2509.23468","citing_title":"Multi-Modal Manipulation via Multi-Modal Policy Consensus","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2512.21898","citing_title":"Flexible Multitask Learning with Factorized Diffusion Policy","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2502.06764","citing_title":"History-Guided Video Diffusion","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03413","citing_title":"Learning to Theorize the World from Observation","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18258","citing_title":"Long-Text-to-Image Generation via Compositional Prompt Decomposition","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/24LQLOOOWNRAXS3ZCP5TM6QHMO","json":"https://pith.science/pith/24LQLOOOWNRAXS3ZCP5TM6QHMO.json","graph_json":"https://pith.science/api/pith-number/24LQLOOOWNRAXS3ZCP5TM6QHMO/graph.json","events_json":"https://pith.science/api/pith-number/24LQLOOOWNRAXS3ZCP5TM6QHMO/events.json","paper":"https://pith.science/paper/24LQLOOO"},"agent_actions":{"view_html":"https://pith.science/pith/24LQLOOOWNRAXS3ZCP5TM6QHMO","download_json":"https://pith.science/pith/24LQLOOOWNRAXS3ZCP5TM6QHMO.json","view_paper":"https://pith.science/paper/24LQLOOO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.01103&json=true","fetch_graph":"https://pith.science/api/pith-number/24LQLOOOWNRAXS3ZCP5TM6QHMO/graph.json","fetch_events":"https://pith.science/api/pith-number/24LQLOOOWNRAXS3ZCP5TM6QHMO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/24LQLOOOWNRAXS3ZCP5TM6QHMO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/24LQLOOOWNRAXS3ZCP5TM6QHMO/action/storage_attestation","attest_author":"https://pith.science/pith/24LQLOOOWNRAXS3ZCP5TM6QHMO/action/author_attestation","sign_citation":"https://pith.science/pith/24LQLOOOWNRAXS3ZCP5TM6QHMO/action/citation_signature","submit_replication":"https://pith.science/pith/24LQLOOOWNRAXS3ZCP5TM6QHMO/action/replication_record"}},"created_at":"2026-07-05T08:26:43.585385+00:00","updated_at":"2026-07-05T08:26:43.585385+00:00"}