{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FNYSPMWWIJ7KIQH4ITZ2TUEBFJ","short_pith_number":"pith:FNYSPMWW","schema_version":"1.0","canonical_sha256":"2b7127b2d6427ea440fc44f3a9d0812a63ce614643bea7587fc9fc0466d99468","source":{"kind":"arxiv","id":"2303.17189","version":2},"attestation_state":"computed","paper":{"title":"LayoutDiffusion: Controllable Diffusion Model for Layout-to-image Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Guangcong Zheng, Xianpan Zhou, Xi Li, Xuewei Li, Ying Shan, Zhongang Qi","submitted_at":"2023-03-30T06:56:12Z","abstract_excerpt":"Recently, diffusion models have achieved great success in image synthesis. However, when it comes to the layout-to-image generation where an image often has a complex scene of multiple objects, how to make strong control over both the global layout map and each detailed object remains a challenging task. In this paper, we propose a diffusion model named LayoutDiffusion that can obtain higher generation quality and greater controllability than the previous works. To overcome the difficult multimodal fusion of image and layout, we propose to construct a structural image patch with region informa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.17189","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2023-03-30T06:56:12Z","cross_cats_sorted":[],"title_canon_sha256":"81fff93d5b263f437342d665b379fa6a3cc42d645b8ea5f240877fb223730482","abstract_canon_sha256":"4f1aae1494deba98a0485a54081eaffbf61a3642a4972ec309ecd7de83a21e9a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:54:48.452657Z","signature_b64":"hJ/QY+qIXnap6Xt1Rsnn5QKotZlug2VJaSYlkGiA1jFnvtUGHdWqcsPiglEBp5OZKLJQ21zyW96VzAZe/WA+DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2b7127b2d6427ea440fc44f3a9d0812a63ce614643bea7587fc9fc0466d99468","last_reissued_at":"2026-07-05T07:54:48.452239Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:54:48.452239Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LayoutDiffusion: Controllable Diffusion Model for Layout-to-image Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Guangcong Zheng, Xianpan Zhou, Xi Li, Xuewei Li, Ying Shan, Zhongang Qi","submitted_at":"2023-03-30T06:56:12Z","abstract_excerpt":"Recently, diffusion models have achieved great success in image synthesis. However, when it comes to the layout-to-image generation where an image often has a complex scene of multiple objects, how to make strong control over both the global layout map and each detailed object remains a challenging task. In this paper, we propose a diffusion model named LayoutDiffusion that can obtain higher generation quality and greater controllability than the previous works. To overcome the difficult multimodal fusion of image and layout, we propose to construct a structural image patch with region informa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.17189","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.17189/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.17189","created_at":"2026-07-05T07:54:48.452291+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.17189v2","created_at":"2026-07-05T07:54:48.452291+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.17189","created_at":"2026-07-05T07:54:48.452291+00:00"},{"alias_kind":"pith_short_12","alias_value":"FNYSPMWWIJ7K","created_at":"2026-07-05T07:54:48.452291+00:00"},{"alias_kind":"pith_short_16","alias_value":"FNYSPMWWIJ7KIQH4","created_at":"2026-07-05T07:54:48.452291+00:00"},{"alias_kind":"pith_short_8","alias_value":"FNYSPMWW","created_at":"2026-07-05T07:54:48.452291+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.23418","citing_title":"Why Settle for Mid: A Probabilistic Viewpoint to Spatial Relationship Alignment in Text-to-image Models","ref_index":83,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FNYSPMWWIJ7KIQH4ITZ2TUEBFJ","json":"https://pith.science/pith/FNYSPMWWIJ7KIQH4ITZ2TUEBFJ.json","graph_json":"https://pith.science/api/pith-number/FNYSPMWWIJ7KIQH4ITZ2TUEBFJ/graph.json","events_json":"https://pith.science/api/pith-number/FNYSPMWWIJ7KIQH4ITZ2TUEBFJ/events.json","paper":"https://pith.science/paper/FNYSPMWW"},"agent_actions":{"view_html":"https://pith.science/pith/FNYSPMWWIJ7KIQH4ITZ2TUEBFJ","download_json":"https://pith.science/pith/FNYSPMWWIJ7KIQH4ITZ2TUEBFJ.json","view_paper":"https://pith.science/paper/FNYSPMWW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.17189&json=true","fetch_graph":"https://pith.science/api/pith-number/FNYSPMWWIJ7KIQH4ITZ2TUEBFJ/graph.json","fetch_events":"https://pith.science/api/pith-number/FNYSPMWWIJ7KIQH4ITZ2TUEBFJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FNYSPMWWIJ7KIQH4ITZ2TUEBFJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FNYSPMWWIJ7KIQH4ITZ2TUEBFJ/action/storage_attestation","attest_author":"https://pith.science/pith/FNYSPMWWIJ7KIQH4ITZ2TUEBFJ/action/author_attestation","sign_citation":"https://pith.science/pith/FNYSPMWWIJ7KIQH4ITZ2TUEBFJ/action/citation_signature","submit_replication":"https://pith.science/pith/FNYSPMWWIJ7KIQH4ITZ2TUEBFJ/action/replication_record"}},"created_at":"2026-07-05T07:54:48.452291+00:00","updated_at":"2026-07-05T07:54:48.452291+00:00"}