{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:JOEZNDQ6MKBH7XF2YF2JTA6LDL","short_pith_number":"pith:JOEZNDQ6","schema_version":"1.0","canonical_sha256":"4b89968e1e62827fdcbac1749983cb1addf0d7607264c83ae6e1ada7a1fcac63","source":{"kind":"arxiv","id":"2305.10855","version":5},"attestation_state":"computed","paper":{"title":"TextDiffuser: Diffusion Models as Text Painters","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Furu Wei, Jingye Chen, Lei Cui, Qifeng Chen, Tengchao Lv, Yupan Huang","submitted_at":"2023-05-18T10:16:19Z","abstract_excerpt":"Diffusion models have gained increasing attention for their impressive generation abilities but currently struggle with rendering accurate and coherent text. To address this issue, we introduce TextDiffuser, focusing on generating images with visually appealing text that is coherent with backgrounds. TextDiffuser consists of two stages: first, a Transformer model generates the layout of keywords extracted from text prompts, and then diffusion models generate images conditioned on the text prompt and the generated layout. Additionally, we contribute the first large-scale text images dataset wit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.10855","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2023-05-18T10:16:19Z","cross_cats_sorted":[],"title_canon_sha256":"920f44c898961c991885ef675a4043c13c4b202df3ef53076fd93c9fd7145f09","abstract_canon_sha256":"9bc471dbae00148e7ec8920b094ff5876107d48953126aef511f024e3aa4d4c9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:06:27.104236Z","signature_b64":"F+RrLKY9jfLvsb9qGU2AYkxyI3KAr56DjnbpFP5cr6lA07z0b6XQw8P+Z7VKhmcqozao9iprr93MLzEsRjlPDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4b89968e1e62827fdcbac1749983cb1addf0d7607264c83ae6e1ada7a1fcac63","last_reissued_at":"2026-07-05T07:06:27.103618Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:06:27.103618Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TextDiffuser: Diffusion Models as Text Painters","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Furu Wei, Jingye Chen, Lei Cui, Qifeng Chen, Tengchao Lv, Yupan Huang","submitted_at":"2023-05-18T10:16:19Z","abstract_excerpt":"Diffusion models have gained increasing attention for their impressive generation abilities but currently struggle with rendering accurate and coherent text. To address this issue, we introduce TextDiffuser, focusing on generating images with visually appealing text that is coherent with backgrounds. TextDiffuser consists of two stages: first, a Transformer model generates the layout of keywords extracted from text prompts, and then diffusion models generate images conditioned on the text prompt and the generated layout. Additionally, we contribute the first large-scale text images dataset wit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.10855","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.10855/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.10855","created_at":"2026-07-05T07:06:27.103695+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.10855v5","created_at":"2026-07-05T07:06:27.103695+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.10855","created_at":"2026-07-05T07:06:27.103695+00:00"},{"alias_kind":"pith_short_12","alias_value":"JOEZNDQ6MKBH","created_at":"2026-07-05T07:06:27.103695+00:00"},{"alias_kind":"pith_short_16","alias_value":"JOEZNDQ6MKBH7XF2","created_at":"2026-07-05T07:06:27.103695+00:00"},{"alias_kind":"pith_short_8","alias_value":"JOEZNDQ6","created_at":"2026-07-05T07:06:27.103695+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06066","citing_title":"FontFusion: Enhancing Generative Text in Diffusion Models with Typographic Conditioning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04479","citing_title":"Evaluating Reasoning Fidelity in Visual Text Generation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2312.16476","citing_title":"SVGDreamer: Text Guided SVG Generation with Diffusion Model","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20807","citing_title":"Decomposing Subject-Driven Image Generation via Intermediate Structural Prediction","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16810","citing_title":"Training-Free Occluded Text Rendering via Glyph Priors and Attention-Guided Semantic Blending","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14708","citing_title":"StyleTextGen: Style-Conditioned Multilingual Scene Text Generation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10954","citing_title":"FineEdit: Fine-Grained Image Edit with Bounding Box Guidance","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JOEZNDQ6MKBH7XF2YF2JTA6LDL","json":"https://pith.science/pith/JOEZNDQ6MKBH7XF2YF2JTA6LDL.json","graph_json":"https://pith.science/api/pith-number/JOEZNDQ6MKBH7XF2YF2JTA6LDL/graph.json","events_json":"https://pith.science/api/pith-number/JOEZNDQ6MKBH7XF2YF2JTA6LDL/events.json","paper":"https://pith.science/paper/JOEZNDQ6"},"agent_actions":{"view_html":"https://pith.science/pith/JOEZNDQ6MKBH7XF2YF2JTA6LDL","download_json":"https://pith.science/pith/JOEZNDQ6MKBH7XF2YF2JTA6LDL.json","view_paper":"https://pith.science/paper/JOEZNDQ6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.10855&json=true","fetch_graph":"https://pith.science/api/pith-number/JOEZNDQ6MKBH7XF2YF2JTA6LDL/graph.json","fetch_events":"https://pith.science/api/pith-number/JOEZNDQ6MKBH7XF2YF2JTA6LDL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JOEZNDQ6MKBH7XF2YF2JTA6LDL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JOEZNDQ6MKBH7XF2YF2JTA6LDL/action/storage_attestation","attest_author":"https://pith.science/pith/JOEZNDQ6MKBH7XF2YF2JTA6LDL/action/author_attestation","sign_citation":"https://pith.science/pith/JOEZNDQ6MKBH7XF2YF2JTA6LDL/action/citation_signature","submit_replication":"https://pith.science/pith/JOEZNDQ6MKBH7XF2YF2JTA6LDL/action/replication_record"}},"created_at":"2026-07-05T07:06:27.103695+00:00","updated_at":"2026-07-05T07:06:27.103695+00:00"}