{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IVMEPYL4ELYTLYNXESQHAGHENC","short_pith_number":"pith:IVMEPYL4","schema_version":"1.0","canonical_sha256":"455847e17c22f135e1b724a07018e468b97a9b92ba9e05e0e9b22a3f69f7883e","source":{"kind":"arxiv","id":"2311.16465","version":1},"attestation_state":"computed","paper":{"title":"TextDiffuser-2: Unleashing the Power of Language Models for Text Rendering","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Furu Wei, Jingye Chen, Lei Cui, Qifeng Chen, Tengchao Lv, Yupan Huang","submitted_at":"2023-11-28T04:02:40Z","abstract_excerpt":"The diffusion model has been proven a powerful generative model in recent years, yet remains a challenge in generating visual text. Several methods alleviated this issue by incorporating explicit text position and content as guidance on where and what text to render. However, these methods still suffer from several drawbacks, such as limited flexibility and automation, constrained capability of layout prediction, and restricted style diversity. In this paper, we present TextDiffuser-2, aiming to unleash the power of language models for text rendering. Firstly, we fine-tune a large language mod"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.16465","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2023-11-28T04:02:40Z","cross_cats_sorted":[],"title_canon_sha256":"53795a9bfcf99e1d0b60959bac6f16c48f66b70a8f65c5cee001e2eb35b0721a","abstract_canon_sha256":"a80f9cd325e7bef97145bfa0cd848cfa7fed8e3ee6ea9e7f47f905c2ba892219"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:17:42.193415Z","signature_b64":"Er+K15oXjUn7oztJ7Vh8nKtBONu/gbTa0REc4SRRrLrVEJWhHq2Q+v5KbIG5DkjHoMwGcYxIJosPDu5X9VZhAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"455847e17c22f135e1b724a07018e468b97a9b92ba9e05e0e9b22a3f69f7883e","last_reissued_at":"2026-07-05T07:17:42.192885Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:17:42.192885Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TextDiffuser-2: Unleashing the Power of Language Models for Text Rendering","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Furu Wei, Jingye Chen, Lei Cui, Qifeng Chen, Tengchao Lv, Yupan Huang","submitted_at":"2023-11-28T04:02:40Z","abstract_excerpt":"The diffusion model has been proven a powerful generative model in recent years, yet remains a challenge in generating visual text. Several methods alleviated this issue by incorporating explicit text position and content as guidance on where and what text to render. However, these methods still suffer from several drawbacks, such as limited flexibility and automation, constrained capability of layout prediction, and restricted style diversity. In this paper, we present TextDiffuser-2, aiming to unleash the power of language models for text rendering. Firstly, we fine-tune a large language mod"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.16465","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.16465/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.16465","created_at":"2026-07-05T07:17:42.192940+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.16465v1","created_at":"2026-07-05T07:17:42.192940+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.16465","created_at":"2026-07-05T07:17:42.192940+00:00"},{"alias_kind":"pith_short_12","alias_value":"IVMEPYL4ELYT","created_at":"2026-07-05T07:17:42.192940+00:00"},{"alias_kind":"pith_short_16","alias_value":"IVMEPYL4ELYTLYNX","created_at":"2026-07-05T07:17:42.192940+00:00"},{"alias_kind":"pith_short_8","alias_value":"IVMEPYL4","created_at":"2026-07-05T07:17:42.192940+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04479","citing_title":"Evaluating Reasoning Fidelity in Visual Text Generation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16810","citing_title":"Training-Free Occluded Text Rendering via Glyph Priors and Attention-Guided Semantic Blending","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14708","citing_title":"StyleTextGen: Style-Conditioned Multilingual Scene Text Generation","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IVMEPYL4ELYTLYNXESQHAGHENC","json":"https://pith.science/pith/IVMEPYL4ELYTLYNXESQHAGHENC.json","graph_json":"https://pith.science/api/pith-number/IVMEPYL4ELYTLYNXESQHAGHENC/graph.json","events_json":"https://pith.science/api/pith-number/IVMEPYL4ELYTLYNXESQHAGHENC/events.json","paper":"https://pith.science/paper/IVMEPYL4"},"agent_actions":{"view_html":"https://pith.science/pith/IVMEPYL4ELYTLYNXESQHAGHENC","download_json":"https://pith.science/pith/IVMEPYL4ELYTLYNXESQHAGHENC.json","view_paper":"https://pith.science/paper/IVMEPYL4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.16465&json=true","fetch_graph":"https://pith.science/api/pith-number/IVMEPYL4ELYTLYNXESQHAGHENC/graph.json","fetch_events":"https://pith.science/api/pith-number/IVMEPYL4ELYTLYNXESQHAGHENC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IVMEPYL4ELYTLYNXESQHAGHENC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IVMEPYL4ELYTLYNXESQHAGHENC/action/storage_attestation","attest_author":"https://pith.science/pith/IVMEPYL4ELYTLYNXESQHAGHENC/action/author_attestation","sign_citation":"https://pith.science/pith/IVMEPYL4ELYTLYNXESQHAGHENC/action/citation_signature","submit_replication":"https://pith.science/pith/IVMEPYL4ELYTLYNXESQHAGHENC/action/replication_record"}},"created_at":"2026-07-05T07:17:42.192940+00:00","updated_at":"2026-07-05T07:17:42.192940+00:00"}