{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DLNUXQRG2UU2GQLICQL7YJGPYO","short_pith_number":"pith:DLNUXQRG","schema_version":"1.0","canonical_sha256":"1adb4bc226d529a341681417fc24cfc38d3b7112c1dc1f41cae1621772a1ef1f","source":{"kind":"arxiv","id":"2405.12914","version":2},"attestation_state":"computed","paper":{"title":"An Empirical Study and Analysis of Text-to-Image Generation Using Large Language Model-Powered Textual Representation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cheng Zhang, Hao Li, Hao Yang, Luozheng Qin, Mengping Yang, Qiang Zhou, Ye Qian, Zhiyu Tan","submitted_at":"2024-05-21T16:35:02Z","abstract_excerpt":"One critical prerequisite for faithful text-to-image generation is the accurate understanding of text inputs. Existing methods leverage the text encoder of the CLIP model to represent input prompts. However, the pre-trained CLIP model can merely encode English with a maximum token length of 77. Moreover, the model capacity of the text encoder from CLIP is relatively limited compared to Large Language Models (LLMs), which offer multilingual input, accommodate longer context, and achieve superior text representation. In this paper, we investigate LLMs as the text encoder to improve the language "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.12914","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-21T16:35:02Z","cross_cats_sorted":[],"title_canon_sha256":"b4e3c74ff175fd35bf00dab13bf6b7053ad0bb18ffe7fb5afff53fe4ba9da107","abstract_canon_sha256":"be877aeb1b1f92f9b7e87e305dfc925b12190e34fbd92f7b055813cb63ba23ef"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:45:28.517581Z","signature_b64":"REx9hEiuIdTOgfMI6a8Nx+CMNj44o1mH1mKXzK1sY13Pq/adZhPvRkmghy/4SPRaguPyXixt5OoHrcVXYBHqBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1adb4bc226d529a341681417fc24cfc38d3b7112c1dc1f41cae1621772a1ef1f","last_reissued_at":"2026-07-05T08:45:28.517108Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:45:28.517108Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Empirical Study and Analysis of Text-to-Image Generation Using Large Language Model-Powered Textual Representation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cheng Zhang, Hao Li, Hao Yang, Luozheng Qin, Mengping Yang, Qiang Zhou, Ye Qian, Zhiyu Tan","submitted_at":"2024-05-21T16:35:02Z","abstract_excerpt":"One critical prerequisite for faithful text-to-image generation is the accurate understanding of text inputs. Existing methods leverage the text encoder of the CLIP model to represent input prompts. However, the pre-trained CLIP model can merely encode English with a maximum token length of 77. Moreover, the model capacity of the text encoder from CLIP is relatively limited compared to Large Language Models (LLMs), which offer multilingual input, accommodate longer context, and achieve superior text representation. In this paper, we investigate LLMs as the text encoder to improve the language "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.12914","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.12914/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.12914","created_at":"2026-07-05T08:45:28.517172+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.12914v2","created_at":"2026-07-05T08:45:28.517172+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.12914","created_at":"2026-07-05T08:45:28.517172+00:00"},{"alias_kind":"pith_short_12","alias_value":"DLNUXQRG2UU2","created_at":"2026-07-05T08:45:28.517172+00:00"},{"alias_kind":"pith_short_16","alias_value":"DLNUXQRG2UU2GQLI","created_at":"2026-07-05T08:45:28.517172+00:00"},{"alias_kind":"pith_short_8","alias_value":"DLNUXQRG","created_at":"2026-07-05T08:45:28.517172+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.15269","citing_title":"Mirage in the Eyes: Hallucination Attack on Multi-modal Large Language Models with Only Attention Sink","ref_index":78,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DLNUXQRG2UU2GQLICQL7YJGPYO","json":"https://pith.science/pith/DLNUXQRG2UU2GQLICQL7YJGPYO.json","graph_json":"https://pith.science/api/pith-number/DLNUXQRG2UU2GQLICQL7YJGPYO/graph.json","events_json":"https://pith.science/api/pith-number/DLNUXQRG2UU2GQLICQL7YJGPYO/events.json","paper":"https://pith.science/paper/DLNUXQRG"},"agent_actions":{"view_html":"https://pith.science/pith/DLNUXQRG2UU2GQLICQL7YJGPYO","download_json":"https://pith.science/pith/DLNUXQRG2UU2GQLICQL7YJGPYO.json","view_paper":"https://pith.science/paper/DLNUXQRG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.12914&json=true","fetch_graph":"https://pith.science/api/pith-number/DLNUXQRG2UU2GQLICQL7YJGPYO/graph.json","fetch_events":"https://pith.science/api/pith-number/DLNUXQRG2UU2GQLICQL7YJGPYO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DLNUXQRG2UU2GQLICQL7YJGPYO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DLNUXQRG2UU2GQLICQL7YJGPYO/action/storage_attestation","attest_author":"https://pith.science/pith/DLNUXQRG2UU2GQLICQL7YJGPYO/action/author_attestation","sign_citation":"https://pith.science/pith/DLNUXQRG2UU2GQLICQL7YJGPYO/action/citation_signature","submit_replication":"https://pith.science/pith/DLNUXQRG2UU2GQLICQL7YJGPYO/action/replication_record"}},"created_at":"2026-07-05T08:45:28.517172+00:00","updated_at":"2026-07-05T08:45:28.517172+00:00"}