{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EVHWQXUFKYPGRRODAYICUJ33JR","short_pith_number":"pith:EVHWQXUF","schema_version":"1.0","canonical_sha256":"254f685e85561e68c5c306102a277b4c67bd59904866fe34f4b7173da58e5127","source":{"kind":"arxiv","id":"2406.10208","version":2},"attestation_state":"computed","paper":{"title":"Glyph-ByT5-v2: A Strong Aesthetic Baseline for Accurate Multilingual Visual Text Rendering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bohan Chen, Ji Li, Lijuan Wang, Lin Liang, Weicong Liang, Yiming Zhao, Yuhui Yuan, Zeyu Liu","submitted_at":"2024-06-14T17:44:09Z","abstract_excerpt":"Recently, Glyph-ByT5 has achieved highly accurate visual text rendering performance in graphic design images. However, it still focuses solely on English and performs relatively poorly in terms of visual appeal. In this work, we address these two fundamental limitations by presenting Glyph-ByT5-v2 and Glyph-SDXL-v2, which not only support accurate visual text rendering for 10 different languages but also achieve much better aesthetic quality. To achieve this, we make the following contributions: (i) creating a high-quality multilingual glyph-text and graphic design dataset consisting of more t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.10208","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-14T17:44:09Z","cross_cats_sorted":[],"title_canon_sha256":"da8edbe5d6eb911bc65a6ac25ee146af5d8f7278420d50fb24401eb989ba61d4","abstract_canon_sha256":"1b65591f6fe803419111f190971d590eb0baa1323dab8e4e044da3c02fe51332"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:43:11.025903Z","signature_b64":"Dq4j1PuwnbSM4xmwdk7hqP6AvWqcb6bc+NZu4QZS3APtdbPKnIoeEJxP5vk0y66dvmi+36fp5hIQav6ZiHlNCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"254f685e85561e68c5c306102a277b4c67bd59904866fe34f4b7173da58e5127","last_reissued_at":"2026-07-05T08:43:11.025472Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:43:11.025472Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Glyph-ByT5-v2: A Strong Aesthetic Baseline for Accurate Multilingual Visual Text Rendering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bohan Chen, Ji Li, Lijuan Wang, Lin Liang, Weicong Liang, Yiming Zhao, Yuhui Yuan, Zeyu Liu","submitted_at":"2024-06-14T17:44:09Z","abstract_excerpt":"Recently, Glyph-ByT5 has achieved highly accurate visual text rendering performance in graphic design images. However, it still focuses solely on English and performs relatively poorly in terms of visual appeal. In this work, we address these two fundamental limitations by presenting Glyph-ByT5-v2 and Glyph-SDXL-v2, which not only support accurate visual text rendering for 10 different languages but also achieve much better aesthetic quality. To achieve this, we make the following contributions: (i) creating a high-quality multilingual glyph-text and graphic design dataset consisting of more t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.10208","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.10208/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.10208","created_at":"2026-07-05T08:43:11.025528+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.10208v2","created_at":"2026-07-05T08:43:11.025528+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.10208","created_at":"2026-07-05T08:43:11.025528+00:00"},{"alias_kind":"pith_short_12","alias_value":"EVHWQXUFKYPG","created_at":"2026-07-05T08:43:11.025528+00:00"},{"alias_kind":"pith_short_16","alias_value":"EVHWQXUFKYPGRROD","created_at":"2026-07-05T08:43:11.025528+00:00"},{"alias_kind":"pith_short_8","alias_value":"EVHWQXUF","created_at":"2026-07-05T08:43:11.025528+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04479","citing_title":"Evaluating Reasoning Fidelity in Visual Text Generation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2503.07703","citing_title":"Seedream 2.0: A Native Chinese-English Bilingual Image Generation Foundation Model","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08163","citing_title":"MULTITEXTEDIT: Benchmarking Cross-Lingual Degradation in Text-in-Image Editing","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24171","citing_title":"POCA: Pareto-Optimal Curriculum Alignment for Visual Text Generation","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EVHWQXUFKYPGRRODAYICUJ33JR","json":"https://pith.science/pith/EVHWQXUFKYPGRRODAYICUJ33JR.json","graph_json":"https://pith.science/api/pith-number/EVHWQXUFKYPGRRODAYICUJ33JR/graph.json","events_json":"https://pith.science/api/pith-number/EVHWQXUFKYPGRRODAYICUJ33JR/events.json","paper":"https://pith.science/paper/EVHWQXUF"},"agent_actions":{"view_html":"https://pith.science/pith/EVHWQXUFKYPGRRODAYICUJ33JR","download_json":"https://pith.science/pith/EVHWQXUFKYPGRRODAYICUJ33JR.json","view_paper":"https://pith.science/paper/EVHWQXUF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.10208&json=true","fetch_graph":"https://pith.science/api/pith-number/EVHWQXUFKYPGRRODAYICUJ33JR/graph.json","fetch_events":"https://pith.science/api/pith-number/EVHWQXUFKYPGRRODAYICUJ33JR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EVHWQXUFKYPGRRODAYICUJ33JR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EVHWQXUFKYPGRRODAYICUJ33JR/action/storage_attestation","attest_author":"https://pith.science/pith/EVHWQXUFKYPGRRODAYICUJ33JR/action/author_attestation","sign_citation":"https://pith.science/pith/EVHWQXUFKYPGRRODAYICUJ33JR/action/citation_signature","submit_replication":"https://pith.science/pith/EVHWQXUFKYPGRRODAYICUJ33JR/action/replication_record"}},"created_at":"2026-07-05T08:43:11.025528+00:00","updated_at":"2026-07-05T08:43:11.025528+00:00"}