{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EPQWU72GZLXFCBKZAVY6I5EOO7","short_pith_number":"pith:EPQWU72G","schema_version":"1.0","canonical_sha256":"23e16a7f46caee5105590571e4748e77f01ba1ffbc1cd189ae660312528be129","source":{"kind":"arxiv","id":"2403.05121","version":1},"attestation_state":"computed","paper":{"title":"CogView3: Finer and Faster Text-to-Image Generation via Relay Diffusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiayan Teng, Jidong Chen, Jie Tang, Ming Ding, Weihan Wang, Wendi Zheng, Xiaotao Gu, Yuxiao Dong, Zhuoyi Yang","submitted_at":"2024-03-08T07:32:50Z","abstract_excerpt":"Recent advancements in text-to-image generative systems have been largely driven by diffusion models. However, single-stage text-to-image diffusion models still face challenges, in terms of computational efficiency and the refinement of image details. To tackle the issue, we propose CogView3, an innovative cascaded framework that enhances the performance of text-to-image diffusion. CogView3 is the first model implementing relay diffusion in the realm of text-to-image generation, executing the task by first creating low-resolution images and subsequently applying relay-based super-resolution. T"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.05121","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-08T07:32:50Z","cross_cats_sorted":[],"title_canon_sha256":"41afc1499f4ba3d814e02aab96f4485099d543977a0c42ff2e4d95dbc87d279c","abstract_canon_sha256":"c7a5996ef1342770c4606c2cd6d0bf13bbccde1ffff5872699d547f6385d7956"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:53:38.860759Z","signature_b64":"CjXA6DlqKCwkUct0dXGYemLLvmxDLKngT4qEPKjMRMrQ+6KmzAHTp8ZfSEaU7gJlpab5kZVApaq5mJUR33TkDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"23e16a7f46caee5105590571e4748e77f01ba1ffbc1cd189ae660312528be129","last_reissued_at":"2026-07-05T07:53:38.860308Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:53:38.860308Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CogView3: Finer and Faster Text-to-Image Generation via Relay Diffusion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiayan Teng, Jidong Chen, Jie Tang, Ming Ding, Weihan Wang, Wendi Zheng, Xiaotao Gu, Yuxiao Dong, Zhuoyi Yang","submitted_at":"2024-03-08T07:32:50Z","abstract_excerpt":"Recent advancements in text-to-image generative systems have been largely driven by diffusion models. However, single-stage text-to-image diffusion models still face challenges, in terms of computational efficiency and the refinement of image details. To tackle the issue, we propose CogView3, an innovative cascaded framework that enhances the performance of text-to-image diffusion. CogView3 is the first model implementing relay diffusion in the realm of text-to-image generation, executing the task by first creating low-resolution images and subsequently applying relay-based super-resolution. T"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.05121","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.05121/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.05121","created_at":"2026-07-05T07:53:38.860373+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.05121v1","created_at":"2026-07-05T07:53:38.860373+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.05121","created_at":"2026-07-05T07:53:38.860373+00:00"},{"alias_kind":"pith_short_12","alias_value":"EPQWU72GZLXF","created_at":"2026-07-05T07:53:38.860373+00:00"},{"alias_kind":"pith_short_16","alias_value":"EPQWU72GZLXFCBKZ","created_at":"2026-07-05T07:53:38.860373+00:00"},{"alias_kind":"pith_short_8","alias_value":"EPQWU72G","created_at":"2026-07-05T07:53:38.860373+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14876","citing_title":"Unlocking Complex Visual Generation via Closed-Loop Verified Reasoning","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12967","citing_title":"ImageAttributionBench: How Far Are We from Generalizable Attribution?","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06170","citing_title":"DynT2I-Eval: A Dynamic Evaluation Framework for Text-to-Image Models","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2511.22699","citing_title":"Z-Image: An Efficient Image Generation Foundation Model with Single-Stream Diffusion Transformer","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2408.06072","citing_title":"CogVideoX: Text-to-Video Diffusion Models with An Expert Transformer","ref_index":113,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EPQWU72GZLXFCBKZAVY6I5EOO7","json":"https://pith.science/pith/EPQWU72GZLXFCBKZAVY6I5EOO7.json","graph_json":"https://pith.science/api/pith-number/EPQWU72GZLXFCBKZAVY6I5EOO7/graph.json","events_json":"https://pith.science/api/pith-number/EPQWU72GZLXFCBKZAVY6I5EOO7/events.json","paper":"https://pith.science/paper/EPQWU72G"},"agent_actions":{"view_html":"https://pith.science/pith/EPQWU72GZLXFCBKZAVY6I5EOO7","download_json":"https://pith.science/pith/EPQWU72GZLXFCBKZAVY6I5EOO7.json","view_paper":"https://pith.science/paper/EPQWU72G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.05121&json=true","fetch_graph":"https://pith.science/api/pith-number/EPQWU72GZLXFCBKZAVY6I5EOO7/graph.json","fetch_events":"https://pith.science/api/pith-number/EPQWU72GZLXFCBKZAVY6I5EOO7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EPQWU72GZLXFCBKZAVY6I5EOO7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EPQWU72GZLXFCBKZAVY6I5EOO7/action/storage_attestation","attest_author":"https://pith.science/pith/EPQWU72GZLXFCBKZAVY6I5EOO7/action/author_attestation","sign_citation":"https://pith.science/pith/EPQWU72GZLXFCBKZAVY6I5EOO7/action/citation_signature","submit_replication":"https://pith.science/pith/EPQWU72GZLXFCBKZAVY6I5EOO7/action/replication_record"}},"created_at":"2026-07-05T07:53:38.860373+00:00","updated_at":"2026-07-05T07:53:38.860373+00:00"}