{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:PJS2OKLJX2CFN3JN6WDFT7HH72","short_pith_number":"pith:PJS2OKLJ","schema_version":"1.0","canonical_sha256":"7a65a72969be8456ed2df58659fce7fe95a1f7d8424ffa8a284a25e2df66451c","source":{"kind":"arxiv","id":"2112.01573","version":1},"attestation_state":"computed","paper":{"title":"FuseDream: Training-Free Text-to-Image Generation with Improved CLIP+GAN Space Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chengyue Gong, Hao Su, Lemeng Wu, Qiang Liu, Shujian Zhang, Xingchao Liu","submitted_at":"2021-12-02T19:27:27Z","abstract_excerpt":"Generating images from natural language instructions is an intriguing yet highly challenging task. We approach text-to-image generation by combining the power of the retrained CLIP representation with an off-the-shelf image generator (GANs), optimizing in the latent space of GAN to find images that achieve maximum CLIP score with the given input text. Compared to traditional methods that train generative models from text to image starting from scratch, the CLIP+GAN approach is training-free, zero shot and can be easily customized with different generators.\n  However, optimizing CLIP score in t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.01573","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-12-02T19:27:27Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"125201f492741c5768989f8462965c5e995a95832fb544b23a5a42d7f1e2a076","abstract_canon_sha256":"c1050115f4475bc89067683de61292850d45e3594e9445359939af3440f5799b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:44:26.240393Z","signature_b64":"uKHO69Li9CGJba4pZlD1/OZvckdQkAKerySK6HMLTnHxqj4upWyolFk/F0f01x+CoYsO0NUuBgFa84qQffytAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7a65a72969be8456ed2df58659fce7fe95a1f7d8424ffa8a284a25e2df66451c","last_reissued_at":"2026-07-05T03:44:26.239956Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:44:26.239956Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FuseDream: Training-Free Text-to-Image Generation with Improved CLIP+GAN Space Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chengyue Gong, Hao Su, Lemeng Wu, Qiang Liu, Shujian Zhang, Xingchao Liu","submitted_at":"2021-12-02T19:27:27Z","abstract_excerpt":"Generating images from natural language instructions is an intriguing yet highly challenging task. We approach text-to-image generation by combining the power of the retrained CLIP representation with an off-the-shelf image generator (GANs), optimizing in the latent space of GAN to find images that achieve maximum CLIP score with the given input text. Compared to traditional methods that train generative models from text to image starting from scratch, the CLIP+GAN approach is training-free, zero shot and can be easily customized with different generators.\n  However, optimizing CLIP score in t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.01573","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.01573/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.01573","created_at":"2026-07-05T03:44:26.240015+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.01573v1","created_at":"2026-07-05T03:44:26.240015+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.01573","created_at":"2026-07-05T03:44:26.240015+00:00"},{"alias_kind":"pith_short_12","alias_value":"PJS2OKLJX2CF","created_at":"2026-07-05T03:44:26.240015+00:00"},{"alias_kind":"pith_short_16","alias_value":"PJS2OKLJX2CFN3JN","created_at":"2026-07-05T03:44:26.240015+00:00"},{"alias_kind":"pith_short_8","alias_value":"PJS2OKLJ","created_at":"2026-07-05T03:44:26.240015+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2306.09341","citing_title":"Human Preference Score v2: A Solid Benchmark for Evaluating Human Preferences of Text-to-Image Synthesis","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2209.03003","citing_title":"Flow Straight and Fast: Learning to Generate and Transfer Data with Rectified Flow","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PJS2OKLJX2CFN3JN6WDFT7HH72","json":"https://pith.science/pith/PJS2OKLJX2CFN3JN6WDFT7HH72.json","graph_json":"https://pith.science/api/pith-number/PJS2OKLJX2CFN3JN6WDFT7HH72/graph.json","events_json":"https://pith.science/api/pith-number/PJS2OKLJX2CFN3JN6WDFT7HH72/events.json","paper":"https://pith.science/paper/PJS2OKLJ"},"agent_actions":{"view_html":"https://pith.science/pith/PJS2OKLJX2CFN3JN6WDFT7HH72","download_json":"https://pith.science/pith/PJS2OKLJX2CFN3JN6WDFT7HH72.json","view_paper":"https://pith.science/paper/PJS2OKLJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.01573&json=true","fetch_graph":"https://pith.science/api/pith-number/PJS2OKLJX2CFN3JN6WDFT7HH72/graph.json","fetch_events":"https://pith.science/api/pith-number/PJS2OKLJX2CFN3JN6WDFT7HH72/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PJS2OKLJX2CFN3JN6WDFT7HH72/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PJS2OKLJX2CFN3JN6WDFT7HH72/action/storage_attestation","attest_author":"https://pith.science/pith/PJS2OKLJX2CFN3JN6WDFT7HH72/action/author_attestation","sign_citation":"https://pith.science/pith/PJS2OKLJX2CFN3JN6WDFT7HH72/action/citation_signature","submit_replication":"https://pith.science/pith/PJS2OKLJX2CFN3JN6WDFT7HH72/action/replication_record"}},"created_at":"2026-07-05T03:44:26.240015+00:00","updated_at":"2026-07-05T03:44:26.240015+00:00"}