{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6MBWA6JUMODVOUDVELX27RTWJE","short_pith_number":"pith:6MBWA6JU","schema_version":"1.0","canonical_sha256":"f303607934638757507522efafc67649201ac0a956cf878238adf658f57ecffc","source":{"kind":"arxiv","id":"2409.19967","version":1},"attestation_state":"computed","paper":{"title":"Magnet: We Never Know How Text-to-Image Diffusion Models Work, Until We Learn How Vision-Language Models Function","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenyi Zhuang, Pan Gao, Ying Hu","submitted_at":"2024-09-30T05:36:24Z","abstract_excerpt":"Text-to-image diffusion models particularly Stable Diffusion, have revolutionized the field of computer vision. However, the synthesis quality often deteriorates when asked to generate images that faithfully represent complex prompts involving multiple attributes and objects. While previous studies suggest that blended text embeddings lead to improper attribute binding, few have explored this in depth. In this work, we critically examine the limitations of the CLIP text encoder in understanding attributes and investigate how this affects diffusion models. We discern a phenomenon of attribute b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.19967","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-09-30T05:36:24Z","cross_cats_sorted":[],"title_canon_sha256":"701f961f7d4e8a7d3182d4ea3d03bf4402adf923805bb7e20466f480feb1e2eb","abstract_canon_sha256":"713607bf3d460208d9b2c80bec9451467da2f55feaaa74eec3d27ff1fc8ad1ea"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:13:25.459863Z","signature_b64":"jYvCa7NFsiZeIrYOTV5TriWRZfuiAkadLT8jZ25I0G+/wO/ijXVle6D/RFJ7fcemHLO0pata06XQtjFRTM52Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f303607934638757507522efafc67649201ac0a956cf878238adf658f57ecffc","last_reissued_at":"2026-07-05T09:13:25.459477Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:13:25.459477Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Magnet: We Never Know How Text-to-Image Diffusion Models Work, Until We Learn How Vision-Language Models Function","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenyi Zhuang, Pan Gao, Ying Hu","submitted_at":"2024-09-30T05:36:24Z","abstract_excerpt":"Text-to-image diffusion models particularly Stable Diffusion, have revolutionized the field of computer vision. However, the synthesis quality often deteriorates when asked to generate images that faithfully represent complex prompts involving multiple attributes and objects. While previous studies suggest that blended text embeddings lead to improper attribute binding, few have explored this in depth. In this work, we critically examine the limitations of the CLIP text encoder in understanding attributes and investigate how this affects diffusion models. We discern a phenomenon of attribute b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.19967","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.19967/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.19967","created_at":"2026-07-05T09:13:25.459531+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.19967v1","created_at":"2026-07-05T09:13:25.459531+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.19967","created_at":"2026-07-05T09:13:25.459531+00:00"},{"alias_kind":"pith_short_12","alias_value":"6MBWA6JUMODV","created_at":"2026-07-05T09:13:25.459531+00:00"},{"alias_kind":"pith_short_16","alias_value":"6MBWA6JUMODVOUDV","created_at":"2026-07-05T09:13:25.459531+00:00"},{"alias_kind":"pith_short_8","alias_value":"6MBWA6JU","created_at":"2026-07-05T09:13:25.459531+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.20206","citing_title":"RAPO++: Cross-Stage Prompt Optimization for Text-to-Video Generation via Data Alignment and Test-Time Scaling","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6MBWA6JUMODVOUDVELX27RTWJE","json":"https://pith.science/pith/6MBWA6JUMODVOUDVELX27RTWJE.json","graph_json":"https://pith.science/api/pith-number/6MBWA6JUMODVOUDVELX27RTWJE/graph.json","events_json":"https://pith.science/api/pith-number/6MBWA6JUMODVOUDVELX27RTWJE/events.json","paper":"https://pith.science/paper/6MBWA6JU"},"agent_actions":{"view_html":"https://pith.science/pith/6MBWA6JUMODVOUDVELX27RTWJE","download_json":"https://pith.science/pith/6MBWA6JUMODVOUDVELX27RTWJE.json","view_paper":"https://pith.science/paper/6MBWA6JU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.19967&json=true","fetch_graph":"https://pith.science/api/pith-number/6MBWA6JUMODVOUDVELX27RTWJE/graph.json","fetch_events":"https://pith.science/api/pith-number/6MBWA6JUMODVOUDVELX27RTWJE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6MBWA6JUMODVOUDVELX27RTWJE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6MBWA6JUMODVOUDVELX27RTWJE/action/storage_attestation","attest_author":"https://pith.science/pith/6MBWA6JUMODVOUDVELX27RTWJE/action/author_attestation","sign_citation":"https://pith.science/pith/6MBWA6JUMODVOUDVELX27RTWJE/action/citation_signature","submit_replication":"https://pith.science/pith/6MBWA6JUMODVOUDVELX27RTWJE/action/replication_record"}},"created_at":"2026-07-05T09:13:25.459531+00:00","updated_at":"2026-07-05T09:13:25.459531+00:00"}