{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:55VZJ4H24CPF6NUGTROTH55WVK","short_pith_number":"pith:55VZJ4H2","schema_version":"1.0","canonical_sha256":"ef6b94f0fae09e5f36869c5d33f7b6aaaf9da2aed08ec9c81a84aedadc14da9e","source":{"kind":"arxiv","id":"2601.02211","version":2},"attestation_state":"computed","paper":{"title":"TexTailor: Inference-Time Textual Guidance Tailoring for Multimodal Diffusion Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Binglei Li, Hao Li, Junping Zhang, Mengping Yang, Zhiyu Tan","submitted_at":"2026-01-05T15:32:53Z","abstract_excerpt":"Recent breakthroughs of transformer-based diffusion models, particularly with Multimodal Diffusion Transformers (MMDiT) driven models like FLUX and Qwen Image, have facilitated thrilling experiences in visual generation. However, these models rely only on the interactions between textual conditions and visual features to produce semantically aligned images. Once the interactions fail to reflect the nuanced compositional structure of the prompt, the generated images might be unsatisfactory. Thus, a comprehensive understanding of how different blocks and their interactions with textual condition"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2601.02211","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2026-01-05T15:32:53Z","cross_cats_sorted":[],"title_canon_sha256":"95594ee8fc7456729ae6878f8f310f1622b15aa2da08b2ce16f965fb70619111","abstract_canon_sha256":"d554e6651f2931be1eba413649f19c532bef99146b5f2192b5d1c539fc48c5f6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-07T02:18:32.628612Z","signature_b64":"DBJAeqM5xsfzJkxPl1wkmy2F/iERv3eXDzJUiEnw9FxDPVo3eunwElkrsvEg/1FsYQ1stwYuAO9cuY9RzxQ4BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ef6b94f0fae09e5f36869c5d33f7b6aaaf9da2aed08ec9c81a84aedadc14da9e","last_reissued_at":"2026-07-07T02:18:32.627625Z","signature_status":"signed_v1","first_computed_at":"2026-07-07T02:18:32.627625Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TexTailor: Inference-Time Textual Guidance Tailoring for Multimodal Diffusion Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Binglei Li, Hao Li, Junping Zhang, Mengping Yang, Zhiyu Tan","submitted_at":"2026-01-05T15:32:53Z","abstract_excerpt":"Recent breakthroughs of transformer-based diffusion models, particularly with Multimodal Diffusion Transformers (MMDiT) driven models like FLUX and Qwen Image, have facilitated thrilling experiences in visual generation. However, these models rely only on the interactions between textual conditions and visual features to produce semantically aligned images. Once the interactions fail to reflect the nuanced compositional structure of the prompt, the generated images might be unsatisfactory. Thus, a comprehensive understanding of how different blocks and their interactions with textual condition"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2601.02211","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2601.02211/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2601.02211","created_at":"2026-07-07T02:18:32.627752+00:00"},{"alias_kind":"arxiv_version","alias_value":"2601.02211v2","created_at":"2026-07-07T02:18:32.627752+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2601.02211","created_at":"2026-07-07T02:18:32.627752+00:00"},{"alias_kind":"pith_short_12","alias_value":"55VZJ4H24CPF","created_at":"2026-07-07T02:18:32.627752+00:00"},{"alias_kind":"pith_short_16","alias_value":"55VZJ4H24CPF6NUG","created_at":"2026-07-07T02:18:32.627752+00:00"},{"alias_kind":"pith_short_8","alias_value":"55VZJ4H2","created_at":"2026-07-07T02:18:32.627752+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2606.06813","citing_title":"Breaking the Lock-in: Diversifying Text-to-Image Generation via Representation Modulation","ref_index":9,"is_internal_anchor":true},{"citing_arxiv_id":"2602.06886","citing_title":"Prompt Reinjection: Alleviating Prompt Forgetting in Multimodal Diffusion Transformers for Text-to-Image Generation","ref_index":8,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/55VZJ4H24CPF6NUGTROTH55WVK","json":"https://pith.science/pith/55VZJ4H24CPF6NUGTROTH55WVK.json","graph_json":"https://pith.science/api/pith-number/55VZJ4H24CPF6NUGTROTH55WVK/graph.json","events_json":"https://pith.science/api/pith-number/55VZJ4H24CPF6NUGTROTH55WVK/events.json","paper":"https://pith.science/paper/55VZJ4H2"},"agent_actions":{"view_html":"https://pith.science/pith/55VZJ4H24CPF6NUGTROTH55WVK","download_json":"https://pith.science/pith/55VZJ4H24CPF6NUGTROTH55WVK.json","view_paper":"https://pith.science/paper/55VZJ4H2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2601.02211&json=true","fetch_graph":"https://pith.science/api/pith-number/55VZJ4H24CPF6NUGTROTH55WVK/graph.json","fetch_events":"https://pith.science/api/pith-number/55VZJ4H24CPF6NUGTROTH55WVK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/55VZJ4H24CPF6NUGTROTH55WVK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/55VZJ4H24CPF6NUGTROTH55WVK/action/storage_attestation","attest_author":"https://pith.science/pith/55VZJ4H24CPF6NUGTROTH55WVK/action/author_attestation","sign_citation":"https://pith.science/pith/55VZJ4H24CPF6NUGTROTH55WVK/action/citation_signature","submit_replication":"https://pith.science/pith/55VZJ4H24CPF6NUGTROTH55WVK/action/replication_record"}},"created_at":"2026-07-07T02:18:32.627752+00:00","updated_at":"2026-07-07T02:18:32.627752+00:00"}