{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:AHPI6N2TKWLIPZNPVCJGPLSECK","short_pith_number":"pith:AHPI6N2T","schema_version":"1.0","canonical_sha256":"01de8f3753559687e5afa89267ae4412a9f5157f4263323f120ec2398d755762","source":{"kind":"arxiv","id":"2506.02161","version":2},"attestation_state":"computed","paper":{"title":"TIIF-Bench: How Does Your T2I Model Follow Your Instructions?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hongyang Wei, Jinrui Zhang, Lei Zhang, Xinyu Wei, Zeqing Wang, Zhen Guo","submitted_at":"2025-06-02T18:44:07Z","abstract_excerpt":"The rapid advancements of Text-to-Image (T2I) models have ushered in a new phase of AI-generated content, marked by their growing ability to interpret and follow user instructions. However, existing T2I model evaluation benchmarks fall short in limited prompt diversity and complexity, as well as coarse evaluation metrics, making it difficult to evaluate the fine-grained alignment performance between textual instructions and generated images. In this paper, we present TIIF-Bench (Text-to-Image Instruction Following Benchmark), aiming to systematically assess T2I models' ability in interpreting "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.02161","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-06-02T18:44:07Z","cross_cats_sorted":[],"title_canon_sha256":"439e77eac1994dcac0d10dc6b0136e17181d7fd16e56d1f11ef2c649ae4e5414","abstract_canon_sha256":"e550e2b308866e5e36ea11a3780220da4b21bb523d30a4eec6e248d3784cb7b3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:26:52.064402Z","signature_b64":"yMLjeFCYHm0BwN5U/Xi10aGl8xNhwUFLu2xS+xjG4KsIHnpfwxwtwHgKP9fJrC3HZxYlR9H5lQKMP3YFjbq3Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"01de8f3753559687e5afa89267ae4412a9f5157f4263323f120ec2398d755762","last_reissued_at":"2026-07-05T11:26:52.063835Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:26:52.063835Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TIIF-Bench: How Does Your T2I Model Follow Your Instructions?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Hongyang Wei, Jinrui Zhang, Lei Zhang, Xinyu Wei, Zeqing Wang, Zhen Guo","submitted_at":"2025-06-02T18:44:07Z","abstract_excerpt":"The rapid advancements of Text-to-Image (T2I) models have ushered in a new phase of AI-generated content, marked by their growing ability to interpret and follow user instructions. However, existing T2I model evaluation benchmarks fall short in limited prompt diversity and complexity, as well as coarse evaluation metrics, making it difficult to evaluate the fine-grained alignment performance between textual instructions and generated images. In this paper, we present TIIF-Bench (Text-to-Image Instruction Following Benchmark), aiming to systematically assess T2I models' ability in interpreting "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.02161","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.02161/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.02161","created_at":"2026-07-05T11:26:52.063906+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.02161v2","created_at":"2026-07-05T11:26:52.063906+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.02161","created_at":"2026-07-05T11:26:52.063906+00:00"},{"alias_kind":"pith_short_12","alias_value":"AHPI6N2TKWLI","created_at":"2026-07-05T11:26:52.063906+00:00"},{"alias_kind":"pith_short_16","alias_value":"AHPI6N2TKWLIPZNP","created_at":"2026-07-05T11:26:52.063906+00:00"},{"alias_kind":"pith_short_8","alias_value":"AHPI6N2T","created_at":"2026-07-05T11:26:52.063906+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":13,"sample":[{"citing_arxiv_id":"2606.24849","citing_title":"IV-CoT: Implicit Visual Chain-of-Thought for Structure-Aware Text-to-Image Generation","ref_index":30,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20100","citing_title":"WeGenBench: A Multidimensional Diagnostic Benchmark towards Text-to-Image Model Optimization","ref_index":27,"is_internal_anchor":true},{"citing_arxiv_id":"2606.03746","citing_title":"Qwen-Image-Flash: Beyond Objective Design","ref_index":13,"is_internal_anchor":true},{"citing_arxiv_id":"2606.31711","citing_title":"Arena-T2I Hard: Benchmarking and Improving Faithfulness with Dependency-Aware Checklist","ref_index":23,"is_internal_anchor":true},{"citing_arxiv_id":"2605.28091","citing_title":"Qwen-Image-Bench: From Generation to Creation in Text-to-Image Evaluation","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2605.17602","citing_title":"AutoRubric-T2I: Robust Rule-Based Reward Model for Text-to-Image Alignment","ref_index":27,"is_internal_anchor":true},{"citing_arxiv_id":"2512.01843","citing_title":"PhyDetEx: Detecting and Explaining the Physical Plausibility of T2V Models","ref_index":51,"is_internal_anchor":true},{"citing_arxiv_id":"2605.17602","citing_title":"AutoRubric-T2I: Robust Rule-Based Reward Model for Text-to-Image Alignment","ref_index":27,"is_internal_anchor":true},{"citing_arxiv_id":"2512.07348","citing_title":"MICo-150K: A Comprehensive Dataset Advancing Multi-Image Composition","ref_index":84,"is_internal_anchor":true},{"citing_arxiv_id":"2605.12500","citing_title":"SenseNova-U1: Unifying Multimodal Understanding and Generation with NEO-unify Architecture","ref_index":138,"is_internal_anchor":true},{"citing_arxiv_id":"2605.09591","citing_title":"From Pixels to Concepts: Do Segmentation Models Understand What They Segment?","ref_index":32,"is_internal_anchor":true},{"citing_arxiv_id":"2605.06170","citing_title":"DynT2I-Eval: A Dynamic Evaluation Framework for Text-to-Image Models","ref_index":43,"is_internal_anchor":true},{"citing_arxiv_id":"2511.22699","citing_title":"Z-Image: An Efficient Image Generation Foundation Model with Single-Stream Diffusion Transformer","ref_index":74,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AHPI6N2TKWLIPZNPVCJGPLSECK","json":"https://pith.science/pith/AHPI6N2TKWLIPZNPVCJGPLSECK.json","graph_json":"https://pith.science/api/pith-number/AHPI6N2TKWLIPZNPVCJGPLSECK/graph.json","events_json":"https://pith.science/api/pith-number/AHPI6N2TKWLIPZNPVCJGPLSECK/events.json","paper":"https://pith.science/paper/AHPI6N2T"},"agent_actions":{"view_html":"https://pith.science/pith/AHPI6N2TKWLIPZNPVCJGPLSECK","download_json":"https://pith.science/pith/AHPI6N2TKWLIPZNPVCJGPLSECK.json","view_paper":"https://pith.science/paper/AHPI6N2T","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.02161&json=true","fetch_graph":"https://pith.science/api/pith-number/AHPI6N2TKWLIPZNPVCJGPLSECK/graph.json","fetch_events":"https://pith.science/api/pith-number/AHPI6N2TKWLIPZNPVCJGPLSECK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AHPI6N2TKWLIPZNPVCJGPLSECK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AHPI6N2TKWLIPZNPVCJGPLSECK/action/storage_attestation","attest_author":"https://pith.science/pith/AHPI6N2TKWLIPZNPVCJGPLSECK/action/author_attestation","sign_citation":"https://pith.science/pith/AHPI6N2TKWLIPZNPVCJGPLSECK/action/citation_signature","submit_replication":"https://pith.science/pith/AHPI6N2TKWLIPZNPVCJGPLSECK/action/replication_record"}},"created_at":"2026-07-05T11:26:52.063906+00:00","updated_at":"2026-07-05T11:26:52.063906+00:00"}