{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ANSISFYA4BYYU5KG3AM24W2544","short_pith_number":"pith:ANSISFYA","schema_version":"1.0","canonical_sha256":"0364891700e0718a7546d819ae5b5de7341fb7dc7e914466c24762d8863f35c6","source":{"kind":"arxiv","id":"2504.07957","version":2},"attestation_state":"computed","paper":{"title":"MM-IFEngine: Towards Multimodal Instruction Following","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dahua Lin, Haodong Duan, Jiaqi Wang, Pan Zhang, Shengyuan Ding, Shenxi Wu, Xiangyu Zhao, Xiaoyi Dong, Yuhang Cao, Yuhang Zang","submitted_at":"2025-04-10T17:59:12Z","abstract_excerpt":"The Instruction Following (IF) ability measures how well Multi-modal Large Language Models (MLLMs) understand exactly what users are telling them and whether they are doing it right. Existing multimodal instruction following training data is scarce, the benchmarks are simple with atomic instructions, and the evaluation strategies are imprecise for tasks demanding exact output constraints. To address this, we present MM-IFEngine, an effective pipeline to generate high-quality image-instruction pairs. Our MM-IFEngine pipeline yields large-scale, diverse, and high-quality training data MM-IFInstr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.07957","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CV","submitted_at":"2025-04-10T17:59:12Z","cross_cats_sorted":[],"title_canon_sha256":"d10006d73d4c20e0efd8997c1f76e5ddf35b423b4f9ff9f958516bef263c4e8f","abstract_canon_sha256":"4e04c3ea9620dd2d10999bca9464184895673ccbbebc3598eb1fd55ceb9fb1ad"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:54:46.105847Z","signature_b64":"yC0DEhoNQirOfr//dHHQGSpwIow/JBww7SfewY4uI6Dq/sVHwKwyI+lOy4URl3YXKI/D7Lxj5bsa+DQEm+l6AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0364891700e0718a7546d819ae5b5de7341fb7dc7e914466c24762d8863f35c6","last_reissued_at":"2026-07-05T10:54:46.105319Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:54:46.105319Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MM-IFEngine: Towards Multimodal Instruction Following","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dahua Lin, Haodong Duan, Jiaqi Wang, Pan Zhang, Shengyuan Ding, Shenxi Wu, Xiangyu Zhao, Xiaoyi Dong, Yuhang Cao, Yuhang Zang","submitted_at":"2025-04-10T17:59:12Z","abstract_excerpt":"The Instruction Following (IF) ability measures how well Multi-modal Large Language Models (MLLMs) understand exactly what users are telling them and whether they are doing it right. Existing multimodal instruction following training data is scarce, the benchmarks are simple with atomic instructions, and the evaluation strategies are imprecise for tasks demanding exact output constraints. To address this, we present MM-IFEngine, an effective pipeline to generate high-quality image-instruction pairs. Our MM-IFEngine pipeline yields large-scale, diverse, and high-quality training data MM-IFInstr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.07957","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.07957/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.07957","created_at":"2026-07-05T10:54:46.105388+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.07957v2","created_at":"2026-07-05T10:54:46.105388+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.07957","created_at":"2026-07-05T10:54:46.105388+00:00"},{"alias_kind":"pith_short_12","alias_value":"ANSISFYA4BYY","created_at":"2026-07-05T10:54:46.105388+00:00"},{"alias_kind":"pith_short_16","alias_value":"ANSISFYA4BYYU5KG","created_at":"2026-07-05T10:54:46.105388+00:00"},{"alias_kind":"pith_short_8","alias_value":"ANSISFYA","created_at":"2026-07-05T10:54:46.105388+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11324","citing_title":"Embodied-R1.5: Evolving Physical Intelligence via Embodied Foundation Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18915","citing_title":"DMN: A Compositional Framework for Jailbreaking Multimodal LLMs with Multi-Image Inputs","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20164","citing_title":"Not Every Rubric Teaches Equally: Policy-Aware Rubric Rewards for RLVR","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2509.18154","citing_title":"MiniCPM-V 4.5: Cooking Efficient MLLMs via Architecture, Data, and Training Recipe","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27393","citing_title":"MiniCPM-o 4.5: Towards Real-Time Full-Duplex Omni-Modal Interaction","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ANSISFYA4BYYU5KG3AM24W2544","json":"https://pith.science/pith/ANSISFYA4BYYU5KG3AM24W2544.json","graph_json":"https://pith.science/api/pith-number/ANSISFYA4BYYU5KG3AM24W2544/graph.json","events_json":"https://pith.science/api/pith-number/ANSISFYA4BYYU5KG3AM24W2544/events.json","paper":"https://pith.science/paper/ANSISFYA"},"agent_actions":{"view_html":"https://pith.science/pith/ANSISFYA4BYYU5KG3AM24W2544","download_json":"https://pith.science/pith/ANSISFYA4BYYU5KG3AM24W2544.json","view_paper":"https://pith.science/paper/ANSISFYA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.07957&json=true","fetch_graph":"https://pith.science/api/pith-number/ANSISFYA4BYYU5KG3AM24W2544/graph.json","fetch_events":"https://pith.science/api/pith-number/ANSISFYA4BYYU5KG3AM24W2544/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ANSISFYA4BYYU5KG3AM24W2544/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ANSISFYA4BYYU5KG3AM24W2544/action/storage_attestation","attest_author":"https://pith.science/pith/ANSISFYA4BYYU5KG3AM24W2544/action/author_attestation","sign_citation":"https://pith.science/pith/ANSISFYA4BYYU5KG3AM24W2544/action/citation_signature","submit_replication":"https://pith.science/pith/ANSISFYA4BYYU5KG3AM24W2544/action/replication_record"}},"created_at":"2026-07-05T10:54:46.105388+00:00","updated_at":"2026-07-05T10:54:46.105388+00:00"}