{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:E66Q2TMYOO4L5HGXWP3AFVKJDJ","short_pith_number":"pith:E66Q2TMY","schema_version":"1.0","canonical_sha256":"27bd0d4d9873b8be9cd7b3f602d5491a644f2f93cfca8fbd8b1bd1fa493f5c56","source":{"kind":"arxiv","id":"2410.12564","version":2},"attestation_state":"computed","paper":{"title":"FTII-Bench: A Comprehensive Multimodal Benchmark for Flow Text with Image Insertion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Feiyu Xiong, Jiacheng Ruan, Yebin Yang, Yuchen Feng, Zehao Lin, Zeyun Tang, Zhiyu Li","submitted_at":"2024-10-16T13:38:31Z","abstract_excerpt":"Benefiting from the revolutionary advances in large language models (LLMs) and foundational vision models, large vision-language models (LVLMs) have also made significant progress. However, current benchmarks focus on tasks that evaluating only a single aspect of LVLM capabilities (e.g., recognition, detection, understanding). These tasks fail to fully demonstrate LVLMs' potential in complex application scenarios. To comprehensively assess the performance of existing LVLMs, we propose a more challenging task called the Flow Text with Image Insertion task (FTII). This task requires LVLMs to sim"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.12564","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-16T13:38:31Z","cross_cats_sorted":[],"title_canon_sha256":"79b60dcb13648a73293098f03cdf3992ddf8b42ab59bee0dabb6809ba3a85a43","abstract_canon_sha256":"bafe22ad79c4c6b7ae65e1ecce2ad3053c44d61f70178bee6d2f13dd41213531"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:39:59.102602Z","signature_b64":"h6dzqb/kfP3LMlEza19YN4zMzGzcrWGtX9dJR7WBYWJzBTS/SC+QXxIoFhIMDA8HNHDDMAZxIR0Dw5vKhCMhCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"27bd0d4d9873b8be9cd7b3f602d5491a644f2f93cfca8fbd8b1bd1fa493f5c56","last_reissued_at":"2026-07-05T09:39:59.102170Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:39:59.102170Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FTII-Bench: A Comprehensive Multimodal Benchmark for Flow Text with Image Insertion","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Feiyu Xiong, Jiacheng Ruan, Yebin Yang, Yuchen Feng, Zehao Lin, Zeyun Tang, Zhiyu Li","submitted_at":"2024-10-16T13:38:31Z","abstract_excerpt":"Benefiting from the revolutionary advances in large language models (LLMs) and foundational vision models, large vision-language models (LVLMs) have also made significant progress. However, current benchmarks focus on tasks that evaluating only a single aspect of LVLM capabilities (e.g., recognition, detection, understanding). These tasks fail to fully demonstrate LVLMs' potential in complex application scenarios. To comprehensively assess the performance of existing LVLMs, we propose a more challenging task called the Flow Text with Image Insertion task (FTII). This task requires LVLMs to sim"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.12564","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.12564/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.12564","created_at":"2026-07-05T09:39:59.102226+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.12564v2","created_at":"2026-07-05T09:39:59.102226+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.12564","created_at":"2026-07-05T09:39:59.102226+00:00"},{"alias_kind":"pith_short_12","alias_value":"E66Q2TMYOO4L","created_at":"2026-07-05T09:39:59.102226+00:00"},{"alias_kind":"pith_short_16","alias_value":"E66Q2TMYOO4L5HGX","created_at":"2026-07-05T09:39:59.102226+00:00"},{"alias_kind":"pith_short_8","alias_value":"E66Q2TMY","created_at":"2026-07-05T09:39:59.102226+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.06328","citing_title":"M2IO-R1: An Efficient RL-Enhanced Reasoning Framework for Multimodal Retrieval Augmented Multimodal Generation","ref_index":28,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E66Q2TMYOO4L5HGXWP3AFVKJDJ","json":"https://pith.science/pith/E66Q2TMYOO4L5HGXWP3AFVKJDJ.json","graph_json":"https://pith.science/api/pith-number/E66Q2TMYOO4L5HGXWP3AFVKJDJ/graph.json","events_json":"https://pith.science/api/pith-number/E66Q2TMYOO4L5HGXWP3AFVKJDJ/events.json","paper":"https://pith.science/paper/E66Q2TMY"},"agent_actions":{"view_html":"https://pith.science/pith/E66Q2TMYOO4L5HGXWP3AFVKJDJ","download_json":"https://pith.science/pith/E66Q2TMYOO4L5HGXWP3AFVKJDJ.json","view_paper":"https://pith.science/paper/E66Q2TMY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.12564&json=true","fetch_graph":"https://pith.science/api/pith-number/E66Q2TMYOO4L5HGXWP3AFVKJDJ/graph.json","fetch_events":"https://pith.science/api/pith-number/E66Q2TMYOO4L5HGXWP3AFVKJDJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E66Q2TMYOO4L5HGXWP3AFVKJDJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E66Q2TMYOO4L5HGXWP3AFVKJDJ/action/storage_attestation","attest_author":"https://pith.science/pith/E66Q2TMYOO4L5HGXWP3AFVKJDJ/action/author_attestation","sign_citation":"https://pith.science/pith/E66Q2TMYOO4L5HGXWP3AFVKJDJ/action/citation_signature","submit_replication":"https://pith.science/pith/E66Q2TMYOO4L5HGXWP3AFVKJDJ/action/replication_record"}},"created_at":"2026-07-05T09:39:59.102226+00:00","updated_at":"2026-07-05T09:39:59.102226+00:00"}