{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QZTJHRX3TCLWTDR4HPJI5UMAKP","short_pith_number":"pith:QZTJHRX3","schema_version":"1.0","canonical_sha256":"866693c6fb9897698e3c3bd28ed18053f33e8be332820428236ac2a3647cd037","source":{"kind":"arxiv","id":"2410.06169","version":3},"attestation_state":"computed","paper":{"title":"Treat Visual Tokens as Text? But Your MLLM Only Needs Fewer Efforts to See","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ajinkya Kale, Chenliang Xu, Daniel Miranda, Jianing Zhou, Kun Wan, Phu Pham, Wentian Zhao, Yu-Jhe Li, Zeliang Zhang","submitted_at":"2024-10-08T16:13:24Z","abstract_excerpt":"By treating visual tokens from visual encoders as text tokens, Multimodal Large Language Models (MLLMs) have achieved remarkable progress across diverse visual understanding tasks, leveraging the robust architectures of Large Language Models (LLMs). However, as token counts grow, the quadratic scaling of computation in LLMs introduces a significant efficiency bottleneck, impeding further scalability. Although recent approaches have explored pruning visual tokens or employing lighter LLM architectures, the computational overhead from an increasing number of visual tokens remains a substantial c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.06169","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-08T16:13:24Z","cross_cats_sorted":[],"title_canon_sha256":"c23b3430e860cd329c660a513a3e5b6fae7e233f5d06d489269c44b2e34a9965","abstract_canon_sha256":"2326a950b8832cfb9c78650f6a578e79d135b6b95f4b26e58a0051a75f0cf369"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:42:31.029557Z","signature_b64":"yL7fJ3t49XZ9Z0bvkI3GUGGvb9yj0ZF4TeJrTGrQs1tY388p0NejtddC5UEelPvHRTwGLKR2PSjN3/fQzMn+CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"866693c6fb9897698e3c3bd28ed18053f33e8be332820428236ac2a3647cd037","last_reissued_at":"2026-07-05T09:42:31.028985Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:42:31.028985Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Treat Visual Tokens as Text? But Your MLLM Only Needs Fewer Efforts to See","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ajinkya Kale, Chenliang Xu, Daniel Miranda, Jianing Zhou, Kun Wan, Phu Pham, Wentian Zhao, Yu-Jhe Li, Zeliang Zhang","submitted_at":"2024-10-08T16:13:24Z","abstract_excerpt":"By treating visual tokens from visual encoders as text tokens, Multimodal Large Language Models (MLLMs) have achieved remarkable progress across diverse visual understanding tasks, leveraging the robust architectures of Large Language Models (LLMs). However, as token counts grow, the quadratic scaling of computation in LLMs introduces a significant efficiency bottleneck, impeding further scalability. Although recent approaches have explored pruning visual tokens or employing lighter LLM architectures, the computational overhead from an increasing number of visual tokens remains a substantial c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.06169","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.06169/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.06169","created_at":"2026-07-05T09:42:31.029049+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.06169v3","created_at":"2026-07-05T09:42:31.029049+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.06169","created_at":"2026-07-05T09:42:31.029049+00:00"},{"alias_kind":"pith_short_12","alias_value":"QZTJHRX3TCLW","created_at":"2026-07-05T09:42:31.029049+00:00"},{"alias_kind":"pith_short_16","alias_value":"QZTJHRX3TCLWTDR4","created_at":"2026-07-05T09:42:31.029049+00:00"},{"alias_kind":"pith_short_8","alias_value":"QZTJHRX3","created_at":"2026-07-05T09:42:31.029049+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.14075","citing_title":"Growing a Multi-head Twig via Distillation and Reinforcement Learning to Accelerate Large Vision-Language Models","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03114","citing_title":"Can VLMs Truly Forget? Benchmarking Training-Free Visual Concept Unlearning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10039","citing_title":"Counting to Four is still a Chore for VLMs","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17087","citing_title":"EvoComp: Learning Visual Token Compression for Multimodal Large Language Models via Semantic-Guided Evolutionary Labeling","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QZTJHRX3TCLWTDR4HPJI5UMAKP","json":"https://pith.science/pith/QZTJHRX3TCLWTDR4HPJI5UMAKP.json","graph_json":"https://pith.science/api/pith-number/QZTJHRX3TCLWTDR4HPJI5UMAKP/graph.json","events_json":"https://pith.science/api/pith-number/QZTJHRX3TCLWTDR4HPJI5UMAKP/events.json","paper":"https://pith.science/paper/QZTJHRX3"},"agent_actions":{"view_html":"https://pith.science/pith/QZTJHRX3TCLWTDR4HPJI5UMAKP","download_json":"https://pith.science/pith/QZTJHRX3TCLWTDR4HPJI5UMAKP.json","view_paper":"https://pith.science/paper/QZTJHRX3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.06169&json=true","fetch_graph":"https://pith.science/api/pith-number/QZTJHRX3TCLWTDR4HPJI5UMAKP/graph.json","fetch_events":"https://pith.science/api/pith-number/QZTJHRX3TCLWTDR4HPJI5UMAKP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QZTJHRX3TCLWTDR4HPJI5UMAKP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QZTJHRX3TCLWTDR4HPJI5UMAKP/action/storage_attestation","attest_author":"https://pith.science/pith/QZTJHRX3TCLWTDR4HPJI5UMAKP/action/author_attestation","sign_citation":"https://pith.science/pith/QZTJHRX3TCLWTDR4HPJI5UMAKP/action/citation_signature","submit_replication":"https://pith.science/pith/QZTJHRX3TCLWTDR4HPJI5UMAKP/action/replication_record"}},"created_at":"2026-07-05T09:42:31.029049+00:00","updated_at":"2026-07-05T09:42:31.029049+00:00"}