{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:N3PFCFKYU42BGYIMCLCFUPI2DD","short_pith_number":"pith:N3PFCFKY","schema_version":"1.0","canonical_sha256":"6ede511558a73413610c12c45a3d1a18f77a1e957396d38d3221b28f3312a5bb","source":{"kind":"arxiv","id":"2504.17892","version":1},"attestation_state":"computed","paper":{"title":"Token Sequence Compression for Efficient Multimodal Computing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Parth Shroff, Thierry Tambe, Yasmine Omri","submitted_at":"2025-04-24T19:11:10Z","abstract_excerpt":"The exponential growth of Large Multimodal Models (LMMs) has driven advancements in cross-modal reasoning but at significant computational costs. In this work, we focus on visual language models. We highlight the redundancy and inefficiency in current vision encoders, and seek to construct an adaptive compression method for multimodal data. In this work, we characterize a panoply of visual token selection and merging approaches through both benchmarking and qualitative analysis. In particular, we demonstrate that simple cluster-level token aggregation outperforms prior state-of-the-art works i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.17892","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-04-24T19:11:10Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"fef537e35fa740e9a61e8b8ab815c25f68300a65a7634e3075c7c0897e95f67b","abstract_canon_sha256":"3048ca0630efe8dabfc680433fd8b981d3e6c37f06eccf3893e5ce1fbd5f1da1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:53:58.269474Z","signature_b64":"yLwq2nZOvNzjGb4e7BpoeyuGjKZD7pKRZGtJND/Y/eb0ryCyIGup95raAjHQkOX05egKY8lFPHM5LDssy/isDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6ede511558a73413610c12c45a3d1a18f77a1e957396d38d3221b28f3312a5bb","last_reissued_at":"2026-07-05T10:53:58.268922Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:53:58.268922Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Token Sequence Compression for Efficient Multimodal Computing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Parth Shroff, Thierry Tambe, Yasmine Omri","submitted_at":"2025-04-24T19:11:10Z","abstract_excerpt":"The exponential growth of Large Multimodal Models (LMMs) has driven advancements in cross-modal reasoning but at significant computational costs. In this work, we focus on visual language models. We highlight the redundancy and inefficiency in current vision encoders, and seek to construct an adaptive compression method for multimodal data. In this work, we characterize a panoply of visual token selection and merging approaches through both benchmarking and qualitative analysis. In particular, we demonstrate that simple cluster-level token aggregation outperforms prior state-of-the-art works i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.17892","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.17892/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.17892","created_at":"2026-07-05T10:53:58.268984+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.17892v1","created_at":"2026-07-05T10:53:58.268984+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.17892","created_at":"2026-07-05T10:53:58.268984+00:00"},{"alias_kind":"pith_short_12","alias_value":"N3PFCFKYU42B","created_at":"2026-07-05T10:53:58.268984+00:00"},{"alias_kind":"pith_short_16","alias_value":"N3PFCFKYU42BGYIM","created_at":"2026-07-05T10:53:58.268984+00:00"},{"alias_kind":"pith_short_8","alias_value":"N3PFCFKY","created_at":"2026-07-05T10:53:58.268984+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.04449","citing_title":"p-MoD: Building Mixture-of-Depths MLLMs via Progressive Ratio Decay","ref_index":39,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/N3PFCFKYU42BGYIMCLCFUPI2DD","json":"https://pith.science/pith/N3PFCFKYU42BGYIMCLCFUPI2DD.json","graph_json":"https://pith.science/api/pith-number/N3PFCFKYU42BGYIMCLCFUPI2DD/graph.json","events_json":"https://pith.science/api/pith-number/N3PFCFKYU42BGYIMCLCFUPI2DD/events.json","paper":"https://pith.science/paper/N3PFCFKY"},"agent_actions":{"view_html":"https://pith.science/pith/N3PFCFKYU42BGYIMCLCFUPI2DD","download_json":"https://pith.science/pith/N3PFCFKYU42BGYIMCLCFUPI2DD.json","view_paper":"https://pith.science/paper/N3PFCFKY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.17892&json=true","fetch_graph":"https://pith.science/api/pith-number/N3PFCFKYU42BGYIMCLCFUPI2DD/graph.json","fetch_events":"https://pith.science/api/pith-number/N3PFCFKYU42BGYIMCLCFUPI2DD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/N3PFCFKYU42BGYIMCLCFUPI2DD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/N3PFCFKYU42BGYIMCLCFUPI2DD/action/storage_attestation","attest_author":"https://pith.science/pith/N3PFCFKYU42BGYIMCLCFUPI2DD/action/author_attestation","sign_citation":"https://pith.science/pith/N3PFCFKYU42BGYIMCLCFUPI2DD/action/citation_signature","submit_replication":"https://pith.science/pith/N3PFCFKYU42BGYIMCLCFUPI2DD/action/replication_record"}},"created_at":"2026-07-05T10:53:58.268984+00:00","updated_at":"2026-07-05T10:53:58.268984+00:00"}