{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TFJ2U5NDIWP2WVU5CZ6HKJOULD","short_pith_number":"pith:TFJ2U5ND","schema_version":"1.0","canonical_sha256":"9953aa75a3459fab569d167c7525d458dd0734159282e094cf1011bb356d3418","source":{"kind":"arxiv","id":"2406.20092","version":2},"attestation_state":"computed","paper":{"title":"Efficient Large Multi-modal Models via Visual Context Compression","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alan Yuille, Daniel Khashabi, Jieneng Chen, Ju He, Luoxin Ye, Zhao-Yang Wang","submitted_at":"2024-06-28T17:57:14Z","abstract_excerpt":"While significant advancements have been made in compressed representations for text embeddings in large language models (LLMs), the compression of visual tokens in multi-modal LLMs (MLLMs) has remained a largely overlooked area. In this work, we present the study on the analysis of redundancy concerning visual tokens and efficient training within these models. Our initial experiments show that eliminating up to 70% of visual tokens at the testing stage by simply average pooling only leads to a minimal 3% reduction in visual question answering accuracy on the GQA benchmark, indicating signific"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.20092","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-28T17:57:14Z","cross_cats_sorted":[],"title_canon_sha256":"6190449daeeb774374337f497b862616481c77d1de85ae18e64c76e359ca293c","abstract_canon_sha256":"bcff5bd92aa317033d4153ae26eac6d52917431f35a2416566cb2e16d5fc156a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:36:12.863713Z","signature_b64":"wKklwBdUxXw+CqDBA2Hj9HD1EW8jLRObgIoM/fpOQwKFtT32JvicvnmBefRnRqb0Ox0eQO4bS8ICRBVHihaoBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9953aa75a3459fab569d167c7525d458dd0734159282e094cf1011bb356d3418","last_reissued_at":"2026-07-05T09:36:12.863217Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:36:12.863217Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Large Multi-modal Models via Visual Context Compression","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alan Yuille, Daniel Khashabi, Jieneng Chen, Ju He, Luoxin Ye, Zhao-Yang Wang","submitted_at":"2024-06-28T17:57:14Z","abstract_excerpt":"While significant advancements have been made in compressed representations for text embeddings in large language models (LLMs), the compression of visual tokens in multi-modal LLMs (MLLMs) has remained a largely overlooked area. In this work, we present the study on the analysis of redundancy concerning visual tokens and efficient training within these models. Our initial experiments show that eliminating up to 70% of visual tokens at the testing stage by simply average pooling only leads to a minimal 3% reduction in visual question answering accuracy on the GQA benchmark, indicating signific"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.20092","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.20092/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.20092","created_at":"2026-07-05T09:36:12.863276+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.20092v2","created_at":"2026-07-05T09:36:12.863276+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.20092","created_at":"2026-07-05T09:36:12.863276+00:00"},{"alias_kind":"pith_short_12","alias_value":"TFJ2U5NDIWP2","created_at":"2026-07-05T09:36:12.863276+00:00"},{"alias_kind":"pith_short_16","alias_value":"TFJ2U5NDIWP2WVU5","created_at":"2026-07-05T09:36:12.863276+00:00"},{"alias_kind":"pith_short_8","alias_value":"TFJ2U5ND","created_at":"2026-07-05T09:36:12.863276+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12195","citing_title":"InternVideo3: Agentify Foundation Models with Multimodal Contextual Reasoning","ref_index":295,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11853","citing_title":"Task-Aware Structured Memory for Dynamic Multi-modal In-Context Learning","ref_index":210,"is_internal_anchor":false},{"citing_arxiv_id":"2503.14075","citing_title":"Growing a Multi-head Twig via Distillation and Reinforcement Learning to Accelerate Large Vision-Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2501.00574","citing_title":"VideoChat-Flash: Hierarchical Compression for Long-Context Video Modeling","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2410.17247","citing_title":"PyramidDrop: Accelerating Your Large Vision-Language Models via Pyramid Visual Redundancy Reduction","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12358","citing_title":"Why and When Visual Token Pruning Fails? A Study on Relevant Visual Information Shift in MLLMs Decoding","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TFJ2U5NDIWP2WVU5CZ6HKJOULD","json":"https://pith.science/pith/TFJ2U5NDIWP2WVU5CZ6HKJOULD.json","graph_json":"https://pith.science/api/pith-number/TFJ2U5NDIWP2WVU5CZ6HKJOULD/graph.json","events_json":"https://pith.science/api/pith-number/TFJ2U5NDIWP2WVU5CZ6HKJOULD/events.json","paper":"https://pith.science/paper/TFJ2U5ND"},"agent_actions":{"view_html":"https://pith.science/pith/TFJ2U5NDIWP2WVU5CZ6HKJOULD","download_json":"https://pith.science/pith/TFJ2U5NDIWP2WVU5CZ6HKJOULD.json","view_paper":"https://pith.science/paper/TFJ2U5ND","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.20092&json=true","fetch_graph":"https://pith.science/api/pith-number/TFJ2U5NDIWP2WVU5CZ6HKJOULD/graph.json","fetch_events":"https://pith.science/api/pith-number/TFJ2U5NDIWP2WVU5CZ6HKJOULD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TFJ2U5NDIWP2WVU5CZ6HKJOULD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TFJ2U5NDIWP2WVU5CZ6HKJOULD/action/storage_attestation","attest_author":"https://pith.science/pith/TFJ2U5NDIWP2WVU5CZ6HKJOULD/action/author_attestation","sign_citation":"https://pith.science/pith/TFJ2U5NDIWP2WVU5CZ6HKJOULD/action/citation_signature","submit_replication":"https://pith.science/pith/TFJ2U5NDIWP2WVU5CZ6HKJOULD/action/replication_record"}},"created_at":"2026-07-05T09:36:12.863276+00:00","updated_at":"2026-07-05T09:36:12.863276+00:00"}