{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MEXINUH36JNLPGEUVLPOTW3P63","short_pith_number":"pith:MEXINUH3","schema_version":"1.0","canonical_sha256":"612e86d0fbf25ab79894aadee9db6ff6cee98be012b7490df68c0e7d9bb448ed","source":{"kind":"arxiv","id":"2407.02392","version":4},"attestation_state":"computed","paper":{"title":"TokenPacker: Efficient Visual Projector for Multimodal LLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dongqi Tang, Jianke Zhu, Jian Liu, Jie Qin, Lei Zhang, Song Wang, Wentong Li, Yuqian Yuan","submitted_at":"2024-07-02T16:10:55Z","abstract_excerpt":"The visual projector serves as an essential bridge between the visual encoder and the Large Language Model (LLM) in a Multimodal LLM (MLLM). Typically, MLLMs adopt a simple MLP to preserve all visual contexts via one-to-one transformation. However, the visual tokens are redundant and can be considerably increased when dealing with high-resolution images, impairing the efficiency of MLLMs significantly. Some recent works have introduced resampler or abstractor to reduce the number of resulting visual tokens. Unfortunately, they fail to capture finer details and undermine the visual reasoning ca"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.02392","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-07-02T16:10:55Z","cross_cats_sorted":[],"title_canon_sha256":"d3a4a5e649b93f12521f2cdb50dc8295f2d9d222900684f3edf0d97b39b1ded4","abstract_canon_sha256":"c2e39f2d2a40a5dc0f5f8ae00e59f911fe7a7183950ad5cdd453278ce2c2e1b4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:00:05.847524Z","signature_b64":"Ii7xj7RSabL/VtbzLKpz7w+5GO8QzwpZy3oQkM8CgPcMg2YBReG7WJtcs2YkxJMau9ubzejozApIw+HYzvFWBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"612e86d0fbf25ab79894aadee9db6ff6cee98be012b7490df68c0e7d9bb448ed","last_reissued_at":"2026-07-05T09:00:05.847107Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:00:05.847107Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TokenPacker: Efficient Visual Projector for Multimodal LLM","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Dongqi Tang, Jianke Zhu, Jian Liu, Jie Qin, Lei Zhang, Song Wang, Wentong Li, Yuqian Yuan","submitted_at":"2024-07-02T16:10:55Z","abstract_excerpt":"The visual projector serves as an essential bridge between the visual encoder and the Large Language Model (LLM) in a Multimodal LLM (MLLM). Typically, MLLMs adopt a simple MLP to preserve all visual contexts via one-to-one transformation. However, the visual tokens are redundant and can be considerably increased when dealing with high-resolution images, impairing the efficiency of MLLMs significantly. Some recent works have introduced resampler or abstractor to reduce the number of resulting visual tokens. Unfortunately, they fail to capture finer details and undermine the visual reasoning ca"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.02392","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.02392/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.02392","created_at":"2026-07-05T09:00:05.847162+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.02392v4","created_at":"2026-07-05T09:00:05.847162+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.02392","created_at":"2026-07-05T09:00:05.847162+00:00"},{"alias_kind":"pith_short_12","alias_value":"MEXINUH36JNL","created_at":"2026-07-05T09:00:05.847162+00:00"},{"alias_kind":"pith_short_16","alias_value":"MEXINUH36JNLPGEU","created_at":"2026-07-05T09:00:05.847162+00:00"},{"alias_kind":"pith_short_8","alias_value":"MEXINUH3","created_at":"2026-07-05T09:00:05.847162+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31383","citing_title":"MS-Resampler: Multi-Scope Visual Resampling for Efficient Multimodal LLMs","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30126","citing_title":"PARCEL: Pool-Anchored Resampling with Conditioned Elastic Queries for Efficient Vision-Language Understanding","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2511.06754","citing_title":"SlotVLA: Towards Modeling of Object-Relation Representations in Robotic Manipulation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2411.10440","citing_title":"LLaVA-CoT: Let Vision Language Models Reason Step-by-Step","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09442","citing_title":"UIPress: Bringing Optical Token Compression to UI-to-Code Generation","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MEXINUH36JNLPGEUVLPOTW3P63","json":"https://pith.science/pith/MEXINUH36JNLPGEUVLPOTW3P63.json","graph_json":"https://pith.science/api/pith-number/MEXINUH36JNLPGEUVLPOTW3P63/graph.json","events_json":"https://pith.science/api/pith-number/MEXINUH36JNLPGEUVLPOTW3P63/events.json","paper":"https://pith.science/paper/MEXINUH3"},"agent_actions":{"view_html":"https://pith.science/pith/MEXINUH36JNLPGEUVLPOTW3P63","download_json":"https://pith.science/pith/MEXINUH36JNLPGEUVLPOTW3P63.json","view_paper":"https://pith.science/paper/MEXINUH3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.02392&json=true","fetch_graph":"https://pith.science/api/pith-number/MEXINUH36JNLPGEUVLPOTW3P63/graph.json","fetch_events":"https://pith.science/api/pith-number/MEXINUH36JNLPGEUVLPOTW3P63/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MEXINUH36JNLPGEUVLPOTW3P63/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MEXINUH36JNLPGEUVLPOTW3P63/action/storage_attestation","attest_author":"https://pith.science/pith/MEXINUH36JNLPGEUVLPOTW3P63/action/author_attestation","sign_citation":"https://pith.science/pith/MEXINUH36JNLPGEUVLPOTW3P63/action/citation_signature","submit_replication":"https://pith.science/pith/MEXINUH36JNLPGEUVLPOTW3P63/action/replication_record"}},"created_at":"2026-07-05T09:00:05.847162+00:00","updated_at":"2026-07-05T09:00:05.847162+00:00"}